1//===- MemProfUse.cpp - memory allocation profile use pass --*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the MemProfUsePass which reads memory profiling data
10// and uses it to add metadata to instructions to guide optimization.
11//
12//===----------------------------------------------------------------------===//
13
14#include "llvm/Transforms/Instrumentation/MemProfUse.h"
15#include "llvm/ADT/DenseSet.h"
16#include "llvm/ADT/SmallVector.h"
17#include "llvm/ADT/Statistic.h"
18#include "llvm/ADT/StringRef.h"
19#include "llvm/Analysis/MemoryProfileInfo.h"
20#include "llvm/Analysis/OptimizationRemarkEmitter.h"
21#include "llvm/Analysis/StaticDataProfileInfo.h"
22#include "llvm/Analysis/TargetLibraryInfo.h"
23#include "llvm/IR/DiagnosticInfo.h"
24#include "llvm/IR/Function.h"
25#include "llvm/IR/IntrinsicInst.h"
26#include "llvm/IR/Module.h"
27#include "llvm/ProfileData/DataAccessProf.h"
28#include "llvm/ProfileData/InstrProf.h"
29#include "llvm/ProfileData/InstrProfReader.h"
30#include "llvm/ProfileData/MemProfCommon.h"
31#include "llvm/Support/BLAKE3.h"
32#include "llvm/Support/CommandLine.h"
33#include "llvm/Support/Debug.h"
34#include "llvm/Support/HashBuilder.h"
35#include "llvm/Support/MD5.h"
36#include "llvm/Support/VirtualFileSystem.h"
37#include "llvm/Transforms/Utils/LongestCommonSequence.h"
38#include <map>
39#include <set>
40
41using namespace llvm;
42using namespace llvm::memprof;
43
44#define DEBUG_TYPE "memprof"
45
46namespace llvm {
47extern cl::opt<bool> PGOWarnMissing;
48extern cl::opt<bool> NoPGOWarnMismatch;
49extern cl::opt<bool> NoPGOWarnMismatchComdatWeak;
50extern cl::opt<bool> AnnotateStringLiteralSectionPrefix;
51} // namespace llvm
52
53// By default disable matching of allocation profiles onto operator new that
54// already explicitly pass a hot/cold hint, since we don't currently
55// override these hints anyway.
56static cl::opt<bool> ClMemProfMatchHotColdNew(
57 "memprof-match-hot-cold-new",
58 cl::desc(
59 "Match allocation profiles onto existing hot/cold operator new calls"),
60 cl::Hidden, cl::init(Val: false));
61
62static cl::opt<bool>
63 ClPrintMemProfMatchInfo("memprof-print-match-info",
64 cl::desc("Print matching stats for each allocation "
65 "context in this module's profiles"),
66 cl::Hidden, cl::init(Val: false));
67
68static cl::opt<bool> PrintMatchedAllocStack(
69 "memprof-print-matched-alloc-stack",
70 cl::desc("Print full stack context for matched "
71 "allocations with -memprof-print-match-info."),
72 cl::Hidden, cl::init(Val: false));
73
74static cl::opt<bool>
75 PrintFunctionGuids("memprof-print-function-guids",
76 cl::desc("Print function GUIDs computed for matching"),
77 cl::Hidden, cl::init(Val: false));
78
79static cl::opt<bool>
80 SalvageStaleProfile("memprof-salvage-stale-profile",
81 cl::desc("Salvage stale MemProf profile"),
82 cl::init(Val: false), cl::Hidden);
83
84static cl::opt<bool> ClMemProfAttachCalleeGuids(
85 "memprof-attach-calleeguids",
86 cl::desc(
87 "Attach calleeguids as value profile metadata for indirect calls."),
88 cl::init(Val: true), cl::Hidden);
89
90static cl::opt<unsigned> MinMatchedColdBytePercent(
91 "memprof-matching-cold-threshold", cl::init(Val: 100), cl::Hidden,
92 cl::desc("Min percent of cold bytes matched to hint allocation cold"));
93
94static cl::opt<bool> AnnotateStaticDataSectionPrefix(
95 "memprof-annotate-static-data-prefix", cl::init(Val: false), cl::Hidden,
96 cl::desc("If true, annotate the static data section prefix"));
97
98// Matching statistics
99STATISTIC(NumOfMemProfMissing, "Number of functions without memory profile.");
100STATISTIC(NumOfMemProfMismatch,
101 "Number of functions having mismatched memory profile hash.");
102STATISTIC(NumOfMemProfFunc, "Number of functions having valid memory profile.");
103STATISTIC(NumOfMemProfAllocContextProfiles,
104 "Number of alloc contexts in memory profile.");
105STATISTIC(NumOfMemProfCallSiteProfiles,
106 "Number of callsites in memory profile.");
107STATISTIC(NumOfMemProfMatchedAllocContexts,
108 "Number of matched memory profile alloc contexts.");
109STATISTIC(NumOfMemProfMatchedAllocs,
110 "Number of matched memory profile allocs.");
111STATISTIC(NumOfMemProfMatchedCallSites,
112 "Number of matched memory profile callsites.");
113STATISTIC(NumOfMemProfHotGlobalVars,
114 "Number of global vars annotated with 'hot' section prefix.");
115STATISTIC(NumOfMemProfColdGlobalVars,
116 "Number of global vars annotated with 'unlikely' section prefix.");
117STATISTIC(NumOfMemProfUnknownGlobalVars,
118 "Number of global vars with unknown hotness (no section prefix).");
119STATISTIC(NumOfMemProfExplicitSectionGlobalVars,
120 "Number of global vars with user-specified section (not annotated).");
121
122static void addCallsiteMetadata(Instruction &I,
123 ArrayRef<uint64_t> InlinedCallStack,
124 LLVMContext &Ctx) {
125 I.setMetadata(KindID: LLVMContext::MD_callsite,
126 Node: buildCallstackMetadata(CallStack: InlinedCallStack, Ctx));
127}
128
129static uint64_t computeStackId(GlobalValue::GUID Function, uint32_t LineOffset,
130 uint32_t Column) {
131 llvm::HashBuilder<llvm::TruncatedBLAKE3<8>, llvm::endianness::little>
132 HashBuilder;
133 HashBuilder.add(Args: Function, Args: LineOffset, Args: Column);
134 llvm::BLAKE3Result<8> Hash = HashBuilder.final();
135 uint64_t Id;
136 std::memcpy(dest: &Id, src: Hash.data(), n: sizeof(Hash));
137 return Id;
138}
139
140static uint64_t computeStackId(const memprof::Frame &Frame) {
141 return computeStackId(Function: Frame.Function, LineOffset: Frame.LineOffset, Column: Frame.Column);
142}
143
144// Compute the line offset of a debug location relative to the start line of
145// its enclosing subprogram, clamped at 0 as the profile reader does (e.g. for
146// locations without debug info whose line is 0).
147static uint32_t getLineOffset(const DILocation *DIL) {
148 unsigned Line = DIL->getLine();
149 unsigned SubprogramLine = DIL->getScope()->getSubprogram()->getLine();
150 if (Line < SubprogramLine)
151 return 0;
152 return Line - SubprogramLine;
153}
154
155static AllocationType getAllocType(const AllocationInfo *AllocInfo) {
156 return getAllocType(TotalLifetimeAccessDensity: AllocInfo->Info.getTotalLifetimeAccessDensity(),
157 AllocCount: AllocInfo->Info.getAllocCount(),
158 TotalLifetime: AllocInfo->Info.getTotalLifetime());
159}
160
161static AllocationType addCallStack(CallStackTrie &AllocTrie,
162 const AllocationInfo *AllocInfo,
163 uint64_t FullStackId) {
164 SmallVector<uint64_t> StackIds;
165 for (const auto &StackFrame : AllocInfo->CallStack)
166 StackIds.push_back(Elt: computeStackId(Frame: StackFrame));
167 auto AllocType = getAllocType(AllocInfo);
168 std::vector<ContextTotalSize> ContextSizeInfo;
169 if (recordContextSizeInfoForAnalysis()) {
170 auto TotalSize = AllocInfo->Info.getTotalSize();
171 assert(TotalSize);
172 assert(FullStackId != 0);
173 ContextSizeInfo.push_back(x: {.FullStackId: FullStackId, .TotalSize: TotalSize});
174 }
175 AllocTrie.addCallStack(AllocType, StackIds, ContextSizeInfo: std::move(ContextSizeInfo));
176 return AllocType;
177}
178
179// Return true if InlinedCallStack, computed from a call instruction's debug
180// info, is a prefix of ProfileCallStack, a list of Frames from profile data
181// (either the allocation data or a callsite).
182static bool
183stackFrameIncludesInlinedCallStack(ArrayRef<Frame> ProfileCallStack,
184 ArrayRef<uint64_t> InlinedCallStack) {
185 return ProfileCallStack.size() >= InlinedCallStack.size() &&
186 llvm::equal(LRange: ProfileCallStack.take_front(N: InlinedCallStack.size()),
187 RRange&: InlinedCallStack, P: [](const Frame &F, uint64_t StackId) {
188 return computeStackId(Frame: F) == StackId;
189 });
190}
191
192static bool isAllocationWithHotColdVariant(const Function *Callee,
193 const TargetLibraryInfo &TLI) {
194 if (!Callee)
195 return false;
196 LibFunc Func = TLI.getLibFunc(FDecl: *Callee);
197 if (Func == NotLibFunc)
198 return false;
199 switch (Func) {
200 case LibFunc_Znwm:
201 case LibFunc_ZnwmRKSt9nothrow_t:
202 case LibFunc_ZnwmSt11align_val_t:
203 case LibFunc_ZnwmSt11align_val_tRKSt9nothrow_t:
204 case LibFunc_Znam:
205 case LibFunc_ZnamRKSt9nothrow_t:
206 case LibFunc_ZnamSt11align_val_t:
207 case LibFunc_ZnamSt11align_val_tRKSt9nothrow_t:
208 case LibFunc_size_returning_new:
209 case LibFunc_size_returning_new_aligned:
210 return true;
211 case LibFunc_Znwm12__hot_cold_t:
212 case LibFunc_ZnwmRKSt9nothrow_t12__hot_cold_t:
213 case LibFunc_ZnwmSt11align_val_t12__hot_cold_t:
214 case LibFunc_ZnwmSt11align_val_tRKSt9nothrow_t12__hot_cold_t:
215 case LibFunc_Znam12__hot_cold_t:
216 case LibFunc_ZnamRKSt9nothrow_t12__hot_cold_t:
217 case LibFunc_ZnamSt11align_val_t12__hot_cold_t:
218 case LibFunc_ZnamSt11align_val_tRKSt9nothrow_t12__hot_cold_t:
219 case LibFunc_size_returning_new_hot_cold:
220 case LibFunc_size_returning_new_aligned_hot_cold:
221 return ClMemProfMatchHotColdNew;
222 default:
223 return false;
224 }
225}
226
227static void HandleUnsupportedAnnotationKinds(GlobalVariable &GVar,
228 AnnotationKind Kind) {
229 assert(Kind != llvm::memprof::AnnotationKind::AnnotationOK &&
230 "Should not handle AnnotationOK here");
231 SmallString<32> Reason;
232 switch (Kind) {
233 case llvm::memprof::AnnotationKind::ExplicitSection:
234 ++NumOfMemProfExplicitSectionGlobalVars;
235 Reason.append(RHS: "explicit section name");
236 break;
237 case llvm::memprof::AnnotationKind::DeclForLinker:
238 Reason.append(RHS: "linker declaration");
239 break;
240 case llvm::memprof::AnnotationKind::ReservedName:
241 Reason.append(RHS: "name starts with `llvm.`");
242 break;
243 default:
244 llvm_unreachable("Unexpected annotation kind");
245 }
246 LLVM_DEBUG(dbgs() << "Skip annotation for " << GVar.getName() << " due to "
247 << Reason << ".\n");
248}
249
250// Computes the LLVM version of MD5 hash for the content of a string
251// literal.
252static std::optional<uint64_t>
253getStringContentHash(const GlobalVariable &GVar) {
254 auto *Initializer = GVar.getInitializer();
255 if (!Initializer)
256 return std::nullopt;
257 if (auto *C = dyn_cast<ConstantDataSequential>(Val: Initializer))
258 if (C->isString()) {
259 // Note the hash computed for the literal would include the null byte.
260 return llvm::MD5Hash(Str: C->getAsString());
261 }
262 return std::nullopt;
263}
264
265// Structure for tracking info about matched allocation contexts for use with
266// -memprof-print-match-info and -memprof-print-matched-alloc-stack.
267struct AllocMatchInfo {
268 // Total size in bytes of matched context.
269 uint64_t TotalSize = 0;
270 // Matched allocation's type.
271 AllocationType AllocType = AllocationType::None;
272 // Number of frames matched to the allocation itself (values will be >1 in
273 // cases where allocation was already inlined). Use a set because there can
274 // be multiple inlined instances and each may have a different inline depth.
275 // Use std::set to iterate in sorted order when printing.
276 std::set<unsigned> MatchedFramesSet;
277 // The full call stack of the allocation, for cases where requested via
278 // -memprof-print-matched-alloc-stack.
279 std::vector<Frame> CallStack;
280
281 // Caller responsible for inserting the matched frames and the call stack when
282 // appropriate.
283 AllocMatchInfo(uint64_t TotalSize, AllocationType AllocType)
284 : TotalSize(TotalSize), AllocType(AllocType) {}
285};
286
287DenseMap<uint64_t, SmallVector<CallEdgeTy, 0>>
288memprof::extractCallsFromIR(Module &M, const TargetLibraryInfo &TLI,
289 function_ref<bool(uint64_t)> IsPresentInProfile) {
290 DenseMap<uint64_t, SmallVector<CallEdgeTy, 0>> Calls;
291
292 for (Function &F : M) {
293 if (F.isDeclaration())
294 continue;
295
296 for (auto &BB : F) {
297 for (auto &I : BB) {
298 if (!isa<CallBase>(Val: &I) || isa<IntrinsicInst>(Val: &I))
299 continue;
300
301 auto *CB = dyn_cast<CallBase>(Val: &I);
302 auto *CalledFunction = CB->getCalledFunction();
303 // Disregard indirect calls and intrinsics.
304 if (!CalledFunction || CalledFunction->isIntrinsic())
305 continue;
306
307 StringRef CalleeName = CalledFunction->getName();
308 // True if we are calling a heap allocation function that supports
309 // hot/cold variants.
310 bool IsAlloc = isAllocationWithHotColdVariant(Callee: CalledFunction, TLI);
311 // True for the first iteration below, indicating that we are looking at
312 // a leaf node.
313 bool IsLeaf = true;
314 for (const DILocation *DIL = I.getDebugLoc(); DIL;
315 DIL = DIL->getInlinedAt()) {
316 StringRef CallerName = DIL->getSubprogramLinkageName();
317 assert(!CallerName.empty() &&
318 "Be sure to enable -fdebug-info-for-profiling");
319 uint64_t CallerGUID = memprof::getGUID(FunctionName: CallerName);
320 uint64_t CalleeGUID = memprof::getGUID(FunctionName: CalleeName);
321 // Pretend that we are calling a function with GUID == 0 if we are
322 // in the inline stack leading to a heap allocation function.
323 if (IsAlloc) {
324 if (IsLeaf) {
325 // For leaf nodes, set CalleeGUID to 0 without consulting
326 // IsPresentInProfile.
327 CalleeGUID = 0;
328 } else if (!IsPresentInProfile(CalleeGUID)) {
329 // In addition to the leaf case above, continue to set CalleeGUID
330 // to 0 as long as we don't see CalleeGUID in the profile.
331 CalleeGUID = 0;
332 } else {
333 // Once we encounter a callee that exists in the profile, stop
334 // setting CalleeGUID to 0.
335 IsAlloc = false;
336 }
337 }
338
339 LineLocation Loc = {getLineOffset(DIL), DIL->getColumn()};
340 Calls[CallerGUID].emplace_back(Args&: Loc, Args&: CalleeGUID);
341 CalleeName = CallerName;
342 IsLeaf = false;
343 }
344 }
345 }
346 }
347
348 // Sort each call list by the source location.
349 for (auto &[CallerGUID, CallList] : Calls) {
350 llvm::sort(C&: CallList);
351 CallList.erase(CS: llvm::unique(R&: CallList), CE: CallList.end());
352 }
353
354 return Calls;
355}
356
357DenseMap<uint64_t, LocToLocMap>
358memprof::computeUndriftMap(Module &M, IndexedInstrProfReader *MemProfReader,
359 const TargetLibraryInfo &TLI) {
360 DenseMap<uint64_t, LocToLocMap> UndriftMaps;
361
362 DenseMap<uint64_t, SmallVector<memprof::CallEdgeTy, 0>> CallsFromProfile =
363 MemProfReader->getMemProfCallerCalleePairs();
364 DenseMap<uint64_t, SmallVector<memprof::CallEdgeTy, 0>> CallsFromIR =
365 extractCallsFromIR(M, TLI, IsPresentInProfile: [&](uint64_t GUID) {
366 return CallsFromProfile.contains(Val: GUID);
367 });
368
369 // Compute an undrift map for each CallerGUID.
370 for (const auto &[CallerGUID, IRAnchors] : CallsFromIR) {
371 auto It = CallsFromProfile.find(Val: CallerGUID);
372 if (It == CallsFromProfile.end())
373 continue;
374 const auto &ProfileAnchors = It->second;
375
376 LocToLocMap Matchings;
377 longestCommonSequence<LineLocation, GlobalValue::GUID>(
378 AnchorList1: ProfileAnchors, AnchorList2: IRAnchors, FunctionMatchesProfile: std::equal_to<GlobalValue::GUID>(),
379 InsertMatching: [&](LineLocation A, LineLocation B) { Matchings.try_emplace(Key: A, Args&: B); });
380 [[maybe_unused]] bool Inserted =
381 UndriftMaps.try_emplace(Key: CallerGUID, Args: std::move(Matchings)).second;
382
383 // The insertion must succeed because we visit each GUID exactly once.
384 assert(Inserted);
385 }
386
387 return UndriftMaps;
388}
389
390// Given a MemProfRecord, undrift all the source locations present in the
391// record in place.
392static void
393undriftMemProfRecord(const DenseMap<uint64_t, LocToLocMap> &UndriftMaps,
394 memprof::MemProfRecord &MemProfRec) {
395 // Undrift a call stack in place.
396 auto UndriftCallStack = [&](std::vector<Frame> &CallStack) {
397 for (auto &F : CallStack) {
398 auto I = UndriftMaps.find(Val: F.Function);
399 if (I == UndriftMaps.end())
400 continue;
401 auto J = I->second.find(Val: LineLocation(F.LineOffset, F.Column));
402 if (J == I->second.end())
403 continue;
404 auto &NewLoc = J->second;
405 F.LineOffset = NewLoc.LineOffset;
406 F.Column = NewLoc.Column;
407 }
408 };
409
410 for (auto &AS : MemProfRec.AllocSites)
411 UndriftCallStack(AS.CallStack);
412
413 for (auto &CS : MemProfRec.CallSites)
414 UndriftCallStack(CS.Frames);
415}
416
417// Helper function to process CalleeGuids and create value profile metadata
418static void addVPMetadata(Module &M, Instruction &I,
419 ArrayRef<GlobalValue::GUID> CalleeGuids) {
420 if (!ClMemProfAttachCalleeGuids || CalleeGuids.empty())
421 return;
422
423 // Prepare the vector of value data, initializing from any existing
424 // value-profile metadata present on the instruction so that we merge the
425 // new CalleeGuids into the existing entries.
426 SmallVector<InstrProfValueData> VDs;
427 uint64_t TotalCount = 0;
428
429 if (I.getMetadata(KindID: LLVMContext::MD_prof)) {
430 // Read all existing entries so we can merge them. Use a large
431 // MaxNumValueData to retrieve all existing entries.
432 VDs = getValueProfDataFromInst(Inst: I, ValueKind: IPVK_IndirectCallTarget,
433 /*MaxNumValueData=*/UINT32_MAX, TotalC&: TotalCount);
434 }
435
436 // Save the original size for use later in detecting whether any were added.
437 const size_t OriginalSize = VDs.size();
438
439 // Initialize the set of existing guids with the original list.
440 DenseSet<uint64_t> ExistingValues(
441 llvm::from_range,
442 llvm::map_range(
443 C&: VDs, F: [](const InstrProfValueData &Entry) { return Entry.Value; }));
444
445 // Merge CalleeGuids into list of existing VDs, by appending any that are not
446 // already included.
447 VDs.reserve(N: OriginalSize + CalleeGuids.size());
448 for (auto G : CalleeGuids) {
449 if (!ExistingValues.insert(V: G).second)
450 continue;
451 InstrProfValueData NewEntry;
452 NewEntry.Value = G;
453 // For MemProf, we don't have actual call counts, so we assign
454 // a weight of 1 to each potential target.
455 // TODO: Consider making this weight configurable or increasing it to
456 // improve effectiveness for ICP.
457 NewEntry.Count = 1;
458 TotalCount += NewEntry.Count;
459 VDs.push_back(Elt: NewEntry);
460 }
461
462 // Update the VP metadata if we added any new callee GUIDs to the list.
463 assert(VDs.size() >= OriginalSize);
464 if (VDs.size() == OriginalSize)
465 return;
466
467 // First clear the existing !prof.
468 I.setMetadata(KindID: LLVMContext::MD_prof, Node: nullptr);
469
470 // No need to sort the updated VDs as all appended entries have the same count
471 // of 1, which is no larger than any existing entries. The incoming list of
472 // CalleeGuids should already be deterministic for a given profile.
473 annotateValueSite(M, Inst&: I, VDs, Sum: TotalCount, ValueKind: IPVK_IndirectCallTarget, MaxMDCount: VDs.size());
474}
475
476static void handleAllocSite(
477 Instruction &I, CallBase *CI, ArrayRef<uint64_t> InlinedCallStack,
478 LLVMContext &Ctx, OptimizationRemarkEmitter &ORE, uint64_t MaxColdSize,
479 const std::set<const AllocationInfo *> &AllocInfoSet,
480 std::map<uint64_t, AllocMatchInfo> &FullStackIdToAllocMatchInfo) {
481 // TODO: Remove this once the profile creation logic deduplicates contexts
482 // that are the same other than the IsInlineFrame bool. Until then, keep the
483 // largest.
484 DenseMap<uint64_t, const AllocationInfo *> UniqueFullContextIdAllocInfo;
485 for (auto *AllocInfo : AllocInfoSet) {
486 auto FullStackId = computeFullStackId(CallStack: AllocInfo->CallStack);
487 auto [It, Inserted] =
488 UniqueFullContextIdAllocInfo.insert(KV: {FullStackId, AllocInfo});
489 // If inserted entry, done.
490 if (Inserted)
491 continue;
492 // Keep the larger one, or the noncold one if they are the same size.
493 auto CurSize = It->second->Info.getTotalSize();
494 auto NewSize = AllocInfo->Info.getTotalSize();
495 if ((CurSize > NewSize) ||
496 (CurSize == NewSize &&
497 getAllocType(AllocInfo) != AllocationType::NotCold))
498 continue;
499 It->second = AllocInfo;
500 }
501 // We may match this instruction's location list to multiple MIB
502 // contexts. Add them to a Trie specialized for trimming the contexts to
503 // the minimal needed to disambiguate contexts with unique behavior.
504 CallStackTrie AllocTrie(&ORE, MaxColdSize);
505 uint64_t TotalSize = 0;
506 uint64_t TotalColdSize = 0;
507 for (auto &[FullStackId, AllocInfo] : UniqueFullContextIdAllocInfo) {
508 // Check the full inlined call stack against this one.
509 // If we found and thus matched all frames on the call, include
510 // this MIB.
511 if (stackFrameIncludesInlinedCallStack(ProfileCallStack: AllocInfo->CallStack,
512 InlinedCallStack)) {
513 NumOfMemProfMatchedAllocContexts++;
514 auto AllocType = addCallStack(AllocTrie, AllocInfo, FullStackId);
515 TotalSize += AllocInfo->Info.getTotalSize();
516 if (AllocType == AllocationType::Cold)
517 TotalColdSize += AllocInfo->Info.getTotalSize();
518 // Record information about the allocation if match info printing
519 // was requested.
520 if (ClPrintMemProfMatchInfo) {
521 assert(FullStackId != 0);
522 auto [Iter, Inserted] = FullStackIdToAllocMatchInfo.try_emplace(
523 k: FullStackId,
524 args: AllocMatchInfo(AllocInfo->Info.getTotalSize(), AllocType));
525 // Always insert the new matched frame count, since it may differ.
526 Iter->second.MatchedFramesSet.insert(x: InlinedCallStack.size());
527 if (Inserted && PrintMatchedAllocStack)
528 Iter->second.CallStack.insert(position: Iter->second.CallStack.begin(),
529 first: AllocInfo->CallStack.begin(),
530 last: AllocInfo->CallStack.end());
531 }
532 ORE.emit(
533 OptDiag: OptimizationRemark(DEBUG_TYPE, "MemProfUse", CI)
534 << ore::NV("AllocationCall", CI) << " in function "
535 << ore::NV("Caller", CI->getFunction())
536 << " matched alloc context with alloc type "
537 << ore::NV("Attribute", getAllocTypeAttributeString(Type: AllocType))
538 << " total size " << ore::NV("Size", AllocInfo->Info.getTotalSize())
539 << " full context id " << ore::NV("Context", FullStackId)
540 << " frame count " << ore::NV("Frames", InlinedCallStack.size()));
541 }
542 }
543 // If the threshold for the percent of cold bytes is less than 100%,
544 // and not all bytes are cold, see if we should still hint this
545 // allocation as cold without context sensitivity.
546 if (TotalColdSize < TotalSize && MinMatchedColdBytePercent < 100 &&
547 TotalColdSize * 100 >= MinMatchedColdBytePercent * TotalSize) {
548 AllocTrie.addSingleAllocTypeAttribute(CI, AT: AllocationType::Cold, Descriptor: "dominant");
549 return;
550 }
551
552 // We might not have matched any to the full inlined call stack.
553 // But if we did, create and attach metadata, or a function attribute if
554 // all contexts have identical profiled behavior.
555 if (!AllocTrie.empty()) {
556 NumOfMemProfMatchedAllocs++;
557 // MemprofMDAttached will be false if a function attribute was
558 // attached.
559 bool MemprofMDAttached = AllocTrie.buildAndAttachMIBMetadata(CI);
560 assert(MemprofMDAttached == I.hasMetadata(LLVMContext::MD_memprof));
561 if (MemprofMDAttached) {
562 // Add callsite metadata for the instruction's location list so that
563 // it simpler later on to identify which part of the MIB contexts
564 // are from this particular instruction (including during inlining,
565 // when the callsite metadata will be updated appropriately).
566 // FIXME: can this be changed to strip out the matching stack
567 // context ids from the MIB contexts and not add any callsite
568 // metadata here to save space?
569 addCallsiteMetadata(I, InlinedCallStack, Ctx);
570 }
571 }
572}
573
574// Helper struct for maintaining refs to callsite data. As an alternative we
575// could store a pointer to the CallSiteInfo struct but we also need the frame
576// index. Using ArrayRefs instead makes it a little easier to read.
577struct CallSiteEntry {
578 // Subset of frames for the corresponding CallSiteInfo.
579 ArrayRef<Frame> Frames;
580 // Potential targets for indirect calls.
581 ArrayRef<GlobalValue::GUID> CalleeGuids;
582};
583
584static void handleCallSite(Instruction &I, const Function *CalledFunction,
585 ArrayRef<uint64_t> InlinedCallStack,
586 const std::vector<CallSiteEntry> &CallSiteEntries,
587 Module &M,
588 std::set<std::vector<uint64_t>> &MatchedCallSites,
589 OptimizationRemarkEmitter &ORE) {
590 auto &Ctx = M.getContext();
591 // Set of Callee GUIDs to attach to indirect calls. We accumulate all of them
592 // to support cases where the instuction's inlined frames match multiple call
593 // site entries, which can happen if the profile was collected from a binary
594 // where this instruction was eventually inlined into multiple callers.
595 SetVector<GlobalValue::GUID> CalleeGuids;
596 bool CallsiteMDAdded = false;
597 for (const auto &CallSiteEntry : CallSiteEntries) {
598 // If we found and thus matched all frames on the call, create and
599 // attach call stack metadata.
600 if (stackFrameIncludesInlinedCallStack(ProfileCallStack: CallSiteEntry.Frames,
601 InlinedCallStack)) {
602 NumOfMemProfMatchedCallSites++;
603 // Only need to find one with a matching call stack and add a single
604 // callsite metadata.
605 if (!CallsiteMDAdded) {
606 addCallsiteMetadata(I, InlinedCallStack, Ctx);
607
608 // Accumulate call site matching information upon request.
609 if (ClPrintMemProfMatchInfo) {
610 std::vector<uint64_t> CallStack;
611 append_range(C&: CallStack, R&: InlinedCallStack);
612 MatchedCallSites.insert(x: std::move(CallStack));
613 }
614 OptimizationRemark Remark(DEBUG_TYPE, "MemProfUse", &I);
615 Remark << ore::NV("CallSite", &I) << " in function "
616 << ore::NV("Caller", I.getFunction())
617 << " matched callsite with frame count "
618 << ore::NV("Frames", InlinedCallStack.size())
619 << " and stack ids";
620 for (uint64_t StackId : InlinedCallStack)
621 Remark << " " << ore::NV("StackId", StackId);
622 ORE.emit(OptDiag&: Remark);
623
624 // If this is a direct call, we're done.
625 if (CalledFunction)
626 break;
627 CallsiteMDAdded = true;
628 }
629
630 assert(!CalledFunction && "Didn't expect direct call");
631
632 // Collect Callee GUIDs from all matching CallSiteEntries.
633 CalleeGuids.insert(Start: CallSiteEntry.CalleeGuids.begin(),
634 End: CallSiteEntry.CalleeGuids.end());
635 }
636 }
637 // Try to attach indirect call metadata if possible.
638 addVPMetadata(M, I, CalleeGuids: CalleeGuids.getArrayRef());
639}
640
641// Dump inline call stack for debugging purposes.
642static void dumpInlineCallStack(Instruction &I, CallBase *CI,
643 OptimizationRemarkEmitter &ORE,
644 DenseSet<uint64_t> &SeenFrames,
645 DenseSet<uint64_t> &SeenStacks,
646 bool ProfileHasColumns) {
647 // Dump frame info. Frames are deduplicated using FrameID.
648 std::string CallStack;
649 raw_string_ostream CallStackOS(CallStack);
650 bool First = true;
651 for (const DILocation *DIL = I.getDebugLoc(); DIL;
652 DIL = DIL->getInlinedAt()) {
653 StringRef Name = DIL->getScope()->getSubprogram()->getLinkageName();
654 if (Name.empty())
655 Name = DIL->getScope()->getSubprogram()->getName();
656 auto CalleeGUID = Function::getGUIDAssumingExternalLinkage(GlobalName: Name);
657 uint64_t FrameID = computeStackId(Function: CalleeGUID, LineOffset: getLineOffset(DIL),
658 Column: ProfileHasColumns ? DIL->getColumn() : 0);
659 if (SeenFrames.insert(V: FrameID).second) {
660 std::string DictMsg;
661 raw_string_ostream DictOS(DictMsg);
662 DictOS << "frame: " << FrameID << " " << Name << ":" << getLineOffset(DIL)
663 << ":" << (ProfileHasColumns ? DIL->getColumn() : 0);
664 ORE.emit(OptDiag: OptimizationRemarkAnalysis(DEBUG_TYPE, "MemProfUse", CI)
665 << DictOS.str());
666 }
667
668 if (First)
669 First = false;
670 else
671 CallStackOS << ",";
672 CallStackOS << FrameID;
673 }
674
675 // Dump inline call stack info. Stacks are deduplicated using StackHash.
676 uint64_t StackHash = llvm::MD5Hash(Str: CallStack);
677 if (SeenStacks.insert(V: StackHash).second) {
678 std::string Msg;
679 raw_string_ostream OS(Msg);
680 OS << "inline call stack: " << CallStack;
681 ORE.emit(OptDiag: OptimizationRemarkAnalysis(DEBUG_TYPE, "MemProfUse", CI)
682 << OS.str());
683 }
684}
685
686static void
687readMemprof(Module &M, Function &F, IndexedInstrProfReader *MemProfReader,
688 const TargetLibraryInfo &TLI,
689 std::map<uint64_t, AllocMatchInfo> &FullStackIdToAllocMatchInfo,
690 std::set<std::vector<uint64_t>> &MatchedCallSites,
691 DenseMap<uint64_t, LocToLocMap> &UndriftMaps,
692 OptimizationRemarkEmitter &ORE, uint64_t MaxColdSize,
693 DenseSet<uint64_t> &SeenStacks, DenseSet<uint64_t> &SeenFrames) {
694 auto &Ctx = M.getContext();
695 // Previously we used getIRPGOObjectName() here. If F is local linkage,
696 // getIRPGOObjectName() returns FuncName with prefix 'FileName;'. But
697 // llvm-profdata uses FuncName in dwarf to create GUID which doesn't
698 // contain FileName's prefix. It caused local linkage function can't
699 // find MemProfRecord. So we use getName() now.
700 // 'unique-internal-linkage-names' can make MemProf work better for local
701 // linkage function.
702 auto FuncName = F.getName();
703 auto FuncGUID = Function::getGUIDAssumingExternalLinkage(GlobalName: FuncName);
704 if (PrintFunctionGuids)
705 errs() << "MemProf: Function GUID " << FuncGUID << " is " << FuncName
706 << "\n";
707 std::optional<memprof::MemProfRecord> MemProfRec;
708 auto Err = MemProfReader->getMemProfRecord(FuncNameHash: FuncGUID).moveInto(Value&: MemProfRec);
709 if (Err) {
710 handleAllErrors(E: std::move(Err), Handlers: [&](const InstrProfError &IPE) {
711 auto Err = IPE.get();
712 bool SkipWarning = false;
713 LLVM_DEBUG(dbgs() << "Error in reading profile for Func " << FuncName
714 << ": ");
715 if (Err == instrprof_error::unknown_function) {
716 NumOfMemProfMissing++;
717 SkipWarning = !PGOWarnMissing;
718 LLVM_DEBUG(dbgs() << "unknown function");
719 } else if (Err == instrprof_error::hash_mismatch) {
720 NumOfMemProfMismatch++;
721 SkipWarning =
722 NoPGOWarnMismatch ||
723 (NoPGOWarnMismatchComdatWeak &&
724 (F.hasComdat() ||
725 F.getLinkage() == GlobalValue::AvailableExternallyLinkage));
726 LLVM_DEBUG(dbgs() << "hash mismatch (skip=" << SkipWarning << ")");
727 }
728
729 if (SkipWarning)
730 return;
731
732 std::string Msg = (IPE.message() + Twine(" ") + F.getName().str() +
733 Twine(" Hash = ") + std::to_string(val: FuncGUID))
734 .str();
735
736 Ctx.diagnose(
737 DI: DiagnosticInfoPGOProfile(M.getName().data(), Msg, DS_Warning));
738 });
739 return;
740 }
741
742 NumOfMemProfFunc++;
743
744 // If requested, undrfit MemProfRecord so that the source locations in it
745 // match those in the IR.
746 if (SalvageStaleProfile)
747 undriftMemProfRecord(UndriftMaps, MemProfRec&: *MemProfRec);
748
749 // Detect if there are non-zero column numbers in the profile. If not,
750 // treat all column numbers as 0 when matching (i.e. ignore any non-zero
751 // columns in the IR). The profiled binary might have been built with
752 // column numbers disabled, for example.
753 bool ProfileHasColumns = false;
754
755 // Build maps of the location hash to all profile data with that leaf location
756 // (allocation info and the callsites).
757 std::map<uint64_t, std::set<const AllocationInfo *>> LocHashToAllocInfo;
758
759 // For the callsites we need to record slices of the frame array (see comments
760 // below where the map entries are added) along with their CalleeGuids.
761 std::map<uint64_t, std::vector<CallSiteEntry>> LocHashToCallSites;
762 for (auto &AI : MemProfRec->AllocSites) {
763 NumOfMemProfAllocContextProfiles++;
764 // Associate the allocation info with the leaf frame. The later matching
765 // code will match any inlined call sequences in the IR with a longer prefix
766 // of call stack frames.
767 uint64_t StackId = computeStackId(Frame: AI.CallStack[0]);
768 LocHashToAllocInfo[StackId].insert(x: &AI);
769 ProfileHasColumns |= AI.CallStack[0].Column;
770 }
771 for (auto &CS : MemProfRec->CallSites) {
772 NumOfMemProfCallSiteProfiles++;
773 // Need to record all frames from leaf up to and including this function,
774 // as any of these may or may not have been inlined at this point.
775 unsigned Idx = 0;
776 for (auto &StackFrame : CS.Frames) {
777 uint64_t StackId = computeStackId(Frame: StackFrame);
778 ArrayRef<Frame> FrameSlice = ArrayRef<Frame>(CS.Frames).drop_front(N: Idx++);
779 // The callee guids for the slice containing all frames (due to the
780 // increment above Idx is now 1) comes from the CalleeGuids recorded in
781 // the CallSite. For the slices not containing the leaf-most frame, the
782 // callee guid is simply the function GUID of the prior frame.
783 LocHashToCallSites[StackId].push_back(
784 x: {.Frames: FrameSlice, .CalleeGuids: (Idx == 1 ? CS.CalleeGuids
785 : ArrayRef<GlobalValue::GUID>(
786 CS.Frames[Idx - 2].Function))});
787
788 ProfileHasColumns |= StackFrame.Column;
789 // Once we find this function, we can stop recording.
790 if (StackFrame.Function == FuncGUID)
791 break;
792 }
793 assert(Idx <= CS.Frames.size() && CS.Frames[Idx - 1].Function == FuncGUID);
794 }
795
796 // Now walk the instructions, looking up the associated profile data using
797 // debug locations.
798 for (auto &BB : F) {
799 for (auto &I : BB) {
800 if (I.isDebugOrPseudoInst())
801 continue;
802 // We are only interested in calls (allocation or interior call stack
803 // context calls).
804 auto *CI = dyn_cast<CallBase>(Val: &I);
805 if (!CI)
806 continue;
807 auto *CalledFunction = CI->getCalledFunction();
808 if (CalledFunction && CalledFunction->isIntrinsic())
809 continue;
810
811 if (ORE.allowExtraAnalysis(DEBUG_TYPE))
812 dumpInlineCallStack(I, CI, ORE, SeenFrames, SeenStacks,
813 ProfileHasColumns);
814
815 // List of call stack ids computed from the location hashes on debug
816 // locations (leaf to inlined at root).
817 SmallVector<uint64_t, 8> InlinedCallStack;
818 // Was the leaf location found in one of the profile maps?
819 bool LeafFound = false;
820 // If leaf was found in a map, iterators pointing to its location in both
821 // of the maps. It might exist in neither, one, or both (the latter case
822 // can happen because we don't currently have discriminators to
823 // distinguish the case when a single line/col maps to both an allocation
824 // and another callsite).
825 auto AllocInfoIter = LocHashToAllocInfo.end();
826 auto CallSitesIter = LocHashToCallSites.end();
827 for (const DILocation *DIL = I.getDebugLoc(); DIL != nullptr;
828 DIL = DIL->getInlinedAt()) {
829 // Use C++ linkage name if possible. Need to compile with
830 // -fdebug-info-for-profiling to get linkage name.
831 StringRef Name = DIL->getScope()->getSubprogram()->getLinkageName();
832 if (Name.empty())
833 Name = DIL->getScope()->getSubprogram()->getName();
834 auto CalleeGUID = Function::getGUIDAssumingExternalLinkage(GlobalName: Name);
835 auto StackId = computeStackId(Function: CalleeGUID, LineOffset: getLineOffset(DIL),
836 Column: ProfileHasColumns ? DIL->getColumn() : 0);
837 // Check if we have found the profile's leaf frame. If yes, collect
838 // the rest of the call's inlined context starting here. If not, see if
839 // we find a match further up the inlined context (in case the profile
840 // was missing debug frames at the leaf).
841 if (!LeafFound) {
842 AllocInfoIter = LocHashToAllocInfo.find(x: StackId);
843 CallSitesIter = LocHashToCallSites.find(x: StackId);
844 if (AllocInfoIter != LocHashToAllocInfo.end() ||
845 CallSitesIter != LocHashToCallSites.end())
846 LeafFound = true;
847 }
848 if (LeafFound)
849 InlinedCallStack.push_back(Elt: StackId);
850 }
851 // If leaf not in either of the maps, skip inst.
852 if (!LeafFound)
853 continue;
854
855 // First add !memprof metadata from allocation info, if we found the
856 // instruction's leaf location in that map, and if the rest of the
857 // instruction's locations match the prefix Frame locations on an
858 // allocation context with the same leaf.
859 if (AllocInfoIter != LocHashToAllocInfo.end() &&
860 // Only consider allocations which support hinting.
861 isAllocationWithHotColdVariant(Callee: CI->getCalledFunction(), TLI))
862 handleAllocSite(I, CI, InlinedCallStack, Ctx, ORE, MaxColdSize,
863 AllocInfoSet: AllocInfoIter->second, FullStackIdToAllocMatchInfo);
864 else if (CallSitesIter != LocHashToCallSites.end())
865 // Otherwise, add callsite metadata. If we reach here then we found the
866 // instruction's leaf location in the callsites map and not the
867 // allocation map.
868 handleCallSite(I, CalledFunction, InlinedCallStack,
869 CallSiteEntries: CallSitesIter->second, M, MatchedCallSites, ORE);
870 }
871 }
872}
873
874MemProfUsePass::MemProfUsePass(std::string MemoryProfileFile,
875 IntrusiveRefCntPtr<vfs::FileSystem> FS)
876 : MemoryProfileFileName(MemoryProfileFile), FS(FS) {
877 if (!FS)
878 this->FS = vfs::getRealFileSystem();
879}
880
881PreservedAnalyses MemProfUsePass::run(Module &M, ModuleAnalysisManager &AM) {
882 // Return immediately if the module doesn't contain any function or global
883 // variables.
884 if (M.empty() && M.globals().empty())
885 return PreservedAnalyses::all();
886
887 LLVM_DEBUG(dbgs() << "Read in memory profile:\n");
888 auto &Ctx = M.getContext();
889 auto ReaderOrErr = IndexedInstrProfReader::create(Path: MemoryProfileFileName, FS&: *FS);
890 if (Error E = ReaderOrErr.takeError()) {
891 handleAllErrors(E: std::move(E), Handlers: [&](const ErrorInfoBase &EI) {
892 Ctx.diagnose(
893 DI: DiagnosticInfoPGOProfile(MemoryProfileFileName.data(), EI.message()));
894 });
895 return PreservedAnalyses::all();
896 }
897
898 std::unique_ptr<IndexedInstrProfReader> MemProfReader =
899 std::move(ReaderOrErr.get());
900 if (!MemProfReader) {
901 Ctx.diagnose(DI: DiagnosticInfoPGOProfile(
902 MemoryProfileFileName.data(), StringRef("Cannot get MemProfReader")));
903 return PreservedAnalyses::all();
904 }
905
906 if (!MemProfReader->hasMemoryProfile()) {
907 Ctx.diagnose(DI: DiagnosticInfoPGOProfile(MemoryProfileFileName.data(),
908 "Not a memory profile"));
909 return PreservedAnalyses::all();
910 }
911
912 const bool Changed =
913 annotateGlobalVariables(M, DataAccessProf: MemProfReader->getDataAccessProfileData());
914
915 // If the module doesn't contain any function, return after we process all
916 // global variables.
917 if (M.empty())
918 return Changed ? PreservedAnalyses::none() : PreservedAnalyses::all();
919
920 auto &FAM = AM.getResult<FunctionAnalysisManagerModuleProxy>(IR&: M).getManager();
921
922 TargetLibraryInfo &TLI = FAM.getResult<TargetLibraryAnalysis>(IR&: *M.begin());
923 DenseMap<uint64_t, LocToLocMap> UndriftMaps;
924 if (SalvageStaleProfile)
925 UndriftMaps = computeUndriftMap(M, MemProfReader: MemProfReader.get(), TLI);
926
927 // Map from the stack hash of each matched allocation context in the function
928 // profiles to match info such as the total profiled size (bytes), allocation
929 // type, number of frames matched to the allocation itself, and the full array
930 // of call stack ids.
931 std::map<uint64_t, AllocMatchInfo> FullStackIdToAllocMatchInfo;
932
933 // Set of the matched call sites, each expressed as a sequence of an inline
934 // call stack.
935 std::set<std::vector<uint64_t>> MatchedCallSites;
936
937 DenseSet<uint64_t> SeenStacks;
938 DenseSet<uint64_t> SeenFrames;
939
940 uint64_t MaxColdSize = 0;
941 if (auto *MemProfSum = MemProfReader->getMemProfSummary())
942 MaxColdSize = MemProfSum->getMaxColdTotalSize();
943
944 for (auto &F : M) {
945 if (F.isDeclaration())
946 continue;
947
948 const TargetLibraryInfo &TLI = FAM.getResult<TargetLibraryAnalysis>(IR&: F);
949 auto &ORE = FAM.getResult<OptimizationRemarkEmitterAnalysis>(IR&: F);
950 readMemprof(M, F, MemProfReader: MemProfReader.get(), TLI, FullStackIdToAllocMatchInfo,
951 MatchedCallSites, UndriftMaps, ORE, MaxColdSize, SeenStacks,
952 SeenFrames);
953 }
954
955 if (ClPrintMemProfMatchInfo) {
956 for (const auto &[Id, Info] : FullStackIdToAllocMatchInfo) {
957 for (auto Frames : Info.MatchedFramesSet) {
958 // TODO: To reduce verbosity, should we change the existing message
959 // so that we emit a list of matched frame counts in a single message
960 // about the context (instead of one message per frame count?
961 errs() << "MemProf " << getAllocTypeAttributeString(Type: Info.AllocType)
962 << " context with id " << Id << " has total profiled size "
963 << Info.TotalSize << " is matched with " << Frames << " frames";
964 if (PrintMatchedAllocStack) {
965 errs() << " and call stack";
966 for (auto &F : Info.CallStack)
967 errs() << " " << computeStackId(Frame: F);
968 }
969 errs() << "\n";
970 }
971 }
972
973 for (const auto &CallStack : MatchedCallSites) {
974 errs() << "MemProf callsite match for inline call stack";
975 for (uint64_t StackId : CallStack)
976 errs() << " " << StackId;
977 errs() << "\n";
978 }
979 }
980
981 return PreservedAnalyses::none();
982}
983
984bool MemProfUsePass::annotateGlobalVariables(
985 Module &M, const memprof::DataAccessProfData *DataAccessProf) {
986 if (!AnnotateStaticDataSectionPrefix || M.globals().empty())
987 return false;
988
989 if (!DataAccessProf) {
990 M.addModuleFlag(Behavior: Module::Warning, Key: "EnableDataAccessProf", Val: 0U);
991 // FIXME: Add a diagnostic message without failing the compilation when
992 // data access profile payload is not available.
993 return false;
994 }
995 M.addModuleFlag(Behavior: Module::Warning, Key: "EnableDataAccessProf", Val: 1U);
996
997 bool Changed = false;
998 // Iterate all global variables in the module and annotate them based on
999 // data access profiles. Note it's up to the linker to decide how to map input
1000 // sections to output sections, and one conservative practice is to map
1001 // unlikely-prefixed ones to unlikely output section, and map the rest
1002 // (hot-prefixed or prefix-less) to the canonical output section.
1003 for (GlobalVariable &GVar : M.globals()) {
1004 assert(!GVar.getSectionPrefix().has_value() &&
1005 "GVar shouldn't have section prefix yet");
1006 auto Kind = llvm::memprof::getAnnotationKind(GV: GVar);
1007 if (Kind != llvm::memprof::AnnotationKind::AnnotationOK) {
1008 HandleUnsupportedAnnotationKinds(GVar, Kind);
1009 continue;
1010 }
1011
1012 StringRef Name = GVar.getName();
1013 SymbolHandleRef Handle = SymbolHandleRef(Name);
1014 // Skip string literals as their mangled names don't stay stable across
1015 // binary releases.
1016 if (!AnnotateStringLiteralSectionPrefix)
1017 if (Name.starts_with(Prefix: ".str"))
1018 continue;
1019
1020 if (Name.starts_with(Prefix: ".str")) {
1021 std::optional<uint64_t> Hash = getStringContentHash(GVar);
1022 if (!Hash) {
1023 LLVM_DEBUG(dbgs() << "Cannot compute content hash for string literal "
1024 << Name << "\n");
1025 continue;
1026 }
1027 Handle = SymbolHandleRef(Hash.value());
1028 }
1029
1030 // DataAccessProfRecord's get* methods will canonicalize the name under the
1031 // hood before looking it up, so optimizer doesn't need to do it.
1032 std::optional<DataAccessProfRecord> Record =
1033 DataAccessProf->getProfileRecord(SymID: Handle);
1034 // Annotate a global variable as hot if it has non-zero sampled count, and
1035 // annotate it as cold if it's seen in the profiled binary
1036 // file but doesn't have any access sample.
1037 // For logging, optimization remark emitter requires a llvm::Function, but
1038 // it's not well defined how to associate a global variable with a function.
1039 // So we just print out the static data section prefix in LLVM_DEBUG.
1040 if (Record && Record->AccessCount > 0) {
1041 ++NumOfMemProfHotGlobalVars;
1042 Changed |= GVar.setSectionPrefix("hot");
1043 LLVM_DEBUG(dbgs() << "Global variable " << Name
1044 << " is annotated as hot\n");
1045 } else if (DataAccessProf->isKnownColdSymbol(SymID: Handle)) {
1046 ++NumOfMemProfColdGlobalVars;
1047 Changed |= GVar.setSectionPrefix("unlikely");
1048 Changed = true;
1049 LLVM_DEBUG(dbgs() << "Global variable " << Name
1050 << " is annotated as unlikely\n");
1051 } else {
1052 ++NumOfMemProfUnknownGlobalVars;
1053 LLVM_DEBUG(dbgs() << "Global variable " << Name << " is not annotated\n");
1054 }
1055 }
1056
1057 return Changed;
1058}
1059