1//===- InlineFunction.cpp - Code to perform function inlining -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements inlining of a function into a call site, resolving
10// parameters and the return value as appropriate.
11//
12//===----------------------------------------------------------------------===//
13
14#include "llvm/ADT/DenseMap.h"
15#include "llvm/ADT/STLExtras.h"
16#include "llvm/ADT/SetVector.h"
17#include "llvm/ADT/SmallPtrSet.h"
18#include "llvm/ADT/SmallVector.h"
19#include "llvm/ADT/StringExtras.h"
20#include "llvm/ADT/iterator_range.h"
21#include "llvm/Analysis/AliasAnalysis.h"
22#include "llvm/Analysis/AssumptionCache.h"
23#include "llvm/Analysis/BlockFrequencyInfo.h"
24#include "llvm/Analysis/CallGraph.h"
25#include "llvm/Analysis/CaptureTracking.h"
26#include "llvm/Analysis/CtxProfAnalysis.h"
27#include "llvm/Analysis/IndirectCallVisitor.h"
28#include "llvm/Analysis/InstructionSimplify.h"
29#include "llvm/Analysis/MemoryProfileInfo.h"
30#include "llvm/Analysis/ObjCARCAnalysisUtils.h"
31#include "llvm/Analysis/ObjCARCUtil.h"
32#include "llvm/Analysis/ProfileSummaryInfo.h"
33#include "llvm/Analysis/ValueTracking.h"
34#include "llvm/Analysis/VectorUtils.h"
35#include "llvm/IR/Argument.h"
36#include "llvm/IR/AttributeMask.h"
37#include "llvm/IR/Attributes.h"
38#include "llvm/IR/BasicBlock.h"
39#include "llvm/IR/CFG.h"
40#include "llvm/IR/Constant.h"
41#include "llvm/IR/ConstantRange.h"
42#include "llvm/IR/Constants.h"
43#include "llvm/IR/DataLayout.h"
44#include "llvm/IR/DebugInfo.h"
45#include "llvm/IR/DebugInfoMetadata.h"
46#include "llvm/IR/DebugLoc.h"
47#include "llvm/IR/DerivedTypes.h"
48#include "llvm/IR/Dominators.h"
49#include "llvm/IR/EHPersonalities.h"
50#include "llvm/IR/Function.h"
51#include "llvm/IR/GlobalVariable.h"
52#include "llvm/IR/IRBuilder.h"
53#include "llvm/IR/InlineAsm.h"
54#include "llvm/IR/InstrTypes.h"
55#include "llvm/IR/Instruction.h"
56#include "llvm/IR/Instructions.h"
57#include "llvm/IR/IntrinsicInst.h"
58#include "llvm/IR/Intrinsics.h"
59#include "llvm/IR/LLVMContext.h"
60#include "llvm/IR/MDBuilder.h"
61#include "llvm/IR/Metadata.h"
62#include "llvm/IR/Module.h"
63#include "llvm/IR/PatternMatch.h"
64#include "llvm/IR/ProfDataUtils.h"
65#include "llvm/IR/Type.h"
66#include "llvm/IR/User.h"
67#include "llvm/IR/Value.h"
68#include "llvm/Support/Casting.h"
69#include "llvm/Support/CommandLine.h"
70#include "llvm/Support/ErrorHandling.h"
71#include "llvm/Transforms/Utils/AssumeBundleBuilder.h"
72#include "llvm/Transforms/Utils/Cloning.h"
73#include "llvm/Transforms/Utils/Local.h"
74#include "llvm/Transforms/Utils/ValueMapper.h"
75#include <algorithm>
76#include <cassert>
77#include <cstdint>
78#include <deque>
79#include <iterator>
80#include <optional>
81#include <string>
82#include <utility>
83#include <vector>
84
85#define DEBUG_TYPE "inline-function"
86
87using namespace llvm;
88using namespace llvm::memprof;
89
90static cl::opt<bool>
91EnableNoAliasConversion("enable-noalias-to-md-conversion", cl::init(Val: true),
92 cl::Hidden,
93 cl::desc("Convert noalias attributes to metadata during inlining."));
94
95static cl::opt<bool>
96 UseNoAliasIntrinsic("use-noalias-intrinsic-during-inlining", cl::Hidden,
97 cl::init(Val: true),
98 cl::desc("Use the llvm.experimental.noalias.scope.decl "
99 "intrinsic during inlining."));
100
101// Disabled by default, because the added alignment assumptions may increase
102// compile-time and block optimizations. This option is not suitable for use
103// with frontends that emit comprehensive parameter alignment annotations.
104static cl::opt<bool>
105PreserveAlignmentAssumptions("preserve-alignment-assumptions-during-inlining",
106 cl::init(Val: false), cl::Hidden,
107 cl::desc("Convert align attributes to assumptions during inlining."));
108
109static cl::opt<unsigned> InlinerAttributeWindow(
110 "max-inst-checked-for-throw-during-inlining", cl::Hidden,
111 cl::desc("the maximum number of instructions analyzed for may throw during "
112 "attribute inference in inlined body"),
113 cl::init(Val: 4));
114
115namespace {
116
117 /// A class for recording information about inlining a landing pad.
118 class LandingPadInliningInfo {
119 /// Destination of the invoke's unwind.
120 BasicBlock *OuterResumeDest;
121
122 /// Destination for the callee's resume.
123 BasicBlock *InnerResumeDest = nullptr;
124
125 /// LandingPadInst associated with the invoke.
126 LandingPadInst *CallerLPad = nullptr;
127
128 /// PHI for EH values from landingpad insts.
129 PHINode *InnerEHValuesPHI = nullptr;
130
131 SmallVector<Value*, 8> UnwindDestPHIValues;
132
133 public:
134 LandingPadInliningInfo(InvokeInst *II)
135 : OuterResumeDest(II->getUnwindDest()) {
136 // If there are PHI nodes in the unwind destination block, we need to keep
137 // track of which values came into them from the invoke before removing
138 // the edge from this block.
139 BasicBlock *InvokeBB = II->getParent();
140 BasicBlock::iterator I = OuterResumeDest->begin();
141 for (; isa<PHINode>(Val: I); ++I) {
142 // Save the value to use for this edge.
143 PHINode *PHI = cast<PHINode>(Val&: I);
144 UnwindDestPHIValues.push_back(Elt: PHI->getIncomingValueForBlock(BB: InvokeBB));
145 }
146
147 CallerLPad = cast<LandingPadInst>(Val&: I);
148 }
149
150 /// The outer unwind destination is the target of
151 /// unwind edges introduced for calls within the inlined function.
152 BasicBlock *getOuterResumeDest() const {
153 return OuterResumeDest;
154 }
155
156 BasicBlock *getInnerResumeDest();
157
158 LandingPadInst *getLandingPadInst() const { return CallerLPad; }
159
160 /// Forward the 'resume' instruction to the caller's landing pad block.
161 /// When the landing pad block has only one predecessor, this is
162 /// a simple branch. When there is more than one predecessor, we need to
163 /// split the landing pad block after the landingpad instruction and jump
164 /// to there.
165 void forwardResume(ResumeInst *RI,
166 SmallPtrSetImpl<LandingPadInst*> &InlinedLPads);
167
168 /// Add incoming-PHI values to the unwind destination block for the given
169 /// basic block, using the values for the original invoke's source block.
170 void addIncomingPHIValuesFor(BasicBlock *BB) const {
171 addIncomingPHIValuesForInto(src: BB, dest: OuterResumeDest);
172 }
173
174 void addIncomingPHIValuesForInto(BasicBlock *src, BasicBlock *dest) const {
175 BasicBlock::iterator I = dest->begin();
176 for (unsigned i = 0, e = UnwindDestPHIValues.size(); i != e; ++i, ++I) {
177 PHINode *phi = cast<PHINode>(Val&: I);
178 phi->addIncoming(V: UnwindDestPHIValues[i], BB: src);
179 }
180 }
181 };
182} // end anonymous namespace
183
184static IntrinsicInst *getConvergenceEntry(BasicBlock &BB) {
185 BasicBlock::iterator It = BB.getFirstNonPHIIt();
186 while (It != BB.end()) {
187 if (auto *IntrinsicCall = dyn_cast<ConvergenceControlInst>(Val&: It)) {
188 if (IntrinsicCall->isEntry()) {
189 return IntrinsicCall;
190 }
191 }
192 It = std::next(x: It);
193 }
194 return nullptr;
195}
196
197/// Get or create a target for the branch from ResumeInsts.
198BasicBlock *LandingPadInliningInfo::getInnerResumeDest() {
199 if (InnerResumeDest) return InnerResumeDest;
200
201 // Split the landing pad.
202 BasicBlock::iterator SplitPoint = ++CallerLPad->getIterator();
203 InnerResumeDest =
204 OuterResumeDest->splitBasicBlock(I: SplitPoint,
205 BBName: OuterResumeDest->getName() + ".body");
206
207 // The number of incoming edges we expect to the inner landing pad.
208 const unsigned PHICapacity = 2;
209
210 // Create corresponding new PHIs for all the PHIs in the outer landing pad.
211 BasicBlock::iterator InsertPoint = InnerResumeDest->begin();
212 BasicBlock::iterator I = OuterResumeDest->begin();
213 for (unsigned i = 0, e = UnwindDestPHIValues.size(); i != e; ++i, ++I) {
214 PHINode *OuterPHI = cast<PHINode>(Val&: I);
215 PHINode *InnerPHI = PHINode::Create(Ty: OuterPHI->getType(), NumReservedValues: PHICapacity,
216 NameStr: OuterPHI->getName() + ".lpad-body");
217 InnerPHI->insertBefore(InsertPos: InsertPoint);
218 OuterPHI->replaceAllUsesWith(V: InnerPHI);
219 InnerPHI->addIncoming(V: OuterPHI, BB: OuterResumeDest);
220 }
221
222 // Create a PHI for the exception values.
223 InnerEHValuesPHI =
224 PHINode::Create(Ty: CallerLPad->getType(), NumReservedValues: PHICapacity, NameStr: "eh.lpad-body");
225 InnerEHValuesPHI->insertBefore(InsertPos: InsertPoint);
226 CallerLPad->replaceAllUsesWith(V: InnerEHValuesPHI);
227 InnerEHValuesPHI->addIncoming(V: CallerLPad, BB: OuterResumeDest);
228
229 // All done.
230 return InnerResumeDest;
231}
232
233/// Forward the 'resume' instruction to the caller's landing pad block.
234/// When the landing pad block has only one predecessor, this is a simple
235/// branch. When there is more than one predecessor, we need to split the
236/// landing pad block after the landingpad instruction and jump to there.
237void LandingPadInliningInfo::forwardResume(
238 ResumeInst *RI, SmallPtrSetImpl<LandingPadInst *> &InlinedLPads) {
239 BasicBlock *Dest = getInnerResumeDest();
240 BasicBlock *Src = RI->getParent();
241
242 auto *BI = UncondBrInst::Create(Target: Dest, InsertBefore: Src);
243 BI->setDebugLoc(RI->getDebugLoc());
244
245 // Update the PHIs in the destination. They were inserted in an order which
246 // makes this work.
247 addIncomingPHIValuesForInto(src: Src, dest: Dest);
248
249 InnerEHValuesPHI->addIncoming(V: RI->getOperand(i_nocapture: 0), BB: Src);
250 RI->eraseFromParent();
251}
252
253/// Helper for getUnwindDestToken/getUnwindDestTokenHelper.
254static Value *getParentPad(Value *EHPad) {
255 if (auto *FPI = dyn_cast<FuncletPadInst>(Val: EHPad))
256 return FPI->getParentPad();
257 return cast<CatchSwitchInst>(Val: EHPad)->getParentPad();
258}
259
260using UnwindDestMemoTy = DenseMap<Instruction *, Value *>;
261
262/// Helper for getUnwindDestToken that does the descendant-ward part of
263/// the search.
264static Value *getUnwindDestTokenHelper(Instruction *EHPad,
265 UnwindDestMemoTy &MemoMap) {
266 SmallVector<Instruction *, 8> Worklist(1, EHPad);
267
268 while (!Worklist.empty()) {
269 Instruction *CurrentPad = Worklist.pop_back_val();
270 // We only put pads on the worklist that aren't in the MemoMap. When
271 // we find an unwind dest for a pad we may update its ancestors, but
272 // the queue only ever contains uncles/great-uncles/etc. of CurrentPad,
273 // so they should never get updated while queued on the worklist.
274 assert(!MemoMap.count(CurrentPad));
275 Value *UnwindDestToken = nullptr;
276 if (auto *CatchSwitch = dyn_cast<CatchSwitchInst>(Val: CurrentPad)) {
277 if (CatchSwitch->hasUnwindDest()) {
278 UnwindDestToken = &*CatchSwitch->getUnwindDest()->getFirstNonPHIIt();
279 } else {
280 // Catchswitch doesn't have a 'nounwind' variant, and one might be
281 // annotated as "unwinds to caller" when really it's nounwind (see
282 // e.g. SimplifyCFGOpt::SimplifyUnreachable), so we can't infer the
283 // parent's unwind dest from this. We can check its catchpads'
284 // descendants, since they might include a cleanuppad with an
285 // "unwinds to caller" cleanupret, which can be trusted.
286 for (auto HI = CatchSwitch->handler_begin(),
287 HE = CatchSwitch->handler_end();
288 HI != HE && !UnwindDestToken; ++HI) {
289 BasicBlock *HandlerBlock = *HI;
290 auto *CatchPad =
291 cast<CatchPadInst>(Val: &*HandlerBlock->getFirstNonPHIIt());
292 for (User *Child : CatchPad->users()) {
293 // Intentionally ignore invokes here -- since the catchswitch is
294 // marked "unwind to caller", it would be a verifier error if it
295 // contained an invoke which unwinds out of it, so any invoke we'd
296 // encounter must unwind to some child of the catch.
297 if (!isa<CleanupPadInst>(Val: Child) && !isa<CatchSwitchInst>(Val: Child))
298 continue;
299
300 Instruction *ChildPad = cast<Instruction>(Val: Child);
301 auto Memo = MemoMap.find(Val: ChildPad);
302 if (Memo == MemoMap.end()) {
303 // Haven't figured out this child pad yet; queue it.
304 Worklist.push_back(Elt: ChildPad);
305 continue;
306 }
307 // We've already checked this child, but might have found that
308 // it offers no proof either way.
309 Value *ChildUnwindDestToken = Memo->second;
310 if (!ChildUnwindDestToken)
311 continue;
312 // We already know the child's unwind dest, which can either
313 // be ConstantTokenNone to indicate unwind to caller, or can
314 // be another child of the catchpad. Only the former indicates
315 // the unwind dest of the catchswitch.
316 if (isa<ConstantTokenNone>(Val: ChildUnwindDestToken)) {
317 UnwindDestToken = ChildUnwindDestToken;
318 break;
319 }
320 assert(getParentPad(ChildUnwindDestToken) == CatchPad);
321 }
322 }
323 }
324 } else {
325 auto *CleanupPad = cast<CleanupPadInst>(Val: CurrentPad);
326 for (User *U : CleanupPad->users()) {
327 if (auto *CleanupRet = dyn_cast<CleanupReturnInst>(Val: U)) {
328 if (BasicBlock *RetUnwindDest = CleanupRet->getUnwindDest())
329 UnwindDestToken = &*RetUnwindDest->getFirstNonPHIIt();
330 else
331 UnwindDestToken = ConstantTokenNone::get(Context&: CleanupPad->getContext());
332 break;
333 }
334 Value *ChildUnwindDestToken;
335 if (auto *Invoke = dyn_cast<InvokeInst>(Val: U)) {
336 ChildUnwindDestToken = &*Invoke->getUnwindDest()->getFirstNonPHIIt();
337 } else if (isa<CleanupPadInst>(Val: U) || isa<CatchSwitchInst>(Val: U)) {
338 Instruction *ChildPad = cast<Instruction>(Val: U);
339 auto Memo = MemoMap.find(Val: ChildPad);
340 if (Memo == MemoMap.end()) {
341 // Haven't resolved this child yet; queue it and keep searching.
342 Worklist.push_back(Elt: ChildPad);
343 continue;
344 }
345 // We've checked this child, but still need to ignore it if it
346 // had no proof either way.
347 ChildUnwindDestToken = Memo->second;
348 if (!ChildUnwindDestToken)
349 continue;
350 } else {
351 // Not a relevant user of the cleanuppad
352 continue;
353 }
354 // In a well-formed program, the child/invoke must either unwind to
355 // an(other) child of the cleanup, or exit the cleanup. In the
356 // first case, continue searching.
357 if (isa<Instruction>(Val: ChildUnwindDestToken) &&
358 getParentPad(EHPad: ChildUnwindDestToken) == CleanupPad)
359 continue;
360 UnwindDestToken = ChildUnwindDestToken;
361 break;
362 }
363 }
364 // If we haven't found an unwind dest for CurrentPad, we may have queued its
365 // children, so move on to the next in the worklist.
366 if (!UnwindDestToken)
367 continue;
368
369 // Now we know that CurrentPad unwinds to UnwindDestToken. It also exits
370 // any ancestors of CurrentPad up to but not including UnwindDestToken's
371 // parent pad. Record this in the memo map, and check to see if the
372 // original EHPad being queried is one of the ones exited.
373 Value *UnwindParent;
374 if (auto *UnwindPad = dyn_cast<Instruction>(Val: UnwindDestToken))
375 UnwindParent = getParentPad(EHPad: UnwindPad);
376 else
377 UnwindParent = nullptr;
378 bool ExitedOriginalPad = false;
379 for (Instruction *ExitedPad = CurrentPad;
380 ExitedPad && ExitedPad != UnwindParent;
381 ExitedPad = dyn_cast<Instruction>(Val: getParentPad(EHPad: ExitedPad))) {
382 // Skip over catchpads since they just follow their catchswitches.
383 if (isa<CatchPadInst>(Val: ExitedPad))
384 continue;
385 MemoMap[ExitedPad] = UnwindDestToken;
386 ExitedOriginalPad |= (ExitedPad == EHPad);
387 }
388
389 if (ExitedOriginalPad)
390 return UnwindDestToken;
391
392 // Continue the search.
393 }
394
395 // No definitive information is contained within this funclet.
396 return nullptr;
397}
398
399/// Given an EH pad, find where it unwinds. If it unwinds to an EH pad,
400/// return that pad instruction. If it unwinds to caller, return
401/// ConstantTokenNone. If it does not have a definitive unwind destination,
402/// return nullptr.
403///
404/// This routine gets invoked for calls in funclets in inlinees when inlining
405/// an invoke. Since many funclets don't have calls inside them, it's queried
406/// on-demand rather than building a map of pads to unwind dests up front.
407/// Determining a funclet's unwind dest may require recursively searching its
408/// descendants, and also ancestors and cousins if the descendants don't provide
409/// an answer. Since most funclets will have their unwind dest immediately
410/// available as the unwind dest of a catchswitch or cleanupret, this routine
411/// searches top-down from the given pad and then up. To avoid worst-case
412/// quadratic run-time given that approach, it uses a memo map to avoid
413/// re-processing funclet trees. The callers that rewrite the IR as they go
414/// take advantage of this, for correctness, by checking/forcing rewritten
415/// pads' entries to match the original callee view.
416static Value *getUnwindDestToken(Instruction *EHPad,
417 UnwindDestMemoTy &MemoMap) {
418 // Catchpads unwind to the same place as their catchswitch;
419 // redirct any queries on catchpads so the code below can
420 // deal with just catchswitches and cleanuppads.
421 if (auto *CPI = dyn_cast<CatchPadInst>(Val: EHPad))
422 EHPad = CPI->getCatchSwitch();
423
424 // Check if we've already determined the unwind dest for this pad.
425 auto Memo = MemoMap.find(Val: EHPad);
426 if (Memo != MemoMap.end())
427 return Memo->second;
428
429 // Search EHPad and, if necessary, its descendants.
430 Value *UnwindDestToken = getUnwindDestTokenHelper(EHPad, MemoMap);
431 assert((UnwindDestToken == nullptr) != (MemoMap.count(EHPad) != 0));
432 if (UnwindDestToken)
433 return UnwindDestToken;
434
435 // No information is available for this EHPad from itself or any of its
436 // descendants. An unwind all the way out to a pad in the caller would
437 // need also to agree with the unwind dest of the parent funclet, so
438 // search up the chain to try to find a funclet with information. Put
439 // null entries in the memo map to avoid re-processing as we go up.
440 MemoMap[EHPad] = nullptr;
441#ifndef NDEBUG
442 SmallPtrSet<Instruction *, 4> TempMemos;
443 TempMemos.insert(EHPad);
444#endif
445 Instruction *LastUselessPad = EHPad;
446 Value *AncestorToken;
447 for (AncestorToken = getParentPad(EHPad);
448 auto *AncestorPad = dyn_cast<Instruction>(Val: AncestorToken);
449 AncestorToken = getParentPad(EHPad: AncestorToken)) {
450 // Skip over catchpads since they just follow their catchswitches.
451 if (isa<CatchPadInst>(Val: AncestorPad))
452 continue;
453 // If the MemoMap had an entry mapping AncestorPad to nullptr, since we
454 // haven't yet called getUnwindDestTokenHelper for AncestorPad in this
455 // call to getUnwindDestToken, that would mean that AncestorPad had no
456 // information in itself, its descendants, or its ancestors. If that
457 // were the case, then we should also have recorded the lack of information
458 // for the descendant that we're coming from. So assert that we don't
459 // find a null entry in the MemoMap for AncestorPad.
460 assert(!MemoMap.count(AncestorPad) || MemoMap[AncestorPad]);
461 auto AncestorMemo = MemoMap.find(Val: AncestorPad);
462 if (AncestorMemo == MemoMap.end()) {
463 UnwindDestToken = getUnwindDestTokenHelper(EHPad: AncestorPad, MemoMap);
464 } else {
465 UnwindDestToken = AncestorMemo->second;
466 }
467 if (UnwindDestToken)
468 break;
469 LastUselessPad = AncestorPad;
470 MemoMap[LastUselessPad] = nullptr;
471#ifndef NDEBUG
472 TempMemos.insert(LastUselessPad);
473#endif
474 }
475
476 // We know that getUnwindDestTokenHelper was called on LastUselessPad and
477 // returned nullptr (and likewise for EHPad and any of its ancestors up to
478 // LastUselessPad), so LastUselessPad has no information from below. Since
479 // getUnwindDestTokenHelper must investigate all downward paths through
480 // no-information nodes to prove that a node has no information like this,
481 // and since any time it finds information it records it in the MemoMap for
482 // not just the immediately-containing funclet but also any ancestors also
483 // exited, it must be the case that, walking downward from LastUselessPad,
484 // visiting just those nodes which have not been mapped to an unwind dest
485 // by getUnwindDestTokenHelper (the nullptr TempMemos notwithstanding, since
486 // they are just used to keep getUnwindDestTokenHelper from repeating work),
487 // any node visited must have been exhaustively searched with no information
488 // for it found.
489 SmallVector<Instruction *, 8> Worklist(1, LastUselessPad);
490 while (!Worklist.empty()) {
491 Instruction *UselessPad = Worklist.pop_back_val();
492 auto Memo = MemoMap.find(Val: UselessPad);
493 if (Memo != MemoMap.end() && Memo->second) {
494 // Here the name 'UselessPad' is a bit of a misnomer, because we've found
495 // that it is a funclet that does have information about unwinding to
496 // a particular destination; its parent was a useless pad.
497 // Since its parent has no information, the unwind edge must not escape
498 // the parent, and must target a sibling of this pad. This local unwind
499 // gives us no information about EHPad. Leave it and the subtree rooted
500 // at it alone.
501 assert(getParentPad(Memo->second) == getParentPad(UselessPad));
502 continue;
503 }
504 // We know we don't have information for UselesPad. If it has an entry in
505 // the MemoMap (mapping it to nullptr), it must be one of the TempMemos
506 // added on this invocation of getUnwindDestToken; if a previous invocation
507 // recorded nullptr, it would have had to prove that the ancestors of
508 // UselessPad, which include LastUselessPad, had no information, and that
509 // in turn would have required proving that the descendants of
510 // LastUselesPad, which include EHPad, have no information about
511 // LastUselessPad, which would imply that EHPad was mapped to nullptr in
512 // the MemoMap on that invocation, which isn't the case if we got here.
513 assert(!MemoMap.count(UselessPad) || TempMemos.count(UselessPad));
514 // Assert as we enumerate users that 'UselessPad' doesn't have any unwind
515 // information that we'd be contradicting by making a map entry for it
516 // (which is something that getUnwindDestTokenHelper must have proved for
517 // us to get here). Just assert on is direct users here; the checks in
518 // this downward walk at its descendants will verify that they don't have
519 // any unwind edges that exit 'UselessPad' either (i.e. they either have no
520 // unwind edges or unwind to a sibling).
521 MemoMap[UselessPad] = UnwindDestToken;
522 if (auto *CatchSwitch = dyn_cast<CatchSwitchInst>(Val: UselessPad)) {
523 assert(CatchSwitch->getUnwindDest() == nullptr && "Expected useless pad");
524 for (BasicBlock *HandlerBlock : CatchSwitch->handlers()) {
525 auto *CatchPad = &*HandlerBlock->getFirstNonPHIIt();
526 for (User *U : CatchPad->users()) {
527 assert((!isa<InvokeInst>(U) ||
528 (getParentPad(&*cast<InvokeInst>(U)
529 ->getUnwindDest()
530 ->getFirstNonPHIIt()) == CatchPad)) &&
531 "Expected useless pad");
532 if (isa<CatchSwitchInst>(Val: U) || isa<CleanupPadInst>(Val: U))
533 Worklist.push_back(Elt: cast<Instruction>(Val: U));
534 }
535 }
536 } else {
537 assert(isa<CleanupPadInst>(UselessPad));
538 for (User *U : UselessPad->users()) {
539 assert(!isa<CleanupReturnInst>(U) && "Expected useless pad");
540 assert(
541 (!isa<InvokeInst>(U) ||
542 (getParentPad(
543 &*cast<InvokeInst>(U)->getUnwindDest()->getFirstNonPHIIt()) ==
544 UselessPad)) &&
545 "Expected useless pad");
546 if (isa<CatchSwitchInst>(Val: U) || isa<CleanupPadInst>(Val: U))
547 Worklist.push_back(Elt: cast<Instruction>(Val: U));
548 }
549 }
550 }
551
552 return UnwindDestToken;
553}
554
555/// When we inline a basic block into an invoke,
556/// we have to turn all of the calls that can throw into invokes.
557/// This function analyze BB to see if there are any calls, and if so,
558/// it rewrites them to be invokes that jump to InvokeDest and fills in the PHI
559/// nodes in that block with the values specified in InvokeDestPHIValues.
560static BasicBlock *HandleCallsInBlockInlinedThroughInvoke(
561 BasicBlock *BB, BasicBlock *UnwindEdge,
562 SmallSetVector<const Value *, 4> &OriginallyIndirectCalls,
563 UnwindDestMemoTy *FuncletUnwindMap = nullptr) {
564 for (Instruction &I : llvm::make_early_inc_range(Range&: *BB)) {
565 // We only need to check for function calls: inlined invoke
566 // instructions require no special handling.
567 CallInst *CI = dyn_cast<CallInst>(Val: &I);
568
569 if (!CI || CI->doesNotThrow())
570 continue;
571
572 // We do not need to (and in fact, cannot) convert possibly throwing calls
573 // to @llvm.experimental_deoptimize (resp. @llvm.experimental.guard) into
574 // invokes. The caller's "segment" of the deoptimization continuation
575 // attached to the newly inlined @llvm.experimental_deoptimize
576 // (resp. @llvm.experimental.guard) call should contain the exception
577 // handling logic, if any.
578 if (auto *F = CI->getCalledFunction())
579 if (F->getIntrinsicID() == Intrinsic::experimental_deoptimize ||
580 F->getIntrinsicID() == Intrinsic::experimental_guard)
581 continue;
582
583 if (auto FuncletBundle = CI->getOperandBundle(ID: LLVMContext::OB_funclet)) {
584 // This call is nested inside a funclet. If that funclet has an unwind
585 // destination within the inlinee, then unwinding out of this call would
586 // be UB. Rewriting this call to an invoke which targets the inlined
587 // invoke's unwind dest would give the call's parent funclet multiple
588 // unwind destinations, which is something that subsequent EH table
589 // generation can't handle and that the veirifer rejects. So when we
590 // see such a call, leave it as a call.
591 auto *FuncletPad = cast<Instruction>(Val: FuncletBundle->Inputs[0]);
592 Value *UnwindDestToken =
593 getUnwindDestToken(EHPad: FuncletPad, MemoMap&: *FuncletUnwindMap);
594 if (UnwindDestToken && !isa<ConstantTokenNone>(Val: UnwindDestToken))
595 continue;
596#ifndef NDEBUG
597 Instruction *MemoKey;
598 if (auto *CatchPad = dyn_cast<CatchPadInst>(FuncletPad))
599 MemoKey = CatchPad->getCatchSwitch();
600 else
601 MemoKey = FuncletPad;
602 assert(FuncletUnwindMap->count(MemoKey) &&
603 (*FuncletUnwindMap)[MemoKey] == UnwindDestToken &&
604 "must get memoized to avoid confusing later searches");
605#endif // NDEBUG
606 }
607
608 bool WasIndirect = OriginallyIndirectCalls.remove(X: CI);
609 changeToInvokeAndSplitBasicBlock(CI, UnwindEdge);
610 if (WasIndirect)
611 OriginallyIndirectCalls.insert(X: BB->getTerminator());
612 return BB;
613 }
614 return nullptr;
615}
616
617/// If we inlined an invoke site, we need to convert calls
618/// in the body of the inlined function into invokes.
619///
620/// II is the invoke instruction being inlined. FirstNewBlock is the first
621/// block of the inlined code (the last block is the end of the function),
622/// and InlineCodeInfo is information about the code that got inlined.
623static void HandleInlinedLandingPad(InvokeInst *II, BasicBlock *FirstNewBlock,
624 ClonedCodeInfo &InlinedCodeInfo) {
625 BasicBlock *InvokeDest = II->getUnwindDest();
626
627 Function *Caller = FirstNewBlock->getParent();
628
629 // The inlined code is currently at the end of the function, scan from the
630 // start of the inlined code to its end, checking for stuff we need to
631 // rewrite.
632 LandingPadInliningInfo Invoke(II);
633
634 // Get all of the inlined landing pad instructions.
635 SmallPtrSet<LandingPadInst*, 16> InlinedLPads;
636 for (Function::iterator I = FirstNewBlock->getIterator(), E = Caller->end();
637 I != E; ++I)
638 if (InvokeInst *II = dyn_cast<InvokeInst>(Val: I->getTerminator()))
639 InlinedLPads.insert(Ptr: II->getLandingPadInst());
640
641 // Append the clauses from the outer landing pad instruction into the inlined
642 // landing pad instructions.
643 LandingPadInst *OuterLPad = Invoke.getLandingPadInst();
644 for (LandingPadInst *InlinedLPad : InlinedLPads) {
645 unsigned OuterNum = OuterLPad->getNumClauses();
646 InlinedLPad->reserveClauses(Size: OuterNum);
647 for (unsigned OuterIdx = 0; OuterIdx != OuterNum; ++OuterIdx)
648 InlinedLPad->addClause(ClauseVal: OuterLPad->getClause(Idx: OuterIdx));
649 if (OuterLPad->isCleanup())
650 InlinedLPad->setCleanup(true);
651 }
652
653 for (Function::iterator BB = FirstNewBlock->getIterator(), E = Caller->end();
654 BB != E; ++BB) {
655 if (InlinedCodeInfo.ContainsCalls)
656 if (BasicBlock *NewBB = HandleCallsInBlockInlinedThroughInvoke(
657 BB: &*BB, UnwindEdge: Invoke.getOuterResumeDest(),
658 OriginallyIndirectCalls&: InlinedCodeInfo.OriginallyIndirectCalls))
659 // Update any PHI nodes in the exceptional block to indicate that there
660 // is now a new entry in them.
661 Invoke.addIncomingPHIValuesFor(BB: NewBB);
662
663 // Forward any resumes that are remaining here.
664 if (ResumeInst *RI = dyn_cast<ResumeInst>(Val: BB->getTerminator()))
665 Invoke.forwardResume(RI, InlinedLPads);
666 }
667
668 // Now that everything is happy, we have one final detail. The PHI nodes in
669 // the exception destination block still have entries due to the original
670 // invoke instruction. Eliminate these entries (which might even delete the
671 // PHI node) now.
672 InvokeDest->removePredecessor(Pred: II->getParent());
673}
674
675/// If we inlined an invoke site, we need to convert calls
676/// in the body of the inlined function into invokes.
677///
678/// II is the invoke instruction being inlined. FirstNewBlock is the first
679/// block of the inlined code (the last block is the end of the function),
680/// and InlineCodeInfo is information about the code that got inlined.
681static void HandleInlinedEHPad(InvokeInst *II, BasicBlock *FirstNewBlock,
682 ClonedCodeInfo &InlinedCodeInfo) {
683 BasicBlock *UnwindDest = II->getUnwindDest();
684 Function *Caller = FirstNewBlock->getParent();
685
686 assert(UnwindDest->getFirstNonPHIIt()->isEHPad() && "unexpected BasicBlock!");
687
688 // If there are PHI nodes in the unwind destination block, we need to keep
689 // track of which values came into them from the invoke before removing the
690 // edge from this block.
691 SmallVector<Value *, 8> UnwindDestPHIValues;
692 BasicBlock *InvokeBB = II->getParent();
693 for (PHINode &PHI : UnwindDest->phis()) {
694 // Save the value to use for this edge.
695 UnwindDestPHIValues.push_back(Elt: PHI.getIncomingValueForBlock(BB: InvokeBB));
696 }
697
698 // Add incoming-PHI values to the unwind destination block for the given basic
699 // block, using the values for the original invoke's source block.
700 auto UpdatePHINodes = [&](BasicBlock *Src) {
701 BasicBlock::iterator I = UnwindDest->begin();
702 for (Value *V : UnwindDestPHIValues) {
703 PHINode *PHI = cast<PHINode>(Val&: I);
704 PHI->addIncoming(V, BB: Src);
705 ++I;
706 }
707 };
708
709 // This connects all the instructions which 'unwind to caller' to the invoke
710 // destination.
711 UnwindDestMemoTy FuncletUnwindMap;
712 for (Function::iterator BB = FirstNewBlock->getIterator(), E = Caller->end();
713 BB != E; ++BB) {
714 if (auto *CRI = dyn_cast<CleanupReturnInst>(Val: BB->getTerminator())) {
715 if (CRI->unwindsToCaller()) {
716 auto *CleanupPad = CRI->getCleanupPad();
717 CleanupReturnInst::Create(CleanupPad, UnwindBB: UnwindDest, InsertBefore: CRI->getIterator());
718 CRI->eraseFromParent();
719 UpdatePHINodes(&*BB);
720 // Finding a cleanupret with an unwind destination would confuse
721 // subsequent calls to getUnwindDestToken, so map the cleanuppad
722 // to short-circuit any such calls and recognize this as an "unwind
723 // to caller" cleanup.
724 assert(!FuncletUnwindMap.count(CleanupPad) ||
725 isa<ConstantTokenNone>(FuncletUnwindMap[CleanupPad]));
726 FuncletUnwindMap[CleanupPad] =
727 ConstantTokenNone::get(Context&: Caller->getContext());
728 }
729 }
730
731 BasicBlock::iterator I = BB->getFirstNonPHIIt();
732 if (!I->isEHPad())
733 continue;
734
735 Instruction *Replacement = nullptr;
736 if (auto *CatchSwitch = dyn_cast<CatchSwitchInst>(Val&: I)) {
737 if (CatchSwitch->unwindsToCaller()) {
738 Value *UnwindDestToken;
739 if (auto *ParentPad =
740 dyn_cast<Instruction>(Val: CatchSwitch->getParentPad())) {
741 // This catchswitch is nested inside another funclet. If that
742 // funclet has an unwind destination within the inlinee, then
743 // unwinding out of this catchswitch would be UB. Rewriting this
744 // catchswitch to unwind to the inlined invoke's unwind dest would
745 // give the parent funclet multiple unwind destinations, which is
746 // something that subsequent EH table generation can't handle and
747 // that the veirifer rejects. So when we see such a call, leave it
748 // as "unwind to caller".
749 UnwindDestToken = getUnwindDestToken(EHPad: ParentPad, MemoMap&: FuncletUnwindMap);
750 if (UnwindDestToken && !isa<ConstantTokenNone>(Val: UnwindDestToken))
751 continue;
752 } else {
753 // This catchswitch has no parent to inherit constraints from, and
754 // none of its descendants can have an unwind edge that exits it and
755 // targets another funclet in the inlinee. It may or may not have a
756 // descendant that definitively has an unwind to caller. In either
757 // case, we'll have to assume that any unwinds out of it may need to
758 // be routed to the caller, so treat it as though it has a definitive
759 // unwind to caller.
760 UnwindDestToken = ConstantTokenNone::get(Context&: Caller->getContext());
761 }
762 auto *NewCatchSwitch = CatchSwitchInst::Create(
763 ParentPad: CatchSwitch->getParentPad(), UnwindDest,
764 NumHandlers: CatchSwitch->getNumHandlers(), NameStr: CatchSwitch->getName(),
765 InsertBefore: CatchSwitch->getIterator());
766 for (BasicBlock *PadBB : CatchSwitch->handlers())
767 NewCatchSwitch->addHandler(Dest: PadBB);
768 // Propagate info for the old catchswitch over to the new one in
769 // the unwind map. This also serves to short-circuit any subsequent
770 // checks for the unwind dest of this catchswitch, which would get
771 // confused if they found the outer handler in the callee.
772 FuncletUnwindMap[NewCatchSwitch] = UnwindDestToken;
773 Replacement = NewCatchSwitch;
774 }
775 } else if (!isa<FuncletPadInst>(Val: I)) {
776 llvm_unreachable("unexpected EHPad!");
777 }
778
779 if (Replacement) {
780 Replacement->takeName(V: &*I);
781 I->replaceAllUsesWith(V: Replacement);
782 I->eraseFromParent();
783 UpdatePHINodes(&*BB);
784 }
785 }
786
787 if (InlinedCodeInfo.ContainsCalls)
788 for (Function::iterator BB = FirstNewBlock->getIterator(),
789 E = Caller->end();
790 BB != E; ++BB)
791 if (BasicBlock *NewBB = HandleCallsInBlockInlinedThroughInvoke(
792 BB: &*BB, UnwindEdge: UnwindDest, OriginallyIndirectCalls&: InlinedCodeInfo.OriginallyIndirectCalls,
793 FuncletUnwindMap: &FuncletUnwindMap))
794 // Update any PHI nodes in the exceptional block to indicate that there
795 // is now a new entry in them.
796 UpdatePHINodes(NewBB);
797
798 // Now that everything is happy, we have one final detail. The PHI nodes in
799 // the exception destination block still have entries due to the original
800 // invoke instruction. Eliminate these entries (which might even delete the
801 // PHI node) now.
802 UnwindDest->removePredecessor(Pred: InvokeBB);
803}
804
805static bool haveCommonPrefix(MDNode *MIBStackContext,
806 MDNode *CallsiteStackContext) {
807 assert(MIBStackContext->getNumOperands() > 0 &&
808 CallsiteStackContext->getNumOperands() > 0);
809 // Because of the context trimming performed during matching, the callsite
810 // context could have more stack ids than the MIB. We match up to the end of
811 // the shortest stack context.
812 for (auto MIBStackIter = MIBStackContext->op_begin(),
813 CallsiteStackIter = CallsiteStackContext->op_begin();
814 MIBStackIter != MIBStackContext->op_end() &&
815 CallsiteStackIter != CallsiteStackContext->op_end();
816 MIBStackIter++, CallsiteStackIter++) {
817 auto *Val1 = mdconst::dyn_extract<ConstantInt>(MD: *MIBStackIter);
818 auto *Val2 = mdconst::dyn_extract<ConstantInt>(MD: *CallsiteStackIter);
819 assert(Val1 && Val2);
820 if (Val1->getZExtValue() != Val2->getZExtValue())
821 return false;
822 }
823 return true;
824}
825
826static void removeMemProfMetadata(CallBase *Call) {
827 Call->setMetadata(KindID: LLVMContext::MD_memprof, Node: nullptr);
828}
829
830static void removeCallsiteMetadata(CallBase *Call) {
831 Call->setMetadata(KindID: LLVMContext::MD_callsite, Node: nullptr);
832}
833
834static void updateMemprofMetadata(CallBase *CI,
835 const std::vector<Metadata *> &MIBList,
836 OptimizationRemarkEmitter *ORE) {
837 assert(!MIBList.empty());
838 // Remove existing memprof, which will either be replaced or may not be needed
839 // if we are able to use a single allocation type function attribute.
840 removeMemProfMetadata(Call: CI);
841 CallStackTrie CallStack(ORE);
842 for (Metadata *MIB : MIBList)
843 CallStack.addCallStack(MIB: cast<MDNode>(Val: MIB));
844 bool MemprofMDAttached = CallStack.buildAndAttachMIBMetadata(CI);
845 assert(MemprofMDAttached == CI->hasMetadata(LLVMContext::MD_memprof));
846 if (!MemprofMDAttached)
847 // If we used a function attribute remove the callsite metadata as well.
848 removeCallsiteMetadata(Call: CI);
849}
850
851// Update the metadata on the inlined copy ClonedCall of a call OrigCall in the
852// inlined callee body, based on the callsite metadata InlinedCallsiteMD from
853// the call that was inlined.
854static void propagateMemProfHelper(const CallBase *OrigCall,
855 CallBase *ClonedCall,
856 MDNode *InlinedCallsiteMD,
857 OptimizationRemarkEmitter *ORE) {
858 MDNode *OrigCallsiteMD = ClonedCall->getMetadata(KindID: LLVMContext::MD_callsite);
859 MDNode *ClonedCallsiteMD = nullptr;
860 // Check if the call originally had callsite metadata, and update it for the
861 // new call in the inlined body.
862 if (OrigCallsiteMD) {
863 // The cloned call's context is now the concatenation of the original call's
864 // callsite metadata and the callsite metadata on the call where it was
865 // inlined.
866 ClonedCallsiteMD = MDNode::concatenate(A: OrigCallsiteMD, B: InlinedCallsiteMD);
867 ClonedCall->setMetadata(KindID: LLVMContext::MD_callsite, Node: ClonedCallsiteMD);
868 }
869
870 // Update any memprof metadata on the cloned call.
871 MDNode *OrigMemProfMD = ClonedCall->getMetadata(KindID: LLVMContext::MD_memprof);
872 if (!OrigMemProfMD)
873 return;
874 // We currently expect that allocations with memprof metadata also have
875 // callsite metadata for the allocation's part of the context.
876 assert(OrigCallsiteMD);
877
878 // New call's MIB list.
879 std::vector<Metadata *> NewMIBList;
880
881 // For each MIB metadata, check if its call stack context starts with the
882 // new clone's callsite metadata. If so, that MIB goes onto the cloned call in
883 // the inlined body. If not, it stays on the out-of-line original call.
884 for (auto &MIBOp : OrigMemProfMD->operands()) {
885 MDNode *MIB = dyn_cast<MDNode>(Val: MIBOp);
886 // Stack is first operand of MIB.
887 MDNode *StackMD = getMIBStackNode(MIB);
888 assert(StackMD);
889 // See if the new cloned callsite context matches this profiled context.
890 if (haveCommonPrefix(MIBStackContext: StackMD, CallsiteStackContext: ClonedCallsiteMD))
891 // Add it to the cloned call's MIB list.
892 NewMIBList.push_back(x: MIB);
893 }
894 if (NewMIBList.empty()) {
895 removeMemProfMetadata(Call: ClonedCall);
896 removeCallsiteMetadata(Call: ClonedCall);
897 return;
898 }
899 if (NewMIBList.size() < OrigMemProfMD->getNumOperands())
900 updateMemprofMetadata(CI: ClonedCall, MIBList: NewMIBList, ORE);
901}
902
903// Update memprof related metadata (!memprof and !callsite) based on the
904// inlining of Callee into the callsite at CB. The updates include merging the
905// inlined callee's callsite metadata with that of the inlined call,
906// and moving the subset of any memprof contexts to the inlined callee
907// allocations if they match the new inlined call stack.
908static void
909propagateMemProfMetadata(Function *Callee, CallBase &CB,
910 bool ContainsMemProfMetadata,
911 const ValueMap<const Value *, WeakTrackingVH> &VMap,
912 OptimizationRemarkEmitter *ORE) {
913 MDNode *CallsiteMD = CB.getMetadata(KindID: LLVMContext::MD_callsite);
914 // Only need to update if the inlined callsite had callsite metadata, or if
915 // there was any memprof metadata inlined.
916 if (!CallsiteMD && !ContainsMemProfMetadata)
917 return;
918
919 // Propagate metadata onto the cloned calls in the inlined callee.
920 for (const auto &Entry : VMap) {
921 // See if this is a call that has been inlined and remapped, and not
922 // simplified away in the process.
923 auto *OrigCall = dyn_cast_or_null<CallBase>(Val: Entry.first);
924 auto *ClonedCall = dyn_cast_or_null<CallBase>(Val: Entry.second);
925 if (!OrigCall || !ClonedCall)
926 continue;
927 // If the inlined callsite did not have any callsite metadata, then it isn't
928 // involved in any profiled call contexts, and we can remove any memprof
929 // metadata on the cloned call.
930 if (!CallsiteMD) {
931 removeMemProfMetadata(Call: ClonedCall);
932 removeCallsiteMetadata(Call: ClonedCall);
933 continue;
934 }
935 propagateMemProfHelper(OrigCall, ClonedCall, InlinedCallsiteMD: CallsiteMD, ORE);
936 }
937}
938
939/// Collect all calls that produce RetVal, following only pointer-preserving
940/// instructions (cast, phi, select).
941static void collectPointerReturningCalls(Value *RetVal,
942 SmallVectorImpl<CallBase *> &Out) {
943 SmallVector<Value *, 8> Worklist{RetVal};
944 SmallPtrSet<Value *, 8> Visited;
945 while (!Worklist.empty()) {
946 Value *V = Worklist.pop_back_val();
947 if (!V->getType()->isPointerTy() || !Visited.insert(Ptr: V).second)
948 continue;
949 if (auto *CB = dyn_cast<CallBase>(Val: V))
950 Out.push_back(Elt: CB);
951 else if (isa<BitCastInst, AddrSpaceCastInst>(Val: V))
952 Worklist.push_back(Elt: cast<CastInst>(Val: V)->getOperand(i_nocapture: 0));
953 else if (auto *PN = dyn_cast<PHINode>(Val: V))
954 append_range(C&: Worklist, R: PN->incoming_values());
955 else if (auto *SI = dyn_cast<SelectInst>(Val: V)) {
956 Worklist.push_back(Elt: SI->getTrueValue());
957 Worklist.push_back(Elt: SI->getFalseValue());
958 }
959 }
960}
961
962/// When inlining a call that carries !alloc_token metadata, propagate that
963/// metadata onto calls exposed by inlining the wrapper body. Propagation is
964/// restricted to return-value producing calls, which avoids instrumenting
965/// unrelated calls in the wrapper body.
966static void
967propagateAllocTokenMetadata(Function *CalledFunc, CallBase &CB,
968 const ValueMap<const Value *, WeakTrackingVH> &VMap,
969 ClonedCodeInfo &InlinedFunctionInfo) {
970 MDNode *AllocTokenMD = CB.getMetadata(KindID: LLVMContext::MD_alloc_token);
971 if (!AllocTokenMD)
972 return;
973
974 SmallVector<CallBase *, 2> AllocCalls;
975 for (BasicBlock &BB : *CalledFunc)
976 if (auto *RI = dyn_cast<ReturnInst>(Val: BB.getTerminator()))
977 if (Value *RV = RI->getReturnValue())
978 collectPointerReturningCalls(RetVal: RV, Out&: AllocCalls);
979
980 for (CallBase *OrigCall : AllocCalls) {
981 auto *ClonedCall = dyn_cast_or_null<CallBase>(Val: VMap.lookup(Val: OrigCall));
982 if (!ClonedCall)
983 continue;
984 // Skip calls simplified during inlining; propagation may be incorrect.
985 if (InlinedFunctionInfo.isSimplified(From: OrigCall, To: ClonedCall))
986 continue;
987 // Fill missing only: never overwrite a more specific token the wrapper
988 // already set on an internal allocation. An unknown type (empty type name
989 // with function name) is not more specific.
990 if (MDNode *MD = ClonedCall->getMetadata(KindID: LLVMContext::MD_alloc_token)) {
991 if (MD->getNumOperands() != 3 ||
992 !cast<MDString>(Val: MD->getOperand(I: 0))->getString().empty())
993 continue;
994 }
995 ClonedCall->setMetadata(KindID: LLVMContext::MD_alloc_token, Node: AllocTokenMD);
996 }
997}
998
999/// When inlining a call site that has !llvm.mem.parallel_loop_access,
1000/// !llvm.access.group, !alias.scope or !noalias metadata, that metadata should
1001/// be propagated to all memory-accessing cloned instructions.
1002static void PropagateCallSiteMetadata(CallBase &CB, Function::iterator FStart,
1003 Function::iterator FEnd) {
1004 MDNode *MemParallelLoopAccess =
1005 CB.getMetadata(KindID: LLVMContext::MD_mem_parallel_loop_access);
1006 MDNode *AccessGroup = CB.getMetadata(KindID: LLVMContext::MD_access_group);
1007 MDNode *AliasScope = CB.getMetadata(KindID: LLVMContext::MD_alias_scope);
1008 MDNode *NoAlias = CB.getMetadata(KindID: LLVMContext::MD_noalias);
1009 if (!MemParallelLoopAccess && !AccessGroup && !AliasScope && !NoAlias)
1010 return;
1011
1012 for (BasicBlock &BB : make_range(x: FStart, y: FEnd)) {
1013 for (Instruction &I : BB) {
1014 // This metadata is only relevant for instructions that access memory.
1015 if (!I.mayReadOrWriteMemory())
1016 continue;
1017
1018 if (MemParallelLoopAccess) {
1019 // TODO: This probably should not overwrite MemParalleLoopAccess.
1020 MemParallelLoopAccess = MDNode::concatenate(
1021 A: I.getMetadata(KindID: LLVMContext::MD_mem_parallel_loop_access),
1022 B: MemParallelLoopAccess);
1023 I.setMetadata(KindID: LLVMContext::MD_mem_parallel_loop_access,
1024 Node: MemParallelLoopAccess);
1025 }
1026
1027 if (AccessGroup)
1028 I.setMetadata(KindID: LLVMContext::MD_access_group, Node: uniteAccessGroups(
1029 AccGroups1: I.getMetadata(KindID: LLVMContext::MD_access_group), AccGroups2: AccessGroup));
1030
1031 if (AliasScope)
1032 I.setMetadata(KindID: LLVMContext::MD_alias_scope, Node: MDNode::concatenate(
1033 A: I.getMetadata(KindID: LLVMContext::MD_alias_scope), B: AliasScope));
1034
1035 if (NoAlias)
1036 I.setMetadata(KindID: LLVMContext::MD_noalias, Node: MDNode::concatenate(
1037 A: I.getMetadata(KindID: LLVMContext::MD_noalias), B: NoAlias));
1038 }
1039 }
1040}
1041
1042/// Track inlining chain via inlined.from metadata for dontcall diagnostics.
1043static void PropagateInlinedFromMetadata(CallBase &CB, StringRef CalledFuncName,
1044 StringRef CallerFuncName,
1045 Function::iterator FStart,
1046 Function::iterator FEnd) {
1047 LLVMContext &Ctx = CB.getContext();
1048 uint64_t InlineSiteLoc = 0;
1049 if (auto *MD = CB.getMetadata(Kind: "srcloc"))
1050 if (auto *CI = mdconst::dyn_extract<ConstantInt>(MD: MD->getOperand(I: 0)))
1051 InlineSiteLoc = CI->getZExtValue();
1052
1053 auto *I64Ty = Type::getInt64Ty(C&: Ctx);
1054 auto MakeMDInt = [&](uint64_t V) {
1055 return ConstantAsMetadata::get(C: ConstantInt::get(Ty: I64Ty, V));
1056 };
1057
1058 for (BasicBlock &BB : make_range(x: FStart, y: FEnd)) {
1059 for (Instruction &I : BB) {
1060 auto *CI = dyn_cast<CallInst>(Val: &I);
1061 if (!CI || !CI->getMetadata(Kind: "srcloc"))
1062 continue;
1063 auto *Callee = CI->getCalledFunction();
1064 if (!Callee || (!Callee->hasFnAttribute(Kind: "dontcall-error") &&
1065 !Callee->hasFnAttribute(Kind: "dontcall-warn")))
1066 continue;
1067
1068 SmallVector<Metadata *, 8> Ops;
1069 if (MDNode *Existing = CI->getMetadata(Kind: "inlined.from"))
1070 append_range(C&: Ops, R: Existing->operands());
1071 else {
1072 Ops.push_back(Elt: MDString::get(Context&: Ctx, Str: CalledFuncName));
1073 Ops.push_back(Elt: MakeMDInt(0));
1074 }
1075 Ops.push_back(Elt: MDString::get(Context&: Ctx, Str: CallerFuncName));
1076 Ops.push_back(Elt: MakeMDInt(InlineSiteLoc));
1077 CI->setMetadata(Kind: "inlined.from", Node: MDNode::get(Context&: Ctx, MDs: Ops));
1078 }
1079 }
1080}
1081
1082/// Bundle operands of the inlined function must be added to inlined call sites.
1083static void PropagateOperandBundles(Function::iterator InlinedBB,
1084 Instruction *CallSiteEHPad) {
1085 for (Instruction &II : llvm::make_early_inc_range(Range&: *InlinedBB)) {
1086 CallBase *I = dyn_cast<CallBase>(Val: &II);
1087 if (!I)
1088 continue;
1089 // Skip call sites which already have a "funclet" bundle.
1090 if (I->getOperandBundle(ID: LLVMContext::OB_funclet))
1091 continue;
1092 // Skip call sites which are nounwind intrinsics (as long as they don't
1093 // lower into regular function calls in the course of IR transformations).
1094 auto *CalledFn =
1095 dyn_cast<Function>(Val: I->getCalledOperand()->stripPointerCasts());
1096 if (CalledFn && CalledFn->isIntrinsic() && I->doesNotThrow() &&
1097 !IntrinsicInst::mayLowerToFunctionCall(IID: CalledFn->getIntrinsicID()))
1098 continue;
1099
1100 SmallVector<OperandBundleDef, 1> OpBundles;
1101 I->getOperandBundlesAsDefs(Defs&: OpBundles);
1102 OpBundles.emplace_back(Args: "funclet", Args&: CallSiteEHPad);
1103
1104 Instruction *NewInst = CallBase::Create(CB: I, Bundles: OpBundles, InsertPt: I->getIterator());
1105 NewInst->takeName(V: I);
1106 I->replaceAllUsesWith(V: NewInst);
1107 I->eraseFromParent();
1108 }
1109}
1110
1111namespace {
1112/// Utility for cloning !noalias and !alias.scope metadata. When a code region
1113/// using scoped alias metadata is inlined, the aliasing relationships may not
1114/// hold between the two version. It is necessary to create a deep clone of the
1115/// metadata, putting the two versions in separate scope domains.
1116class ScopedAliasMetadataDeepCloner {
1117 using MetadataMap = DenseMap<const MDNode *, TrackingMDNodeRef>;
1118 SetVector<const MDNode *> MD;
1119 MetadataMap MDMap;
1120 void addRecursiveMetadataUses();
1121
1122public:
1123 ScopedAliasMetadataDeepCloner(const Function *F);
1124
1125 /// Create a new clone of the scoped alias metadata, which will be used by
1126 /// subsequent remap() calls.
1127 void clone();
1128
1129 /// Remap instructions in the given range from the original to the cloned
1130 /// metadata.
1131 void remap(Function::iterator FStart, Function::iterator FEnd);
1132};
1133} // namespace
1134
1135ScopedAliasMetadataDeepCloner::ScopedAliasMetadataDeepCloner(
1136 const Function *F) {
1137 for (const BasicBlock &BB : *F) {
1138 for (const Instruction &I : BB) {
1139 if (const MDNode *M = I.getMetadata(KindID: LLVMContext::MD_alias_scope))
1140 MD.insert(X: M);
1141 if (const MDNode *M = I.getMetadata(KindID: LLVMContext::MD_noalias))
1142 MD.insert(X: M);
1143
1144 // We also need to clone the metadata in noalias intrinsics.
1145 if (const auto *Decl = dyn_cast<NoAliasScopeDeclInst>(Val: &I))
1146 MD.insert(X: Decl->getScopeList());
1147 }
1148 }
1149 addRecursiveMetadataUses();
1150}
1151
1152void ScopedAliasMetadataDeepCloner::addRecursiveMetadataUses() {
1153 SmallVector<const Metadata *, 16> Queue(MD.begin(), MD.end());
1154 while (!Queue.empty()) {
1155 const MDNode *M = cast<MDNode>(Val: Queue.pop_back_val());
1156 for (const Metadata *Op : M->operands())
1157 if (const MDNode *OpMD = dyn_cast<MDNode>(Val: Op))
1158 if (MD.insert(X: OpMD))
1159 Queue.push_back(Elt: OpMD);
1160 }
1161}
1162
1163void ScopedAliasMetadataDeepCloner::clone() {
1164 assert(MDMap.empty() && "clone() already called ?");
1165
1166 SmallVector<TempMDTuple, 16> DummyNodes;
1167 for (const MDNode *I : MD) {
1168 DummyNodes.push_back(Elt: MDTuple::getTemporary(Context&: I->getContext(), MDs: {}));
1169 MDMap[I].reset(MD: DummyNodes.back().get());
1170 }
1171
1172 // Create new metadata nodes to replace the dummy nodes, replacing old
1173 // metadata references with either a dummy node or an already-created new
1174 // node.
1175 SmallVector<Metadata *, 4> NewOps;
1176 for (const MDNode *I : MD) {
1177 for (const Metadata *Op : I->operands()) {
1178 if (const MDNode *M = dyn_cast<MDNode>(Val: Op))
1179 NewOps.push_back(Elt: MDMap[M]);
1180 else
1181 NewOps.push_back(Elt: const_cast<Metadata *>(Op));
1182 }
1183
1184 MDNode *NewM = MDNode::get(Context&: I->getContext(), MDs: NewOps);
1185 MDTuple *TempM = cast<MDTuple>(Val&: MDMap[I]);
1186 assert(TempM->isTemporary() && "Expected temporary node");
1187
1188 TempM->replaceAllUsesWith(MD: NewM);
1189 NewOps.clear();
1190 }
1191}
1192
1193void ScopedAliasMetadataDeepCloner::remap(Function::iterator FStart,
1194 Function::iterator FEnd) {
1195 if (MDMap.empty())
1196 return; // Nothing to do.
1197
1198 for (BasicBlock &BB : make_range(x: FStart, y: FEnd)) {
1199 for (Instruction &I : BB) {
1200 // TODO: The null checks for the MDMap.lookup() results should no longer
1201 // be necessary.
1202 if (MDNode *M = I.getMetadata(KindID: LLVMContext::MD_alias_scope))
1203 if (MDNode *MNew = MDMap.lookup(Val: M))
1204 I.setMetadata(KindID: LLVMContext::MD_alias_scope, Node: MNew);
1205
1206 if (MDNode *M = I.getMetadata(KindID: LLVMContext::MD_noalias))
1207 if (MDNode *MNew = MDMap.lookup(Val: M))
1208 I.setMetadata(KindID: LLVMContext::MD_noalias, Node: MNew);
1209
1210 if (auto *Decl = dyn_cast<NoAliasScopeDeclInst>(Val: &I))
1211 if (MDNode *MNew = MDMap.lookup(Val: Decl->getScopeList()))
1212 Decl->setScopeList(MNew);
1213 }
1214 }
1215}
1216
1217/// If the inlined function has noalias arguments,
1218/// then add new alias scopes for each noalias argument, tag the mapped noalias
1219/// parameters with noalias metadata specifying the new scope, and tag all
1220/// non-derived loads, stores and memory intrinsics with the new alias scopes.
1221static void AddAliasScopeMetadata(CallBase &CB, ValueToValueMapTy &VMap,
1222 const DataLayout &DL, AAResults *CalleeAAR,
1223 ClonedCodeInfo &InlinedFunctionInfo) {
1224 if (!EnableNoAliasConversion)
1225 return;
1226
1227 const Function *CalledFunc = CB.getCalledFunction();
1228 SmallVector<const Argument *, 4> NoAliasArgs;
1229
1230 for (const Argument &Arg : CalledFunc->args())
1231 if (CB.paramHasAttr(ArgNo: Arg.getArgNo(), Kind: Attribute::NoAlias) && !Arg.use_empty())
1232 NoAliasArgs.push_back(Elt: &Arg);
1233
1234 if (NoAliasArgs.empty())
1235 return;
1236
1237 // To do a good job, if a noalias variable is captured, we need to know if
1238 // the capture point dominates the particular use we're considering.
1239 DominatorTree DT;
1240 DT.recalculate(Func&: const_cast<Function&>(*CalledFunc));
1241
1242 // noalias indicates that pointer values based on the argument do not alias
1243 // pointer values which are not based on it. So we add a new "scope" for each
1244 // noalias function argument. Accesses using pointers based on that argument
1245 // become part of that alias scope, accesses using pointers not based on that
1246 // argument are tagged as noalias with that scope.
1247
1248 DenseMap<const Argument *, MDNode *> NewScopes;
1249 MDBuilder MDB(CalledFunc->getContext());
1250
1251 // Create a new scope domain for this function.
1252 MDNode *NewDomain =
1253 MDB.createAnonymousAliasScopeDomain(Description: CalledFunc->getName());
1254 for (unsigned i = 0, e = NoAliasArgs.size(); i != e; ++i) {
1255 const Argument *A = NoAliasArgs[i];
1256
1257 std::string Name = std::string(CalledFunc->getName());
1258 if (A->hasName()) {
1259 Name += ": %";
1260 Name += A->getName();
1261 } else {
1262 Name += ": argument ";
1263 Name += utostr(X: i);
1264 }
1265
1266 // Note: We always create a new anonymous root here. This is true regardless
1267 // of the linkage of the callee because the aliasing "scope" is not just a
1268 // property of the callee, but also all control dependencies in the caller.
1269 MDNode *NewScope = MDB.createAnonymousAliasScope(Domain: NewDomain, Name);
1270 NewScopes.insert(KV: std::make_pair(x&: A, y&: NewScope));
1271
1272 if (UseNoAliasIntrinsic) {
1273 // Introduce a llvm.experimental.noalias.scope.decl for the noalias
1274 // argument.
1275 MDNode *AScopeList = MDNode::get(Context&: CalledFunc->getContext(), MDs: NewScope);
1276 auto *NoAliasDecl =
1277 IRBuilder<>(&CB).CreateNoAliasScopeDeclaration(ScopeTag: AScopeList);
1278 // Ignore the result for now. The result will be used when the
1279 // llvm.noalias intrinsic is introduced.
1280 (void)NoAliasDecl;
1281 }
1282 }
1283
1284 // Iterate over all new instructions in the map; for all memory-access
1285 // instructions, add the alias scope metadata.
1286 for (ValueToValueMapTy::iterator VMI = VMap.begin(), VMIE = VMap.end();
1287 VMI != VMIE; ++VMI) {
1288 if (const Instruction *I = dyn_cast<Instruction>(Val: VMI->first)) {
1289 if (!VMI->second)
1290 continue;
1291
1292 Instruction *NI = dyn_cast<Instruction>(Val&: VMI->second);
1293 if (!NI || InlinedFunctionInfo.isSimplified(From: I, To: NI))
1294 continue;
1295
1296 bool IsArgMemOnlyCall = false, IsFuncCall = false;
1297 SmallVector<const Value *, 2> PtrArgs;
1298
1299 if (const LoadInst *LI = dyn_cast<LoadInst>(Val: I))
1300 PtrArgs.push_back(Elt: LI->getPointerOperand());
1301 else if (const StoreInst *SI = dyn_cast<StoreInst>(Val: I))
1302 PtrArgs.push_back(Elt: SI->getPointerOperand());
1303 else if (const VAArgInst *VAAI = dyn_cast<VAArgInst>(Val: I))
1304 PtrArgs.push_back(Elt: VAAI->getPointerOperand());
1305 else if (const AtomicCmpXchgInst *CXI = dyn_cast<AtomicCmpXchgInst>(Val: I))
1306 PtrArgs.push_back(Elt: CXI->getPointerOperand());
1307 else if (const AtomicRMWInst *RMWI = dyn_cast<AtomicRMWInst>(Val: I))
1308 PtrArgs.push_back(Elt: RMWI->getPointerOperand());
1309 else if (const auto *Call = dyn_cast<CallBase>(Val: I)) {
1310 // If we know that the call does not access memory, then we'll still
1311 // know that about the inlined clone of this call site, and we don't
1312 // need to add metadata.
1313 if (Call->doesNotAccessMemory())
1314 continue;
1315
1316 IsFuncCall = true;
1317 if (CalleeAAR) {
1318 MemoryEffects ME = CalleeAAR->getMemoryEffects(Call);
1319
1320 // We'll retain this knowledge without additional metadata.
1321 if (ME.onlyAccessesInaccessibleMem())
1322 continue;
1323
1324 if (ME.onlyAccessesArgPointees())
1325 IsArgMemOnlyCall = true;
1326 }
1327
1328 for (Value *Arg : Call->args()) {
1329 // Only care about pointer arguments. If a noalias argument is
1330 // accessed through a non-pointer argument, it must be captured
1331 // first (e.g. via ptrtoint), and we protect against captures below.
1332 if (!Arg->getType()->isPointerTy())
1333 continue;
1334
1335 PtrArgs.push_back(Elt: Arg);
1336 }
1337 }
1338
1339 // If we found no pointers, then this instruction is not suitable for
1340 // pairing with an instruction to receive aliasing metadata.
1341 // However, if this is a call, this we might just alias with none of the
1342 // noalias arguments.
1343 if (PtrArgs.empty() && !IsFuncCall)
1344 continue;
1345
1346 // It is possible that there is only one underlying object, but you
1347 // need to go through several PHIs to see it, and thus could be
1348 // repeated in the Objects list.
1349 SmallPtrSet<const Value *, 4> ObjSet;
1350 SmallVector<Metadata *, 4> Scopes, NoAliases;
1351
1352 for (const Value *V : PtrArgs) {
1353 SmallVector<const Value *, 4> Objects;
1354 getUnderlyingObjects(V, Objects, /* LI = */ nullptr);
1355
1356 ObjSet.insert_range(R&: Objects);
1357 }
1358
1359 // Figure out if we're derived from anything that is not a noalias
1360 // argument.
1361 bool RequiresNoCaptureBefore = false, UsesAliasingPtr = false,
1362 UsesUnknownObject = false;
1363 for (const Value *V : ObjSet) {
1364 // Is this value a constant that cannot be derived from any pointer
1365 // value (we need to exclude constant expressions, for example, that
1366 // are formed from arithmetic on global symbols).
1367 bool IsNonPtrConst = isa<ConstantInt>(Val: V) || isa<ConstantFP>(Val: V) ||
1368 isa<ConstantPointerNull>(Val: V) ||
1369 isa<ConstantDataVector>(Val: V) || isa<UndefValue>(Val: V);
1370 if (IsNonPtrConst)
1371 continue;
1372
1373 // If this is anything other than a noalias argument, then we cannot
1374 // completely describe the aliasing properties using alias.scope
1375 // metadata (and, thus, won't add any).
1376 if (const Argument *A = dyn_cast<Argument>(Val: V)) {
1377 if (!CB.paramHasAttr(ArgNo: A->getArgNo(), Kind: Attribute::NoAlias))
1378 UsesAliasingPtr = true;
1379 } else {
1380 UsesAliasingPtr = true;
1381 }
1382
1383 if (isEscapeSource(V)) {
1384 // An escape source can only alias with a noalias argument if it has
1385 // been captured beforehand.
1386 RequiresNoCaptureBefore = true;
1387 } else if (!isa<Argument>(Val: V) && !isIdentifiedObject(V)) {
1388 // If this is neither an escape source, nor some identified object
1389 // (which cannot directly alias a noalias argument), nor some other
1390 // argument (which, by definition, also cannot alias a noalias
1391 // argument), conservatively do not make any assumptions.
1392 UsesUnknownObject = true;
1393 }
1394 }
1395
1396 // Nothing we can do if the used underlying object cannot be reliably
1397 // determined.
1398 if (UsesUnknownObject)
1399 continue;
1400
1401 // A function call can always get captured noalias pointers (via other
1402 // parameters, globals, etc.).
1403 if (IsFuncCall && !IsArgMemOnlyCall)
1404 RequiresNoCaptureBefore = true;
1405
1406 // First, we want to figure out all of the sets with which we definitely
1407 // don't alias. Iterate over all noalias set, and add those for which:
1408 // 1. The noalias argument is not in the set of objects from which we
1409 // definitely derive.
1410 // 2. The noalias argument has not yet been captured.
1411 // An arbitrary function that might load pointers could see captured
1412 // noalias arguments via other noalias arguments or globals, and so we
1413 // must always check for prior capture.
1414 for (const Argument *A : NoAliasArgs) {
1415 if (ObjSet.contains(Ptr: A))
1416 continue; // May be based on a noalias argument.
1417
1418 // It might be tempting to skip the PointerMayBeCapturedBefore check if
1419 // A->hasNoCaptureAttr() is true, but this is incorrect because
1420 // nocapture only guarantees that no copies outlive the function, not
1421 // that the value cannot be locally captured.
1422 if (!RequiresNoCaptureBefore ||
1423 !capturesAnything(CC: PointerMayBeCapturedBefore(
1424 V: A, /*ReturnCaptures=*/false, I, DT: &DT, /*IncludeI=*/false,
1425 Mask: CaptureComponents::Provenance)))
1426 NoAliases.push_back(Elt: NewScopes[A]);
1427 }
1428
1429 if (!NoAliases.empty())
1430 NI->setMetadata(KindID: LLVMContext::MD_noalias,
1431 Node: MDNode::concatenate(
1432 A: NI->getMetadata(KindID: LLVMContext::MD_noalias),
1433 B: MDNode::get(Context&: CalledFunc->getContext(), MDs: NoAliases)));
1434
1435 // Next, we want to figure out all of the sets to which we might belong.
1436 // We might belong to a set if the noalias argument is in the set of
1437 // underlying objects. If there is some non-noalias argument in our list
1438 // of underlying objects, then we cannot add a scope because the fact
1439 // that some access does not alias with any set of our noalias arguments
1440 // cannot itself guarantee that it does not alias with this access
1441 // (because there is some pointer of unknown origin involved and the
1442 // other access might also depend on this pointer). We also cannot add
1443 // scopes to arbitrary functions unless we know they don't access any
1444 // non-parameter pointer-values.
1445 bool CanAddScopes = !UsesAliasingPtr;
1446 if (CanAddScopes && IsFuncCall)
1447 CanAddScopes = IsArgMemOnlyCall;
1448
1449 if (CanAddScopes)
1450 for (const Argument *A : NoAliasArgs) {
1451 if (ObjSet.count(Ptr: A))
1452 Scopes.push_back(Elt: NewScopes[A]);
1453 }
1454
1455 if (!Scopes.empty())
1456 NI->setMetadata(
1457 KindID: LLVMContext::MD_alias_scope,
1458 Node: MDNode::concatenate(A: NI->getMetadata(KindID: LLVMContext::MD_alias_scope),
1459 B: MDNode::get(Context&: CalledFunc->getContext(), MDs: Scopes)));
1460 }
1461 }
1462}
1463
1464static bool MayContainThrowingOrExitingCallAfterCB(CallBase *Begin,
1465 ReturnInst *End) {
1466
1467 assert(Begin->getParent() == End->getParent() &&
1468 "Expected to be in same basic block!");
1469 auto BeginIt = Begin->getIterator();
1470 assert(BeginIt != End->getIterator() && "Non-empty BB has empty iterator");
1471 return !llvm::isGuaranteedToTransferExecutionToSuccessor(
1472 Begin: ++BeginIt, End: End->getIterator(), ScanLimit: InlinerAttributeWindow + 1);
1473}
1474
1475// Add attributes from CB params and Fn attributes that can always be propagated
1476// to the corresponding argument / inner callbases.
1477static void AddParamAndFnBasicAttributes(const CallBase &CB,
1478 ValueToValueMapTy &VMap,
1479 ClonedCodeInfo &InlinedFunctionInfo) {
1480 auto *CalledFunction = CB.getCalledFunction();
1481 auto &Context = CalledFunction->getContext();
1482
1483 // Collect valid attributes for all params.
1484 SmallVector<AttrBuilder> ValidObjParamAttrs, ValidExactParamAttrs;
1485 bool HasAttrToPropagate = false;
1486
1487 // Attributes we can only propagate if the exact parameter is forwarded.
1488 // We can propagate both poison generating and UB generating attributes
1489 // without any extra checks. The only attribute that is tricky to propagate
1490 // is `noundef` (skipped for now) as that can create new UB where previous
1491 // behavior was just using a poison value.
1492 static const Attribute::AttrKind ExactAttrsToPropagate[] = {
1493 Attribute::Dereferenceable, Attribute::DereferenceableOrNull,
1494 Attribute::NonNull, Attribute::NoFPClass,
1495 Attribute::Alignment, Attribute::Range};
1496
1497 for (unsigned I = 0, E = CB.arg_size(); I < E; ++I) {
1498 ValidObjParamAttrs.emplace_back(Args: AttrBuilder{CB.getContext()});
1499 ValidExactParamAttrs.emplace_back(Args: AttrBuilder{CB.getContext()});
1500 // Access attributes can be propagated to any param with the same underlying
1501 // object as the argument.
1502 if (CB.paramHasAttr(ArgNo: I, Kind: Attribute::ReadNone))
1503 ValidObjParamAttrs.back().addAttribute(Val: Attribute::ReadNone);
1504 if (CB.paramHasAttr(ArgNo: I, Kind: Attribute::ReadOnly))
1505 ValidObjParamAttrs.back().addAttribute(Val: Attribute::ReadOnly);
1506
1507 for (Attribute::AttrKind AK : ExactAttrsToPropagate) {
1508 Attribute Attr = CB.getParamAttr(ArgNo: I, Kind: AK);
1509 if (Attr.isValid())
1510 ValidExactParamAttrs.back().addAttribute(A: Attr);
1511 }
1512
1513 HasAttrToPropagate |= ValidObjParamAttrs.back().hasAttributes();
1514 HasAttrToPropagate |= ValidExactParamAttrs.back().hasAttributes();
1515 }
1516
1517 // Won't be able to propagate anything.
1518 if (!HasAttrToPropagate)
1519 return;
1520
1521 for (BasicBlock &BB : *CalledFunction) {
1522 for (Instruction &Ins : BB) {
1523 const auto *InnerCB = dyn_cast<CallBase>(Val: &Ins);
1524 if (!InnerCB)
1525 continue;
1526 auto *NewInnerCB = dyn_cast_or_null<CallBase>(Val: VMap.lookup(Val: InnerCB));
1527 if (!NewInnerCB)
1528 continue;
1529 // The InnerCB might have be simplified during the inlining
1530 // process which can make propagation incorrect.
1531 if (InlinedFunctionInfo.isSimplified(From: InnerCB, To: NewInnerCB))
1532 continue;
1533
1534 AttributeList AL = NewInnerCB->getAttributes();
1535 for (unsigned I = 0, E = InnerCB->arg_size(); I < E; ++I) {
1536 // It's unsound or requires special handling to propagate
1537 // attributes to byval arguments. Even if CalledFunction
1538 // doesn't e.g. write to the argument (readonly), the call to
1539 // NewInnerCB may write to its by-value copy.
1540 if (NewInnerCB->isByValArgument(ArgNo: I))
1541 continue;
1542
1543 // Don't bother propagating attrs to constants.
1544 if (match(V: NewInnerCB->getArgOperand(i: I),
1545 P: llvm::PatternMatch::m_ImmConstant()))
1546 continue;
1547
1548 // Check if the underlying value for the parameter is an argument.
1549 const Argument *Arg = dyn_cast<Argument>(Val: InnerCB->getArgOperand(i: I));
1550 unsigned ArgNo;
1551 if (Arg) {
1552 ArgNo = Arg->getArgNo();
1553 // For dereferenceable, dereferenceable_or_null, align, etc...
1554 // we don't want to propagate if the existing param has the same
1555 // attribute with "better" constraints. So remove from the
1556 // new AL if the region of the existing param is larger than
1557 // what we can propagate.
1558 AttrBuilder NewAB{
1559 Context, AttributeSet::get(C&: Context, B: ValidExactParamAttrs[ArgNo])};
1560 if (AL.getParamDereferenceableBytes(Index: I) >
1561 NewAB.getDereferenceableBytes())
1562 NewAB.removeAttribute(Val: Attribute::Dereferenceable);
1563 if (AL.getParamDereferenceableOrNullBytes(ArgNo: I) >
1564 NewAB.getDereferenceableOrNullBytes())
1565 NewAB.removeAttribute(Val: Attribute::DereferenceableOrNull);
1566 if (AL.getParamAlignment(ArgNo: I).valueOrOne() >
1567 NewAB.getAlignment().valueOrOne())
1568 NewAB.removeAttribute(Val: Attribute::Alignment);
1569 if (auto ExistingRange = AL.getParamRange(ArgNo: I)) {
1570 if (auto NewRange = NewAB.getRange()) {
1571 ConstantRange CombinedRange =
1572 ExistingRange->intersectWith(CR: *NewRange);
1573 NewAB.removeAttribute(Val: Attribute::Range);
1574 NewAB.addRangeAttr(CR: CombinedRange);
1575 }
1576 }
1577
1578 if (FPClassTest ExistingNoFP = AL.getParamNoFPClass(ArgNo: I))
1579 NewAB.addNoFPClassAttr(NoFPClassMask: ExistingNoFP | NewAB.getNoFPClass());
1580
1581 AL = AL.addParamAttributes(C&: Context, ArgNo: I, B: NewAB);
1582 } else if (NewInnerCB->getArgOperand(i: I)->getType()->isPointerTy()) {
1583 // Check if the underlying value for the parameter is an argument.
1584 const Value *UnderlyingV =
1585 getUnderlyingObject(V: InnerCB->getArgOperand(i: I));
1586 Arg = dyn_cast<Argument>(Val: UnderlyingV);
1587 if (!Arg)
1588 continue;
1589 ArgNo = Arg->getArgNo();
1590 } else {
1591 continue;
1592 }
1593
1594 // If so, propagate its access attributes.
1595 AL = AL.addParamAttributes(C&: Context, ArgNo: I, B: ValidObjParamAttrs[ArgNo]);
1596
1597 // We can have conflicting attributes from the inner callsite and
1598 // to-be-inlined callsite. In that case, choose the most
1599 // restrictive.
1600
1601 // readonly + writeonly means we can never deref so make readnone.
1602 if (AL.hasParamAttr(ArgNo: I, Kind: Attribute::ReadOnly) &&
1603 AL.hasParamAttr(ArgNo: I, Kind: Attribute::WriteOnly))
1604 AL = AL.addParamAttribute(C&: Context, ArgNo: I, Kind: Attribute::ReadNone);
1605
1606 // If have readnone, need to clear readonly/writeonly
1607 if (AL.hasParamAttr(ArgNo: I, Kind: Attribute::ReadNone)) {
1608 AL = AL.removeParamAttribute(C&: Context, ArgNo: I, Kind: Attribute::ReadOnly);
1609 AL = AL.removeParamAttribute(C&: Context, ArgNo: I, Kind: Attribute::WriteOnly);
1610 }
1611
1612 // Writable cannot exist in conjunction w/ readonly/readnone
1613 if (AL.hasParamAttr(ArgNo: I, Kind: Attribute::ReadOnly) ||
1614 AL.hasParamAttr(ArgNo: I, Kind: Attribute::ReadNone))
1615 AL = AL.removeParamAttribute(C&: Context, ArgNo: I, Kind: Attribute::Writable);
1616 }
1617 NewInnerCB->setAttributes(AL);
1618 }
1619 }
1620}
1621
1622// Only allow these white listed attributes to be propagated back to the
1623// callee. This is because other attributes may only be valid on the call
1624// itself, i.e. attributes such as signext and zeroext.
1625
1626// Attributes that are always okay to propagate as if they are violated its
1627// immediate UB.
1628static AttrBuilder IdentifyValidUBGeneratingAttributes(CallBase &CB) {
1629 AttrBuilder Valid(CB.getContext());
1630 if (auto DerefBytes = CB.getRetDereferenceableBytes())
1631 Valid.addDereferenceableAttr(Bytes: DerefBytes);
1632 if (auto DerefOrNullBytes = CB.getRetDereferenceableOrNullBytes())
1633 Valid.addDereferenceableOrNullAttr(Bytes: DerefOrNullBytes);
1634 if (CB.hasRetAttr(Kind: Attribute::NoAlias))
1635 Valid.addAttribute(Val: Attribute::NoAlias);
1636 if (CB.hasRetAttr(Kind: Attribute::NoUndef))
1637 Valid.addAttribute(Val: Attribute::NoUndef);
1638 return Valid;
1639}
1640
1641// Attributes that need additional checks as propagating them may change
1642// behavior or cause new UB.
1643static AttrBuilder IdentifyValidPoisonGeneratingAttributes(CallBase &CB) {
1644 AttrBuilder Valid(CB.getContext());
1645 if (CB.hasRetAttr(Kind: Attribute::NonNull))
1646 Valid.addAttribute(Val: Attribute::NonNull);
1647 if (CB.hasRetAttr(Kind: Attribute::Alignment))
1648 Valid.addAlignmentAttr(Align: CB.getRetAlign());
1649 if (std::optional<ConstantRange> Range = CB.getRange())
1650 Valid.addRangeAttr(CR: *Range);
1651 if (CB.hasRetAttr(Kind: Attribute::NoFPClass))
1652 Valid.addNoFPClassAttr(NoFPClassMask: CB.getRetNoFPClass());
1653 return Valid;
1654}
1655
1656static void AddReturnAttributes(CallBase &CB, ValueToValueMapTy &VMap,
1657 ClonedCodeInfo &InlinedFunctionInfo) {
1658 AttrBuilder CallSiteValidUB = IdentifyValidUBGeneratingAttributes(CB);
1659 AttrBuilder CallSiteValidPG = IdentifyValidPoisonGeneratingAttributes(CB);
1660 if (!CallSiteValidUB.hasAttributes() && !CallSiteValidPG.hasAttributes())
1661 return;
1662 auto *CalledFunction = CB.getCalledFunction();
1663 auto &Context = CalledFunction->getContext();
1664
1665 for (auto &BB : *CalledFunction) {
1666 auto *RI = dyn_cast<ReturnInst>(Val: BB.getTerminator());
1667 if (!RI || !isa<CallBase>(Val: RI->getOperand(i_nocapture: 0)))
1668 continue;
1669 auto *RetVal = cast<CallBase>(Val: RI->getOperand(i_nocapture: 0));
1670 // Check that the cloned RetVal exists and is a call, otherwise we cannot
1671 // add the attributes on the cloned RetVal. Simplification during inlining
1672 // could have transformed the cloned instruction.
1673 auto *NewRetVal = dyn_cast_or_null<CallBase>(Val: VMap.lookup(Val: RetVal));
1674 if (!NewRetVal)
1675 continue;
1676
1677 // The RetVal might have be simplified during the inlining
1678 // process which can make propagation incorrect.
1679 if (InlinedFunctionInfo.isSimplified(From: RetVal, To: NewRetVal))
1680 continue;
1681 // Backward propagation of attributes to the returned value may be incorrect
1682 // if it is control flow dependent.
1683 // Consider:
1684 // @callee {
1685 // %rv = call @foo()
1686 // %rv2 = call @bar()
1687 // if (%rv2 != null)
1688 // return %rv2
1689 // if (%rv == null)
1690 // exit()
1691 // return %rv
1692 // }
1693 // caller() {
1694 // %val = call nonnull @callee()
1695 // }
1696 // Here we cannot add the nonnull attribute on either foo or bar. So, we
1697 // limit the check to both RetVal and RI are in the same basic block and
1698 // there are no throwing/exiting instructions between these instructions.
1699 if (RI->getParent() != RetVal->getParent() ||
1700 MayContainThrowingOrExitingCallAfterCB(Begin: RetVal, End: RI))
1701 continue;
1702 // Add to the existing attributes of NewRetVal, i.e. the cloned call
1703 // instruction.
1704 // NB! When we have the same attribute already existing on NewRetVal, but
1705 // with a differing value, the AttributeList's merge API honours the already
1706 // existing attribute value (i.e. attributes such as dereferenceable,
1707 // dereferenceable_or_null etc). See AttrBuilder::merge for more details.
1708 AttrBuilder ValidUB = IdentifyValidUBGeneratingAttributes(CB);
1709 AttrBuilder ValidPG = IdentifyValidPoisonGeneratingAttributes(CB);
1710 AttributeList AL = NewRetVal->getAttributes();
1711 if (ValidUB.getDereferenceableBytes() < AL.getRetDereferenceableBytes())
1712 ValidUB.removeAttribute(Val: Attribute::Dereferenceable);
1713 if (ValidUB.getDereferenceableOrNullBytes() <
1714 AL.getRetDereferenceableOrNullBytes())
1715 ValidUB.removeAttribute(Val: Attribute::DereferenceableOrNull);
1716 AttributeList NewAL = AL.addRetAttributes(C&: Context, B: ValidUB);
1717 // Attributes that may generate poison returns are a bit tricky. If we
1718 // propagate them, other uses of the callsite might have their behavior
1719 // change or cause UB (if they have noundef) b.c of the new potential
1720 // poison.
1721 // Take the following three cases:
1722 //
1723 // 1)
1724 // define nonnull ptr @foo() {
1725 // %p = call ptr @bar()
1726 // call void @use(ptr %p) willreturn nounwind
1727 // ret ptr %p
1728 // }
1729 //
1730 // 2)
1731 // define noundef nonnull ptr @foo() {
1732 // %p = call ptr @bar()
1733 // call void @use(ptr %p) willreturn nounwind
1734 // ret ptr %p
1735 // }
1736 //
1737 // 3)
1738 // define nonnull ptr @foo() {
1739 // %p = call noundef ptr @bar()
1740 // ret ptr %p
1741 // }
1742 //
1743 // In case 1, we can't propagate nonnull because poison value in @use may
1744 // change behavior or trigger UB.
1745 // In case 2, we don't need to be concerned about propagating nonnull, as
1746 // any new poison at @use will trigger UB anyways.
1747 // In case 3, we can never propagate nonnull because it may create UB due to
1748 // the noundef on @bar.
1749 if (ValidPG.getAlignment().valueOrOne() < AL.getRetAlignment().valueOrOne())
1750 ValidPG.removeAttribute(Val: Attribute::Alignment);
1751 if (ValidPG.hasAttributes()) {
1752 Attribute CBRange = ValidPG.getAttribute(Kind: Attribute::Range);
1753 if (CBRange.isValid()) {
1754 Attribute NewRange = AL.getRetAttr(Kind: Attribute::Range);
1755 if (NewRange.isValid()) {
1756 ValidPG.addRangeAttr(
1757 CR: CBRange.getRange().intersectWith(CR: NewRange.getRange()));
1758 }
1759 }
1760
1761 Attribute CBNoFPClass = ValidPG.getAttribute(Kind: Attribute::NoFPClass);
1762 if (CBNoFPClass.isValid() && AL.hasRetAttr(Kind: Attribute::NoFPClass)) {
1763 ValidPG.addNoFPClassAttr(
1764 NoFPClassMask: CBNoFPClass.getNoFPClass() |
1765 AL.getRetAttr(Kind: Attribute::NoFPClass).getNoFPClass());
1766 }
1767
1768 // Three checks.
1769 // If the callsite has `noundef`, then a poison due to violating the
1770 // return attribute will create UB anyways so we can always propagate.
1771 // Otherwise, if the return value (callee to be inlined) has `noundef`, we
1772 // can't propagate as a new poison return will cause UB.
1773 // Finally, check if the return value has no uses whose behavior may
1774 // change/may cause UB if we potentially return poison. At the moment this
1775 // is implemented overly conservatively with a single-use check.
1776 // TODO: Update the single-use check to iterate through uses and only bail
1777 // if we have a potentially dangerous use.
1778
1779 if (CB.hasRetAttr(Kind: Attribute::NoUndef) ||
1780 (RetVal->hasOneUse() && !RetVal->hasRetAttr(Kind: Attribute::NoUndef)))
1781 NewAL = NewAL.addRetAttributes(C&: Context, B: ValidPG);
1782 }
1783 NewRetVal->setAttributes(NewAL);
1784 }
1785}
1786
1787/// If the inlined function has non-byval align arguments, then
1788/// add @llvm.assume-based alignment assumptions to preserve this information.
1789static void AddAlignmentAssumptions(CallBase &CB, InlineFunctionInfo &IFI) {
1790 if (!PreserveAlignmentAssumptions || !IFI.GetAssumptionCache)
1791 return;
1792
1793 AssumptionCache *AC = &IFI.GetAssumptionCache(*CB.getCaller());
1794 auto &DL = CB.getDataLayout();
1795
1796 // To avoid inserting redundant assumptions, we should check for assumptions
1797 // already in the caller. To do this, we might need a DT of the caller.
1798 DominatorTree DT;
1799 bool DTCalculated = false;
1800
1801 Function *CalledFunc = CB.getCalledFunction();
1802 for (Argument &Arg : CalledFunc->args()) {
1803 if (!Arg.getType()->isPointerTy() || Arg.hasPassPointeeByValueCopyAttr() ||
1804 Arg.use_empty())
1805 continue;
1806 MaybeAlign Alignment = Arg.getParamAlign();
1807 if (!Alignment)
1808 continue;
1809
1810 if (!DTCalculated) {
1811 DT.recalculate(Func&: *CB.getCaller());
1812 DTCalculated = true;
1813 }
1814 // If we can already prove the asserted alignment in the context of the
1815 // caller, then don't bother inserting the assumption.
1816 Value *ArgVal = CB.getArgOperand(i: Arg.getArgNo());
1817 if (getKnownAlignment(V: ArgVal, DL, CtxI: &CB, AC, DT: &DT) >= *Alignment)
1818 continue;
1819
1820 CallInst *NewAsmp = IRBuilder<>(&CB).CreateAlignmentAssumption(
1821 DL, PtrValue: ArgVal, Alignment: Alignment->value());
1822 AC->registerAssumption(CI: cast<AssumeInst>(Val: NewAsmp));
1823 }
1824}
1825
1826static void HandleByValArgumentInit(Type *ByValType, Value *Dst, Value *Src,
1827 MaybeAlign SrcAlign, Module *M,
1828 BasicBlock *InsertBlock,
1829 InlineFunctionInfo &IFI,
1830 Function *CalledFunc) {
1831 IRBuilder<> Builder(InsertBlock->begin());
1832
1833 Value *Size =
1834 Builder.getInt64(C: M->getDataLayout().getTypeStoreSize(Ty: ByValType));
1835
1836 Align DstAlign = Dst->getPointerAlignment(DL: M->getDataLayout());
1837
1838 // Generate a memcpy with the correct alignments.
1839 CallInst *CI = Builder.CreateMemCpy(Dst, DstAlign, Src, SrcAlign, Size);
1840
1841 // The verifier requires that all calls of debug-info-bearing functions
1842 // from debug-info-bearing functions have a debug location (for inlining
1843 // purposes). Assign a dummy location to satisfy the constraint.
1844 if (!CI->getDebugLoc() && InsertBlock->getParent()->getSubprogram())
1845 if (DISubprogram *SP = CalledFunc->getSubprogram())
1846 CI->setDebugLoc(DILocation::get(Context&: SP->getContext(), Line: 0, Column: 0, Scope: SP));
1847}
1848
1849/// When inlining a call site that has a byval argument,
1850/// we have to make the implicit memcpy explicit by adding it.
1851static Value *HandleByValArgument(Type *ByValType, Value *Arg,
1852 Instruction *TheCall,
1853 const Function *CalledFunc,
1854 InlineFunctionInfo &IFI,
1855 MaybeAlign ByValAlignment) {
1856 Function *Caller = TheCall->getFunction();
1857 const DataLayout &DL = Caller->getDataLayout();
1858
1859 // If the called function is readonly, then it could not mutate the caller's
1860 // copy of the byval'd memory. In this case, it is safe to elide the copy and
1861 // temporary.
1862 if (CalledFunc->onlyReadsMemory()) {
1863 // If the byval argument has a specified alignment that is greater than the
1864 // passed in pointer, then we either have to round up the input pointer or
1865 // give up on this transformation.
1866 if (ByValAlignment.valueOrOne() == 1)
1867 return Arg;
1868
1869 AssumptionCache *AC =
1870 IFI.GetAssumptionCache ? &IFI.GetAssumptionCache(*Caller) : nullptr;
1871
1872 // If the pointer is already known to be sufficiently aligned, or if we can
1873 // round it up to a larger alignment, then we don't need a temporary.
1874 if (getOrEnforceKnownAlignment(V: Arg, PrefAlign: *ByValAlignment, DL, CtxI: TheCall, AC) >=
1875 *ByValAlignment)
1876 return Arg;
1877
1878 // Otherwise, we have to make a memcpy to get a safe alignment. This is bad
1879 // for code quality, but rarely happens and is required for correctness.
1880 }
1881
1882 // Create the alloca. If we have DataLayout, use nice alignment.
1883 Align Alignment = DL.getPrefTypeAlign(Ty: ByValType);
1884
1885 // If the byval had an alignment specified, we *must* use at least that
1886 // alignment, as it is required by the byval argument (and uses of the
1887 // pointer inside the callee).
1888 if (ByValAlignment)
1889 Alignment = std::max(a: Alignment, b: *ByValAlignment);
1890
1891 AllocaInst *NewAlloca =
1892 new AllocaInst(ByValType, Arg->getType()->getPointerAddressSpace(),
1893 nullptr, Alignment, Arg->getName());
1894 NewAlloca->setDebugLoc(DebugLoc::getCompilerGenerated());
1895 NewAlloca->insertBefore(InsertPos: Caller->begin()->begin());
1896 IFI.StaticAllocas.push_back(Elt: NewAlloca);
1897
1898 // Uses of the argument in the function should use our new alloca
1899 // instead.
1900 return NewAlloca;
1901}
1902
1903// Check whether this Value is used by a lifetime intrinsic.
1904static bool isUsedByLifetimeMarker(Value *V) {
1905 for (User *U : V->users())
1906 if (isa<LifetimeIntrinsic>(Val: U))
1907 return true;
1908 return false;
1909}
1910
1911// Check whether the given alloca already has
1912// lifetime.start or lifetime.end intrinsics.
1913static bool hasLifetimeMarkers(AllocaInst *AI) {
1914 Type *Ty = AI->getType();
1915 Type *Int8PtrTy =
1916 PointerType::get(C&: Ty->getContext(), AddressSpace: Ty->getPointerAddressSpace());
1917 if (Ty == Int8PtrTy)
1918 return isUsedByLifetimeMarker(V: AI);
1919
1920 // Do a scan to find all the casts to i8*.
1921 for (User *U : AI->users()) {
1922 if (U->getType() != Int8PtrTy) continue;
1923 if (U->stripPointerCasts() != AI) continue;
1924 if (isUsedByLifetimeMarker(V: U))
1925 return true;
1926 }
1927 return false;
1928}
1929
1930/// Return the result of AI->isStaticAlloca() if AI were moved to the entry
1931/// block. Allocas used in inalloca calls and allocas of dynamic array size
1932/// cannot be static.
1933static bool allocaWouldBeStaticInEntry(const AllocaInst *AI ) {
1934 return isa<Constant>(Val: AI->getArraySize()) && !AI->isUsedWithInAlloca();
1935}
1936
1937/// Returns a DebugLoc for a new DILocation which is a clone of \p OrigDL
1938/// inlined at \p InlinedAt. \p IANodes is an inlined-at cache.
1939static DebugLoc inlineDebugLoc(
1940 DebugLoc OrigDL, DILocation *InlinedAt, LLVMContext &Ctx,
1941 DenseMap<const MDNode *, MDNode *> &IANodes,
1942 SmallDenseMap<const DILocation *, DILocation *, 16> &InlineLocs) {
1943 if (DILocation *Cached = InlineLocs.lookup(Val: OrigDL.get()))
1944 return DebugLoc(Cached);
1945 DILocation *IA =
1946 OrigDL->getInlinedAt()
1947 ? DebugLoc::appendInlinedAt(DL: OrigDL, InlinedAt, Ctx, Cache&: IANodes).get()
1948 : InlinedAt;
1949 DILocation *Result = DILocation::getDistinct(
1950 Context&: Ctx, Line: OrigDL.getLine(), Column: OrigDL.getCol(), Scope: OrigDL.getScope(), InlinedAt: IA,
1951 ImplicitCode: OrigDL.isImplicitCode(), AtomGroup: OrigDL->getAtomGroup(), AtomRank: OrigDL->getAtomRank());
1952 InlineLocs[OrigDL.get()] = Result;
1953 return DebugLoc(Result);
1954}
1955
1956/// Update inlined instructions' line numbers to
1957/// to encode location where these instructions are inlined.
1958static void fixupLineNumbers(Function *Fn, Function::iterator FI,
1959 Instruction *TheCall, bool CalleeHasDebugInfo) {
1960 if (!TheCall->getDebugLoc())
1961 return;
1962
1963 // Don't propagate the source location atom from the call to inlined nodebug
1964 // instructions, and avoid putting it in the InlinedAt field of inlined
1965 // not-nodebug instructions. FIXME: Possibly worth transferring/generating
1966 // an atom for the returned value, otherwise we miss stepping on inlined
1967 // nodebug functions (which is different to existing behaviour).
1968 DebugLoc TheCallDL = TheCall->getDebugLoc()->getWithoutAtom();
1969
1970 // A call receiving the outer callsite's fallback location has no usable
1971 // callsite probe of its own. Do not let it inherit the outer call's probe
1972 // discriminator, but preserve any packed DWARF base discriminator.
1973 DebugLoc TheCallDLForInlinedCall = TheCallDL;
1974 uint32_t Discriminator = TheCallDLForInlinedCall->getDiscriminator();
1975 if (DILocation::isPseudoProbeDiscriminator(Discriminator)) {
1976 std::optional<uint32_t> DwarfDiscriminator =
1977 PseudoProbeDwarfDiscriminator::extractDwarfBaseDiscriminator(
1978 Value: Discriminator);
1979 TheCallDLForInlinedCall = TheCallDLForInlinedCall->cloneWithDiscriminator(
1980 Discriminator: DwarfDiscriminator.value_or(u: 0));
1981 }
1982
1983 auto &Ctx = Fn->getContext();
1984 DILocation *InlinedAtNode = TheCallDL;
1985
1986 // Create a unique call site, not to be confused with any other call from the
1987 // same location.
1988 InlinedAtNode = DILocation::getDistinct(
1989 Context&: Ctx, Line: InlinedAtNode->getLine(), Column: InlinedAtNode->getColumn(),
1990 Scope: InlinedAtNode->getScope(), InlinedAt: InlinedAtNode->getInlinedAt());
1991
1992 // Cache the inlined-at nodes as they're built so they are reused, without
1993 // this every instruction's inlined-at chain would become distinct from each
1994 // other.
1995 DenseMap<const MDNode *, MDNode *> IANodes;
1996 SmallDenseMap<const DILocation *, DILocation *, 16> InlineLocs;
1997
1998 // Check if we are not generating inline line tables and want to use
1999 // the call site location instead.
2000 bool NoInlineLineTables = Fn->hasFnAttribute(Kind: "no-inline-line-tables");
2001
2002 // Helper-util for updating the metadata attached to an instruction.
2003 auto UpdateInst = [&](Instruction &I) {
2004 // Loop metadata needs to be updated so that the start and end locs
2005 // reference inlined-at locations.
2006 auto updateLoopInfoLoc = [&Ctx, &InlinedAtNode, &IANodes,
2007 &InlineLocs](Metadata *MD) -> Metadata * {
2008 if (auto *Loc = dyn_cast_or_null<DILocation>(Val: MD))
2009 return inlineDebugLoc(OrigDL: Loc, InlinedAt: InlinedAtNode, Ctx, IANodes, InlineLocs)
2010 .get();
2011 return MD;
2012 };
2013 updateLoopMetadataDebugLocations(I, Updater: updateLoopInfoLoc);
2014
2015 if (!NoInlineLineTables)
2016 if (DebugLoc DL = I.getDebugLoc()) {
2017 DebugLoc IDL = inlineDebugLoc(OrigDL: DL, InlinedAt: InlinedAtNode, Ctx&: I.getContext(),
2018 IANodes, InlineLocs);
2019 I.setDebugLoc(IDL);
2020 return;
2021 }
2022
2023 if (CalleeHasDebugInfo && !NoInlineLineTables)
2024 return;
2025
2026 // If the inlined instruction has no line number, or if inline info
2027 // is not being generated, make it look as if it originates from the call
2028 // location. This is important for ((__always_inline, __nodebug__))
2029 // functions which must use caller location for all instructions in their
2030 // function body.
2031
2032 // Don't update static allocas, as they may get moved later.
2033 if (auto *AI = dyn_cast<AllocaInst>(Val: &I))
2034 if (allocaWouldBeStaticInEntry(AI))
2035 return;
2036
2037 // Do not force a debug loc for pseudo probes, since they do not need to
2038 // be debuggable, and also they are expected to have a zero/null dwarf
2039 // discriminator at this point which could be violated otherwise.
2040 if (isa<PseudoProbeInst>(Val: I))
2041 return;
2042
2043 I.setDebugLoc(isa<CallBase>(Val: I) ? TheCallDLForInlinedCall : TheCallDL);
2044 };
2045
2046 // Helper-util for updating debug-info records attached to instructions.
2047 auto UpdateDVR = [&](DbgRecord *DVR) {
2048 assert(DVR->getDebugLoc() && "Debug Value must have debug loc");
2049 if (NoInlineLineTables) {
2050 DVR->setDebugLoc(TheCallDL);
2051 return;
2052 }
2053 DebugLoc DL = DVR->getDebugLoc();
2054 DebugLoc IDL = inlineDebugLoc(OrigDL: DL, InlinedAt: InlinedAtNode,
2055 Ctx&: DVR->getMarker()->getParent()->getContext(),
2056 IANodes, InlineLocs);
2057 DVR->setDebugLoc(IDL);
2058 };
2059
2060 // Iterate over all instructions, updating metadata and debug-info records.
2061 for (; FI != Fn->end(); ++FI) {
2062 for (Instruction &I : *FI) {
2063 UpdateInst(I);
2064 for (DbgRecord &DVR : I.getDbgRecordRange()) {
2065 UpdateDVR(&DVR);
2066 }
2067 }
2068
2069 // Remove debug info records if we're not keeping inline info.
2070 if (NoInlineLineTables) {
2071 BasicBlock::iterator BI = FI->begin();
2072 while (BI != FI->end()) {
2073 BI->dropDbgRecords();
2074 ++BI;
2075 }
2076 }
2077 }
2078}
2079
2080#undef DEBUG_TYPE
2081#define DEBUG_TYPE "assignment-tracking"
2082/// Find Alloca and linked DbgAssignIntrinsic for locals escaped by \p CB.
2083static at::StorageToVarsMap collectEscapedLocals(const DataLayout &DL,
2084 const CallBase &CB) {
2085 at::StorageToVarsMap EscapedLocals;
2086 SmallPtrSet<const Value *, 4> SeenBases;
2087
2088 LLVM_DEBUG(
2089 errs() << "# Finding caller local variables escaped by callee\n");
2090 for (const Value *Arg : CB.args()) {
2091 LLVM_DEBUG(errs() << "INSPECT: " << *Arg << "\n");
2092 if (!Arg->getType()->isPointerTy()) {
2093 LLVM_DEBUG(errs() << " | SKIP: Not a pointer\n");
2094 continue;
2095 }
2096
2097 const Instruction *I = dyn_cast<Instruction>(Val: Arg);
2098 if (!I) {
2099 LLVM_DEBUG(errs() << " | SKIP: Not result of instruction\n");
2100 continue;
2101 }
2102
2103 // Walk back to the base storage.
2104 assert(Arg->getType()->isPtrOrPtrVectorTy());
2105 APInt TmpOffset(DL.getIndexTypeSizeInBits(Ty: Arg->getType()), 0, false);
2106 const AllocaInst *Base = dyn_cast<AllocaInst>(
2107 Val: Arg->stripAndAccumulateConstantOffsets(DL, Offset&: TmpOffset, AllowNonInbounds: true));
2108 if (!Base) {
2109 LLVM_DEBUG(errs() << " | SKIP: Couldn't walk back to base storage\n");
2110 continue;
2111 }
2112
2113 assert(Base);
2114 LLVM_DEBUG(errs() << " | BASE: " << *Base << "\n");
2115 // We only need to process each base address once - skip any duplicates.
2116 if (!SeenBases.insert(Ptr: Base).second)
2117 continue;
2118
2119 // Find all local variables associated with the backing storage.
2120 auto CollectAssignsForStorage = [&](DbgVariableRecord *DbgAssign) {
2121 // Skip variables from inlined functions - they are not local variables.
2122 if (DbgAssign->getDebugLoc().getInlinedAt())
2123 return;
2124 LLVM_DEBUG(errs() << " > DEF : " << *DbgAssign << "\n");
2125 EscapedLocals[Base].insert(X: at::VarRecord(DbgAssign));
2126 };
2127 for_each(Range: at::getDVRAssignmentMarkers(Inst: Base), F: CollectAssignsForStorage);
2128 }
2129 return EscapedLocals;
2130}
2131
2132static void trackInlinedStores(Function::iterator Start, Function::iterator End,
2133 const CallBase &CB) {
2134 LLVM_DEBUG(errs() << "trackInlinedStores into "
2135 << Start->getParent()->getName() << " from "
2136 << CB.getCalledFunction()->getName() << "\n");
2137 const DataLayout &DL = CB.getDataLayout();
2138 at::trackAssignments(Start, End, Vars: collectEscapedLocals(DL, CB), DL);
2139}
2140
2141/// Update inlined instructions' DIAssignID metadata. We need to do this
2142/// otherwise a function inlined more than once into the same function
2143/// will cause DIAssignID to be shared by many instructions.
2144static void fixupAssignments(Function::iterator Start, Function::iterator End) {
2145 DenseMap<DIAssignID *, DIAssignID *> Map;
2146 // Loop over all the inlined instructions. If we find a DIAssignID
2147 // attachment or use, replace it with a new version.
2148 for (auto BBI = Start; BBI != End; ++BBI) {
2149 for (Instruction &I : *BBI)
2150 at::remapAssignID(Map, I);
2151 }
2152}
2153#undef DEBUG_TYPE
2154#define DEBUG_TYPE "inline-function"
2155
2156/// Update the block frequencies of the caller after a callee has been inlined.
2157///
2158/// Each block cloned into the caller has its block frequency scaled by the
2159/// ratio of CallSiteFreq/CalleeEntryFreq. This ensures that the cloned copy of
2160/// callee's entry block gets the same frequency as the callsite block and the
2161/// relative frequencies of all cloned blocks remain the same after cloning.
2162static void updateCallerBFI(BasicBlock *CallSiteBlock,
2163 const ValueToValueMapTy &VMap,
2164 BlockFrequencyInfo *CallerBFI,
2165 BlockFrequencyInfo *CalleeBFI,
2166 const BasicBlock &CalleeEntryBlock) {
2167 SmallPtrSet<BasicBlock *, 16> ClonedBBs;
2168 for (auto Entry : VMap) {
2169 if (!isa<BasicBlock>(Val: Entry.first) || !Entry.second)
2170 continue;
2171 auto *OrigBB = cast<BasicBlock>(Val: Entry.first);
2172 auto *ClonedBB = cast<BasicBlock>(Val: Entry.second);
2173 BlockFrequency Freq = CalleeBFI->getBlockFreq(BB: OrigBB);
2174 if (!ClonedBBs.insert(Ptr: ClonedBB).second) {
2175 // Multiple blocks in the callee might get mapped to one cloned block in
2176 // the caller since we prune the callee as we clone it. When that happens,
2177 // we want to use the maximum among the original blocks' frequencies.
2178 BlockFrequency NewFreq = CallerBFI->getBlockFreq(BB: ClonedBB);
2179 if (NewFreq > Freq)
2180 Freq = NewFreq;
2181 }
2182 CallerBFI->setBlockFreq(BB: ClonedBB, Freq);
2183 }
2184 BasicBlock *EntryClone = cast<BasicBlock>(Val: VMap.lookup(Val: &CalleeEntryBlock));
2185 CallerBFI->setBlockFreqAndScale(
2186 ReferenceBB: EntryClone, Freq: CallerBFI->getBlockFreq(BB: CallSiteBlock), BlocksToScale&: ClonedBBs);
2187}
2188
2189/// Update the branch metadata for cloned call instructions.
2190static void updateCallProfile(Function *Callee, const ValueToValueMapTy &VMap,
2191 const uint64_t &CalleeEntryCount,
2192 const CallBase &TheCall, ProfileSummaryInfo *PSI,
2193 BlockFrequencyInfo *CallerBFI) {
2194 if (CalleeEntryCount < 1)
2195 return;
2196 auto CallSiteCount =
2197 PSI ? PSI->getProfileCount(CallInst: TheCall, BFI: CallerBFI) : std::nullopt;
2198 int64_t CallCount = std::min(a: CallSiteCount.value_or(u: 0), b: CalleeEntryCount);
2199 updateProfileCallee(Callee, EntryDelta: -CallCount, VMap: &VMap);
2200}
2201
2202void llvm::updateProfileCallee(
2203 Function *Callee, int64_t EntryDelta,
2204 const ValueMap<const Value *, WeakTrackingVH> *VMap) {
2205 auto CalleeCount = Callee->getEntryCount();
2206 if (!CalleeCount)
2207 return;
2208
2209 // Since CallSiteCount is an estimate, it could exceed the original callee
2210 // count and has to be set to 0 so guard against underflow.
2211 const uint64_t NewEntryCount =
2212 (EntryDelta < 0 && static_cast<uint64_t>(-EntryDelta) > *CalleeCount)
2213 ? 0
2214 : *CalleeCount + EntryDelta;
2215
2216 auto updateVTableProfWeight = [](CallBase *CB, const uint64_t NewEntryCount,
2217 const uint64_t PriorEntryCount) {
2218 Instruction *VPtr = PGOIndirectCallVisitor::tryGetVTableInstruction(CB);
2219 if (VPtr)
2220 scaleProfData(I&: *VPtr, S: NewEntryCount, T: PriorEntryCount);
2221 };
2222
2223 // During inlining ?
2224 if (VMap) {
2225 uint64_t CloneEntryCount = *CalleeCount - NewEntryCount;
2226 for (auto Entry : *VMap) {
2227 if (isa<CallInst>(Val: Entry.first))
2228 if (auto *CI = dyn_cast_or_null<CallInst>(Val: Entry.second)) {
2229 CI->updateProfWeight(S: CloneEntryCount, T: *CalleeCount);
2230 updateVTableProfWeight(CI, CloneEntryCount, *CalleeCount);
2231 }
2232
2233 if (isa<InvokeInst>(Val: Entry.first))
2234 if (auto *II = dyn_cast_or_null<InvokeInst>(Val: Entry.second)) {
2235 II->updateProfWeight(S: CloneEntryCount, T: *CalleeCount);
2236 updateVTableProfWeight(II, CloneEntryCount, *CalleeCount);
2237 }
2238 }
2239 }
2240
2241 if (EntryDelta) {
2242 Callee->setEntryCount(Count: NewEntryCount);
2243
2244 for (BasicBlock &BB : *Callee)
2245 // No need to update the callsite if it is pruned during inlining.
2246 if (!VMap || VMap->count(Val: &BB))
2247 for (Instruction &I : BB) {
2248 if (CallInst *CI = dyn_cast<CallInst>(Val: &I)) {
2249 CI->updateProfWeight(S: NewEntryCount, T: *CalleeCount);
2250 updateVTableProfWeight(CI, NewEntryCount, *CalleeCount);
2251 }
2252 if (InvokeInst *II = dyn_cast<InvokeInst>(Val: &I)) {
2253 II->updateProfWeight(S: NewEntryCount, T: *CalleeCount);
2254 updateVTableProfWeight(II, NewEntryCount, *CalleeCount);
2255 }
2256 }
2257 }
2258}
2259
2260/// An operand bundle "clang.arc.attachedcall" on a call indicates the call
2261/// result is implicitly consumed by a call to retainRV or claimRV immediately
2262/// after the call. This function inlines the retainRV/claimRV calls.
2263///
2264/// There are three cases to consider:
2265///
2266/// 1. If there is a call to autoreleaseRV that takes a pointer to the returned
2267/// object in the callee return block, the autoreleaseRV call and the
2268/// retainRV/claimRV call in the caller cancel out. If the call in the caller
2269/// is a claimRV call, a call to objc_release is emitted.
2270///
2271/// 2. If there is a call in the callee return block that doesn't have operand
2272/// bundle "clang.arc.attachedcall", the operand bundle on the original call
2273/// is transferred to the call in the callee.
2274///
2275/// 3. Otherwise, a call to objc_retain is inserted if the call in the caller is
2276/// a retainRV call.
2277static void
2278inlineRetainOrClaimRVCalls(CallBase &CB, objcarc::ARCInstKind RVCallKind,
2279 const SmallVectorImpl<ReturnInst *> &Returns) {
2280 assert(objcarc::isRetainOrClaimRV(RVCallKind) && "unexpected ARC function");
2281 bool IsRetainRV = RVCallKind == objcarc::ARCInstKind::RetainRV,
2282 IsUnsafeClaimRV = !IsRetainRV;
2283
2284 for (auto *RI : Returns) {
2285 Value *RetOpnd = objcarc::GetRCIdentityRoot(V: RI->getOperand(i_nocapture: 0));
2286 bool InsertRetainCall = IsRetainRV;
2287 IRBuilder<> Builder(*RI->getModule());
2288
2289 // Walk backwards through the basic block looking for either a matching
2290 // autoreleaseRV call or an unannotated call.
2291 auto InstRange = llvm::make_range(x: ++(RI->getIterator().getReverse()),
2292 y: RI->getParent()->rend());
2293 for (Instruction &I : llvm::make_early_inc_range(Range&: InstRange)) {
2294 // Ignore casts.
2295 if (isa<CastInst>(Val: I))
2296 continue;
2297
2298 if (auto *II = dyn_cast<IntrinsicInst>(Val: &I)) {
2299 if (II->getIntrinsicID() != Intrinsic::objc_autoreleaseReturnValue ||
2300 !II->use_empty() ||
2301 objcarc::GetRCIdentityRoot(V: II->getOperand(i_nocapture: 0)) != RetOpnd)
2302 break;
2303
2304 // If we've found a matching authoreleaseRV call:
2305 // - If claimRV is attached to the call, insert a call to objc_release
2306 // and erase the autoreleaseRV call.
2307 // - If retainRV is attached to the call, just erase the autoreleaseRV
2308 // call.
2309 if (IsUnsafeClaimRV) {
2310 Builder.SetInsertPoint(II);
2311 Builder.CreateIntrinsic(ID: Intrinsic::objc_release, Args: RetOpnd);
2312 }
2313 II->eraseFromParent();
2314 InsertRetainCall = false;
2315 break;
2316 }
2317
2318 auto *CI = dyn_cast<CallInst>(Val: &I);
2319
2320 if (!CI)
2321 break;
2322
2323 if (objcarc::GetRCIdentityRoot(V: CI) != RetOpnd ||
2324 objcarc::hasAttachedCallOpBundle(CB: CI))
2325 break;
2326
2327 // If we've found an unannotated call that defines RetOpnd, add a
2328 // "clang.arc.attachedcall" operand bundle.
2329 Value *BundleArgs[] = {*objcarc::getAttachedARCFunction(CB: &CB)};
2330 OperandBundleDef OB("clang.arc.attachedcall", BundleArgs);
2331 auto *NewCall = CallBase::addOperandBundle(
2332 CB: CI, ID: LLVMContext::OB_clang_arc_attachedcall, OB, InsertPt: CI->getIterator());
2333 NewCall->copyMetadata(SrcInst: *CI);
2334 CI->replaceAllUsesWith(V: NewCall);
2335 CI->eraseFromParent();
2336 InsertRetainCall = false;
2337 break;
2338 }
2339
2340 if (InsertRetainCall) {
2341 // The retainRV is attached to the call and we've failed to find a
2342 // matching autoreleaseRV or an annotated call in the callee. Emit a call
2343 // to objc_retain.
2344 Builder.SetInsertPoint(RI);
2345 Builder.CreateIntrinsic(ID: Intrinsic::objc_retain, Args: RetOpnd);
2346 }
2347 }
2348}
2349
2350// In contextual profiling, when an inline succeeds, we want to remap the
2351// indices of the callee into the index space of the caller. We can't just leave
2352// them as-is because the same callee may appear in other places in this caller
2353// (other callsites), and its (callee's) counters and sub-contextual profile
2354// tree would be potentially different.
2355// Not all BBs of the callee may survive the opportunistic DCE InlineFunction
2356// does (same goes for callsites in the callee).
2357// We will return a pair of vectors, one for basic block IDs and one for
2358// callsites. For such a vector V, V[Idx] will be -1 if the callee
2359// instrumentation with index Idx did not survive inlining, and a new value
2360// otherwise.
2361// This function will update the caller's instrumentation intrinsics
2362// accordingly, mapping indices as described above. We also replace the "name"
2363// operand because we use it to distinguish between "own" instrumentation and
2364// "from callee" instrumentation when performing the traversal of the CFG of the
2365// caller. We traverse depth-first from the callsite's BB and up to the point we
2366// hit BBs owned by the caller.
2367// The return values will be then used to update the contextual
2368// profile. Note: we only update the "name" and "index" operands in the
2369// instrumentation intrinsics, we leave the hash and total nr of indices as-is,
2370// it's not worth updating those.
2371static std::pair<std::vector<int64_t>, std::vector<int64_t>>
2372remapIndices(Function &Caller, BasicBlock *StartBB,
2373 PGOContextualProfile &CtxProf, uint32_t CalleeCounters,
2374 uint32_t CalleeCallsites) {
2375 // We'll allocate a new ID to imported callsite counters and callsites. We're
2376 // using -1 to indicate a counter we delete. Most likely the entry ID, for
2377 // example, will be deleted - we don't want 2 IDs in the same BB, and the
2378 // entry would have been cloned in the callsite's old BB.
2379 std::vector<int64_t> CalleeCounterMap;
2380 std::vector<int64_t> CalleeCallsiteMap;
2381 CalleeCounterMap.resize(new_size: CalleeCounters, x: -1);
2382 CalleeCallsiteMap.resize(new_size: CalleeCallsites, x: -1);
2383
2384 auto RewriteInstrIfNeeded = [&](InstrProfIncrementInst &Ins) -> bool {
2385 if (Ins.getNameValue() == &Caller)
2386 return false;
2387 const auto OldID = static_cast<uint32_t>(Ins.getIndex()->getZExtValue());
2388 if (CalleeCounterMap[OldID] == -1)
2389 CalleeCounterMap[OldID] = CtxProf.allocateNextCounterIndex(F: Caller);
2390 const auto NewID = static_cast<uint32_t>(CalleeCounterMap[OldID]);
2391
2392 Ins.setNameValue(&Caller);
2393 Ins.setIndex(NewID);
2394 return true;
2395 };
2396
2397 auto RewriteCallsiteInsIfNeeded = [&](InstrProfCallsite &Ins) -> bool {
2398 if (Ins.getNameValue() == &Caller)
2399 return false;
2400 const auto OldID = static_cast<uint32_t>(Ins.getIndex()->getZExtValue());
2401 if (CalleeCallsiteMap[OldID] == -1)
2402 CalleeCallsiteMap[OldID] = CtxProf.allocateNextCallsiteIndex(F: Caller);
2403 const auto NewID = static_cast<uint32_t>(CalleeCallsiteMap[OldID]);
2404
2405 Ins.setNameValue(&Caller);
2406 Ins.setIndex(NewID);
2407 return true;
2408 };
2409
2410 std::deque<BasicBlock *> Worklist;
2411 DenseSet<const BasicBlock *> Seen;
2412 // We will traverse the BBs starting from the callsite BB. The callsite BB
2413 // will have at least a BB ID - maybe its own, and in any case the one coming
2414 // from the cloned function's entry BB. The other BBs we'll start seeing from
2415 // there on may or may not have BB IDs. BBs with IDs belonging to our caller
2416 // are definitely not coming from the imported function and form a boundary
2417 // past which we don't need to traverse anymore. BBs may have no
2418 // instrumentation (because we originally inserted instrumentation as per
2419 // MST), in which case we'll traverse past them. An invariant we'll keep is
2420 // that a BB will have at most 1 BB ID. For example, in the callsite BB, we
2421 // will delete the callee BB's instrumentation. This doesn't result in
2422 // information loss: the entry BB of the callee will have the same count as
2423 // the callsite's BB. At the end of this traversal, all the callee's
2424 // instrumentation would be mapped into the caller's instrumentation index
2425 // space. Some of the callee's counters may be deleted (as mentioned, this
2426 // should result in no loss of information).
2427 Worklist.push_back(x: StartBB);
2428 while (!Worklist.empty()) {
2429 auto *BB = Worklist.front();
2430 Worklist.pop_front();
2431 bool Changed = false;
2432 auto *BBID = CtxProfAnalysis::getBBInstrumentation(BB&: *BB);
2433 if (BBID) {
2434 Changed |= RewriteInstrIfNeeded(*BBID);
2435 // this may be the entryblock from the inlined callee, coming into a BB
2436 // that didn't have instrumentation because of MST decisions. Let's make
2437 // sure it's placed accordingly. This is a noop elsewhere.
2438 BBID->moveBefore(InsertPos: BB->getFirstInsertionPt());
2439 }
2440 for (auto &I : llvm::make_early_inc_range(Range&: *BB)) {
2441 if (auto *Inc = dyn_cast<InstrProfIncrementInst>(Val: &I)) {
2442 if (isa<InstrProfIncrementInstStep>(Val: Inc)) {
2443 // Step instrumentation is used for select instructions. Inlining may
2444 // have propagated a constant resulting in the condition of the select
2445 // being resolved, case in which function cloning resolves the value
2446 // of the select, and elides the select instruction. If that is the
2447 // case, the step parameter of the instrumentation will reflect that.
2448 // We can delete the instrumentation in that case.
2449 if (isa<Constant>(Val: Inc->getStep())) {
2450 assert(!Inc->getNextNode() || !isa<SelectInst>(Inc->getNextNode()));
2451 Inc->eraseFromParent();
2452 } else {
2453 assert(isa_and_nonnull<SelectInst>(Inc->getNextNode()));
2454 RewriteInstrIfNeeded(*Inc);
2455 }
2456 } else if (Inc != BBID) {
2457 // If we're here it means that the BB had more than 1 IDs, presumably
2458 // some coming from the callee. We "made up our mind" to keep the
2459 // first one (which may or may not have been originally the caller's).
2460 // All the others are superfluous and we delete them.
2461 Inc->eraseFromParent();
2462 Changed = true;
2463 }
2464 } else if (auto *CS = dyn_cast<InstrProfCallsite>(Val: &I)) {
2465 Changed |= RewriteCallsiteInsIfNeeded(*CS);
2466 }
2467 }
2468 if (!BBID || Changed)
2469 for (auto *Succ : successors(BB))
2470 if (Seen.insert(V: Succ).second)
2471 Worklist.push_back(x: Succ);
2472 }
2473
2474 assert(!llvm::is_contained(CalleeCounterMap, 0) &&
2475 "Counter index mapping should be either to -1 or to non-zero index, "
2476 "because the 0 "
2477 "index corresponds to the entry BB of the caller");
2478 assert(!llvm::is_contained(CalleeCallsiteMap, 0) &&
2479 "Callsite index mapping should be either to -1 or to non-zero index, "
2480 "because there should have been at least a callsite - the inlined one "
2481 "- which would have had a 0 index.");
2482
2483 return {std::move(CalleeCounterMap), std::move(CalleeCallsiteMap)};
2484}
2485
2486// Inline. If successful, update the contextual profile (if a valid one is
2487// given).
2488// The contextual profile data is organized in trees, as follows:
2489// - each node corresponds to a function
2490// - the root of each tree corresponds to an "entrypoint" - e.g.
2491// RPC handler for server side
2492// - the path from the root to a node is a particular call path
2493// - the counters stored in a node are counter values observed in that
2494// particular call path ("context")
2495// - the edges between nodes are annotated with callsite IDs.
2496//
2497// Updating the contextual profile after an inlining means, at a high level,
2498// copying over the data of the callee, **intentionally without any value
2499// scaling**, and copying over the callees of the inlined callee.
2500llvm::InlineResult
2501llvm::InlineFunction(CallBase &CB, InlineFunctionInfo &IFI,
2502 PGOContextualProfile &CtxProf, bool MergeAttributes,
2503 AAResults *CalleeAAR, bool InsertLifetime,
2504 bool TrackInlineHistory, Function *ForwardVarArgsTo,
2505 OptimizationRemarkEmitter *ORE) {
2506 if (!CtxProf.isInSpecializedModule())
2507 return InlineFunction(CB, IFI, MergeAttributes, CalleeAAR, InsertLifetime,
2508 TrackInlineHistory, ForwardVarArgsTo, ORE);
2509
2510 auto &Caller = *CB.getCaller();
2511 auto &Callee = *CB.getCalledFunction();
2512 auto *StartBB = CB.getParent();
2513
2514 // Get some preliminary data about the callsite before it might get inlined.
2515 // Inlining shouldn't delete the callee, but it's cleaner (and low-cost) to
2516 // get this data upfront and rely less on InlineFunction's behavior.
2517 const auto CalleeGUID = Callee.getGUID();
2518 auto *CallsiteIDIns = CtxProfAnalysis::getCallsiteInstrumentation(CB);
2519 const auto CallsiteID =
2520 static_cast<uint32_t>(CallsiteIDIns->getIndex()->getZExtValue());
2521
2522 const auto NumCalleeCounters = CtxProf.getNumCounters(F: Callee);
2523 const auto NumCalleeCallsites = CtxProf.getNumCallsites(F: Callee);
2524
2525 auto Ret = InlineFunction(CB, IFI, MergeAttributes, CalleeAAR, InsertLifetime,
2526 TrackInlineHistory, ForwardVarArgsTo, ORE);
2527 if (!Ret.isSuccess())
2528 return Ret;
2529
2530 // Inlining succeeded, we don't need the instrumentation of the inlined
2531 // callsite.
2532 CallsiteIDIns->eraseFromParent();
2533
2534 // Assinging Maps and then capturing references into it in the lambda because
2535 // captured structured bindings are a C++20 extension. We do also need a
2536 // capture here, though.
2537 const auto IndicesMaps = remapIndices(Caller, StartBB, CtxProf,
2538 CalleeCounters: NumCalleeCounters, CalleeCallsites: NumCalleeCallsites);
2539 const uint32_t NewCountersSize = CtxProf.getNumCounters(F: Caller);
2540
2541 auto Updater = [&](PGOCtxProfContext &Ctx) {
2542 assert(Ctx.guid() == Caller.getGUID());
2543 const auto &[CalleeCounterMap, CalleeCallsiteMap] = IndicesMaps;
2544 assert(
2545 (Ctx.counters().size() +
2546 llvm::count_if(CalleeCounterMap, [](auto V) { return V != -1; }) ==
2547 NewCountersSize) &&
2548 "The caller's counters size should have grown by the number of new "
2549 "distinct counters inherited from the inlined callee.");
2550 Ctx.resizeCounters(Size: NewCountersSize);
2551 // If the callsite wasn't exercised in this context, the value of the
2552 // counters coming from it is 0 - which it is right now, after resizing them
2553 // - and so we're done.
2554 auto CSIt = Ctx.callsites().find(x: CallsiteID);
2555 if (CSIt == Ctx.callsites().end())
2556 return;
2557 auto CalleeCtxIt = CSIt->second.find(x: CalleeGUID);
2558 // The callsite was exercised, but not with this callee (so presumably this
2559 // is an indirect callsite). Again, we're done here.
2560 if (CalleeCtxIt == CSIt->second.end())
2561 return;
2562
2563 // Let's pull in the counter values and the subcontexts coming from the
2564 // inlined callee.
2565 auto &CalleeCtx = CalleeCtxIt->second;
2566 assert(CalleeCtx.guid() == CalleeGUID);
2567
2568 for (auto I = 0U; I < CalleeCtx.counters().size(); ++I) {
2569 const int64_t NewIndex = CalleeCounterMap[I];
2570 if (NewIndex >= 0) {
2571 assert(NewIndex != 0 && "counter index mapping shouldn't happen to a 0 "
2572 "index, that's the caller's entry BB");
2573 Ctx.counters()[NewIndex] = CalleeCtx.counters()[I];
2574 }
2575 }
2576 for (auto &[I, OtherSet] : CalleeCtx.callsites()) {
2577 const int64_t NewCSIdx = CalleeCallsiteMap[I];
2578 if (NewCSIdx >= 0) {
2579 assert(NewCSIdx != 0 &&
2580 "callsite index mapping shouldn't happen to a 0 index, the "
2581 "caller must've had at least one callsite (with such an index)");
2582 Ctx.ingestAllContexts(CSId: NewCSIdx, Other: std::move(OtherSet));
2583 }
2584 }
2585 // We know the traversal is preorder, so it wouldn't have yet looked at the
2586 // sub-contexts of this context that it's currently visiting. Meaning, the
2587 // erase below invalidates no iterators.
2588 auto Deleted = Ctx.callsites().erase(x: CallsiteID);
2589 assert(Deleted);
2590 (void)Deleted;
2591 };
2592 CtxProf.update(Updater, F: Caller);
2593 return Ret;
2594}
2595
2596llvm::InlineResult llvm::CanInlineCallSite(const CallBase &CB,
2597 InlineFunctionInfo &IFI) {
2598 assert(CB.getParent() && CB.getFunction() && "Instruction not in function!");
2599
2600 // FIXME: we don't inline callbr yet.
2601 if (isa<CallBrInst>(Val: CB))
2602 return InlineResult::failure(Reason: "We don't inline callbr yet.");
2603
2604 // If IFI has any state in it, zap it before we fill it in.
2605 IFI.reset();
2606
2607 Function *CalledFunc = CB.getCalledFunction();
2608 if (!CalledFunc || // Can't inline external function or indirect
2609 CalledFunc->isDeclaration()) // call!
2610 return InlineResult::failure(Reason: "external or indirect");
2611
2612 // Don't inline if we've already inlined this callee through this call site
2613 // before to prevent infinite inlining through mutually recursive functions.
2614 if (MDNode *InlineHistory = CB.getMetadata(KindID: LLVMContext::MD_inline_history)) {
2615 for (const auto &Op : InlineHistory->operands()) {
2616 if (auto *MD = dyn_cast_or_null<ValueAsMetadata>(Val: Op)) {
2617 if (MD->getValue() == CalledFunc) {
2618 return InlineResult::failure(Reason: "inline history");
2619 }
2620 }
2621 }
2622 }
2623
2624 // The inliner does not know how to inline through calls with operand bundles
2625 // in general ...
2626 if (CB.hasOperandBundles()) {
2627 for (int i = 0, e = CB.getNumOperandBundles(); i != e; ++i) {
2628 auto OBUse = CB.getOperandBundleAt(Index: i);
2629 uint32_t Tag = OBUse.getTagID();
2630 // ... but it knows how to inline through "deopt" operand bundles ...
2631 if (Tag == LLVMContext::OB_deopt)
2632 continue;
2633 // ... and "funclet" operand bundles.
2634 if (Tag == LLVMContext::OB_funclet)
2635 continue;
2636 if (Tag == LLVMContext::OB_clang_arc_attachedcall)
2637 continue;
2638 if (Tag == LLVMContext::OB_kcfi)
2639 continue;
2640 if (Tag == LLVMContext::OB_convergencectrl) {
2641 IFI.ConvergenceControlToken = OBUse.Inputs[0].get();
2642 continue;
2643 }
2644
2645 return InlineResult::failure(Reason: "unsupported operand bundle");
2646 }
2647 }
2648
2649 // FIXME: The check below is redundant and incomplete. According to spec, if a
2650 // convergent call is missing a token, then the caller is using uncontrolled
2651 // convergence. If the callee has an entry intrinsic, then the callee is using
2652 // controlled convergence, and the call cannot be inlined. A proper
2653 // implemenation of this check requires a whole new analysis that identifies
2654 // convergence in every function. For now, we skip that and just do this one
2655 // cursory check. The underlying assumption is that in a compiler flow that
2656 // fully implements convergence control tokens, there is no mixing of
2657 // controlled and uncontrolled convergent operations in the whole program.
2658 if (CB.isConvergent()) {
2659 if (!IFI.ConvergenceControlToken &&
2660 getConvergenceEntry(BB&: CalledFunc->getEntryBlock())) {
2661 return InlineResult::failure(
2662 Reason: "convergent call needs convergencectrl operand");
2663 }
2664 }
2665
2666 const BasicBlock *OrigBB = CB.getParent();
2667 const Function *Caller = OrigBB->getParent();
2668
2669 // GC poses two hazards to inlining, which only occur when the callee has GC:
2670 // 1. If the caller has no GC, then the callee's GC must be propagated to the
2671 // caller.
2672 // 2. If the caller has a differing GC, it is invalid to inline.
2673 if (CalledFunc->hasGC()) {
2674 if (Caller->hasGC() && CalledFunc->getGC() != Caller->getGC())
2675 return InlineResult::failure(Reason: "incompatible GC");
2676 }
2677
2678 // Get the personality function from the callee if it contains a landing pad.
2679 Constant *CalledPersonality =
2680 CalledFunc->hasPersonalityFn()
2681 ? CalledFunc->getPersonalityFn()->stripPointerCasts()
2682 : nullptr;
2683
2684 // Find the personality function used by the landing pads of the caller. If it
2685 // exists, then check to see that it matches the personality function used in
2686 // the callee.
2687 Constant *CallerPersonality =
2688 Caller->hasPersonalityFn()
2689 ? Caller->getPersonalityFn()->stripPointerCasts()
2690 : nullptr;
2691 if (CalledPersonality) {
2692 // If the personality functions match, then we can perform the
2693 // inlining. Otherwise, we can't inline.
2694 // TODO: This isn't 100% true. Some personality functions are proper
2695 // supersets of others and can be used in place of the other.
2696 if (CallerPersonality && CalledPersonality != CallerPersonality)
2697 return InlineResult::failure(Reason: "incompatible personality");
2698 }
2699
2700 // We need to figure out which funclet the callsite was in so that we may
2701 // properly nest the callee.
2702 if (CallerPersonality) {
2703 EHPersonality Personality = classifyEHPersonality(Pers: CallerPersonality);
2704 if (isScopedEHPersonality(Pers: Personality)) {
2705 std::optional<OperandBundleUse> ParentFunclet =
2706 CB.getOperandBundle(ID: LLVMContext::OB_funclet);
2707 if (ParentFunclet)
2708 IFI.CallSiteEHPad = cast<FuncletPadInst>(Val: ParentFunclet->Inputs.front());
2709
2710 // OK, the inlining site is legal. What about the target function?
2711
2712 if (IFI.CallSiteEHPad) {
2713 if (Personality == EHPersonality::MSVC_CXX) {
2714 // The MSVC personality cannot tolerate catches getting inlined into
2715 // cleanup funclets.
2716 if (isa<CleanupPadInst>(Val: IFI.CallSiteEHPad)) {
2717 // Ok, the call site is within a cleanuppad. Let's check the callee
2718 // for catchpads.
2719 for (const BasicBlock &CalledBB : *CalledFunc) {
2720 if (isa<CatchSwitchInst>(Val: CalledBB.getFirstNonPHIIt()))
2721 return InlineResult::failure(Reason: "catch in cleanup funclet");
2722 }
2723 }
2724 } else if (isAsynchronousEHPersonality(Pers: Personality)) {
2725 // SEH is even less tolerant, there may not be any sort of exceptional
2726 // funclet in the callee.
2727 for (const BasicBlock &CalledBB : *CalledFunc) {
2728 if (CalledBB.isEHPad())
2729 return InlineResult::failure(Reason: "SEH in cleanup funclet");
2730 }
2731 }
2732 }
2733 }
2734 }
2735
2736 return InlineResult::success();
2737}
2738
2739/// This function inlines the called function into the basic block of the
2740/// caller. This returns false if it is not possible to inline this call.
2741/// The program is still in a well defined state if this occurs though.
2742///
2743/// Note that this only does one level of inlining. For example, if the
2744/// instruction 'call B' is inlined, and 'B' calls 'C', then the call to 'C' now
2745/// exists in the instruction stream. Similarly this will inline a recursive
2746/// function by one level.
2747void llvm::InlineFunctionImpl(CallBase &CB, InlineFunctionInfo &IFI,
2748 bool MergeAttributes, AAResults *CalleeAAR,
2749 bool InsertLifetime, bool TrackInlineHistory,
2750 Function *ForwardVarArgsTo,
2751 OptimizationRemarkEmitter *ORE) {
2752 BasicBlock *OrigBB = CB.getParent();
2753 Function *Caller = OrigBB->getParent();
2754 Function *CalledFunc = CB.getCalledFunction();
2755 assert(CalledFunc && !CalledFunc->isDeclaration() &&
2756 "CanInlineCallSite should have verified direct call to definition");
2757
2758 // Determine if we are dealing with a call in an EHPad which does not unwind
2759 // to caller.
2760 bool EHPadForCallUnwindsLocally = false;
2761 if (IFI.CallSiteEHPad && isa<CallInst>(Val: CB)) {
2762 UnwindDestMemoTy FuncletUnwindMap;
2763 Value *CallSiteUnwindDestToken =
2764 getUnwindDestToken(EHPad: IFI.CallSiteEHPad, MemoMap&: FuncletUnwindMap);
2765
2766 EHPadForCallUnwindsLocally =
2767 CallSiteUnwindDestToken &&
2768 !isa<ConstantTokenNone>(Val: CallSiteUnwindDestToken);
2769 }
2770
2771 // Get an iterator to the last basic block in the function, which will have
2772 // the new function inlined after it.
2773 Function::iterator LastBlock = --Caller->end();
2774
2775 // Make sure to capture all of the return instructions from the cloned
2776 // function.
2777 SmallVector<ReturnInst*, 8> Returns;
2778 ClonedCodeInfo InlinedFunctionInfo;
2779 Function::iterator FirstNewBlock;
2780
2781 // GC poses two hazards to inlining, which only occur when the callee has GC:
2782 // 1. If the caller has no GC, then the callee's GC must be propagated to the
2783 // caller.
2784 // 2. If the caller has a differing GC, it is invalid to inline.
2785 if (CalledFunc->hasGC()) {
2786 if (!Caller->hasGC())
2787 Caller->setGC(CalledFunc->getGC());
2788 else {
2789 assert(CalledFunc->getGC() == Caller->getGC() &&
2790 "CanInlineCallSite should have verified compatible GCs");
2791 }
2792 }
2793
2794 if (CalledFunc->hasPersonalityFn()) {
2795 Constant *CalledPersonality =
2796 CalledFunc->getPersonalityFn()->stripPointerCasts();
2797 if (!Caller->hasPersonalityFn()) {
2798 Caller->setPersonalityFn(CalledPersonality);
2799 } else
2800 assert(Caller->getPersonalityFn()->stripPointerCasts() ==
2801 CalledPersonality &&
2802 "CanInlineCallSite should have verified compatible personality");
2803 }
2804
2805 { // Scope to destroy VMap after cloning.
2806 ValueToValueMapTy VMap;
2807 struct ByValInit {
2808 Value *Dst;
2809 Value *Src;
2810 MaybeAlign SrcAlign;
2811 Type *Ty;
2812 };
2813 // Keep a list of tuples (dst, src, src_align) to emit byval
2814 // initializations. Src Alignment is only available though the callbase,
2815 // therefore has to be saved.
2816 SmallVector<ByValInit, 4> ByValInits;
2817
2818 // When inlining a function that contains noalias scope metadata,
2819 // this metadata needs to be cloned so that the inlined blocks
2820 // have different "unique scopes" at every call site.
2821 // Track the metadata that must be cloned. Do this before other changes to
2822 // the function, so that we do not get in trouble when inlining caller ==
2823 // callee.
2824 ScopedAliasMetadataDeepCloner SAMetadataCloner(CB.getCalledFunction());
2825
2826 auto &DL = Caller->getDataLayout();
2827
2828 // Calculate the vector of arguments to pass into the function cloner, which
2829 // matches up the formal to the actual argument values.
2830 auto AI = CB.arg_begin();
2831 unsigned ArgNo = 0;
2832 for (Function::arg_iterator I = CalledFunc->arg_begin(),
2833 E = CalledFunc->arg_end(); I != E; ++I, ++AI, ++ArgNo) {
2834 Value *ActualArg = *AI;
2835
2836 // When byval arguments actually inlined, we need to make the copy implied
2837 // by them explicit. However, we don't do this if the callee is readonly
2838 // or readnone, because the copy would be unneeded: the callee doesn't
2839 // modify the struct.
2840 if (CB.isByValArgument(ArgNo)) {
2841 ActualArg = HandleByValArgument(ByValType: CB.getParamByValType(ArgNo), Arg: ActualArg,
2842 TheCall: &CB, CalledFunc, IFI,
2843 ByValAlignment: CalledFunc->getParamAlign(ArgNo));
2844 if (ActualArg != *AI)
2845 ByValInits.push_back(Elt: {.Dst: ActualArg, .Src: (Value *)*AI,
2846 .SrcAlign: CB.getParamAlign(ArgNo),
2847 .Ty: CB.getParamByValType(ArgNo)});
2848 }
2849
2850 VMap[&*I] = ActualArg;
2851 }
2852
2853 // TODO: Remove this when users have been updated to the assume bundles.
2854 // Add alignment assumptions if necessary. We do this before the inlined
2855 // instructions are actually cloned into the caller so that we can easily
2856 // check what will be known at the start of the inlined code.
2857 AddAlignmentAssumptions(CB, IFI);
2858
2859 AssumptionCache *AC =
2860 IFI.GetAssumptionCache ? &IFI.GetAssumptionCache(*Caller) : nullptr;
2861
2862 /// Preserve all attributes on of the call and its parameters.
2863 salvageKnowledge(I: &CB, AC);
2864
2865 // We want the inliner to prune the code as it copies. We would LOVE to
2866 // have no dead or constant instructions leftover after inlining occurs
2867 // (which can happen, e.g., because an argument was constant), but we'll be
2868 // happy with whatever the cloner can do.
2869 CloneAndPruneFunctionInto(NewFunc: Caller, OldFunc: CalledFunc, VMap,
2870 /*ModuleLevelChanges=*/false, Returns, NameSuffix: ".i",
2871 CodeInfo&: InlinedFunctionInfo);
2872 // Remember the first block that is newly cloned over.
2873 FirstNewBlock = LastBlock; ++FirstNewBlock;
2874
2875 // Insert retainRV/clainRV runtime calls.
2876 objcarc::ARCInstKind RVCallKind = objcarc::getAttachedARCFunctionKind(CB: &CB);
2877 if (RVCallKind != objcarc::ARCInstKind::None)
2878 inlineRetainOrClaimRVCalls(CB, RVCallKind, Returns);
2879
2880 // Updated caller/callee profiles only when requested. For sample loader
2881 // inlining, the context-sensitive inlinee profile doesn't need to be
2882 // subtracted from callee profile, and the inlined clone also doesn't need
2883 // to be scaled based on call site count.
2884 if (IFI.UpdateProfile) {
2885 if (IFI.CallerBFI != nullptr && IFI.CalleeBFI != nullptr)
2886 // Update the BFI of blocks cloned into the caller.
2887 updateCallerBFI(CallSiteBlock: OrigBB, VMap, CallerBFI: IFI.CallerBFI, CalleeBFI: IFI.CalleeBFI,
2888 CalleeEntryBlock: CalledFunc->front());
2889
2890 if (auto Profile = CalledFunc->getEntryCount())
2891 updateCallProfile(Callee: CalledFunc, VMap, CalleeEntryCount: *Profile, TheCall: CB, PSI: IFI.PSI,
2892 CallerBFI: IFI.CallerBFI);
2893 }
2894
2895 // Inject byval arguments initialization.
2896 for (ByValInit &Init : ByValInits)
2897 HandleByValArgumentInit(ByValType: Init.Ty, Dst: Init.Dst, Src: Init.Src, SrcAlign: Init.SrcAlign,
2898 M: Caller->getParent(), InsertBlock: &*FirstNewBlock, IFI,
2899 CalledFunc);
2900
2901 std::optional<OperandBundleUse> ParentDeopt =
2902 CB.getOperandBundle(ID: LLVMContext::OB_deopt);
2903 if (ParentDeopt) {
2904 SmallVector<OperandBundleDef, 2> OpDefs;
2905
2906 for (auto &VH : InlinedFunctionInfo.OperandBundleCallSites) {
2907 CallBase *ICS = dyn_cast_or_null<CallBase>(Val&: VH);
2908 if (!ICS)
2909 continue; // instruction was DCE'd or RAUW'ed to undef
2910
2911 OpDefs.clear();
2912
2913 OpDefs.reserve(N: ICS->getNumOperandBundles());
2914
2915 for (unsigned COBi = 0, COBe = ICS->getNumOperandBundles(); COBi < COBe;
2916 ++COBi) {
2917 auto ChildOB = ICS->getOperandBundleAt(Index: COBi);
2918 if (ChildOB.getTagID() != LLVMContext::OB_deopt) {
2919 // If the inlined call has other operand bundles, let them be
2920 OpDefs.emplace_back(Args&: ChildOB);
2921 continue;
2922 }
2923
2924 // It may be useful to separate this logic (of handling operand
2925 // bundles) out to a separate "policy" component if this gets crowded.
2926 // Prepend the parent's deoptimization continuation to the newly
2927 // inlined call's deoptimization continuation.
2928 std::vector<Value *> MergedDeoptArgs;
2929 MergedDeoptArgs.reserve(n: ParentDeopt->Inputs.size() +
2930 ChildOB.Inputs.size());
2931
2932 llvm::append_range(C&: MergedDeoptArgs, R&: ParentDeopt->Inputs);
2933 llvm::append_range(C&: MergedDeoptArgs, R&: ChildOB.Inputs);
2934
2935 OpDefs.emplace_back(Args: "deopt", Args: std::move(MergedDeoptArgs));
2936 }
2937
2938 Instruction *NewI = CallBase::Create(CB: ICS, Bundles: OpDefs, InsertPt: ICS->getIterator());
2939
2940 // Note: the RAUW does the appropriate fixup in VMap, so we need to do
2941 // this even if the call returns void.
2942 ICS->replaceAllUsesWith(V: NewI);
2943
2944 VH = nullptr;
2945 ICS->eraseFromParent();
2946 }
2947 }
2948
2949 // For 'nodebug' functions, the associated DISubprogram is always null.
2950 // Conservatively avoid propagating the callsite debug location to
2951 // instructions inlined from a function whose DISubprogram is not null.
2952 fixupLineNumbers(Fn: Caller, FI: FirstNewBlock, TheCall: &CB,
2953 CalleeHasDebugInfo: CalledFunc->getSubprogram() != nullptr);
2954
2955 if (isAssignmentTrackingEnabled(M: *Caller->getParent())) {
2956 // Interpret inlined stores to caller-local variables as assignments.
2957 trackInlinedStores(Start: FirstNewBlock, End: Caller->end(), CB);
2958
2959 // Update DIAssignID metadata attachments and uses so that they are
2960 // unique to this inlined instance.
2961 fixupAssignments(Start: FirstNewBlock, End: Caller->end());
2962 }
2963
2964 // Now clone the inlined noalias scope metadata.
2965 SAMetadataCloner.clone();
2966 SAMetadataCloner.remap(FStart: FirstNewBlock, FEnd: Caller->end());
2967
2968 // Add noalias metadata if necessary.
2969 AddAliasScopeMetadata(CB, VMap, DL, CalleeAAR, InlinedFunctionInfo);
2970
2971 // Clone return attributes on the callsite into the calls within the inlined
2972 // function which feed into its return value.
2973 AddReturnAttributes(CB, VMap, InlinedFunctionInfo);
2974
2975 // Clone attributes on the params of the callsite to calls within the
2976 // inlined function which use the same param.
2977 AddParamAndFnBasicAttributes(CB, VMap, InlinedFunctionInfo);
2978
2979 propagateMemProfMetadata(
2980 Callee: CalledFunc, CB, ContainsMemProfMetadata: InlinedFunctionInfo.ContainsMemProfMetadata, VMap, ORE);
2981
2982 // Propagate metadata on the callsite if necessary.
2983 PropagateCallSiteMetadata(CB, FStart: FirstNewBlock, FEnd: Caller->end());
2984
2985 // Propagate an allocation wrapper's !alloc_token if necessary.
2986 propagateAllocTokenMetadata(CalledFunc, CB, VMap, InlinedFunctionInfo);
2987
2988 // Propagate implicit ref metadata.
2989 if (CalledFunc->hasMetadata(KindID: LLVMContext::MD_implicit_ref)) {
2990 SmallVector<MDNode *> MDs;
2991 CalledFunc->getMetadata(KindID: LLVMContext::MD_implicit_ref, MDs);
2992 for (MDNode *MD : MDs) {
2993 Caller->addMetadata(KindID: LLVMContext::MD_implicit_ref, MD&: *MD);
2994 }
2995 }
2996
2997 // Propagate inlined.from metadata for dontcall diagnostics.
2998 PropagateInlinedFromMetadata(CB, CalledFuncName: CalledFunc->getName(), CallerFuncName: Caller->getName(),
2999 FStart: FirstNewBlock, FEnd: Caller->end());
3000
3001 // Register any cloned assumptions.
3002 if (IFI.GetAssumptionCache)
3003 for (BasicBlock &NewBlock :
3004 make_range(x: FirstNewBlock->getIterator(), y: Caller->end()))
3005 for (Instruction &I : NewBlock)
3006 if (auto *II = dyn_cast<AssumeInst>(Val: &I))
3007 IFI.GetAssumptionCache(*Caller).registerAssumption(CI: II);
3008 }
3009
3010 if (IFI.ConvergenceControlToken) {
3011 IntrinsicInst *IntrinsicCall = getConvergenceEntry(BB&: *FirstNewBlock);
3012 if (IntrinsicCall) {
3013 IntrinsicCall->replaceAllUsesWith(V: IFI.ConvergenceControlToken);
3014 IntrinsicCall->eraseFromParent();
3015 }
3016 }
3017
3018 // If there are any alloca instructions in the block that used to be the entry
3019 // block for the callee, move them to the entry block of the caller. First
3020 // calculate which instruction they should be inserted before. We insert the
3021 // instructions at the end of the current alloca list.
3022 {
3023 BasicBlock::iterator InsertPoint = Caller->begin()->begin();
3024 for (BasicBlock::iterator I = FirstNewBlock->begin(),
3025 E = FirstNewBlock->end(); I != E; ) {
3026 AllocaInst *AI = dyn_cast<AllocaInst>(Val: I++);
3027 if (!AI) continue;
3028
3029 // If the alloca is now dead, remove it. This often occurs due to code
3030 // specialization.
3031 if (AI->use_empty()) {
3032 AI->eraseFromParent();
3033 continue;
3034 }
3035
3036 if (!allocaWouldBeStaticInEntry(AI))
3037 continue;
3038
3039 // Keep track of the static allocas that we inline into the caller.
3040 IFI.StaticAllocas.push_back(Elt: AI);
3041
3042 // Scan for the block of allocas that we can move over, and move them
3043 // all at once.
3044 while (isa<AllocaInst>(Val: I) &&
3045 !cast<AllocaInst>(Val&: I)->use_empty() &&
3046 allocaWouldBeStaticInEntry(AI: cast<AllocaInst>(Val&: I))) {
3047 IFI.StaticAllocas.push_back(Elt: cast<AllocaInst>(Val&: I));
3048 ++I;
3049 }
3050
3051 // Transfer all of the allocas over in a block. Using splice means
3052 // that the instructions aren't removed from the symbol table, then
3053 // reinserted.
3054 I.setTailBit(true);
3055 Caller->getEntryBlock().splice(ToIt: InsertPoint, FromBB: &*FirstNewBlock,
3056 FromBeginIt: AI->getIterator(), FromEndIt: I);
3057 }
3058 }
3059
3060 // If the call to the callee cannot throw, set the 'nounwind' flag on any
3061 // calls that we inline.
3062 bool MarkNoUnwind = CB.doesNotThrow();
3063
3064 SmallVector<Value*,4> VarArgsToForward;
3065 SmallVector<AttributeSet, 4> VarArgsAttrs;
3066 for (unsigned i = CalledFunc->getFunctionType()->getNumParams();
3067 i < CB.arg_size(); i++) {
3068 VarArgsToForward.push_back(Elt: CB.getArgOperand(i));
3069 VarArgsAttrs.push_back(Elt: CB.getAttributes().getParamAttrs(ArgNo: i));
3070 }
3071
3072 bool InlinedMustTailCalls = false, InlinedDeoptimizeCalls = false;
3073 if (InlinedFunctionInfo.ContainsCalls) {
3074 CallInst::TailCallKind CallSiteTailKind = CallInst::TCK_None;
3075 if (CallInst *CI = dyn_cast<CallInst>(Val: &CB))
3076 CallSiteTailKind = CI->getTailCallKind();
3077
3078 // For inlining purposes, the "notail" marker is the same as no marker.
3079 if (CallSiteTailKind == CallInst::TCK_NoTail)
3080 CallSiteTailKind = CallInst::TCK_None;
3081
3082 for (Function::iterator BB = FirstNewBlock, E = Caller->end(); BB != E;
3083 ++BB) {
3084 for (Instruction &I : llvm::make_early_inc_range(Range&: *BB)) {
3085 CallInst *CI = dyn_cast<CallInst>(Val: &I);
3086 if (!CI)
3087 continue;
3088
3089 // Forward varargs from inlined call site to calls to the
3090 // ForwardVarArgsTo function, if requested, and to musttail calls.
3091 if (!VarArgsToForward.empty() &&
3092 ((ForwardVarArgsTo &&
3093 CI->getCalledFunction() == ForwardVarArgsTo) ||
3094 CI->isMustTailCall())) {
3095 // Collect attributes for non-vararg parameters.
3096 AttributeList Attrs = CI->getAttributes();
3097 SmallVector<AttributeSet, 8> ArgAttrs;
3098 if (!Attrs.isEmpty() || !VarArgsAttrs.empty()) {
3099 for (unsigned ArgNo = 0;
3100 ArgNo < CI->getFunctionType()->getNumParams(); ++ArgNo)
3101 ArgAttrs.push_back(Elt: Attrs.getParamAttrs(ArgNo));
3102 }
3103
3104 // Add VarArg attributes.
3105 ArgAttrs.append(in_start: VarArgsAttrs.begin(), in_end: VarArgsAttrs.end());
3106 Attrs = AttributeList::get(C&: CI->getContext(), FnAttrs: Attrs.getFnAttrs(),
3107 RetAttrs: Attrs.getRetAttrs(), ArgAttrs);
3108 // Add VarArgs to existing parameters.
3109 SmallVector<Value *, 6> Params(CI->args());
3110 Params.append(in_start: VarArgsToForward.begin(), in_end: VarArgsToForward.end());
3111 CallInst *NewCI = CallInst::Create(
3112 Ty: CI->getFunctionType(), Func: CI->getCalledOperand(), Args: Params, NameStr: "", InsertBefore: CI->getIterator());
3113 NewCI->setDebugLoc(CI->getDebugLoc());
3114 NewCI->setAttributes(Attrs);
3115 NewCI->setCallingConv(CI->getCallingConv());
3116 CI->replaceAllUsesWith(V: NewCI);
3117 CI->eraseFromParent();
3118 CI = NewCI;
3119 }
3120
3121 if (Function *F = CI->getCalledFunction())
3122 InlinedDeoptimizeCalls |=
3123 F->getIntrinsicID() == Intrinsic::experimental_deoptimize;
3124
3125 // We need to reduce the strength of any inlined tail calls. For
3126 // musttail, we have to avoid introducing potential unbounded stack
3127 // growth. For example, if functions 'f' and 'g' are mutually recursive
3128 // with musttail, we can inline 'g' into 'f' so long as we preserve
3129 // musttail on the cloned call to 'f'. If either the inlined call site
3130 // or the cloned call site is *not* musttail, the program already has
3131 // one frame of stack growth, so it's safe to remove musttail. Here is
3132 // a table of example transformations:
3133 //
3134 // f -> musttail g -> musttail f ==> f -> musttail f
3135 // f -> musttail g -> tail f ==> f -> tail f
3136 // f -> g -> musttail f ==> f -> f
3137 // f -> g -> tail f ==> f -> f
3138 //
3139 // Inlined notail calls should remain notail calls.
3140 CallInst::TailCallKind ChildTCK = CI->getTailCallKind();
3141 if (ChildTCK != CallInst::TCK_NoTail)
3142 ChildTCK = std::min(a: CallSiteTailKind, b: ChildTCK);
3143 CI->setTailCallKind(ChildTCK);
3144 InlinedMustTailCalls |= CI->isMustTailCall();
3145
3146 // Call sites inlined through a 'nounwind' call site should be
3147 // 'nounwind' as well. However, avoid marking call sites explicitly
3148 // where possible. This helps expose more opportunities for CSE after
3149 // inlining, commonly when the callee is an intrinsic.
3150 if (MarkNoUnwind && !CI->doesNotThrow())
3151 CI->setDoesNotThrow();
3152 }
3153 }
3154 }
3155
3156 // Leave lifetime markers for the static alloca's, scoping them to the
3157 // function we just inlined.
3158 // We need to insert lifetime intrinsics even at O0 to avoid invalid
3159 // access caused by multithreaded coroutines. The check
3160 // `Caller->isPresplitCoroutine()` would affect AlwaysInliner at O0 only.
3161 if ((InsertLifetime || Caller->isPresplitCoroutine()) &&
3162 !IFI.StaticAllocas.empty()) {
3163 IRBuilder<> builder(FirstNewBlock->begin());
3164 for (AllocaInst *AI : IFI.StaticAllocas) {
3165 // Don't mark swifterror allocas. They can't have bitcast uses.
3166 if (AI->isSwiftError())
3167 continue;
3168
3169 // If the alloca is already scoped to something smaller than the whole
3170 // function then there's no need to add redundant, less accurate markers.
3171 if (hasLifetimeMarkers(AI))
3172 continue;
3173
3174 std::optional<TypeSize> Size = AI->getAllocationSize(DL: AI->getDataLayout());
3175 if (Size && Size->isZero())
3176 continue;
3177
3178 builder.CreateLifetimeStart(Ptr: AI);
3179 for (ReturnInst *RI : Returns) {
3180 // Don't insert llvm.lifetime.end calls between a musttail or deoptimize
3181 // call and a return. The return kills all local allocas.
3182 if (InlinedMustTailCalls &&
3183 RI->getParent()->getTerminatingMustTailCall())
3184 continue;
3185 if (InlinedDeoptimizeCalls &&
3186 RI->getParent()->getTerminatingDeoptimizeCall())
3187 continue;
3188 IRBuilder<>(RI).CreateLifetimeEnd(Ptr: AI);
3189 }
3190 }
3191 }
3192
3193 // If the inlined code contained dynamic alloca instructions, wrap the inlined
3194 // code with llvm.stacksave/llvm.stackrestore intrinsics.
3195 if (InlinedFunctionInfo.ContainsDynamicAllocas) {
3196 // Insert the llvm.stacksave.
3197 CallInst *SavedPtr =
3198 IRBuilder<>(FirstNewBlock->begin()).CreateStackSave(Name: "savedstack");
3199
3200 // Insert a call to llvm.stackrestore before any return instructions in the
3201 // inlined function.
3202 for (ReturnInst *RI : Returns) {
3203 // Don't insert llvm.stackrestore calls between a musttail or deoptimize
3204 // call and a return. The return will restore the stack pointer.
3205 if (InlinedMustTailCalls && RI->getParent()->getTerminatingMustTailCall())
3206 continue;
3207 if (InlinedDeoptimizeCalls && RI->getParent()->getTerminatingDeoptimizeCall())
3208 continue;
3209 IRBuilder<>(RI).CreateStackRestore(Ptr: SavedPtr);
3210 }
3211 }
3212
3213 // If we are inlining for an invoke instruction, we must make sure to rewrite
3214 // any call instructions into invoke instructions. This is sensitive to which
3215 // funclet pads were top-level in the inlinee, so must be done before
3216 // rewriting the "parent pad" links.
3217 if (auto *II = dyn_cast<InvokeInst>(Val: &CB)) {
3218 BasicBlock *UnwindDest = II->getUnwindDest();
3219 BasicBlock::iterator FirstNonPHI = UnwindDest->getFirstNonPHIIt();
3220 if (isa<LandingPadInst>(Val: FirstNonPHI)) {
3221 HandleInlinedLandingPad(II, FirstNewBlock: &*FirstNewBlock, InlinedCodeInfo&: InlinedFunctionInfo);
3222 } else {
3223 HandleInlinedEHPad(II, FirstNewBlock: &*FirstNewBlock, InlinedCodeInfo&: InlinedFunctionInfo);
3224 }
3225 }
3226
3227 // Update the lexical scopes of the new funclets and callsites.
3228 // Anything that had 'none' as its parent is now nested inside the callsite's
3229 // EHPad.
3230 if (IFI.CallSiteEHPad) {
3231 for (Function::iterator BB = FirstNewBlock->getIterator(),
3232 E = Caller->end();
3233 BB != E; ++BB) {
3234 // Add bundle operands to inlined call sites.
3235 PropagateOperandBundles(InlinedBB: BB, CallSiteEHPad: IFI.CallSiteEHPad);
3236
3237 // It is problematic if the inlinee has a cleanupret which unwinds to
3238 // caller and we inline it into a call site which doesn't unwind but into
3239 // an EH pad that does. Such an edge must be dynamically unreachable.
3240 // As such, we replace the cleanupret with unreachable.
3241 if (auto *CleanupRet = dyn_cast<CleanupReturnInst>(Val: BB->getTerminator()))
3242 if (CleanupRet->unwindsToCaller() && EHPadForCallUnwindsLocally)
3243 changeToUnreachable(I: CleanupRet);
3244
3245 BasicBlock::iterator I = BB->getFirstNonPHIIt();
3246 if (!I->isEHPad())
3247 continue;
3248
3249 if (auto *CatchSwitch = dyn_cast<CatchSwitchInst>(Val&: I)) {
3250 if (isa<ConstantTokenNone>(Val: CatchSwitch->getParentPad()))
3251 CatchSwitch->setParentPad(IFI.CallSiteEHPad);
3252 } else {
3253 auto *FPI = cast<FuncletPadInst>(Val&: I);
3254 if (isa<ConstantTokenNone>(Val: FPI->getParentPad()))
3255 FPI->setParentPad(IFI.CallSiteEHPad);
3256 }
3257 }
3258 }
3259
3260 if (InlinedDeoptimizeCalls) {
3261 // We need to at least remove the deoptimizing returns from the Return set,
3262 // so that the control flow from those returns does not get merged into the
3263 // caller (but terminate it instead). If the caller's return type does not
3264 // match the callee's return type, we also need to change the return type of
3265 // the intrinsic.
3266 if (Caller->getReturnType() == CB.getType()) {
3267 llvm::erase_if(C&: Returns, P: [](ReturnInst *RI) {
3268 return RI->getParent()->getTerminatingDeoptimizeCall() != nullptr;
3269 });
3270 } else {
3271 SmallVector<ReturnInst *, 8> NormalReturns;
3272 Function *NewDeoptIntrinsic = Intrinsic::getOrInsertDeclaration(
3273 M: Caller->getParent(), id: Intrinsic::experimental_deoptimize,
3274 OverloadTys: {Caller->getReturnType()});
3275
3276 for (ReturnInst *RI : Returns) {
3277 CallInst *DeoptCall = RI->getParent()->getTerminatingDeoptimizeCall();
3278 if (!DeoptCall) {
3279 NormalReturns.push_back(Elt: RI);
3280 continue;
3281 }
3282
3283 // The calling convention on the deoptimize call itself may be bogus,
3284 // since the code we're inlining may have undefined behavior (and may
3285 // never actually execute at runtime); but all
3286 // @llvm.experimental.deoptimize declarations have to have the same
3287 // calling convention in a well-formed module.
3288 auto CallingConv = DeoptCall->getCalledFunction()->getCallingConv();
3289 NewDeoptIntrinsic->setCallingConv(CallingConv);
3290 auto *CurBB = RI->getParent();
3291 RI->eraseFromParent();
3292
3293 SmallVector<Value *, 4> CallArgs(DeoptCall->args());
3294
3295 SmallVector<OperandBundleDef, 1> OpBundles;
3296 DeoptCall->getOperandBundlesAsDefs(Defs&: OpBundles);
3297 auto DeoptAttributes = DeoptCall->getAttributes();
3298 DeoptCall->eraseFromParent();
3299 assert(!OpBundles.empty() &&
3300 "Expected at least the deopt operand bundle");
3301
3302 IRBuilder<> Builder(CurBB);
3303 CallInst *NewDeoptCall =
3304 Builder.CreateCall(Callee: NewDeoptIntrinsic, Args: CallArgs, OpBundles);
3305 NewDeoptCall->setCallingConv(CallingConv);
3306 NewDeoptCall->setAttributes(DeoptAttributes);
3307 if (NewDeoptCall->getType()->isVoidTy())
3308 Builder.CreateRetVoid();
3309 else
3310 Builder.CreateRet(V: NewDeoptCall);
3311 // Since the ret type is changed, remove the incompatible attributes.
3312 NewDeoptCall->removeRetAttrs(AttrsToRemove: AttributeFuncs::typeIncompatible(
3313 Ty: NewDeoptCall->getType(), AS: NewDeoptCall->getRetAttributes()));
3314 }
3315
3316 // Leave behind the normal returns so we can merge control flow.
3317 std::swap(LHS&: Returns, RHS&: NormalReturns);
3318 }
3319 }
3320
3321 // Handle any inlined musttail call sites. In order for a new call site to be
3322 // musttail, the source of the clone and the inlined call site must have been
3323 // musttail. Therefore it's safe to return without merging control into the
3324 // phi below.
3325 if (InlinedMustTailCalls) {
3326 // Handle the returns preceded by musttail calls separately.
3327 SmallVector<ReturnInst *, 8> NormalReturns;
3328 for (ReturnInst *RI : Returns) {
3329 CallInst *ReturnedMustTail =
3330 RI->getParent()->getTerminatingMustTailCall();
3331 if (!ReturnedMustTail)
3332 NormalReturns.push_back(Elt: RI);
3333 }
3334
3335 // Leave behind the normal returns so we can merge control flow.
3336 std::swap(LHS&: Returns, RHS&: NormalReturns);
3337 }
3338
3339 // Now that all of the transforms on the inlined code have taken place but
3340 // before we splice the inlined code into the CFG and lose track of which
3341 // blocks were actually inlined, collect the call sites. We only do this if
3342 // call graph updates weren't requested, as those provide value handle based
3343 // tracking of inlined call sites instead. Calls to intrinsics are not
3344 // collected because they are not inlineable.
3345 if (InlinedFunctionInfo.ContainsCalls) {
3346 // Otherwise just collect the raw call sites that were inlined.
3347 for (BasicBlock &NewBB :
3348 make_range(x: FirstNewBlock->getIterator(), y: Caller->end()))
3349 for (Instruction &I : NewBB)
3350 if (auto *CB = dyn_cast<CallBase>(Val: &I))
3351 if (!(CB->getCalledFunction() &&
3352 CB->getCalledFunction()->isIntrinsic()))
3353 IFI.InlinedCallSites.push_back(Elt: CB);
3354 }
3355
3356 for (CallBase *ICB : IFI.InlinedCallSites) {
3357 // We only track inline history if requested, or if the inlined call site
3358 // was originally an indirect call (it may have become a direct call
3359 // during inlining).
3360 if (TrackInlineHistory ||
3361 InlinedFunctionInfo.OriginallyIndirectCalls.contains(key: ICB)) {
3362 // !inline_history is {Callee, CB.inline_history, ICB.inline_history}.
3363 // Metadata nodes may be null if the referenced function was erased from
3364 // the module.
3365 SmallVector<Metadata *, 4> History;
3366 History.push_back(Elt: ValueAsMetadata::get(V: CalledFunc));
3367 if (MDNode *CBHistory = CB.getMetadata(KindID: LLVMContext::MD_inline_history)) {
3368 for (const auto &Op : CBHistory->operands()) {
3369 if (Op)
3370 History.push_back(Elt: Op.get());
3371 }
3372 }
3373 if (MDNode *CBHistory =
3374 ICB->getMetadata(KindID: LLVMContext::MD_inline_history)) {
3375 for (const auto &Op : CBHistory->operands()) {
3376 if (Op)
3377 History.push_back(Elt: Op.get());
3378 }
3379 }
3380 MDNode *NewHistory = MDNode::get(Context&: Caller->getContext(), MDs: History);
3381 ICB->setMetadata(KindID: LLVMContext::MD_inline_history, Node: NewHistory);
3382 }
3383 }
3384
3385 // If we cloned in _exactly one_ basic block, and if that block ends in a
3386 // return instruction, we splice the body of the inlined callee directly into
3387 // the calling basic block.
3388 if (Returns.size() == 1 && std::distance(first: FirstNewBlock, last: Caller->end()) == 1) {
3389 // Move all of the instructions right before the call.
3390 OrigBB->splice(ToIt: CB.getIterator(), FromBB: &*FirstNewBlock, FromBeginIt: FirstNewBlock->begin(),
3391 FromEndIt: FirstNewBlock->end());
3392 // Remove the cloned basic block.
3393 Caller->back().eraseFromParent();
3394
3395 // If the call site was an invoke instruction, add a branch to the normal
3396 // destination.
3397 if (InvokeInst *II = dyn_cast<InvokeInst>(Val: &CB)) {
3398 UncondBrInst *NewBr =
3399 UncondBrInst::Create(Target: II->getNormalDest(), InsertBefore: CB.getIterator());
3400 NewBr->setDebugLoc(Returns[0]->getDebugLoc());
3401 }
3402
3403 // If the return instruction returned a value, replace uses of the call with
3404 // uses of the returned value.
3405 if (!CB.use_empty()) {
3406 ReturnInst *R = Returns[0];
3407 if (&CB == R->getReturnValue())
3408 CB.replaceAllUsesWith(V: PoisonValue::get(T: CB.getType()));
3409 else
3410 CB.replaceAllUsesWith(V: R->getReturnValue());
3411 }
3412 // Since we are now done with the Call/Invoke, we can delete it.
3413 CB.eraseFromParent();
3414
3415 // Since we are now done with the return instruction, delete it also.
3416 Returns[0]->eraseFromParent();
3417
3418 if (MergeAttributes)
3419 AttributeFuncs::mergeAttributesForInlining(Caller&: *Caller, Callee: *CalledFunc);
3420
3421 // We are now done with the inlining.
3422 return;
3423 }
3424
3425 // Otherwise, we have the normal case, of more than one block to inline or
3426 // multiple return sites.
3427
3428 // We want to clone the entire callee function into the hole between the
3429 // "starter" and "ender" blocks. How we accomplish this depends on whether
3430 // this is an invoke instruction or a call instruction.
3431 BasicBlock *AfterCallBB;
3432 UncondBrInst *CreatedBranchToNormalDest = nullptr;
3433 if (InvokeInst *II = dyn_cast<InvokeInst>(Val: &CB)) {
3434
3435 // Add an unconditional branch to make this look like the CallInst case...
3436 CreatedBranchToNormalDest =
3437 UncondBrInst::Create(Target: II->getNormalDest(), InsertBefore: CB.getIterator());
3438 // We intend to replace this DebugLoc with another later.
3439 CreatedBranchToNormalDest->setDebugLoc(DebugLoc::getTemporary());
3440
3441 // Split the basic block. This guarantees that no PHI nodes will have to be
3442 // updated due to new incoming edges, and make the invoke case more
3443 // symmetric to the call case.
3444 AfterCallBB =
3445 OrigBB->splitBasicBlock(I: CreatedBranchToNormalDest->getIterator(),
3446 BBName: CalledFunc->getName() + ".exit");
3447
3448 } else { // It's a call
3449 // If this is a call instruction, we need to split the basic block that
3450 // the call lives in.
3451 //
3452 AfterCallBB = OrigBB->splitBasicBlock(I: CB.getIterator(),
3453 BBName: CalledFunc->getName() + ".exit");
3454 }
3455
3456 if (IFI.CallerBFI) {
3457 // Copy original BB's block frequency to AfterCallBB
3458 IFI.CallerBFI->setBlockFreq(BB: AfterCallBB,
3459 Freq: IFI.CallerBFI->getBlockFreq(BB: OrigBB));
3460 }
3461
3462 // Change the branch that used to go to AfterCallBB to branch to the first
3463 // basic block of the inlined function.
3464 //
3465 UncondBrInst *Br = cast<UncondBrInst>(Val: OrigBB->getTerminator());
3466 Br->setSuccessor(&*FirstNewBlock);
3467
3468 // Now that the function is correct, make it a little bit nicer. In
3469 // particular, move the basic blocks inserted from the end of the function
3470 // into the space made by splitting the source basic block.
3471 Caller->splice(ToIt: AfterCallBB->getIterator(), FromF: Caller, FromBeginIt: FirstNewBlock,
3472 FromEndIt: Caller->end());
3473
3474 // Handle all of the return instructions that we just cloned in, and eliminate
3475 // any users of the original call/invoke instruction.
3476 Type *RTy = CalledFunc->getReturnType();
3477
3478 PHINode *PHI = nullptr;
3479 if (Returns.size() > 1) {
3480 // The PHI node should go at the front of the new basic block to merge all
3481 // possible incoming values.
3482 if (!CB.use_empty()) {
3483 PHI = PHINode::Create(Ty: RTy, NumReservedValues: Returns.size(), NameStr: CB.getName());
3484 PHI->insertBefore(InsertPos: AfterCallBB->begin());
3485 // Anything that used the result of the function call should now use the
3486 // PHI node as their operand.
3487 CB.replaceAllUsesWith(V: PHI);
3488 }
3489
3490 // Loop over all of the return instructions adding entries to the PHI node
3491 // as appropriate.
3492 if (PHI) {
3493 for (ReturnInst *RI : Returns) {
3494 assert(RI->getReturnValue()->getType() == PHI->getType() &&
3495 "Ret value not consistent in function!");
3496 PHI->addIncoming(V: RI->getReturnValue(), BB: RI->getParent());
3497 }
3498 }
3499
3500 // Add a branch to the merge points and remove return instructions.
3501 DebugLoc Loc;
3502 for (ReturnInst *RI : Returns) {
3503 UncondBrInst *BI = UncondBrInst::Create(Target: AfterCallBB, InsertBefore: RI->getIterator());
3504 Loc = RI->getDebugLoc();
3505 BI->setDebugLoc(Loc);
3506 RI->eraseFromParent();
3507 }
3508 // We need to set the debug location to *somewhere* inside the
3509 // inlined function. The line number may be nonsensical, but the
3510 // instruction will at least be associated with the right
3511 // function.
3512 if (CreatedBranchToNormalDest)
3513 CreatedBranchToNormalDest->setDebugLoc(Loc);
3514 } else if (!Returns.empty()) {
3515 // Otherwise, if there is exactly one return value, just replace anything
3516 // using the return value of the call with the computed value.
3517 if (!CB.use_empty()) {
3518 if (&CB == Returns[0]->getReturnValue())
3519 CB.replaceAllUsesWith(V: PoisonValue::get(T: CB.getType()));
3520 else
3521 CB.replaceAllUsesWith(V: Returns[0]->getReturnValue());
3522 }
3523
3524 // Update PHI nodes that use the ReturnBB to use the AfterCallBB.
3525 BasicBlock *ReturnBB = Returns[0]->getParent();
3526 ReturnBB->replaceAllUsesWith(V: AfterCallBB);
3527
3528 // Splice the code from the return block into the block that it will return
3529 // to, which contains the code that was after the call.
3530 AfterCallBB->splice(ToIt: AfterCallBB->begin(), FromBB: ReturnBB);
3531
3532 if (CreatedBranchToNormalDest)
3533 CreatedBranchToNormalDest->setDebugLoc(Returns[0]->getDebugLoc());
3534
3535 // Delete the return instruction now and empty ReturnBB now.
3536 Returns[0]->eraseFromParent();
3537 ReturnBB->eraseFromParent();
3538 } else if (!CB.use_empty()) {
3539 // In this case there are no returns to use, so there is no clear source
3540 // location for the "return".
3541 // FIXME: It may be correct to use the scope end line of the function here,
3542 // since this likely means we are falling out of the function.
3543 if (CreatedBranchToNormalDest)
3544 CreatedBranchToNormalDest->setDebugLoc(DebugLoc::getUnknown());
3545 // No returns, but something is using the return value of the call. Just
3546 // nuke the result.
3547 CB.replaceAllUsesWith(V: PoisonValue::get(T: CB.getType()));
3548 }
3549
3550 // Since we are now done with the Call/Invoke, we can delete it.
3551 CB.eraseFromParent();
3552
3553 // If we inlined any musttail calls and the original return is now
3554 // unreachable, delete it. It can only contain a ret.
3555 if (InlinedMustTailCalls && pred_empty(BB: AfterCallBB))
3556 AfterCallBB->eraseFromParent();
3557
3558 // We should always be able to fold the entry block of the function into the
3559 // single predecessor of the block...
3560 BasicBlock *CalleeEntry = Br->getSuccessor();
3561
3562 // Splice the code entry block into calling block, right before the
3563 // unconditional branch.
3564 CalleeEntry->replaceAllUsesWith(V: OrigBB); // Update PHI nodes
3565 OrigBB->splice(ToIt: Br->getIterator(), FromBB: CalleeEntry);
3566
3567 // Remove the unconditional branch.
3568 Br->eraseFromParent();
3569
3570 // Now we can remove the CalleeEntry block, which is now empty.
3571 CalleeEntry->eraseFromParent();
3572
3573 // If we inserted a phi node, check to see if it has a single value (e.g. all
3574 // the entries are the same or undef). If so, remove the PHI so it doesn't
3575 // block other optimizations.
3576 if (PHI) {
3577 AssumptionCache *AC =
3578 IFI.GetAssumptionCache ? &IFI.GetAssumptionCache(*Caller) : nullptr;
3579 auto &DL = Caller->getDataLayout();
3580 if (Value *V = simplifyInstruction(I: PHI, Q: {DL, nullptr, nullptr, AC})) {
3581 PHI->replaceAllUsesWith(V);
3582 PHI->eraseFromParent();
3583 }
3584 }
3585
3586 if (MergeAttributes)
3587 AttributeFuncs::mergeAttributesForInlining(Caller&: *Caller, Callee: *CalledFunc);
3588}
3589
3590llvm::InlineResult llvm::InlineFunction(
3591 CallBase &CB, InlineFunctionInfo &IFI, bool MergeAttributes,
3592 AAResults *CalleeAAR, bool InsertLifetime, bool TrackInlineHistory,
3593 Function *ForwardVarArgsTo, OptimizationRemarkEmitter *ORE) {
3594 llvm::InlineResult Result = CanInlineCallSite(CB, IFI);
3595 if (Result.isSuccess()) {
3596 InlineFunctionImpl(CB, IFI, MergeAttributes, CalleeAAR, InsertLifetime,
3597 TrackInlineHistory, ForwardVarArgsTo, ORE);
3598 }
3599
3600 return Result;
3601}
3602