1//===- Construction of pass pipelines -------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9///
10/// This file provides the implementation of the PassBuilder based on our
11/// static pass registry as well as related functionality. It also provides
12/// helpers to aid in analyzing, debugging, and testing passes and pass
13/// pipelines.
14///
15//===----------------------------------------------------------------------===//
16
17#include "PassesOptions.h"
18#include "llvm/ADT/Statistic.h"
19#include "llvm/Analysis/AliasAnalysis.h"
20#include "llvm/Analysis/BasicAliasAnalysis.h"
21#include "llvm/Analysis/CGSCCPassManager.h"
22#include "llvm/Analysis/FunctionPropertiesAnalysis.h"
23#include "llvm/Analysis/GlobalsModRef.h"
24#include "llvm/Analysis/InlineAdvisor.h"
25#include "llvm/Analysis/InstCount.h"
26#include "llvm/Analysis/ProfileSummaryInfo.h"
27#include "llvm/Analysis/ScopedNoAliasAA.h"
28#include "llvm/Analysis/TypeBasedAliasAnalysis.h"
29#include "llvm/IR/PassManager.h"
30#include "llvm/IR/Verifier.h"
31#include "llvm/Pass.h"
32#include "llvm/Passes/OptimizationLevel.h"
33#include "llvm/Passes/PassBuilder.h"
34#include "llvm/Passes/TriggerCrashPasses.h"
35#include "llvm/Support/CommandLine.h"
36#include "llvm/Support/ErrorHandling.h"
37#include "llvm/Support/PGOOptions.h"
38#include "llvm/Target/TargetMachine.h"
39#include "llvm/Transforms/AggressiveInstCombine/AggressiveInstCombine.h"
40#include "llvm/Transforms/Coroutines/CoroAnnotationElide.h"
41#include "llvm/Transforms/Coroutines/CoroCleanup.h"
42#include "llvm/Transforms/Coroutines/CoroConditionalWrapper.h"
43#include "llvm/Transforms/Coroutines/CoroEarly.h"
44#include "llvm/Transforms/Coroutines/CoroElide.h"
45#include "llvm/Transforms/Coroutines/CoroSplit.h"
46#include "llvm/Transforms/HipStdPar/HipStdPar.h"
47#include "llvm/Transforms/IPO/AlwaysInliner.h"
48#include "llvm/Transforms/IPO/Annotation2Metadata.h"
49#include "llvm/Transforms/IPO/ArgumentPromotion.h"
50#include "llvm/Transforms/IPO/Attributor.h"
51#include "llvm/Transforms/IPO/CalledValuePropagation.h"
52#include "llvm/Transforms/IPO/ConstantMerge.h"
53#include "llvm/Transforms/IPO/CrossDSOCFI.h"
54#include "llvm/Transforms/IPO/DeadArgumentElimination.h"
55#include "llvm/Transforms/IPO/ElimAvailExtern.h"
56#include "llvm/Transforms/IPO/EmbedBitcodePass.h"
57#include "llvm/Transforms/IPO/ExpandVariadics.h"
58#include "llvm/Transforms/IPO/FatLTOCleanup.h"
59#include "llvm/Transforms/IPO/ForceFunctionAttrs.h"
60#include "llvm/Transforms/IPO/FunctionAttrs.h"
61#include "llvm/Transforms/IPO/GlobalDCE.h"
62#include "llvm/Transforms/IPO/GlobalOpt.h"
63#include "llvm/Transforms/IPO/GlobalSplit.h"
64#include "llvm/Transforms/IPO/HotColdSplitting.h"
65#include "llvm/Transforms/IPO/InferFunctionAttrs.h"
66#include "llvm/Transforms/IPO/Inliner.h"
67#include "llvm/Transforms/IPO/Instrumentor.h"
68#include "llvm/Transforms/IPO/LowerTypeTests.h"
69#include "llvm/Transforms/IPO/MemProfContextDisambiguation.h"
70#include "llvm/Transforms/IPO/MergeFunctions.h"
71#include "llvm/Transforms/IPO/ModuleInliner.h"
72#include "llvm/Transforms/IPO/OpenMPOpt.h"
73#include "llvm/Transforms/IPO/PartialInlining.h"
74#include "llvm/Transforms/IPO/SCCP.h"
75#include "llvm/Transforms/IPO/SampleProfile.h"
76#include "llvm/Transforms/IPO/SampleProfileProbe.h"
77#include "llvm/Transforms/IPO/WholeProgramDevirt.h"
78#include "llvm/Transforms/InstCombine/InstCombine.h"
79#include "llvm/Transforms/Instrumentation/AllocToken.h"
80#include "llvm/Transforms/Instrumentation/CGProfile.h"
81#include "llvm/Transforms/Instrumentation/ControlHeightReduction.h"
82#include "llvm/Transforms/Instrumentation/InstrProfiling.h"
83#include "llvm/Transforms/Instrumentation/MemProfInstrumentation.h"
84#include "llvm/Transforms/Instrumentation/MemProfUse.h"
85#include "llvm/Transforms/Instrumentation/PGOCtxProfFlattening.h"
86#include "llvm/Transforms/Instrumentation/PGOCtxProfLowering.h"
87#include "llvm/Transforms/Instrumentation/PGOForceFunctionAttrs.h"
88#include "llvm/Transforms/Instrumentation/PGOInstrumentation.h"
89#include "llvm/Transforms/Scalar/ADCE.h"
90#include "llvm/Transforms/Scalar/AlignmentFromAssumptions.h"
91#include "llvm/Transforms/Scalar/AnnotationRemarks.h"
92#include "llvm/Transforms/Scalar/BDCE.h"
93#include "llvm/Transforms/Scalar/CallSiteSplitting.h"
94#include "llvm/Transforms/Scalar/ConstraintElimination.h"
95#include "llvm/Transforms/Scalar/CorrelatedValuePropagation.h"
96#include "llvm/Transforms/Scalar/DFAJumpThreading.h"
97#include "llvm/Transforms/Scalar/DeadStoreElimination.h"
98#include "llvm/Transforms/Scalar/DivRemPairs.h"
99#include "llvm/Transforms/Scalar/DropUnnecessaryAssumes.h"
100#include "llvm/Transforms/Scalar/EarlyCSE.h"
101#include "llvm/Transforms/Scalar/ExpandMemCmp.h"
102#include "llvm/Transforms/Scalar/Float2Int.h"
103#include "llvm/Transforms/Scalar/GVN.h"
104#include "llvm/Transforms/Scalar/GVNHoist.h"
105#include "llvm/Transforms/Scalar/GVNSink.h"
106#include "llvm/Transforms/Scalar/IndVarSimplify.h"
107#include "llvm/Transforms/Scalar/InferAlignment.h"
108#include "llvm/Transforms/Scalar/InstSimplifyPass.h"
109#include "llvm/Transforms/Scalar/JumpTableToSwitch.h"
110#include "llvm/Transforms/Scalar/JumpThreading.h"
111#include "llvm/Transforms/Scalar/LICM.h"
112#include "llvm/Transforms/Scalar/LoopDeletion.h"
113#include "llvm/Transforms/Scalar/LoopDistribute.h"
114#include "llvm/Transforms/Scalar/LoopFlatten.h"
115#include "llvm/Transforms/Scalar/LoopFuse.h"
116#include "llvm/Transforms/Scalar/LoopIdiomRecognize.h"
117#include "llvm/Transforms/Scalar/LoopInstSimplify.h"
118#include "llvm/Transforms/Scalar/LoopInterchange.h"
119#include "llvm/Transforms/Scalar/LoopLoadElimination.h"
120#include "llvm/Transforms/Scalar/LoopPassManager.h"
121#include "llvm/Transforms/Scalar/LoopRotation.h"
122#include "llvm/Transforms/Scalar/LoopSimplifyCFG.h"
123#include "llvm/Transforms/Scalar/LoopSink.h"
124#include "llvm/Transforms/Scalar/LoopUnrollAndJamPass.h"
125#include "llvm/Transforms/Scalar/LoopUnrollPass.h"
126#include "llvm/Transforms/Scalar/LoopVersioningLICM.h"
127#include "llvm/Transforms/Scalar/LowerConstantIntrinsics.h"
128#include "llvm/Transforms/Scalar/LowerExpectIntrinsic.h"
129#include "llvm/Transforms/Scalar/LowerMatrixIntrinsics.h"
130#include "llvm/Transforms/Scalar/MemCpyOptimizer.h"
131#include "llvm/Transforms/Scalar/MergeICmps.h"
132#include "llvm/Transforms/Scalar/MergedLoadStoreMotion.h"
133#include "llvm/Transforms/Scalar/NewGVN.h"
134#include "llvm/Transforms/Scalar/Reassociate.h"
135#include "llvm/Transforms/Scalar/SCCP.h"
136#include "llvm/Transforms/Scalar/SROA.h"
137#include "llvm/Transforms/Scalar/SimpleLoopUnswitch.h"
138#include "llvm/Transforms/Scalar/SimplifyCFG.h"
139#include "llvm/Transforms/Scalar/SpeculativeExecution.h"
140#include "llvm/Transforms/Scalar/TailRecursionElimination.h"
141#include "llvm/Transforms/Scalar/WarnMissedTransforms.h"
142#include "llvm/Transforms/Utils/AddDiscriminators.h"
143#include "llvm/Transforms/Utils/AssignGUID.h"
144#include "llvm/Transforms/Utils/AssumeBundleBuilder.h"
145#include "llvm/Transforms/Utils/CanonicalizeAliases.h"
146#include "llvm/Transforms/Utils/CountVisits.h"
147#include "llvm/Transforms/Utils/EntryExitInstrumenter.h"
148#include "llvm/Transforms/Utils/ExtraPassManager.h"
149#include "llvm/Transforms/Utils/InjectTLIMappings.h"
150#include "llvm/Transforms/Utils/LibCallsShrinkWrap.h"
151#include "llvm/Transforms/Utils/LowerCommentStringPass.h"
152#include "llvm/Transforms/Utils/Mem2Reg.h"
153#include "llvm/Transforms/Utils/MoveAutoInit.h"
154#include "llvm/Transforms/Utils/NameAnonGlobals.h"
155#include "llvm/Transforms/Utils/RelLookupTableConverter.h"
156#include "llvm/Transforms/Utils/SimplifyCFGOptions.h"
157#include "llvm/Transforms/Vectorize/LoopVectorize.h"
158#include "llvm/Transforms/Vectorize/SLPVectorizer.h"
159#include "llvm/Transforms/Vectorize/VectorCombine.h"
160
161using namespace llvm;
162
163namespace llvm {
164extern cl::opt<std::string> UseCtxProfile;
165
166extern cl::opt<bool> EnableMemProfContextDisambiguation;
167} // namespace llvm
168
169PipelineTuningOptions::PipelineTuningOptions() {
170 const PassesOptions &Opts = PassesOptions::Global;
171 LoopInterleaving = true;
172 LoopVectorization = true;
173 SLPVectorization = false;
174 LoopUnrolling = true;
175 LoopInterchange = Opts.enable_loopinterchange;
176 LoopFusion = false;
177 ForgetAllSCEVInLoopUnroll = getForgetSCEVInLoopUnroll();
178 LicmMssaOptCap = getLicmMssaOptCap();
179 LicmMssaNoAccForPromotionCap = getLicmMssaNoAccForPromotionCap();
180 CallGraphProfile = true;
181 UnifiedLTO = false;
182 MergeFunctions = Opts.enable_merge_functions;
183 InlinerThreshold = -1;
184 EagerlyInvalidateAnalyses = Opts.eagerly_invalidate_analyses;
185 DevirtualizeSpeculatively = Opts.enable_devirtualize_speculatively;
186}
187
188namespace llvm {
189extern cl::opt<unsigned> MaxDevirtIterations;
190} // namespace llvm
191
192void PassBuilder::invokePeepholeEPCallbacks(FunctionPassManager &FPM,
193 OptimizationLevel Level) {
194 for (auto &C : PeepholeEPCallbacks)
195 C(FPM, Level);
196}
197void PassBuilder::invokeLateLoopOptimizationsEPCallbacks(
198 LoopPassManager &LPM, OptimizationLevel Level) {
199 for (auto &C : LateLoopOptimizationsEPCallbacks)
200 C(LPM, Level);
201}
202void PassBuilder::invokeLoopOptimizerEndEPCallbacks(LoopPassManager &LPM,
203 OptimizationLevel Level) {
204 for (auto &C : LoopOptimizerEndEPCallbacks)
205 C(LPM, Level);
206}
207void PassBuilder::invokeScalarOptimizerLateEPCallbacks(
208 FunctionPassManager &FPM, OptimizationLevel Level) {
209 for (auto &C : ScalarOptimizerLateEPCallbacks)
210 C(FPM, Level);
211}
212void PassBuilder::invokeCGSCCOptimizerLateEPCallbacks(CGSCCPassManager &CGPM,
213 OptimizationLevel Level) {
214 for (auto &C : CGSCCOptimizerLateEPCallbacks)
215 C(CGPM, Level);
216}
217void PassBuilder::invokeVectorizerStartEPCallbacks(FunctionPassManager &FPM,
218 OptimizationLevel Level) {
219 for (auto &C : VectorizerStartEPCallbacks)
220 C(FPM, Level);
221}
222void PassBuilder::invokeVectorizerEndEPCallbacks(FunctionPassManager &FPM,
223 OptimizationLevel Level) {
224 for (auto &C : VectorizerEndEPCallbacks)
225 C(FPM, Level);
226}
227void PassBuilder::invokeOptimizerEarlyEPCallbacks(ModulePassManager &MPM,
228 OptimizationLevel Level,
229 ThinOrFullLTOPhase Phase) {
230 for (auto &C : OptimizerEarlyEPCallbacks)
231 C(MPM, Level, Phase);
232}
233void PassBuilder::invokeOptimizerLastEPCallbacks(ModulePassManager &MPM,
234 OptimizationLevel Level,
235 ThinOrFullLTOPhase Phase) {
236 for (auto &C : OptimizerLastEPCallbacks)
237 C(MPM, Level, Phase);
238}
239void PassBuilder::invokeFullLinkTimeOptimizationEarlyEPCallbacks(
240 ModulePassManager &MPM, OptimizationLevel Level) {
241 for (auto &C : FullLinkTimeOptimizationEarlyEPCallbacks)
242 C(MPM, Level);
243}
244void PassBuilder::invokeFullLinkTimeOptimizationLastEPCallbacks(
245 ModulePassManager &MPM, OptimizationLevel Level) {
246 for (auto &C : FullLinkTimeOptimizationLastEPCallbacks)
247 C(MPM, Level);
248}
249void PassBuilder::invokeThinLinkTimeOptimizationEarlyEPCallbacks(
250 ModulePassManager &MPM, OptimizationLevel Level) {
251 for (auto &C : ThinLinkTimeOptimizationEarlyEPCallbacks)
252 C(MPM, Level);
253}
254void PassBuilder::invokeThinLinkTimeOptimizationLastEPCallbacks(
255 ModulePassManager &MPM, OptimizationLevel Level) {
256 for (auto &C : ThinLinkTimeOptimizationLastEPCallbacks)
257 C(MPM, Level);
258}
259void PassBuilder::invokePipelineStartEPCallbacks(ModulePassManager &MPM,
260 OptimizationLevel Level) {
261 for (auto &C : PipelineStartEPCallbacks)
262 C(MPM, Level);
263}
264void PassBuilder::invokePipelineEarlySimplificationEPCallbacks(
265 ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase) {
266 for (auto &C : PipelineEarlySimplificationEPCallbacks)
267 C(MPM, Level, Phase);
268}
269
270// Get IR stats with InstCount before/after the optimization pipeline
271static void instructionCountersPass(ModulePassManager &MPM,
272 bool IsPreOptimization) {
273 if (AreStatisticsEnabled()) {
274 MPM.addPass(
275 Pass: createModuleToFunctionPassAdaptor(Pass: InstCountPass(IsPreOptimization)));
276 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
277 Pass: FunctionPropertiesStatisticsPass(IsPreOptimization)));
278 }
279}
280
281// Helper to add AnnotationRemarksPass.
282static void addAnnotationRemarksPass(ModulePassManager &MPM) {
283 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: AnnotationRemarksPass()));
284}
285
286// Helper to check if the current compilation phase is preparing for LTO
287static bool isLTOPreLink(ThinOrFullLTOPhase Phase) {
288 return Phase == ThinOrFullLTOPhase::ThinLTOPreLink ||
289 Phase == ThinOrFullLTOPhase::FullLTOPreLink;
290}
291
292// Helper to check if the current compilation phase is preparing for FullLTO
293[[maybe_unused]] static bool isFullLTOPreLink(ThinOrFullLTOPhase Phase) {
294 return Phase == ThinOrFullLTOPhase::FullLTOPreLink;
295}
296
297// Helper to check if the current compilation phase is preparing for ThinLTO
298static bool isThinLTOPreLink(ThinOrFullLTOPhase Phase) {
299 return Phase == ThinOrFullLTOPhase::ThinLTOPreLink;
300}
301
302// Helper to check if the current compilation phase is LTO backend
303static bool isLTOPostLink(ThinOrFullLTOPhase Phase) {
304 return Phase == ThinOrFullLTOPhase::ThinLTOPostLink ||
305 Phase == ThinOrFullLTOPhase::FullLTOPostLink;
306}
307
308// Helper to check if the current compilation phase is FullLTO backend
309static bool isFullLTOPostLink(ThinOrFullLTOPhase Phase) {
310 return Phase == ThinOrFullLTOPhase::FullLTOPostLink;
311}
312
313// Helper to check if the current compilation phase is ThinLTO backend
314static bool isThinLTOPostLink(ThinOrFullLTOPhase Phase) {
315 return Phase == ThinOrFullLTOPhase::ThinLTOPostLink;
316}
317
318// Helper to wrap conditionally Coro passes.
319static CoroConditionalWrapper buildCoroWrapper(ThinOrFullLTOPhase Phase) {
320 // TODO: Skip passes according to Phase.
321 ModulePassManager CoroPM;
322 CoroPM.addPass(Pass: CoroEarlyPass());
323 CGSCCPassManager CGPM;
324 CGPM.addPass(Pass: CoroSplitPass());
325 CoroPM.addPass(Pass: createModuleToPostOrderCGSCCPassAdaptor(Pass: std::move(CGPM)));
326 CoroPM.addPass(Pass: CoroCleanupPass());
327 CoroPM.addPass(Pass: GlobalDCEPass());
328 return CoroConditionalWrapper(std::move(CoroPM));
329}
330
331static InlineParams getInlineParamsFromOptLevel(OptimizationLevel Level) {
332 return getInlineParamsFromOptLevel(OptLevel: static_cast<unsigned>(Level));
333}
334
335static void addModuleInlinerPass(ModulePassManager &MPM,
336 const PassesOptions &Opts,
337 OptimizationLevel Level,
338 ThinOrFullLTOPhase Phase) {
339 InlineParams IP = ::getInlineParamsFromOptLevel(Level);
340 if (Opts.enable_module_inliner)
341 MPM.addPass(Pass: ModuleInlinerPass(IP, Opts.enable_ml_inliner, Phase));
342 else
343 MPM.addPass(Pass: ModuleInlinerWrapperPass(
344 IP,
345 /* MandatoryFirst */ true,
346 InlineContext{.LTOPhase: Phase, .Pass: InlinePass::CGSCCInliner}));
347}
348
349// TODO: Investigate the cost/benefit of tail call elimination on debugging.
350FunctionPassManager
351PassBuilder::buildO1FunctionSimplificationPipeline(OptimizationLevel Level,
352 ThinOrFullLTOPhase Phase) {
353
354 FunctionPassManager FPM;
355
356 if (AreStatisticsEnabled())
357 FPM.addPass(Pass: CountVisitsPass());
358
359 // Form SSA out of local memory accesses after breaking apart aggregates into
360 // scalars.
361 FPM.addPass(Pass: SROAPass(SROAOptions::ModifyCFG));
362
363 // Catch trivial redundancies
364 FPM.addPass(Pass: EarlyCSEPass(true /* Enable mem-ssa. */));
365
366 // Hoisting of scalars and load expressions.
367 FPM.addPass(
368 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
369 FPM.addPass(Pass: InstCombinePass());
370
371 FPM.addPass(Pass: LibCallsShrinkWrapPass());
372
373 invokePeepholeEPCallbacks(FPM, Level);
374
375 FPM.addPass(
376 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
377
378 // Form canonically associated expression trees, and simplify the trees using
379 // basic mathematical properties. For example, this will form (nearly)
380 // minimal multiplication trees.
381 FPM.addPass(Pass: ReassociatePass());
382
383 // Add the primary loop simplification pipeline.
384 // FIXME: Currently this is split into two loop pass pipelines because we run
385 // some function passes in between them. These can and should be removed
386 // and/or replaced by scheduling the loop pass equivalents in the correct
387 // positions. But those equivalent passes aren't powerful enough yet.
388 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
389 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
390 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
391 // `LoopInstSimplify`.
392 LoopPassManager LPM1, LPM2;
393
394 // Simplify the loop body. We do this initially to clean up after other loop
395 // passes run, either when iterating on a loop or on inner loops with
396 // implications on the outer loop.
397 LPM1.addPass(Pass: LoopInstSimplifyPass());
398 LPM1.addPass(Pass: LoopSimplifyCFGPass());
399
400 // Try to remove as much code from the loop header as possible,
401 // to reduce amount of IR that will have to be duplicated. However,
402 // do not perform speculative hoisting the first time as LICM
403 // will destroy metadata that may not need to be destroyed if run
404 // after loop rotation.
405 // TODO: Investigate promotion cap for O1.
406 LPM1.addPass(Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
407 /*AllowSpeculation=*/false));
408
409 LPM1.addPass(
410 Pass: LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
411 // TODO: Investigate promotion cap for O1.
412 LPM1.addPass(Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
413 /*AllowSpeculation=*/true));
414 LPM1.addPass(Pass: SimpleLoopUnswitchPass());
415 if (Opts.enable_loop_flatten)
416 LPM1.addPass(Pass: LoopFlattenPass());
417
418 LPM2.addPass(Pass: LoopIdiomRecognizePass());
419 LPM2.addPass(Pass: IndVarSimplifyPass());
420
421 invokeLateLoopOptimizationsEPCallbacks(LPM&: LPM2, Level);
422
423 LPM2.addPass(Pass: LoopDeletionPass());
424
425 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
426 // because it changes IR to makes profile annotation in back compile
427 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
428 // attributes so we need to make sure and allow the full unroll pass to pay
429 // attention to it.
430 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
431 PGOOpt->Action != PGOOptions::SampleUse)
432 LPM2.addPass(Pass: LoopFullUnrollPass(static_cast<int>(Level),
433 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
434 PTO.ForgetAllSCEVInLoopUnroll));
435
436 invokeLoopOptimizerEndEPCallbacks(LPM&: LPM2, Level);
437
438 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM1),
439 /*UseMemorySSA=*/true));
440 FPM.addPass(
441 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
442 FPM.addPass(Pass: InstCombinePass());
443 // The loop passes in LPM2 (LoopFullUnrollPass) do not preserve MemorySSA.
444 // *All* loop passes must preserve it, in order to be able to use it.
445 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM2),
446 /*UseMemorySSA=*/false));
447
448 // Delete small array after loop unroll.
449 FPM.addPass(Pass: SROAPass(SROAOptions::ModifyCFG));
450
451 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
452 FPM.addPass(Pass: MemCpyOptPass());
453
454 // Sparse conditional constant propagation.
455 // FIXME: It isn't clear why we do this *after* loop passes rather than
456 // before...
457 FPM.addPass(Pass: SCCPPass());
458
459 // Delete dead bit computations (instcombine runs after to fold away the dead
460 // computations, and then ADCE will run later to exploit any new DCE
461 // opportunities that creates).
462 FPM.addPass(Pass: BDCEPass());
463
464 // Run instcombine after redundancy and dead bit elimination to exploit
465 // opportunities opened up by them.
466 FPM.addPass(Pass: InstCombinePass());
467 invokePeepholeEPCallbacks(FPM, Level);
468
469 FPM.addPass(Pass: CoroElidePass());
470
471 invokeScalarOptimizerLateEPCallbacks(FPM, Level);
472
473 // Finally, do an expensive DCE pass to catch all the dead code exposed by
474 // the simplifications and basic cleanup after all the simplifications.
475 // TODO: Investigate if this is too expensive.
476 FPM.addPass(Pass: ADCEPass());
477 FPM.addPass(
478 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
479 FPM.addPass(Pass: InstCombinePass());
480 invokePeepholeEPCallbacks(FPM, Level);
481
482 return FPM;
483}
484
485FunctionPassManager
486PassBuilder::buildFunctionSimplificationPipeline(OptimizationLevel Level,
487 ThinOrFullLTOPhase Phase) {
488 assert(Level != OptimizationLevel::O0 && "Must request optimizations!");
489
490 // The O1 pipeline has a separate pipeline creation function to simplify
491 // construction readability.
492 if (Level == OptimizationLevel::O1)
493 return buildO1FunctionSimplificationPipeline(Level, Phase);
494
495 FunctionPassManager FPM;
496
497 if (AreStatisticsEnabled())
498 FPM.addPass(Pass: CountVisitsPass());
499
500 // Form SSA out of local memory accesses after breaking apart aggregates into
501 // scalars.
502 FPM.addPass(Pass: SROAPass(SROAOptions::ModifyCFG));
503
504 // Catch trivial redundancies
505 FPM.addPass(Pass: EarlyCSEPass(true /* Enable mem-ssa. */));
506 if (EnableKnowledgeRetention)
507 FPM.addPass(Pass: AssumeSimplifyPass());
508
509 // Hoisting of scalars and load expressions.
510 if (Opts.enable_gvn_hoist)
511 FPM.addPass(Pass: GVNHoistPass());
512
513 // Global value numbering based sinking.
514 if (Opts.enable_gvn_sink) {
515 FPM.addPass(Pass: GVNSinkPass());
516 FPM.addPass(
517 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
518 }
519
520 // Speculative execution if the target has divergent branches; otherwise nop.
521 FPM.addPass(Pass: SpeculativeExecutionPass(/* OnlyIfDivergentTarget =*/true));
522
523 // Optimize based on known information about branches, and cleanup afterward.
524 FPM.addPass(Pass: JumpThreadingPass());
525 FPM.addPass(Pass: CorrelatedValuePropagationPass());
526
527 // Jump table to switch conversion.
528 if (Opts.enable_jump_table_to_switch)
529 FPM.addPass(Pass: JumpTableToSwitchPass(/*InLTO=*/isLTOPostLink(Phase)));
530
531 FPM.addPass(
532 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
533 FPM.addPass(Pass: InstCombinePass());
534 FPM.addPass(Pass: AggressiveInstCombinePass());
535 FPM.addPass(Pass: LibCallsShrinkWrapPass());
536
537 invokePeepholeEPCallbacks(FPM, Level);
538
539 // For PGO use pipeline, try to optimize memory intrinsics such as memcpy
540 // using the size value profile. Don't perform this when optimizing for size.
541 if (PGOOpt && PGOOpt->Action == PGOOptions::IRUse)
542 FPM.addPass(Pass: PGOMemOPSizeOpt());
543
544 FPM.addPass(Pass: TailCallElimPass(/*UpdateFunctionEntryCount=*/
545 isInstrumentedPGOUse()));
546 FPM.addPass(
547 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
548
549 // Form canonically associated expression trees, and simplify the trees using
550 // basic mathematical properties. For example, this will form (nearly)
551 // minimal multiplication trees.
552 FPM.addPass(Pass: ReassociatePass());
553
554 if (Opts.enable_constraint_elimination)
555 FPM.addPass(Pass: ConstraintEliminationPass());
556
557 // Add the primary loop simplification pipeline.
558 // FIXME: Currently this is split into two loop pass pipelines because we run
559 // some function passes in between them. These can and should be removed
560 // and/or replaced by scheduling the loop pass equivalents in the correct
561 // positions. But those equivalent passes aren't powerful enough yet.
562 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
563 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
564 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
565 // `LoopInstSimplify`.
566 LoopPassManager LPM1, LPM2;
567
568 // Simplify the loop body. We do this initially to clean up after other loop
569 // passes run, either when iterating on a loop or on inner loops with
570 // implications on the outer loop.
571 LPM1.addPass(Pass: LoopInstSimplifyPass());
572 LPM1.addPass(Pass: LoopSimplifyCFGPass());
573
574 // Try to remove as much code from the loop header as possible,
575 // to reduce amount of IR that will have to be duplicated. However,
576 // do not perform speculative hoisting the first time as LICM
577 // will destroy metadata that may not need to be destroyed if run
578 // after loop rotation.
579 // TODO: Investigate promotion cap for O1.
580 LPM1.addPass(Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
581 /*AllowSpeculation=*/false));
582
583 LPM1.addPass(
584 Pass: LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
585 // TODO: Investigate promotion cap for O1.
586 LPM1.addPass(Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
587 /*AllowSpeculation=*/true));
588 LPM1.addPass(
589 Pass: SimpleLoopUnswitchPass(/* NonTrivial */ Level == OptimizationLevel::O3));
590 if (Opts.enable_loop_flatten)
591 LPM1.addPass(Pass: LoopFlattenPass());
592
593 LPM2.addPass(Pass: LoopIdiomRecognizePass());
594 LPM2.addPass(Pass: IndVarSimplifyPass());
595
596 {
597 ExtraLoopPassManager<ShouldRunExtraSimpleLoopUnswitch> ExtraPasses;
598 ExtraPasses.addPass(Pass: SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
599 OptimizationLevel::O3));
600 LPM2.addPass(Pass: std::move(ExtraPasses));
601 }
602
603 invokeLateLoopOptimizationsEPCallbacks(LPM&: LPM2, Level);
604
605 LPM2.addPass(Pass: LoopDeletionPass());
606
607 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
608 // because it changes IR to makes profile annotation in back compile
609 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
610 // attributes so we need to make sure and allow the full unroll pass to pay
611 // attention to it.
612 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
613 PGOOpt->Action != PGOOptions::SampleUse)
614 LPM2.addPass(Pass: LoopFullUnrollPass(static_cast<int>(Level),
615 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
616 PTO.ForgetAllSCEVInLoopUnroll));
617
618 invokeLoopOptimizerEndEPCallbacks(LPM&: LPM2, Level);
619
620 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM1),
621 /*UseMemorySSA=*/true));
622 FPM.addPass(
623 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
624 FPM.addPass(Pass: InstCombinePass());
625 // The loop passes in LPM2 (LoopIdiomRecognizePass, IndVarSimplifyPass,
626 // LoopDeletionPass and LoopFullUnrollPass) do not preserve MemorySSA.
627 // *All* loop passes must preserve it, in order to be able to use it.
628 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM2),
629 /*UseMemorySSA=*/false));
630
631 // Delete small array after loop unroll.
632 FPM.addPass(Pass: SROAPass(SROAOptions::ModifyCFG));
633
634 // Try vectorization/scalarization transforms that are both improvements
635 // themselves and can allow further folds with GVN and InstCombine.
636 FPM.addPass(Pass: VectorCombinePass(/*TryEarlyFoldsOnly=*/true));
637
638 // Eliminate redundancies.
639 FPM.addPass(Pass: MergedLoadStoreMotionPass());
640 if (Opts.enable_newgvn)
641 FPM.addPass(Pass: NewGVNPass());
642 else
643 FPM.addPass(Pass: GVNPass());
644
645 // Sparse conditional constant propagation.
646 // FIXME: It isn't clear why we do this *after* loop passes rather than
647 // before...
648 FPM.addPass(Pass: SCCPPass());
649
650 // Delete dead bit computations (instcombine runs after to fold away the dead
651 // computations, and then ADCE will run later to exploit any new DCE
652 // opportunities that creates).
653 FPM.addPass(Pass: BDCEPass());
654
655 // Run instcombine after redundancy and dead bit elimination to exploit
656 // opportunities opened up by them.
657 FPM.addPass(Pass: InstCombinePass());
658 invokePeepholeEPCallbacks(FPM, Level);
659
660 // Re-consider control flow based optimizations after redundancy elimination,
661 // redo DCE, etc.
662 if (Opts.enable_dfa_jump_thread)
663 FPM.addPass(Pass: DFAJumpThreadingPass());
664
665 FPM.addPass(Pass: JumpThreadingPass());
666 FPM.addPass(Pass: CorrelatedValuePropagationPass());
667
668 // Finally, do an expensive DCE pass to catch all the dead code exposed by
669 // the simplifications and basic cleanup after all the simplifications.
670 // TODO: Investigate if this is too expensive.
671 FPM.addPass(Pass: ADCEPass());
672
673 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
674 FPM.addPass(Pass: MemCpyOptPass());
675
676 FPM.addPass(Pass: DSEPass());
677 FPM.addPass(Pass: MoveAutoInitPass());
678
679 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(
680 Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
681 /*AllowSpeculation=*/true),
682 /*UseMemorySSA=*/true));
683
684 FPM.addPass(Pass: CoroElidePass());
685
686 invokeScalarOptimizerLateEPCallbacks(FPM, Level);
687
688 FPM.addPass(Pass: SimplifyCFGPass(SimplifyCFGOptions()
689 .convertSwitchRangeToICmp(B: true)
690 .convertSwitchToArithmetic(B: true)
691 .hoistCommonInsts(B: true)
692 .sinkCommonInsts(B: true)));
693 FPM.addPass(Pass: InstCombinePass());
694 invokePeepholeEPCallbacks(FPM, Level);
695
696 return FPM;
697}
698
699void PassBuilder::addRequiredLTOPreLinkPasses(ModulePassManager &MPM) {
700 MPM.addPass(Pass: CanonicalizeAliasesPass());
701 MPM.addPass(Pass: NameAnonGlobalPass());
702 MPM.addPass(Pass: AssignGUIDPass());
703}
704
705void PassBuilder::addPreInlinerPasses(ModulePassManager &MPM,
706 OptimizationLevel Level,
707 ThinOrFullLTOPhase LTOPhase) {
708 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
709 if (Opts.disable_preinline)
710 return;
711 InlineParams IP;
712
713 IP.DefaultThreshold = Opts.preinline_threshold;
714
715 // FIXME: The hint threshold has the same value used by the regular inliner
716 // when not optimzing for size. This should probably be lowered after
717 // performance testing.
718 // FIXME: this comment is cargo culted from the old pass manager, revisit).
719 IP.HintThreshold = 325;
720 IP.OptSizeHintThreshold = Opts.preinline_threshold;
721 ModuleInlinerWrapperPass MIWP(
722 IP, /* MandatoryFirst */ true,
723 InlineContext{.LTOPhase: LTOPhase, .Pass: InlinePass::EarlyInliner});
724 CGSCCPassManager &CGPipeline = MIWP.getPM();
725
726 FunctionPassManager FPM;
727 FPM.addPass(Pass: SROAPass(SROAOptions::ModifyCFG));
728 FPM.addPass(Pass: EarlyCSEPass()); // Catch trivial redundancies.
729 FPM.addPass(Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(
730 B: true))); // Merge & remove basic blocks.
731 FPM.addPass(Pass: InstCombinePass()); // Combine silly sequences.
732 invokePeepholeEPCallbacks(FPM, Level);
733
734 CGPipeline.addPass(Pass: createCGSCCToFunctionPassAdaptor(
735 Pass: std::move(FPM), EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
736
737 MPM.addPass(Pass: std::move(MIWP));
738
739 // Delete anything that is now dead to make sure that we don't instrument
740 // dead code. Instrumentation can end up keeping dead code around and
741 // dramatically increase code size.
742 MPM.addPass(Pass: GlobalDCEPass());
743}
744
745void PassBuilder::addPostPGOLoopRotation(ModulePassManager &MPM,
746 OptimizationLevel Level) {
747 if (Opts.enable_post_pgo_loop_rotation) {
748 // Disable header duplication in loop rotation at -Oz.
749 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
750 Pass: createFunctionToLoopPassAdaptor(Pass: LoopRotatePass(),
751 /*UseMemorySSA=*/false),
752 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
753 }
754}
755
756void PassBuilder::addPGOInstrPasses(ModulePassManager &MPM,
757 OptimizationLevel Level, bool RunProfileGen,
758 bool IsCS, bool AtomicCounterUpdate,
759 std::string ProfileFile,
760 std::string ProfileRemappingFile) {
761 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
762
763 if (!RunProfileGen) {
764 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
765 MPM.addPass(
766 Pass: PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
767 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
768 // RequireAnalysisPass for PSI before subsequent non-module passes.
769 MPM.addPass(Pass: RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
770 return;
771 }
772
773 // Perform PGO instrumentation.
774 MPM.addPass(Pass: PGOInstrumentationGen(IsCS ? PGOInstrumentationType::CSFDO
775 : PGOInstrumentationType::FDO));
776
777 addPostPGOLoopRotation(MPM, Level);
778 // Add the profile lowering pass.
779 InstrProfOptions Options;
780 if (!ProfileFile.empty())
781 Options.InstrProfileOutput = ProfileFile;
782 // Do counter promotion at Level greater than O0.
783 Options.DoCounterPromotion = true;
784 Options.UseBFIInPromotion = IsCS;
785 if (Opts.enable_sampled_instrumentation) {
786 Options.Sampling = true;
787 // With sampling, there is little beneifit to enable counter promotion.
788 // But note that sampling does work with counter promotion.
789 Options.DoCounterPromotion = false;
790 }
791 Options.Atomic = AtomicCounterUpdate;
792 MPM.addPass(Pass: InstrProfilingLoweringPass(Options, IsCS));
793}
794
795void PassBuilder::addPGOInstrPassesForO0(ModulePassManager &MPM,
796 bool RunProfileGen, bool IsCS,
797 bool AtomicCounterUpdate,
798 std::string ProfileFile,
799 std::string ProfileRemappingFile) {
800 if (!RunProfileGen) {
801 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
802 MPM.addPass(
803 Pass: PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
804 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
805 // RequireAnalysisPass for PSI before subsequent non-module passes.
806 MPM.addPass(Pass: RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
807 return;
808 }
809
810 // Perform PGO instrumentation.
811 MPM.addPass(Pass: PGOInstrumentationGen(IsCS ? PGOInstrumentationType::CSFDO
812 : PGOInstrumentationType::FDO));
813 // Add the profile lowering pass.
814 InstrProfOptions Options;
815 if (!ProfileFile.empty())
816 Options.InstrProfileOutput = ProfileFile;
817 // Do not do counter promotion at O0.
818 Options.DoCounterPromotion = false;
819 Options.UseBFIInPromotion = IsCS;
820 Options.Atomic = AtomicCounterUpdate;
821 MPM.addPass(Pass: InstrProfilingLoweringPass(Options, IsCS));
822}
823
824ModuleInlinerWrapperPass
825PassBuilder::buildInlinerPipeline(OptimizationLevel Level,
826 ThinOrFullLTOPhase Phase) {
827 InlineParams IP;
828 if (PTO.InlinerThreshold == -1)
829 IP = ::getInlineParamsFromOptLevel(Level);
830 else
831 IP = getInlineParams(Threshold: PTO.InlinerThreshold);
832 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
833 // set hot-caller threshold to 0 to disable hot
834 // callsite inline (as much as possible [1]) because it makes
835 // profile annotation in the backend inaccurate.
836 //
837 // [1] Note the cost of a function could be below zero due to erased
838 // prologue / epilogue.
839 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
840 IP.HotCallSiteThreshold = 0;
841
842 if (PGOOpt)
843 IP.EnableDeferral = Opts.enable_npm_pgo_inline_deferral;
844
845 ModuleInlinerWrapperPass MIWP(IP, Opts.mandatory_inlining_first,
846 InlineContext{.LTOPhase: Phase, .Pass: InlinePass::CGSCCInliner},
847 Opts.enable_ml_inliner, MaxDevirtIterations);
848
849 // Require the GlobalsAA analysis for the module so we can query it within
850 // the CGSCC pipeline.
851 if (Opts.enable_global_analyses) {
852 MIWP.addModulePass(Pass: RequireAnalysisPass<GlobalsAA, Module>());
853 // Invalidate AAManager so it can be recreated and pick up the newly
854 // available GlobalsAA.
855 MIWP.addModulePass(
856 Pass: createModuleToFunctionPassAdaptor(Pass: InvalidateAnalysisPass<AAManager>()));
857 }
858
859 // Require the ProfileSummaryAnalysis for the module so we can query it within
860 // the inliner pass.
861 MIWP.addModulePass(Pass: RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
862
863 // Now begin the main postorder CGSCC pipeline.
864 // FIXME: The current CGSCC pipeline has its origins in the legacy pass
865 // manager and trying to emulate its precise behavior. Much of this doesn't
866 // make a lot of sense and we should revisit the core CGSCC structure.
867 CGSCCPassManager &MainCGPipeline = MIWP.getPM();
868
869 // Note: historically, the PruneEH pass was run first to deduce nounwind and
870 // generally clean up exception handling overhead. It isn't clear this is
871 // valuable as the inliner doesn't currently care whether it is inlining an
872 // invoke or a call.
873
874 if (Opts.attributor_enable & AttributorRunOption::CGSCC)
875 MainCGPipeline.addPass(Pass: AttributorCGSCCPass());
876 else if (Opts.attributor_enable & AttributorRunOption::CGSCC_LIGHT)
877 MainCGPipeline.addPass(Pass: AttributorLightCGSCCPass());
878
879 // Deduce function attributes. We do another run of this after the function
880 // simplification pipeline, so this only needs to run when it could affect the
881 // function simplification pipeline, which is only the case with recursive
882 // functions.
883 MainCGPipeline.addPass(Pass: PostOrderFunctionAttrsPass(/*SkipNonRecursive*/ true));
884
885 // When at O3 add argument promotion to the pass pipeline.
886 // FIXME: It isn't at all clear why this should be limited to O3.
887 if (Level == OptimizationLevel::O3)
888 MainCGPipeline.addPass(Pass: ArgumentPromotionPass());
889
890 // Try to perform OpenMP specific optimizations. This is a (quick!) no-op if
891 // there are no OpenMP runtime calls present in the module.
892 if (Level == OptimizationLevel::O2 || Level == OptimizationLevel::O3)
893 MainCGPipeline.addPass(Pass: OpenMPOptCGSCCPass(Phase));
894
895 invokeCGSCCOptimizerLateEPCallbacks(CGPM&: MainCGPipeline, Level);
896
897 // Add the core function simplification pipeline nested inside the
898 // CGSCC walk.
899 MainCGPipeline.addPass(Pass: createCGSCCToFunctionPassAdaptor(
900 Pass: buildFunctionSimplificationPipeline(Level, Phase),
901 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses, /*NoRerun=*/true));
902
903 // Finally, deduce any function attributes based on the fully simplified
904 // function.
905 MainCGPipeline.addPass(Pass: PostOrderFunctionAttrsPass());
906
907 // Mark that the function is fully simplified and that it shouldn't be
908 // simplified again if we somehow revisit it due to CGSCC mutations unless
909 // it's been modified since.
910 MainCGPipeline.addPass(Pass: createCGSCCToFunctionPassAdaptor(
911 Pass: RequireAnalysisPass<ShouldNotRunFunctionPassesAnalysis, Function>()));
912
913 if (!isThinLTOPreLink(Phase)) {
914 MainCGPipeline.addPass(Pass: CoroSplitPass(Level != OptimizationLevel::O0));
915 MainCGPipeline.addPass(Pass: CoroAnnotationElidePass());
916 }
917
918 // Make sure we don't affect potential future NoRerun CGSCC adaptors.
919 MIWP.addLateModulePass(Pass: createModuleToFunctionPassAdaptor(
920 Pass: InvalidateAnalysisPass<ShouldNotRunFunctionPassesAnalysis>()));
921
922 return MIWP;
923}
924
925ModulePassManager
926PassBuilder::buildModuleInlinerPipeline(OptimizationLevel Level,
927 ThinOrFullLTOPhase Phase) {
928 ModulePassManager MPM;
929
930 InlineParams IP = ::getInlineParamsFromOptLevel(Level);
931 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
932 // set hot-caller threshold to 0 to disable hot
933 // callsite inline (as much as possible [1]) because it makes
934 // profile annotation in the backend inaccurate.
935 //
936 // [1] Note the cost of a function could be below zero due to erased
937 // prologue / epilogue.
938 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
939 IP.HotCallSiteThreshold = 0;
940
941 if (PGOOpt)
942 IP.EnableDeferral = Opts.enable_npm_pgo_inline_deferral;
943
944 // The inline deferral logic is used to avoid losing some
945 // inlining chance in future. It is helpful in SCC inliner, in which
946 // inlining is processed in bottom-up order.
947 // While in module inliner, the inlining order is a priority-based order
948 // by default. The inline deferral is unnecessary there. So we disable the
949 // inline deferral logic in module inliner.
950 IP.EnableDeferral = false;
951
952 MPM.addPass(Pass: ModuleInlinerPass(IP, Opts.enable_ml_inliner, Phase));
953 if (!UseCtxProfile.empty() && Phase == ThinOrFullLTOPhase::ThinLTOPostLink) {
954 MPM.addPass(Pass: GlobalOptPass());
955 MPM.addPass(Pass: GlobalDCEPass());
956 MPM.addPass(Pass: AssignGUIDPass());
957 MPM.addPass(Pass: PGOCtxProfFlatteningPass(/*IsPreThinlink=*/false));
958 }
959
960 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
961 Pass: buildFunctionSimplificationPipeline(Level, Phase),
962 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
963
964 if (!isThinLTOPreLink(Phase)) {
965 MPM.addPass(Pass: createModuleToPostOrderCGSCCPassAdaptor(
966 Pass: CoroSplitPass(Level != OptimizationLevel::O0)));
967 MPM.addPass(
968 Pass: createModuleToPostOrderCGSCCPassAdaptor(Pass: CoroAnnotationElidePass()));
969 }
970
971 return MPM;
972}
973
974ModulePassManager
975PassBuilder::buildModuleSimplificationPipeline(OptimizationLevel Level,
976 ThinOrFullLTOPhase Phase) {
977 assert(Level != OptimizationLevel::O0 &&
978 "Should not be used for O0 pipeline");
979
980 assert(!isFullLTOPostLink(Phase) &&
981 "FullLTOPostLink shouldn't call buildModuleSimplificationPipeline!");
982
983 ModulePassManager MPM;
984
985 // Place pseudo probe instrumentation as the first pass of the pipeline to
986 // minimize the impact of optimization changes.
987 if (PGOOpt && PGOOpt->PseudoProbeForProfiling && !isThinLTOPostLink(Phase))
988 MPM.addPass(Pass: SampleProfileProbePass(TM));
989
990 bool HasSampleProfile = PGOOpt && (PGOOpt->Action == PGOOptions::SampleUse);
991
992 // In ThinLTO mode, when flattened profile is used, all the available
993 // profile information will be annotated in PreLink phase so there is
994 // no need to load the profile again in PostLink.
995 bool LoadSampleProfile = HasSampleProfile && !(Opts.flattened_profile_used &&
996 isThinLTOPostLink(Phase));
997
998 // During the ThinLTO backend phase we perform early indirect call promotion
999 // here, before globalopt. Otherwise imported available_externally functions
1000 // look unreferenced and are removed. If we are going to load the sample
1001 // profile then defer until later.
1002 // TODO: See if we can move later and consolidate with the location where
1003 // we perform ICP when we are loading a sample profile.
1004 // TODO: We pass HasSampleProfile (whether there was a sample profile file
1005 // passed to the compile) to the SamplePGO flag of ICP. This is used to
1006 // determine whether the new direct calls are annotated with prof metadata.
1007 // Ideally this should be determined from whether the IR is annotated with
1008 // sample profile, and not whether the a sample profile was provided on the
1009 // command line. E.g. for flattened profiles where we will not be reloading
1010 // the sample profile in the ThinLTO backend, we ideally shouldn't have to
1011 // provide the sample profile file.
1012 if (isThinLTOPostLink(Phase) && !LoadSampleProfile)
1013 MPM.addPass(Pass: PGOIndirectCallPromotion(true /* InLTO */, HasSampleProfile));
1014
1015 // Create an early function pass manager to cleanup the output of the
1016 // frontend. Not necessary with LTO post link pipelines since the pre link
1017 // pipeline already cleaned up the frontend output.
1018 if (!isThinLTOPostLink(Phase)) {
1019 // Do basic inference of function attributes from known properties of system
1020 // libraries and other oracles.
1021 MPM.addPass(Pass: InferFunctionAttrsPass());
1022 MPM.addPass(Pass: CoroEarlyPass());
1023
1024 FunctionPassManager EarlyFPM;
1025 EarlyFPM.addPass(Pass: EntryExitInstrumenterPass(/*PostInlining=*/false));
1026 // Lower llvm.expect to metadata before attempting transforms.
1027 // Compare/branch metadata may alter the behavior of passes like
1028 // SimplifyCFG.
1029 EarlyFPM.addPass(Pass: LowerExpectIntrinsicPass());
1030 EarlyFPM.addPass(Pass: SimplifyCFGPass());
1031 EarlyFPM.addPass(Pass: SROAPass(SROAOptions::ModifyCFG));
1032 EarlyFPM.addPass(Pass: EarlyCSEPass());
1033 if (Level == OptimizationLevel::O3)
1034 EarlyFPM.addPass(Pass: CallSiteSplittingPass());
1035 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
1036 Pass: std::move(EarlyFPM), EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
1037 }
1038
1039 if (LoadSampleProfile) {
1040 // Annotate sample profile right after early FPM to ensure freshness of
1041 // the debug info.
1042 MPM.addPass(Pass: SampleProfileLoaderPass(
1043 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile, Phase, FS));
1044 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
1045 // RequireAnalysisPass for PSI before subsequent non-module passes.
1046 MPM.addPass(Pass: RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
1047 // Do not invoke ICP in the LTOPrelink phase as it makes it hard
1048 // for the profile annotation to be accurate in the LTO backend.
1049 if (!isLTOPreLink(Phase))
1050 // We perform early indirect call promotion here, before globalopt.
1051 // This is important for the ThinLTO backend phase because otherwise
1052 // imported available_externally functions look unreferenced and are
1053 // removed.
1054 MPM.addPass(
1055 Pass: PGOIndirectCallPromotion(true /* IsInLTO */, true /* SamplePGO */));
1056 }
1057
1058 // Try to perform OpenMP specific optimizations on the module. This is a
1059 // (quick!) no-op if there are no OpenMP runtime calls present in the module.
1060 MPM.addPass(Pass: OpenMPOptPass(Phase));
1061
1062 if (Opts.attributor_enable & AttributorRunOption::MODULE)
1063 MPM.addPass(Pass: AttributorPass());
1064 else if (Opts.attributor_enable & AttributorRunOption::MODULE_LIGHT)
1065 MPM.addPass(Pass: AttributorLightPass());
1066
1067 // Lower type metadata and the type.test intrinsic in the ThinLTO
1068 // post link pipeline after ICP. This is to enable usage of the type
1069 // tests in ICP sequences.
1070 if (isThinLTOPostLink(Phase))
1071 MPM.addPass(Pass: DropTypeTestsPass());
1072
1073 invokePipelineEarlySimplificationEPCallbacks(MPM, Level, Phase);
1074
1075 // Interprocedural constant propagation now that basic cleanup has occurred
1076 // and prior to optimizing globals.
1077 // FIXME: This position in the pipeline hasn't been carefully considered in
1078 // years, it should be re-analyzed.
1079 MPM.addPass(
1080 Pass: IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/!isLTOPreLink(Phase))));
1081
1082 // Attach metadata to indirect call sites indicating the set of functions
1083 // they may target at run-time. This should follow IPSCCP.
1084 MPM.addPass(Pass: CalledValuePropagationPass());
1085
1086 // Optimize globals to try and fold them into constants.
1087 MPM.addPass(Pass: GlobalOptPass());
1088
1089 // Create a small function pass pipeline to cleanup after all the global
1090 // optimizations.
1091 FunctionPassManager GlobalCleanupPM;
1092 // FIXME: Should this instead by a run of SROA?
1093 GlobalCleanupPM.addPass(Pass: PromotePass());
1094 GlobalCleanupPM.addPass(Pass: InstCombinePass());
1095 invokePeepholeEPCallbacks(FPM&: GlobalCleanupPM, Level);
1096 GlobalCleanupPM.addPass(
1097 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
1098 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(GlobalCleanupPM),
1099 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
1100
1101 // We already asserted this happens in non-FullLTOPostLink earlier.
1102 const bool IsPreLink = !isThinLTOPostLink(Phase);
1103 // Enable contextual profiling instrumentation.
1104 const bool IsCtxProfGen =
1105 IsPreLink && PGOCtxProfLoweringPass::isCtxIRPGOInstrEnabled();
1106 const bool IsPGOPreLink = !IsCtxProfGen && PGOOpt && IsPreLink;
1107 const bool IsPGOInstrGen =
1108 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRInstr;
1109 const bool IsPGOInstrUse =
1110 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRUse;
1111 const bool IsMemprofUse = IsPGOPreLink && !PGOOpt->MemoryProfile.empty();
1112 // We don't want to mix pgo ctx gen and pgo gen; we also don't currently
1113 // enable ctx profiling from the frontend.
1114 assert(!(IsPGOInstrGen && PGOCtxProfLoweringPass::isCtxIRPGOInstrEnabled()) &&
1115 "Enabling both instrumented PGO and contextual instrumentation is not "
1116 "supported.");
1117 const bool IsCtxProfUse = !UseCtxProfile.empty() && isThinLTOPreLink(Phase);
1118
1119 assert((Opts.instrument_cold_function_only_path.empty() ||
1120 isPGOInstrumentColdFunctionOnly()) &&
1121 "--instrument-cold-function-only-path is provided but "
1122 "--pgo-instrument-cold-function-only is not enabled");
1123 const bool IsColdFuncOnlyInstrGen =
1124 isPGOInstrumentColdFunctionOnly() && IsPGOPreLink &&
1125 !Opts.instrument_cold_function_only_path.empty();
1126
1127 if (IsPGOInstrGen || IsPGOInstrUse || IsMemprofUse || IsCtxProfGen ||
1128 IsCtxProfUse || IsColdFuncOnlyInstrGen)
1129 addPreInlinerPasses(MPM, Level, LTOPhase: Phase);
1130
1131 // Add all the requested passes for instrumentation PGO, if requested.
1132 if (IsPGOInstrGen || IsPGOInstrUse) {
1133 addPGOInstrPasses(MPM, Level,
1134 /*RunProfileGen=*/IsPGOInstrGen,
1135 /*IsCS=*/false, AtomicCounterUpdate: PGOOpt->AtomicCounterUpdate,
1136 ProfileFile: PGOOpt->ProfileFile, ProfileRemappingFile: PGOOpt->ProfileRemappingFile);
1137 } else if (IsCtxProfGen || IsCtxProfUse) {
1138 MPM.addPass(Pass: PGOInstrumentationGen(PGOInstrumentationType::CTXPROF));
1139 // In pre-link, we just want the instrumented IR. We use the contextual
1140 // profile in the post-thinlink phase.
1141 // The instrumentation will be removed in post-thinlink after IPO.
1142 if (IsCtxProfUse) {
1143 MPM.addPass(Pass: AssignGUIDPass());
1144 MPM.addPass(Pass: PGOCtxProfFlatteningPass(/*IsPreThinlink=*/true));
1145 return MPM;
1146 }
1147 // Block further inlining in the instrumented ctxprof case. This avoids
1148 // confusingly collecting profiles for the same GUID corresponding to
1149 // different variants of the function. We could do like PGO and identify
1150 // functions by a (GUID, Hash) tuple, but since the ctxprof "use" waits for
1151 // thinlto to happen before performing any further optimizations, it's
1152 // unnecessary to collect profiles for non-prevailing copies.
1153 MPM.addPass(Pass: NoinlineNonPrevailing());
1154 addPostPGOLoopRotation(MPM, Level);
1155 MPM.addPass(Pass: AssignGUIDPass());
1156 MPM.addPass(Pass: PGOCtxProfLoweringPass());
1157 } else if (IsColdFuncOnlyInstrGen) {
1158 addPGOInstrPasses(MPM, Level, /* RunProfileGen */ true, /* IsCS */ false,
1159 /* AtomicCounterUpdate */ false,
1160 ProfileFile: Opts.instrument_cold_function_only_path.str(),
1161 /* ProfileRemappingFile */ "");
1162 }
1163
1164 if (IsPGOInstrGen || IsPGOInstrUse || IsCtxProfGen)
1165 MPM.addPass(Pass: PGOIndirectCallPromotion(false, false));
1166
1167 if (IsPGOPreLink && PGOOpt->CSAction == PGOOptions::CSIRInstr)
1168 MPM.addPass(Pass: PGOInstrumentationGenCreateVar(
1169 PGOOpt->CSProfileGenFile, Opts.enable_sampled_instrumentation));
1170
1171 if (IsMemprofUse)
1172 MPM.addPass(Pass: MemProfUsePass(PGOOpt->MemoryProfile, FS));
1173
1174 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRUse ||
1175 PGOOpt->Action == PGOOptions::SampleUse))
1176 MPM.addPass(Pass: PGOForceFunctionAttrsPass(PGOOpt->ColdOptType));
1177
1178 MPM.addPass(Pass: AlwaysInlinerPass(/*InsertLifetimeIntrinsics=*/true));
1179
1180 if (Opts.enable_module_inliner)
1181 MPM.addPass(Pass: buildModuleInlinerPipeline(Level, Phase));
1182 else
1183 MPM.addPass(Pass: buildInlinerPipeline(Level, Phase));
1184
1185 // Remove any dead arguments exposed by cleanups, constant folding globals,
1186 // and argument promotion.
1187 MPM.addPass(Pass: DeadArgumentEliminationPass());
1188
1189 if (isThinLTOPostLink(Phase))
1190 MPM.addPass(Pass: SimplifyTypeTestsPass());
1191
1192 if (!isThinLTOPreLink(Phase))
1193 MPM.addPass(Pass: CoroCleanupPass());
1194
1195 // Optimize globals now that functions are fully simplified.
1196 MPM.addPass(Pass: GlobalOptPass());
1197 MPM.addPass(Pass: GlobalDCEPass());
1198
1199 return MPM;
1200}
1201
1202/// TODO: Should LTO cause any differences to this set of passes?
1203void PassBuilder::addVectorPasses(OptimizationLevel Level,
1204 FunctionPassManager &FPM,
1205 ThinOrFullLTOPhase LTOPhase) {
1206 FPM.addPass(Pass: LoopVectorizePass(
1207 LoopVectorizeOptions(!PTO.LoopInterleaving, !PTO.LoopVectorization)));
1208
1209 // Drop dereferenceable assumes after vectorization, as they are no longer
1210 // needed and can inhibit further optimization.
1211 if (!isLTOPreLink(Phase: LTOPhase))
1212 FPM.addPass(Pass: DropUnnecessaryAssumesPass(/*DropDereferenceable=*/true));
1213
1214 FPM.addPass(Pass: InferAlignmentPass());
1215 if (isFullLTOPostLink(Phase: LTOPhase)) {
1216 // The vectorizer may have significantly shortened a loop body; unroll
1217 // again. Unroll small loops to hide loop backedge latency and saturate any
1218 // parallel execution resources of an out-of-order processor. We also then
1219 // need to clean up redundancies and loop invariant code.
1220 // FIXME: It would be really good to use a loop-integrated instruction
1221 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1222 // across the loop nests.
1223 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1224 if (Opts.enable_unroll_and_jam && PTO.LoopUnrolling)
1225 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(
1226 Pass: LoopUnrollAndJamPass(static_cast<int>(Level))));
1227 FPM.addPass(Pass: LoopUnrollPass(LoopUnrollOptions(
1228 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1229 PTO.ForgetAllSCEVInLoopUnroll)));
1230 FPM.addPass(Pass: WarnMissedTransformationsPass());
1231 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1232 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1233 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1234 // NOTE: we are very late in the pipeline, and we don't have any LICM
1235 // or SimplifyCFG passes scheduled after us, that would cleanup
1236 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1237
1238 // We also turn on struct to vector canonicalization here, which allows
1239 // converting allocas of homogeneous structs into vector allocas when the
1240 // allocas' users are all memory intrinsics. This allows promotion in some
1241 // cases because structs cannot promote to SSA values, but vectors can. We
1242 // only turn this on after memcpyopt runs because this might hinder
1243 // memcpyopt's optimizations if done before. Look at the documentation for
1244 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1245 FPM.addPass(Pass: SROAPass(SROAOptions(SROAOptions::PreserveCFG,
1246 /*AggregateToVector=*/true)));
1247 }
1248
1249 if (!isFullLTOPostLink(Phase: LTOPhase)) {
1250 // Eliminate loads by forwarding stores from the previous iteration to loads
1251 // of the current iteration.
1252 FPM.addPass(Pass: LoopLoadEliminationPass());
1253 }
1254 // Cleanup after the loop optimization passes.
1255 FPM.addPass(Pass: InstCombinePass());
1256
1257 if (Level > OptimizationLevel::O1 && Opts.extra_vectorizer_passes) {
1258 ExtraFunctionPassManager<ShouldRunExtraVectorPasses> ExtraPasses;
1259 // At higher optimization levels, try to clean up any runtime overlap and
1260 // alignment checks inserted by the vectorizer. We want to track correlated
1261 // runtime checks for two inner loops in the same outer loop, fold any
1262 // common computations, hoist loop-invariant aspects out of any outer loop,
1263 // and unswitch the runtime checks if possible. Once hoisted, we may have
1264 // dead (or speculatable) control flows or more combining opportunities.
1265 ExtraPasses.addPass(Pass: EarlyCSEPass());
1266 ExtraPasses.addPass(Pass: CorrelatedValuePropagationPass());
1267 ExtraPasses.addPass(Pass: InstCombinePass());
1268 LoopPassManager LPM;
1269 LPM.addPass(Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1270 /*AllowSpeculation=*/true));
1271 LPM.addPass(Pass: SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
1272 OptimizationLevel::O3));
1273 ExtraPasses.addPass(
1274 Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM), /*UseMemorySSA=*/true));
1275 ExtraPasses.addPass(
1276 Pass: SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(B: true)));
1277 ExtraPasses.addPass(Pass: InstCombinePass());
1278 FPM.addPass(Pass: std::move(ExtraPasses));
1279 }
1280
1281 // Now that we've formed fast to execute loop structures, we do further
1282 // optimizations. These are run afterward as they might block doing complex
1283 // analyses and transforms such as what are needed for loop vectorization.
1284
1285 // Cleanup after loop vectorization, etc. Simplification passes like CVP and
1286 // GVN, loop transforms, and others have already run, so it's now better to
1287 // convert to more optimized IR using more aggressive simplify CFG options.
1288 // The extra sinking transform can create larger basic blocks, so do this
1289 // before SLP vectorization.
1290 FPM.addPass(Pass: SimplifyCFGPass(SimplifyCFGOptions()
1291 .forwardSwitchCondToPhi(B: true)
1292 .convertSwitchRangeToICmp(B: true)
1293 .convertSwitchToArithmetic(B: true)
1294 .convertSwitchToLookupTable(B: true)
1295 .needCanonicalLoops(B: false)
1296 .hoistCommonInsts(B: true)
1297 .sinkCommonInsts(B: true)));
1298
1299 if (isFullLTOPostLink(Phase: LTOPhase)) {
1300 FPM.addPass(Pass: SCCPPass());
1301 FPM.addPass(Pass: InstCombinePass());
1302 FPM.addPass(Pass: BDCEPass());
1303 }
1304
1305 // Optimize parallel scalar instruction chains into SIMD instructions.
1306 if (PTO.SLPVectorization) {
1307 FPM.addPass(Pass: SLPVectorizerPass());
1308 if (Level >= OptimizationLevel::O2 && Opts.extra_vectorizer_passes) {
1309 FPM.addPass(Pass: EarlyCSEPass());
1310 }
1311 }
1312 // Enhance/cleanup vector code.
1313 FPM.addPass(Pass: VectorCombinePass());
1314
1315 if (!isFullLTOPostLink(Phase: LTOPhase)) {
1316 FPM.addPass(Pass: InstCombinePass());
1317 // Unroll small loops to hide loop backedge latency and saturate any
1318 // parallel execution resources of an out-of-order processor. We also then
1319 // need to clean up redundancies and loop invariant code.
1320 // FIXME: It would be really good to use a loop-integrated instruction
1321 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1322 // across the loop nests.
1323 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1324 if (Opts.enable_unroll_and_jam && PTO.LoopUnrolling) {
1325 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(
1326 Pass: LoopUnrollAndJamPass(static_cast<int>(Level))));
1327 }
1328 FPM.addPass(Pass: LoopUnrollPass(LoopUnrollOptions(
1329 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1330 PTO.ForgetAllSCEVInLoopUnroll)));
1331 FPM.addPass(Pass: WarnMissedTransformationsPass());
1332 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1333 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1334 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1335 // NOTE: we are very late in the pipeline, and we don't have any LICM
1336 // or SimplifyCFG passes scheduled after us, that would cleanup
1337 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1338
1339 // We also turn on struct to vector canonicalization here, which allows
1340 // converting allocas of homogeneous structs into vector allocas when the
1341 // allocas' users are all memory intrinsics. This allows promotion in some
1342 // cases because structs cannot promote to SSA values, but vectors can. We
1343 // only turn this on after memcpyopt runs because this might hinder
1344 // memcpyopt's optimizations if done before. Look at the documentation for
1345 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1346 FPM.addPass(Pass: SROAPass(SROAOptions(SROAOptions::PreserveCFG,
1347 /*AggregateToVector=*/true)));
1348 }
1349
1350 FPM.addPass(Pass: InferAlignmentPass());
1351 FPM.addPass(Pass: InstCombinePass());
1352
1353 // This is needed for two reasons:
1354 // 1. It works around problems that instcombine introduces, such as sinking
1355 // expensive FP divides into loops containing multiplications using the
1356 // divide result.
1357 // 2. It helps to clean up some loop-invariant code created by the loop
1358 // unroll pass when IsFullLTO=false.
1359 FPM.addPass(Pass: createFunctionToLoopPassAdaptor(
1360 Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1361 /*AllowSpeculation=*/true),
1362 /*UseMemorySSA=*/true));
1363
1364 // Now that we've vectorized and unrolled loops, we may have more refined
1365 // alignment information, try to re-derive it here.
1366 FPM.addPass(Pass: AlignmentFromAssumptionsPass());
1367}
1368
1369ModulePassManager
1370PassBuilder::buildModuleOptimizationPipeline(OptimizationLevel Level,
1371 ThinOrFullLTOPhase LTOPhase) {
1372 ModulePassManager MPM;
1373
1374 // Run partial inlining pass to partially inline functions that have
1375 // large bodies.
1376 if (Opts.enable_partial_inlining)
1377 MPM.addPass(Pass: PartialInlinerPass());
1378
1379 // Remove avail extern fns and globals definitions since we aren't compiling
1380 // an object file for later LTO. For LTO we want to preserve these so they
1381 // are eligible for inlining at link-time. Note if they are unreferenced they
1382 // will be removed by GlobalDCE later, so this only impacts referenced
1383 // available externally globals. Eventually they will be suppressed during
1384 // codegen, but eliminating here enables more opportunity for GlobalDCE as it
1385 // may make globals referenced by available external functions dead and saves
1386 // running remaining passes on the eliminated functions. These should be
1387 // preserved during prelinking for link-time inlining decisions.
1388 if (!isLTOPreLink(Phase: LTOPhase))
1389 MPM.addPass(Pass: EliminateAvailableExternallyPass());
1390
1391 // Do RPO function attribute inference across the module to forward-propagate
1392 // attributes where applicable.
1393 // FIXME: Is this really an optimization rather than a canonicalization?
1394 MPM.addPass(Pass: ReversePostOrderFunctionAttrsPass());
1395
1396 // Do a post inline PGO instrumentation and use pass. This is a context
1397 // sensitive PGO pass. We don't want to do this in LTOPreLink phrase as
1398 // cross-module inline has not been done yet. The context sensitive
1399 // instrumentation is after all the inlines are done.
1400 if (!isLTOPreLink(Phase: LTOPhase) && PGOOpt) {
1401 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
1402 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
1403 /*IsCS=*/true, AtomicCounterUpdate: PGOOpt->AtomicCounterUpdate,
1404 ProfileFile: PGOOpt->CSProfileGenFile, ProfileRemappingFile: PGOOpt->ProfileRemappingFile);
1405 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
1406 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
1407 /*IsCS=*/true, AtomicCounterUpdate: PGOOpt->AtomicCounterUpdate,
1408 ProfileFile: PGOOpt->ProfileFile, ProfileRemappingFile: PGOOpt->ProfileRemappingFile);
1409 }
1410
1411 // Re-compute GlobalsAA here prior to function passes. This is particularly
1412 // useful as the above will have inlined, DCE'ed, and function-attr
1413 // propagated everything. We should at this point have a reasonably minimal
1414 // and richly annotated call graph. By computing aliasing and mod/ref
1415 // information for all local globals here, the late loop passes and notably
1416 // the vectorizer will be able to use them to help recognize vectorizable
1417 // memory operations.
1418 if (Opts.enable_global_analyses)
1419 MPM.addPass(Pass: RecomputeGlobalsAAPass());
1420
1421 invokeOptimizerEarlyEPCallbacks(MPM, Level, Phase: LTOPhase);
1422
1423 FunctionPassManager OptimizePM;
1424
1425 // Only drop unnecessary assumes post-inline and post-link, as otherwise
1426 // additional uses of the affected value may be introduced through inlining
1427 // and CSE.
1428 if (!isLTOPreLink(Phase: LTOPhase))
1429 OptimizePM.addPass(Pass: DropUnnecessaryAssumesPass());
1430
1431 // Scheduling LoopVersioningLICM when inlining is over, because after that
1432 // we may see more accurate aliasing. Reason to run this late is that too
1433 // early versioning may prevent further inlining due to increase of code
1434 // size. Other optimizations which runs later might get benefit of no-alias
1435 // assumption in clone loop.
1436 if (Opts.enable_loop_versioning_licm) {
1437 OptimizePM.addPass(
1438 Pass: createFunctionToLoopPassAdaptor(Pass: LoopVersioningLICMPass()));
1439 // LoopVersioningLICM pass might increase new LICM opportunities.
1440 OptimizePM.addPass(Pass: createFunctionToLoopPassAdaptor(
1441 Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1442 /*AllowSpeculation=*/true),
1443 /*USeMemorySSA=*/UseMemorySSA: true));
1444 }
1445
1446 OptimizePM.addPass(Pass: Float2IntPass());
1447 // Defer until LTO post-link where some constants may become known.
1448 if (!isLTOPreLink(Phase: LTOPhase))
1449 OptimizePM.addPass(Pass: LowerConstantIntrinsicsPass());
1450
1451 if (Opts.enable_matrix) {
1452 OptimizePM.addPass(Pass: LowerMatrixIntrinsicsPass());
1453 OptimizePM.addPass(Pass: EarlyCSEPass());
1454 }
1455
1456 // CHR pass should only be applied with the profile information.
1457 // The check is to check the profile summary information in CHR.
1458 if (Opts.enable_chr && Level == OptimizationLevel::O3)
1459 OptimizePM.addPass(Pass: ControlHeightReductionPass());
1460
1461 // FIXME: We need to run some loop optimizations to re-rotate loops after
1462 // simplifycfg and others undo their rotation.
1463
1464 // Optimize the loop execution. These passes operate on entire loop nests
1465 // rather than on each loop in an inside-out manner, and so they are actually
1466 // function passes.
1467
1468 invokeVectorizerStartEPCallbacks(FPM&: OptimizePM, Level);
1469
1470 LoopPassManager LPM;
1471 // First rotate loops that may have been un-rotated by prior passes.
1472 // Disable header duplication at -Oz.
1473 LPM.addPass(Pass: LoopRotatePass(/*EnableLoopHeaderDuplication=*/true,
1474 isLTOPreLink(Phase: LTOPhase),
1475 /*CheckExitCount=*/true));
1476 // Some loops may have become dead by now. Try to delete them.
1477 // FIXME: see discussion in https://reviews.llvm.org/D112851,
1478 // this may need to be revisited once we run GVN before loop deletion
1479 // in the simplification pipeline.
1480 LPM.addPass(Pass: LoopDeletionPass());
1481
1482 if (PTO.LoopInterchange)
1483 LPM.addPass(Pass: LoopInterchangePass());
1484
1485 OptimizePM.addPass(
1486 Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM), /*UseMemorySSA=*/false));
1487
1488 // FIXME: This may not be the right place in the pipeline.
1489 // We need to have the data to support the right place.
1490 if (PTO.LoopFusion)
1491 OptimizePM.addPass(Pass: LoopFusePass());
1492
1493 // Distribute loops to allow partial vectorization. I.e. isolate dependences
1494 // into separate loop that would otherwise inhibit vectorization. This is
1495 // currently only performed for loops marked with the metadata
1496 // llvm.loop.distribute=true or when -enable-loop-distribute is specified.
1497 OptimizePM.addPass(Pass: LoopDistributePass());
1498
1499 // Populates the VFABI attribute with the scalar-to-vector mappings
1500 // from the TargetLibraryInfo.
1501 OptimizePM.addPass(Pass: InjectTLIMappings());
1502
1503 addVectorPasses(Level, FPM&: OptimizePM, LTOPhase);
1504
1505 invokeVectorizerEndEPCallbacks(FPM&: OptimizePM, Level);
1506
1507 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
1508 // canonicalization pass that enables other optimizations. As a result,
1509 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
1510 // result too early.
1511 OptimizePM.addPass(Pass: LoopSinkPass());
1512
1513 // And finally clean up LCSSA form before generating code.
1514 OptimizePM.addPass(Pass: InstSimplifyPass());
1515
1516 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
1517 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
1518 // flattening of blocks.
1519 OptimizePM.addPass(Pass: DivRemPairsPass());
1520
1521 // Merge adjacent icmps into memcmp, then expand memcmp to loads/compares.
1522 // TODO: move this furter up so that it can be optimized by GVN, etc.
1523 if (Opts.enable_mergeicmps)
1524 OptimizePM.addPass(Pass: MergeICmpsPass());
1525 OptimizePM.addPass(Pass: ExpandMemCmpPass());
1526
1527 // Try to annotate calls that were created during optimization.
1528 OptimizePM.addPass(
1529 Pass: TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
1530
1531 // LoopSink (and other loop passes since the last simplifyCFG) might have
1532 // resulted in single-entry-single-exit or empty blocks. Clean up the CFG.
1533 OptimizePM.addPass(
1534 Pass: SimplifyCFGPass(SimplifyCFGOptions()
1535 .convertSwitchRangeToICmp(B: true)
1536 .convertSwitchToArithmetic(B: true)
1537 .speculateUnpredictables(B: true)
1538 .hoistLoadsStoresWithCondFaulting(B: true)));
1539
1540 // Add the core optimizing pipeline.
1541 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(OptimizePM),
1542 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
1543
1544 // AllocToken transforms heap allocation calls; this needs to run late after
1545 // other allocation call transformations (such as those in InstCombine).
1546 if (!isLTOPreLink(Phase: LTOPhase))
1547 MPM.addPass(Pass: AllocTokenPass());
1548
1549 invokeOptimizerLastEPCallbacks(MPM, Level, Phase: LTOPhase);
1550
1551 // Run the Instrumentor pass late.
1552 if (Opts.enable_instrumentor)
1553 MPM.addPass(Pass: InstrumentorPass(FS));
1554
1555 // Split out cold code. Splitting is done late to avoid hiding context from
1556 // other optimizations and inadvertently regressing performance. The tradeoff
1557 // is that this has a higher code size cost than splitting early.
1558 if (Opts.hot_cold_split && !isLTOPreLink(Phase: LTOPhase))
1559 MPM.addPass(Pass: HotColdSplittingPass());
1560
1561 // Now we need to do some global optimization transforms.
1562 // FIXME: It would seem like these should come first in the optimization
1563 // pipeline and maybe be the bottom of the canonicalization pipeline? Weird
1564 // ordering here.
1565 MPM.addPass(Pass: GlobalDCEPass());
1566 MPM.addPass(Pass: ConstantMergePass());
1567
1568 // Merge functions if requested. It has a better chance to merge functions
1569 // after ConstantMerge folded jump tables.
1570 if (PTO.MergeFunctions)
1571 MPM.addPass(Pass: MergeFunctionsPass());
1572
1573 if (PTO.CallGraphProfile && !isLTOPreLink(Phase: LTOPhase))
1574 MPM.addPass(Pass: CGProfilePass(isLTOPostLink(Phase: LTOPhase)));
1575
1576 // RelLookupTableConverterPass runs later in LTO post-link pipeline.
1577 if (!isLTOPreLink(Phase: LTOPhase))
1578 MPM.addPass(Pass: RelLookupTableConverterPass());
1579
1580 // Add devirtualization pass only when LTO is not enabled, as otherwise
1581 // the pass is already enabled in the LTO pipeline.
1582 if (PTO.DevirtualizeSpeculatively && LTOPhase == ThinOrFullLTOPhase::None) {
1583 // TODO: explore a better pipeline configuration that can improve
1584 // compilation time overhead.
1585 // FIXME: move this earlier (lots of pass ordering tests will need fixing)
1586 MPM.addPass(Pass: AssignGUIDPass());
1587 MPM.addPass(Pass: WholeProgramDevirtPass(
1588 /*ExportSummary*/ nullptr,
1589 /*ImportSummary*/ nullptr,
1590 /*DevirtSpeculatively*/ PTO.DevirtualizeSpeculatively));
1591 MPM.addPass(Pass: DropTypeTestsPass());
1592 // Given that the devirtualization creates more opportunities for inlining,
1593 // we run the Inliner again here to maximize the optimization gain we
1594 // get from devirtualization.
1595 // Also, we can't run devirtualization before inlining because the
1596 // devirtualization depends on the passes optimizing/eliminating vtable GVs
1597 // and those passes are only effective after inlining.
1598 addModuleInlinerPass(MPM, Opts, Level, Phase: ThinOrFullLTOPhase::None);
1599 }
1600
1601 // Attach !implicit.ref metadata from all functions to copyright strings.
1602 MPM.addPass(Pass: LowerCommentStringPass());
1603
1604 return MPM;
1605}
1606
1607ModulePassManager
1608PassBuilder::buildPerModuleDefaultPipeline(OptimizationLevel Level,
1609 ThinOrFullLTOPhase Phase) {
1610 if (Level == OptimizationLevel::O0)
1611 return buildO0DefaultPipeline(Level, Phase);
1612
1613 ModulePassManager MPM;
1614 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1615 // Currently this pipeline is only invoked in an LTO pre link pass or when we
1616 // are not running LTO. If that changes the below checks may need updating.
1617 assert(isLTOPreLink(Phase) || Phase == ThinOrFullLTOPhase::None);
1618
1619 // If we are invoking this in non-LTO mode, remove any MemProf related
1620 // attributes and metadata, as we don't know whether we are linking with
1621 // a library containing the necessary interfaces.
1622 if (Phase == ThinOrFullLTOPhase::None)
1623 MPM.addPass(Pass: MemProfRemoveInfo());
1624
1625 // Convert @llvm.global.annotations to !annotation metadata.
1626 MPM.addPass(Pass: Annotation2MetadataPass());
1627
1628 // Force any function attributes we want the rest of the pipeline to observe.
1629 MPM.addPass(Pass: ForceFunctionAttrsPass());
1630
1631 if (Opts.opt_pipeline_trigger_crash)
1632 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: TriggerCrashFunctionPass()));
1633
1634 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1635 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: AddDiscriminatorsPass()));
1636
1637 // Apply module pipeline start EP callback.
1638 invokePipelineStartEPCallbacks(MPM, Level);
1639
1640 // Add the core simplification pipeline.
1641 MPM.addPass(Pass: buildModuleSimplificationPipeline(Level, Phase));
1642
1643 // Now add the optimization pipeline.
1644 MPM.addPass(Pass: buildModuleOptimizationPipeline(Level, LTOPhase: Phase));
1645
1646 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1647 PGOOpt->Action == PGOOptions::SampleUse)
1648 MPM.addPass(Pass: PseudoProbeUpdatePass());
1649
1650 // Emit annotation remarks.
1651 addAnnotationRemarksPass(MPM);
1652
1653 if (isLTOPreLink(Phase))
1654 addRequiredLTOPreLinkPasses(MPM);
1655
1656 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1657 return MPM;
1658}
1659
1660ModulePassManager
1661PassBuilder::buildFatLTODefaultPipeline(OptimizationLevel Level, bool ThinLTO,
1662 bool EmitSummary, bool Verify) {
1663 ModulePassManager MPM;
1664
1665 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1666
1667 if (ThinLTO)
1668 MPM.addPass(Pass: buildThinLTOPreLinkDefaultPipeline(Level));
1669 else
1670 MPM.addPass(Pass: buildLTOPreLinkDefaultPipeline(Level));
1671 // AssignGUIDPass attaches !guid metadata (MD_unique_id) to global objects,
1672 // triggering the bitcode writer to emit a METADATA_KIND_BLOCK. Standard LTO
1673 // bitcode emission runs VerifierPass by default, which registers metadata
1674 // kind IDs in LLVMContext. Running VerifierPass here before EmbedBitcodePass
1675 // to get the same behavior.
1676 if (Verify)
1677 MPM.addPass(Pass: VerifierPass());
1678 MPM.addPass(Pass: EmbedBitcodePass(ThinLTO, EmitSummary));
1679
1680 // Perform any cleanups to the IR that aren't suitable for per TU compilation,
1681 // like removing CFI/WPD related instructions. Note, we reuse
1682 // DropTypeTestsPass to clean up type tests rather than duplicate that logic
1683 // in FatLtoCleanup.
1684 MPM.addPass(Pass: FatLtoCleanup());
1685
1686 // If we're doing FatLTO w/ CFI enabled, we don't want the type tests in the
1687 // object code, only in the bitcode section, so drop it before we run
1688 // module optimization and generate machine code. If llvm.type.test() isn't in
1689 // the IR, this won't do anything.
1690 MPM.addPass(Pass: DropTypeTestsPass(lowertypetests::DropTestKind::All));
1691
1692 // Use the ThinLTO post-link pipeline with sample profiling
1693 if (ThinLTO && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
1694 MPM.addPass(Pass: buildThinLTODefaultPipeline(Level, /*ImportSummary=*/nullptr));
1695 else {
1696 // ModuleSimplification does not run the coroutine passes for
1697 // ThinLTOPreLink, so we need the coroutine passes to run for ThinLTO
1698 // builds, otherwise they will miscompile.
1699 if (ThinLTO) {
1700 // TODO: replace w/ buildCoroWrapper() when it takes phase and level into
1701 // consideration.
1702 CGSCCPassManager CGPM;
1703 CGPM.addPass(Pass: CoroSplitPass(Level != OptimizationLevel::O0));
1704 CGPM.addPass(Pass: CoroAnnotationElidePass());
1705 MPM.addPass(Pass: createModuleToPostOrderCGSCCPassAdaptor(Pass: std::move(CGPM)));
1706 MPM.addPass(Pass: CoroCleanupPass());
1707 }
1708
1709 // otherwise, just use module optimization
1710 MPM.addPass(
1711 Pass: buildModuleOptimizationPipeline(Level, LTOPhase: ThinOrFullLTOPhase::None));
1712 // Emit annotation remarks.
1713 addAnnotationRemarksPass(MPM);
1714 }
1715
1716 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1717
1718 return MPM;
1719}
1720
1721ModulePassManager
1722PassBuilder::buildThinLTOPreLinkDefaultPipeline(OptimizationLevel Level) {
1723 if (Level == OptimizationLevel::O0)
1724 return buildO0DefaultPipeline(Level, Phase: ThinOrFullLTOPhase::ThinLTOPreLink);
1725
1726 ModulePassManager MPM;
1727
1728 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1729
1730 // Convert @llvm.global.annotations to !annotation metadata.
1731 MPM.addPass(Pass: Annotation2MetadataPass());
1732
1733 // Force any function attributes we want the rest of the pipeline to observe.
1734 MPM.addPass(Pass: ForceFunctionAttrsPass());
1735
1736 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1737 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: AddDiscriminatorsPass()));
1738
1739 // Apply module pipeline start EP callback.
1740 invokePipelineStartEPCallbacks(MPM, Level);
1741
1742 // If we are planning to perform ThinLTO later, we don't bloat the code with
1743 // unrolling/vectorization/... now. Just simplify the module as much as we
1744 // can.
1745 MPM.addPass(Pass: buildModuleSimplificationPipeline(
1746 Level, Phase: ThinOrFullLTOPhase::ThinLTOPreLink));
1747 // In pre-link, for ctx prof use, we stop here with an instrumented IR. We let
1748 // thinlto use the contextual info to perform imports; then use the contextual
1749 // profile in the post-thinlink phase.
1750 if (!UseCtxProfile.empty()) {
1751 addRequiredLTOPreLinkPasses(MPM);
1752 return MPM;
1753 }
1754
1755 // Run partial inlining pass to partially inline functions that have
1756 // large bodies.
1757 // FIXME: It isn't clear whether this is really the right place to run this
1758 // in ThinLTO. Because there is another canonicalization and simplification
1759 // phase that will run after the thin link, running this here ends up with
1760 // less information than will be available later and it may grow functions in
1761 // ways that aren't beneficial.
1762 if (Opts.enable_partial_inlining)
1763 MPM.addPass(Pass: PartialInlinerPass());
1764
1765 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1766 PGOOpt->Action == PGOOptions::SampleUse)
1767 MPM.addPass(Pass: PseudoProbeUpdatePass());
1768
1769 // Handle Optimizer{Early,Last}EPCallbacks added by clang on PreLink. Actual
1770 // optimization is going to be done in PostLink stage, but clang can't add
1771 // callbacks there in case of in-process ThinLTO called by linker.
1772 invokeOptimizerEarlyEPCallbacks(MPM, Level,
1773 /*Phase=*/ThinOrFullLTOPhase::ThinLTOPreLink);
1774 invokeOptimizerLastEPCallbacks(MPM, Level,
1775 /*Phase=*/ThinOrFullLTOPhase::ThinLTOPreLink);
1776
1777 // Emit annotation remarks.
1778 addAnnotationRemarksPass(MPM);
1779
1780 // Attach !implicit.ref metadata from all functions to copyright strings.
1781 MPM.addPass(Pass: LowerCommentStringPass());
1782
1783 addRequiredLTOPreLinkPasses(MPM);
1784
1785 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1786
1787 return MPM;
1788}
1789
1790ModulePassManager PassBuilder::buildThinLTODefaultPipeline(
1791 OptimizationLevel Level, const ModuleSummaryIndex *ImportSummary) {
1792 ModulePassManager MPM;
1793
1794 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1795
1796 invokeThinLinkTimeOptimizationEarlyEPCallbacks(MPM, Level);
1797
1798 // If we are invoking this without a summary index noting that we are linking
1799 // with a library containing the necessary APIs, remove any MemProf related
1800 // attributes and metadata.
1801 if (!ImportSummary || !ImportSummary->withSupportsHotColdNew())
1802 MPM.addPass(Pass: MemProfRemoveInfo());
1803
1804 if (ImportSummary) {
1805 // For ThinLTO we must apply the context disambiguation decisions early, to
1806 // ensure we can correctly match the callsites to summary data.
1807 if (EnableMemProfContextDisambiguation)
1808 MPM.addPass(Pass: MemProfContextDisambiguation(
1809 ImportSummary, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
1810
1811 // These passes import type identifier resolutions for whole-program
1812 // devirtualization and CFI. They must run early because other passes may
1813 // disturb the specific instruction patterns that these passes look for,
1814 // creating dependencies on resolutions that may not appear in the summary.
1815 //
1816 // For example, GVN may transform the pattern assume(type.test) appearing in
1817 // two basic blocks into assume(phi(type.test, type.test)), which would
1818 // transform a dependency on a WPD resolution into a dependency on a type
1819 // identifier resolution for CFI.
1820 //
1821 // Also, WPD has access to more precise information than ICP and can
1822 // devirtualize more effectively, so it should operate on the IR first.
1823 //
1824 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
1825 // metadata and intrinsics.
1826 MPM.addPass(Pass: WholeProgramDevirtPass(nullptr, ImportSummary));
1827 MPM.addPass(Pass: LowerTypeTestsPass(nullptr, ImportSummary));
1828 }
1829
1830 if (Level == OptimizationLevel::O0) {
1831 // Run a second time to clean up any type tests left behind by WPD for use
1832 // in ICP.
1833 MPM.addPass(Pass: DropTypeTestsPass());
1834 MPM.addPass(Pass: buildCoroWrapper(Phase: ThinOrFullLTOPhase::ThinLTOPostLink));
1835
1836 // AllocToken transforms heap allocation calls; this needs to run late after
1837 // other allocation call transformations (such as those in InstCombine).
1838 MPM.addPass(Pass: AllocTokenPass());
1839
1840 // Drop available_externally and unreferenced globals. This is necessary
1841 // with ThinLTO in order to avoid leaving undefined references to dead
1842 // globals in the object file.
1843 MPM.addPass(Pass: EliminateAvailableExternallyPass());
1844 MPM.addPass(Pass: GlobalDCEPass());
1845
1846 invokeThinLinkTimeOptimizationLastEPCallbacks(MPM, Level);
1847
1848 return MPM;
1849 }
1850 if (!UseCtxProfile.empty()) {
1851 MPM.addPass(
1852 Pass: buildModuleInlinerPipeline(Level, Phase: ThinOrFullLTOPhase::ThinLTOPostLink));
1853 } else {
1854 // Add the core simplification pipeline.
1855 MPM.addPass(Pass: buildModuleSimplificationPipeline(
1856 Level, Phase: ThinOrFullLTOPhase::ThinLTOPostLink));
1857 }
1858 // Now add the optimization pipeline.
1859 MPM.addPass(Pass: buildModuleOptimizationPipeline(
1860 Level, LTOPhase: ThinOrFullLTOPhase::ThinLTOPostLink));
1861
1862 invokeThinLinkTimeOptimizationLastEPCallbacks(MPM, Level);
1863
1864 // Emit annotation remarks.
1865 addAnnotationRemarksPass(MPM);
1866
1867 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1868
1869 return MPM;
1870}
1871
1872ModulePassManager
1873PassBuilder::buildLTOPreLinkDefaultPipeline(OptimizationLevel Level) {
1874 // FIXME: We should use a customized pre-link pipeline!
1875 return buildPerModuleDefaultPipeline(Level,
1876 Phase: ThinOrFullLTOPhase::FullLTOPreLink);
1877}
1878
1879ModulePassManager
1880PassBuilder::buildLTODefaultPipeline(OptimizationLevel Level,
1881 ModuleSummaryIndex *ExportSummary) {
1882 ModulePassManager MPM;
1883
1884 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1885
1886 invokeFullLinkTimeOptimizationEarlyEPCallbacks(MPM, Level);
1887
1888 // If we are invoking this without a summary index noting that we are linking
1889 // with a library containing the necessary APIs, remove any MemProf related
1890 // attributes and metadata.
1891 if (!ExportSummary || !ExportSummary->withSupportsHotColdNew())
1892 MPM.addPass(Pass: MemProfRemoveInfo());
1893
1894 // Create a function that performs CFI checks for cross-DSO calls with targets
1895 // in the current module.
1896 MPM.addPass(Pass: CrossDSOCFIPass());
1897
1898 if (Level == OptimizationLevel::O0) {
1899 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
1900 // metadata and intrinsics.
1901 MPM.addPass(Pass: WholeProgramDevirtPass(ExportSummary, nullptr));
1902 MPM.addPass(Pass: LowerTypeTestsPass(ExportSummary, nullptr));
1903 // Run a second time to clean up any type tests left behind by WPD for use
1904 // in ICP.
1905 MPM.addPass(Pass: DropTypeTestsPass());
1906
1907 MPM.addPass(Pass: buildCoroWrapper(Phase: ThinOrFullLTOPhase::FullLTOPostLink));
1908
1909 // AllocToken transforms heap allocation calls; this needs to run late after
1910 // other allocation call transformations (such as those in InstCombine).
1911 MPM.addPass(Pass: AllocTokenPass());
1912
1913 invokeFullLinkTimeOptimizationLastEPCallbacks(MPM, Level);
1914
1915 // Emit annotation remarks.
1916 addAnnotationRemarksPass(MPM);
1917
1918 return MPM;
1919 }
1920
1921 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
1922 // Load sample profile before running the LTO optimization pipeline.
1923 MPM.addPass(Pass: SampleProfileLoaderPass(PGOOpt->ProfileFile,
1924 PGOOpt->ProfileRemappingFile,
1925 ThinOrFullLTOPhase::FullLTOPostLink));
1926 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
1927 // RequireAnalysisPass for PSI before subsequent non-module passes.
1928 MPM.addPass(Pass: RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
1929 }
1930
1931 // Try to run OpenMP optimizations, quick no-op if no OpenMP metadata present.
1932 MPM.addPass(Pass: OpenMPOptPass(ThinOrFullLTOPhase::FullLTOPostLink));
1933
1934 // Remove unused virtual tables to improve the quality of code generated by
1935 // whole-program devirtualization and bitset lowering.
1936 MPM.addPass(Pass: GlobalDCEPass(/*InLTOPostLink=*/true));
1937
1938 // Do basic inference of function attributes from known properties of system
1939 // libraries and other oracles.
1940 MPM.addPass(Pass: InferFunctionAttrsPass());
1941
1942 if (Level >= OptimizationLevel::O2) {
1943 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
1944 Pass: CallSiteSplittingPass(), EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
1945
1946 // Indirect call promotion. This should promote all the targets that are
1947 // left by the earlier promotion pass that promotes intra-module targets.
1948 // This two-step promotion is to save the compile time. For LTO, it should
1949 // produce the same result as if we only do promotion here.
1950 MPM.addPass(Pass: PGOIndirectCallPromotion(
1951 true /* InLTO */, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
1952
1953 // Promoting by-reference arguments to by-value exposes more constants to
1954 // IPSCCP.
1955 CGSCCPassManager CGPM;
1956 CGPM.addPass(Pass: PostOrderFunctionAttrsPass());
1957 CGPM.addPass(Pass: ArgumentPromotionPass());
1958 CGPM.addPass(
1959 Pass: createCGSCCToFunctionPassAdaptor(Pass: SROAPass(SROAOptions::ModifyCFG)));
1960 MPM.addPass(Pass: createModuleToPostOrderCGSCCPassAdaptor(Pass: std::move(CGPM)));
1961
1962 // Propagate constants at call sites into the functions they call. This
1963 // opens opportunities for globalopt (and inlining) by substituting function
1964 // pointers passed as arguments to direct uses of functions.
1965 MPM.addPass(Pass: IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/true)));
1966
1967 // Attach metadata to indirect call sites indicating the set of functions
1968 // they may target at run-time. This should follow IPSCCP.
1969 MPM.addPass(Pass: CalledValuePropagationPass());
1970 }
1971
1972 // Do RPO function attribute inference across the module to forward-propagate
1973 // attributes where applicable.
1974 // FIXME: Is this really an optimization rather than a canonicalization?
1975 MPM.addPass(Pass: ReversePostOrderFunctionAttrsPass());
1976
1977 // Use in-range annotations on GEP indices to split globals where beneficial.
1978 MPM.addPass(Pass: GlobalSplitPass());
1979
1980 // Run whole program optimization of virtual call when the list of callees
1981 // is fixed.
1982 MPM.addPass(Pass: WholeProgramDevirtPass(ExportSummary, nullptr));
1983
1984 MPM.addPass(Pass: NoRecurseLTOInferencePass());
1985 // Stop here at -O1.
1986 if (Level == OptimizationLevel::O1) {
1987 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
1988 Pass: LowerConstantIntrinsicsPass(), EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
1989
1990 // The LowerTypeTestsPass needs to run to lower type metadata and the
1991 // type.test intrinsics. The pass does nothing if CFI is disabled.
1992 MPM.addPass(Pass: LowerTypeTestsPass(ExportSummary, nullptr));
1993 // Run a second time to clean up any type tests left behind by WPD for use
1994 // in ICP (which is performed earlier than this in the regular LTO
1995 // pipeline).
1996 MPM.addPass(Pass: DropTypeTestsPass());
1997
1998 MPM.addPass(Pass: buildCoroWrapper(Phase: ThinOrFullLTOPhase::FullLTOPostLink));
1999
2000 // AllocToken transforms heap allocation calls; this needs to run late after
2001 // other allocation call transformations (such as those in InstCombine).
2002 MPM.addPass(Pass: AllocTokenPass());
2003
2004 invokeFullLinkTimeOptimizationLastEPCallbacks(MPM, Level);
2005
2006 // Emit annotation remarks.
2007 addAnnotationRemarksPass(MPM);
2008
2009 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2010
2011 return MPM;
2012 }
2013
2014 // TODO: Skip to match buildCoroWrapper.
2015 MPM.addPass(Pass: CoroEarlyPass());
2016
2017 // Optimize globals to try and fold them into constants.
2018 MPM.addPass(Pass: GlobalOptPass());
2019
2020 // Promote any localized globals to SSA registers.
2021 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: PromotePass()));
2022
2023 // Linking modules together can lead to duplicate global constant, only
2024 // keep one copy of each constant.
2025 MPM.addPass(Pass: ConstantMergePass());
2026
2027 // Remove unused arguments from functions.
2028 MPM.addPass(Pass: DeadArgumentEliminationPass());
2029
2030 // Reduce the code after globalopt and ipsccp. Both can open up significant
2031 // simplification opportunities, and both can propagate functions through
2032 // function pointers. When this happens, we often have to resolve varargs
2033 // calls, etc, so let instcombine do this.
2034 FunctionPassManager PeepholeFPM;
2035 PeepholeFPM.addPass(Pass: InstCombinePass());
2036 if (Level >= OptimizationLevel::O2)
2037 PeepholeFPM.addPass(Pass: AggressiveInstCombinePass());
2038 invokePeepholeEPCallbacks(FPM&: PeepholeFPM, Level);
2039
2040 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(PeepholeFPM),
2041 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
2042
2043 // Lower variadic functions for supported targets prior to inlining.
2044 MPM.addPass(Pass: ExpandVariadicsPass(ExpandVariadicsMode::Optimize));
2045
2046 // Note: historically, the PruneEH pass was run first to deduce nounwind and
2047 // generally clean up exception handling overhead. It isn't clear this is
2048 // valuable as the inliner doesn't currently care whether it is inlining an
2049 // invoke or a call.
2050 // Run the inliner now.
2051 addModuleInlinerPass(MPM, Opts, Level, Phase: ThinOrFullLTOPhase::FullLTOPostLink);
2052
2053 // Perform context disambiguation after inlining, since that would reduce the
2054 // amount of additional cloning required to distinguish the allocation
2055 // contexts.
2056 if (EnableMemProfContextDisambiguation)
2057 MPM.addPass(Pass: MemProfContextDisambiguation(
2058 /*Summary=*/nullptr,
2059 PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
2060
2061 // Optimize globals again after we ran the inliner.
2062 MPM.addPass(Pass: GlobalOptPass());
2063
2064 // Run the OpenMPOpt pass again after global optimizations.
2065 MPM.addPass(Pass: OpenMPOptPass(ThinOrFullLTOPhase::FullLTOPostLink));
2066
2067 // Garbage collect dead functions.
2068 MPM.addPass(Pass: GlobalDCEPass(/*InLTOPostLink=*/true));
2069
2070 // If we didn't decide to inline a function, check to see if we can
2071 // transform it to pass arguments by value instead of by reference.
2072 CGSCCPassManager CGPM;
2073 CGPM.addPass(Pass: ArgumentPromotionPass());
2074 CGPM.addPass(Pass: CoroSplitPass(Level != OptimizationLevel::O0));
2075 CGPM.addPass(Pass: CoroAnnotationElidePass());
2076 invokeCGSCCOptimizerLateEPCallbacks(CGPM, Level);
2077 MPM.addPass(Pass: createModuleToPostOrderCGSCCPassAdaptor(Pass: std::move(CGPM)));
2078
2079 FunctionPassManager FPM;
2080 // The IPO Passes may leave cruft around. Clean up after them.
2081 FPM.addPass(Pass: InstCombinePass());
2082 invokePeepholeEPCallbacks(FPM, Level);
2083
2084 if (Opts.enable_constraint_elimination)
2085 FPM.addPass(Pass: ConstraintEliminationPass());
2086
2087 FPM.addPass(Pass: JumpThreadingPass());
2088
2089 // Do a post inline PGO instrumentation and use pass. This is a context
2090 // sensitive PGO pass.
2091 if (PGOOpt) {
2092 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
2093 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
2094 /*IsCS=*/true, AtomicCounterUpdate: PGOOpt->AtomicCounterUpdate,
2095 ProfileFile: PGOOpt->CSProfileGenFile, ProfileRemappingFile: PGOOpt->ProfileRemappingFile);
2096 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
2097 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
2098 /*IsCS=*/true, AtomicCounterUpdate: PGOOpt->AtomicCounterUpdate,
2099 ProfileFile: PGOOpt->ProfileFile, ProfileRemappingFile: PGOOpt->ProfileRemappingFile);
2100 }
2101
2102 // Break up allocas
2103 FPM.addPass(Pass: SROAPass(SROAOptions::ModifyCFG));
2104
2105 // LTO provides additional opportunities for tailcall elimination due to
2106 // link-time inlining, and visibility of nocapture attribute.
2107 FPM.addPass(
2108 Pass: TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
2109
2110 // Run a few AA driver optimizations here and now to cleanup the code.
2111 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(FPM),
2112 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
2113
2114 MPM.addPass(
2115 Pass: createModuleToPostOrderCGSCCPassAdaptor(Pass: PostOrderFunctionAttrsPass()));
2116
2117 // Require the GlobalsAA analysis for the module so we can query it within
2118 // MainFPM.
2119 if (Opts.enable_global_analyses) {
2120 MPM.addPass(Pass: RequireAnalysisPass<GlobalsAA, Module>());
2121 // Invalidate AAManager so it can be recreated and pick up the newly
2122 // available GlobalsAA.
2123 MPM.addPass(
2124 Pass: createModuleToFunctionPassAdaptor(Pass: InvalidateAnalysisPass<AAManager>()));
2125 }
2126
2127 FunctionPassManager MainFPM;
2128 MainFPM.addPass(Pass: createFunctionToLoopPassAdaptor(
2129 Pass: LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
2130 /*AllowSpeculation=*/true),
2131 /*USeMemorySSA=*/UseMemorySSA: true));
2132
2133 if (Opts.enable_newgvn)
2134 MainFPM.addPass(Pass: NewGVNPass());
2135 else
2136 MainFPM.addPass(Pass: GVNPass());
2137
2138 // Remove dead memcpy()'s.
2139 MainFPM.addPass(Pass: MemCpyOptPass());
2140
2141 // Nuke dead stores.
2142 MainFPM.addPass(Pass: DSEPass());
2143 MainFPM.addPass(Pass: MoveAutoInitPass());
2144 MainFPM.addPass(Pass: MergedLoadStoreMotionPass());
2145
2146 MainFPM.addPass(Pass: LowerConstantIntrinsicsPass());
2147
2148 invokeVectorizerStartEPCallbacks(FPM&: MainFPM, Level);
2149
2150 LoopPassManager LPM;
2151 if (Opts.enable_loop_flatten && Level >= OptimizationLevel::O2)
2152 LPM.addPass(Pass: LoopFlattenPass());
2153 LPM.addPass(Pass: IndVarSimplifyPass());
2154 LPM.addPass(Pass: LoopDeletionPass());
2155 // FIXME: Add loop interchange.
2156
2157 // Unroll small loops and perform peeling.
2158 LPM.addPass(Pass: LoopFullUnrollPass(static_cast<int>(Level),
2159 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
2160 PTO.ForgetAllSCEVInLoopUnroll));
2161 // The loop passes in LPM (LoopFullUnrollPass) do not preserve MemorySSA.
2162 // *All* loop passes must preserve it, in order to be able to use it.
2163 MainFPM.addPass(
2164 Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM), /*UseMemorySSA=*/false));
2165
2166 MainFPM.addPass(Pass: LoopDistributePass());
2167
2168 addVectorPasses(Level, FPM&: MainFPM, LTOPhase: ThinOrFullLTOPhase::FullLTOPostLink);
2169
2170 invokeVectorizerEndEPCallbacks(FPM&: MainFPM, Level);
2171
2172 // Run the OpenMPOpt CGSCC pass again late.
2173 MPM.addPass(Pass: createModuleToPostOrderCGSCCPassAdaptor(
2174 Pass: OpenMPOptCGSCCPass(ThinOrFullLTOPhase::FullLTOPostLink)));
2175
2176 invokePeepholeEPCallbacks(FPM&: MainFPM, Level);
2177 MainFPM.addPass(Pass: JumpThreadingPass());
2178 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(MainFPM),
2179 EagerlyInvalidate: PTO.EagerlyInvalidateAnalyses));
2180
2181 // Lower type metadata and the type.test intrinsic. This pass supports
2182 // clang's control flow integrity mechanisms (-fsanitize=cfi*) and needs
2183 // to be run at link time if CFI is enabled. This pass does nothing if
2184 // CFI is disabled.
2185 MPM.addPass(Pass: LowerTypeTestsPass(ExportSummary, nullptr));
2186 // Run a second time to clean up any type tests left behind by WPD for use
2187 // in ICP (which is performed earlier than this in the regular LTO pipeline).
2188 MPM.addPass(Pass: DropTypeTestsPass());
2189
2190 // Enable splitting late in the FullLTO post-link pipeline.
2191 if (Opts.hot_cold_split)
2192 MPM.addPass(Pass: HotColdSplittingPass());
2193
2194 // Add late LTO optimization passes.
2195 FunctionPassManager LateFPM;
2196
2197 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
2198 // canonicalization pass that enables other optimizations. As a result,
2199 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
2200 // result too early.
2201 LateFPM.addPass(Pass: LoopSinkPass());
2202
2203 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
2204 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
2205 // flattening of blocks.
2206 LateFPM.addPass(Pass: DivRemPairsPass());
2207
2208 // Delete basic blocks, which optimization passes may have killed.
2209 LateFPM.addPass(Pass: SimplifyCFGPass(SimplifyCFGOptions()
2210 .convertSwitchRangeToICmp(B: true)
2211 .convertSwitchToArithmetic(B: true)
2212 .hoistCommonInsts(B: true)
2213 .speculateUnpredictables(B: true)));
2214 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(LateFPM)));
2215
2216 // Drop bodies of available eternally objects to improve GlobalDCE.
2217 MPM.addPass(Pass: EliminateAvailableExternallyPass());
2218
2219 // Now that we have optimized the program, discard unreachable functions.
2220 MPM.addPass(Pass: GlobalDCEPass(/*InLTOPostLink=*/true));
2221
2222 if (PTO.MergeFunctions)
2223 MPM.addPass(Pass: MergeFunctionsPass());
2224
2225 MPM.addPass(Pass: RelLookupTableConverterPass());
2226
2227 if (PTO.CallGraphProfile)
2228 MPM.addPass(Pass: CGProfilePass(/*InLTOPostLink=*/true));
2229
2230 MPM.addPass(Pass: CoroCleanupPass());
2231
2232 // AllocToken transforms heap allocation calls; this needs to run late after
2233 // other allocation call transformations (such as those in InstCombine).
2234 MPM.addPass(Pass: AllocTokenPass());
2235
2236 invokeFullLinkTimeOptimizationLastEPCallbacks(MPM, Level);
2237
2238 // Emit annotation remarks.
2239 addAnnotationRemarksPass(MPM);
2240
2241 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2242
2243 return MPM;
2244}
2245
2246ModulePassManager
2247PassBuilder::buildO0DefaultPipeline(OptimizationLevel Level,
2248 ThinOrFullLTOPhase Phase) {
2249 assert(Level == OptimizationLevel::O0 &&
2250 "buildO0DefaultPipeline should only be used with O0");
2251
2252 ModulePassManager MPM;
2253
2254 instructionCountersPass(MPM, /* IsPreOptimization */ true);
2255
2256 // Perform pseudo probe instrumentation in O0 mode. This is for the
2257 // consistency between different build modes. For example, a LTO build can be
2258 // mixed with an O0 prelink and an O2 postlink. Loading a sample profile in
2259 // the postlink will require pseudo probe instrumentation in the prelink.
2260 if (PGOOpt && PGOOpt->PseudoProbeForProfiling)
2261 MPM.addPass(Pass: SampleProfileProbePass(TM));
2262
2263 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRInstr ||
2264 PGOOpt->Action == PGOOptions::IRUse))
2265 addPGOInstrPassesForO0(
2266 MPM,
2267 /*RunProfileGen=*/(PGOOpt->Action == PGOOptions::IRInstr),
2268 /*IsCS=*/false, AtomicCounterUpdate: PGOOpt->AtomicCounterUpdate, ProfileFile: PGOOpt->ProfileFile,
2269 ProfileRemappingFile: PGOOpt->ProfileRemappingFile);
2270
2271 // Instrument function entry and exit before all inlining.
2272 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
2273 Pass: EntryExitInstrumenterPass(/*PostInlining=*/false)));
2274
2275 invokePipelineStartEPCallbacks(MPM, Level);
2276
2277 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
2278 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: AddDiscriminatorsPass()));
2279
2280 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
2281 // Explicitly disable sample loader inlining and use flattened profile in O0
2282 // pipeline.
2283 MPM.addPass(Pass: SampleProfileLoaderPass(PGOOpt->ProfileFile,
2284 PGOOpt->ProfileRemappingFile,
2285 ThinOrFullLTOPhase::None, FS,
2286 /*DisableSampleProfileInlining=*/true,
2287 /*UseFlattenedProfile=*/true));
2288 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
2289 // RequireAnalysisPass for PSI before subsequent non-module passes.
2290 MPM.addPass(Pass: RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
2291 }
2292
2293 invokePipelineEarlySimplificationEPCallbacks(MPM, Level, Phase);
2294
2295 // Build a minimal pipeline based on the semantics required by LLVM,
2296 // which is just that always inlining occurs. Further, disable generating
2297 // lifetime intrinsics to avoid enabling further optimizations during
2298 // code generation.
2299 MPM.addPass(Pass: AlwaysInlinerPass(
2300 /*InsertLifetimeIntrinsics=*/false));
2301
2302 if (PTO.MergeFunctions)
2303 MPM.addPass(Pass: MergeFunctionsPass());
2304
2305 if (Opts.enable_matrix)
2306 MPM.addPass(
2307 Pass: createModuleToFunctionPassAdaptor(Pass: LowerMatrixIntrinsicsPass(true)));
2308
2309 if (!CGSCCOptimizerLateEPCallbacks.empty()) {
2310 CGSCCPassManager CGPM;
2311 invokeCGSCCOptimizerLateEPCallbacks(CGPM, Level);
2312 if (!CGPM.isEmpty())
2313 MPM.addPass(Pass: createModuleToPostOrderCGSCCPassAdaptor(Pass: std::move(CGPM)));
2314 }
2315 if (!LateLoopOptimizationsEPCallbacks.empty()) {
2316 LoopPassManager LPM;
2317 invokeLateLoopOptimizationsEPCallbacks(LPM, Level);
2318 if (!LPM.isEmpty()) {
2319 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
2320 Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM))));
2321 }
2322 }
2323 if (!LoopOptimizerEndEPCallbacks.empty()) {
2324 LoopPassManager LPM;
2325 invokeLoopOptimizerEndEPCallbacks(LPM, Level);
2326 if (!LPM.isEmpty()) {
2327 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(
2328 Pass: createFunctionToLoopPassAdaptor(Pass: std::move(LPM))));
2329 }
2330 }
2331 if (!ScalarOptimizerLateEPCallbacks.empty()) {
2332 FunctionPassManager FPM;
2333 invokeScalarOptimizerLateEPCallbacks(FPM, Level);
2334 if (!FPM.isEmpty())
2335 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(FPM)));
2336 }
2337
2338 invokeOptimizerEarlyEPCallbacks(MPM, Level, Phase);
2339
2340 if (!VectorizerStartEPCallbacks.empty()) {
2341 FunctionPassManager FPM;
2342 invokeVectorizerStartEPCallbacks(FPM, Level);
2343 if (!FPM.isEmpty())
2344 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(FPM)));
2345 }
2346
2347 if (!VectorizerEndEPCallbacks.empty()) {
2348 FunctionPassManager FPM;
2349 invokeVectorizerEndEPCallbacks(FPM, Level);
2350 if (!FPM.isEmpty())
2351 MPM.addPass(Pass: createModuleToFunctionPassAdaptor(Pass: std::move(FPM)));
2352 }
2353
2354 MPM.addPass(Pass: buildCoroWrapper(Phase));
2355
2356 // AllocToken transforms heap allocation calls; this needs to run late after
2357 // other allocation call transformations (such as those in InstCombine).
2358 if (!isLTOPreLink(Phase))
2359 MPM.addPass(Pass: AllocTokenPass());
2360
2361 invokeOptimizerLastEPCallbacks(MPM, Level, Phase);
2362
2363 if (Opts.enable_instrumentor)
2364 MPM.addPass(Pass: InstrumentorPass(FS));
2365
2366 // Attach !implicit.ref metadata from all functions to copyright strings.
2367 MPM.addPass(Pass: LowerCommentStringPass());
2368
2369 if (isLTOPreLink(Phase))
2370 addRequiredLTOPreLinkPasses(MPM);
2371
2372 // Emit annotation remarks.
2373 addAnnotationRemarksPass(MPM);
2374
2375 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2376
2377 return MPM;
2378}
2379
2380AAManager PassBuilder::buildDefaultAAPipeline() {
2381 AAManager AA;
2382
2383 // The order in which these are registered determines their priority when
2384 // being queried.
2385
2386 // Add any target-specific alias analyses that should be run early.
2387 if (TM)
2388 TM->registerEarlyDefaultAliasAnalyses(AA);
2389
2390 // First we register the basic alias analysis that provides the majority of
2391 // per-function local AA logic. This is a stateless, on-demand local set of
2392 // AA techniques.
2393 AA.registerFunctionAnalysis<BasicAA>();
2394
2395 // Next we query fast, specialized alias analyses that wrap IR-embedded
2396 // information about aliasing.
2397 AA.registerFunctionAnalysis<ScopedNoAliasAA>();
2398 AA.registerFunctionAnalysis<TypeBasedAA>();
2399
2400 // Add support for querying global aliasing information when available.
2401 // Because the `AAManager` is a function analysis and `GlobalsAA` is a module
2402 // analysis, all that the `AAManager` can do is query for any *cached*
2403 // results from `GlobalsAA` through a readonly proxy.
2404 if (Opts.enable_global_analyses)
2405 AA.registerModuleAnalysis<GlobalsAA>();
2406
2407 // Add target-specific alias analyses.
2408 if (TM)
2409 TM->registerDefaultAliasAnalyses(AA);
2410
2411 return AA;
2412}
2413
2414bool PassBuilder::isInstrumentedPGOUse() const {
2415 return (PGOOpt && PGOOpt->Action == PGOOptions::IRUse) ||
2416 !UseCtxProfile.empty();
2417}
2418