1//===-- AArch64CodeLayoutOpt.cpp - Code Layout Optimizations --===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This pass runs after instruction scheduling and employs code layout
10// optimizations for certain patterns.
11//
12// Option -aarch64-code-layout-opt-enable selects instruction pairs to optimize:
13// cmp-csel: Enable CMP/CMN-CSEL code layout optimization
14// fcmp-fcsel: Enable FCMP-FCSEL code layout optimization
15//
16// The initial implementation induces function alignment when a supported
17// pattern is detected, and possibly instruction-alignment when a pair would
18// straddle cache-lines.
19//===----------------------------------------------------------------------===//
20
21#include "AArch64.h"
22#include "AArch64InstrInfo.h"
23#include "AArch64Subtarget.h"
24#include "llvm/ADT/BitmaskEnum.h"
25#include "llvm/ADT/SmallVector.h"
26#include "llvm/ADT/Statistic.h"
27#include "llvm/CodeGen/MachineBasicBlock.h"
28#include "llvm/CodeGen/MachineFunctionPass.h"
29#include "llvm/Support/CommandLine.h"
30#include "llvm/Support/Debug.h"
31#include "llvm/Support/ErrorHandling.h"
32#include "llvm/Support/MathExtras.h"
33
34using namespace llvm;
35
36#define DEBUG_TYPE "aarch64-code-layout-opt"
37#define DBG(...) LLVM_DEBUG(dbgs() << DEBUG_TYPE ": " << __VA_ARGS__)
38#define AARCH64_CODE_LAYOUT_OPT_NAME "AArch64 Code Layout Optimization"
39
40enum CodeLayoutOpt {
41 None = 0,
42 CmpCsel = 1 << 0, // Align CMP/CMN-CSEL pairs
43 FcmpFcsel = 1 << 1, // Align FCMP-FCSEL pairs
44 LLVM_MARK_AS_BITMASK_ENUM(FcmpFcsel)
45};
46
47static cl::list<CodeLayoutOpt> EnableCodeAlignment(
48 "aarch64-code-layout-opt-enable", cl::Hidden, cl::CommaSeparated,
49 cl::desc("Enable code alignment optimization for instruction pairs"),
50 cl::values(
51 clEnumValN(None, "none", "Disable the code alignment pass"),
52 clEnumValN(CmpCsel, "cmp-csel", "CMP/CMN-CSEL pair alignment (32-bit)"),
53 clEnumValN(FcmpFcsel, "fcmp-fcsel", "FCMP-FCSEL pair alignment")));
54
55STATISTIC(NumFunctionsAligned,
56 "Number of functions with aligned (to 64-bytes by default)");
57STATISTIC(NumCmpCselPairsDetected,
58 "Number of CMP/CMN-CSEL pairs detected for alignment");
59STATISTIC(NumFcmpFcselPairsDetected,
60 "Number of FCMP-FCSEL pairs detected for alignment");
61
62namespace {
63
64class AArch64CodeLayoutOpt : public MachineFunctionPass {
65public:
66 static char ID;
67 AArch64CodeLayoutOpt() : MachineFunctionPass(ID) {}
68 void getAnalysisUsage(AnalysisUsage &AU) const override;
69 bool runOnMachineFunction(MachineFunction &MF) override;
70 StringRef getPassName() const override {
71 return AARCH64_CODE_LAYOUT_OPT_NAME;
72 }
73
74private:
75 const AArch64InstrInfo *TII = nullptr;
76
77 /// Align each fusible CMP/CMN-CSEL or FCMP-FCSEL pair in \p MBB by emitting
78 /// .p2align before the lead instruction (splitting the block if needed).
79 /// \returns true iff at least one pair was found and aligned.
80 bool alignLayoutSensitivePatterns(MachineBasicBlock *MBB, CodeLayoutOpt CLO);
81
82 /// Emit .p2align before MI. Splits the block if MI is not at its start.
83 void emitP2Align(MachineInstr &MI, Align DesiredAlign,
84 unsigned MaxSkipBytes = 4);
85
86 bool optimizeForCodeLayout(MachineFunction &MF, CodeLayoutOpt CLO);
87};
88
89} // end anonymous namespace
90
91char AArch64CodeLayoutOpt::ID = 0;
92
93INITIALIZE_PASS(AArch64CodeLayoutOpt, "aarch64-code-layout-opt",
94 AARCH64_CODE_LAYOUT_OPT_NAME, false, false)
95
96void AArch64CodeLayoutOpt::getAnalysisUsage(AnalysisUsage &AU) const {
97 AU.setPreservesAll();
98 MachineFunctionPass::getAnalysisUsage(AU);
99}
100
101FunctionPass *llvm::createAArch64CodeLayoutOptPass() {
102 return new AArch64CodeLayoutOpt();
103}
104
105/// \returns true iff Opc is a floating-point comparison (FCMP/FCMPE).
106static bool isFloatingPointCompare(unsigned Opc) {
107 switch (Opc) {
108 case AArch64::FCMPSrr:
109 case AArch64::FCMPDrr:
110 case AArch64::FCMPESrr:
111 case AArch64::FCMPEDrr:
112 case AArch64::FCMPHrr:
113 case AArch64::FCMPEHrr:
114 return true;
115 default:
116 return false;
117 }
118}
119
120/// \returns true iff Opc is a floating-point conditional select (FCSEL).
121static bool isFloatingPointConditionalSelect(unsigned Opc) {
122 switch (Opc) {
123 case AArch64::FCSELSrrr:
124 case AArch64::FCSELDrrr:
125 case AArch64::FCSELHrrr:
126 return true;
127 default:
128 return false;
129 }
130}
131
132/// \returns true if MI is a qualifying 32-bit CMP or CMN instruction.
133/// CMP is encoded as SUBS with WZR destination, CMN as ADDS with WZR.
134/// Only simple variants (no shifted/extended reg) qualify, and immediate
135/// variants require no LSL shift and small immediates (<=15).
136static bool isQualifyingIntCompare(const MachineInstr &MI) {
137 switch (MI.getOpcode()) {
138 case AArch64::SUBSWrr:
139 case AArch64::ADDSWrr:
140 return MI.definesRegister(Reg: AArch64::WZR, /*TRI=*/nullptr);
141 case AArch64::SUBSWri:
142 case AArch64::ADDSWri:
143 return MI.definesRegister(Reg: AArch64::WZR, /*TRI=*/nullptr) &&
144 MI.getOperand(i: 3).getImm() == 0 && MI.getOperand(i: 2).getImm() <= 15;
145 case AArch64::SUBSWrs:
146 case AArch64::ADDSWrs:
147 return MI.definesRegister(Reg: AArch64::WZR, /*TRI=*/nullptr) &&
148 !AArch64InstrInfo::hasShiftedReg(MI);
149 case AArch64::SUBSWrx:
150 return MI.definesRegister(Reg: AArch64::WZR, /*TRI=*/nullptr) &&
151 !AArch64InstrInfo::hasExtendedReg(MI);
152 default:
153 return false;
154 }
155}
156
157bool AArch64CodeLayoutOpt::runOnMachineFunction(MachineFunction &MF) {
158 const Function &F = MF.getFunction();
159 // hasOptSize() returns true for both -Os and -Oz.
160 if (F.hasOptSize())
161 return false;
162
163 const auto *Subtarget = &MF.getSubtarget<AArch64Subtarget>();
164
165 // Aligning basic blocks currently isn't compatible with Windows unwind info.
166 if (Subtarget->isTargetWindows())
167 return false;
168
169 TII = Subtarget->getInstrInfo();
170
171 CodeLayoutOpt CLO = None;
172 if (EnableCodeAlignment.getNumOccurrences()) {
173 if (is_contained(Range&: EnableCodeAlignment, Element: CodeLayoutOpt::CmpCsel))
174 CLO |= CodeLayoutOpt::CmpCsel;
175 if (is_contained(Range&: EnableCodeAlignment, Element: CodeLayoutOpt::FcmpFcsel))
176 CLO |= CodeLayoutOpt::FcmpFcsel;
177 } else {
178 // Default: enable when the subtarget opts in via FeatureAlignCmpCSelPairs.
179 if (Subtarget->hasAlignCmpCSelPairs()) {
180 if (Subtarget->hasFuseCmpCSel())
181 CLO |= CodeLayoutOpt::CmpCsel;
182 if (Subtarget->hasFuseFCmpFCSel())
183 CLO |= CodeLayoutOpt::FcmpFcsel;
184 }
185 }
186
187 if (CLO == None)
188 return false;
189
190 return optimizeForCodeLayout(MF, CLO);
191}
192
193void AArch64CodeLayoutOpt::emitP2Align(MachineInstr &MI, Align DesiredAlign,
194 unsigned MaxSkipBytes) {
195 MachineBasicBlock *MBB = MI.getParent();
196
197 auto FirstReal =
198 skipDebugInstructionsForward(It: MBB->instr_begin(), End: MBB->instr_end());
199 if (&*FirstReal != &MI) {
200 auto PrevIt = prev_nodbg(It: MI.getIterator(), Begin: MBB->instr_begin());
201 MBB = MBB->splitAt(SplitInst&: *PrevIt, /*UpdateLiveIns=*/true);
202 }
203
204 MBB->setAlignment(DesiredAlign);
205 MBB->setMaxBytesForAlignment(MaxSkipBytes);
206}
207
208// Align each fusible CMP/CMN-CSEL or FCMP-FCSEL pair in MBB by emitting
209// .p2align before the lead instruction (splitting the block if needed).
210// A pair is: a qualifying lead instruction immediately followed by its
211// consumer (CMP/CMN→CSEL or FCMP→FCSEL), with no intervening instructions.
212// Returns true iff at least one pair was found and aligned.
213bool AArch64CodeLayoutOpt::alignLayoutSensitivePatterns(MachineBasicBlock *MBB,
214 CodeLayoutOpt CLO) {
215 auto End = MBB->instr_end();
216 SmallVector<std::pair<MachineInstr *, bool>, 4> Pairs;
217
218 for (auto &MI : instructionsWithoutDebug(It: MBB->begin(), End: MBB->end())) {
219 auto NextIt =
220 skipDebugInstructionsForward(It: std::next(x: MI.getIterator()), End);
221 if (NextIt == End)
222 break;
223
224 // --- CMP/CMN-CSEL detection ---
225 if ((CLO & CodeLayoutOpt::CmpCsel) && isQualifyingIntCompare(MI) &&
226 NextIt->getOpcode() == AArch64::CSELWr) {
227 Pairs.push_back(Elt: {&MI, true});
228 continue;
229 }
230
231 // --- FCMP-FCSEL detection ---
232 if ((CLO & CodeLayoutOpt::FcmpFcsel) &&
233 isFloatingPointCompare(Opc: MI.getOpcode()) &&
234 isFloatingPointConditionalSelect(Opc: NextIt->getOpcode())) {
235 Pairs.push_back(Elt: {&MI, false});
236 continue;
237 }
238 }
239
240 for (auto &[MI, IsCmpCsel] : Pairs) {
241 emitP2Align(MI&: *MI, DesiredAlign: Align(64));
242 DBG(".p2align 6, , 4 before " << *MI);
243 ++(IsCmpCsel ? NumCmpCselPairsDetected : NumFcmpFcselPairsDetected);
244 }
245
246 return !Pairs.empty();
247}
248
249bool AArch64CodeLayoutOpt::optimizeForCodeLayout(MachineFunction &MF,
250 CodeLayoutOpt CLO) {
251 DBG("optimizeForCodeLayout: " << MF.getName() << "\n");
252
253 bool Changed = false;
254 for (auto &MBB : MF)
255 Changed |= alignLayoutSensitivePatterns(MBB: &MBB, CLO);
256
257 if (!Changed)
258 return false;
259
260 unsigned FunctionAlignBytes = MF.getSubtarget<AArch64Subtarget>()
261 .getCLOpts()
262 .code_layout_opt_align_functions;
263 if (!isPowerOf2_32(Value: FunctionAlignBytes))
264 reportFatalUsageError(
265 reason: "aarch64-code-layout-opt-align-functions must be a power of 2");
266 if (MF.getAlignment() < Align(FunctionAlignBytes)) {
267 MF.setAlignment(Align(FunctionAlignBytes));
268 ++NumFunctionsAligned;
269 DBG("Set " << FunctionAlignBytes << "-byte alignment for function "
270 << MF.getName() << "\n");
271 } else {
272 DBG("Function " << MF.getName() << " already has sufficient alignment\n");
273 }
274 return true;
275}
276