1//===-- RISCVFrameLowering.cpp - RISC-V Frame Information -----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains the RISC-V implementation of TargetFrameLowering class.
10//
11//===----------------------------------------------------------------------===//
12
13#include "RISCVFrameLowering.h"
14#include "MCTargetDesc/RISCVBaseInfo.h"
15#include "MCTargetDesc/RISCVMCTargetDesc.h"
16#include "RISCVMachineFunctionInfo.h"
17#include "RISCVSubtarget.h"
18#include "llvm/BinaryFormat/Dwarf.h"
19#include "llvm/CodeGen/CFIInstBuilder.h"
20#include "llvm/CodeGen/LivePhysRegs.h"
21#include "llvm/CodeGen/MachineFrameInfo.h"
22#include "llvm/CodeGen/MachineFunction.h"
23#include "llvm/CodeGen/MachineInstrBuilder.h"
24#include "llvm/CodeGen/MachineRegisterInfo.h"
25#include "llvm/CodeGen/RegisterScavenging.h"
26#include "llvm/CodeGen/TargetFrameLowering.h"
27#include "llvm/IR/DiagnosticInfo.h"
28#include "llvm/MC/MCDwarf.h"
29#include "llvm/Support/LEB128.h"
30
31#include <algorithm>
32#include <cstdint>
33
34#define DEBUG_TYPE "riscv-frame"
35
36using namespace llvm;
37
38static Align getABIStackAlignment(RISCVABI::ABI ABI) {
39 if (ABI == RISCVABI::ABI_ILP32E)
40 return Align(4);
41 if (ABI == RISCVABI::ABI_LP64E)
42 return Align(8);
43 return Align(16);
44}
45
46RISCVFrameLowering::RISCVFrameLowering(const RISCVSubtarget &STI)
47 : TargetFrameLowering(
48 StackGrowsDown, getABIStackAlignment(ABI: STI.getTargetABI()),
49 /*LocalAreaOffset=*/0,
50 /*TransientStackAlignment=*/getABIStackAlignment(ABI: STI.getTargetABI())),
51 STI(STI) {}
52
53// The register used to hold the frame pointer.
54static constexpr MCPhysReg FPReg = RISCV::X8;
55
56// The register used to hold the stack pointer.
57static constexpr MCPhysReg SPReg = RISCV::X2;
58
59// The register used to hold the return address.
60static constexpr MCPhysReg RAReg = RISCV::X1;
61
62// LIst of CSRs that are given a fixed location by save/restore libcalls or
63// Zcmp/Xqccmp Push/Pop. The order in this table indicates the order the
64// registers are saved on the stack. Zcmp uses the reverse order of save/restore
65// and Xqccmp on the stack, but this is handled when offsets are calculated.
66static const MCPhysReg FixedCSRFIMap[] = {
67 /*ra*/ RAReg, /*s0*/ FPReg, /*s1*/ RISCV::X9,
68 /*s2*/ RISCV::X18, /*s3*/ RISCV::X19, /*s4*/ RISCV::X20,
69 /*s5*/ RISCV::X21, /*s6*/ RISCV::X22, /*s7*/ RISCV::X23,
70 /*s8*/ RISCV::X24, /*s9*/ RISCV::X25, /*s10*/ RISCV::X26,
71 /*s11*/ RISCV::X27};
72
73// The number of stack bytes allocated by `QC.C.MIENTER(.NEST)` and popped by
74// `QC.C.MILEAVERET`.
75static constexpr uint64_t QCIInterruptPushAmount = 96;
76
77static const std::pair<MCPhysReg, int8_t> FixedCSRFIQCIInterruptMap[] = {
78 /* -1 is a gap for mepc/mnepc */
79 {/*fp*/ FPReg, -2},
80 /* -3 is a gap for qc.mcause */
81 {/*ra*/ RAReg, -4},
82 /* -5 is reserved */
83 {/*t0*/ RISCV::X5, -6},
84 {/*t1*/ RISCV::X6, -7},
85 {/*t2*/ RISCV::X7, -8},
86 {/*a0*/ RISCV::X10, -9},
87 {/*a1*/ RISCV::X11, -10},
88 {/*a2*/ RISCV::X12, -11},
89 {/*a3*/ RISCV::X13, -12},
90 {/*a4*/ RISCV::X14, -13},
91 {/*a5*/ RISCV::X15, -14},
92 {/*a6*/ RISCV::X16, -15},
93 {/*a7*/ RISCV::X17, -16},
94 {/*t3*/ RISCV::X28, -17},
95 {/*t4*/ RISCV::X29, -18},
96 {/*t5*/ RISCV::X30, -19},
97 {/*t6*/ RISCV::X31, -20},
98 /* -21, -22, -23, -24 are reserved */
99};
100
101/// Returns true if DWARF CFI instructions ("frame moves") should be emitted.
102static bool needsDwarfCFI(const MachineFunction &MF) {
103 return MF.needsFrameMoves();
104}
105
106// For now we use x3, a.k.a gp, as pointer to shadow call stack.
107// User should not use x3 in their asm.
108static void emitSCSPrologue(MachineFunction &MF, MachineBasicBlock &MBB,
109 MachineBasicBlock::iterator MI,
110 const DebugLoc &DL) {
111 const auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
112 RISCVMachineFunctionInfo::ShadowStackKind SSK = RVFI->getShadowStackKind(MF);
113 if (SSK == RISCVMachineFunctionInfo::ShadowStackKind::None)
114 return;
115
116 const auto &STI = MF.getSubtarget<RISCVSubtarget>();
117 const llvm::RISCVRegisterInfo *TRI = STI.getRegisterInfo();
118
119 // Do not save RA to the SCS if it's not saved to the regular stack,
120 // i.e. RA is not at risk of being overwritten.
121 std::vector<CalleeSavedInfo> &CSI = MF.getFrameInfo().getCalleeSavedInfo();
122 if (llvm::none_of(
123 Range&: CSI, P: [&](CalleeSavedInfo &CSR) { return CSR.getReg() == RAReg; }))
124 return;
125
126 const RISCVInstrInfo *TII = STI.getInstrInfo();
127 if (SSK == RISCVMachineFunctionInfo::ShadowStackKind::Hardware) {
128 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII->get(Opcode: RISCV::SSPUSH))
129 .addReg(RegNo: RAReg)
130 .setMIFlag(MachineInstr::FrameSetup);
131 return;
132 }
133
134 assert(SSK == RISCVMachineFunctionInfo::ShadowStackKind::Software &&
135 "Unexpected Shadow Stack Kind");
136
137 Register SCSPReg = RISCVABI::getSCSPReg();
138
139 bool IsRV64 = STI.is64Bit();
140 int64_t SlotSize = STI.getXLen() / 8;
141 // Store return address to shadow call stack
142 // addi gp, gp, [4|8]
143 // s[w|d] ra, -[4|8](gp)
144 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII->get(Opcode: RISCV::ADDI), DestReg: SCSPReg)
145 .addReg(RegNo: SCSPReg)
146 .addImm(Val: SlotSize)
147 .setMIFlag(MachineInstr::FrameSetup);
148 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII->get(Opcode: IsRV64 ? RISCV::SD : RISCV::SW))
149 .addReg(RegNo: RAReg)
150 .addReg(RegNo: SCSPReg)
151 .addImm(Val: -SlotSize)
152 .setMIFlag(MachineInstr::FrameSetup);
153
154 if (!needsDwarfCFI(MF))
155 return;
156
157 // Emit a CFI instruction that causes SlotSize to be subtracted from the value
158 // of the shadow stack pointer when unwinding past this frame.
159 char DwarfSCSReg = TRI->getDwarfRegNum(Reg: SCSPReg, /*IsEH*/ isEH: true);
160 assert(DwarfSCSReg < 32 && "SCS Register should be < 32 (X3).");
161
162 char Offset = static_cast<char>(-SlotSize) & 0x7f;
163 const char CFIInst[] = {
164 dwarf::DW_CFA_val_expression,
165 DwarfSCSReg, // register
166 2, // length
167 static_cast<char>(unsigned(dwarf::DW_OP_breg0 + DwarfSCSReg)),
168 Offset, // addend (sleb128)
169 };
170
171 CFIInstBuilder(MBB, MI, MachineInstr::FrameSetup)
172 .buildEscape(Bytes: StringRef(CFIInst, sizeof(CFIInst)));
173}
174
175static void emitSCSEpilogue(MachineFunction &MF, MachineBasicBlock &MBB,
176 MachineBasicBlock::iterator MI,
177 const DebugLoc &DL) {
178 const auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
179 RISCVMachineFunctionInfo::ShadowStackKind SSK = RVFI->getShadowStackKind(MF);
180 if (SSK == RISCVMachineFunctionInfo::ShadowStackKind::None)
181 return;
182
183 const auto &STI = MF.getSubtarget<RISCVSubtarget>();
184
185 // See emitSCSPrologue() above.
186 std::vector<CalleeSavedInfo> &CSI = MF.getFrameInfo().getCalleeSavedInfo();
187 if (llvm::none_of(
188 Range&: CSI, P: [&](CalleeSavedInfo &CSR) { return CSR.getReg() == RAReg; }))
189 return;
190
191 // The shadow call stack popchk needs to happen after cm.pop that loads ra.
192 if (MI != MBB.end() &&
193 (MI->getOpcode() == RISCV::CM_POP || MI->getOpcode() == RISCV::QC_CM_POP))
194 ++MI;
195 const RISCVInstrInfo *TII = STI.getInstrInfo();
196 if (SSK == RISCVMachineFunctionInfo::ShadowStackKind::Hardware) {
197 // `sspopchk x5` is the only compressible form, but this would require using
198 // `x5` as the return address everywhere, which would also mean reserving
199 // it. We prefer to just use `x1` to avoid that complexity.
200 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII->get(Opcode: RISCV::SSPOPCHK))
201 .addReg(RegNo: RAReg)
202 .setMIFlag(MachineInstr::FrameDestroy);
203 return;
204 }
205
206 assert(SSK == RISCVMachineFunctionInfo::ShadowStackKind::Software &&
207 "Unexpected Shadow Stack Kind");
208
209 Register SCSPReg = RISCVABI::getSCSPReg();
210
211 bool IsRV64 = STI.is64Bit();
212 int64_t SlotSize = STI.getXLen() / 8;
213 // Load return address from shadow call stack
214 // l[w|d] ra, -[4|8](gp)
215 // addi gp, gp, -[4|8]
216 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII->get(Opcode: IsRV64 ? RISCV::LD : RISCV::LW), DestReg: RAReg)
217 .addReg(RegNo: SCSPReg)
218 .addImm(Val: -SlotSize)
219 .setMIFlag(MachineInstr::FrameDestroy);
220 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII->get(Opcode: RISCV::ADDI), DestReg: SCSPReg)
221 .addReg(RegNo: SCSPReg)
222 .addImm(Val: -SlotSize)
223 .setMIFlag(MachineInstr::FrameDestroy);
224 if (needsDwarfCFI(MF)) {
225 // Restore the SCS pointer
226 CFIInstBuilder(MBB, MI, MachineInstr::FrameDestroy).buildRestore(Reg: SCSPReg);
227 }
228}
229
230// Insert instruction to swap mscratchsw with sp
231static void emitSiFiveCLICStackSwap(MachineFunction &MF, MachineBasicBlock &MBB,
232 MachineBasicBlock::iterator MBBI,
233 const DebugLoc &DL,
234 MachineInstr::MIFlag FrameFlag) {
235 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
236
237 if (!RVFI->isSiFiveStackSwapInterrupt(MF))
238 return;
239
240 const auto &STI = MF.getSubtarget<RISCVSubtarget>();
241 const RISCVInstrInfo *TII = STI.getInstrInfo();
242
243 assert(STI.hasVendorXSfmclic() && "Stack Swapping Requires XSfmclic");
244
245 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::CSRRW), DestReg: SPReg)
246 .addImm(Val: RISCVSysReg::sf_mscratchcsw)
247 .addReg(RegNo: SPReg, Flags: RegState::Kill)
248 .setMIFlag(FrameFlag);
249
250 // FIXME: CFI Information for this swap.
251}
252
253static void
254createSiFivePreemptibleInterruptFrameEntries(MachineFunction &MF,
255 RISCVMachineFunctionInfo &RVFI) {
256 if (!RVFI.isSiFivePreemptibleInterrupt(MF))
257 return;
258
259 const TargetRegisterClass &RC = RISCV::GPRRegClass;
260 const TargetRegisterInfo &TRI =
261 *MF.getSubtarget<RISCVSubtarget>().getRegisterInfo();
262 MachineFrameInfo &MFI = MF.getFrameInfo();
263
264 // Create two frame objects for saving `mcause` and `mepc`.
265 for (int I = 0; I < 2; ++I) {
266 int FI = MFI.CreateStackObject(Size: TRI.getSpillSize(RC), Alignment: TRI.getSpillAlign(RC),
267 isSpillSlot: true);
268 RVFI.pushInterruptCSRFrameIndex(FI);
269 }
270}
271
272// The scratch register retains an ordinary CSI slot, but its save and restore
273// are emitted explicitly as part of the SiFive CLIC interrupt sequence.
274static int getSiFiveCLICScratchFrameIndex(const MachineFunction &MF) {
275 const auto &CSI = MF.getFrameInfo().getCalleeSavedInfo();
276 auto ScratchCS = llvm::find_if(
277 Range: CSI, P: [](const CalleeSavedInfo &CS) { return CS.getReg() == RISCV::X5; });
278 assert(ScratchCS != CSI.end() && "Missing SiFive CLIC scratch spill slot");
279 return ScratchCS->getFrameIdx();
280}
281
282static void emitSiFiveCLICPreemptibleSaves(MachineFunction &MF,
283 MachineBasicBlock &MBB,
284 MachineBasicBlock::iterator MBBI,
285 const DebugLoc &DL) {
286 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
287
288 if (!RVFI->isSiFivePreemptibleInterrupt(MF))
289 return;
290
291 const auto &STI = MF.getSubtarget<RISCVSubtarget>();
292 const RISCVInstrInfo *TII = STI.getInstrInfo();
293
294 // FIXME: CFI information for `mcause` and `mepc` is missing.
295
296 // Preserve X5 before using it to save the interrupt CSRs. Other GPRs
297 // are saved by the ordinary spill sequence after preemption is enabled.
298 int ScratchFI = getSiFiveCLICScratchFrameIndex(MF);
299 TII->storeRegToStackSlot(MBB, MBBI, SrcReg: RISCV::X5, /*IsKill=*/true, FrameIndex: ScratchFI,
300 RC: &RISCV::GPRRegClass, VReg: Register(),
301 Flags: MachineInstr::FrameSetup);
302 if (needsDwarfCFI(MF))
303 CFIInstBuilder(MBB, MBBI, MachineInstr::FrameSetup)
304 .buildOffset(Reg: RISCV::X5, Offset: MF.getFrameInfo().getObjectOffset(ObjectIdx: ScratchFI));
305
306 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::CSRRS), DestReg: RISCV::X5)
307 .addImm(Val: RISCVSysReg::mcause)
308 .addReg(RegNo: RISCV::X0)
309 .setMIFlag(MachineInstr::FrameSetup);
310 TII->storeRegToStackSlot(MBB, MBBI, SrcReg: RISCV::X5, /* IsKill=*/true,
311 FrameIndex: RVFI->getInterruptCSRFrameIndex(Idx: 0),
312 RC: &RISCV::GPRRegClass, VReg: Register(),
313 Flags: MachineInstr::FrameSetup);
314
315 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::CSRRS), DestReg: RISCV::X5)
316 .addImm(Val: RISCVSysReg::mepc)
317 .addReg(RegNo: RISCV::X0)
318 .setMIFlag(MachineInstr::FrameSetup);
319
320 // Enable interrupts.
321 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::CSRRSI), DestReg: RISCV::X0)
322 .addImm(Val: RISCVSysReg::mstatus)
323 .addImm(Val: 8)
324 .setMIFlag(MachineInstr::FrameSetup);
325 TII->storeRegToStackSlot(MBB, MBBI, SrcReg: RISCV::X5, /* IsKill=*/true,
326 FrameIndex: RVFI->getInterruptCSRFrameIndex(Idx: 1),
327 RC: &RISCV::GPRRegClass, VReg: Register(),
328 Flags: MachineInstr::FrameSetup);
329}
330
331static void emitSiFiveCLICPreemptibleRestores(MachineFunction &MF,
332 MachineBasicBlock &MBB,
333 MachineBasicBlock::iterator MBBI,
334 CFIInstBuilder &CFIBuilder,
335 const DebugLoc &DL) {
336 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
337
338 if (!RVFI->isSiFivePreemptibleInterrupt(MF))
339 return;
340
341 const auto &STI = MF.getSubtarget<RISCVSubtarget>();
342 const RISCVInstrInfo *TII = STI.getInstrInfo();
343
344 // FIXME: CFI information for `mcause` and `mepc` is missing.
345
346 // Load mepc while preemption is still enabled. A nested handler preserves
347 // X5. Interrupts only need to be disabled before writing the CSRs back.
348 TII->loadRegFromStackSlot(MBB, MBBI, DstReg: RISCV::X5,
349 FrameIndex: RVFI->getInterruptCSRFrameIndex(Idx: 1),
350 RC: &RISCV::GPRRegClass, VReg: Register(),
351 SubReg: RISCV::NoSubRegister, Flags: MachineInstr::FrameDestroy);
352
353 // Disable interrupts.
354 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::CSRRCI), DestReg: RISCV::X0)
355 .addImm(Val: RISCVSysReg::mstatus)
356 .addImm(Val: 8)
357 .setMIFlag(MachineInstr::FrameDestroy);
358
359 // Restore `mepc` and `mcause` through X5, then restore the value X5 held
360 // on entry to the handler.
361 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::CSRRW), DestReg: RISCV::X0)
362 .addImm(Val: RISCVSysReg::mepc)
363 .addReg(RegNo: RISCV::X5, Flags: RegState::Kill)
364 .setMIFlag(MachineInstr::FrameDestroy);
365
366 TII->loadRegFromStackSlot(MBB, MBBI, DstReg: RISCV::X5,
367 FrameIndex: RVFI->getInterruptCSRFrameIndex(Idx: 0),
368 RC: &RISCV::GPRRegClass, VReg: Register(),
369 SubReg: RISCV::NoSubRegister, Flags: MachineInstr::FrameDestroy);
370 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::CSRRW), DestReg: RISCV::X0)
371 .addImm(Val: RISCVSysReg::mcause)
372 .addReg(RegNo: RISCV::X5, Flags: RegState::Kill)
373 .setMIFlag(MachineInstr::FrameDestroy);
374
375 // The ordinary reloads have finished. Recover the interrupted value of X5
376 // only after it has restored both CSRs.
377 TII->loadRegFromStackSlot(MBB, MBBI, DstReg: RISCV::X5,
378 FrameIndex: getSiFiveCLICScratchFrameIndex(MF),
379 RC: &RISCV::GPRRegClass, VReg: Register(),
380 SubReg: RISCV::NoSubRegister, Flags: MachineInstr::FrameDestroy);
381 if (needsDwarfCFI(MF))
382 CFIBuilder.buildRestore(Reg: RISCV::X5);
383}
384
385// Get the ID of the libcall used for spilling and restoring callee saved
386// registers. The ID is representative of the number of registers saved or
387// restored by the libcall, except it is zero-indexed - ID 0 corresponds to a
388// single register.
389static int getLibCallID(const MachineFunction &MF,
390 const std::vector<CalleeSavedInfo> &CSI) {
391 const auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
392
393 if (CSI.empty() || !RVFI->useSaveRestoreLibCalls(MF))
394 return -1;
395
396 MCRegister MaxReg;
397 for (auto &CS : CSI)
398 // assignCalleeSavedSpillSlots assigns negative frame indexes to
399 // registers which can be saved by libcall.
400 if (CS.getFrameIdx() < 0)
401 MaxReg = std::max(a: MaxReg.id(), b: CS.getReg().id());
402
403 if (!MaxReg)
404 return -1;
405
406 switch (MaxReg.id()) {
407 default:
408 llvm_unreachable("Something has gone wrong!");
409 // clang-format off
410 case /*s11*/ RISCV::X27: return 12;
411 case /*s10*/ RISCV::X26: return 11;
412 case /*s9*/ RISCV::X25: return 10;
413 case /*s8*/ RISCV::X24: return 9;
414 case /*s7*/ RISCV::X23: return 8;
415 case /*s6*/ RISCV::X22: return 7;
416 case /*s5*/ RISCV::X21: return 6;
417 case /*s4*/ RISCV::X20: return 5;
418 case /*s3*/ RISCV::X19: return 4;
419 case /*s2*/ RISCV::X18: return 3;
420 case /*s1*/ RISCV::X9: return 2;
421 case /*s0*/ FPReg: return 1;
422 case /*ra*/ RAReg: return 0;
423 // clang-format on
424 }
425}
426
427// Get the name of the libcall used for spilling callee saved registers.
428// If this function will not use save/restore libcalls, then return a nullptr.
429static const char *
430getSpillLibCallName(const MachineFunction &MF,
431 const std::vector<CalleeSavedInfo> &CSI) {
432 static const char *const SpillLibCalls[] = {
433 "__riscv_save_0",
434 "__riscv_save_1",
435 "__riscv_save_2",
436 "__riscv_save_3",
437 "__riscv_save_4",
438 "__riscv_save_5",
439 "__riscv_save_6",
440 "__riscv_save_7",
441 "__riscv_save_8",
442 "__riscv_save_9",
443 "__riscv_save_10",
444 "__riscv_save_11",
445 "__riscv_save_12"
446 };
447
448 int LibCallID = getLibCallID(MF, CSI);
449 if (LibCallID == -1)
450 return nullptr;
451 return SpillLibCalls[LibCallID];
452}
453
454// Get the name of the libcall used for restoring callee saved registers.
455// If this function will not use save/restore libcalls, then return a nullptr.
456static const char *
457getRestoreLibCallName(const MachineFunction &MF,
458 const std::vector<CalleeSavedInfo> &CSI) {
459 static const char *const RestoreLibCalls[] = {
460 "__riscv_restore_0",
461 "__riscv_restore_1",
462 "__riscv_restore_2",
463 "__riscv_restore_3",
464 "__riscv_restore_4",
465 "__riscv_restore_5",
466 "__riscv_restore_6",
467 "__riscv_restore_7",
468 "__riscv_restore_8",
469 "__riscv_restore_9",
470 "__riscv_restore_10",
471 "__riscv_restore_11",
472 "__riscv_restore_12"
473 };
474
475 int LibCallID = getLibCallID(MF, CSI);
476 if (LibCallID == -1)
477 return nullptr;
478 return RestoreLibCalls[LibCallID];
479}
480
481// Get the max reg of Push/Pop for restoring callee saved registers.
482static unsigned getNumPushPopRegs(const std::vector<CalleeSavedInfo> &CSI) {
483 unsigned NumPushPopRegs = 0;
484 for (auto &CS : CSI) {
485 auto *FII = llvm::find_if(Range: FixedCSRFIMap,
486 P: [&](MCPhysReg P) { return P == CS.getReg(); });
487 if (FII != std::end(arr: FixedCSRFIMap)) {
488 unsigned RegNum = std::distance(first: std::begin(arr: FixedCSRFIMap), last: FII);
489 NumPushPopRegs = std::max(a: NumPushPopRegs, b: RegNum + 1);
490 }
491 }
492 assert(NumPushPopRegs != 12 && "x26 requires x27 to also be pushed");
493 return NumPushPopRegs;
494}
495
496// Return true if the specified function should have a dedicated frame
497// pointer register. This is true if frame pointer elimination is
498// disabled, if it needs dynamic stack realignment, if the function has
499// variable sized allocas, or if the frame address is taken.
500bool RISCVFrameLowering::hasFPImpl(const MachineFunction &MF) const {
501 const TargetRegisterInfo *RegInfo = MF.getSubtarget().getRegisterInfo();
502
503 const MachineFrameInfo &MFI = MF.getFrameInfo();
504 if (MF.disableFramePointerElim() || RegInfo->hasStackRealignment(MF) ||
505 MFI.hasVarSizedObjects() || MFI.isFrameAddressTaken())
506 return true;
507
508 // With large callframes around we may need to use FP to access the scavenging
509 // emergency spillslot.
510 //
511 // We calculate the MaxCallFrameSize at the end of isel so this value should
512 // be stable for the whole post-isel MIR pipeline.
513 //
514 // NOTE: The idea of forcing a frame pointer is copied from AArch64, but they
515 // conservatively return true when the call frame size hasd not been
516 // computed yet. On RISC-V that caused MachineOutliner tests to fail the
517 // MachineVerifier due to outlined functions not computing max call frame
518 // size thus the frame pointer would always be reserved.
519 if (MFI.isMaxCallFrameSizeComputed() && MFI.getMaxCallFrameSize() > 2047)
520 return true;
521
522 return false;
523}
524
525bool RISCVFrameLowering::hasBP(const MachineFunction &MF) const {
526 const MachineFrameInfo &MFI = MF.getFrameInfo();
527 const TargetRegisterInfo *TRI = STI.getRegisterInfo();
528
529 // If we do not reserve stack space for outgoing arguments in prologue,
530 // we will adjust the stack pointer before call instruction. After the
531 // adjustment, we can not use SP to access the stack objects for the
532 // arguments. Instead, use BP to access these stack objects.
533 return (MFI.hasVarSizedObjects() ||
534 (!hasReservedCallFrame(MF) && (!MFI.isMaxCallFrameSizeComputed() ||
535 MFI.getMaxCallFrameSize() != 0))) &&
536 TRI->hasStackRealignment(MF);
537}
538
539// Determines the size of the frame and maximum call frame size.
540void RISCVFrameLowering::determineFrameLayout(MachineFunction &MF) const {
541 MachineFrameInfo &MFI = MF.getFrameInfo();
542 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
543
544 // Get the number of bytes to allocate from the FrameInfo.
545 uint64_t FrameSize = MFI.getStackSize();
546
547 // QCI Interrupts use at least 96 bytes of stack space
548 if (RVFI->useQCIInterrupt(MF))
549 FrameSize = std::max(a: FrameSize, b: QCIInterruptPushAmount);
550
551 // Get the alignment.
552 Align StackAlign = getStackAlign();
553
554 // Make sure the frame is aligned.
555 FrameSize = alignTo(Size: FrameSize, A: StackAlign);
556
557 // Update frame info.
558 MFI.setStackSize(FrameSize);
559
560 // When using SP or BP to access stack objects, we may require extra padding
561 // to ensure the bottom of the RVV stack is correctly aligned within the main
562 // stack. We calculate this as the amount required to align the scalar local
563 // variable section up to the RVV alignment.
564 const TargetRegisterInfo *TRI = STI.getRegisterInfo();
565 if (RVFI->getRVVStackSize() && (!hasFP(MF) || TRI->hasStackRealignment(MF))) {
566 int ScalarLocalVarSize = FrameSize - RVFI->getCalleeSavedStackSize() -
567 RVFI->getVarArgsSaveSize();
568 if (auto RVVPadding =
569 offsetToAlignment(Value: ScalarLocalVarSize, Alignment: RVFI->getRVVStackAlign()))
570 RVFI->setRVVPadding(RVVPadding);
571 }
572}
573
574// Returns the stack size including RVV padding (when required), rounded back
575// up to the required stack alignment.
576uint64_t RISCVFrameLowering::getStackSizeWithRVVPadding(
577 const MachineFunction &MF) const {
578 const MachineFrameInfo &MFI = MF.getFrameInfo();
579 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
580 return alignTo(Size: MFI.getStackSize() + RVFI->getRVVPadding(), A: getStackAlign());
581}
582
583static SmallVector<CalleeSavedInfo, 8>
584getUnmanagedCSI(const MachineFunction &MF,
585 const std::vector<CalleeSavedInfo> &CSI,
586 bool ReverseOrder = false) {
587 const MachineFrameInfo &MFI = MF.getFrameInfo();
588 SmallVector<CalleeSavedInfo, 8> NonLibcallCSI;
589
590 for (auto &CS : CSI) {
591 int FI = CS.getFrameIdx();
592 if (FI >= 0 && MFI.getStackID(ObjectIdx: FI) == TargetStackID::Default)
593 NonLibcallCSI.push_back(Elt: CS);
594 }
595
596 // Reverse the order so that load/store operations use ascending addresses,
597 // enabling better load/store clustering and fusion.
598 if (ReverseOrder)
599 std::reverse(first: NonLibcallCSI.begin(), last: NonLibcallCSI.end());
600
601 return NonLibcallCSI;
602}
603
604// Exclude X5 from ordinary spills and restores for SiFive CLIC preemptible
605// handlers, which save and restore it explicitly.
606static SmallVector<CalleeSavedInfo, 8>
607getUnmanagedInterruptCSI(const MachineFunction &MF,
608 const std::vector<CalleeSavedInfo> &CSI,
609 bool ReverseOrder = false) {
610 auto InterruptCSI = getUnmanagedCSI(MF, CSI, ReverseOrder);
611 if (MF.getInfo<RISCVMachineFunctionInfo>()->isSiFivePreemptibleInterrupt(MF))
612 llvm::erase_if(C&: InterruptCSI, P: [](const CalleeSavedInfo &CS) {
613 return CS.getReg() == RISCV::X5;
614 });
615 return InterruptCSI;
616}
617
618static SmallVector<CalleeSavedInfo, 8>
619getRVVCalleeSavedInfo(const MachineFunction &MF,
620 const std::vector<CalleeSavedInfo> &CSI) {
621 const MachineFrameInfo &MFI = MF.getFrameInfo();
622 SmallVector<CalleeSavedInfo, 8> RVVCSI;
623
624 for (auto &CS : CSI) {
625 int FI = CS.getFrameIdx();
626 if (FI >= 0 && MFI.getStackID(ObjectIdx: FI) == TargetStackID::ScalableVector)
627 RVVCSI.push_back(Elt: CS);
628 }
629
630 return RVVCSI;
631}
632
633static SmallVector<CalleeSavedInfo, 8>
634getPushOrLibCallsSavedInfo(const MachineFunction &MF,
635 const std::vector<CalleeSavedInfo> &CSI) {
636 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
637
638 SmallVector<CalleeSavedInfo, 8> PushOrLibCallsCSI;
639 if (!RVFI->useSaveRestoreLibCalls(MF) && !RVFI->isPushable(MF))
640 return PushOrLibCallsCSI;
641
642 for (const auto &CS : CSI) {
643 if (RVFI->useQCIInterrupt(MF)) {
644 // Some registers are saved by both `QC.C.MIENTER(.NEST)` and
645 // `QC.CM.PUSH(FP)`. In these cases, prioritise the CFI info that points
646 // to the versions saved by `QC.C.MIENTER(.NEST)` which is what FP
647 // unwinding would use.
648 if (llvm::is_contained(Range: llvm::make_first_range(c: FixedCSRFIQCIInterruptMap),
649 Element: CS.getReg()))
650 continue;
651 }
652
653 if (llvm::is_contained(Range: FixedCSRFIMap, Element: CS.getReg()))
654 PushOrLibCallsCSI.push_back(Elt: CS);
655 }
656
657 return PushOrLibCallsCSI;
658}
659
660static SmallVector<CalleeSavedInfo, 8>
661getQCISavedInfo(const MachineFunction &MF,
662 const std::vector<CalleeSavedInfo> &CSI) {
663 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
664
665 SmallVector<CalleeSavedInfo, 8> QCIInterruptCSI;
666 if (!RVFI->useQCIInterrupt(MF))
667 return QCIInterruptCSI;
668
669 for (const auto &CS : CSI) {
670 if (llvm::is_contained(Range: llvm::make_first_range(c: FixedCSRFIQCIInterruptMap),
671 Element: CS.getReg()))
672 QCIInterruptCSI.push_back(Elt: CS);
673 }
674
675 return QCIInterruptCSI;
676}
677
678static void getLiveRegsForEntryMBB(LivePhysRegs &LiveRegs,
679 const MachineBasicBlock &MBB) {
680 const MachineFunction *MF = MBB.getParent();
681 LiveRegs.addLiveIns(MBB);
682 const MCPhysReg *CSRegs = MF->getRegInfo().getCalleeSavedRegs();
683 for (unsigned i = 0; CSRegs[i]; ++i)
684 LiveRegs.addReg(Reg: CSRegs[i]);
685}
686
687Register RISCVFrameLowering::findScratchNonCalleeSaveRegister(
688 MachineBasicBlock *MBB, Register PreferredReg, Register DontUseReg) const {
689 MachineFunction *MF = MBB->getParent();
690
691 // Stack protection code is being inserted at beginning of function, use
692 // register which has been historically used
693 if (&MF->front() == MBB)
694 return PreferredReg;
695
696 const RISCVSubtarget &Subtarget = MF->getSubtarget<RISCVSubtarget>();
697 const TargetRegisterInfo &TRI = *Subtarget.getRegisterInfo();
698 LivePhysRegs LiveRegs(TRI);
699 getLiveRegsForEntryMBB(LiveRegs, MBB: *MBB);
700
701 const MachineRegisterInfo &MRI = MF->getRegInfo();
702 // Prefer the register which has been historically used for stack protector
703 if (LiveRegs.available(MRI, Reg: PreferredReg))
704 return PreferredReg;
705
706 static const MCPhysReg CandidateRegs[] = {
707 RISCV::X5, RISCV::X6, RISCV::X7, RISCV::X28,
708 RISCV::X29, RISCV::X30, RISCV::X31,
709 };
710
711 for (unsigned Reg : CandidateRegs) {
712 if (Reg != DontUseReg && LiveRegs.available(MRI, Reg))
713 return Reg;
714 }
715
716 return Register();
717}
718
719void RISCVFrameLowering::allocateAndProbeStackForRVV(
720 MachineFunction &MF, MachineBasicBlock &MBB,
721 MachineBasicBlock::iterator MBBI, const DebugLoc &DL, int64_t Amount,
722 MachineInstr::MIFlag Flag, bool EmitCFI, bool DynAllocation) const {
723 assert(Amount != 0 && "Did not need to adjust stack pointer for RVV.");
724
725 // Emit a variable-length allocation probing loop.
726
727 // Get VLEN in TargetReg
728 Register TargetReg = findScratchNonCalleeSaveRegister(MBB: &MBB, PreferredReg: RISCV::X6);
729 assert(TargetReg.isValid() &&
730 "No available scratch register for stack probing");
731 const RISCVInstrInfo *TII = STI.getInstrInfo();
732 uint32_t NumOfVReg = Amount / RISCV::RVVBytesPerBlock;
733 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::PseudoReadVLENB), DestReg: TargetReg)
734 .setMIFlag(Flag);
735 TII->mulImm(MF, MBB, II: MBBI, DL, DestReg: TargetReg, Amt: NumOfVReg, Flag);
736
737 CFIInstBuilder CFIBuilder(MBB, MBBI, MachineInstr::FrameSetup);
738 if (EmitCFI) {
739 // Set the CFA register to TargetReg.
740 CFIBuilder.buildDefCFA(Reg: TargetReg, Offset: -Amount);
741 }
742
743 // It will be expanded to a probe loop in `inlineStackProbe`.
744 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::PROBED_STACKALLOC_RVV))
745 .addReg(RegNo: TargetReg);
746
747 if (EmitCFI) {
748 // Set the CFA register back to SP.
749 CFIBuilder.buildDefCFARegister(Reg: SPReg);
750 }
751
752 // SUB SP, SP, T1
753 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::SUB), DestReg: SPReg)
754 .addReg(RegNo: SPReg)
755 .addReg(RegNo: TargetReg)
756 .setMIFlag(Flag);
757
758 // If we have a dynamic allocation later we need to probe any residuals.
759 if (DynAllocation) {
760 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: STI.is64Bit() ? RISCV::SD : RISCV::SW))
761 .addReg(RegNo: RISCV::X0)
762 .addReg(RegNo: SPReg)
763 .addImm(Val: 0)
764 .setMIFlags(MachineInstr::FrameSetup);
765 }
766}
767
768static void appendScalableVectorExpression(const TargetRegisterInfo &TRI,
769 SmallVectorImpl<char> &Expr,
770 StackOffset Offset,
771 llvm::raw_string_ostream &Comment) {
772 int64_t FixedOffset = Offset.getFixed();
773 int64_t ScalableOffset = Offset.getScalable();
774 unsigned DwarfVLenB = TRI.getDwarfRegNum(Reg: RISCV::VLENB, isEH: true);
775 if (FixedOffset) {
776 Expr.push_back(Elt: dwarf::DW_OP_consts);
777 appendLEB128<LEB128Sign::Signed>(Buffer&: Expr, Value: FixedOffset);
778 Expr.push_back(Elt: (uint8_t)dwarf::DW_OP_plus);
779 Comment << (FixedOffset < 0 ? " - " : " + ") << std::abs(i: FixedOffset);
780 }
781
782 Expr.push_back(Elt: (uint8_t)dwarf::DW_OP_consts);
783 appendLEB128<LEB128Sign::Signed>(Buffer&: Expr, Value: ScalableOffset);
784
785 Expr.push_back(Elt: (uint8_t)dwarf::DW_OP_bregx);
786 appendLEB128<LEB128Sign::Unsigned>(Buffer&: Expr, Value: DwarfVLenB);
787 Expr.push_back(Elt: 0);
788
789 Expr.push_back(Elt: (uint8_t)dwarf::DW_OP_mul);
790 Expr.push_back(Elt: (uint8_t)dwarf::DW_OP_plus);
791
792 Comment << (ScalableOffset < 0 ? " - " : " + ") << std::abs(i: ScalableOffset)
793 << " * vlenb";
794}
795
796static MCCFIInstruction createDefCFAExpression(const TargetRegisterInfo &TRI,
797 Register Reg,
798 StackOffset Offset) {
799 assert(Offset.getScalable() != 0 && "Did not need to adjust CFA for RVV");
800 SmallString<64> Expr;
801 std::string CommentBuffer;
802 llvm::raw_string_ostream Comment(CommentBuffer);
803 // Build up the expression (Reg + FixedOffset + ScalableOffset * VLENB).
804 unsigned DwarfReg = TRI.getDwarfRegNum(Reg, isEH: true);
805 Expr.push_back(Elt: (uint8_t)(dwarf::DW_OP_breg0 + DwarfReg));
806 Expr.push_back(Elt: 0);
807 if (Reg == SPReg)
808 Comment << "sp";
809 else
810 Comment << printReg(Reg, TRI: &TRI);
811
812 appendScalableVectorExpression(TRI, Expr, Offset, Comment);
813
814 SmallString<64> DefCfaExpr;
815 DefCfaExpr.push_back(Elt: dwarf::DW_CFA_def_cfa_expression);
816 appendLEB128<LEB128Sign::Unsigned>(Buffer&: DefCfaExpr, Value: Expr.size());
817 DefCfaExpr.append(RHS: Expr.str());
818
819 return MCCFIInstruction::createEscape(L: nullptr, Vals: DefCfaExpr.str(), Loc: SMLoc(),
820 Comment: Comment.str());
821}
822
823static MCCFIInstruction createDefCFAOffset(const TargetRegisterInfo &TRI,
824 Register Reg, StackOffset Offset) {
825 assert(Offset.getScalable() != 0 && "Did not need to adjust CFA for RVV");
826 SmallString<64> Expr;
827 std::string CommentBuffer;
828 llvm::raw_string_ostream Comment(CommentBuffer);
829 Comment << printReg(Reg, TRI: &TRI) << " @ cfa";
830
831 // Build up the expression (FixedOffset + ScalableOffset * VLENB).
832 appendScalableVectorExpression(TRI, Expr, Offset, Comment);
833
834 SmallString<64> DefCfaExpr;
835 unsigned DwarfReg = TRI.getDwarfRegNum(Reg, isEH: true);
836 DefCfaExpr.push_back(Elt: dwarf::DW_CFA_expression);
837 appendLEB128<LEB128Sign::Unsigned>(Buffer&: DefCfaExpr, Value: DwarfReg);
838 appendLEB128<LEB128Sign::Unsigned>(Buffer&: DefCfaExpr, Value: Expr.size());
839 DefCfaExpr.append(RHS: Expr.str());
840
841 return MCCFIInstruction::createEscape(L: nullptr, Vals: DefCfaExpr.str(), Loc: SMLoc(),
842 Comment: Comment.str());
843}
844
845// Allocate stack space and probe it if necessary.
846void RISCVFrameLowering::allocateStack(MachineBasicBlock &MBB,
847 MachineBasicBlock::iterator MBBI,
848 MachineFunction &MF, uint64_t Offset,
849 uint64_t RealStackSize, bool EmitCFI,
850 bool NeedProbe, uint64_t ProbeSize,
851 bool DynAllocation,
852 MachineInstr::MIFlag Flag) const {
853 DebugLoc DL;
854 const RISCVRegisterInfo *RI = STI.getRegisterInfo();
855 const RISCVInstrInfo *TII = STI.getInstrInfo();
856 bool IsRV64 = STI.is64Bit();
857 CFIInstBuilder CFIBuilder(MBB, MBBI, MachineInstr::FrameSetup);
858
859 // Simply allocate the stack if it's not big enough to require a probe.
860 if (!NeedProbe || Offset <= ProbeSize) {
861 RI->adjustReg(MBB, II: MBBI, DL, DestReg: SPReg, SrcReg: SPReg, Offset: StackOffset::getFixed(Fixed: -Offset),
862 Flag, RequiredAlign: getStackAlign());
863
864 if (EmitCFI)
865 CFIBuilder.buildDefCFAOffset(Offset: RealStackSize);
866
867 if (NeedProbe && DynAllocation) {
868 // s[d|w] zero, 0(sp)
869 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: IsRV64 ? RISCV::SD : RISCV::SW))
870 .addReg(RegNo: RISCV::X0)
871 .addReg(RegNo: SPReg)
872 .addImm(Val: 0)
873 .setMIFlags(Flag);
874 }
875
876 return;
877 }
878
879 // The amount of stack that was already allocated before this allocation.
880 uint64_t CFAAdjust = RealStackSize - Offset;
881
882 // Unroll the probe loop depending on the number of iterations.
883 if (Offset < ProbeSize * 5) {
884 uint64_t CurrentOffset = 0;
885 while (CurrentOffset + ProbeSize <= Offset) {
886 RI->adjustReg(MBB, II: MBBI, DL, DestReg: SPReg, SrcReg: SPReg,
887 Offset: StackOffset::getFixed(Fixed: -ProbeSize), Flag, RequiredAlign: getStackAlign());
888 // s[d|w] zero, 0(sp)
889 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: IsRV64 ? RISCV::SD : RISCV::SW))
890 .addReg(RegNo: RISCV::X0)
891 .addReg(RegNo: SPReg)
892 .addImm(Val: 0)
893 .setMIFlags(Flag);
894
895 CurrentOffset += ProbeSize;
896 if (EmitCFI)
897 CFIBuilder.buildDefCFAOffset(Offset: CurrentOffset + CFAAdjust);
898 }
899
900 uint64_t Residual = Offset - CurrentOffset;
901 if (Residual) {
902 RI->adjustReg(MBB, II: MBBI, DL, DestReg: SPReg, SrcReg: SPReg,
903 Offset: StackOffset::getFixed(Fixed: -Residual), Flag, RequiredAlign: getStackAlign());
904 if (EmitCFI)
905 CFIBuilder.buildDefCFAOffset(Offset: RealStackSize);
906
907 if (DynAllocation) {
908 // s[d|w] zero, 0(sp)
909 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: IsRV64 ? RISCV::SD : RISCV::SW))
910 .addReg(RegNo: RISCV::X0)
911 .addReg(RegNo: SPReg)
912 .addImm(Val: 0)
913 .setMIFlags(Flag);
914 }
915 }
916
917 return;
918 }
919
920 // Emit a variable-length allocation probing loop.
921 uint64_t RoundedSize = alignDown(Value: Offset, Align: ProbeSize);
922 uint64_t Residual = Offset - RoundedSize;
923
924 Register TargetReg = findScratchNonCalleeSaveRegister(MBB: &MBB, PreferredReg: RISCV::X6);
925 assert(TargetReg.isValid() &&
926 "No available scratch register for stack probing");
927 // SUB TargetReg, SP, RoundedSize
928 RI->adjustReg(MBB, II: MBBI, DL, DestReg: TargetReg, SrcReg: SPReg,
929 Offset: StackOffset::getFixed(Fixed: -RoundedSize), Flag, RequiredAlign: getStackAlign());
930
931 if (EmitCFI) {
932 // Set the CFA register to TargetReg.
933 CFIBuilder.buildDefCFA(Reg: TargetReg, Offset: RoundedSize + CFAAdjust);
934 }
935
936 // It will be expanded to a probe loop in `inlineStackProbe`.
937 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::PROBED_STACKALLOC)).addReg(RegNo: TargetReg);
938
939 if (EmitCFI) {
940 // Set the CFA register back to SP.
941 CFIBuilder.buildDefCFARegister(Reg: SPReg);
942 }
943
944 if (Residual) {
945 RI->adjustReg(MBB, II: MBBI, DL, DestReg: SPReg, SrcReg: SPReg, Offset: StackOffset::getFixed(Fixed: -Residual),
946 Flag, RequiredAlign: getStackAlign());
947 if (DynAllocation) {
948 // s[d|w] zero, 0(sp)
949 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: IsRV64 ? RISCV::SD : RISCV::SW))
950 .addReg(RegNo: RISCV::X0)
951 .addReg(RegNo: SPReg)
952 .addImm(Val: 0)
953 .setMIFlags(Flag);
954 }
955 }
956
957 if (EmitCFI)
958 CFIBuilder.buildDefCFAOffset(Offset: RealStackSize);
959}
960
961static bool isPush(unsigned Opcode) {
962 switch (Opcode) {
963 case RISCV::CM_PUSH:
964 case RISCV::QC_CM_PUSH:
965 case RISCV::QC_CM_PUSHFP:
966 return true;
967 default:
968 return false;
969 }
970}
971
972static bool isPop(unsigned Opcode) {
973 // There are other pops but these are the only ones introduced during this
974 // pass.
975 switch (Opcode) {
976 case RISCV::CM_POP:
977 case RISCV::QC_CM_POP:
978 return true;
979 default:
980 return false;
981 }
982}
983
984static unsigned getPushOpcode(RISCVMachineFunctionInfo::PushPopKind Kind,
985 bool UpdateFP) {
986 switch (Kind) {
987 case RISCVMachineFunctionInfo::PushPopKind::StdExtZcmp:
988 return RISCV::CM_PUSH;
989 case RISCVMachineFunctionInfo::PushPopKind::VendorXqccmp:
990 return UpdateFP ? RISCV::QC_CM_PUSHFP : RISCV::QC_CM_PUSH;
991 default:
992 llvm_unreachable("Unhandled PushPopKind");
993 }
994}
995
996static unsigned getPopOpcode(RISCVMachineFunctionInfo::PushPopKind Kind) {
997 // There are other pops but they are introduced later by the Push/Pop
998 // Optimizer.
999 switch (Kind) {
1000 case RISCVMachineFunctionInfo::PushPopKind::StdExtZcmp:
1001 return RISCV::CM_POP;
1002 case RISCVMachineFunctionInfo::PushPopKind::VendorXqccmp:
1003 return RISCV::QC_CM_POP;
1004 default:
1005 llvm_unreachable("Unhandled PushPopKind");
1006 }
1007}
1008
1009void RISCVFrameLowering::emitPrologue(MachineFunction &MF,
1010 MachineBasicBlock &MBB) const {
1011 MachineFrameInfo &MFI = MF.getFrameInfo();
1012 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
1013 const RISCVRegisterInfo *RI = STI.getRegisterInfo();
1014 MachineBasicBlock::iterator MBBI = MBB.begin();
1015 bool PreferAscendingLS = STI.preferAscendingLoadStore();
1016
1017 Register BPReg = RISCVABI::getBPReg();
1018
1019 // Debug location must be unknown since the first debug location is used
1020 // to determine the end of the prologue.
1021 DebugLoc DL;
1022
1023 // All calls are tail calls in GHC calling conv, and functions have no
1024 // prologue/epilogue.
1025 if (MF.getFunction().getCallingConv() == CallingConv::GHC)
1026 return;
1027
1028 // SiFive CLIC needs to swap `sp` into `sf.mscratchcsw`
1029 emitSiFiveCLICStackSwap(MF, MBB, MBBI, DL, FrameFlag: MachineInstr::FrameSetup);
1030
1031 // Emit prologue for shadow call stack.
1032 emitSCSPrologue(MF, MBB, MI: MBBI, DL);
1033
1034 // We keep track of the first instruction because it might be a
1035 // `(QC.)CM.PUSH(FP)`, and we may need to adjust the immediate rather than
1036 // inserting an `addi sp, sp, -N*16`
1037 auto PossiblePush = MBBI;
1038
1039 // Skip past all callee-saved register spill instructions.
1040 while (MBBI != MBB.end() && MBBI->getFlag(Flag: MachineInstr::FrameSetup))
1041 ++MBBI;
1042
1043 // Determine the correct frame layout
1044 determineFrameLayout(MF);
1045
1046 const auto &CSI = MFI.getCalleeSavedInfo();
1047
1048 // Skip to before the spills of scalar callee-saved registers
1049 // FIXME: assumes exactly one instruction is used to restore each
1050 // callee-saved register.
1051 MBBI = std::prev(
1052 x: MBBI, n: getRVVCalleeSavedInfo(MF, CSI).size() +
1053 getUnmanagedInterruptCSI(MF, CSI, ReverseOrder: PreferAscendingLS).size());
1054 CFIInstBuilder CFIBuilder(MBB, MBBI, MachineInstr::FrameSetup);
1055 bool NeedsDwarfCFI = needsDwarfCFI(MF);
1056
1057 // If libcalls are used to spill and restore callee-saved registers, the frame
1058 // has two sections; the opaque section managed by the libcalls, and the
1059 // section managed by MachineFrameInfo which can also hold callee saved
1060 // registers in fixed stack slots, both of which have negative frame indices.
1061 // This gets even more complicated when incoming arguments are passed via the
1062 // stack, as these too have negative frame indices. An example is detailed
1063 // below:
1064 //
1065 // | incoming arg | <- FI[-3]
1066 // | libcallspill |
1067 // | calleespill | <- FI[-2]
1068 // | calleespill | <- FI[-1]
1069 // | this_frame | <- FI[0]
1070 //
1071 // For negative frame indices, the offset from the frame pointer will differ
1072 // depending on which of these groups the frame index applies to.
1073 // The following calculates the correct offset knowing the number of callee
1074 // saved registers spilt by the two methods.
1075 if (int LibCallRegs = getLibCallID(MF, CSI: MFI.getCalleeSavedInfo()) + 1) {
1076 // Calculate the size of the frame managed by the libcall. The stack
1077 // alignment of these libcalls should be the same as how we set it in
1078 // getABIStackAlignment.
1079 unsigned LibCallFrameSize =
1080 alignTo(Size: (STI.getXLen() / 8) * LibCallRegs, A: getStackAlign());
1081 RVFI->setLibCallStackSize(LibCallFrameSize);
1082
1083 if (NeedsDwarfCFI) {
1084 CFIBuilder.buildDefCFAOffset(Offset: LibCallFrameSize);
1085 for (const CalleeSavedInfo &CS : getPushOrLibCallsSavedInfo(MF, CSI))
1086 CFIBuilder.buildOffset(Reg: CS.getReg(),
1087 Offset: MFI.getObjectOffset(ObjectIdx: CS.getFrameIdx()));
1088 }
1089 }
1090
1091 // FIXME (note copied from Lanai): This appears to be overallocating. Needs
1092 // investigation. Get the number of bytes to allocate from the FrameInfo.
1093 uint64_t RealStackSize = getStackSizeWithRVVPadding(MF);
1094 uint64_t StackSize = RealStackSize - RVFI->getReservedSpillsSize();
1095 uint64_t RVVStackSize = RVFI->getRVVStackSize();
1096
1097 // Early exit if there is no need to allocate on the stack
1098 if (RealStackSize == 0 && !MFI.adjustsStack() && RVVStackSize == 0)
1099 return;
1100
1101 // If the stack pointer has been marked as reserved, then produce an error if
1102 // the frame requires stack allocation
1103 if (STI.isRegisterReservedByUser(i: SPReg))
1104 MF.getFunction().getContext().diagnose(DI: DiagnosticInfoUnsupported{
1105 MF.getFunction(), "Stack pointer required, but has been reserved."});
1106
1107 uint64_t FirstSPAdjustAmount = getFirstSPAdjustAmount(MF);
1108 // Split the SP adjustment to reduce the offsets of callee saved spill.
1109 if (FirstSPAdjustAmount) {
1110 StackSize = FirstSPAdjustAmount;
1111 RealStackSize = FirstSPAdjustAmount;
1112 }
1113
1114 if (RVFI->useQCIInterrupt(MF)) {
1115 // The function starts with `QC.C.MIENTER(.NEST)`, so the `(QC.)CM.PUSH(FP)`
1116 // could only be the next instruction.
1117 ++PossiblePush;
1118
1119 if (NeedsDwarfCFI) {
1120 // Insert the CFI metadata before where we think the `(QC.)CM.PUSH(FP)`
1121 // could be. The PUSH will also get its own CFI metadata for its own
1122 // modifications, which should come after the PUSH.
1123 CFIInstBuilder PushCFIBuilder(MBB, PossiblePush,
1124 MachineInstr::FrameSetup);
1125 PushCFIBuilder.buildDefCFAOffset(Offset: QCIInterruptPushAmount);
1126 for (const CalleeSavedInfo &CS : getQCISavedInfo(MF, CSI))
1127 PushCFIBuilder.buildOffset(Reg: CS.getReg(),
1128 Offset: MFI.getObjectOffset(ObjectIdx: CS.getFrameIdx()));
1129 }
1130 }
1131
1132 if (RVFI->isPushable(MF) && PossiblePush != MBB.end() &&
1133 isPush(Opcode: PossiblePush->getOpcode())) {
1134 // Use available stack adjustment in push instruction to allocate additional
1135 // stack space. Align the stack size down to a multiple of 16. This is
1136 // needed for RVE.
1137 // FIXME: Can we increase the stack size to a multiple of 16 instead?
1138 uint64_t StackAdj =
1139 std::min(a: alignDown(Value: StackSize, Align: 16), b: static_cast<uint64_t>(48));
1140 PossiblePush->getOperand(i: 1).setImm(StackAdj);
1141 StackSize -= StackAdj;
1142
1143 if (NeedsDwarfCFI) {
1144 CFIBuilder.buildDefCFAOffset(Offset: RealStackSize - StackSize);
1145 for (const CalleeSavedInfo &CS : getPushOrLibCallsSavedInfo(MF, CSI))
1146 CFIBuilder.buildOffset(Reg: CS.getReg(),
1147 Offset: MFI.getObjectOffset(ObjectIdx: CS.getFrameIdx()));
1148 }
1149 }
1150
1151 // Allocate space on the stack if necessary.
1152 auto &Subtarget = MF.getSubtarget<RISCVSubtarget>();
1153 const RISCVTargetLowering *TLI = Subtarget.getTargetLowering();
1154 bool NeedProbe = TLI->hasInlineStackProbe(MF);
1155 uint64_t ProbeSize = TLI->getStackProbeSize(MF, StackAlign: getStackAlign());
1156 bool DynAllocation =
1157 MF.getInfo<RISCVMachineFunctionInfo>()->hasDynamicAllocation();
1158 if (StackSize != 0)
1159 allocateStack(MBB, MBBI, MF, Offset: StackSize, RealStackSize, EmitCFI: NeedsDwarfCFI,
1160 NeedProbe, ProbeSize, DynAllocation,
1161 Flag: MachineInstr::FrameSetup);
1162
1163 // Save SiFive CLIC CSRs into Stack
1164 emitSiFiveCLICPreemptibleSaves(MF, MBB, MBBI, DL);
1165
1166 // The frame pointer is callee-saved, and code has been generated for us to
1167 // save it to the stack. We need to skip over the storing of callee-saved
1168 // registers as the frame pointer must be modified after it has been saved
1169 // to the stack, not before.
1170 // FIXME: assumes exactly one instruction is used to save each callee-saved
1171 // register.
1172 std::advance(i&: MBBI,
1173 n: getUnmanagedInterruptCSI(MF, CSI, ReverseOrder: PreferAscendingLS).size());
1174 CFIBuilder.setInsertPoint(MBBI);
1175
1176 // Iterate over list of callee-saved registers and emit .cfi_offset
1177 // directives.
1178 if (NeedsDwarfCFI) {
1179 for (const CalleeSavedInfo &CS :
1180 getUnmanagedInterruptCSI(MF, CSI, ReverseOrder: PreferAscendingLS)) {
1181 MCRegister Reg = CS.getReg();
1182 int64_t Offset = MFI.getObjectOffset(ObjectIdx: CS.getFrameIdx());
1183 // Emit CFI for both sub-registers. The even register is at the base
1184 // offset and odd at base+4.
1185 if (RISCV::GPRPairRegClass.contains(Reg)) {
1186 MCRegister EvenReg = RI->getSubReg(Reg, Idx: RISCV::sub_gpr_even);
1187 MCRegister OddReg = RI->getSubReg(Reg, Idx: RISCV::sub_gpr_odd);
1188 CFIBuilder.buildOffset(Reg: EvenReg, Offset);
1189 CFIBuilder.buildOffset(Reg: OddReg, Offset: Offset + 4);
1190 } else {
1191 CFIBuilder.buildOffset(Reg, Offset);
1192 }
1193 }
1194 }
1195
1196 // Generate new FP.
1197 if (hasFP(MF)) {
1198 if (STI.isRegisterReservedByUser(i: FPReg))
1199 MF.getFunction().getContext().diagnose(DI: DiagnosticInfoUnsupported{
1200 MF.getFunction(), "Frame pointer required, but has been reserved."});
1201 // The frame pointer does need to be reserved from register allocation.
1202 assert(MF.getRegInfo().isReserved(FPReg) && "FP not reserved");
1203
1204 // Some stack management variants automatically keep FP updated, so we don't
1205 // need an instruction to do so.
1206 if (!RVFI->hasImplicitFPUpdates(MF)) {
1207 RI->adjustReg(
1208 MBB, II: MBBI, DL, DestReg: FPReg, SrcReg: SPReg,
1209 Offset: StackOffset::getFixed(Fixed: RealStackSize - RVFI->getVarArgsSaveSize()),
1210 Flag: MachineInstr::FrameSetup, RequiredAlign: getStackAlign());
1211 }
1212
1213 if (NeedsDwarfCFI)
1214 CFIBuilder.buildDefCFA(Reg: FPReg, Offset: RVFI->getVarArgsSaveSize());
1215 }
1216
1217 uint64_t SecondSPAdjustAmount = 0;
1218 // Emit the second SP adjustment after saving callee saved registers.
1219 if (FirstSPAdjustAmount) {
1220 SecondSPAdjustAmount = getStackSizeWithRVVPadding(MF) - FirstSPAdjustAmount;
1221 assert(SecondSPAdjustAmount > 0 &&
1222 "SecondSPAdjustAmount should be greater than zero");
1223
1224 allocateStack(MBB, MBBI, MF, Offset: SecondSPAdjustAmount,
1225 RealStackSize: getStackSizeWithRVVPadding(MF), EmitCFI: NeedsDwarfCFI && !hasFP(MF),
1226 NeedProbe, ProbeSize, DynAllocation,
1227 Flag: MachineInstr::FrameSetup);
1228 }
1229
1230 if (RVVStackSize) {
1231 if (NeedProbe) {
1232 allocateAndProbeStackForRVV(MF, MBB, MBBI, DL, Amount: RVVStackSize,
1233 Flag: MachineInstr::FrameSetup,
1234 EmitCFI: NeedsDwarfCFI && !hasFP(MF), DynAllocation);
1235 } else {
1236 // We must keep the stack pointer aligned through any intermediate
1237 // updates.
1238 RI->adjustReg(MBB, II: MBBI, DL, DestReg: SPReg, SrcReg: SPReg,
1239 Offset: StackOffset::getScalable(Scalable: -RVVStackSize),
1240 Flag: MachineInstr::FrameSetup, RequiredAlign: getStackAlign());
1241 }
1242
1243 if (NeedsDwarfCFI && !hasFP(MF)) {
1244 // Emit .cfi_def_cfa_expression "sp + StackSize + RVVStackSize * vlenb".
1245 CFIBuilder.insertCFIInst(CFIInst: createDefCFAExpression(
1246 TRI: *RI, Reg: SPReg,
1247 Offset: StackOffset::get(Fixed: getStackSizeWithRVVPadding(MF), Scalable: RVVStackSize / 8)));
1248 }
1249
1250 std::advance(i&: MBBI, n: getRVVCalleeSavedInfo(MF, CSI).size());
1251 if (NeedsDwarfCFI)
1252 emitCalleeSavedRVVPrologCFI(MBB, MI: MBBI, HasFP: hasFP(MF));
1253 }
1254
1255 if (hasFP(MF)) {
1256 // Realign Stack
1257 const RISCVRegisterInfo *RI = STI.getRegisterInfo();
1258 if (RI->hasStackRealignment(MF)) {
1259 Align MaxAlignment = MFI.getMaxAlign();
1260
1261 const RISCVInstrInfo *TII = STI.getInstrInfo();
1262 if (isInt<12>(x: -(int64_t)MaxAlignment.value())) {
1263 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::ANDI), DestReg: SPReg)
1264 .addReg(RegNo: SPReg)
1265 .addImm(Val: -(int64_t)MaxAlignment.value())
1266 .setMIFlag(MachineInstr::FrameSetup);
1267 } else {
1268 unsigned ShiftAmount = Log2(A: MaxAlignment);
1269 Register VR =
1270 MF.getRegInfo().createVirtualRegister(RegClass: &RISCV::GPRRegClass);
1271 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::SRLI), DestReg: VR)
1272 .addReg(RegNo: SPReg)
1273 .addImm(Val: ShiftAmount)
1274 .setMIFlag(MachineInstr::FrameSetup);
1275 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::SLLI), DestReg: SPReg)
1276 .addReg(RegNo: VR)
1277 .addImm(Val: ShiftAmount)
1278 .setMIFlag(MachineInstr::FrameSetup);
1279 }
1280 if (NeedProbe && RVVStackSize == 0) {
1281 // Do a probe if the align + size allocated just passed the probe size
1282 // and was not yet probed.
1283 if (SecondSPAdjustAmount < ProbeSize &&
1284 SecondSPAdjustAmount + MaxAlignment.value() >= ProbeSize) {
1285 bool IsRV64 = STI.is64Bit();
1286 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: IsRV64 ? RISCV::SD : RISCV::SW))
1287 .addReg(RegNo: RISCV::X0)
1288 .addReg(RegNo: SPReg)
1289 .addImm(Val: 0)
1290 .setMIFlags(MachineInstr::FrameSetup);
1291 }
1292 }
1293 // FP will be used to restore the frame in the epilogue, so we need
1294 // another base register BP to record SP after re-alignment. SP will
1295 // track the current stack after allocating variable sized objects.
1296 if (hasBP(MF)) {
1297 // move BP, SP
1298 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII->get(Opcode: RISCV::ADDI), DestReg: BPReg)
1299 .addReg(RegNo: SPReg)
1300 .addImm(Val: 0)
1301 .setMIFlag(MachineInstr::FrameSetup);
1302 }
1303 }
1304 }
1305}
1306
1307void RISCVFrameLowering::deallocateStack(MachineFunction &MF,
1308 MachineBasicBlock &MBB,
1309 MachineBasicBlock::iterator MBBI,
1310 const DebugLoc &DL,
1311 uint64_t &StackSize,
1312 int64_t CFAOffset) const {
1313 const RISCVRegisterInfo *RI = STI.getRegisterInfo();
1314
1315 RI->adjustReg(MBB, II: MBBI, DL, DestReg: SPReg, SrcReg: SPReg, Offset: StackOffset::getFixed(Fixed: StackSize),
1316 Flag: MachineInstr::FrameDestroy, RequiredAlign: getStackAlign());
1317 StackSize = 0;
1318
1319 if (needsDwarfCFI(MF))
1320 CFIInstBuilder(MBB, MBBI, MachineInstr::FrameDestroy)
1321 .buildDefCFAOffset(Offset: CFAOffset);
1322}
1323
1324void RISCVFrameLowering::emitEpilogue(MachineFunction &MF,
1325 MachineBasicBlock &MBB) const {
1326 const RISCVRegisterInfo *RI = STI.getRegisterInfo();
1327 MachineFrameInfo &MFI = MF.getFrameInfo();
1328 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
1329 bool PreferAscendingLS = STI.preferAscendingLoadStore();
1330
1331 // All calls are tail calls in GHC calling conv, and functions have no
1332 // prologue/epilogue.
1333 if (MF.getFunction().getCallingConv() == CallingConv::GHC)
1334 return;
1335
1336 // Get the insert location for the epilogue. If there were no terminators in
1337 // the block, get the last instruction.
1338 MachineBasicBlock::iterator MBBI = MBB.end();
1339 DebugLoc DL;
1340 if (!MBB.empty()) {
1341 MBBI = MBB.getLastNonDebugInstr();
1342 if (MBBI != MBB.end())
1343 DL = MBBI->getDebugLoc();
1344
1345 MBBI = MBB.getFirstTerminator();
1346
1347 // Skip to before the restores of all callee-saved registers.
1348 while (MBBI != MBB.begin() &&
1349 std::prev(x: MBBI)->getFlag(Flag: MachineInstr::FrameDestroy))
1350 --MBBI;
1351 }
1352
1353 const auto &CSI = MFI.getCalleeSavedInfo();
1354
1355 // Skip to before the restores of scalar callee-saved registers
1356 // FIXME: assumes exactly one instruction is used to restore each
1357 // callee-saved register.
1358 auto FirstScalarCSRRestoreInsn =
1359 std::next(x: MBBI, n: getRVVCalleeSavedInfo(MF, CSI).size());
1360 CFIInstBuilder CFIBuilder(MBB, FirstScalarCSRRestoreInsn,
1361 MachineInstr::FrameDestroy);
1362 bool NeedsDwarfCFI = needsDwarfCFI(MF);
1363
1364 uint64_t FirstSPAdjustAmount = getFirstSPAdjustAmount(MF);
1365 uint64_t RealStackSize = FirstSPAdjustAmount ? FirstSPAdjustAmount
1366 : getStackSizeWithRVVPadding(MF);
1367 uint64_t StackSize = FirstSPAdjustAmount ? FirstSPAdjustAmount
1368 : getStackSizeWithRVVPadding(MF) -
1369 RVFI->getReservedSpillsSize();
1370 uint64_t FPOffset = RealStackSize - RVFI->getVarArgsSaveSize();
1371 uint64_t RVVStackSize = RVFI->getRVVStackSize();
1372
1373 bool RestoreSPFromFP = RI->hasStackRealignment(MF) ||
1374 MFI.hasVarSizedObjects() || !hasReservedCallFrame(MF);
1375 if (RVVStackSize) {
1376 // If RestoreSPFromFP the stack pointer will be restored using the frame
1377 // pointer value.
1378 if (!RestoreSPFromFP)
1379 RI->adjustReg(MBB, II: FirstScalarCSRRestoreInsn, DL, DestReg: SPReg, SrcReg: SPReg,
1380 Offset: StackOffset::getScalable(Scalable: RVVStackSize),
1381 Flag: MachineInstr::FrameDestroy, RequiredAlign: getStackAlign());
1382
1383 if (NeedsDwarfCFI) {
1384 if (!hasFP(MF))
1385 CFIBuilder.buildDefCFA(Reg: SPReg, Offset: RealStackSize);
1386 emitCalleeSavedRVVEpilogCFI(MBB, MI: FirstScalarCSRRestoreInsn);
1387 }
1388 }
1389
1390 if (FirstSPAdjustAmount) {
1391 uint64_t SecondSPAdjustAmount =
1392 getStackSizeWithRVVPadding(MF) - FirstSPAdjustAmount;
1393 assert(SecondSPAdjustAmount > 0 &&
1394 "SecondSPAdjustAmount should be greater than zero");
1395
1396 // If RestoreSPFromFP the stack pointer will be restored using the frame
1397 // pointer value.
1398 if (!RestoreSPFromFP)
1399 RI->adjustReg(MBB, II: FirstScalarCSRRestoreInsn, DL, DestReg: SPReg, SrcReg: SPReg,
1400 Offset: StackOffset::getFixed(Fixed: SecondSPAdjustAmount),
1401 Flag: MachineInstr::FrameDestroy, RequiredAlign: getStackAlign());
1402
1403 if (NeedsDwarfCFI && !hasFP(MF))
1404 CFIBuilder.buildDefCFAOffset(Offset: FirstSPAdjustAmount);
1405 }
1406
1407 // Restore the stack pointer using the value of the frame pointer. Only
1408 // necessary if the stack pointer was modified, meaning the stack size is
1409 // unknown.
1410 //
1411 // In order to make sure the stack point is right through the EH region,
1412 // we also need to restore stack pointer from the frame pointer if we
1413 // don't preserve stack space within prologue/epilogue for outgoing variables,
1414 // normally it's just checking the variable sized object is present or not
1415 // is enough, but we also don't preserve that at prologue/epilogue when
1416 // have vector objects in stack.
1417 if (RestoreSPFromFP) {
1418 assert(hasFP(MF) && "frame pointer should not have been eliminated");
1419 RI->adjustReg(MBB, II: FirstScalarCSRRestoreInsn, DL, DestReg: SPReg, SrcReg: FPReg,
1420 Offset: StackOffset::getFixed(Fixed: -FPOffset), Flag: MachineInstr::FrameDestroy,
1421 RequiredAlign: getStackAlign());
1422 }
1423
1424 if (NeedsDwarfCFI && hasFP(MF))
1425 CFIBuilder.buildDefCFA(Reg: SPReg, Offset: RealStackSize);
1426
1427 // Skip to after the restores of scalar callee-saved registers
1428 // FIXME: assumes exactly one instruction is used to restore each
1429 // callee-saved register.
1430 MBBI = std::next(x: FirstScalarCSRRestoreInsn,
1431 n: getUnmanagedInterruptCSI(MF, CSI, ReverseOrder: PreferAscendingLS).size());
1432 CFIBuilder.setInsertPoint(MBBI);
1433 emitSiFiveCLICPreemptibleRestores(MF, MBB, MBBI, CFIBuilder, DL);
1434
1435 auto emitRestoreCFI = [&](auto CSInfo) {
1436 for (auto &CS : CSInfo) {
1437 MCRegister Reg = CS.getReg();
1438 // Emit CFI for both sub-registers.
1439 if (RISCV::GPRPairRegClass.contains(Reg)) {
1440 MCRegister EvenReg = RI->getSubReg(Reg, Idx: RISCV::sub_gpr_even);
1441 MCRegister OddReg = RI->getSubReg(Reg, Idx: RISCV::sub_gpr_odd);
1442 CFIBuilder.buildRestore(Reg: EvenReg);
1443 CFIBuilder.buildRestore(Reg: OddReg);
1444 } else {
1445 CFIBuilder.buildRestore(Reg);
1446 }
1447 }
1448 };
1449
1450 if (RVFI->useSaveRestoreLibCalls(MF)) {
1451 if (RVFI->hasShadowStack(MF)) {
1452 // We don't use `__riscv_restore_<N>` to restore in this case, so we need
1453 // to make sure we're correctly accounting for the stack deallocation we
1454 // need to do, which now must include the allocation made by
1455 // `__riscv_save_<N>`.
1456 StackSize += RVFI->getLibCallStackSize();
1457
1458 SmallVector<CalleeSavedInfo, 8> LibcallCSI =
1459 getPushOrLibCallsSavedInfo(MF, CSI);
1460 MBBI = std::next(x: MBBI, n: LibcallCSI.size());
1461 CFIBuilder.setInsertPoint(MBBI);
1462
1463 if (NeedsDwarfCFI)
1464 emitRestoreCFI(LibcallCSI);
1465 } else {
1466 // tail __riscv_restore_[0-12] instruction is considered as a terminator,
1467 // therefore it is unnecessary to place any CFI instructions after it.
1468 // Just deallocate stack if needed and return.
1469 if (StackSize != 0)
1470 deallocateStack(MF, MBB, MBBI, DL, StackSize,
1471 CFAOffset: RVFI->getLibCallStackSize());
1472
1473 // Emit epilogue for shadow call stack.
1474 emitSCSEpilogue(MF, MBB, MI: MBBI, DL);
1475 return;
1476 }
1477 }
1478
1479 // Recover callee-saved registers.
1480 if (NeedsDwarfCFI)
1481 emitRestoreCFI(getUnmanagedInterruptCSI(MF, CSI, ReverseOrder: PreferAscendingLS));
1482
1483 if (RVFI->isPushable(MF) && MBBI != MBB.end() && isPop(Opcode: MBBI->getOpcode())) {
1484 // Use available stack adjustment in pop instruction to deallocate stack
1485 // space. Align the stack size down to a multiple of 16. This is needed for
1486 // RVE.
1487 // FIXME: Can we increase the stack size to a multiple of 16 instead?
1488 uint64_t StackAdj =
1489 std::min(a: alignDown(Value: StackSize, Align: 16), b: static_cast<uint64_t>(48));
1490 MBBI->getOperand(i: 1).setImm(StackAdj);
1491 StackSize -= StackAdj;
1492
1493 if (StackSize != 0)
1494 deallocateStack(MF, MBB, MBBI, DL, StackSize,
1495 /*stack_adj of cm.pop instr*/ CFAOffset: RealStackSize - StackSize);
1496
1497 auto NextI = next_nodbg(It: MBBI, End: MBB.end());
1498 if (NextI == MBB.end() || NextI->getOpcode() != RISCV::PseudoRET) {
1499 ++MBBI;
1500 if (NeedsDwarfCFI) {
1501 CFIBuilder.setInsertPoint(MBBI);
1502
1503 for (const CalleeSavedInfo &CS : getPushOrLibCallsSavedInfo(MF, CSI))
1504 CFIBuilder.buildRestore(Reg: CS.getReg());
1505
1506 // Update CFA Offset. If this is a QCI interrupt function, there will
1507 // be a leftover offset which is deallocated by `QC.C.MILEAVERET`,
1508 // otherwise getQCIInterruptStackSize() will be 0.
1509 CFIBuilder.buildDefCFAOffset(Offset: RVFI->getQCIInterruptStackSize());
1510 }
1511 }
1512 }
1513
1514 // Deallocate stack if StackSize isn't a zero yet. If this is a QCI interrupt
1515 // function, there will be a leftover offset which is deallocated by
1516 // `QC.C.MILEAVERET`, otherwise getQCIInterruptStackSize() will be 0.
1517 if (StackSize != 0)
1518 deallocateStack(MF, MBB, MBBI, DL, StackSize,
1519 CFAOffset: RVFI->getQCIInterruptStackSize());
1520
1521 // Emit epilogue for shadow call stack.
1522 emitSCSEpilogue(MF, MBB, MI: MBBI, DL);
1523
1524 // SiFive CLIC needs to swap `sf.mscratchcsw` into `sp`
1525 emitSiFiveCLICStackSwap(MF, MBB, MBBI, DL, FrameFlag: MachineInstr::FrameDestroy);
1526}
1527
1528static MCRegister getPhysicalGPR(const TargetRegisterInfo &TRI,
1529 MCRegister Reg) {
1530 if (RISCV::GPRRegClass.contains(Reg))
1531 return Reg;
1532
1533 std::array<TargetRegisterClass const *, 2> RegisterClasses = {
1534 &RISCV::GPRF16RegClass, &RISCV::GPRF32RegClass};
1535 std::array<unsigned, 2> SubIdx = {RISCV::sub_16, RISCV::sub_32};
1536
1537 for (auto [RegClass, SubReg] : zip(t&: RegisterClasses, u&: SubIdx)) {
1538 if (RegClass->contains(Reg)) {
1539 if (MCRegister Super =
1540 TRI.getMatchingSuperReg(Reg, SubIdx: SubReg, RC: &RISCV::GPRRegClass))
1541 return Super;
1542 }
1543 }
1544
1545 llvm::reportFatalInternalError(
1546 reason: "getPhysicalGPR called with unsupported register");
1547}
1548
1549static MCRegister getLargestFPRegisterOrZero(const RISCVSubtarget &STI,
1550 const TargetRegisterInfo &TRI,
1551 MCRegister Reg) {
1552 if (!STI.hasStdExtF())
1553 return MCRegister();
1554
1555 TargetRegisterClass const *LargestFPRegClass = STI.getLargestFPRegClass();
1556 assert(LargestFPRegClass);
1557
1558 if (LargestFPRegClass->contains(Reg))
1559 return Reg;
1560
1561 std::array<TargetRegisterClass const *, 3> RegisterClasses = {
1562 &RISCV::FPR16RegClass, &RISCV::FPR32RegClass, &RISCV::FPR64RegClass};
1563 std::array<unsigned, 3> SubIdx = {RISCV::sub_16, RISCV::sub_32,
1564 RISCV::sub_64};
1565
1566 for (auto [RegClass, SubReg] : zip(t&: RegisterClasses, u&: SubIdx)) {
1567 if (RegClass->contains(Reg)) {
1568 if (MCRegister Super =
1569 TRI.getMatchingSuperReg(Reg, SubIdx: SubReg, RC: LargestFPRegClass))
1570 return Super;
1571 }
1572 }
1573
1574 // Reg is bigger than what's currently available for the target, we can ignore
1575 // it.
1576 return MCRegister();
1577}
1578
1579void RISCVFrameLowering::emitZeroCallUsedRegs(BitVector RegsToZero,
1580 MachineBasicBlock &MBB,
1581 RegScavenger *RS) const {
1582 // Insertion point.
1583 MachineBasicBlock::iterator MBBI = MBB.getFirstTerminator();
1584
1585 // Fake a debug loc.
1586 DebugLoc DL;
1587 if (MBBI != MBB.end())
1588 DL = MBBI->getDebugLoc();
1589
1590 const MachineFunction &MF = *MBB.getParent();
1591 const RISCVRegisterInfo &TRI = *STI.getRegisterInfo();
1592 const RISCVInstrInfo &TII = *STI.getInstrInfo();
1593
1594 BitVector FinalRegsToZero(TRI.getNumRegs());
1595
1596 bool HasVRegister = false;
1597
1598 for (MCRegister Reg : RegsToZero.set_bits()) {
1599 if (TRI.isGeneralPurposeRegister(MF, Reg)) {
1600 FinalRegsToZero.set(getPhysicalGPR(TRI, Reg).id());
1601 } else if (RISCV::GPRPairRegClass.contains(Reg)) {
1602 FinalRegsToZero.set(
1603 getPhysicalGPR(TRI, Reg: TRI.getSubReg(Reg, Idx: RISCV::sub_gpr_even)).id());
1604 FinalRegsToZero.set(
1605 getPhysicalGPR(TRI, Reg: TRI.getSubReg(Reg, Idx: RISCV::sub_gpr_odd)).id());
1606 } else if (TRI.isFPRegister(Reg)) {
1607 if (MCRegister MaybeReg = getLargestFPRegisterOrZero(STI, TRI, Reg))
1608 FinalRegsToZero.set(MaybeReg.id());
1609 } else if (RISCVRegisterInfo::isRVVRegClass(
1610 RC: TRI.getMinimalPhysRegClass(Reg))) {
1611 if (!STI.hasVInstructions())
1612 continue;
1613 HasVRegister = true;
1614
1615 for (MCRegister SubReg : TRI.subregs_inclusive(Reg)) {
1616 if (TRI.subregs(Reg: SubReg).empty())
1617 FinalRegsToZero.set(SubReg.id());
1618 }
1619 }
1620 }
1621
1622 if (HasVRegister) {
1623 RISCVVType::VLMUL VLMUL = RISCVVType::encodeLMUL(LMUL: 1, /*Fractional=*/false);
1624 unsigned VTypeImm = RISCVVType::encodeVTYPE(
1625 VLMUL, /*SEW=*/32, /*TailAgnostic=*/true, /*MaskAgnostic=*/true);
1626
1627 MCRegister TemporaryReg = RISCV::NoRegister;
1628 for (MCRegister Reg : FinalRegsToZero.set_bits()) {
1629 if (TRI.isGeneralPurposeRegister(MF, Reg)) {
1630 TemporaryReg = Reg;
1631 break;
1632 }
1633 }
1634
1635 if (TemporaryReg == RISCV::NoRegister) {
1636 RS->enterBasicBlockEnd(MBB);
1637 TemporaryReg = RS->scavengeRegisterBackwards(RC: RISCV::GPRRegClass, To: MBBI,
1638 /*RestoreAfter=*/false,
1639 /*SPAdj=*/0);
1640 }
1641
1642 if (MBB.getParent()
1643 ->getFunction()
1644 .getFnAttribute(Kind: "zero-call-used-regs")
1645 .getValueAsString() == "used")
1646 FinalRegsToZero.set(TemporaryReg.id());
1647
1648 BuildMI(BB&: MBB, I: MBBI, MIMD: DL, MCID: TII.get(Opcode: RISCV::VSETVLI), DestReg: TemporaryReg)
1649 .addReg(RegNo: RISCV::X0)
1650 .addImm(Val: VTypeImm)
1651 .addReg(RegNo: RISCV::VL, Flags: RegState::ImplicitDefine)
1652 .addReg(RegNo: RISCV::VTYPE, Flags: RegState::ImplicitDefine);
1653 }
1654
1655 for (MCRegister Reg : FinalRegsToZero.set_bits())
1656 TII.buildClearRegister(Reg, MBB, Iter: MBBI, DL);
1657}
1658
1659StackOffset
1660RISCVFrameLowering::getFrameIndexReference(const MachineFunction &MF, int FI,
1661 Register &FrameReg) const {
1662 const MachineFrameInfo &MFI = MF.getFrameInfo();
1663 const TargetRegisterInfo *RI = MF.getSubtarget().getRegisterInfo();
1664 const auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
1665
1666 // Callee-saved registers should be referenced relative to the stack
1667 // pointer (positive offset), otherwise use the frame pointer (negative
1668 // offset).
1669 const auto &CSI = getUnmanagedCSI(MF, CSI: MFI.getCalleeSavedInfo(),
1670 ReverseOrder: STI.preferAscendingLoadStore());
1671 int MinCSFI = 0;
1672 int MaxCSFI = -1;
1673 StackOffset Offset;
1674 auto StackID = MFI.getStackID(ObjectIdx: FI);
1675
1676 assert((StackID == TargetStackID::Default ||
1677 StackID == TargetStackID::ScalableVector) &&
1678 "Unexpected stack ID for the frame object.");
1679 if (StackID == TargetStackID::Default) {
1680 assert(getOffsetOfLocalArea() == 0 && "LocalAreaOffset is not 0!");
1681 Offset = StackOffset::getFixed(Fixed: MFI.getObjectOffset(ObjectIdx: FI) +
1682 MFI.getOffsetAdjustment());
1683 } else if (StackID == TargetStackID::ScalableVector) {
1684 Offset = StackOffset::getScalable(Scalable: MFI.getObjectOffset(ObjectIdx: FI));
1685 }
1686
1687 uint64_t FirstSPAdjustAmount = getFirstSPAdjustAmount(MF);
1688
1689 if (CSI.size()) {
1690 MinCSFI = std::min(a: CSI.front().getFrameIdx(), b: CSI.back().getFrameIdx());
1691 MaxCSFI = std::max(a: CSI.front().getFrameIdx(), b: CSI.back().getFrameIdx());
1692 }
1693
1694 bool IsInterruptCSR = RVFI->isSiFivePreemptibleInterrupt(MF) &&
1695 (FI == RVFI->getInterruptCSRFrameIndex(Idx: 0) ||
1696 FI == RVFI->getInterruptCSRFrameIndex(Idx: 1));
1697 if ((FI >= MinCSFI && FI <= MaxCSFI) || IsInterruptCSR) {
1698 FrameReg = SPReg;
1699
1700 if (FirstSPAdjustAmount)
1701 Offset += StackOffset::getFixed(Fixed: FirstSPAdjustAmount);
1702 else
1703 Offset += StackOffset::getFixed(Fixed: getStackSizeWithRVVPadding(MF));
1704 return Offset;
1705 }
1706
1707 if (RI->hasStackRealignment(MF) && !MFI.isFixedObjectIndex(ObjectIdx: FI)) {
1708 // If the stack was realigned, the frame pointer is set in order to allow
1709 // SP to be restored, so we need another base register to record the stack
1710 // after realignment.
1711 // |--------------------------| --
1712 // | callee-allocated save | | <----|
1713 // | area for register varargs| | |
1714 // |--------------------------| <-- FP |
1715 // | callee-saved registers | | |
1716 // |--------------------------| -- |
1717 // | realignment (the size of | | |
1718 // | this area is not counted | | |
1719 // | in MFI.getStackSize()) | | |
1720 // |--------------------------| -- |-- MFI.getStackSize()
1721 // | RVV alignment padding | | |
1722 // | (not counted in | | |
1723 // | MFI.getStackSize() but | | |
1724 // | counted in | | |
1725 // | RVFI.getRVVStackSize()) | | |
1726 // |--------------------------| -- |
1727 // | RVV objects | | |
1728 // | (not counted in | | |
1729 // | MFI.getStackSize()) | | |
1730 // |--------------------------| -- |
1731 // | padding before RVV | | |
1732 // | (not counted in | | |
1733 // | MFI.getStackSize() or in | | |
1734 // | RVFI.getRVVStackSize()) | | |
1735 // |--------------------------| -- |
1736 // | scalar local variables | | <----'
1737 // |--------------------------| -- <-- BP (if var sized objects present)
1738 // | VarSize objects | |
1739 // |--------------------------| -- <-- SP
1740 if (hasBP(MF)) {
1741 FrameReg = RISCVABI::getBPReg();
1742 } else {
1743 // VarSize objects must be empty in this case!
1744 assert(!MFI.hasVarSizedObjects());
1745 FrameReg = SPReg;
1746 }
1747 } else if (!RI->hasStackRealignment(MF)) {
1748 // Note: Keeping the following as multiple 'if' statements rather than
1749 // merging to a single expression for readability.
1750 if (!hasFP(MF)) {
1751 // No FP available, must use SP.
1752 FrameReg = SPReg;
1753 } else {
1754 FrameReg = FPReg;
1755 // SP-relative addressing is only valid when SP is stable throughout
1756 // the function body: no dynamic SP adjustments for outgoing call args,
1757 // no variable-sized objects, and no RVV scalable stack regions.
1758 // hasReservedCallFrame() conservatively encompasses all these checks.
1759 if (hasReservedCallFrame(MF)) {
1760 // Both FP and SP are candidates.
1761 // Prefer SP when the SP-relative offset fits in the compressed
1762 // instruction immediate range.
1763 int64_t SPOff = Offset.getFixed() + MFI.getStackSize();
1764 int64_t CLWSPMaxOffset = 252;
1765 int64_t CLDSPMaxOffset = 504;
1766 int64_t SPThreshold = STI.is64Bit() ? CLDSPMaxOffset : CLWSPMaxOffset;
1767 if (SPOff >= 0 && SPOff <= SPThreshold)
1768 FrameReg = SPReg;
1769 }
1770 }
1771 } else {
1772 assert(RI->hasStackRealignment(MF) && MFI.isFixedObjectIndex(FI) &&
1773 "Expected fixed object with stack realignment");
1774 assert(hasFP(MF) && "Re-aligned stack must have frame pointer");
1775 FrameReg = FPReg;
1776 }
1777
1778 if (FrameReg == FPReg) {
1779 Offset += StackOffset::getFixed(Fixed: RVFI->getVarArgsSaveSize());
1780 // When using FP to access scalable vector objects, we need to minus
1781 // the frame size.
1782 //
1783 // |--------------------------| --
1784 // | callee-allocated save | |
1785 // | area for register varargs| |
1786 // |--------------------------| | -- <-- FP
1787 // | callee-saved registers | |
1788 // |--------------------------| | MFI.getStackSize()
1789 // | scalar local variables | |
1790 // |--------------------------| -- (Offset of RVV objects is from here.)
1791 // | RVV objects |
1792 // |--------------------------|
1793 // | VarSize objects |
1794 // |--------------------------| <-- SP
1795 if (StackID == TargetStackID::ScalableVector) {
1796 assert(!RI->hasStackRealignment(MF) &&
1797 "Can't index across variable sized realign");
1798 // We don't expect any extra RVV alignment padding, as the stack size
1799 // and RVV object sections should be correct aligned in their own
1800 // right.
1801 assert(MFI.getStackSize() == getStackSizeWithRVVPadding(MF) &&
1802 "Inconsistent stack layout");
1803 Offset -= StackOffset::getFixed(Fixed: MFI.getStackSize());
1804 }
1805 return Offset;
1806 }
1807
1808 // This case handles indexing off both SP and BP.
1809 // If indexing off SP, there must not be any var sized objects
1810 assert(FrameReg == RISCVABI::getBPReg() || !MFI.hasVarSizedObjects());
1811
1812 // When using SP to access frame objects, we need to add RVV stack size.
1813 //
1814 // |--------------------------| --
1815 // | callee-allocated save | | <----|
1816 // | area for register varargs| | |
1817 // |--------------------------| | | <-- FP
1818 // | callee-saved registers | | |
1819 // |--------------------------| -- |
1820 // | RVV alignment padding | | |
1821 // | (not counted in | | |
1822 // | MFI.getStackSize() but | | |
1823 // | counted in | | |
1824 // | RVFI.getRVVStackSize()) | | |
1825 // |--------------------------| -- |
1826 // | RVV objects | | |-- MFI.getStackSize()
1827 // | (not counted in | | |
1828 // | MFI.getStackSize()) | | |
1829 // |--------------------------| -- |
1830 // | padding before RVV | | |
1831 // | (not counted in | | |
1832 // | MFI.getStackSize()) | | |
1833 // |--------------------------| -- |
1834 // | scalar local variables | | <----'
1835 // |--------------------------| -- <-- BP (if var sized objects present)
1836 // | VarSize objects | |
1837 // |--------------------------| -- <-- SP
1838 //
1839 // The total amount of padding surrounding RVV objects is described by
1840 // RVV->getRVVPadding() and it can be zero. It allows us to align the RVV
1841 // objects to the required alignment.
1842 if (MFI.getStackID(ObjectIdx: FI) == TargetStackID::Default) {
1843 if (MFI.isFixedObjectIndex(ObjectIdx: FI)) {
1844 assert(!RI->hasStackRealignment(MF) &&
1845 "Can't index across variable sized realign");
1846 Offset += StackOffset::get(Fixed: getStackSizeWithRVVPadding(MF),
1847 Scalable: RVFI->getRVVStackSize());
1848 } else {
1849 Offset += StackOffset::getFixed(Fixed: MFI.getStackSize());
1850 }
1851 } else if (MFI.getStackID(ObjectIdx: FI) == TargetStackID::ScalableVector) {
1852 // Ensure the base of the RVV stack is correctly aligned: add on the
1853 // alignment padding.
1854 int64_t ScalarLocalVarSize =
1855 MFI.getStackSize() - RVFI->getCalleeSavedStackSize() -
1856 RVFI->getVarArgsSaveSize() + RVFI->getRVVPadding();
1857 Offset += StackOffset::get(Fixed: ScalarLocalVarSize, Scalable: RVFI->getRVVStackSize());
1858 }
1859 return Offset;
1860}
1861
1862static MCRegister getRVVBaseRegister(const RISCVRegisterInfo &TRI,
1863 const Register &Reg) {
1864 MCRegister BaseReg = TRI.getSubReg(Reg, Idx: RISCV::sub_vrm1_0);
1865 // If it's not a grouped vector register, it doesn't have subregister, so
1866 // the base register is just itself.
1867 if (!BaseReg.isValid())
1868 BaseReg = Reg;
1869 return BaseReg;
1870}
1871
1872void RISCVFrameLowering::determineCalleeSaves(MachineFunction &MF,
1873 BitVector &SavedRegs,
1874 RegScavenger *RS) const {
1875 TargetFrameLowering::determineCalleeSaves(MF, SavedRegs, RS);
1876
1877 // In TargetFrameLowering::determineCalleeSaves, any vector register is marked
1878 // as saved if any of its subregister is clobbered, this is not correct in
1879 // vector registers. We only want the vector register to be marked as saved
1880 // if all of its subregisters are clobbered.
1881 // For example:
1882 // Original behavior: If v24 is marked, v24m2, v24m4, v24m8 are also marked.
1883 // Correct behavior: v24m2 is marked only if v24 and v25 are marked.
1884 MachineRegisterInfo &MRI = MF.getRegInfo();
1885 const MCPhysReg *CSRegs = MRI.getCalleeSavedRegs();
1886 const RISCVRegisterInfo &TRI = *STI.getRegisterInfo();
1887 for (unsigned i = 0; CSRegs[i]; ++i) {
1888 unsigned CSReg = CSRegs[i];
1889 // Only vector registers need special care.
1890 if (!RISCV::VRRegClass.contains(Reg: getRVVBaseRegister(TRI, Reg: CSReg)))
1891 continue;
1892
1893 SavedRegs.reset(Idx: CSReg);
1894
1895 auto SubRegs = TRI.subregs(Reg: CSReg);
1896 // Set the register and all its subregisters.
1897 if (!MRI.def_empty(RegNo: CSReg) || MRI.getUsedPhysRegsMask().test(Idx: CSReg)) {
1898 SavedRegs.set(CSReg);
1899 for (unsigned Reg : SubRegs)
1900 SavedRegs.set(Reg);
1901 }
1902
1903 }
1904
1905 // Unconditionally spill RA and FP only if the function uses a frame
1906 // pointer.
1907 if (hasFP(MF)) {
1908 SavedRegs.set(RAReg);
1909 SavedRegs.set(FPReg);
1910 }
1911 // Mark BP as used if function has dedicated base pointer.
1912 if (hasBP(MF))
1913 SavedRegs.set(RISCVABI::getBPReg());
1914
1915 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
1916 // X5 is used as a temporary for saving and restoring `mcause` and `mepc`.
1917 if (RVFI->isSiFivePreemptibleInterrupt(MF))
1918 SavedRegs.set(RISCV::X5);
1919
1920 // When using cm.push/pop we must save X27 if we save X26.
1921 if (RVFI->isPushable(MF) && SavedRegs.test(Idx: RISCV::X26))
1922 SavedRegs.set(RISCV::X27);
1923
1924 // For Zilsd on RV32, append GPRPair registers to the CSR list. This prevents
1925 // the need to create register sets for each abi which is a lot more complex.
1926 // Don't use Zilsd for callee-saved coalescing if the required alignment
1927 // exceeds the stack alignment or when Zcmp/Xqccmp or save/restore libcalls
1928 // are enabled.
1929 bool UseZilsd = !STI.is64Bit() && STI.hasStdExtZilsd() &&
1930 STI.getZilsdAlign() <= getStackAlign() &&
1931 !RVFI->isPushable(MF) && !RVFI->useSaveRestoreLibCalls(MF);
1932 if (UseZilsd) {
1933 SmallVector<MCPhysReg, 32> NewCSRs;
1934 SmallSet<MCPhysReg, 16> CSRSet;
1935 for (unsigned i = 0; CSRegs[i]; ++i) {
1936 NewCSRs.push_back(Elt: CSRegs[i]);
1937 CSRSet.insert(V: CSRegs[i]);
1938 }
1939
1940 // Append GPRPair registers for pairs where both sub-registers are in CSR
1941 // list. Iterate through all GPRPairs and check if both sub-regs are CSRs.
1942 for (MCPhysReg Pair : RISCV::GPRPairRegClass) {
1943 // Do not append a pair that's already in the CSR list.
1944 if (CSRSet.contains(V: Pair))
1945 continue;
1946 MCRegister EvenReg = TRI.getSubReg(Reg: Pair, Idx: RISCV::sub_gpr_even);
1947 MCRegister OddReg = TRI.getSubReg(Reg: Pair, Idx: RISCV::sub_gpr_odd);
1948 if (CSRSet.contains(V: EvenReg.id()) && CSRSet.contains(V: OddReg.id())) {
1949 NewCSRs.push_back(Elt: Pair);
1950 CSRSet.insert(V: Pair);
1951 }
1952 }
1953
1954 MRI.setCalleeSavedRegs(NewCSRs);
1955 CSRegs = MRI.getCalleeSavedRegs();
1956 }
1957
1958 // Check if all subregisters are marked for saving. If so, set the super
1959 // register bit. For GPRPair, only check sub_gpr_even and sub_gpr_odd, not
1960 // aliases like X8_W or X8_H which are not set in SavedRegs.
1961 for (unsigned i = 0; CSRegs[i]; ++i) {
1962 MCRegister CSReg = CSRegs[i];
1963 bool CombineToSuperReg;
1964 if (RISCV::GPRPairRegClass.contains(Reg: CSReg)) {
1965 MCRegister EvenReg = TRI.getSubReg(Reg: CSReg, Idx: RISCV::sub_gpr_even);
1966 MCRegister OddReg = TRI.getSubReg(Reg: CSReg, Idx: RISCV::sub_gpr_odd);
1967 CombineToSuperReg =
1968 SavedRegs.test(Idx: EvenReg.id()) && SavedRegs.test(Idx: OddReg.id());
1969 // If s0(x8) is used as FP we can't generate load/store pair because it
1970 // breaks the frame chain.
1971 if (hasFP(MF) && CSReg == RISCV::X8_X9)
1972 CombineToSuperReg = false;
1973 } else {
1974 auto SubRegs = TRI.subregs(Reg: CSReg);
1975 CombineToSuperReg =
1976 !SubRegs.empty() && llvm::all_of(Range&: SubRegs, P: [&](unsigned Reg) {
1977 return SavedRegs.test(Idx: Reg);
1978 });
1979 }
1980
1981 if (CombineToSuperReg)
1982 SavedRegs.set(CSReg);
1983 }
1984
1985 // SiFive Preemptible Interrupt Handlers need additional frame entries
1986 createSiFivePreemptibleInterruptFrameEntries(MF, RVFI&: *RVFI);
1987}
1988
1989std::pair<int64_t, Align>
1990RISCVFrameLowering::assignRVVStackObjectOffsets(MachineFunction &MF) const {
1991 MachineFrameInfo &MFI = MF.getFrameInfo();
1992 // Create a buffer of RVV objects to allocate.
1993 SmallVector<int, 8> ObjectsToAllocate;
1994 auto pushRVVObjects = [&](int FIBegin, int FIEnd) {
1995 for (int I = FIBegin, E = FIEnd; I != E; ++I) {
1996 unsigned StackID = MFI.getStackID(ObjectIdx: I);
1997 if (StackID != TargetStackID::ScalableVector)
1998 continue;
1999 if (MFI.isDeadObjectIndex(ObjectIdx: I))
2000 continue;
2001
2002 ObjectsToAllocate.push_back(Elt: I);
2003 }
2004 };
2005 // First push RVV Callee Saved object, then push RVV stack object
2006 std::vector<CalleeSavedInfo> &CSI = MF.getFrameInfo().getCalleeSavedInfo();
2007 const auto &RVVCSI = getRVVCalleeSavedInfo(MF, CSI);
2008 if (!RVVCSI.empty())
2009 pushRVVObjects(RVVCSI[0].getFrameIdx(),
2010 RVVCSI[RVVCSI.size() - 1].getFrameIdx() + 1);
2011 pushRVVObjects(0, MFI.getObjectIndexEnd() - RVVCSI.size());
2012
2013 // The minimum alignment is 16 bytes.
2014 Align RVVStackAlign(16);
2015 const auto &ST = MF.getSubtarget<RISCVSubtarget>();
2016
2017 if (!ST.hasVInstructions()) {
2018 assert(ObjectsToAllocate.empty() &&
2019 "Can't allocate scalable-vector objects without V instructions");
2020 return std::make_pair(x: 0, y&: RVVStackAlign);
2021 }
2022
2023 // Allocate all RVV locals and spills
2024 int64_t Offset = 0;
2025 for (int FI : ObjectsToAllocate) {
2026 // ObjectSize in bytes.
2027 int64_t ObjectSize = MFI.getObjectSize(ObjectIdx: FI);
2028 auto ObjectAlign =
2029 std::max(a: Align(RISCV::RVVBytesPerBlock), b: MFI.getObjectAlign(ObjectIdx: FI));
2030 // If the data type is the fractional vector type, reserve one vector
2031 // register for it.
2032 if (ObjectSize < RISCV::RVVBytesPerBlock)
2033 ObjectSize = RISCV::RVVBytesPerBlock;
2034 Offset = alignTo(Size: Offset + ObjectSize, A: ObjectAlign);
2035 MFI.setObjectOffset(ObjectIdx: FI, SPOffset: -Offset);
2036 // Update the maximum alignment of the RVV stack section
2037 RVVStackAlign = std::max(a: RVVStackAlign, b: ObjectAlign);
2038 }
2039
2040 uint64_t StackSize = Offset;
2041
2042 // Ensure the alignment of the RVV stack. Since we want the most-aligned
2043 // object right at the bottom (i.e., any padding at the top of the frame),
2044 // readjust all RVV objects down by the alignment padding.
2045 // Stack size and offsets are multiples of vscale, stack alignment is in
2046 // bytes, we can divide stack alignment by minimum vscale to get a maximum
2047 // stack alignment multiple of vscale.
2048 auto VScale =
2049 std::max<uint64_t>(a: ST.getRealMinVLen() / RISCV::RVVBitsPerBlock, b: 1);
2050 if (auto RVVStackAlignVScale = RVVStackAlign.value() / VScale) {
2051 if (auto AlignmentPadding =
2052 offsetToAlignment(Value: StackSize, Alignment: Align(RVVStackAlignVScale))) {
2053 StackSize += AlignmentPadding;
2054 for (int FI : ObjectsToAllocate)
2055 MFI.setObjectOffset(ObjectIdx: FI, SPOffset: MFI.getObjectOffset(ObjectIdx: FI) - AlignmentPadding);
2056 }
2057 }
2058
2059 return std::make_pair(x&: StackSize, y&: RVVStackAlign);
2060}
2061
2062static unsigned getScavSlotsNumForRVV(MachineFunction &MF) {
2063 // For RVV spill, scalable stack offsets computing requires up to two scratch
2064 // registers
2065 static constexpr unsigned ScavSlotsNumRVVSpillScalableObject = 2;
2066
2067 // For RVV spill, non-scalable stack offsets computing requires up to one
2068 // scratch register.
2069 static constexpr unsigned ScavSlotsNumRVVSpillNonScalableObject = 1;
2070
2071 // ADDI instruction's destination register can be used for computing
2072 // offsets. So Scalable stack offsets require up to one scratch register.
2073 static constexpr unsigned ScavSlotsADDIScalableObject = 1;
2074
2075 static constexpr unsigned MaxScavSlotsNumKnown =
2076 std::max(l: {ScavSlotsADDIScalableObject, ScavSlotsNumRVVSpillScalableObject,
2077 ScavSlotsNumRVVSpillNonScalableObject});
2078
2079 unsigned MaxScavSlotsNum = 0;
2080 if (!MF.getSubtarget<RISCVSubtarget>().hasVInstructions())
2081 return false;
2082 for (const MachineBasicBlock &MBB : MF)
2083 for (const MachineInstr &MI : MBB) {
2084 bool IsRVVSpill = RISCV::isRVVSpill(MI);
2085 for (auto &MO : MI.operands()) {
2086 if (!MO.isFI())
2087 continue;
2088 bool IsScalableVectorID = MF.getFrameInfo().getStackID(ObjectIdx: MO.getIndex()) ==
2089 TargetStackID::ScalableVector;
2090 if (IsRVVSpill) {
2091 MaxScavSlotsNum = std::max(
2092 a: MaxScavSlotsNum, b: IsScalableVectorID
2093 ? ScavSlotsNumRVVSpillScalableObject
2094 : ScavSlotsNumRVVSpillNonScalableObject);
2095 } else if (MI.getOpcode() == RISCV::ADDI && IsScalableVectorID) {
2096 MaxScavSlotsNum =
2097 std::max(a: MaxScavSlotsNum, b: ScavSlotsADDIScalableObject);
2098 }
2099 }
2100 if (MaxScavSlotsNum == MaxScavSlotsNumKnown)
2101 return MaxScavSlotsNumKnown;
2102 }
2103 return MaxScavSlotsNum;
2104}
2105
2106static bool hasRVVFrameObject(const MachineFunction &MF) {
2107 // Originally, the function will scan all the stack objects to check whether
2108 // if there is any scalable vector object on the stack or not. However, it
2109 // causes errors in the register allocator. In issue 53016, it returns false
2110 // before RA because there is no RVV stack objects. After RA, it returns true
2111 // because there are spilling slots for RVV values during RA. It will not
2112 // reserve BP during register allocation and generate BP access in the PEI
2113 // pass due to the inconsistent behavior of the function.
2114 //
2115 // The function is changed to use hasVInstructions() as the return value. It
2116 // is not precise, but it can make the register allocation correct.
2117 //
2118 // FIXME: Find a better way to make the decision or revisit the solution in
2119 // D103622.
2120 //
2121 // Refer to https://github.com/llvm/llvm-project/issues/53016.
2122 return MF.getSubtarget<RISCVSubtarget>().hasVInstructions();
2123}
2124
2125static unsigned estimateFunctionSizeInBytes(const MachineFunction &MF,
2126 const RISCVInstrInfo &TII) {
2127 unsigned FnSize = 0;
2128 for (auto &MBB : MF) {
2129 for (auto &MI : MBB) {
2130 // Far branches over 20-bit offset will be relaxed in branch relaxation
2131 // pass. In the worst case, conditional branches will be relaxed into
2132 // the following instruction sequence. Unconditional branches are
2133 // relaxed in the same way, with the exception that there is no first
2134 // branch instruction.
2135 //
2136 // foo
2137 // bne t5, t6, .rev_cond # `TII->getInstSizeInBytes(MI)` bytes
2138 // sd s11, 0(sp) # 4 bytes, or 2 bytes with Zca
2139 // jump .restore, s11 # 8 bytes
2140 // .rev_cond
2141 // bar
2142 // j .dest_bb # 4 bytes, or 2 bytes with Zca
2143 // .restore:
2144 // ld s11, 0(sp) # 4 bytes, or 2 bytes with Zca
2145 // .dest:
2146 // baz
2147 if (MI.isConditionalBranch())
2148 FnSize += TII.getInstSizeInBytes(MI);
2149 if (MI.isConditionalBranch() || MI.isUnconditionalBranch()) {
2150 if (MF.getSubtarget<RISCVSubtarget>().hasStdExtZca())
2151 FnSize += 2 + 8 + 2 + 2;
2152 else
2153 FnSize += 4 + 8 + 4 + 4;
2154 continue;
2155 }
2156
2157 FnSize += TII.getInstSizeInBytes(MI);
2158 }
2159 }
2160 return FnSize;
2161}
2162
2163void RISCVFrameLowering::processFunctionBeforeFrameFinalized(
2164 MachineFunction &MF, RegScavenger *RS) const {
2165 const RISCVRegisterInfo *RegInfo =
2166 MF.getSubtarget<RISCVSubtarget>().getRegisterInfo();
2167 const RISCVInstrInfo *TII = MF.getSubtarget<RISCVSubtarget>().getInstrInfo();
2168 MachineFrameInfo &MFI = MF.getFrameInfo();
2169 const TargetRegisterClass *RC = &RISCV::GPRRegClass;
2170 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
2171
2172 int64_t RVVStackSize;
2173 Align RVVStackAlign;
2174 std::tie(args&: RVVStackSize, args&: RVVStackAlign) = assignRVVStackObjectOffsets(MF);
2175
2176 RVFI->setRVVStackSize(RVVStackSize);
2177 RVFI->setRVVStackAlign(RVVStackAlign);
2178
2179 if (hasRVVFrameObject(MF)) {
2180 // Ensure the entire stack is aligned to at least the RVV requirement: some
2181 // scalable-vector object alignments are not considered by the
2182 // target-independent code.
2183 MFI.ensureMaxAlignment(Alignment: RVVStackAlign);
2184 }
2185
2186 unsigned ScavSlotsNum = 0;
2187
2188 // estimateStackSize has been observed to under-estimate the final stack
2189 // size, so give ourselves wiggle-room by checking for stack size
2190 // representable an 11-bit signed field rather than 12-bits.
2191 if (!isInt<11>(x: MFI.estimateStackSize(MF)))
2192 ScavSlotsNum = 1;
2193
2194 // Far branches over 20-bit offset require a spill slot for scratch register.
2195 bool IsLargeFunction = !isInt<20>(x: estimateFunctionSizeInBytes(MF, TII: *TII));
2196 if (IsLargeFunction)
2197 ScavSlotsNum = std::max(a: ScavSlotsNum, b: 1u);
2198
2199 // RVV loads & stores have no capacity to hold the immediate address offsets
2200 // so we must always reserve an emergency spill slot if the MachineFunction
2201 // contains any RVV spills.
2202 ScavSlotsNum = std::max(a: ScavSlotsNum, b: getScavSlotsNumForRVV(MF));
2203
2204 for (unsigned I = 0; I < ScavSlotsNum; I++) {
2205 int FI = MFI.CreateSpillStackObject(Size: RegInfo->getSpillSize(RC: *RC),
2206 Alignment: RegInfo->getSpillAlign(RC: *RC));
2207 RS->addScavengingFrameIndex(FI);
2208
2209 if (IsLargeFunction && RVFI->getBranchRelaxationScratchFrameIndex() == -1)
2210 RVFI->setBranchRelaxationScratchFrameIndex(FI);
2211 }
2212
2213 unsigned Size = RVFI->getReservedSpillsSize();
2214 for (const auto &Info : MFI.getCalleeSavedInfo()) {
2215 int FrameIdx = Info.getFrameIdx();
2216 if (FrameIdx < 0 || MFI.getStackID(ObjectIdx: FrameIdx) != TargetStackID::Default)
2217 continue;
2218
2219 Size += MFI.getObjectSize(ObjectIdx: FrameIdx);
2220 }
2221 RVFI->setCalleeSavedStackSize(Size);
2222}
2223
2224// Not preserve stack space within prologue for outgoing variables when the
2225// function contains variable size objects or there are vector objects accessed
2226// by the frame pointer.
2227// Let eliminateCallFramePseudoInstr preserve stack space for it.
2228bool RISCVFrameLowering::hasReservedCallFrame(const MachineFunction &MF) const {
2229 return !MF.getFrameInfo().hasVarSizedObjects() &&
2230 !(hasFP(MF) && hasRVVFrameObject(MF));
2231}
2232
2233// Eliminate ADJCALLSTACKDOWN, ADJCALLSTACKUP pseudo instructions.
2234MachineBasicBlock::iterator RISCVFrameLowering::eliminateCallFramePseudoInstr(
2235 MachineFunction &MF, MachineBasicBlock &MBB,
2236 MachineBasicBlock::iterator MI) const {
2237 DebugLoc DL = MI->getDebugLoc();
2238
2239 if (!hasReservedCallFrame(MF)) {
2240 // If space has not been reserved for a call frame, ADJCALLSTACKDOWN and
2241 // ADJCALLSTACKUP must be converted to instructions manipulating the stack
2242 // pointer. This is necessary when there is a variable length stack
2243 // allocation (e.g. alloca), which means it's not possible to allocate
2244 // space for outgoing arguments from within the function prologue.
2245 int64_t Amount = MI->getOperand(i: 0).getImm();
2246
2247 if (Amount != 0) {
2248 // Ensure the stack remains aligned after adjustment.
2249 Amount = alignSPAdjust(SPAdj: Amount);
2250
2251 if (MI->getOpcode() == RISCV::ADJCALLSTACKDOWN)
2252 Amount = -Amount;
2253
2254 const RISCVTargetLowering *TLI =
2255 MF.getSubtarget<RISCVSubtarget>().getTargetLowering();
2256 int64_t ProbeSize = TLI->getStackProbeSize(MF, StackAlign: getStackAlign());
2257 if (TLI->hasInlineStackProbe(MF) && -Amount >= ProbeSize) {
2258 // When stack probing is enabled, the decrement of SP may need to be
2259 // probed. We can handle both the decrement and the probing in
2260 // allocateStack.
2261 bool DynAllocation =
2262 MF.getInfo<RISCVMachineFunctionInfo>()->hasDynamicAllocation();
2263 allocateStack(MBB, MBBI: MI, MF, Offset: -Amount, RealStackSize: -Amount,
2264 EmitCFI: needsDwarfCFI(MF) && !hasFP(MF),
2265 /*NeedProbe=*/true, ProbeSize, DynAllocation,
2266 Flag: MachineInstr::NoFlags);
2267 inlineStackProbe(MF, PrologueMBB&: MBB);
2268 } else {
2269 const RISCVRegisterInfo &RI = *STI.getRegisterInfo();
2270 RI.adjustReg(MBB, II: MI, DL, DestReg: SPReg, SrcReg: SPReg, Offset: StackOffset::getFixed(Fixed: Amount),
2271 Flag: MachineInstr::NoFlags, RequiredAlign: getStackAlign());
2272 }
2273 }
2274 }
2275
2276 return MBB.erase(I: MI);
2277}
2278
2279// We would like to split the SP adjustment to reduce prologue/epilogue
2280// as following instructions. In this way, the offset of the callee saved
2281// register could fit in a single store. Supposed that the first sp adjust
2282// amount is 2032.
2283// add sp,sp,-2032
2284// sw ra,2028(sp)
2285// sw s0,2024(sp)
2286// sw s1,2020(sp)
2287// sw s3,2012(sp)
2288// sw s4,2008(sp)
2289// add sp,sp,-64
2290uint64_t
2291RISCVFrameLowering::getFirstSPAdjustAmount(const MachineFunction &MF) const {
2292 const auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
2293 const MachineFrameInfo &MFI = MF.getFrameInfo();
2294 const std::vector<CalleeSavedInfo> &CSI = MFI.getCalleeSavedInfo();
2295 uint64_t StackSize = getStackSizeWithRVVPadding(MF);
2296
2297 // Disable SplitSPAdjust if save-restore libcall, push/pop or QCI interrupts
2298 // are used. The callee-saved registers will be pushed by the save-restore
2299 // libcalls, so we don't have to split the SP adjustment in this case.
2300 if (RVFI->getReservedSpillsSize())
2301 return 0;
2302
2303 // Return the FirstSPAdjustAmount if the StackSize can not fit in a signed
2304 // 12-bit and there exists a callee-saved register needing to be pushed.
2305 if (!isInt<12>(x: StackSize) && (CSI.size() > 0)) {
2306 // FirstSPAdjustAmount is chosen at most as (2048 - StackAlign) because
2307 // 2048 will cause sp = sp + 2048 in the epilogue to be split into multiple
2308 // instructions. Offsets smaller than 2048 can fit in a single load/store
2309 // instruction, and we have to stick with the stack alignment. 2048 has
2310 // 16-byte alignment. The stack alignment for RV32 and RV64 is 16 and for
2311 // RV32E it is 4. So (2048 - StackAlign) will satisfy the stack alignment.
2312 const uint64_t StackAlign = getStackAlign().value();
2313
2314 // Amount of (2048 - StackAlign) will prevent callee saved and restored
2315 // instructions be compressed, so try to adjust the amount to the largest
2316 // offset that stack compression instructions accept when target supports
2317 // compression instructions.
2318 if (STI.hasStdExtZca()) {
2319 // The compression extensions may support the following instructions:
2320 // riscv32: c.lwsp rd, offset[7:2] => 2^(6 + 2)
2321 // c.swsp rs2, offset[7:2] => 2^(6 + 2)
2322 // c.flwsp rd, offset[7:2] => 2^(6 + 2)
2323 // c.fswsp rs2, offset[7:2] => 2^(6 + 2)
2324 // riscv64: c.ldsp rd, offset[8:3] => 2^(6 + 3)
2325 // c.sdsp rs2, offset[8:3] => 2^(6 + 3)
2326 // c.fldsp rd, offset[8:3] => 2^(6 + 3)
2327 // c.fsdsp rs2, offset[8:3] => 2^(6 + 3)
2328 const uint64_t RVCompressLen = STI.getXLen() * 8;
2329 // Compared with amount (2048 - StackAlign), StackSize needs to
2330 // satisfy the following conditions to avoid using more instructions
2331 // to adjust the sp after adjusting the amount, such as
2332 // StackSize meets the condition (StackSize <= 2048 + RVCompressLen),
2333 // case1: Amount is 2048 - StackAlign: use addi + addi to adjust sp.
2334 // case2: Amount is RVCompressLen: use addi + addi to adjust sp.
2335 auto CanCompress = [&](uint64_t CompressLen) -> bool {
2336 if (StackSize <= 2047 + CompressLen ||
2337 (StackSize > 2048 * 2 - StackAlign &&
2338 StackSize <= 2047 * 2 + CompressLen) ||
2339 StackSize > 2048 * 3 - StackAlign)
2340 return true;
2341
2342 return false;
2343 };
2344 // In the epilogue, addi sp, sp, 496 is used to recover the sp and it
2345 // can be compressed(C.ADDI16SP, offset can be [-512, 496]), but
2346 // addi sp, sp, 512 can not be compressed. So try to use 496 first.
2347 const uint64_t ADDI16SPCompressLen = 496;
2348 if (STI.is64Bit() && CanCompress(ADDI16SPCompressLen))
2349 return ADDI16SPCompressLen;
2350 if (CanCompress(RVCompressLen))
2351 return RVCompressLen;
2352 }
2353 return 2048 - StackAlign;
2354 }
2355 return 0;
2356}
2357
2358bool RISCVFrameLowering::assignCalleeSavedSpillSlots(
2359 MachineFunction &MF, const TargetRegisterInfo *TRI,
2360 std::vector<CalleeSavedInfo> &CSI) const {
2361 auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
2362 MachineFrameInfo &MFI = MF.getFrameInfo();
2363 const TargetRegisterInfo *RegInfo = MF.getSubtarget().getRegisterInfo();
2364
2365 // Preemptible Interrupts have two additional Callee-save Frame Indexes,
2366 // not tracked by `CSI`.
2367 if (RVFI->isSiFivePreemptibleInterrupt(MF)) {
2368 for (int I = 0; I < 2; ++I) {
2369 int FI = RVFI->getInterruptCSRFrameIndex(Idx: I);
2370 MFI.setIsCalleeSavedObjectIndex(ObjectIdx: FI, IsCalleeSaved: true);
2371 }
2372 }
2373
2374 // Early exit if no callee saved registers are modified!
2375 if (CSI.empty())
2376 return true;
2377
2378 if (RVFI->useQCIInterrupt(MF)) {
2379 RVFI->setQCIInterruptStackSize(QCIInterruptPushAmount);
2380 }
2381
2382 if (RVFI->isPushable(MF)) {
2383 // Determine how many GPRs we need to push and save it to RVFI.
2384 unsigned PushedRegNum = getNumPushPopRegs(CSI);
2385
2386 // `QC.C.MIENTER(.NEST)` will save `ra` and `s0`, so we should only push if
2387 // we want to push more than 2 registers. Otherwise, we should push if we
2388 // want to push more than 0 registers.
2389 unsigned OnlyPushIfMoreThan = RVFI->useQCIInterrupt(MF) ? 2 : 0;
2390 if (PushedRegNum > OnlyPushIfMoreThan) {
2391 RVFI->setRVPushRegs(PushedRegNum);
2392 RVFI->setRVPushStackSize(alignTo(Value: (STI.getXLen() / 8) * PushedRegNum, Align: 16));
2393 }
2394 }
2395
2396 for (auto &CS : CSI) {
2397 MCRegister Reg = CS.getReg();
2398 const TargetRegisterClass *RC = RegInfo->getMinimalPhysRegClass(Reg);
2399 unsigned Size = RegInfo->getSpillSize(RC: *RC);
2400
2401 if (RVFI->useQCIInterrupt(MF)) {
2402 const auto *FFI = llvm::find_if(Range: FixedCSRFIQCIInterruptMap, P: [&](auto P) {
2403 return P.first == CS.getReg();
2404 });
2405 if (FFI != std::end(arr: FixedCSRFIQCIInterruptMap)) {
2406 int64_t Offset = FFI->second * (int64_t)Size;
2407
2408 int FrameIdx = MFI.CreateFixedSpillStackObject(Size, SPOffset: Offset);
2409 assert(FrameIdx < 0);
2410 CS.setFrameIdx(FrameIdx);
2411 continue;
2412 }
2413 }
2414
2415 if (RVFI->useSaveRestoreLibCalls(MF) || RVFI->isPushable(MF)) {
2416 const auto *FII = llvm::find_if(
2417 Range: FixedCSRFIMap, P: [&](MCPhysReg P) { return P == CS.getReg(); });
2418 unsigned RegNum = std::distance(first: std::begin(arr: FixedCSRFIMap), last: FII);
2419
2420 if (FII != std::end(arr: FixedCSRFIMap)) {
2421 int64_t Offset;
2422 if (RVFI->getPushPopKind(MF) ==
2423 RISCVMachineFunctionInfo::PushPopKind::StdExtZcmp)
2424 Offset = -int64_t(RVFI->getRVPushRegs() - RegNum) * Size;
2425 else
2426 Offset = -int64_t(RegNum + 1) * Size;
2427
2428 if (RVFI->useQCIInterrupt(MF))
2429 Offset -= QCIInterruptPushAmount;
2430
2431 int FrameIdx = MFI.CreateFixedSpillStackObject(Size, SPOffset: Offset);
2432 assert(FrameIdx < 0);
2433 CS.setFrameIdx(FrameIdx);
2434 continue;
2435 }
2436 }
2437
2438 // For GPRPair registers, use 8-byte slots with required alignment by zilsd.
2439 if (!STI.is64Bit() && STI.hasStdExtZilsd() &&
2440 RISCV::GPRPairRegClass.contains(Reg)) {
2441 Align PairAlign = STI.getZilsdAlign();
2442 int FrameIdx = MFI.CreateStackObject(Size: 8, Alignment: PairAlign, isSpillSlot: true);
2443 MFI.setIsCalleeSavedObjectIndex(ObjectIdx: FrameIdx, IsCalleeSaved: true);
2444 CS.setFrameIdx(FrameIdx);
2445 continue;
2446 }
2447
2448 // Not a fixed slot.
2449 Align Alignment = RegInfo->getSpillAlign(RC: *RC);
2450 // We may not be able to satisfy the desired alignment specification of
2451 // the TargetRegisterClass if the stack alignment is smaller. Use the
2452 // min.
2453 Alignment = std::min(a: Alignment, b: getStackAlign());
2454 int FrameIdx = MFI.CreateStackObject(Size, Alignment, isSpillSlot: true);
2455 MFI.setIsCalleeSavedObjectIndex(ObjectIdx: FrameIdx, IsCalleeSaved: true);
2456 CS.setFrameIdx(FrameIdx);
2457 if (RISCVRegisterInfo::isRVVRegClass(RC))
2458 MFI.setStackID(ObjectIdx: FrameIdx, ID: TargetStackID::ScalableVector);
2459 }
2460
2461 if (RVFI->useQCIInterrupt(MF)) {
2462 // Allocate a fixed object that covers the entire QCI stack allocation,
2463 // because there are gaps which are reserved for future use.
2464 MFI.CreateFixedSpillStackObject(
2465 Size: QCIInterruptPushAmount, SPOffset: -static_cast<int64_t>(QCIInterruptPushAmount));
2466 }
2467
2468 if (RVFI->isPushable(MF)) {
2469 int64_t QCIOffset = RVFI->useQCIInterrupt(MF) ? QCIInterruptPushAmount : 0;
2470 // Allocate a fixed object that covers the full push.
2471 if (int64_t PushSize = RVFI->getRVPushStackSize())
2472 MFI.CreateFixedSpillStackObject(Size: PushSize, SPOffset: -PushSize - QCIOffset);
2473 } else if (int LibCallRegs = getLibCallID(MF, CSI) + 1) {
2474 int64_t LibCallFrameSize =
2475 alignTo(Size: (STI.getXLen() / 8) * LibCallRegs, A: getStackAlign());
2476 MFI.CreateFixedSpillStackObject(Size: LibCallFrameSize, SPOffset: -LibCallFrameSize);
2477 }
2478
2479 return true;
2480}
2481
2482bool RISCVFrameLowering::spillCalleeSavedRegisters(
2483 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
2484 ArrayRef<CalleeSavedInfo> CSI, const TargetRegisterInfo *TRI) const {
2485 if (CSI.empty())
2486 return true;
2487
2488 MachineFunction *MF = MBB.getParent();
2489 const TargetInstrInfo &TII = *MF->getSubtarget().getInstrInfo();
2490 DebugLoc DL;
2491 if (MI != MBB.end() && !MI->isDebugInstr())
2492 DL = MI->getDebugLoc();
2493
2494 RISCVMachineFunctionInfo *RVFI = MF->getInfo<RISCVMachineFunctionInfo>();
2495 if (RVFI->useQCIInterrupt(MF: *MF)) {
2496 // Emit QC.C.MIENTER(.NEST)
2497 BuildMI(
2498 BB&: MBB, I: MI, MIMD: DL,
2499 MCID: TII.get(Opcode: RVFI->getInterruptStackKind(MF: *MF) ==
2500 RISCVMachineFunctionInfo::InterruptStackKind::QCINest
2501 ? RISCV::QC_C_MIENTER_NEST
2502 : RISCV::QC_C_MIENTER))
2503 .setMIFlag(MachineInstr::FrameSetup);
2504
2505 for (auto [Reg, _Offset] : FixedCSRFIQCIInterruptMap)
2506 MBB.addLiveIn(PhysReg: Reg);
2507 }
2508
2509 if (RVFI->isPushable(MF: *MF)) {
2510 // Emit CM.PUSH with base StackAdj & evaluate Push stack
2511 unsigned PushedRegNum = RVFI->getRVPushRegs();
2512 if (PushedRegNum > 0) {
2513 // Use encoded number to represent registers to spill.
2514 unsigned Opcode = getPushOpcode(
2515 Kind: RVFI->getPushPopKind(MF: *MF), UpdateFP: hasFP(MF: *MF) && !RVFI->useQCIInterrupt(MF: *MF));
2516 unsigned RegEnc = RISCVZC::encodeRegListNumRegs(NumRegs: PushedRegNum);
2517 MachineInstrBuilder PushBuilder =
2518 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII.get(Opcode))
2519 .setMIFlag(MachineInstr::FrameSetup);
2520 PushBuilder.addImm(Val: RegEnc);
2521 PushBuilder.addImm(Val: 0);
2522
2523 for (unsigned i = 0; i < PushedRegNum; i++)
2524 PushBuilder.addUse(RegNo: FixedCSRFIMap[i], Flags: RegState::Implicit);
2525 }
2526 } else if (const char *SpillLibCall = getSpillLibCallName(MF: *MF, CSI)) {
2527 // Add spill libcall via non-callee-saved register t0.
2528 MachineInstrBuilder NewMI =
2529 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII.get(Opcode: RISCV::PseudoCALLReg), DestReg: RISCV::X5)
2530 .addExternalSymbol(FnName: SpillLibCall, TargetFlags: RISCVII::MO_CALL)
2531 .setMIFlag(MachineInstr::FrameSetup)
2532 .addUse(RegNo: RISCV::X2, Flags: RegState::Implicit)
2533 .addDef(RegNo: RISCV::X2, Flags: RegState::ImplicitDefine);
2534
2535 // Add registers spilled as implicit used.
2536 for (auto &CS : CSI)
2537 NewMI.addUse(RegNo: CS.getReg(), Flags: RegState::Implicit);
2538 }
2539
2540 // Manually spill values not spilled by libcall & Push/Pop.
2541 const auto &UnmanagedCSI =
2542 getUnmanagedInterruptCSI(MF: *MF, CSI, ReverseOrder: STI.preferAscendingLoadStore());
2543 const auto &RVVCSI = getRVVCalleeSavedInfo(MF: *MF, CSI);
2544
2545 auto storeRegsToStackSlots = [&](ArrayRef<CalleeSavedInfo> CSInfo) {
2546 for (auto &CS : CSInfo) {
2547 // Insert the spill to the stack frame.
2548 MCRegister Reg = CS.getReg();
2549 const TargetRegisterClass *RC = TRI->getMinimalPhysRegClass(Reg);
2550 TII.storeRegToStackSlot(MBB, MI, SrcReg: Reg, isKill: !MBB.isLiveIn(Reg),
2551 FrameIndex: CS.getFrameIdx(), RC, VReg: Register(),
2552 Flags: MachineInstr::FrameSetup);
2553 }
2554 };
2555 storeRegsToStackSlots(UnmanagedCSI);
2556 storeRegsToStackSlots(RVVCSI);
2557
2558 return true;
2559}
2560
2561static unsigned getCalleeSavedRVVNumRegs(const Register &BaseReg) {
2562 return RISCV::VRRegClass.contains(Reg: BaseReg) ? 1
2563 : RISCV::VRM2RegClass.contains(Reg: BaseReg) ? 2
2564 : RISCV::VRM4RegClass.contains(Reg: BaseReg) ? 4
2565 : 8;
2566}
2567
2568void RISCVFrameLowering::emitCalleeSavedRVVPrologCFI(
2569 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, bool HasFP) const {
2570 MachineFunction *MF = MBB.getParent();
2571 const MachineFrameInfo &MFI = MF->getFrameInfo();
2572 RISCVMachineFunctionInfo *RVFI = MF->getInfo<RISCVMachineFunctionInfo>();
2573 const RISCVRegisterInfo &TRI = *STI.getRegisterInfo();
2574
2575 const auto &RVVCSI = getRVVCalleeSavedInfo(MF: *MF, CSI: MFI.getCalleeSavedInfo());
2576 if (RVVCSI.empty())
2577 return;
2578
2579 uint64_t FixedSize = getStackSizeWithRVVPadding(MF: *MF);
2580 if (!HasFP) {
2581 uint64_t ScalarLocalVarSize =
2582 MFI.getStackSize() - RVFI->getCalleeSavedStackSize() -
2583 RVFI->getVarArgsSaveSize() + RVFI->getRVVPadding();
2584 FixedSize -= ScalarLocalVarSize;
2585 }
2586
2587 CFIInstBuilder CFIBuilder(MBB, MI, MachineInstr::FrameSetup);
2588 for (auto &CS : RVVCSI) {
2589 // Insert the spill to the stack frame.
2590 int FI = CS.getFrameIdx();
2591 MCRegister BaseReg = getRVVBaseRegister(TRI, Reg: CS.getReg());
2592 unsigned NumRegs = getCalleeSavedRVVNumRegs(BaseReg: CS.getReg());
2593 for (unsigned i = 0; i < NumRegs; ++i) {
2594 CFIBuilder.insertCFIInst(CFIInst: createDefCFAOffset(
2595 TRI, Reg: BaseReg + i,
2596 Offset: StackOffset::get(Fixed: -FixedSize, Scalable: MFI.getObjectOffset(ObjectIdx: FI) / 8 + i)));
2597 }
2598 }
2599}
2600
2601void RISCVFrameLowering::emitCalleeSavedRVVEpilogCFI(
2602 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const {
2603 MachineFunction *MF = MBB.getParent();
2604 const MachineFrameInfo &MFI = MF->getFrameInfo();
2605 const RISCVRegisterInfo &TRI = *STI.getRegisterInfo();
2606
2607 CFIInstBuilder CFIHelper(MBB, MI, MachineInstr::FrameDestroy);
2608 const auto &RVVCSI = getRVVCalleeSavedInfo(MF: *MF, CSI: MFI.getCalleeSavedInfo());
2609 for (auto &CS : RVVCSI) {
2610 MCRegister BaseReg = getRVVBaseRegister(TRI, Reg: CS.getReg());
2611 unsigned NumRegs = getCalleeSavedRVVNumRegs(BaseReg: CS.getReg());
2612 for (unsigned i = 0; i < NumRegs; ++i)
2613 CFIHelper.buildRestore(Reg: BaseReg + i);
2614 }
2615}
2616
2617bool RISCVFrameLowering::restoreCalleeSavedRegisters(
2618 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
2619 MutableArrayRef<CalleeSavedInfo> CSI, const TargetRegisterInfo *TRI) const {
2620 if (CSI.empty())
2621 return true;
2622
2623 MachineFunction *MF = MBB.getParent();
2624 const TargetInstrInfo &TII = *MF->getSubtarget().getInstrInfo();
2625 DebugLoc DL;
2626 if (MI != MBB.end() && !MI->isDebugInstr())
2627 DL = MI->getDebugLoc();
2628
2629 // Manually restore values not restored by libcall & Push/Pop.
2630 // Reverse the restore order in epilog. In addition, the return
2631 // address will be restored first in the epilogue. It increases
2632 // the opportunity to avoid the load-to-use data hazard between
2633 // loading RA and return by RA. loadRegFromStackSlot can insert
2634 // multiple instructions.
2635 const auto &UnmanagedCSI =
2636 getUnmanagedInterruptCSI(MF: *MF, CSI, ReverseOrder: STI.preferAscendingLoadStore());
2637 const auto &RVVCSI = getRVVCalleeSavedInfo(MF: *MF, CSI);
2638
2639 auto loadRegFromStackSlot = [&](ArrayRef<CalleeSavedInfo> CSInfo) {
2640 for (auto &CS : CSInfo) {
2641 MCRegister Reg = CS.getReg();
2642 const TargetRegisterClass *RC = TRI->getMinimalPhysRegClass(Reg);
2643 TII.loadRegFromStackSlot(MBB, MI, DestReg: Reg, FrameIndex: CS.getFrameIdx(), RC, VReg: Register(),
2644 SubReg: RISCV::NoSubRegister,
2645 Flags: MachineInstr::FrameDestroy);
2646 assert(MI != MBB.begin() &&
2647 "loadRegFromStackSlot didn't insert any code!");
2648 }
2649 };
2650 loadRegFromStackSlot(RVVCSI);
2651 loadRegFromStackSlot(UnmanagedCSI);
2652
2653 RISCVMachineFunctionInfo *RVFI = MF->getInfo<RISCVMachineFunctionInfo>();
2654 if (RVFI->useQCIInterrupt(MF: *MF)) {
2655 // Don't emit anything here because restoration is handled by
2656 // QC.C.MILEAVERET which we already inserted to return.
2657 assert(MI->getOpcode() == RISCV::QC_C_MILEAVERET &&
2658 "Unexpected QCI Interrupt Return Instruction");
2659 }
2660
2661 if (RVFI->isPushable(MF: *MF)) {
2662 unsigned PushedRegNum = RVFI->getRVPushRegs();
2663 if (PushedRegNum > 0) {
2664 unsigned Opcode = getPopOpcode(Kind: RVFI->getPushPopKind(MF: *MF));
2665 unsigned RegEnc = RISCVZC::encodeRegListNumRegs(NumRegs: PushedRegNum);
2666 MachineInstrBuilder PopBuilder =
2667 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII.get(Opcode))
2668 .setMIFlag(MachineInstr::FrameDestroy);
2669 // Use encoded number to represent registers to restore.
2670 PopBuilder.addImm(Val: RegEnc);
2671 PopBuilder.addImm(Val: 0);
2672
2673 for (unsigned i = 0; i < RVFI->getRVPushRegs(); i++)
2674 PopBuilder.addDef(RegNo: FixedCSRFIMap[i], Flags: RegState::ImplicitDefine);
2675 }
2676 } else if (const char *RestoreLibCall = getRestoreLibCallName(MF: *MF, CSI)) {
2677 // Restore is not compatible with shadow stacks, so do the restore manually.
2678 // This ensures we still get the code size saving of `__riscv_save_<N>`.
2679 if (RVFI->hasShadowStack(MF: *MF)) {
2680 loadRegFromStackSlot(CSI);
2681 return true;
2682 }
2683
2684 // Add restore libcall via tail call.
2685 MachineInstrBuilder NewMI =
2686 BuildMI(BB&: MBB, I: MI, MIMD: DL, MCID: TII.get(Opcode: RISCV::PseudoTAIL))
2687 .addExternalSymbol(FnName: RestoreLibCall, TargetFlags: RISCVII::MO_CALL)
2688 .setMIFlag(MachineInstr::FrameDestroy)
2689 .addDef(RegNo: RISCV::X2, Flags: RegState::ImplicitDefine);
2690
2691 // Add registers restored as implicit defined.
2692 for (auto &CS : CSI)
2693 NewMI.addDef(RegNo: CS.getReg(), Flags: RegState::ImplicitDefine);
2694
2695 // Remove trailing returns, since the terminator is now a tail call to the
2696 // restore function.
2697 if (MI != MBB.end() && MI->getOpcode() == RISCV::PseudoRET) {
2698 NewMI.getInstr()->copyImplicitOps(MF&: *MF, MI: *MI);
2699 MI->eraseFromParent();
2700 }
2701 }
2702 return true;
2703}
2704
2705bool RISCVFrameLowering::enableShrinkWrapping(const MachineFunction &MF) const {
2706 // Keep the conventional code flow when not optimizing.
2707 if (MF.getFunction().hasOptNone())
2708 return false;
2709
2710 // QCI and SiFive CLIC interrupt entry sequences must precede all handler
2711 // code.
2712 const auto *RVFI = MF.getInfo<RISCVMachineFunctionInfo>();
2713 if (RVFI->useQCIInterrupt(MF) || RVFI->useSiFiveInterrupt(MF))
2714 return false;
2715
2716 return true;
2717}
2718
2719bool RISCVFrameLowering::canUseAsPrologue(const MachineBasicBlock &MBB) const {
2720 MachineBasicBlock *TmpMBB = const_cast<MachineBasicBlock *>(&MBB);
2721 const MachineFunction *MF = MBB.getParent();
2722 const auto *RVFI = MF->getInfo<RISCVMachineFunctionInfo>();
2723
2724 // Make sure VTYPE and VL are not live-in since we will use vsetvli in the
2725 // prologue to get the VLEN, and that will clobber these registers.
2726 //
2727 // We may do also check the stack contains objects with scalable vector type,
2728 // but this will require iterating over all the stack objects, but this may
2729 // not worth since the situation is rare, we could do further check in future
2730 // if we find it is necessary.
2731 if (STI.preferVsetvliOverReadVLENB() &&
2732 (MBB.isLiveIn(Reg: RISCV::VTYPE) || MBB.isLiveIn(Reg: RISCV::VL)))
2733 return false;
2734
2735 if (!RVFI->useSaveRestoreLibCalls(MF: *MF))
2736 return true;
2737
2738 // Inserting a call to a __riscv_save libcall requires the use of the register
2739 // t0 (X5) to hold the return address. Therefore if this register is already
2740 // used we can't insert the call.
2741
2742 RegScavenger RS;
2743 RS.enterBasicBlock(MBB&: *TmpMBB);
2744 return !RS.isRegUsed(Reg: RISCV::X5);
2745}
2746
2747bool RISCVFrameLowering::canUseAsEpilogue(const MachineBasicBlock &MBB) const {
2748 const MachineFunction *MF = MBB.getParent();
2749 MachineBasicBlock *TmpMBB = const_cast<MachineBasicBlock *>(&MBB);
2750 const auto *RVFI = MF->getInfo<RISCVMachineFunctionInfo>();
2751
2752 // If we have a shadow stack, we don't use `__riscv_restore` libcalls.
2753 if (RVFI->hasShadowStack(MF: *MF))
2754 return true;
2755
2756 if (!RVFI->useSaveRestoreLibCalls(MF: *MF))
2757 return true;
2758
2759 // Using the __riscv_restore libcalls to restore CSRs requires a tail call.
2760 // This means if we still need to continue executing code within this function
2761 // the restore cannot take place in this basic block.
2762
2763 if (MBB.succ_size() > 1)
2764 return false;
2765
2766 MachineBasicBlock *SuccMBB =
2767 MBB.succ_empty() ? TmpMBB->getFallThrough() : *MBB.succ_begin();
2768
2769 // Doing a tail call should be safe if there are no successors, because either
2770 // we have a returning block or the end of the block is unreachable, so the
2771 // restore will be eliminated regardless.
2772 if (!SuccMBB)
2773 return true;
2774
2775 // The successor can only contain a return and debug instructions, since we
2776 // would effectively replace it with our own tail return at the end of this
2777 // block. The debug instructions would not execute on the tail-return path.
2778 return SuccMBB->isReturnBlock() &&
2779 llvm::count_if(Range: SuccMBB->instrs(), P: [](const MachineInstr &MI) {
2780 return !MI.isDebugInstr();
2781 }) == 1;
2782}
2783
2784bool RISCVFrameLowering::isSupportedStackID(TargetStackID::Value ID) const {
2785 return ID == TargetStackID::Default || ID == TargetStackID::ScalableVector;
2786}
2787
2788TargetStackID::Value RISCVFrameLowering::getStackIDForScalableVectors() const {
2789 return TargetStackID::ScalableVector;
2790}
2791
2792// Synthesize the probe loop.
2793static void emitStackProbeInline(MachineBasicBlock::iterator MBBI, DebugLoc DL,
2794 Register TargetReg, Register ScratchReg,
2795 bool IsRVV) {
2796 assert(TargetReg != RISCV::X2 && "New top of stack cannot already be in SP");
2797 assert(ScratchReg != RISCV::X2 && "Scratch register cannot be SP");
2798 assert(TargetReg != ScratchReg && "Target and scratch must be different");
2799
2800 MachineBasicBlock &MBB = *MBBI->getParent();
2801 MachineFunction &MF = *MBB.getParent();
2802
2803 auto &Subtarget = MF.getSubtarget<RISCVSubtarget>();
2804 const RISCVInstrInfo *TII = Subtarget.getInstrInfo();
2805 bool IsRV64 = Subtarget.is64Bit();
2806 Align StackAlign = Subtarget.getFrameLowering()->getStackAlign();
2807 const RISCVTargetLowering *TLI = Subtarget.getTargetLowering();
2808 uint64_t ProbeSize = TLI->getStackProbeSize(MF, StackAlign);
2809
2810 MachineFunction::iterator MBBInsertPoint = std::next(x: MBB.getIterator());
2811 MachineBasicBlock *LoopTestMBB =
2812 MF.CreateMachineBasicBlock(BB: MBB.getBasicBlock());
2813 MF.insert(MBBI: MBBInsertPoint, MBB: LoopTestMBB);
2814 MachineBasicBlock *ExitMBB = MF.CreateMachineBasicBlock(BB: MBB.getBasicBlock());
2815 MF.insert(MBBI: MBBInsertPoint, MBB: ExitMBB);
2816 MachineInstr::MIFlag Flags = MachineInstr::FrameSetup;
2817
2818 // ScratchReg = ProbeSize
2819 TII->movImm(MBB, MBBI, DL, DstReg: ScratchReg, Val: ProbeSize, Flag: Flags);
2820
2821 // LoopTest:
2822 // SUB SP, SP, ProbeSize
2823 BuildMI(BB&: *LoopTestMBB, I: LoopTestMBB->end(), MIMD: DL, MCID: TII->get(Opcode: RISCV::SUB), DestReg: SPReg)
2824 .addReg(RegNo: SPReg)
2825 .addReg(RegNo: ScratchReg)
2826 .setMIFlags(Flags);
2827
2828 // s[d|w] zero, 0(sp)
2829 BuildMI(BB&: *LoopTestMBB, I: LoopTestMBB->end(), MIMD: DL,
2830 MCID: TII->get(Opcode: IsRV64 ? RISCV::SD : RISCV::SW))
2831 .addReg(RegNo: RISCV::X0)
2832 .addReg(RegNo: SPReg)
2833 .addImm(Val: 0)
2834 .setMIFlags(Flags);
2835
2836 if (IsRVV) {
2837 // SUB TargetReg, TargetReg, ProbeSize
2838 BuildMI(BB&: *LoopTestMBB, I: LoopTestMBB->end(), MIMD: DL, MCID: TII->get(Opcode: RISCV::SUB),
2839 DestReg: TargetReg)
2840 .addReg(RegNo: TargetReg)
2841 .addReg(RegNo: ScratchReg)
2842 .setMIFlags(Flags);
2843
2844 // BGE TargetReg, ProbeSize, LoopTest
2845 BuildMI(BB&: *LoopTestMBB, I: LoopTestMBB->end(), MIMD: DL, MCID: TII->get(Opcode: RISCV::BGE))
2846 .addReg(RegNo: TargetReg)
2847 .addReg(RegNo: ScratchReg)
2848 .addMBB(MBB: LoopTestMBB)
2849 .setMIFlags(Flags);
2850
2851 } else {
2852 // BNE SP, TargetReg, LoopTest
2853 BuildMI(BB&: *LoopTestMBB, I: LoopTestMBB->end(), MIMD: DL, MCID: TII->get(Opcode: RISCV::BNE))
2854 .addReg(RegNo: SPReg)
2855 .addReg(RegNo: TargetReg)
2856 .addMBB(MBB: LoopTestMBB)
2857 .setMIFlags(Flags);
2858 }
2859
2860 ExitMBB->splice(Where: ExitMBB->end(), Other: &MBB, From: std::next(x: MBBI), To: MBB.end());
2861 ExitMBB->transferSuccessorsAndUpdatePHIs(FromMBB: &MBB);
2862
2863 LoopTestMBB->addSuccessor(Succ: ExitMBB);
2864 LoopTestMBB->addSuccessor(Succ: LoopTestMBB);
2865 MBB.addSuccessor(Succ: LoopTestMBB);
2866 // Update liveins.
2867 fullyRecomputeLiveIns(MBBs: {ExitMBB, LoopTestMBB});
2868}
2869
2870void RISCVFrameLowering::inlineStackProbe(MachineFunction &MF,
2871 MachineBasicBlock &MBB) const {
2872 // Get the instructions that need to be replaced. We emit at most two of
2873 // these. Remember them in order to avoid complications coming from the need
2874 // to traverse the block while potentially creating more blocks.
2875 SmallVector<MachineInstr *, 4> ToReplace;
2876 for (MachineInstr &MI : MBB) {
2877 unsigned Opc = MI.getOpcode();
2878 if (Opc == RISCV::PROBED_STACKALLOC ||
2879 Opc == RISCV::PROBED_STACKALLOC_RVV) {
2880 ToReplace.push_back(Elt: &MI);
2881 }
2882 }
2883
2884 for (MachineInstr *MI : ToReplace) {
2885 if (MI->getOpcode() == RISCV::PROBED_STACKALLOC ||
2886 MI->getOpcode() == RISCV::PROBED_STACKALLOC_RVV) {
2887 MachineBasicBlock::iterator MBBI = MI->getIterator();
2888 DebugLoc DL = MBB.findDebugLoc(MBBI);
2889 Register TargetReg = MI->getOperand(i: 0).getReg();
2890
2891 Register ScratchReg =
2892 findScratchNonCalleeSaveRegister(MBB: &MBB, PreferredReg: RISCV::X7, DontUseReg: TargetReg);
2893
2894 assert(ScratchReg.isValid() &&
2895 "No available scratch register for stack probe loop");
2896
2897 emitStackProbeInline(MBBI, DL, TargetReg, ScratchReg,
2898 IsRVV: (MI->getOpcode() == RISCV::PROBED_STACKALLOC_RVV));
2899 MBBI->eraseFromParent();
2900 }
2901 }
2902}
2903
2904int RISCVFrameLowering::getInitialCFAOffset(const MachineFunction &MF) const {
2905 return 0;
2906}
2907
2908Register
2909RISCVFrameLowering::getInitialCFARegister(const MachineFunction &MF) const {
2910 return RISCV::X2;
2911}
2912
2913// On 64-bit systems the fixed stack can hold INT64_MAX bytes, since
2914// stack-offset calculation is done in 2s-complement.
2915// NOTE: In theory a register can hold any 64-bit number, so this constraint
2916// might be relaxed to UINT64_MAX in the future, if anyone actually needs
2917// that.
2918uint64_t RISCVFrameLowering::getStackThreshold() const {
2919 return STI.is64Bit() ? INT64_MAX : UINT32_MAX;
2920}
2921