1//===-- X86AsmBackend.cpp - X86 Assembler Backend -------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "MCTargetDesc/X86BaseInfo.h"
10#include "MCTargetDesc/X86EncodingOptimization.h"
11#include "MCTargetDesc/X86FixupKinds.h"
12#include "MCTargetDesc/X86MCAsmInfo.h"
13#include "MCTargetDesc/X86MCOptions.h"
14#include "llvm/ADT/StringSwitch.h"
15#include "llvm/BinaryFormat/ELF.h"
16#include "llvm/BinaryFormat/MachO.h"
17#include "llvm/MC/MCAsmBackend.h"
18#include "llvm/MC/MCAssembler.h"
19#include "llvm/MC/MCCodeEmitter.h"
20#include "llvm/MC/MCContext.h"
21#include "llvm/MC/MCDwarf.h"
22#include "llvm/MC/MCELFObjectWriter.h"
23#include "llvm/MC/MCELFStreamer.h"
24#include "llvm/MC/MCExpr.h"
25#include "llvm/MC/MCInst.h"
26#include "llvm/MC/MCInstrInfo.h"
27#include "llvm/MC/MCLFIRewriter.h"
28#include "llvm/MC/MCObjectStreamer.h"
29#include "llvm/MC/MCObjectWriter.h"
30#include "llvm/MC/MCRegisterInfo.h"
31#include "llvm/MC/MCSection.h"
32#include "llvm/MC/MCSubtargetInfo.h"
33#include "llvm/MC/MCTargetOptions.h"
34#include "llvm/MC/MCValue.h"
35#include "llvm/MC/TargetRegistry.h"
36#include "llvm/Support/ErrorHandling.h"
37#include "llvm/Support/raw_ostream.h"
38
39using namespace llvm;
40
41namespace {
42/// A wrapper for holding a mask of the values from X86::AlignBranchBoundaryKind
43class X86AlignBranchKind {
44private:
45 uint8_t AlignBranchKind = 0;
46
47public:
48 X86AlignBranchKind() = default;
49 explicit X86AlignBranchKind(StringRef Val) {
50 SmallVector<StringRef, 6> BranchTypes;
51 Val.split(A&: BranchTypes, Separator: '+', MaxSplit: -1, KeepEmpty: false);
52 for (auto BranchType : BranchTypes) {
53 if (BranchType == "fused")
54 addKind(Value: X86::AlignBranchFused);
55 else if (BranchType == "jcc")
56 addKind(Value: X86::AlignBranchJcc);
57 else if (BranchType == "jmp")
58 addKind(Value: X86::AlignBranchJmp);
59 else if (BranchType == "call")
60 addKind(Value: X86::AlignBranchCall);
61 else if (BranchType == "ret")
62 addKind(Value: X86::AlignBranchRet);
63 else if (BranchType == "indirect")
64 addKind(Value: X86::AlignBranchIndirect);
65 else {
66 errs() << "invalid argument " << BranchType.str()
67 << " to -x86-align-branch=; each element must be one of: fused, "
68 "jcc, jmp, call, ret, indirect.(plus separated)\n";
69 }
70 }
71 }
72
73 operator uint8_t() const { return AlignBranchKind; }
74 void addKind(X86::AlignBranchBoundaryKind Value) { AlignBranchKind |= Value; }
75};
76
77class X86AsmBackend : public MCAsmBackend {
78 const X86MCOptions &CLOpts;
79 const MCSubtargetInfo &STI;
80 std::unique_ptr<const MCInstrInfo> MCII;
81 X86AlignBranchKind AlignBranchType;
82 Align AlignBoundary;
83 unsigned TargetPrefixMax = 0;
84
85 MCInst PrevInst;
86 unsigned PrevInstOpcode = 0;
87 bool PrefixEndsBundleLock = false;
88 MCBoundaryAlignFragment *PendingBA = nullptr;
89 std::pair<MCFragment *, size_t> PrevInstPosition;
90
91 uint8_t determinePaddingPrefix(const MCInst &Inst) const;
92 bool isMacroFused(const MCInst &Cmp, const MCInst &Jcc) const;
93 bool needAlign(const MCInst &Inst) const;
94 bool canPadBranches(MCObjectStreamer &OS) const;
95 bool canPadInst(const MCInst &Inst, MCObjectStreamer &OS) const;
96 void emitInstructionBeginBundle(MCObjectStreamer &OS);
97 void emitInstructionEndBundle(MCObjectStreamer &OS);
98
99public:
100 X86AsmBackend(const Target &T, const MCSubtargetInfo &STI)
101 : MCAsmBackend(llvm::endianness::little), CLOpts(X86MCOptions::Global),
102 STI(STI), MCII(T.createMCInstrInfo()) {
103 if (CLOpts.branches_within_32B_boundaries) {
104 // At the moment, this defaults to aligning fused branches, unconditional
105 // jumps, and (unfused) conditional jumps with nops. Both the
106 // instructions aligned and the alignment method (nop vs prefix) may
107 // change in the future.
108 AlignBoundary = assumeAligned(Value: 32);
109 AlignBranchType.addKind(Value: X86::AlignBranchFused);
110 AlignBranchType.addKind(Value: X86::AlignBranchJcc);
111 AlignBranchType.addKind(Value: X86::AlignBranchJmp);
112 }
113 // Allow overriding defaults set by main flag
114 if (CLOpts.align_branch_boundary)
115 AlignBoundary = assumeAligned(Value: *CLOpts.align_branch_boundary);
116 if (CLOpts.align_branch)
117 AlignBranchType = X86AlignBranchKind(*CLOpts.align_branch);
118 if (CLOpts.pad_max_prefix_size)
119 TargetPrefixMax = *CLOpts.pad_max_prefix_size;
120
121 AllowAutoPadding =
122 AlignBoundary != Align(1) && AlignBranchType != X86::AlignBranchNone;
123 AllowEnhancedRelaxation =
124 AllowAutoPadding && TargetPrefixMax != 0 && CLOpts.pad_for_branch_align;
125 AllowBundling = true;
126 }
127
128 // The streamer frees the fragments these point into.
129 void reset() override {
130 PrevInst = MCInst();
131 PrevInstOpcode = 0;
132 PrefixEndsBundleLock = false;
133 PendingBA = nullptr;
134 PrevInstPosition = {};
135 }
136
137 void emitInstructionBegin(MCObjectStreamer &OS, const MCInst &Inst,
138 const MCSubtargetInfo &STI);
139 void emitInstructionEnd(MCObjectStreamer &OS, const MCInst &Inst);
140
141
142 std::optional<MCFixupKind> getFixupKind(StringRef Name) const override;
143
144 MCFixupKindInfo getFixupKindInfo(MCFixupKind Kind) const override;
145
146 std::optional<bool> evaluateFixup(const MCFragment &, MCFixup &, MCValue &,
147 uint64_t &) override;
148 void applyFixup(const MCFragment &, const MCFixup &, const MCValue &Target,
149 uint8_t *Data, uint64_t Value, bool IsResolved) override;
150
151 bool mayNeedRelaxation(unsigned Opcode, ArrayRef<MCOperand> Operands,
152 const MCSubtargetInfo &STI) const override;
153
154 bool fixupNeedsRelaxationAdvanced(const MCFragment &, const MCFixup &,
155 const MCValue &, uint64_t,
156 bool) const override;
157
158 void relaxInstruction(MCInst &Inst,
159 const MCSubtargetInfo &STI) const override;
160
161 bool padInstructionViaRelaxation(MCFragment &RF, MCCodeEmitter &Emitter,
162 unsigned &RemainingSize) const;
163
164 bool padInstructionViaPrefix(MCFragment &RF, MCCodeEmitter &Emitter,
165 unsigned &RemainingSize) const;
166
167 bool padInstructionEncoding(MCFragment &RF, MCCodeEmitter &Emitter,
168 unsigned &RemainingSize) const;
169
170 bool finishLayout() const override;
171
172 bool padInstsBackward(SmallVectorImpl<MCFragment *> &Relaxable,
173 unsigned &RemainingSize) const;
174 bool foldBundlePad(const MCAssembler &Asm, MCBoundaryAlignFragment &BF,
175 SmallVectorImpl<MCFragment *> &Relaxable) const;
176 bool optimizeBundleNops(const MCAssembler &Asm) const;
177
178 unsigned getMaximumNopSize(const MCSubtargetInfo &STI) const override;
179
180 bool writeNopData(raw_ostream &OS, uint64_t Count,
181 const MCSubtargetInfo *STI) const override;
182};
183} // end anonymous namespace
184
185static bool isRelaxableBranch(unsigned Opcode) {
186 return Opcode == X86::JCC_1 || Opcode == X86::JMP_1;
187}
188
189static unsigned getRelaxedOpcodeBranch(unsigned Opcode,
190 bool Is16BitMode = false) {
191 switch (Opcode) {
192 default:
193 llvm_unreachable("invalid opcode for branch");
194 case X86::JCC_1:
195 return (Is16BitMode) ? X86::JCC_2 : X86::JCC_4;
196 case X86::JMP_1:
197 return (Is16BitMode) ? X86::JMP_2 : X86::JMP_4;
198 }
199}
200
201static unsigned getRelaxedOpcode(const MCInst &MI, bool Is16BitMode) {
202 unsigned Opcode = MI.getOpcode();
203 return isRelaxableBranch(Opcode) ? getRelaxedOpcodeBranch(Opcode, Is16BitMode)
204 : X86::getOpcodeForLongImmediateForm(Opcode);
205}
206
207static X86::CondCode getCondFromBranch(const MCInst &MI,
208 const MCInstrInfo &MCII) {
209 unsigned Opcode = MI.getOpcode();
210 switch (Opcode) {
211 default:
212 return X86::COND_INVALID;
213 case X86::JCC_1: {
214 const MCInstrDesc &Desc = MCII.get(Opcode);
215 return static_cast<X86::CondCode>(
216 MI.getOperand(i: Desc.getNumOperands() - 1).getImm());
217 }
218 }
219}
220
221static X86::SecondMacroFusionInstKind
222classifySecondInstInMacroFusion(const MCInst &MI, const MCInstrInfo &MCII) {
223 X86::CondCode CC = getCondFromBranch(MI, MCII);
224 return classifySecondCondCodeInMacroFusion(CC);
225}
226
227/// Check if the instruction uses RIP relative addressing.
228static bool isRIPRelative(const MCInst &MI, const MCInstrInfo &MCII) {
229 const MCInstrDesc &Desc = MCII.get(Opcode: MI.getOpcode());
230 int MemoryOperand = X86II::getMemoryOperandIdx(Desc);
231 if (MemoryOperand < 0)
232 return false;
233 unsigned BaseRegNum = MemoryOperand + X86::AddrBaseReg;
234 MCRegister BaseReg = MI.getOperand(i: BaseRegNum).getReg();
235 return (BaseReg == X86::RIP);
236}
237
238/// Check if the instruction is a prefix.
239static bool isPrefix(unsigned Opcode, const MCInstrInfo &MCII) {
240 return X86II::isPrefix(TSFlags: MCII.get(Opcode).TSFlags);
241}
242
243/// Check if the instruction is valid as the first instruction in macro fusion.
244static bool isFirstMacroFusibleInst(const MCInst &Inst,
245 const MCInstrInfo &MCII) {
246 // An Intel instruction with RIP relative addressing is not macro fusible.
247 if (isRIPRelative(MI: Inst, MCII))
248 return false;
249 X86::FirstMacroFusionInstKind FIK =
250 X86::classifyFirstOpcodeInMacroFusion(Opcode: Inst.getOpcode());
251 return FIK != X86::FirstMacroFusionInstKind::Invalid;
252}
253
254/// X86 can reduce the bytes of NOP by padding instructions with prefixes to
255/// get a better peformance in some cases. Here, we determine which prefix is
256/// the most suitable.
257///
258/// If the instruction has a segment override prefix, use the existing one.
259/// If the target is 64-bit, use the CS.
260/// If the target is 32-bit,
261/// - If the instruction has a ESP/EBP base register, use SS.
262/// - Otherwise use DS.
263uint8_t X86AsmBackend::determinePaddingPrefix(const MCInst &Inst) const {
264 assert((STI.hasFeature(X86::Is32Bit) || STI.hasFeature(X86::Is64Bit)) &&
265 "Prefixes can be added only in 32-bit or 64-bit mode.");
266 const MCInstrDesc &Desc = MCII->get(Opcode: Inst.getOpcode());
267 uint64_t TSFlags = Desc.TSFlags;
268
269 // Determine where the memory operand starts, if present.
270 int MemoryOperand = X86II::getMemoryOperandIdx(Desc);
271
272 MCRegister SegmentReg;
273 if (MemoryOperand >= 0) {
274 // Check for explicit segment override on memory operand.
275 SegmentReg = Inst.getOperand(i: MemoryOperand + X86::AddrSegmentReg).getReg();
276 }
277
278 switch (TSFlags & X86II::FormMask) {
279 default:
280 break;
281 case X86II::RawFrmDstSrc: {
282 // Check segment override opcode prefix as needed (not for %ds).
283 if (Inst.getOperand(i: 2).getReg() != X86::DS)
284 SegmentReg = Inst.getOperand(i: 2).getReg();
285 break;
286 }
287 case X86II::RawFrmSrc: {
288 // Check segment override opcode prefix as needed (not for %ds).
289 if (Inst.getOperand(i: 1).getReg() != X86::DS)
290 SegmentReg = Inst.getOperand(i: 1).getReg();
291 break;
292 }
293 case X86II::RawFrmMemOffs: {
294 // Check segment override opcode prefix as needed.
295 SegmentReg = Inst.getOperand(i: 1).getReg();
296 break;
297 }
298 }
299
300 if (SegmentReg)
301 return X86::getSegmentOverridePrefixForReg(Reg: SegmentReg);
302
303 if (STI.hasFeature(Feature: X86::Is64Bit))
304 return X86::CS_Encoding;
305
306 if (MemoryOperand >= 0) {
307 unsigned BaseRegNum = MemoryOperand + X86::AddrBaseReg;
308 MCRegister BaseReg = Inst.getOperand(i: BaseRegNum).getReg();
309 if (BaseReg == X86::ESP || BaseReg == X86::EBP)
310 return X86::SS_Encoding;
311 }
312 return X86::DS_Encoding;
313}
314
315/// Check if the two instructions will be macro-fused on the target cpu.
316bool X86AsmBackend::isMacroFused(const MCInst &Cmp, const MCInst &Jcc) const {
317 const MCInstrDesc &InstDesc = MCII->get(Opcode: Jcc.getOpcode());
318 if (!InstDesc.isConditionalBranch())
319 return false;
320 if (!isFirstMacroFusibleInst(Inst: Cmp, MCII: *MCII))
321 return false;
322 const X86::FirstMacroFusionInstKind CmpKind =
323 X86::classifyFirstOpcodeInMacroFusion(Opcode: Cmp.getOpcode());
324 const X86::SecondMacroFusionInstKind BranchKind =
325 classifySecondInstInMacroFusion(MI: Jcc, MCII: *MCII);
326 return X86::isMacroFused(FirstKind: CmpKind, SecondKind: BranchKind);
327}
328
329/// Check if the instruction has a variant symbol operand.
330static bool hasVariantSymbol(const MCInst &MI) {
331 for (auto &Operand : MI) {
332 if (!Operand.isExpr())
333 continue;
334 const MCExpr &Expr = *Operand.getExpr();
335 if (Expr.getKind() == MCExpr::SymbolRef &&
336 cast<MCSymbolRefExpr>(Val: &Expr)->getSpecifier())
337 return true;
338 }
339 return false;
340}
341
342/// X86 has certain instructions which enable interrupts exactly one
343/// instruction *after* the instruction which stores to SS. Return true if the
344/// given instruction may have such an interrupt delay slot.
345static bool mayHaveInterruptDelaySlot(unsigned InstOpcode) {
346 switch (InstOpcode) {
347 case X86::POPSS16:
348 case X86::POPSS32:
349 case X86::STI:
350 return true;
351
352 case X86::MOV16sr:
353 case X86::MOV32sr:
354 case X86::MOV64sr:
355 case X86::MOV16sm:
356 // In fact, this is only the case if the first operand is SS. However, as
357 // segment moves occur extremely rarely, this is just a minor pessimization.
358 return true;
359 }
360 return false;
361}
362
363/// Return true if we can insert NOP or prefixes automatically before the
364/// the instruction to be emitted.
365bool X86AsmBackend::canPadInst(const MCInst &Inst, MCObjectStreamer &OS) const {
366 if (hasVariantSymbol(MI: Inst))
367 // Linker may rewrite the instruction with variant symbol operand(e.g.
368 // TLSCALL).
369 return false;
370
371 if (mayHaveInterruptDelaySlot(InstOpcode: PrevInstOpcode))
372 // If this instruction follows an interrupt enabling instruction with a one
373 // instruction delay, inserting a nop would change behavior.
374 return false;
375
376 if (isPrefix(Opcode: PrevInstOpcode, MCII: *MCII))
377 // If this instruction follows a prefix, inserting a nop/prefix would change
378 // semantic.
379 return false;
380
381 if (isPrefix(Opcode: Inst.getOpcode(), MCII: *MCII))
382 // If this instruction is a prefix, inserting a prefix would change
383 // semantic.
384 return false;
385
386 // If this instruction follows any data, there is no clear instruction
387 // boundary, inserting a nop/prefix would change semantic.
388 auto Offset = OS.getCurFragSize();
389 if (Offset && (OS.getCurrentFragment() != PrevInstPosition.first ||
390 Offset != PrevInstPosition.second))
391 return false;
392
393 return true;
394}
395
396bool X86AsmBackend::canPadBranches(MCObjectStreamer &OS) const {
397 if (!OS.getAllowAutoPadding())
398 return false;
399 assert(allowAutoPadding() && "incorrect initialization!");
400
401 // We only pad in text section.
402 if (!OS.getCurrentSectionOnly()->isText())
403 return false;
404
405 // Branches only need to be aligned in 32-bit or 64-bit mode.
406 if (!(STI.hasFeature(Feature: X86::Is64Bit) || STI.hasFeature(Feature: X86::Is32Bit)))
407 return false;
408
409 return true;
410}
411
412/// Check if the instruction operand needs to be aligned.
413bool X86AsmBackend::needAlign(const MCInst &Inst) const {
414 const MCInstrDesc &Desc = MCII->get(Opcode: Inst.getOpcode());
415 return (Desc.isConditionalBranch() &&
416 (AlignBranchType & X86::AlignBranchJcc)) ||
417 (Desc.isUnconditionalBranch() &&
418 (AlignBranchType & X86::AlignBranchJmp)) ||
419 (Desc.isCall() && (AlignBranchType & X86::AlignBranchCall)) ||
420 (Desc.isReturn() && (AlignBranchType & X86::AlignBranchRet)) ||
421 (Desc.isIndirectBranch() &&
422 (AlignBranchType & X86::AlignBranchIndirect));
423}
424
425void X86_MC::emitInstruction(MCObjectStreamer &S, const MCInst &Inst,
426 const MCSubtargetInfo &STI) {
427 bool AutoPadding = S.getAllowAutoPadding();
428 if (LLVM_LIKELY(!AutoPadding && !X86MCOptions::Global.pad_for_align)) {
429 S.MCObjectStreamer::emitInstruction(Inst, STI);
430 return;
431 }
432
433 // Run the LFI rewriter outside of emitInstructionBegin/End so that nested
434 // instructions or bundle_lock/unlock directives do not corrupt the Begin/End
435 // bookkeeping for the original instruction.
436 if (S.getLFIRewriter() && S.getLFIRewriter()->rewriteInst(Inst, Out&: S, STI))
437 return;
438
439 auto &Backend = static_cast<X86AsmBackend &>(S.getAssembler().getBackend());
440 Backend.emitInstructionBegin(OS&: S, Inst, STI);
441 S.MCObjectStreamer::emitInstruction(Inst, STI);
442 Backend.emitInstructionEnd(OS&: S, Inst);
443}
444
445/// Open a MCBoundaryAlignFragment for the upcoming instruction so that layout
446/// can pad it into the next bundle. Within .bundle_lock the group's fragment
447/// already covers it.
448void X86AsmBackend::emitInstructionBeginBundle(MCObjectStreamer &OS) {
449 assert(Asm->isBundlingEnabled());
450
451 // The prefix stays in the group while this instruction gets its own
452 // fragment, so padding may land between the two.
453 if (PrefixEndsBundleLock && OS.getCurrentFragment() != PrevInstPosition.first)
454 getContext().reportError(L: OS.getStartTokLoc(),
455 Msg: "instruction prefix cannot be the last "
456 "instruction of a .bundle_lock group");
457
458 if (OS.isBundleLocked())
459 return;
460 // A pending fragment means the previous MCInst was a prefix, which must stay
461 // with this one: extend its range. Adjacency rejects a fragment left stale by
462 // an intervening .bundle_lock group.
463 if (PendingBA &&
464 PendingBA->getLastFragment()->getNext() == OS.getCurrentFragment()) {
465 PendingBA->setLastFragment(OS.getCurrentFragment());
466 return;
467 }
468 PendingBA = OS.newSpecialFragment<MCBoundaryAlignFragment>(
469 args: Asm->getBundleAlign(), args: STI);
470 // We can set LastFragment now, before the instruction is emitted, as bundling
471 // emits one fragment per instruction. Deferring setLastFragment to
472 // post-emitInstruction would risk capturing a fragment that a subsequent
473 // emitCodeAlignment repurposes in-place to FT_Align, corrupting the BA's
474 // boundary range.
475 PendingBA->setLastFragment(OS.getCurrentFragment());
476}
477
478/// Close the fragment opened by emitInstructionBeginBundle, unless the
479/// instruction was a prefix, in which case the next one extends it.
480void X86AsmBackend::emitInstructionEndBundle(MCObjectStreamer &OS) {
481 assert(Asm->isBundlingEnabled());
482
483 if (OS.isBundleLocked()) {
484 PrefixEndsBundleLock = isPrefix(Opcode: PrevInstOpcode, MCII: *MCII);
485 return;
486 }
487 PrefixEndsBundleLock = false;
488 assert(PendingBA && "MCBoundaryAlignFragment is expected for every "
489 "instruction if it is not bundle-locked");
490
491 OS.getCurrentSectionOnly()->ensureMinAlignment(MinAlignment: Asm->getBundleAlign());
492
493 if (!isPrefix(Opcode: PrevInstOpcode, MCII: *MCII))
494 PendingBA = nullptr;
495}
496
497/// Insert BoundaryAlignFragment before instructions to align branches.
498void X86AsmBackend::emitInstructionBegin(MCObjectStreamer &OS,
499 const MCInst &Inst,
500 const MCSubtargetInfo &STI) {
501 bool CanPadInst = canPadInst(Inst, OS);
502 if (Asm->isBundlingEnabled()) {
503 emitInstructionBeginBundle(OS);
504 OS.getCurrentFragment()->setAllowAutoPadding(CanPadInst);
505 return;
506 }
507 if (CanPadInst)
508 OS.getCurrentFragment()->setAllowAutoPadding(true);
509
510 if (!canPadBranches(OS))
511 return;
512
513 // NB: PrevInst only valid if canPadBranches is true.
514 if (!isMacroFused(Cmp: PrevInst, Jcc: Inst))
515 // Macro fusion doesn't happen indeed, clear the pending.
516 PendingBA = nullptr;
517
518 // When branch padding is enabled (basically the skx102 erratum => unlikely),
519 // we call canPadInst (not cheap) twice. However, in the common case, we can
520 // avoid unnecessary calls to that, as this is otherwise only used for
521 // relaxable fragments.
522 if (!CanPadInst)
523 return;
524
525 if (PendingBA) {
526 auto *NextFragment = PendingBA->getNext();
527 assert(NextFragment && "NextFragment should not be null");
528 if (NextFragment == OS.getCurrentFragment())
529 return;
530 // We eagerly create an empty fragment when inserting a fragment
531 // with a variable-size tail.
532 if (NextFragment->getNext() == OS.getCurrentFragment())
533 return;
534
535 // Macro fusion actually happens and there is no other fragment inserted
536 // after the previous instruction.
537 //
538 // Do nothing here since we already inserted a BoudaryAlign fragment when
539 // we met the first instruction in the fused pair and we'll tie them
540 // together in emitInstructionEnd.
541 //
542 // Note: When there is at least one fragment, such as MCAlignFragment,
543 // inserted after the previous instruction, e.g.
544 //
545 // \code
546 // cmp %rax %rcx
547 // .align 16
548 // je .Label0
549 // \ endcode
550 //
551 // We will treat the JCC as a unfused branch although it may be fused
552 // with the CMP.
553 return;
554 }
555
556 if (needAlign(Inst) || ((AlignBranchType & X86::AlignBranchFused) &&
557 isFirstMacroFusibleInst(Inst, MCII: *MCII))) {
558 // If we meet a unfused branch or the first instuction in a fusiable pair,
559 // insert a BoundaryAlign fragment.
560 PendingBA =
561 OS.newSpecialFragment<MCBoundaryAlignFragment>(args&: AlignBoundary, args: STI);
562 }
563}
564
565/// Set the last fragment to be aligned for the BoundaryAlignFragment.
566void X86AsmBackend::emitInstructionEnd(MCObjectStreamer &OS,
567 const MCInst &Inst) {
568 // Update PrevInstOpcode here, canPadInst() reads that.
569 MCFragment *CF = OS.getCurrentFragment();
570 PrevInstOpcode = Inst.getOpcode();
571 PrevInstPosition = std::make_pair(x&: CF, y: OS.getCurFragSize());
572 if (Asm->isBundlingEnabled())
573 return emitInstructionEndBundle(OS);
574
575 if (!canPadBranches(OS))
576 return;
577
578 // PrevInst is only needed if canPadBranches. Copying an MCInst isn't cheap.
579 PrevInst = Inst;
580
581 if (!needAlign(Inst) || !PendingBA)
582 return;
583
584 // Tie the aligned instructions into a pending BoundaryAlign.
585 PendingBA->setLastFragment(CF);
586 PendingBA = nullptr;
587
588 // We need to ensure that further data isn't added to the current
589 // DataFragment, so that we can get the size of instructions later in
590 // MCAssembler::relaxBoundaryAlign. The easiest way is to insert a new empty
591 // DataFragment.
592 OS.newFragment();
593
594 // Update the maximum alignment on the current section if necessary.
595 CF->getParent()->ensureMinAlignment(MinAlignment: AlignBoundary);
596}
597
598std::optional<MCFixupKind> X86AsmBackend::getFixupKind(StringRef Name) const {
599 if (STI.getTargetTriple().isOSBinFormatELF()) {
600 unsigned Type;
601 if (STI.getTargetTriple().isX86_64()) {
602 Type = llvm::StringSwitch<unsigned>(Name)
603#define ELF_RELOC(X, Y) .Case(#X, Y)
604#include "llvm/BinaryFormat/ELFRelocs/x86_64.def"
605#undef ELF_RELOC
606 .Case(S: "BFD_RELOC_NONE", Value: ELF::R_X86_64_NONE)
607 .Case(S: "BFD_RELOC_8", Value: ELF::R_X86_64_8)
608 .Case(S: "BFD_RELOC_16", Value: ELF::R_X86_64_16)
609 .Case(S: "BFD_RELOC_32", Value: ELF::R_X86_64_32)
610 .Case(S: "BFD_RELOC_64", Value: ELF::R_X86_64_64)
611 .Default(Value: -1u);
612 } else {
613 Type = llvm::StringSwitch<unsigned>(Name)
614#define ELF_RELOC(X, Y) .Case(#X, Y)
615#include "llvm/BinaryFormat/ELFRelocs/i386.def"
616#undef ELF_RELOC
617 .Case(S: "BFD_RELOC_NONE", Value: ELF::R_386_NONE)
618 .Case(S: "BFD_RELOC_8", Value: ELF::R_386_8)
619 .Case(S: "BFD_RELOC_16", Value: ELF::R_386_16)
620 .Case(S: "BFD_RELOC_32", Value: ELF::R_386_32)
621 .Default(Value: -1u);
622 }
623 if (Type == -1u)
624 return std::nullopt;
625 return static_cast<MCFixupKind>(FirstLiteralRelocationKind + Type);
626 }
627 return MCAsmBackend::getFixupKind(Name);
628}
629
630MCFixupKindInfo X86AsmBackend::getFixupKindInfo(MCFixupKind Kind) const {
631 const static MCFixupKindInfo Infos[X86::NumTargetFixupKinds] = {
632 // clang-format off
633 {.Name: "reloc_riprel_4byte", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
634 {.Name: "reloc_riprel_4byte_movq_load", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
635 {.Name: "reloc_riprel_4byte_movq_load_rex2", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
636 {.Name: "reloc_riprel_4byte_relax", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
637 {.Name: "reloc_riprel_4byte_relax_rex", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
638 {.Name: "reloc_riprel_4byte_relax_rex2", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
639 {.Name: "reloc_riprel_4byte_relax_evex", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
640 {.Name: "reloc_signed_4byte", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
641 {.Name: "reloc_signed_4byte_relax", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
642 {.Name: "reloc_global_offset_table", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
643 {.Name: "reloc_branch_4byte_pcrel", .TargetOffset: 0, .TargetSize: 32, .Flags: 0},
644 // clang-format on
645 };
646
647 // Fixup kinds from .reloc directive are like R_386_NONE/R_X86_64_NONE. They
648 // do not require any extra processing.
649 if (mc::isRelocation(FixupKind: Kind))
650 return {};
651
652 if (Kind < FirstTargetFixupKind)
653 return MCAsmBackend::getFixupKindInfo(Kind);
654
655 assert(unsigned(Kind - FirstTargetFixupKind) < X86::NumTargetFixupKinds &&
656 "Invalid kind!");
657 assert(Infos[Kind - FirstTargetFixupKind].Name && "Empty fixup name!");
658 return Infos[Kind - FirstTargetFixupKind];
659}
660
661static unsigned getFixupKindSize(unsigned Kind) {
662 switch (Kind) {
663 default:
664 llvm_unreachable("invalid fixup kind!");
665 case FK_NONE:
666 return 0;
667 case FK_SecRel_1:
668 case FK_Data_1:
669 return 1;
670 case FK_SecRel_2:
671 case FK_Data_2:
672 return 2;
673 case X86::reloc_riprel_4byte:
674 case X86::reloc_riprel_4byte_relax:
675 case X86::reloc_riprel_4byte_relax_rex:
676 case X86::reloc_riprel_4byte_relax_rex2:
677 case X86::reloc_riprel_4byte_movq_load:
678 case X86::reloc_riprel_4byte_movq_load_rex2:
679 case X86::reloc_riprel_4byte_relax_evex:
680 case X86::reloc_signed_4byte:
681 case X86::reloc_signed_4byte_relax:
682 case X86::reloc_global_offset_table:
683 case X86::reloc_branch_4byte_pcrel:
684 case FK_SecRel_4:
685 case FK_Data_4:
686 return 4;
687 case FK_SecRel_8:
688 case FK_Data_8:
689 return 8;
690 }
691}
692
693constexpr char GotSymName[] = "_GLOBAL_OFFSET_TABLE_";
694
695// Adjust PC-relative fixup offsets, which are calculated from the start of the
696// next instruction.
697std::optional<bool> X86AsmBackend::evaluateFixup(const MCFragment &,
698 MCFixup &Fixup,
699 MCValue &Target, uint64_t &) {
700 if (Fixup.isPCRel()) {
701 switch (Fixup.getKind()) {
702 case FK_Data_1:
703 Target.setConstant(Target.getConstant() - 1);
704 break;
705 case FK_Data_2:
706 Target.setConstant(Target.getConstant() - 2);
707 break;
708 default: {
709 Target.setConstant(Target.getConstant() - 4);
710 auto *Add = Target.getAddSym();
711 // If this is a pc-relative load off _GLOBAL_OFFSET_TABLE_:
712 // leaq _GLOBAL_OFFSET_TABLE_(%rip), %r15
713 // this needs to be a GOTPC32 relocation.
714 if (Add && Add->getName() == GotSymName)
715 Fixup = MCFixup::create(Offset: Fixup.getOffset(), Value: Fixup.getValue(),
716 Kind: X86::reloc_global_offset_table);
717 } break;
718 }
719 }
720 // Use default handling for `Value` and `IsResolved`.
721 return {};
722}
723
724void X86AsmBackend::applyFixup(const MCFragment &F, const MCFixup &Fixup,
725 const MCValue &Target, uint8_t *Data,
726 uint64_t Value, bool IsResolved) {
727 // Force relocation when there is a specifier. This might be too conservative
728 // - GAS doesn't emit a relocation for call local@plt; local:.
729 if (Target.getSpecifier())
730 IsResolved = false;
731 maybeAddReloc(F, Fixup, Target, Value, IsResolved);
732
733 auto Kind = Fixup.getKind();
734 if (mc::isRelocation(FixupKind: Kind))
735 return;
736 unsigned Size = getFixupKindSize(Kind);
737
738 assert(Fixup.getOffset() + Size <= F.getSize() && "Invalid fixup offset!");
739
740 // Check fixup value overflow similar to GAS (fixups emitted as RELA
741 // relocations have a value of 0).
742 // - Unknown signedness: the range (-2^N, 2^N) is allowed,
743 // accommodating intN_t, uintN_t, and a non-positive value type.
744 // - Signed (intN_t): the range [-2^(N-1), 2^(N-1)) is allowed.
745 //
746 // Currently only resolved PC-relative fixups are treated as signed. GAS
747 // treats more as signed (e.g. unresolved R_X86_64_32S).
748 // Unresolved fixups have unknown signedness to allow `jmp foo+0xffffffff`.
749 if (Size && Size < 8) {
750 bool Signed = IsResolved && Fixup.isPCRel();
751 uint64_t Mask = ~uint64_t(0) << (Size * 8 - (Signed ? 1 : 0));
752 if ((Value & Mask) && (Signed ? (Value & Mask) != Mask : (-Value & Mask)))
753 getContext().reportError(L: Fixup.getLoc(),
754 Msg: "value of " + Twine(int64_t(Value)) +
755 " is too large for field of " + Twine(Size) +
756 (Size == 1 ? " byte" : " bytes"));
757 }
758
759 for (unsigned i = 0; i != Size; ++i)
760 Data[i] = uint8_t(Value >> (i * 8));
761}
762
763bool X86AsmBackend::mayNeedRelaxation(unsigned Opcode,
764 ArrayRef<MCOperand> Operands,
765 const MCSubtargetInfo &STI) const {
766 unsigned SkipOperands = X86::isCCMPCC(Opcode) ? 2 : 0;
767 return isRelaxableBranch(Opcode) ||
768 (X86::getOpcodeForLongImmediateForm(Opcode) != Opcode &&
769 Operands[Operands.size() - 1 - SkipOperands].isExpr());
770}
771
772bool X86AsmBackend::fixupNeedsRelaxationAdvanced(const MCFragment &F,
773 const MCFixup &Fixup,
774 const MCValue &Target,
775 uint64_t Value,
776 bool Resolved) const {
777 // If resolved, relax if the value is too big for a (signed) i8.
778 //
779 // Currently, `jmp local@plt` relaxes JMP even if the offset is small,
780 // different from gas.
781 if (Resolved) {
782 // finishLayout folds padding into encodings after relaxation, shifting a
783 // branch and its target within their bundles. Keep a bundle of headroom.
784 // Immediates do not shift.
785 int64_t Slack = Asm->isBundlingEnabled() && TargetPrefixMax != 0 &&
786 isRelaxableBranch(Opcode: F.getOpcode())
787 ? Asm->getBundleAlign().value()
788 : 0;
789 return !isInt<8>(x: int64_t(Value) + Slack) ||
790 !isInt<8>(x: int64_t(Value) - Slack) || Target.getSpecifier();
791 }
792
793 // Otherwise, relax unless there is a @ABS8 specifier.
794 if (Fixup.getKind() == FK_Data_1 && Target.getAddSym() &&
795 Target.getSpecifier() == X86::S_ABS8)
796 return false;
797 return true;
798}
799
800// FIXME: Can tblgen help at all here to verify there aren't other instructions
801// we can relax?
802void X86AsmBackend::relaxInstruction(MCInst &Inst,
803 const MCSubtargetInfo &STI) const {
804 // The only relaxations X86 does is from a 1byte pcrel to a 4byte pcrel.
805 bool Is16BitMode = STI.hasFeature(Feature: X86::Is16Bit);
806 unsigned RelaxedOp = getRelaxedOpcode(MI: Inst, Is16BitMode);
807 assert(RelaxedOp != Inst.getOpcode());
808 Inst.setOpcode(RelaxedOp);
809}
810
811bool X86AsmBackend::padInstructionViaPrefix(MCFragment &RF,
812 MCCodeEmitter &Emitter,
813 unsigned &RemainingSize) const {
814 if (!RF.getAllowAutoPadding())
815 return false;
816 // If the instruction isn't fully relaxed, shifting it around might require a
817 // larger value for one of the fixups then can be encoded. The outer loop
818 // will also catch this before moving to the next instruction, but we need to
819 // prevent padding this single instruction as well.
820 if (mayNeedRelaxation(Opcode: RF.getOpcode(), Operands: RF.getOperands(),
821 STI: *RF.getSubtargetInfo()))
822 return false;
823
824 const unsigned OldSize = RF.getVarSize();
825 if (OldSize == 15)
826 return false;
827
828 const unsigned MaxPossiblePad = std::min(a: 15 - OldSize, b: RemainingSize);
829 const unsigned RemainingPrefixSize = [&]() -> unsigned {
830 SmallString<15> Code;
831 X86_MC::emitPrefix(MCE&: Emitter, MI: RF.getInst(), CB&: Code, STI);
832 assert(Code.size() < 15 && "The number of prefixes must be less than 15.");
833
834 // TODO: It turns out we need a decent amount of plumbing for the target
835 // specific bits to determine number of prefixes its safe to add. Various
836 // targets (older chips mostly, but also Atom family) encounter decoder
837 // stalls with too many prefixes. For testing purposes, we set the value
838 // externally for the moment.
839 unsigned ExistingPrefixSize = Code.size();
840 if (TargetPrefixMax <= ExistingPrefixSize)
841 return 0;
842 return TargetPrefixMax - ExistingPrefixSize;
843 }();
844 const unsigned PrefixBytesToAdd =
845 std::min(a: MaxPossiblePad, b: RemainingPrefixSize);
846 if (PrefixBytesToAdd == 0)
847 return false;
848
849 const uint8_t Prefix = determinePaddingPrefix(Inst: RF.getInst());
850
851 SmallString<256> Code;
852 Code.append(NumInputs: PrefixBytesToAdd, Elt: Prefix);
853 Code.append(in_start: RF.getVarContents().begin(), in_end: RF.getVarContents().end());
854 RF.setVarContents(Code);
855
856 // Adjust the fixups for the change in offsets
857 for (auto &F : RF.getVarFixups())
858 F.setOffset(PrefixBytesToAdd + F.getOffset());
859
860 RemainingSize -= PrefixBytesToAdd;
861 return true;
862}
863
864bool X86AsmBackend::padInstructionViaRelaxation(MCFragment &RF,
865 MCCodeEmitter &Emitter,
866 unsigned &RemainingSize) const {
867 if (!mayNeedRelaxation(Opcode: RF.getOpcode(), Operands: RF.getOperands(),
868 STI: *RF.getSubtargetInfo()))
869 // TODO: There are lots of other tricks we could apply for increasing
870 // encoding size without impacting performance.
871 return false;
872
873 MCInst Relaxed = RF.getInst();
874 relaxInstruction(Inst&: Relaxed, STI: *RF.getSubtargetInfo());
875
876 SmallVector<MCFixup, 4> Fixups;
877 SmallString<15> Code;
878 Emitter.encodeInstruction(Inst: Relaxed, CB&: Code, Fixups, STI: *RF.getSubtargetInfo());
879 const unsigned OldSize = RF.getVarContents().size();
880 const unsigned NewSize = Code.size();
881 assert(NewSize >= OldSize && "size decrease during relaxation?");
882 unsigned Delta = NewSize - OldSize;
883 if (Delta > RemainingSize)
884 return false;
885 RF.setInst(Relaxed);
886 RF.setVarContents(Code);
887 RF.setVarFixups(Fixups);
888 RemainingSize -= Delta;
889 return true;
890}
891
892bool X86AsmBackend::padInstructionEncoding(MCFragment &RF,
893 MCCodeEmitter &Emitter,
894 unsigned &RemainingSize) const {
895 bool Changed = false;
896 if (RemainingSize != 0)
897 Changed |= padInstructionViaRelaxation(RF, Emitter, RemainingSize);
898 if (RemainingSize != 0)
899 Changed |= padInstructionViaPrefix(RF, Emitter, RemainingSize);
900 return Changed;
901}
902
903bool X86AsmBackend::padInstsBackward(SmallVectorImpl<MCFragment *> &Relaxable,
904 unsigned &RemainingSize) const {
905 bool Changed = false;
906 while (!Relaxable.empty() && RemainingSize != 0) {
907 auto &RF = *Relaxable.pop_back_val();
908 // Give the backend a chance to play any tricks it wishes to increase
909 // the encoding size of the given instruction. Target independent code
910 // will try further relaxation, but target's may play further tricks.
911 Changed |= padInstructionEncoding(RF, Emitter&: Asm->getEmitter(), RemainingSize);
912
913 // If we have an instruction which hasn't been fully relaxed, we can't
914 // skip past it and insert bytes before it. Changing its starting
915 // offset might require a larger negative offset than it can encode.
916 // We don't need to worry about larger positive offsets as none of the
917 // possible offsets between this and our align are visible, and the
918 // ones afterwards aren't changing.
919 if (mayNeedRelaxation(Opcode: RF.getOpcode(), Operands: RF.getOperands(),
920 STI: *RF.getSubtargetInfo()))
921 break;
922 }
923 Relaxable.clear();
924 return Changed;
925}
926
927/// Trade the padding held by \p BF for ignored prefixes on the instructions
928/// around it. Padding never leaves its own bundle, so no instruction or
929/// bundle-locked group moves across a boundary. \p Relaxable holds the
930/// preceding instructions in that bundle and is consumed.
931bool X86AsmBackend::foldBundlePad(
932 const MCAssembler &Asm, MCBoundaryAlignFragment &BF,
933 SmallVectorImpl<MCFragment *> &Relaxable) const {
934 const uint64_t BundleSize = Asm.getBundleAlign().value();
935 const uint64_t PadStart = Asm.getFragmentOffset(F: BF);
936 unsigned Remaining = BF.getSize();
937
938 // Only padding in PadStart's own bundle may move backward; an align_to_end
939 // group can push the rest into the next bundle.
940 unsigned Budget =
941 std::min<uint64_t>(a: Remaining, b: BundleSize - PadStart % BundleSize);
942 unsigned Left = Budget;
943 bool Changed = padInstsBackward(Relaxable, RemainingSize&: Left);
944 Remaining -= Budget - Left;
945
946 // Absorbing padding moves the group's start earlier while its end is pinned,
947 // so it may only grow into the slack in its bundle: BundleSize - GroupSize
948 // for align_to_end, zero for a group that already starts on a boundary.
949 if (BF.isAlignToEnd() && Remaining) {
950 uint64_t GroupSize = 0;
951 for (const MCFragment *F = BF.getNext();; F = F->getNext()) {
952 GroupSize += Asm.computeFragmentSize(F: *F);
953 if (F == BF.getLastFragment())
954 break;
955 }
956 if (GroupSize < BundleSize) {
957 Left = Budget = std::min<uint64_t>(a: Remaining, b: BundleSize - GroupSize);
958 for (MCFragment *F = BF.getNext(); F; F = F->getNext()) {
959 if (F->getKind() == MCFragment::FT_Relaxable)
960 Changed |= padInstructionEncoding(RF&: *F, Emitter&: Asm.getEmitter(), RemainingSize&: Left);
961 if (F == BF.getLastFragment() || Left == 0)
962 break;
963 }
964 Remaining -= Budget - Left;
965 }
966 }
967
968 BF.setSize(Remaining);
969 return Changed;
970}
971
972bool X86AsmBackend::optimizeBundleNops(const MCAssembler &Asm) const {
973 const uint64_t BundleSize = Asm.getBundleAlign().value();
974 bool Changed = false;
975 for (MCSection &Sec : Asm) {
976 if (!Sec.isText())
977 continue;
978
979 // Instructions preceding the next padding and sharing its bundle.
980 SmallVector<MCFragment *, 8> Relaxable;
981 // Folding leaves stale offsets until the next layout, so skip the
982 // rewritten range.
983 const MCFragment *ResumeAfter = nullptr;
984 for (MCFragment &F : Sec) {
985 if (ResumeAfter) {
986 if (&F == ResumeAfter)
987 ResumeAfter = nullptr;
988 continue;
989 }
990 uint64_t Offset = Asm.getFragmentOffset(F);
991 if (!Relaxable.empty() &&
992 Asm.getFragmentOffset(F: *Relaxable.front()) / BundleSize !=
993 Offset / BundleSize)
994 Relaxable.clear();
995
996 switch (F.getKind()) {
997 case MCFragment::FT_BoundaryAlign: {
998 auto &BF = static_cast<MCBoundaryAlignFragment &>(F);
999 if (!BF.getSize())
1000 break; // Nothing to fold, and not a barrier.
1001 Changed |= foldBundlePad(Asm, BF, Relaxable);
1002 ResumeAfter = BF.getLastFragment();
1003 break;
1004 }
1005 case MCFragment::FT_Relaxable:
1006 Relaxable.push_back(Elt: &F);
1007 break;
1008 case MCFragment::FT_Data:
1009 break; // Fixed bytes, safe to shift.
1010 default:
1011 // Other kinds may change size when shifted (.p2align, .org, LEBs).
1012 Relaxable.clear();
1013 break;
1014 }
1015 }
1016 }
1017
1018 return Changed;
1019}
1020
1021bool X86AsmBackend::finishLayout() const {
1022 // With bundling, padding is fully determined during layout and the only
1023 // post-layout optimization is prefix padding.
1024 if (Asm->isBundlingEnabled())
1025 return TargetPrefixMax != 0 && optimizeBundleNops(Asm: *Asm);
1026 // See if we can further relax some instructions to cut down on the number of
1027 // nop bytes required for code alignment. The actual win is in reducing
1028 // instruction count, not number of bytes. Modern X86-64 can easily end up
1029 // decode limited. It is often better to reduce the number of instructions
1030 // (i.e. eliminate nops) even at the cost of increasing the size and
1031 // complexity of others.
1032 if (!CLOpts.pad_for_align && !CLOpts.pad_for_branch_align)
1033 return false;
1034
1035 // The processed regions are delimitered by LabeledFragments. -g may have more
1036 // MCSymbols and therefore different relaxation results. -x86-pad-for-align is
1037 // disabled by default to eliminate the -g vs non -g difference.
1038 DenseSet<MCFragment *> LabeledFragments;
1039 for (const MCSymbol &S : Asm->symbols())
1040 LabeledFragments.insert(V: S.getFragment());
1041
1042 bool Changed = false;
1043 for (MCSection &Sec : *Asm) {
1044 if (!Sec.isText())
1045 continue;
1046
1047 SmallVector<MCFragment *, 4> Relaxable;
1048 for (MCSection::iterator I = Sec.begin(), IE = Sec.end(); I != IE; ++I) {
1049 MCFragment &F = *I;
1050
1051 if (LabeledFragments.count(V: &F))
1052 Relaxable.clear();
1053
1054 if (F.getKind() == MCFragment::FT_Data) // Skip and ignore
1055 continue;
1056
1057 if (F.getKind() == MCFragment::FT_Relaxable) {
1058 auto &RF = cast<MCFragment>(Val&: *I);
1059 Relaxable.push_back(Elt: &RF);
1060 continue;
1061 }
1062
1063 auto canHandle = [&](MCFragment &F) -> bool {
1064 switch (F.getKind()) {
1065 default:
1066 return false;
1067 case MCFragment::FT_Align:
1068 return CLOpts.pad_for_align;
1069 case MCFragment::FT_BoundaryAlign:
1070 return CLOpts.pad_for_branch_align;
1071 }
1072 };
1073 // For any unhandled kind, assume we can't change layout.
1074 if (!canHandle(F)) {
1075 Relaxable.clear();
1076 continue;
1077 }
1078
1079 // To keep the effects local, prefer to relax instructions closest to
1080 // the align directive. This is purely about human understandability
1081 // of the resulting code. If we later find a reason to expand
1082 // particular instructions over others, we can adjust.
1083 unsigned RemainingSize = Asm->computeFragmentSize(F) - F.getFixedSize();
1084 Changed |= padInstsBackward(Relaxable, RemainingSize);
1085
1086 // If we're looking at a boundary align, make sure we don't try to pad
1087 // its target instructions for some following directive. Doing so would
1088 // break the alignment of the current boundary align.
1089 if (auto *BF = dyn_cast<MCBoundaryAlignFragment>(Val: &F)) {
1090 cast<MCBoundaryAlignFragment>(Val&: F).setSize(RemainingSize);
1091 Changed = true;
1092 const MCFragment *LastFragment = BF->getLastFragment();
1093 if (!LastFragment)
1094 continue;
1095 while (&*I != LastFragment)
1096 ++I;
1097 }
1098 }
1099 }
1100
1101 return Changed;
1102}
1103
1104unsigned X86AsmBackend::getMaximumNopSize(const MCSubtargetInfo &STI) const {
1105 if (STI.hasFeature(Feature: X86::Is16Bit))
1106 return 4;
1107 if (!STI.hasFeature(Feature: X86::FeatureNOPL) && !STI.hasFeature(Feature: X86::Is64Bit))
1108 return 1;
1109 if (STI.hasFeature(Feature: X86::TuningFast7ByteNOP))
1110 return 7;
1111 if (STI.hasFeature(Feature: X86::TuningFast15ByteNOP))
1112 return 15;
1113 if (STI.hasFeature(Feature: X86::TuningFast11ByteNOP))
1114 return 11;
1115 // FIXME: handle 32-bit mode
1116 // 15-bytes is the longest single NOP instruction, but 10-bytes is
1117 // commonly the longest that can be efficiently decoded.
1118 return 10;
1119}
1120
1121/// Write a sequence of optimal nops to the output, covering \p Count
1122/// bytes.
1123/// \return - true on success, false on failure
1124bool X86AsmBackend::writeNopData(raw_ostream &OS, uint64_t Count,
1125 const MCSubtargetInfo *STI) const {
1126 static const char Nops32Bit[10][11] = {
1127 // nop
1128 "\x90",
1129 // xchg %ax,%ax
1130 "\x66\x90",
1131 // nopl (%[re]ax)
1132 "\x0f\x1f\x00",
1133 // nopl 0(%[re]ax)
1134 "\x0f\x1f\x40\x00",
1135 // nopl 0(%[re]ax,%[re]ax,1)
1136 "\x0f\x1f\x44\x00\x00",
1137 // nopw 0(%[re]ax,%[re]ax,1)
1138 "\x66\x0f\x1f\x44\x00\x00",
1139 // nopl 0L(%[re]ax)
1140 "\x0f\x1f\x80\x00\x00\x00\x00",
1141 // nopl 0L(%[re]ax,%[re]ax,1)
1142 "\x0f\x1f\x84\x00\x00\x00\x00\x00",
1143 // nopw 0L(%[re]ax,%[re]ax,1)
1144 "\x66\x0f\x1f\x84\x00\x00\x00\x00\x00",
1145 // nopw %cs:0L(%[re]ax,%[re]ax,1)
1146 "\x66\x2e\x0f\x1f\x84\x00\x00\x00\x00\x00",
1147 };
1148
1149 // 16-bit mode uses different nop patterns than 32-bit.
1150 static const char Nops16Bit[4][11] = {
1151 // nop
1152 "\x90",
1153 // xchg %eax,%eax
1154 "\x66\x90",
1155 // lea 0(%si),%si
1156 "\x8d\x74\x00",
1157 // lea 0w(%si),%si
1158 "\x8d\xb4\x00\x00",
1159 };
1160
1161 const char(*Nops)[11] =
1162 STI->hasFeature(Feature: X86::Is16Bit) ? Nops16Bit : Nops32Bit;
1163
1164 uint64_t MaxNopLength = (uint64_t)getMaximumNopSize(STI: *STI);
1165
1166 // Emit as many MaxNopLength NOPs as needed, then emit a NOP of the remaining
1167 // length.
1168 do {
1169 const uint8_t ThisNopLength = (uint8_t) std::min(a: Count, b: MaxNopLength);
1170 const uint8_t Prefixes = ThisNopLength <= 10 ? 0 : ThisNopLength - 10;
1171 for (uint8_t i = 0; i < Prefixes; i++)
1172 OS << '\x66';
1173 const uint8_t Rest = ThisNopLength - Prefixes;
1174 if (Rest != 0)
1175 OS.write(Ptr: Nops[Rest - 1], Size: Rest);
1176 Count -= ThisNopLength;
1177 } while (Count != 0);
1178
1179 return true;
1180}
1181
1182/* *** */
1183
1184namespace {
1185
1186class ELFX86AsmBackend : public X86AsmBackend {
1187public:
1188 uint8_t OSABI;
1189 ELFX86AsmBackend(const Target &T, uint8_t OSABI, const MCSubtargetInfo &STI)
1190 : X86AsmBackend(T, STI), OSABI(OSABI) {}
1191};
1192
1193class ELFX86_32AsmBackend : public ELFX86AsmBackend {
1194public:
1195 ELFX86_32AsmBackend(const Target &T, uint8_t OSABI,
1196 const MCSubtargetInfo &STI)
1197 : ELFX86AsmBackend(T, OSABI, STI) {}
1198
1199 std::unique_ptr<MCObjectTargetWriter>
1200 createObjectTargetWriter() const override {
1201 return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI, EMachine: ELF::EM_386);
1202 }
1203};
1204
1205class ELFX86_X32AsmBackend : public ELFX86AsmBackend {
1206public:
1207 ELFX86_X32AsmBackend(const Target &T, uint8_t OSABI,
1208 const MCSubtargetInfo &STI)
1209 : ELFX86AsmBackend(T, OSABI, STI) {}
1210
1211 std::unique_ptr<MCObjectTargetWriter>
1212 createObjectTargetWriter() const override {
1213 return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI,
1214 EMachine: ELF::EM_X86_64);
1215 }
1216};
1217
1218class ELFX86_IAMCUAsmBackend : public ELFX86AsmBackend {
1219public:
1220 ELFX86_IAMCUAsmBackend(const Target &T, uint8_t OSABI,
1221 const MCSubtargetInfo &STI)
1222 : ELFX86AsmBackend(T, OSABI, STI) {}
1223
1224 std::unique_ptr<MCObjectTargetWriter>
1225 createObjectTargetWriter() const override {
1226 return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI,
1227 EMachine: ELF::EM_IAMCU);
1228 }
1229};
1230
1231class ELFX86_64AsmBackend : public ELFX86AsmBackend {
1232public:
1233 ELFX86_64AsmBackend(const Target &T, uint8_t OSABI,
1234 const MCSubtargetInfo &STI)
1235 : ELFX86AsmBackend(T, OSABI, STI) {}
1236
1237 std::unique_ptr<MCObjectTargetWriter>
1238 createObjectTargetWriter() const override {
1239 return createX86ELFObjectWriter(/*IsELF64*/ true, OSABI, EMachine: ELF::EM_X86_64);
1240 }
1241};
1242
1243class WindowsX86AsmBackend : public X86AsmBackend {
1244 bool Is64Bit;
1245
1246public:
1247 WindowsX86AsmBackend(const Target &T, bool is64Bit,
1248 const MCSubtargetInfo &STI)
1249 : X86AsmBackend(T, STI)
1250 , Is64Bit(is64Bit) {
1251 }
1252
1253 std::optional<MCFixupKind> getFixupKind(StringRef Name) const override {
1254 return StringSwitch<std::optional<MCFixupKind>>(Name)
1255 .Case(S: "dir32", Value: FK_Data_4)
1256 .Case(S: "secrel32", Value: FK_SecRel_4)
1257 .Case(S: "secidx", Value: FK_SecRel_2)
1258 .Default(Value: MCAsmBackend::getFixupKind(Name));
1259 }
1260
1261 std::unique_ptr<MCObjectTargetWriter>
1262 createObjectTargetWriter() const override {
1263 return createX86WinCOFFObjectWriter(Is64Bit);
1264 }
1265};
1266
1267namespace CU {
1268
1269 /// Compact unwind encoding values.
1270 enum CompactUnwindEncodings {
1271 /// [RE]BP based frame where [RE]BP is pused on the stack immediately after
1272 /// the return address, then [RE]SP is moved to [RE]BP.
1273 UNWIND_MODE_BP_FRAME = 0x01000000,
1274
1275 /// A frameless function with a small constant stack size.
1276 UNWIND_MODE_STACK_IMMD = 0x02000000,
1277
1278 /// A frameless function with a large constant stack size.
1279 UNWIND_MODE_STACK_IND = 0x03000000,
1280
1281 /// No compact unwind encoding is available.
1282 UNWIND_MODE_DWARF = 0x04000000,
1283
1284 /// Mask for encoding the frame registers.
1285 UNWIND_BP_FRAME_REGISTERS = 0x00007FFF,
1286
1287 /// Mask for encoding the frameless registers.
1288 UNWIND_FRAMELESS_STACK_REG_PERMUTATION = 0x000003FF
1289 };
1290
1291} // namespace CU
1292
1293class DarwinX86AsmBackend : public X86AsmBackend {
1294 const MCRegisterInfo &MRI;
1295
1296 /// Number of registers that can be saved in a compact unwind encoding.
1297 enum { CU_NUM_SAVED_REGS = 6 };
1298
1299 mutable unsigned SavedRegs[CU_NUM_SAVED_REGS];
1300 Triple TT;
1301 bool Is64Bit;
1302
1303 unsigned OffsetSize; ///< Offset of a "push" instruction.
1304 unsigned MoveInstrSize; ///< Size of a "move" instruction.
1305 unsigned StackDivide; ///< Amount to adjust stack size by.
1306protected:
1307 /// Size of a "push" instruction for the given register.
1308 unsigned PushInstrSize(MCRegister Reg) const {
1309 switch (Reg.id()) {
1310 case X86::EBX:
1311 case X86::ECX:
1312 case X86::EDX:
1313 case X86::EDI:
1314 case X86::ESI:
1315 case X86::EBP:
1316 case X86::RBX:
1317 case X86::RBP:
1318 return 1;
1319 case X86::R12:
1320 case X86::R13:
1321 case X86::R14:
1322 case X86::R15:
1323 return 2;
1324 }
1325 return 1;
1326 }
1327
1328private:
1329 /// Get the compact unwind number for a given register. The number
1330 /// corresponds to the enum lists in compact_unwind_encoding.h.
1331 int getCompactUnwindRegNum(unsigned Reg) const {
1332 static const MCPhysReg CU32BitRegs[7] = {
1333 X86::EBX, X86::ECX, X86::EDX, X86::EDI, X86::ESI, X86::EBP, 0
1334 };
1335 static const MCPhysReg CU64BitRegs[] = {
1336 X86::RBX, X86::R12, X86::R13, X86::R14, X86::R15, X86::RBP, 0
1337 };
1338 const MCPhysReg *CURegs = Is64Bit ? CU64BitRegs : CU32BitRegs;
1339 for (int Idx = 1; *CURegs; ++CURegs, ++Idx)
1340 if (*CURegs == Reg)
1341 return Idx;
1342
1343 return -1;
1344 }
1345
1346 /// Return the registers encoded for a compact encoding with a frame
1347 /// pointer.
1348 uint32_t encodeCompactUnwindRegistersWithFrame() const {
1349 // Encode the registers in the order they were saved --- 3-bits per
1350 // register. The list of saved registers is assumed to be in reverse
1351 // order. The registers are numbered from 1 to CU_NUM_SAVED_REGS.
1352 uint32_t RegEnc = 0;
1353 for (int i = 0, Idx = 0; i != CU_NUM_SAVED_REGS; ++i) {
1354 unsigned Reg = SavedRegs[i];
1355 if (Reg == 0) break;
1356
1357 int CURegNum = getCompactUnwindRegNum(Reg);
1358 if (CURegNum == -1) return ~0U;
1359
1360 // Encode the 3-bit register number in order, skipping over 3-bits for
1361 // each register.
1362 RegEnc |= (CURegNum & 0x7) << (Idx++ * 3);
1363 }
1364
1365 assert((RegEnc & 0x3FFFF) == RegEnc &&
1366 "Invalid compact register encoding!");
1367 return RegEnc;
1368 }
1369
1370 /// Create the permutation encoding used with frameless stacks. It is
1371 /// passed the number of registers to be saved and an array of the registers
1372 /// saved.
1373 uint32_t encodeCompactUnwindRegistersWithoutFrame(unsigned RegCount) const {
1374 // The saved registers are numbered from 1 to 6. In order to encode the
1375 // order in which they were saved, we re-number them according to their
1376 // place in the register order. The re-numbering is relative to the last
1377 // re-numbered register. E.g., if we have registers {6, 2, 4, 5} saved in
1378 // that order:
1379 //
1380 // Orig Re-Num
1381 // ---- ------
1382 // 6 6
1383 // 2 2
1384 // 4 3
1385 // 5 3
1386 //
1387 for (unsigned i = 0; i < RegCount; ++i) {
1388 int CUReg = getCompactUnwindRegNum(Reg: SavedRegs[i]);
1389 if (CUReg == -1) return ~0U;
1390 SavedRegs[i] = CUReg;
1391 }
1392
1393 // Reverse the list.
1394 std::reverse(first: &SavedRegs[0], last: &SavedRegs[CU_NUM_SAVED_REGS]);
1395
1396 uint32_t RenumRegs[CU_NUM_SAVED_REGS];
1397 for (unsigned i = CU_NUM_SAVED_REGS - RegCount; i < CU_NUM_SAVED_REGS; ++i){
1398 unsigned Countless = 0;
1399 for (unsigned j = CU_NUM_SAVED_REGS - RegCount; j < i; ++j)
1400 if (SavedRegs[j] < SavedRegs[i])
1401 ++Countless;
1402
1403 RenumRegs[i] = SavedRegs[i] - Countless - 1;
1404 }
1405
1406 // Take the renumbered values and encode them into a 10-bit number.
1407 uint32_t permutationEncoding = 0;
1408 switch (RegCount) {
1409 case 6:
1410 permutationEncoding |= 120 * RenumRegs[0] + 24 * RenumRegs[1]
1411 + 6 * RenumRegs[2] + 2 * RenumRegs[3]
1412 + RenumRegs[4];
1413 break;
1414 case 5:
1415 permutationEncoding |= 120 * RenumRegs[1] + 24 * RenumRegs[2]
1416 + 6 * RenumRegs[3] + 2 * RenumRegs[4]
1417 + RenumRegs[5];
1418 break;
1419 case 4:
1420 permutationEncoding |= 60 * RenumRegs[2] + 12 * RenumRegs[3]
1421 + 3 * RenumRegs[4] + RenumRegs[5];
1422 break;
1423 case 3:
1424 permutationEncoding |= 20 * RenumRegs[3] + 4 * RenumRegs[4]
1425 + RenumRegs[5];
1426 break;
1427 case 2:
1428 permutationEncoding |= 5 * RenumRegs[4] + RenumRegs[5];
1429 break;
1430 case 1:
1431 permutationEncoding |= RenumRegs[5];
1432 break;
1433 }
1434
1435 assert((permutationEncoding & 0x3FF) == permutationEncoding &&
1436 "Invalid compact register encoding!");
1437 return permutationEncoding;
1438 }
1439
1440public:
1441 DarwinX86AsmBackend(const Target &T, const MCRegisterInfo &MRI,
1442 const MCSubtargetInfo &STI)
1443 : X86AsmBackend(T, STI), MRI(MRI), TT(STI.getTargetTriple()),
1444 Is64Bit(TT.isX86_64()) {
1445 memset(s: SavedRegs, c: 0, n: sizeof(SavedRegs));
1446 OffsetSize = Is64Bit ? 8 : 4;
1447 MoveInstrSize = Is64Bit ? 3 : 2;
1448 StackDivide = Is64Bit ? 8 : 4;
1449 }
1450
1451 std::unique_ptr<MCObjectTargetWriter>
1452 createObjectTargetWriter() const override {
1453 uint32_t CPUType = cantFail(ValOrErr: MachO::getCPUType(T: TT));
1454 uint32_t CPUSubType = cantFail(ValOrErr: MachO::getCPUSubType(T: TT));
1455 return createX86MachObjectWriter(Is64Bit, CPUType, CPUSubtype: CPUSubType);
1456 }
1457
1458 /// Implementation of algorithm to generate the compact unwind encoding
1459 /// for the CFI instructions.
1460 uint64_t generateCompactUnwindEncoding(const MCDwarfFrameInfo *FI,
1461 const MCContext *Ctxt) const override {
1462 if (Ctxt->emitDwarfUnwindInfo() == EmitDwarfUnwindType::DwarfOnly)
1463 return CU::UNWIND_MODE_DWARF;
1464
1465 // Signal frames cannot be encoded in compact unwind.
1466 if (FI->IsSignalFrame)
1467 return CU::UNWIND_MODE_DWARF;
1468
1469 ArrayRef<MCCFIInstruction> Instrs = FI->Instructions;
1470 if (Instrs.empty()) return 0;
1471 if (!isDarwinCanonicalPersonality(Sym: FI->Personality) &&
1472 !Ctxt->emitCompactUnwindNonCanonical())
1473 return CU::UNWIND_MODE_DWARF;
1474
1475 // Reset the saved registers.
1476 unsigned SavedRegIdx = 0;
1477 memset(s: SavedRegs, c: 0, n: sizeof(SavedRegs));
1478
1479 bool HasFP = false;
1480
1481 // Encode that we are using EBP/RBP as the frame pointer.
1482 uint64_t CompactUnwindEncoding = 0;
1483
1484 unsigned SubtractInstrIdx = Is64Bit ? 3 : 2;
1485 unsigned InstrOffset = 0;
1486 unsigned StackAdjust = 0;
1487 uint64_t StackSize = 0;
1488 int64_t MinAbsOffset = std::numeric_limits<int64_t>::max();
1489
1490 for (const MCCFIInstruction &Inst : Instrs) {
1491 switch (Inst.getOperation()) {
1492 default:
1493 // Any other CFI directives indicate a frame that we aren't prepared
1494 // to represent via compact unwind, so just bail out.
1495 return CU::UNWIND_MODE_DWARF;
1496 case MCCFIInstruction::OpDefCfaRegister: {
1497 // Defines a frame pointer. E.g.
1498 //
1499 // movq %rsp, %rbp
1500 // L0:
1501 // .cfi_def_cfa_register %rbp
1502 //
1503 HasFP = true;
1504
1505 // If the frame pointer is other than esp/rsp, we do not have a way to
1506 // generate a compact unwinding representation, so bail out.
1507 if (*MRI.getLLVMRegNum(RegNum: Inst.getRegister(), isEH: true) !=
1508 (Is64Bit ? X86::RBP : X86::EBP))
1509 return CU::UNWIND_MODE_DWARF;
1510
1511 // Reset the counts.
1512 memset(s: SavedRegs, c: 0, n: sizeof(SavedRegs));
1513 StackAdjust = 0;
1514 SavedRegIdx = 0;
1515 MinAbsOffset = std::numeric_limits<int64_t>::max();
1516 InstrOffset += MoveInstrSize;
1517 break;
1518 }
1519 case MCCFIInstruction::OpDefCfaOffset: {
1520 // Defines a new offset for the CFA. E.g.
1521 //
1522 // With frame:
1523 //
1524 // pushq %rbp
1525 // L0:
1526 // .cfi_def_cfa_offset 16
1527 //
1528 // Without frame:
1529 //
1530 // subq $72, %rsp
1531 // L0:
1532 // .cfi_def_cfa_offset 80
1533 //
1534 StackSize = Inst.getOffset() / StackDivide;
1535 break;
1536 }
1537 case MCCFIInstruction::OpOffset: {
1538 // Defines a "push" of a callee-saved register. E.g.
1539 //
1540 // pushq %r15
1541 // pushq %r14
1542 // pushq %rbx
1543 // L0:
1544 // subq $120, %rsp
1545 // L1:
1546 // .cfi_offset %rbx, -40
1547 // .cfi_offset %r14, -32
1548 // .cfi_offset %r15, -24
1549 //
1550 if (SavedRegIdx == CU_NUM_SAVED_REGS)
1551 // If there are too many saved registers, we cannot use a compact
1552 // unwind encoding.
1553 return CU::UNWIND_MODE_DWARF;
1554
1555 MCRegister Reg = *MRI.getLLVMRegNum(RegNum: Inst.getRegister(), isEH: true);
1556 SavedRegs[SavedRegIdx++] = Reg.id();
1557 StackAdjust += OffsetSize;
1558 MinAbsOffset = std::min(a: MinAbsOffset, b: std::abs(i: Inst.getOffset()));
1559 InstrOffset += PushInstrSize(Reg);
1560 break;
1561 }
1562 }
1563 }
1564
1565 StackAdjust /= StackDivide;
1566
1567 if (HasFP) {
1568 if ((StackAdjust & 0xFF) != StackAdjust)
1569 // Offset was too big for a compact unwind encoding.
1570 return CU::UNWIND_MODE_DWARF;
1571
1572 // We don't attempt to track a real StackAdjust, so if the saved registers
1573 // aren't adjacent to rbp we can't cope.
1574 if (SavedRegIdx != 0 && MinAbsOffset != 3 * (int)OffsetSize)
1575 return CU::UNWIND_MODE_DWARF;
1576
1577 // Get the encoding of the saved registers when we have a frame pointer.
1578 uint32_t RegEnc = encodeCompactUnwindRegistersWithFrame();
1579 if (RegEnc == ~0U) return CU::UNWIND_MODE_DWARF;
1580
1581 CompactUnwindEncoding |= CU::UNWIND_MODE_BP_FRAME;
1582 CompactUnwindEncoding |= (StackAdjust & 0xFF) << 16;
1583 CompactUnwindEncoding |= RegEnc & CU::UNWIND_BP_FRAME_REGISTERS;
1584 } else {
1585 SubtractInstrIdx += InstrOffset;
1586 ++StackAdjust;
1587
1588 if ((StackSize & 0xFF) == StackSize) {
1589 // Frameless stack with a small stack size.
1590 CompactUnwindEncoding |= CU::UNWIND_MODE_STACK_IMMD;
1591
1592 // Encode the stack size.
1593 CompactUnwindEncoding |= (StackSize & 0xFF) << 16;
1594 } else {
1595 if ((StackAdjust & 0x7) != StackAdjust)
1596 // The extra stack adjustments are too big for us to handle.
1597 return CU::UNWIND_MODE_DWARF;
1598
1599 // Frameless stack with an offset too large for us to encode compactly.
1600 CompactUnwindEncoding |= CU::UNWIND_MODE_STACK_IND;
1601
1602 // Encode the offset to the nnnnnn value in the 'subl $nnnnnn, ESP'
1603 // instruction.
1604 CompactUnwindEncoding |= (SubtractInstrIdx & 0xFF) << 16;
1605
1606 // Encode any extra stack adjustments (done via push instructions).
1607 CompactUnwindEncoding |= (StackAdjust & 0x7) << 13;
1608 }
1609
1610 // Encode the number of registers saved. (Reverse the list first.)
1611 std::reverse(first: &SavedRegs[0], last: &SavedRegs[SavedRegIdx]);
1612 CompactUnwindEncoding |= (SavedRegIdx & 0x7) << 10;
1613
1614 // Get the encoding of the saved registers when we don't have a frame
1615 // pointer.
1616 uint32_t RegEnc = encodeCompactUnwindRegistersWithoutFrame(RegCount: SavedRegIdx);
1617 if (RegEnc == ~0U) return CU::UNWIND_MODE_DWARF;
1618
1619 // Encode the register encoding.
1620 CompactUnwindEncoding |=
1621 RegEnc & CU::UNWIND_FRAMELESS_STACK_REG_PERMUTATION;
1622 }
1623
1624 return CompactUnwindEncoding;
1625 }
1626};
1627
1628} // end anonymous namespace
1629
1630MCAsmBackend *llvm::createX86_32AsmBackend(const Target &T,
1631 const MCSubtargetInfo &STI,
1632 const MCRegisterInfo &MRI,
1633 const MCTargetOptions &Options) {
1634 const Triple &TheTriple = STI.getTargetTriple();
1635 if (TheTriple.isOSBinFormatMachO())
1636 return new DarwinX86AsmBackend(T, MRI, STI);
1637
1638 if (TheTriple.isOSWindows() && TheTriple.isOSBinFormatCOFF())
1639 return new WindowsX86AsmBackend(T, false, STI);
1640
1641 uint8_t OSABI = MCELFObjectTargetWriter::getOSABI(OSType: TheTriple.getOS());
1642
1643 if (TheTriple.isOSIAMCU())
1644 return new ELFX86_IAMCUAsmBackend(T, OSABI, STI);
1645
1646 return new ELFX86_32AsmBackend(T, OSABI, STI);
1647}
1648
1649MCAsmBackend *llvm::createX86_64AsmBackend(const Target &T,
1650 const MCSubtargetInfo &STI,
1651 const MCRegisterInfo &MRI,
1652 const MCTargetOptions &Options) {
1653 const Triple &TheTriple = STI.getTargetTriple();
1654 if (TheTriple.isOSBinFormatMachO())
1655 return new DarwinX86AsmBackend(T, MRI, STI);
1656
1657 if (TheTriple.isOSWindows() && TheTriple.isOSBinFormatCOFF())
1658 return new WindowsX86AsmBackend(T, true, STI);
1659
1660 if (TheTriple.isUEFI()) {
1661 assert(TheTriple.isOSBinFormatCOFF() &&
1662 "Only COFF format is supported in UEFI environment.");
1663 return new WindowsX86AsmBackend(T, true, STI);
1664 }
1665
1666 uint8_t OSABI = MCELFObjectTargetWriter::getOSABI(OSType: TheTriple.getOS());
1667
1668 if (TheTriple.isX32())
1669 return new ELFX86_X32AsmBackend(T, OSABI, STI);
1670 return new ELFX86_64AsmBackend(T, OSABI, STI);
1671}
1672
1673namespace {
1674class X86ELFStreamer : public MCELFStreamer {
1675public:
1676 X86ELFStreamer(MCContext &Context, std::unique_ptr<MCAsmBackend> TAB,
1677 std::unique_ptr<MCObjectWriter> OW,
1678 std::unique_ptr<MCCodeEmitter> Emitter)
1679 : MCELFStreamer(Context, std::move(TAB), std::move(OW),
1680 std::move(Emitter)) {}
1681
1682 void emitInstruction(const MCInst &Inst, const MCSubtargetInfo &STI) override;
1683};
1684} // end anonymous namespace
1685
1686void X86ELFStreamer::emitInstruction(const MCInst &Inst,
1687 const MCSubtargetInfo &STI) {
1688 X86_MC::emitInstruction(S&: *this, Inst, STI);
1689}
1690
1691MCStreamer *llvm::createX86ELFStreamer(const Triple &T, MCContext &Context,
1692 std::unique_ptr<MCAsmBackend> &&MAB,
1693 std::unique_ptr<MCObjectWriter> &&MOW,
1694 std::unique_ptr<MCCodeEmitter> &&MCE) {
1695 return new X86ELFStreamer(Context, std::move(MAB), std::move(MOW),
1696 std::move(MCE));
1697}
1698