1//===- AArch64.cpp --------------------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "InputFiles.h"
10#include "OutputSections.h"
11#include "RelocScan.h"
12#include "Symbols.h"
13#include "SyntheticSections.h"
14#include "Target.h"
15#include "TargetImpl.h"
16#include "llvm/BinaryFormat/ELF.h"
17#include "llvm/Support/Endian.h"
18
19using namespace llvm;
20using namespace llvm::support::endian;
21using namespace llvm::ELF;
22using namespace lld;
23using namespace lld::elf;
24
25// Page(Expr) is the page address of the expression Expr, defined
26// as (Expr & ~0xFFF). (This applies even if the machine page size
27// supported by the platform has a different value.)
28uint64_t elf::getAArch64Page(uint64_t expr) {
29 return expr & ~static_cast<uint64_t>(0xFFF);
30}
31
32// A BTI landing pad is a valid target for an indirect branch when the Branch
33// Target Identification has been enabled. As linker generated branches are
34// via x16 the BTI landing pads are defined as: BTI C, BTI J, BTI JC, PACIASP,
35// PACIBSP.
36bool elf::isAArch64BTILandingPad(Ctx &ctx, Symbol &s, int64_t a) {
37 // PLT entries accessed indirectly have a BTI c.
38 if (s.isInPlt(ctx))
39 return true;
40 Defined *d = dyn_cast<Defined>(Val: &s);
41 if (!isa_and_nonnull<InputSection>(Val: d->section))
42 // All places that we cannot disassemble are responsible for making
43 // the target a BTI landing pad.
44 return true;
45 InputSection *isec = cast<InputSection>(Val: d->section);
46 uint64_t off = d->value + a;
47 // Likely user error, but protect ourselves against out of bounds
48 // access.
49 if (off >= isec->getSize())
50 return true;
51 const uint8_t *buf = isec->content().begin();
52 // Synthetic sections may have a size but empty data - Assume that they won't
53 // contain a landing pad
54 if (buf == nullptr && isa<SyntheticSection>(Val: isec))
55 return false;
56
57 const uint32_t instr = read32le(P: buf + off);
58 // All BTI instructions are HINT instructions which all have same encoding
59 // apart from bits [11:5]
60 if ((instr & 0xd503201f) == 0xd503201f &&
61 is_contained(Set: {/*PACIASP*/ 0xd503233f, /*PACIBSP*/ 0xd503237f,
62 /*BTI C*/ 0xd503245f, /*BTI J*/ 0xd503249f,
63 /*BTI JC*/ 0xd50324df},
64 Element: instr))
65 return true;
66 return false;
67}
68
69namespace {
70class AArch64 : public TargetInfo {
71public:
72 AArch64(Ctx &);
73 RelExpr getRelExpr(RelType type, const Symbol &s,
74 const uint8_t *loc) const override;
75 RelType getDynRel(RelType type) const override;
76 int64_t getImplicitAddend(const uint8_t *buf, RelType type) const override;
77 void writeGotPlt(uint8_t *buf, const Symbol &s) const override;
78 void writeIgotPlt(uint8_t *buf, const Symbol &s) const override;
79 void writePltHeader(uint8_t *buf) const override;
80 void writePlt(uint8_t *buf, const Symbol &sym,
81 uint64_t pltEntryAddr) const override;
82 template <class ELFT, class RelTy>
83 void scanSectionImpl(InputSectionBase &sec, Relocs<RelTy> rels,
84 unsigned shard);
85 void scanSection(InputSectionBase &sec, unsigned shard) override {
86 if (ctx.arg.ekind == ELF64BEKind)
87 elf::scanSection1<AArch64, ELF64BE>(target&: *this, sec, shard);
88 else
89 elf::scanSection1<AArch64, ELF64LE>(target&: *this, sec, shard);
90 }
91 bool needsThunk(RelExpr expr, RelType type, const InputFile *file,
92 uint64_t branchAddr, const Symbol &s,
93 int64_t a) const override;
94 uint32_t getThunkSectionSpacing() const override;
95 bool inBranchRange(RelType type, uint64_t src, uint64_t dst) const override;
96 bool usesOnlyLowPageBits(RelType type) const override;
97 void relocate(uint8_t *loc, const Relocation &rel,
98 uint64_t val) const override;
99 void relocateAlloc(InputSection &sec, uint8_t *buf) const override;
100 void applyBranchToBranchOpt() const override;
101
102private:
103 void relaxTlsGdToLe(uint8_t *loc, const Relocation &rel, uint64_t val) const;
104 void relaxTlsGdToIe(uint8_t *loc, const Relocation &rel, uint64_t val) const;
105 void relaxTlsIeToLe(uint8_t *loc, const Relocation &rel, uint64_t val) const;
106 void relaxAuthTlsDescForNonPreemptibleUndefined(uint8_t *loc,
107 const Relocation &rel) const;
108};
109
110struct AArch64Relaxer {
111 Ctx &ctx;
112 SmallPtrSet<Symbol *, 32> unsafeToRelaxAdrpLdr;
113
114 AArch64Relaxer(Ctx &ctx, ArrayRef<Relocation> relocs, uint64_t secAddr,
115 uint8_t *buf);
116 bool tryRelaxAdrpAdd(const Relocation &adrpRel, const Relocation &addRel,
117 uint64_t secAddr, uint8_t *buf) const;
118 bool tryRelaxAdrpLdr(const Relocation &adrpRel, const Relocation &ldrRel,
119 uint64_t secAddr, uint8_t *buf) const;
120 bool isLegalAdrpLdrRelaxationCandidate(const Relocation &adrpRel,
121 const Relocation &ldrRel,
122 uint64_t secAddr, uint8_t *buf) const;
123};
124} // namespace
125
126// Return the bits [Start, End] from Val shifted Start bits.
127// For instance, getBits(0xF0, 4, 8) returns 0xF.
128static uint64_t getBits(uint64_t val, int start, int end) {
129 uint64_t mask = ((uint64_t)1 << (end + 1 - start)) - 1;
130 return (val >> start) & mask;
131}
132
133AArch64::AArch64(Ctx &ctx) : TargetInfo(ctx) {
134 copyRel = R_AARCH64_COPY;
135 relativeRel = R_AARCH64_RELATIVE;
136 iRelativeRel = R_AARCH64_IRELATIVE;
137 iRelSymbolicRel = R_AARCH64_FUNCINIT64;
138 gotRel = R_AARCH64_GLOB_DAT;
139 pltRel = R_AARCH64_JUMP_SLOT;
140 symbolicRel = R_AARCH64_ABS64;
141 tlsDescRel = R_AARCH64_TLSDESC;
142 tlsGotRel = R_AARCH64_TLS_TPREL64;
143 pltHeaderSize = 32;
144 pltEntrySize = 16;
145 ipltEntrySize = 16;
146 defaultMaxPageSize = 65536;
147
148 // Align to the 2 MiB page size (known as a superpage or huge page).
149 // FreeBSD automatically promotes 2 MiB-aligned allocations.
150 defaultImageBase = 0x200000;
151
152 needsThunks = true;
153}
154
155// Only needed to support relocations used by relocateNonAlloc and
156// preprocessRelocs.
157RelExpr AArch64::getRelExpr(RelType type, const Symbol &s,
158 const uint8_t *loc) const {
159 switch (type) {
160 case R_AARCH64_ABS32:
161 case R_AARCH64_ABS64:
162 return R_ABS;
163 case R_AARCH64_PREL32:
164 case R_AARCH64_PREL64:
165 return R_PC;
166 case R_AARCH64_TLS_DTPREL64:
167 return R_DTPREL;
168 case R_AARCH64_NONE:
169 return R_NONE;
170 default:
171 Err(ctx) << getErrorLoc(ctx, loc) << "unknown relocation (" << type.v
172 << ") against symbol " << &s;
173 return R_NONE;
174 }
175}
176
177bool AArch64::usesOnlyLowPageBits(RelType type) const {
178 switch (type) {
179 default:
180 return false;
181 case R_AARCH64_ADD_ABS_LO12_NC:
182 case R_AARCH64_LD64_GOT_LO12_NC:
183 case R_AARCH64_AUTH_LD64_GOT_LO12_NC:
184 case R_AARCH64_AUTH_GOT_ADD_LO12_NC:
185 case R_AARCH64_LDST128_ABS_LO12_NC:
186 case R_AARCH64_LDST16_ABS_LO12_NC:
187 case R_AARCH64_LDST32_ABS_LO12_NC:
188 case R_AARCH64_LDST64_ABS_LO12_NC:
189 case R_AARCH64_LDST8_ABS_LO12_NC:
190 case R_AARCH64_TLSDESC_ADD_LO12:
191 case R_AARCH64_TLSDESC_LD64_LO12:
192 case R_AARCH64_TLSIE_LD64_GOTTPREL_LO12_NC:
193 return true;
194 }
195}
196
197template <class ELFT, class RelTy>
198void AArch64::scanSectionImpl(InputSectionBase &sec, Relocs<RelTy> rels,
199 unsigned shard) {
200 RelocScan rs(ctx, &sec, shard);
201 sec.relocations.reserve(N: rels.size());
202
203 for (auto it = rels.begin(); it != rels.end(); ++it) {
204 const RelTy &rel = *it;
205 uint32_t symIdx = rel.getSymbol(false);
206 Symbol &sym = sec.getFile<ELFT>()->getSymbol(symIdx);
207 uint64_t offset = rel.r_offset;
208 RelType type = rel.getType(false);
209 if (sym.isUndefined() && symIdx != 0 &&
210 rs.maybeReportUndefined(sym&: cast<Undefined>(Val&: sym), offset))
211 continue;
212 int64_t addend = rs.getAddend<ELFT>(rel, type);
213 RelExpr expr;
214 // Relocation types that only need a RelExpr set `expr` and break out of
215 // the switch to reach rs.process(). Types that need special handling
216 // (fast-path helpers, TLS) call a handler and use `continue`.
217
218 auto handleTlsDescAuth = [&sym, &sec, type, offset,
219 addend](RelExpr tlsdescExpr) {
220 if (sym.isUndefined() && !sym.isPreemptible) {
221 // Resolves to `addend`. Handle in
222 // relaxAuthTlsDescForNonPreemptibleUndefined
223 sec.addReloc(r: {.expr: R_TPREL, .type: type, .offset: offset, .addend: addend, .sym: &sym});
224 } else {
225 sym.setFlags(NEEDS_TLSDESC_AUTH);
226 sec.addReloc(r: {.expr: tlsdescExpr, .type: type, .offset: offset, .addend: addend, .sym: &sym});
227 }
228 };
229
230 switch (type) {
231 case R_AARCH64_NONE:
232 continue;
233
234 // Absolute relocations:
235 case R_AARCH64_ABS16:
236 case R_AARCH64_ABS32:
237 case R_AARCH64_ABS64:
238 case R_AARCH64_FUNCINIT64:
239 case R_AARCH64_ADD_ABS_LO12_NC:
240 case R_AARCH64_LDST128_ABS_LO12_NC:
241 case R_AARCH64_LDST16_ABS_LO12_NC:
242 case R_AARCH64_LDST32_ABS_LO12_NC:
243 case R_AARCH64_LDST64_ABS_LO12_NC:
244 case R_AARCH64_LDST8_ABS_LO12_NC:
245 case R_AARCH64_MOVW_SABS_G0:
246 case R_AARCH64_MOVW_SABS_G1:
247 case R_AARCH64_MOVW_SABS_G2:
248 case R_AARCH64_MOVW_UABS_G0:
249 case R_AARCH64_MOVW_UABS_G0_NC:
250 case R_AARCH64_MOVW_UABS_G1:
251 case R_AARCH64_MOVW_UABS_G1_NC:
252 case R_AARCH64_MOVW_UABS_G2:
253 case R_AARCH64_MOVW_UABS_G2_NC:
254 case R_AARCH64_MOVW_UABS_G3:
255 expr = R_ABS;
256 break;
257
258 case R_AARCH64_AUTH_ABS64:
259 expr = RE_AARCH64_AUTH;
260 break;
261
262 case R_AARCH64_PATCHINST:
263 if (!isAbsolute(sym))
264 Err(ctx) << getErrorLoc(ctx, loc: sec.content().data() + offset)
265 << "R_AARCH64_PATCHINST relocation against non-absolute "
266 "symbol "
267 << &sym;
268 expr = R_ABS;
269 break;
270
271 // PC-relative relocations:
272 case R_AARCH64_PREL16:
273 case R_AARCH64_PREL32:
274 case R_AARCH64_PREL64:
275 case R_AARCH64_ADR_PREL_LO21:
276 case R_AARCH64_LD_PREL_LO19:
277 case R_AARCH64_MOVW_PREL_G0:
278 case R_AARCH64_MOVW_PREL_G0_NC:
279 case R_AARCH64_MOVW_PREL_G1:
280 case R_AARCH64_MOVW_PREL_G1_NC:
281 case R_AARCH64_MOVW_PREL_G2:
282 case R_AARCH64_MOVW_PREL_G2_NC:
283 case R_AARCH64_MOVW_PREL_G3:
284 rs.processR_PC(type, offset, addend, sym);
285 continue;
286
287 // Page-PC relocations:
288 case R_AARCH64_ADR_PREL_PG_HI21:
289 case R_AARCH64_ADR_PREL_PG_HI21_NC:
290 expr = RE_AARCH64_PAGE_PC;
291 break;
292
293 // PLT-generating relocations:
294 case R_AARCH64_PLT32:
295 sym.thunkAccessed = true;
296 [[fallthrough]];
297 case R_AARCH64_CALL26:
298 case R_AARCH64_CONDBR19:
299 case R_AARCH64_JUMP26:
300 case R_AARCH64_TSTBR14:
301 rs.processR_PLT_PC(type, offset, addend, sym);
302 continue;
303
304 // GOT relocations:
305 case R_AARCH64_ADR_GOT_PAGE:
306 expr = RE_AARCH64_GOT_PAGE_PC;
307 break;
308 case R_AARCH64_LD64_GOT_LO12_NC:
309 expr = R_GOT;
310 break;
311 case R_AARCH64_LD64_GOTPAGE_LO15:
312 expr = RE_AARCH64_GOT_PAGE;
313 break;
314 case R_AARCH64_GOTPCREL32:
315 case R_AARCH64_GOT_LD_PREL19:
316 expr = R_GOT_PC;
317 break;
318
319 // AUTH GOT relocations. Handle flags here, as rs.process assumes R_GOT is a
320 // normal non-AUTH GOT entry.
321 case R_AARCH64_AUTH_LD64_GOT_LO12_NC:
322 case R_AARCH64_AUTH_GOT_ADD_LO12_NC:
323 sym.setFlags(NEEDS_GOT_AUTH);
324 rs.processAux(expr: R_GOT, type, offset, sym, addend);
325 continue;
326 case R_AARCH64_AUTH_GOT_LD_PREL19:
327 case R_AARCH64_AUTH_GOT_ADR_PREL_LO21:
328 sym.setFlags(NEEDS_GOT_AUTH);
329 rs.processAux(expr: R_GOT_PC, type, offset, sym, addend);
330 continue;
331 case R_AARCH64_AUTH_ADR_GOT_PAGE:
332 sym.setFlags(NEEDS_GOT_AUTH);
333 rs.processAux(expr: RE_AARCH64_GOT_PAGE_PC, type, offset, sym, addend);
334 continue;
335
336 // TLS LE relocations:
337 case R_AARCH64_TLSLE_ADD_TPREL_HI12:
338 case R_AARCH64_TLSLE_ADD_TPREL_LO12:
339 case R_AARCH64_TLSLE_ADD_TPREL_LO12_NC:
340 case R_AARCH64_TLSLE_LDST8_TPREL_LO12:
341 case R_AARCH64_TLSLE_LDST8_TPREL_LO12_NC:
342 case R_AARCH64_TLSLE_LDST16_TPREL_LO12:
343 case R_AARCH64_TLSLE_LDST16_TPREL_LO12_NC:
344 case R_AARCH64_TLSLE_LDST32_TPREL_LO12:
345 case R_AARCH64_TLSLE_LDST32_TPREL_LO12_NC:
346 case R_AARCH64_TLSLE_LDST64_TPREL_LO12:
347 case R_AARCH64_TLSLE_LDST64_TPREL_LO12_NC:
348 case R_AARCH64_TLSLE_LDST128_TPREL_LO12:
349 case R_AARCH64_TLSLE_LDST128_TPREL_LO12_NC:
350 case R_AARCH64_TLSLE_MOVW_TPREL_G0:
351 case R_AARCH64_TLSLE_MOVW_TPREL_G0_NC:
352 case R_AARCH64_TLSLE_MOVW_TPREL_G1:
353 case R_AARCH64_TLSLE_MOVW_TPREL_G1_NC:
354 case R_AARCH64_TLSLE_MOVW_TPREL_G2:
355 if (rs.checkTlsLe(offset, sym, type))
356 continue;
357 expr = R_TPREL;
358 break;
359
360 // TLS IE relocations:
361 case R_AARCH64_TLSIE_ADR_GOTTPREL_PAGE21:
362 rs.handleTlsIe(ieExpr: RE_AARCH64_GOT_PAGE_PC, type, offset, addend, sym);
363 continue;
364 case R_AARCH64_TLSIE_LD64_GOTTPREL_LO12_NC:
365 rs.handleTlsIe(ieExpr: R_GOT, type, offset, addend, sym);
366 continue;
367
368 // TLSDESC relocations:
369 case R_AARCH64_TLSDESC_ADR_PAGE21:
370 rs.handleTlsDesc(sharedExpr: RE_AARCH64_TLSDESC_PAGE, ieExpr: RE_AARCH64_GOT_PAGE_PC, type,
371 offset, addend, sym);
372 continue;
373 case R_AARCH64_TLSDESC_LD64_LO12:
374 case R_AARCH64_TLSDESC_ADD_LO12:
375 rs.handleTlsDesc(sharedExpr: R_TLSDESC, ieExpr: R_GOT, type, offset, addend, sym);
376 continue;
377 case R_AARCH64_TLSDESC_CALL:
378 if (!ctx.arg.shared)
379 sec.addReloc(r: {.expr: R_TPREL, .type: type, .offset: offset, .addend: addend, .sym: &sym});
380 continue;
381
382 // AUTH TLSDESC relocations. Do not optimize to LE/IE because PAUTHELF64
383 // only supports the descriptor based TLS (TLSDESC).
384 // https://github.com/ARM-software/abi-aa/blob/main/pauthabielf64/pauthabielf64.rst#general-restrictions
385 case R_AARCH64_AUTH_TLSDESC_ADR_PAGE21:
386 handleTlsDescAuth(RE_AARCH64_TLSDESC_PAGE);
387 continue;
388 case R_AARCH64_AUTH_TLSDESC_LD64_LO12:
389 case R_AARCH64_AUTH_TLSDESC_ADD_LO12:
390 handleTlsDescAuth(R_TLSDESC);
391 continue;
392 case R_AARCH64_AUTH_TLSDESC_CALL:
393 if (sym.isUndefined() && !sym.isPreemptible)
394 sec.addReloc(r: {.expr: R_TPREL, .type: type, .offset: offset, .addend: addend, .sym: &sym});
395 else
396 sym.setFlags(NEEDS_TLSDESC_AUTH);
397 continue;
398
399 default:
400 Err(ctx) << getErrorLoc(ctx, loc: sec.content().data() + offset)
401 << "unknown relocation (" << type.v << ") against symbol "
402 << &sym;
403 continue;
404 }
405 rs.process(expr, type, offset, sym, addend);
406 }
407
408 if (ctx.arg.branchToBranch)
409 llvm::stable_sort(sec.relocs(),
410 [](auto &l, auto &r) { return l.offset < r.offset; });
411}
412
413RelType AArch64::getDynRel(RelType type) const {
414 if (type == R_AARCH64_ABS64 || type == R_AARCH64_AUTH_ABS64 ||
415 type == R_AARCH64_FUNCINIT64)
416 return type;
417 return R_AARCH64_NONE;
418}
419
420int64_t AArch64::getImplicitAddend(const uint8_t *buf, RelType type) const {
421 switch (type) {
422 case R_AARCH64_TLSDESC:
423 return read64(ctx, p: buf + 8);
424 case R_AARCH64_NONE:
425 case R_AARCH64_GLOB_DAT:
426 case R_AARCH64_AUTH_GLOB_DAT:
427 case R_AARCH64_JUMP_SLOT:
428 return 0;
429 case R_AARCH64_ABS16:
430 case R_AARCH64_PREL16:
431 return SignExtend64<16>(x: read16(ctx, p: buf));
432 case R_AARCH64_ABS32:
433 case R_AARCH64_PREL32:
434 return SignExtend64<32>(x: read32(ctx, p: buf));
435 case R_AARCH64_ABS64:
436 case R_AARCH64_PREL64:
437 case R_AARCH64_RELATIVE:
438 case R_AARCH64_IRELATIVE:
439 case R_AARCH64_TLS_TPREL64:
440 return read64(ctx, p: buf);
441
442 // The following relocation types all point at instructions, and
443 // relocate an immediate field in the instruction.
444 //
445 // The general rule, from AAELF64 §5.7.2 "Addends and PC-bias",
446 // says: "If the relocation relocates an instruction the immediate
447 // field of the instruction is extracted, scaled as required by
448 // the instruction field encoding, and sign-extended to 64 bits".
449
450 // The R_AARCH64_MOVW family operates on wide MOV/MOVK/MOVZ
451 // instructions, which have a 16-bit immediate field with its low
452 // bit in bit 5 of the instruction encoding. When the immediate
453 // field is used as an implicit addend for REL-type relocations,
454 // it is treated as added to the low bits of the output value, not
455 // shifted depending on the relocation type.
456 //
457 // This allows REL relocations to express the requirement 'please
458 // add 12345 to this symbol value and give me the four 16-bit
459 // chunks of the result', by putting the same addend 12345 in all
460 // four instructions. Carries between the 16-bit chunks are
461 // handled correctly, because the whole 64-bit addition is done
462 // once per relocation.
463 case R_AARCH64_MOVW_UABS_G0:
464 case R_AARCH64_MOVW_UABS_G0_NC:
465 case R_AARCH64_MOVW_UABS_G1:
466 case R_AARCH64_MOVW_UABS_G1_NC:
467 case R_AARCH64_MOVW_UABS_G2:
468 case R_AARCH64_MOVW_UABS_G2_NC:
469 case R_AARCH64_MOVW_UABS_G3:
470 return SignExtend64<16>(x: getBits(val: read32le(P: buf), start: 5, end: 20));
471
472 // R_AARCH64_TSTBR14 points at a TBZ or TBNZ instruction, which
473 // has a 14-bit offset measured in instructions, i.e. shifted left
474 // by 2.
475 case R_AARCH64_TSTBR14:
476 return SignExtend64<16>(x: getBits(val: read32le(P: buf), start: 5, end: 18) << 2);
477
478 // R_AARCH64_CONDBR19 operates on the ordinary B.cond instruction,
479 // which has a 19-bit offset measured in instructions.
480 //
481 // R_AARCH64_LD_PREL_LO19 operates on the LDR (literal)
482 // instruction, which also has a 19-bit offset, measured in 4-byte
483 // chunks. So the calculation is the same as for
484 // R_AARCH64_CONDBR19.
485 case R_AARCH64_CONDBR19:
486 case R_AARCH64_LD_PREL_LO19:
487 return SignExtend64<21>(x: getBits(val: read32le(P: buf), start: 5, end: 23) << 2);
488
489 // R_AARCH64_ADD_ABS_LO12_NC operates on ADD (immediate). The
490 // immediate can optionally be shifted left by 12 bits, but this
491 // relocation is intended for the case where it is not.
492 case R_AARCH64_ADD_ABS_LO12_NC:
493 return SignExtend64<12>(x: getBits(val: read32le(P: buf), start: 10, end: 21));
494
495 // R_AARCH64_ADR_PREL_LO21 operates on an ADR instruction, whose
496 // 21-bit immediate is split between two bits high up in the word
497 // (in fact the two _lowest_ order bits of the value) and 19 bits
498 // lower down.
499 //
500 // R_AARCH64_ADR_PREL_PG_HI21[_NC] operate on an ADRP instruction,
501 // which encodes the immediate in the same way, but will shift it
502 // left by 12 bits when the instruction executes. For the same
503 // reason as the MOVW family, we don't apply that left shift here.
504 case R_AARCH64_ADR_PREL_LO21:
505 case R_AARCH64_ADR_PREL_PG_HI21:
506 case R_AARCH64_ADR_PREL_PG_HI21_NC:
507 return SignExtend64<21>(x: (getBits(val: read32le(P: buf), start: 5, end: 23) << 2) |
508 getBits(val: read32le(P: buf), start: 29, end: 30));
509
510 // R_AARCH64_{JUMP,CALL}26 operate on B and BL, which have a
511 // 26-bit offset measured in instructions.
512 case R_AARCH64_JUMP26:
513 case R_AARCH64_CALL26:
514 return SignExtend64<28>(x: getBits(val: read32le(P: buf), start: 0, end: 25) << 2);
515
516 default:
517 InternalErr(ctx, buf) << "cannot read addend for relocation " << type;
518 return 0;
519 }
520}
521
522void AArch64::writeGotPlt(uint8_t *buf, const Symbol &) const {
523 write64(ctx, p: buf, v: ctx.in.plt->getVA());
524}
525
526void AArch64::writeIgotPlt(uint8_t *buf, const Symbol &s) const {
527 if (ctx.arg.writeAddends)
528 write64(ctx, p: buf, v: s.getVA(ctx));
529}
530
531void AArch64::writePltHeader(uint8_t *buf) const {
532 const uint8_t pltData[] = {
533 0xf0, 0x7b, 0xbf, 0xa9, // stp x16, x30, [sp,#-16]!
534 0x10, 0x00, 0x00, 0x90, // adrp x16, Page(&(.got.plt[2]))
535 0x11, 0x02, 0x40, 0xf9, // ldr x17, [x16, Offset(&(.got.plt[2]))]
536 0x10, 0x02, 0x00, 0x91, // add x16, x16, Offset(&(.got.plt[2]))
537 0x20, 0x02, 0x1f, 0xd6, // br x17
538 0x1f, 0x20, 0x03, 0xd5, // nop
539 0x1f, 0x20, 0x03, 0xd5, // nop
540 0x1f, 0x20, 0x03, 0xd5 // nop
541 };
542 memcpy(dest: buf, src: pltData, n: sizeof(pltData));
543
544 uint64_t got = ctx.in.gotPlt->getVA();
545 uint64_t plt = ctx.in.plt->getVA();
546 relocateNoSym(loc: buf + 4, type: R_AARCH64_ADR_PREL_PG_HI21,
547 val: getAArch64Page(expr: got + 16) - getAArch64Page(expr: plt + 4));
548 relocateNoSym(loc: buf + 8, type: R_AARCH64_LDST64_ABS_LO12_NC, val: got + 16);
549 relocateNoSym(loc: buf + 12, type: R_AARCH64_ADD_ABS_LO12_NC, val: got + 16);
550}
551
552void AArch64::writePlt(uint8_t *buf, const Symbol &sym,
553 uint64_t pltEntryAddr) const {
554 const uint8_t inst[] = {
555 0x10, 0x00, 0x00, 0x90, // adrp x16, Page(&(.got.plt[n]))
556 0x11, 0x02, 0x40, 0xf9, // ldr x17, [x16, Offset(&(.got.plt[n]))]
557 0x10, 0x02, 0x00, 0x91, // add x16, x16, Offset(&(.got.plt[n]))
558 0x20, 0x02, 0x1f, 0xd6 // br x17
559 };
560 memcpy(dest: buf, src: inst, n: sizeof(inst));
561
562 uint64_t gotPltEntryAddr = sym.getGotPltVA(ctx);
563 relocateNoSym(loc: buf, type: R_AARCH64_ADR_PREL_PG_HI21,
564 val: getAArch64Page(expr: gotPltEntryAddr) - getAArch64Page(expr: pltEntryAddr));
565 relocateNoSym(loc: buf + 4, type: R_AARCH64_LDST64_ABS_LO12_NC, val: gotPltEntryAddr);
566 relocateNoSym(loc: buf + 8, type: R_AARCH64_ADD_ABS_LO12_NC, val: gotPltEntryAddr);
567}
568
569bool AArch64::needsThunk(RelExpr expr, RelType type, const InputFile *file,
570 uint64_t branchAddr, const Symbol &s,
571 int64_t a) const {
572 // If s is an undefined weak symbol and does not have a PLT entry then it will
573 // be resolved as a branch to the next instruction. If it is hidden, its
574 // binding has been converted to local, so we just check isUndefined() here. A
575 // undefined non-weak symbol will have been errored.
576 if (s.isUndefined() && !s.isInPlt(ctx))
577 return false;
578 // ELF for the ARM 64-bit architecture, section Call and Jump relocations
579 // only permits range extension thunks for R_AARCH64_CALL26 and
580 // R_AARCH64_JUMP26 relocation types.
581 if (type != R_AARCH64_CALL26 && type != R_AARCH64_JUMP26 &&
582 type != R_AARCH64_PLT32)
583 return false;
584 uint64_t dst = expr == R_PLT_PC ? s.getPltVA(ctx) : s.getVA(ctx, addend: a);
585 return !inBranchRange(type, src: branchAddr, dst);
586}
587
588uint32_t AArch64::getThunkSectionSpacing() const {
589 // See comment in Arch/ARM.cpp for a more detailed explanation of
590 // getThunkSectionSpacing(). For AArch64 the only branches we are permitted to
591 // Thunk have a range of +/- 128 MiB
592 return (128 * 1024 * 1024) - 0x30000;
593}
594
595bool AArch64::inBranchRange(RelType type, uint64_t src, uint64_t dst) const {
596 if (type != R_AARCH64_CALL26 && type != R_AARCH64_JUMP26 &&
597 type != R_AARCH64_PLT32)
598 return true;
599 // The AArch64 call and unconditional branch instructions have a range of
600 // +/- 128 MiB. The PLT32 relocation supports a range up to +/- 2 GiB.
601 uint64_t range =
602 type == R_AARCH64_PLT32 ? (UINT64_C(1) << 31) : (128 * 1024 * 1024);
603 if (dst > src) {
604 // Immediate of branch is signed.
605 range -= 4;
606 return dst - src <= range;
607 }
608 return src - dst <= range;
609}
610
611static void write32AArch64Addr(uint8_t *l, uint64_t imm) {
612 uint32_t immLo = (imm & 0x3) << 29;
613 uint32_t immHi = (imm & 0x1FFFFC) << 3;
614 uint64_t mask = (0x3 << 29) | (0x1FFFFC << 3);
615 write32le(P: l, V: (read32le(P: l) & ~mask) | immLo | immHi);
616}
617
618static void writeMaskedBits32le(uint8_t *p, int32_t v, uint32_t mask) {
619 write32le(P: p, V: (read32le(P: p) & ~mask) | v);
620}
621
622// Update the immediate field in a AARCH64 ldr, str, and add instruction.
623static void write32Imm12(uint8_t *l, uint64_t imm) {
624 writeMaskedBits32le(p: l, v: (imm & 0xFFF) << 10, mask: 0xFFF << 10);
625}
626
627// Update the immediate field in an AArch64 movk, movn or movz instruction
628// for a signed relocation, and update the opcode of a movn or movz instruction
629// to match the sign of the operand.
630static void writeSMovWImm(uint8_t *loc, uint32_t imm) {
631 uint32_t inst = read32le(P: loc);
632 // Opcode field is bits 30, 29, with 10 = movz, 00 = movn and 11 = movk.
633 if (!(inst & (1 << 29))) {
634 // movn or movz.
635 if (imm & 0x10000) {
636 // Change opcode to movn, which takes an inverted operand.
637 imm ^= 0xFFFF;
638 inst &= ~(1 << 30);
639 } else {
640 // Change opcode to movz.
641 inst |= 1 << 30;
642 }
643 }
644 write32le(P: loc, V: inst | ((imm & 0xFFFF) << 5));
645}
646
647void AArch64::relocate(uint8_t *loc, const Relocation &rel,
648 uint64_t val) const {
649 switch (rel.type) {
650 case R_AARCH64_ABS16:
651 checkIntUInt(ctx, loc, v: val, n: 16, rel);
652 write16(ctx, p: loc, v: val);
653 break;
654 case R_AARCH64_PREL16:
655 checkInt(ctx, loc, v: val, n: 16, rel);
656 write16(ctx, p: loc, v: val);
657 break;
658 case R_AARCH64_ABS32:
659 checkIntUInt(ctx, loc, v: val, n: 32, rel);
660 write32(ctx, p: loc, v: val);
661 break;
662 case R_AARCH64_PATCHINST:
663 if (!rel.sym->isUndefined()) {
664 checkUInt(ctx, loc, v: val, n: 32, rel);
665 write32le(P: loc, V: val);
666 }
667 break;
668 case R_AARCH64_PREL32:
669 case R_AARCH64_PLT32:
670 case R_AARCH64_GOTPCREL32:
671 checkInt(ctx, loc, v: val, n: 32, rel);
672 write32(ctx, p: loc, v: val);
673 break;
674 case R_AARCH64_ABS64:
675 write64(ctx, p: loc, v: val);
676 break;
677 case R_AARCH64_PREL64:
678 write64(ctx, p: loc, v: val);
679 break;
680 case R_AARCH64_AUTH_ABS64:
681 if (rel.sym->isUndefined() && !rel.sym->isPreemptible) {
682 // Resolve to the addend. No dynamic relocation and corresponding signing
683 // schema encoding is needed.
684 write64(ctx, p: loc, v: val);
685 } else {
686 // This is used for the addend of a .relr.auth.dyn entry,
687 // which is a 32-bit value; the upper 32 bits are used to
688 // encode the schema.
689 checkInt(ctx, loc, v: val, n: 32, rel);
690 write32(ctx, p: loc, v: val);
691 }
692 break;
693 case R_AARCH64_TLS_DTPREL64:
694 write64(ctx, p: loc, v: val);
695 break;
696 case R_AARCH64_ADD_ABS_LO12_NC:
697 case R_AARCH64_AUTH_GOT_ADD_LO12_NC:
698 write32Imm12(l: loc, imm: val);
699 break;
700 case R_AARCH64_ADR_GOT_PAGE:
701 case R_AARCH64_AUTH_ADR_GOT_PAGE:
702 case R_AARCH64_ADR_PREL_PG_HI21:
703 case R_AARCH64_TLSIE_ADR_GOTTPREL_PAGE21:
704 case R_AARCH64_TLSDESC_ADR_PAGE21:
705 case R_AARCH64_AUTH_TLSDESC_ADR_PAGE21:
706 checkInt(ctx, loc, v: val, n: 33, rel);
707 [[fallthrough]];
708 case R_AARCH64_ADR_PREL_PG_HI21_NC:
709 write32AArch64Addr(l: loc, imm: val >> 12);
710 break;
711 case R_AARCH64_ADR_PREL_LO21:
712 case R_AARCH64_AUTH_GOT_ADR_PREL_LO21:
713 checkInt(ctx, loc, v: val, n: 21, rel);
714 write32AArch64Addr(l: loc, imm: val);
715 break;
716 case R_AARCH64_JUMP26:
717 // Normally we would just write the bits of the immediate field, however
718 // when patching instructions for the cpu errata fix -fix-cortex-a53-843419
719 // we want to replace a non-branch instruction with a branch immediate
720 // instruction. By writing all the bits of the instruction including the
721 // opcode and the immediate (0 001 | 01 imm26) we can do this
722 // transformation by placing a R_AARCH64_JUMP26 relocation at the offset of
723 // the instruction we want to patch.
724 write32le(P: loc, V: 0x14000000);
725 [[fallthrough]];
726 case R_AARCH64_CALL26:
727 checkInt(ctx, loc, v: val, n: 28, rel);
728 writeMaskedBits32le(p: loc, v: (val & 0x0FFFFFFC) >> 2, mask: 0x0FFFFFFC >> 2);
729 break;
730 case R_AARCH64_CONDBR19:
731 case R_AARCH64_LD_PREL_LO19:
732 case R_AARCH64_GOT_LD_PREL19:
733 case R_AARCH64_AUTH_GOT_LD_PREL19:
734 checkAlignment(ctx, loc, v: val, n: 4, rel);
735 checkInt(ctx, loc, v: val, n: 21, rel);
736 writeMaskedBits32le(p: loc, v: (val & 0x1FFFFC) << 3, mask: 0x1FFFFC << 3);
737 break;
738 case R_AARCH64_TLSLE_LDST8_TPREL_LO12:
739 checkUInt(ctx, loc, v: val, n: 12, rel);
740 [[fallthrough]];
741 case R_AARCH64_LDST8_ABS_LO12_NC:
742 case R_AARCH64_TLSLE_LDST8_TPREL_LO12_NC:
743 write32Imm12(l: loc, imm: getBits(val, start: 0, end: 11));
744 break;
745 case R_AARCH64_TLSLE_LDST16_TPREL_LO12:
746 checkUInt(ctx, loc, v: val, n: 12, rel);
747 [[fallthrough]];
748 case R_AARCH64_LDST16_ABS_LO12_NC:
749 case R_AARCH64_TLSLE_LDST16_TPREL_LO12_NC:
750 checkAlignment(ctx, loc, v: val, n: 2, rel);
751 write32Imm12(l: loc, imm: getBits(val, start: 1, end: 11));
752 break;
753 case R_AARCH64_TLSLE_LDST32_TPREL_LO12:
754 checkUInt(ctx, loc, v: val, n: 12, rel);
755 [[fallthrough]];
756 case R_AARCH64_LDST32_ABS_LO12_NC:
757 case R_AARCH64_TLSLE_LDST32_TPREL_LO12_NC:
758 checkAlignment(ctx, loc, v: val, n: 4, rel);
759 write32Imm12(l: loc, imm: getBits(val, start: 2, end: 11));
760 break;
761 case R_AARCH64_TLSLE_LDST64_TPREL_LO12:
762 checkUInt(ctx, loc, v: val, n: 12, rel);
763 [[fallthrough]];
764 case R_AARCH64_LDST64_ABS_LO12_NC:
765 case R_AARCH64_LD64_GOT_LO12_NC:
766 case R_AARCH64_AUTH_LD64_GOT_LO12_NC:
767 case R_AARCH64_TLSIE_LD64_GOTTPREL_LO12_NC:
768 case R_AARCH64_TLSLE_LDST64_TPREL_LO12_NC:
769 case R_AARCH64_TLSDESC_LD64_LO12:
770 case R_AARCH64_AUTH_TLSDESC_LD64_LO12:
771 checkAlignment(ctx, loc, v: val, n: 8, rel);
772 write32Imm12(l: loc, imm: getBits(val, start: 3, end: 11));
773 break;
774 case R_AARCH64_TLSLE_LDST128_TPREL_LO12:
775 checkUInt(ctx, loc, v: val, n: 12, rel);
776 [[fallthrough]];
777 case R_AARCH64_LDST128_ABS_LO12_NC:
778 case R_AARCH64_TLSLE_LDST128_TPREL_LO12_NC:
779 checkAlignment(ctx, loc, v: val, n: 16, rel);
780 write32Imm12(l: loc, imm: getBits(val, start: 4, end: 11));
781 break;
782 case R_AARCH64_LD64_GOTPAGE_LO15:
783 checkAlignment(ctx, loc, v: val, n: 8, rel);
784 write32Imm12(l: loc, imm: getBits(val, start: 3, end: 14));
785 break;
786 case R_AARCH64_MOVW_UABS_G0:
787 checkUInt(ctx, loc, v: val, n: 16, rel);
788 [[fallthrough]];
789 case R_AARCH64_MOVW_UABS_G0_NC:
790 writeMaskedBits32le(p: loc, v: (val & 0xFFFF) << 5, mask: 0xFFFF << 5);
791 break;
792 case R_AARCH64_MOVW_UABS_G1:
793 checkUInt(ctx, loc, v: val, n: 32, rel);
794 [[fallthrough]];
795 case R_AARCH64_MOVW_UABS_G1_NC:
796 writeMaskedBits32le(p: loc, v: (val & 0xFFFF0000) >> 11, mask: 0xFFFF0000 >> 11);
797 break;
798 case R_AARCH64_MOVW_UABS_G2:
799 checkUInt(ctx, loc, v: val, n: 48, rel);
800 [[fallthrough]];
801 case R_AARCH64_MOVW_UABS_G2_NC:
802 writeMaskedBits32le(p: loc, v: (val & 0xFFFF00000000) >> 27,
803 mask: 0xFFFF00000000 >> 27);
804 break;
805 case R_AARCH64_MOVW_UABS_G3:
806 writeMaskedBits32le(p: loc, v: (val & 0xFFFF000000000000) >> 43,
807 mask: 0xFFFF000000000000 >> 43);
808 break;
809 case R_AARCH64_MOVW_PREL_G0:
810 case R_AARCH64_MOVW_SABS_G0:
811 case R_AARCH64_TLSLE_MOVW_TPREL_G0:
812 checkInt(ctx, loc, v: val, n: 17, rel);
813 [[fallthrough]];
814 case R_AARCH64_MOVW_PREL_G0_NC:
815 case R_AARCH64_TLSLE_MOVW_TPREL_G0_NC:
816 writeSMovWImm(loc, imm: val);
817 break;
818 case R_AARCH64_MOVW_PREL_G1:
819 case R_AARCH64_MOVW_SABS_G1:
820 case R_AARCH64_TLSLE_MOVW_TPREL_G1:
821 checkInt(ctx, loc, v: val, n: 33, rel);
822 [[fallthrough]];
823 case R_AARCH64_MOVW_PREL_G1_NC:
824 case R_AARCH64_TLSLE_MOVW_TPREL_G1_NC:
825 writeSMovWImm(loc, imm: val >> 16);
826 break;
827 case R_AARCH64_MOVW_PREL_G2:
828 case R_AARCH64_MOVW_SABS_G2:
829 case R_AARCH64_TLSLE_MOVW_TPREL_G2:
830 checkInt(ctx, loc, v: val, n: 49, rel);
831 [[fallthrough]];
832 case R_AARCH64_MOVW_PREL_G2_NC:
833 writeSMovWImm(loc, imm: val >> 32);
834 break;
835 case R_AARCH64_MOVW_PREL_G3:
836 writeSMovWImm(loc, imm: val >> 48);
837 break;
838 case R_AARCH64_TSTBR14:
839 checkInt(ctx, loc, v: val, n: 16, rel);
840 writeMaskedBits32le(p: loc, v: (val & 0xFFFC) << 3, mask: 0xFFFC << 3);
841 break;
842 case R_AARCH64_TLSLE_ADD_TPREL_HI12:
843 checkUInt(ctx, loc, v: val, n: 24, rel);
844 if (ctx.arg.relax && (val >> 12) == 0) {
845 uint32_t inst = read32le(P: loc);
846 // The W-form zero-extends Xd, so only the X-form is a nop.
847 if ((inst & (1u << 31)) && (inst & 0x1f) == ((inst >> 5) & 0x1f)) {
848 write32le(P: loc, V: 0xd503201f); // nop
849 break;
850 }
851 }
852 write32Imm12(l: loc, imm: val >> 12);
853 break;
854 case R_AARCH64_TLSLE_ADD_TPREL_LO12:
855 checkUInt(ctx, loc, v: val, n: 12, rel);
856 write32Imm12(l: loc, imm: val);
857 break;
858 case R_AARCH64_TLSLE_ADD_TPREL_LO12_NC:
859 case R_AARCH64_TLSDESC_ADD_LO12:
860 case R_AARCH64_AUTH_TLSDESC_ADD_LO12:
861 write32Imm12(l: loc, imm: val);
862 break;
863 case R_AARCH64_TLSDESC:
864 // For R_AARCH64_TLSDESC the addend is stored in the second 64-bit word.
865 write64(ctx, p: loc + 8, v: val);
866 break;
867 default:
868 llvm_unreachable("unknown relocation");
869 }
870}
871
872void AArch64::relaxAuthTlsDescForNonPreemptibleUndefined(
873 uint8_t *loc, const Relocation &rel) const {
874 // AUTH TLSDESC relocations are in the form:
875 // adrp x0, :tlsdesc_auth:v [R_AARCH64_AUTH_TLSDESC_ADR_PAGE21]
876 // ldr x16, [x0, :tlsdesc_auth_lo12:v] [R_AARCH64_AUTH_TLSDESC_LD64_LO12]
877 // add x0, x0, :tlsdesc_auth_lo12:v [R_AARCH64_AUTH_TLSDESC_ADD_LO12]
878 // .tlsauthdesccall v [R_AARCH64_AUTH_TLSDESC_CALL]
879 // blraa x16, x0
880 // And it can optimized to:
881 // mrs x0, tpidr_el0
882 // neg x0, x0
883 // nop
884 // nop
885
886 switch (rel.type) {
887 case R_AARCH64_AUTH_TLSDESC_ADR_PAGE21:
888 write32le(P: loc, V: 0xd53bd040); // mrs x0, tpidr_el0
889 return;
890 case R_AARCH64_AUTH_TLSDESC_LD64_LO12:
891 write32le(P: loc, V: 0xcb0003e0); // neg x0, x0
892 return;
893 case R_AARCH64_AUTH_TLSDESC_ADD_LO12:
894 case R_AARCH64_AUTH_TLSDESC_CALL:
895 write32le(P: loc, V: 0xd503201f); // nop
896 return;
897 default:
898 llvm_unreachable("unsupported relocation for non-preemptible undefined "
899 "AUTH TLSDESC relaxation");
900 }
901}
902
903void AArch64::relaxTlsGdToLe(uint8_t *loc, const Relocation &rel,
904 uint64_t val) const {
905 // TLSDESC Global-Dynamic relocation are in the form:
906 // adrp x0, :tlsdesc:v [R_AARCH64_TLSDESC_ADR_PAGE21]
907 // ldr x1, [x0, #:tlsdesc_lo12:v [R_AARCH64_TLSDESC_LD64_LO12]
908 // add x0, x0, :tlsdesc_los:v [R_AARCH64_TLSDESC_ADD_LO12]
909 // .tlsdesccall [R_AARCH64_TLSDESC_CALL]
910 // blr x1
911 // And it can optimized to:
912 // movz x0, #0x0, lsl #16
913 // movk x0, #0x10
914 // nop
915 // nop
916 checkUInt(ctx, loc, v: val, n: 32, rel);
917
918 switch (rel.type) {
919 case R_AARCH64_TLSDESC_ADD_LO12:
920 case R_AARCH64_TLSDESC_CALL:
921 write32le(P: loc, V: 0xd503201f); // nop
922 return;
923 case R_AARCH64_TLSDESC_ADR_PAGE21:
924 write32le(P: loc, V: 0xd2a00000 | (((val >> 16) & 0xffff) << 5)); // movz
925 return;
926 case R_AARCH64_TLSDESC_LD64_LO12:
927 write32le(P: loc, V: 0xf2800000 | ((val & 0xffff) << 5)); // movk
928 return;
929 default:
930 llvm_unreachable("unsupported relocation for TLS GD to LE relaxation");
931 }
932}
933
934void AArch64::relaxTlsGdToIe(uint8_t *loc, const Relocation &rel,
935 uint64_t val) const {
936 // TLSDESC Global-Dynamic relocation are in the form:
937 // adrp x0, :tlsdesc:v [R_AARCH64_TLSDESC_ADR_PAGE21]
938 // ldr x1, [x0, #:tlsdesc_lo12:v [R_AARCH64_TLSDESC_LD64_LO12]
939 // add x0, x0, :tlsdesc_los:v [R_AARCH64_TLSDESC_ADD_LO12]
940 // .tlsdesccall [R_AARCH64_TLSDESC_CALL]
941 // blr x1
942 // And it can optimized to:
943 // adrp x0, :gottprel:v
944 // ldr x0, [x0, :gottprel_lo12:v]
945 // nop
946 // nop
947
948 switch (rel.type) {
949 case R_AARCH64_TLSDESC_ADD_LO12:
950 case R_AARCH64_TLSDESC_CALL:
951 write32le(P: loc, V: 0xd503201f); // nop
952 break;
953 case R_AARCH64_TLSDESC_ADR_PAGE21:
954 write32le(P: loc, V: 0x90000000); // adrp
955 relocateNoSym(loc, type: R_AARCH64_TLSIE_ADR_GOTTPREL_PAGE21, val);
956 break;
957 case R_AARCH64_TLSDESC_LD64_LO12:
958 write32le(P: loc, V: 0xf9400000); // ldr
959 relocateNoSym(loc, type: R_AARCH64_TLSIE_LD64_GOTTPREL_LO12_NC, val);
960 break;
961 default:
962 llvm_unreachable("unsupported relocation for TLS GD to IE relaxation");
963 }
964}
965
966void AArch64::relaxTlsIeToLe(uint8_t *loc, const Relocation &rel,
967 uint64_t val) const {
968 checkUInt(ctx, loc, v: val, n: 32, rel);
969
970 if (rel.type == R_AARCH64_TLSIE_ADR_GOTTPREL_PAGE21) {
971 // Generate MOVZ.
972 uint32_t regNo = read32le(P: loc) & 0x1f;
973 write32le(P: loc, V: (0xd2a00000 | regNo) | (((val >> 16) & 0xffff) << 5));
974 return;
975 }
976 if (rel.type == R_AARCH64_TLSIE_LD64_GOTTPREL_LO12_NC) {
977 // Generate MOVK.
978 uint32_t regNo = read32le(P: loc) & 0x1f;
979 write32le(P: loc, V: (0xf2800000 | regNo) | ((val & 0xffff) << 5));
980 return;
981 }
982 llvm_unreachable("invalid relocation for TLS IE to LE relaxation");
983}
984
985AArch64Relaxer::AArch64Relaxer(Ctx &ctx, ArrayRef<Relocation> relocs,
986 uint64_t secAddr, uint8_t *buf)
987 : ctx(ctx) {
988 if (!ctx.arg.relax)
989 return;
990 // For a given symbol R_AARCH64_ADR_GOT_PAGE and R_AARCH64_LD64_GOT_LO12_NC
991 // relaxation is all-or-nothing. We can't relax only some of them, as there
992 // may be a jump destination between the two relocations.
993 size_t i = 0;
994 const size_t size = relocs.size();
995 for (; i != size; ++i) {
996 if (relocs[i].type == R_AARCH64_ADR_GOT_PAGE) {
997 if (i + 1 < size && relocs[i + 1].type == R_AARCH64_LD64_GOT_LO12_NC &&
998 !unsafeToRelaxAdrpLdr.contains(Ptr: relocs[i].sym) &&
999 isLegalAdrpLdrRelaxationCandidate(adrpRel: relocs[i], ldrRel: relocs[i + 1], secAddr,
1000 buf)) {
1001 ++i;
1002 continue;
1003 }
1004 unsafeToRelaxAdrpLdr.insert(Ptr: relocs[i].sym);
1005 } else if (relocs[i].type == R_AARCH64_LD64_GOT_LO12_NC) {
1006 unsafeToRelaxAdrpLdr.insert(Ptr: relocs[i].sym);
1007 }
1008 }
1009}
1010
1011bool AArch64Relaxer::tryRelaxAdrpAdd(const Relocation &adrpRel,
1012 const Relocation &addRel, uint64_t secAddr,
1013 uint8_t *buf) const {
1014 // When the address of sym is within the range of ADR then
1015 // we may relax
1016 // ADRP xn, sym
1017 // ADD xn, xn, :lo12: sym
1018 // to
1019 // NOP
1020 // ADR xn, sym
1021 if (!ctx.arg.relax || addRel.type != R_AARCH64_ADD_ABS_LO12_NC)
1022 return false;
1023 // Check if the relocations apply to consecutive instructions.
1024 if (adrpRel.offset + 4 != addRel.offset)
1025 return false;
1026 if (adrpRel.sym != addRel.sym)
1027 return false;
1028 if (adrpRel.addend != 0 || addRel.addend != 0)
1029 return false;
1030
1031 uint32_t adrpInstr = read32le(P: buf + adrpRel.offset);
1032 uint32_t addInstr = read32le(P: buf + addRel.offset);
1033 // Check if the first instruction is ADRP and the second instruction is ADD.
1034 if ((adrpInstr & 0x9f000000) != 0x90000000 ||
1035 (addInstr & 0xffc00000) != 0x91000000)
1036 return false;
1037 uint32_t adrpDestReg = adrpInstr & 0x1f;
1038 uint32_t addDestReg = addInstr & 0x1f;
1039 uint32_t addSrcReg = (addInstr >> 5) & 0x1f;
1040 if (adrpDestReg != addDestReg || adrpDestReg != addSrcReg)
1041 return false;
1042
1043 Symbol &sym = *adrpRel.sym;
1044 // Check if the address difference is within 1MiB range.
1045 int64_t val = sym.getVA(ctx) - (secAddr + addRel.offset);
1046 if (val < -1024 * 1024 || val >= 1024 * 1024)
1047 return false;
1048
1049 Relocation adrRel = {.expr: R_ABS, .type: R_AARCH64_ADR_PREL_LO21, .offset: addRel.offset,
1050 /*addend=*/0, .sym: &sym};
1051 // nop
1052 write32le(P: buf + adrpRel.offset, V: 0xd503201f);
1053 // adr x_<dest_reg>
1054 write32le(P: buf + adrRel.offset, V: 0x10000000 | adrpDestReg);
1055 ctx.target->relocate(loc: buf + adrRel.offset, rel: adrRel, val);
1056 return true;
1057}
1058
1059bool AArch64Relaxer::isLegalAdrpLdrRelaxationCandidate(
1060 const Relocation &adrpRel, const Relocation &ldrRel, uint64_t secAddr,
1061 uint8_t *buf) const {
1062 // Check if the relocations apply to consecutive instructions.
1063 if (adrpRel.offset + 4 != ldrRel.offset)
1064 return false;
1065 // Check if the relocations reference the same symbol and
1066 // skip undefined, preemptible and STT_GNU_IFUNC symbols.
1067 if (!adrpRel.sym || adrpRel.sym != ldrRel.sym || !adrpRel.sym->isDefined() ||
1068 adrpRel.sym->isPreemptible || adrpRel.sym->isGnuIFunc())
1069 return false;
1070 // Check if the addends of the both relocations are zero.
1071 if (adrpRel.addend != 0 || ldrRel.addend != 0)
1072 return false;
1073 uint32_t adrpInstr = read32le(P: buf + adrpRel.offset);
1074 uint32_t ldrInstr = read32le(P: buf + ldrRel.offset);
1075 // Check if the first instruction is ADRP and the second instruction is LDR.
1076 if ((adrpInstr & 0x9f000000) != 0x90000000 ||
1077 (ldrInstr & 0x3b000000) != 0x39000000)
1078 return false;
1079 // Check the value of the sf bit.
1080 if (!(ldrInstr >> 31))
1081 return false;
1082 uint32_t adrpDestReg = adrpInstr & 0x1f;
1083 uint32_t ldrDestReg = ldrInstr & 0x1f;
1084 uint32_t ldrSrcReg = (ldrInstr >> 5) & 0x1f;
1085 // Check if ADPR and LDR use the same register.
1086 if (adrpDestReg != ldrDestReg || adrpDestReg != ldrSrcReg)
1087 return false;
1088
1089 Symbol &sym = *adrpRel.sym;
1090 // GOT references to absolute symbols can't be relaxed to use ADRP/ADD in
1091 // position-independent code because these instructions produce a relative
1092 // address.
1093 if (ctx.arg.isPic && !cast<Defined>(Val&: sym).section)
1094 return false;
1095 // Check if the address difference is within 4GB range.
1096 int64_t val =
1097 getAArch64Page(expr: sym.getVA(ctx)) - getAArch64Page(expr: secAddr + adrpRel.offset);
1098 if (val != llvm::SignExtend64(X: val, B: 33))
1099 return false;
1100
1101 return true;
1102}
1103
1104bool AArch64Relaxer::tryRelaxAdrpLdr(const Relocation &adrpRel,
1105 const Relocation &ldrRel, uint64_t secAddr,
1106 uint8_t *buf) const {
1107 // When the definition of sym is not preemptible then we may
1108 // be able to relax
1109 // ADRP xn, :got: sym
1110 // LDR xn, [ xn :got_lo12: sym]
1111 // to
1112 // ADRP xn, sym
1113 // ADD xn, xn, :lo_12: sym
1114
1115 if (!ctx.arg.relax || adrpRel.type != R_AARCH64_ADR_GOT_PAGE ||
1116 ldrRel.type != R_AARCH64_LD64_GOT_LO12_NC)
1117 return false;
1118
1119 Symbol *sym = adrpRel.sym;
1120 if (unsafeToRelaxAdrpLdr.contains(Ptr: sym))
1121 return false;
1122
1123 assert(isLegalAdrpLdrRelaxationCandidate(adrpRel, ldrRel, secAddr, buf) &&
1124 "Should have been marked as unsafe");
1125
1126 uint32_t adrpInstr = read32le(P: buf + adrpRel.offset);
1127 uint32_t adrpDestReg = adrpInstr & 0x1f;
1128 Relocation adrpSymRel = {.expr: RE_AARCH64_PAGE_PC, .type: R_AARCH64_ADR_PREL_PG_HI21,
1129 .offset: adrpRel.offset, /*addend=*/0, .sym: sym};
1130 Relocation addRel = {.expr: R_ABS, .type: R_AARCH64_ADD_ABS_LO12_NC, .offset: ldrRel.offset,
1131 /*addend=*/0, .sym: sym};
1132
1133 // adrp x_<dest_reg>
1134 write32le(P: buf + adrpSymRel.offset, V: 0x90000000 | adrpDestReg);
1135 // add x_<dest reg>, x_<dest reg>
1136 write32le(P: buf + addRel.offset, V: 0x91000000 | adrpDestReg | (adrpDestReg << 5));
1137
1138 ctx.target->relocate(
1139 loc: buf + adrpSymRel.offset, rel: adrpSymRel,
1140 val: SignExtend64(X: getAArch64Page(expr: sym->getVA(ctx)) -
1141 getAArch64Page(expr: secAddr + adrpSymRel.offset),
1142 B: 64));
1143 ctx.target->relocate(loc: buf + addRel.offset, rel: addRel,
1144 val: SignExtend64(X: sym->getVA(ctx), B: 64));
1145 tryRelaxAdrpAdd(adrpRel: adrpSymRel, addRel, secAddr, buf);
1146 return true;
1147}
1148
1149// Tagged symbols have upper address bits that are added by the dynamic loader,
1150// and thus need the full 64-bit GOT entry. Do not relax such symbols.
1151static bool needsGotForMemtag(const Relocation &rel) {
1152 return rel.sym->isTagged() && needsGot(expr: rel.expr);
1153}
1154
1155void AArch64::relocateAlloc(InputSection &sec, uint8_t *buf) const {
1156 uint64_t secAddr = sec.getOutputSection()->addr + sec.outSecOff;
1157 const ArrayRef<Relocation> relocs = sec.relocs();
1158 AArch64Relaxer relaxer(ctx, relocs, secAddr, buf);
1159 for (size_t i = 0, size = relocs.size(); i != size; ++i) {
1160 const Relocation &rel = relocs[i];
1161 if (rel.expr == R_NONE) // See finalizeAddressDependentContent()
1162 continue;
1163 uint8_t *loc = buf + rel.offset;
1164 const uint64_t val = sec.getRelocTargetVA(ctx, r: rel, p: secAddr + rel.offset);
1165
1166 if (needsGotForMemtag(rel)) {
1167 relocate(loc, rel, val);
1168 continue;
1169 }
1170
1171 switch (rel.type) {
1172 case R_AARCH64_ADR_GOT_PAGE:
1173 if (i + 1 < size &&
1174 relaxer.tryRelaxAdrpLdr(adrpRel: rel, ldrRel: relocs[i + 1], secAddr, buf)) {
1175 ++i;
1176 continue;
1177 }
1178 break;
1179 case R_AARCH64_ADR_PREL_PG_HI21:
1180 if (i + 1 < size &&
1181 relaxer.tryRelaxAdrpAdd(adrpRel: rel, addRel: relocs[i + 1], secAddr, buf)) {
1182 ++i;
1183 continue;
1184 }
1185 break;
1186
1187 case R_AARCH64_TLSDESC_ADR_PAGE21:
1188 case R_AARCH64_TLSDESC_LD64_LO12:
1189 case R_AARCH64_TLSDESC_ADD_LO12:
1190 case R_AARCH64_TLSDESC_CALL:
1191 if (rel.expr == R_TPREL)
1192 relaxTlsGdToLe(loc, rel, val);
1193 else if (rel.expr == RE_AARCH64_GOT_PAGE_PC || rel.expr == R_GOT)
1194 relaxTlsGdToIe(loc, rel, val);
1195 else
1196 relocate(loc, rel, val);
1197 continue;
1198 case R_AARCH64_AUTH_TLSDESC_ADR_PAGE21:
1199 case R_AARCH64_AUTH_TLSDESC_LD64_LO12:
1200 case R_AARCH64_AUTH_TLSDESC_ADD_LO12:
1201 case R_AARCH64_AUTH_TLSDESC_CALL:
1202 if (rel.expr == R_TPREL)
1203 relaxAuthTlsDescForNonPreemptibleUndefined(loc, rel);
1204 else
1205 relocate(loc, rel, val);
1206 continue;
1207 case R_AARCH64_TLSIE_ADR_GOTTPREL_PAGE21:
1208 case R_AARCH64_TLSIE_LD64_GOTTPREL_LO12_NC:
1209 if (rel.expr == R_TPREL)
1210 relaxTlsIeToLe(loc, rel, val);
1211 else
1212 relocate(loc, rel, val);
1213 continue;
1214 default:
1215 break;
1216 }
1217
1218 relocate(loc, rel, val);
1219 }
1220}
1221
1222static std::optional<uint64_t> getControlTransferAddend(InputSection &is,
1223 Relocation &r) {
1224 // Identify a control transfer relocation for the branch-to-branch
1225 // optimization. A "control transfer relocation" means a B or BL
1226 // target but it also includes relative vtable relocations for example.
1227 //
1228 // We require the relocation type to be JUMP26, CALL26 or PLT32. With a
1229 // relocation type of PLT32 the value may be assumed to be used for branching
1230 // directly to the symbol and the addend is only used to produce the relocated
1231 // value (hence the effective addend is always 0). This is because if a PLT is
1232 // needed the addend will be added to the address of the PLT, and it doesn't
1233 // make sense to branch into the middle of a PLT. For example, relative vtable
1234 // relocations use PLT32 and 0 or a positive value as the addend but still are
1235 // used to branch to the symbol.
1236 //
1237 // With JUMP26 or CALL26 the only reasonable interpretation of a non-zero
1238 // addend is that we are branching to symbol+addend so that becomes the
1239 // effective addend.
1240 if (r.type == R_AARCH64_PLT32)
1241 return 0;
1242 if (r.type == R_AARCH64_JUMP26 || r.type == R_AARCH64_CALL26)
1243 return r.addend;
1244 return std::nullopt;
1245}
1246
1247static std::pair<Relocation *, uint64_t>
1248getBranchInfoAtTarget(InputSection &is, uint64_t offset) {
1249 auto *i = llvm::partition_point(
1250 Range&: is.relocations, P: [&](Relocation &r) { return r.offset < offset; });
1251 if (i != is.relocations.end() && i->offset == offset &&
1252 i->type == R_AARCH64_JUMP26) {
1253 return {i, i->addend};
1254 }
1255 return {nullptr, 0};
1256}
1257
1258static void redirectControlTransferRelocations(Relocation &r1,
1259 const Relocation &r2) {
1260 r1.expr = r2.expr;
1261 r1.sym = r2.sym;
1262 // With PLT32 we must respect the original addend as that affects the value's
1263 // interpretation. With the other relocation types the original addend is
1264 // irrelevant because it referred to an offset within the original target
1265 // section so we overwrite it.
1266 if (r1.type == R_AARCH64_PLT32)
1267 r1.addend += r2.addend;
1268 else
1269 r1.addend = r2.addend;
1270}
1271
1272void AArch64::applyBranchToBranchOpt() const {
1273 applyBranchToBranchOptImpl(ctx, getControlTransferAddend,
1274 getBranchInfoAtTarget,
1275 redirectControlTransferRelocations);
1276}
1277
1278// AArch64 may use security features in variant PLT sequences. These are:
1279// Pointer Authentication (PAC), introduced in armv8.3-a and Branch Target
1280// Indicator (BTI) introduced in armv8.5-a. The additional instructions used
1281// in the variant Plt sequences are encoded in the Hint space so they can be
1282// deployed on older architectures, which treat the instructions as a nop.
1283// PAC and BTI can be combined leading to the following combinations:
1284// writePltHeader
1285// writePltHeaderBti (no PAC Header needed)
1286// writePlt
1287// writePltBti (BTI only)
1288// writePltPac (PAC only)
1289// writePltBtiPac (BTI and PAC)
1290//
1291// When PAC is enabled the dynamic loader encrypts the address that it places
1292// in the .got.plt using the pacia1716 instruction which encrypts the value in
1293// x17 using the modifier in x16. The static linker places autia1716 before the
1294// indirect branch to x17 to authenticate the address in x17 with the modifier
1295// in x16. This makes it more difficult for an attacker to modify the value in
1296// the .got.plt.
1297//
1298// When BTI is enabled all indirect branches must land on a bti instruction.
1299// The static linker must place a bti instruction at the start of any PLT entry
1300// that may be the target of an indirect branch. As the PLT entries call the
1301// lazy resolver indirectly this must have a bti instruction at start. In
1302// general a bti instruction is not needed for a PLT entry as indirect calls
1303// are resolved to the function address and not the PLT entry for the function.
1304// There are a small number of cases where the PLT address can escape, such as
1305// taking the address of a function or ifunc via a non got-generating
1306// relocation, and a shared library refers to that symbol.
1307//
1308// We use the bti c variant of the instruction which permits indirect branches
1309// (br) via x16/x17 and indirect function calls (blr) via any register. The ABI
1310// guarantees that all indirect branches from code requiring BTI protection
1311// will go via x16/x17
1312
1313namespace {
1314class AArch64BtiPac final : public AArch64 {
1315public:
1316 AArch64BtiPac(Ctx &);
1317 void writePltHeader(uint8_t *buf) const override;
1318 void writePlt(uint8_t *buf, const Symbol &sym,
1319 uint64_t pltEntryAddr) const override;
1320
1321private:
1322 bool btiHeader; // bti instruction needed in PLT Header and Entry
1323 enum {
1324 PEK_NoAuth,
1325 PEK_AuthHint, // use autia1716 instr for authenticated branch in PLT entry
1326 PEK_Auth, // use braa instr for authenticated branch in PLT entry
1327 } pacEntryKind;
1328};
1329} // namespace
1330
1331AArch64BtiPac::AArch64BtiPac(Ctx &ctx) : AArch64(ctx) {
1332 btiHeader = (ctx.arg.andFeatures & GNU_PROPERTY_AARCH64_FEATURE_1_BTI);
1333 // A BTI (Branch Target Indicator) Plt Entry is only required if the
1334 // address of the PLT entry can be taken by the program, which permits an
1335 // indirect jump to the PLT entry. This can happen when the address
1336 // of the PLT entry for a function is canonicalised due to the address of
1337 // the function in an executable being taken by a shared library, or
1338 // non-preemptible ifunc referenced by non-GOT-generating, non-PLT-generating
1339 // relocations.
1340 // The PAC PLT entries require dynamic loader support and this isn't known
1341 // from properties in the objects, so we use the command line flag.
1342 // By default we only use hint-space instructions, but if we detect the
1343 // PAuthABI, which requires v8.3-A, we can use the non-hint space
1344 // instructions.
1345
1346 if (ctx.arg.zPacPlt) {
1347 if (ctx.aarch64PauthAbiCoreInfo && ctx.aarch64PauthAbiCoreInfo->isValid())
1348 pacEntryKind = PEK_Auth;
1349 else
1350 pacEntryKind = PEK_AuthHint;
1351 } else {
1352 pacEntryKind = PEK_NoAuth;
1353 }
1354
1355 if (btiHeader || (pacEntryKind != PEK_NoAuth)) {
1356 pltEntrySize = 24;
1357 ipltEntrySize = 24;
1358 }
1359}
1360
1361void AArch64BtiPac::writePltHeader(uint8_t *buf) const {
1362 const uint8_t btiData[] = { 0x5f, 0x24, 0x03, 0xd5 }; // bti c
1363 const uint8_t pltData[] = {
1364 0xf0, 0x7b, 0xbf, 0xa9, // stp x16, x30, [sp,#-16]!
1365 0x10, 0x00, 0x00, 0x90, // adrp x16, Page(&(.got.plt[2]))
1366 0x11, 0x02, 0x40, 0xf9, // ldr x17, [x16, Offset(&(.got.plt[2]))]
1367 0x10, 0x02, 0x00, 0x91, // add x16, x16, Offset(&(.got.plt[2]))
1368 0x20, 0x02, 0x1f, 0xd6, // br x17
1369 0x1f, 0x20, 0x03, 0xd5, // nop
1370 0x1f, 0x20, 0x03, 0xd5 // nop
1371 };
1372 const uint8_t nopData[] = { 0x1f, 0x20, 0x03, 0xd5 }; // nop
1373
1374 uint64_t got = ctx.in.gotPlt->getVA();
1375 uint64_t plt = ctx.in.plt->getVA();
1376
1377 if (btiHeader) {
1378 // PltHeader is called indirectly by plt[N]. Prefix pltData with a BTI C
1379 // instruction.
1380 memcpy(dest: buf, src: btiData, n: sizeof(btiData));
1381 buf += sizeof(btiData);
1382 plt += sizeof(btiData);
1383 }
1384 memcpy(dest: buf, src: pltData, n: sizeof(pltData));
1385
1386 relocateNoSym(loc: buf + 4, type: R_AARCH64_ADR_PREL_PG_HI21,
1387 val: getAArch64Page(expr: got + 16) - getAArch64Page(expr: plt + 4));
1388 relocateNoSym(loc: buf + 8, type: R_AARCH64_LDST64_ABS_LO12_NC, val: got + 16);
1389 relocateNoSym(loc: buf + 12, type: R_AARCH64_ADD_ABS_LO12_NC, val: got + 16);
1390 if (!btiHeader)
1391 // We didn't add the BTI c instruction so round out size with NOP.
1392 memcpy(dest: buf + sizeof(pltData), src: nopData, n: sizeof(nopData));
1393}
1394
1395void AArch64BtiPac::writePlt(uint8_t *buf, const Symbol &sym,
1396 uint64_t pltEntryAddr) const {
1397 // The PLT entry is of the form:
1398 // [btiData] addrInst (pacBr | stdBr) [nopData]
1399 const uint8_t btiData[] = { 0x5f, 0x24, 0x03, 0xd5 }; // bti c
1400 const uint8_t addrInst[] = {
1401 0x10, 0x00, 0x00, 0x90, // adrp x16, Page(&(.got.plt[n]))
1402 0x11, 0x02, 0x40, 0xf9, // ldr x17, [x16, Offset(&(.got.plt[n]))]
1403 0x10, 0x02, 0x00, 0x91 // add x16, x16, Offset(&(.got.plt[n]))
1404 };
1405 const uint8_t pacHintBr[] = {
1406 0x9f, 0x21, 0x03, 0xd5, // autia1716
1407 0x20, 0x02, 0x1f, 0xd6 // br x17
1408 };
1409 const uint8_t pacBr[] = {
1410 0x30, 0x0a, 0x1f, 0xd7, // braa x17, x16
1411 0x1f, 0x20, 0x03, 0xd5 // nop
1412 };
1413 const uint8_t stdBr[] = {
1414 0x20, 0x02, 0x1f, 0xd6, // br x17
1415 0x1f, 0x20, 0x03, 0xd5 // nop
1416 };
1417 const uint8_t nopData[] = { 0x1f, 0x20, 0x03, 0xd5 }; // nop
1418
1419 // NEEDS_COPY indicates a non-ifunc canonical PLT entry whose address may
1420 // escape to shared objects. isInIplt indicates a non-preemptible ifunc. Its
1421 // address may escape if referenced by a direct relocation. If relative
1422 // vtables are used then if the vtable is in a shared object the offsets will
1423 // be to the PLT entry. The condition is conservative.
1424 bool hasBti = btiHeader &&
1425 (sym.hasFlag(bit: NEEDS_COPY) || sym.isInIplt || sym.thunkAccessed);
1426 if (hasBti) {
1427 memcpy(dest: buf, src: btiData, n: sizeof(btiData));
1428 buf += sizeof(btiData);
1429 pltEntryAddr += sizeof(btiData);
1430 }
1431
1432 uint64_t gotPltEntryAddr = sym.getGotPltVA(ctx);
1433 memcpy(dest: buf, src: addrInst, n: sizeof(addrInst));
1434 relocateNoSym(loc: buf, type: R_AARCH64_ADR_PREL_PG_HI21,
1435 val: getAArch64Page(expr: gotPltEntryAddr) - getAArch64Page(expr: pltEntryAddr));
1436 relocateNoSym(loc: buf + 4, type: R_AARCH64_LDST64_ABS_LO12_NC, val: gotPltEntryAddr);
1437 relocateNoSym(loc: buf + 8, type: R_AARCH64_ADD_ABS_LO12_NC, val: gotPltEntryAddr);
1438
1439 if (pacEntryKind != PEK_NoAuth)
1440 memcpy(dest: buf + sizeof(addrInst),
1441 src: pacEntryKind == PEK_AuthHint ? pacHintBr : pacBr,
1442 n: sizeof(pacEntryKind == PEK_AuthHint ? pacHintBr : pacBr));
1443 else
1444 memcpy(dest: buf + sizeof(addrInst), src: stdBr, n: sizeof(stdBr));
1445 if (!hasBti)
1446 // We didn't add the BTI c instruction so round out size with NOP.
1447 memcpy(dest: buf + sizeof(addrInst) + sizeof(stdBr), src: nopData, n: sizeof(nopData));
1448}
1449
1450template <class ELFT>
1451static void
1452addTaggedSymbolReferences(Ctx &ctx, InputSectionBase &sec,
1453 DenseMap<Symbol *, unsigned> &referenceCount) {
1454 assert(sec.type == SHT_AARCH64_MEMTAG_GLOBALS_STATIC);
1455
1456 const RelsOrRelas<ELFT> rels = sec.relsOrRelas<ELFT>();
1457 if (rels.areRelocsRel())
1458 ErrAlways(ctx)
1459 << "non-RELA relocations are not allowed with memtag globals";
1460
1461 for (const typename ELFT::Rela &rel : rels.relas) {
1462 Symbol &sym = sec.file->getRelocTargetSym(rel);
1463 // Linker-synthesized symbols such as __executable_start may be referenced
1464 // as tagged in input objfiles, and we don't want them to be tagged. A
1465 // cheap way to exclude them is the type check, but their type is
1466 // STT_NOTYPE. In addition, this save us from checking untaggable symbols,
1467 // like functions or TLS symbols.
1468 if (sym.type != STT_OBJECT)
1469 continue;
1470 // STB_LOCAL symbols can't be referenced from outside the object file, and
1471 // thus don't need to be checked for references from other object files.
1472 if (sym.binding == STB_LOCAL) {
1473 sym.setIsTagged(true);
1474 continue;
1475 }
1476 ++referenceCount[&sym];
1477 }
1478 sec.markDead();
1479}
1480
1481// A tagged symbol must be denoted as being tagged by all references and the
1482// chosen definition. For simplicity, here, it must also be denoted as tagged
1483// for all definitions. Otherwise:
1484//
1485// 1. A tagged definition can be used by an untagged declaration, in which case
1486// the untagged access may be PC-relative, causing a tag mismatch at
1487// runtime.
1488// 2. An untagged definition can be used by a tagged declaration, where the
1489// compiler has taken advantage of the increased alignment of the tagged
1490// declaration, but the alignment at runtime is wrong, causing a fault.
1491//
1492// Ideally, this isn't a problem, as any TU that imports or exports tagged
1493// symbols should also be built with tagging. But, to handle these cases, we
1494// demote the symbol to be untagged.
1495void elf::createTaggedSymbols(Ctx &ctx) {
1496 assert(hasMemtag(ctx));
1497
1498 // First, collect all symbols that are marked as tagged, and count how many
1499 // times they're marked as tagged.
1500 DenseMap<Symbol *, unsigned> taggedSymbolReferenceCount;
1501 for (InputFile *file : ctx.objectFiles) {
1502 if (file->kind() != InputFile::ObjKind)
1503 continue;
1504 for (InputSectionBase *section : file->getSections()) {
1505 if (!section || section->type != SHT_AARCH64_MEMTAG_GLOBALS_STATIC ||
1506 section == &InputSection::discarded)
1507 continue;
1508 invokeELFT(addTaggedSymbolReferences, ctx, *section,
1509 taggedSymbolReferenceCount);
1510 }
1511 }
1512
1513 // Now, go through all the symbols. If the number of declarations +
1514 // definitions to a symbol exceeds the amount of times they're marked as
1515 // tagged, it means we have an objfile that uses the untagged variant of the
1516 // symbol.
1517 for (InputFile *file : ctx.objectFiles) {
1518 if (file->kind() != InputFile::BinaryKind &&
1519 file->kind() != InputFile::ObjKind)
1520 continue;
1521
1522 for (Symbol *symbol : file->getSymbols()) {
1523 // See `addTaggedSymbolReferences` for more details.
1524 if (symbol->type != STT_OBJECT ||
1525 symbol->binding == STB_LOCAL)
1526 continue;
1527 auto it = taggedSymbolReferenceCount.find(Val: symbol);
1528 if (it == taggedSymbolReferenceCount.end()) continue;
1529 unsigned &remainingAllowedTaggedRefs = it->second;
1530 if (remainingAllowedTaggedRefs == 0) {
1531 taggedSymbolReferenceCount.erase(I: it);
1532 continue;
1533 }
1534 --remainingAllowedTaggedRefs;
1535 }
1536 }
1537
1538 // `addTaggedSymbolReferences` has already checked that we have RELA
1539 // relocations, the only other way to get written addends is with
1540 // --apply-dynamic-relocs.
1541 if (!taggedSymbolReferenceCount.empty() && ctx.arg.writeAddends)
1542 ErrAlways(ctx) << "--apply-dynamic-relocs cannot be used with MTE globals";
1543
1544 // Now, `taggedSymbolReferenceCount` should only contain symbols that are
1545 // defined as tagged exactly the same amount as it's referenced, meaning all
1546 // uses are tagged.
1547 for (auto &[symbol, remainingTaggedRefs] : taggedSymbolReferenceCount) {
1548 assert(remainingTaggedRefs == 0 &&
1549 "Symbol is defined as tagged more times than it's used");
1550 symbol->setIsTagged(true);
1551 }
1552}
1553
1554void elf::setAArch64TargetInfo(Ctx &ctx) {
1555 if ((ctx.arg.andFeatures & GNU_PROPERTY_AARCH64_FEATURE_1_BTI) ||
1556 ctx.arg.zPacPlt)
1557 ctx.target.reset(p: new AArch64BtiPac(ctx));
1558 else
1559 ctx.target.reset(p: new AArch64(ctx));
1560}
1561