1//===- SyntheticSections.cpp ----------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains linker-synthesized sections. Currently,
10// synthetic sections are created either output sections or input sections,
11// but we are rewriting code so that all synthetic sections are created as
12// input sections.
13//
14//===----------------------------------------------------------------------===//
15
16#include "SyntheticSections.h"
17#include "Config.h"
18#include "DWARF.h"
19#include "EhFrame.h"
20#include "InputFiles.h"
21#include "LinkerScript.h"
22#include "OutputSections.h"
23#include "SymbolTable.h"
24#include "Symbols.h"
25#include "Target.h"
26#include "Thunks.h"
27#include "Writer.h"
28#include "lld/Common/Version.h"
29#include "llvm/ADT/STLExtras.h"
30#include "llvm/ADT/Sequence.h"
31#include "llvm/ADT/SetOperations.h"
32#include "llvm/ADT/StringExtras.h"
33#include "llvm/BinaryFormat/Dwarf.h"
34#include "llvm/BinaryFormat/ELF.h"
35#include "llvm/DebugInfo/DWARF/DWARFAcceleratorTable.h"
36#include "llvm/DebugInfo/DWARF/DWARFDebugPubTable.h"
37#include "llvm/Support/DJB.h"
38#include "llvm/Support/Endian.h"
39#include "llvm/Support/LEB128.h"
40#include "llvm/Support/Parallel.h"
41#include "llvm/Support/TimeProfiler.h"
42#include <cinttypes>
43#include <cstdlib>
44
45using namespace llvm;
46using namespace llvm::dwarf;
47using namespace llvm::ELF;
48using namespace llvm::object;
49using namespace llvm::support;
50using namespace lld;
51using namespace lld::elf;
52
53using llvm::support::endian::read32le;
54using llvm::support::endian::write32le;
55using llvm::support::endian::write64le;
56
57static uint64_t readUint(Ctx &ctx, uint8_t *buf) {
58 return ctx.arg.is64 ? read64(ctx, p: buf) : read32(ctx, p: buf);
59}
60
61static void writeUint(Ctx &ctx, uint8_t *buf, uint64_t val) {
62 if (ctx.arg.is64)
63 write64(ctx, p: buf, v: val);
64 else
65 write32(ctx, p: buf, v: val);
66}
67
68// Returns an LLD version string.
69static ArrayRef<uint8_t> getVersion(Ctx &ctx) {
70 // Check LLD_VERSION first for ease of testing.
71 // You can get consistent output by using the environment variable.
72 // This is only for testing.
73 StringRef s = getenv(name: "LLD_VERSION");
74 if (s.empty())
75 s = ctx.saver.save(S: Twine("Linker: ") + getLLDVersion());
76
77 // +1 to include the terminating '\0'.
78 return {(const uint8_t *)s.data(), s.size() + 1};
79}
80
81// Creates a .comment section containing LLD version info.
82// With this feature, you can identify LLD-generated binaries easily
83// by "readelf --string-dump .comment <file>".
84// The returned object is a mergeable string section.
85MergeInputSection *elf::createCommentSection(Ctx &ctx) {
86 auto *sec =
87 make<MergeInputSection>(args&: ctx, args: ".comment", args: SHT_PROGBITS,
88 args: SHF_MERGE | SHF_STRINGS, args: 1, args: getVersion(ctx));
89 sec->splitIntoPieces();
90 return sec;
91}
92
93InputSection *elf::createInterpSection(Ctx &ctx) {
94 // StringSaver guarantees that the returned string ends with '\0'.
95 StringRef s = ctx.saver.save(S: ctx.arg.dynamicLinker);
96 ArrayRef<uint8_t> contents = {(const uint8_t *)s.data(), s.size() + 1};
97
98 return make<InputSection>(args&: ctx.internalFile, args: ".interp", args: SHT_PROGBITS,
99 args: SHF_ALLOC,
100 /*addralign=*/args: 1, /*entsize=*/args: 0, args&: contents);
101}
102
103Defined *elf::addSyntheticLocal(Ctx &ctx, StringRef name, uint8_t type,
104 uint64_t value, uint64_t size,
105 SectionBase &section) {
106 Defined *s = makeDefined(args&: ctx, args&: section.file, args&: name, args: STB_LOCAL, args: STV_DEFAULT,
107 args&: type, args&: value, args&: size, args: &section);
108 if (ctx.in.symTab)
109 ctx.in.symTab->addSymbol(sym: s);
110
111 if (ctx.arg.emachine == EM_ARM && !ctx.arg.isLE && ctx.arg.armBe8 &&
112 (section.flags & SHF_EXECINSTR))
113 // Adding Linker generated mapping symbols to the arm specific mapping
114 // symbols list.
115 addArmSyntheticSectionMappingSymbol(s);
116
117 return s;
118}
119
120static size_t getHashSize(Ctx &ctx) {
121 switch (ctx.arg.buildId) {
122 case BuildIdKind::Fast:
123 return 8;
124 case BuildIdKind::Md5:
125 case BuildIdKind::Uuid:
126 return 16;
127 case BuildIdKind::Sha1:
128 return 20;
129 case BuildIdKind::Hexstring:
130 return ctx.arg.buildIdVector.size();
131 default:
132 llvm_unreachable("unknown BuildIdKind");
133 }
134}
135
136// This class represents a linker-synthesized .note.gnu.property section.
137//
138// In x86 and AArch64, object files may contain feature flags indicating the
139// features that they have used. The flags are stored in a .note.gnu.property
140// section.
141//
142// lld reads the sections from input files and merges them by computing AND of
143// the flags. The result is written as a new .note.gnu.property section.
144//
145// If the flag is zero (which indicates that the intersection of the feature
146// sets is empty, or some input files didn't have .note.gnu.property sections),
147// we don't create this section.
148GnuPropertySection::GnuPropertySection(Ctx &ctx)
149 : SyntheticSection(ctx, ".note.gnu.property", SHT_NOTE, SHF_ALLOC,
150 ctx.arg.wordsize) {}
151
152void GnuPropertySection::writeTo(uint8_t *buf) {
153 uint32_t featureAndType;
154 switch (ctx.arg.emachine) {
155 case EM_386:
156 case EM_X86_64:
157 featureAndType = GNU_PROPERTY_X86_FEATURE_1_AND;
158 break;
159 case EM_AARCH64:
160 featureAndType = GNU_PROPERTY_AARCH64_FEATURE_1_AND;
161 break;
162 case EM_RISCV:
163 featureAndType = GNU_PROPERTY_RISCV_FEATURE_1_AND;
164 break;
165 default:
166 llvm_unreachable(
167 "target machine does not support .note.gnu.property section");
168 }
169
170 write32(ctx, p: buf, v: 4); // Name size
171 write32(ctx, p: buf + 4, v: getSize() - 16); // Content size
172 write32(ctx, p: buf + 8, v: NT_GNU_PROPERTY_TYPE_0); // Type
173 memcpy(dest: buf + 12, src: "GNU", n: 4); // Name string
174
175 unsigned offset = 16;
176 if (ctx.arg.andFeatures != 0) {
177 write32(ctx, p: buf + offset + 0, v: featureAndType); // Feature type
178 write32(ctx, p: buf + offset + 4, v: 4); // Feature size
179 write32(ctx, p: buf + offset + 8, v: ctx.arg.andFeatures); // Feature flags
180 if (ctx.arg.is64)
181 write32(ctx, p: buf + offset + 12, v: 0); // Padding
182 offset += 16;
183 }
184
185 if (ctx.aarch64PauthAbiCoreInfo) {
186 write32(ctx, p: buf + offset + 0, v: GNU_PROPERTY_AARCH64_FEATURE_PAUTH);
187 write32(ctx, p: buf + offset + 4, v: AArch64PauthAbiCoreInfo::size());
188 write64(ctx, p: buf + offset + 8, v: ctx.aarch64PauthAbiCoreInfo->platform);
189 write64(ctx, p: buf + offset + 16, v: ctx.aarch64PauthAbiCoreInfo->version);
190 }
191}
192
193size_t GnuPropertySection::getSize() const {
194 uint32_t contentSize = 0;
195 if (ctx.arg.andFeatures != 0)
196 contentSize += ctx.arg.is64 ? 16 : 12;
197 if (ctx.aarch64PauthAbiCoreInfo)
198 contentSize += 4 + 4 + AArch64PauthAbiCoreInfo::size();
199 assert(contentSize != 0);
200 return contentSize + 16;
201}
202
203BuildIdSection::BuildIdSection(Ctx &ctx)
204 : SyntheticSection(ctx, ".note.gnu.build-id", SHT_NOTE, SHF_ALLOC, 4),
205 hashSize(getHashSize(ctx)) {}
206
207void BuildIdSection::writeTo(uint8_t *buf) {
208 write32(ctx, p: buf, v: 4); // Name size
209 write32(ctx, p: buf + 4, v: hashSize); // Content size
210 write32(ctx, p: buf + 8, v: NT_GNU_BUILD_ID); // Type
211 memcpy(dest: buf + 12, src: "GNU", n: 4); // Name string
212 hashBuf = buf + 16;
213}
214
215void BuildIdSection::writeBuildId(ArrayRef<uint8_t> buf) {
216 assert(buf.size() == hashSize);
217 memcpy(dest: hashBuf, src: buf.data(), n: hashSize);
218}
219
220BssSection::BssSection(Ctx &ctx, StringRef name, uint64_t size,
221 uint32_t alignment)
222 : SyntheticSection(ctx, name, SHT_NOBITS, SHF_ALLOC | SHF_WRITE,
223 alignment) {
224 this->bss = true;
225 this->size = size;
226}
227
228EhFrameSection::EhFrameSection(Ctx &ctx)
229 : SyntheticSection(ctx, ".eh_frame", SHT_PROGBITS, SHF_ALLOC, 1) {}
230
231// Search for an existing CIE record or create a new one.
232// CIE records from input object files are uniquified by their contents
233// and where their relocations point to.
234CieRecord *EhFrameSection::addCie(EhSectionPiece &cie,
235 ArrayRef<Relocation> rels) {
236 Symbol *personality = nullptr;
237 unsigned firstRelI = cie.firstRelocation;
238 if (firstRelI != (unsigned)-1)
239 personality = rels[firstRelI].sym;
240
241 // Search for an existing CIE by CIE contents/relocation target pair.
242 CieRecord *&rec = cieMap[{cie.data(), personality}];
243
244 // If not found, create a new one.
245 if (!rec) {
246 rec = make<CieRecord>();
247 rec->cie = &cie;
248 cieRecords.push_back(Elt: rec);
249 }
250 return rec;
251}
252
253// There is one FDE per function. Returns a non-null pointer to the function
254// symbol if the given FDE points to a live function.
255Defined *EhFrameSection::isFdeLive(EhSectionPiece &fde,
256 ArrayRef<Relocation> rels) {
257 // An FDE should point to some function because FDEs are to describe
258 // functions. That's however not always the case due to an issue of
259 // ld.gold with -r. ld.gold may discard only functions and leave their
260 // corresponding FDEs, which results in creating bad .eh_frame sections.
261 // To deal with that, we ignore such FDEs.
262 unsigned firstRelI = fde.firstRelocation;
263 if (firstRelI == (unsigned)-1)
264 return nullptr;
265
266 // FDEs for garbage-collected or merged-by-ICF sections are dead.
267 if (auto *d = dyn_cast<Defined>(Val: rels[firstRelI].sym))
268 if (!d->folded && d->section && d->section->partition == partition)
269 return d;
270 return nullptr;
271}
272
273// .eh_frame is a sequence of CIE or FDE records. In general, there
274// is one CIE record per input object file which is followed by
275// a list of FDEs. This function searches an existing CIE or create a new
276// one and associates FDEs to the CIE.
277template <endianness e> void EhFrameSection::addRecords(EhInputSection *sec) {
278 auto rels = sec->rels;
279 offsetToCie.clear();
280 for (EhSectionPiece &cie : sec->cies)
281 offsetToCie[cie.inputOff] = addCie(cie, rels);
282 for (EhSectionPiece &fde : sec->fdes) {
283 uint32_t id = endian::read32<e>(fde.data().data() + 4);
284 CieRecord *rec = offsetToCie[fde.inputOff + 4 - id];
285 if (!rec)
286 Fatal(ctx) << sec << ": invalid CIE reference";
287
288 if (!isFdeLive(fde, rels))
289 continue;
290 rec->fdes.push_back(Elt: &fde);
291 numFdes++;
292 }
293}
294
295// Used by ICF<ELFT>::handleLSDA(). This function is very similar to
296// EhFrameSection::addRecords().
297template <class ELFT>
298void EhFrameSection::iterateFDEWithLSDAAux(
299 EhInputSection &sec, DenseSet<size_t> &ciesWithLSDA,
300 llvm::function_ref<void(InputSection &)> fn) {
301 for (EhSectionPiece &cie : sec.cies)
302 if (hasLSDA(p: cie))
303 ciesWithLSDA.insert(V: cie.inputOff);
304 for (EhSectionPiece &fde : sec.fdes) {
305 uint32_t id = endian::read32<ELFT::Endianness>(fde.data().data() + 4);
306 if (!ciesWithLSDA.contains(V: fde.inputOff + 4 - id))
307 continue;
308
309 // The CIE has a LSDA argument. Call fn with d's section.
310 if (Defined *d = isFdeLive(fde, rels: sec.rels))
311 if (auto *s = dyn_cast_or_null<InputSection>(Val: d->section))
312 fn(*s);
313 }
314}
315
316template <class ELFT>
317void EhFrameSection::iterateFDEWithLSDA(
318 llvm::function_ref<void(InputSection &)> fn) {
319 DenseSet<size_t> ciesWithLSDA;
320 for (EhInputSection *sec : sections) {
321 ciesWithLSDA.clear();
322 iterateFDEWithLSDAAux<ELFT>(*sec, ciesWithLSDA, fn);
323 }
324}
325
326static void writeCieFde(Ctx &ctx, uint8_t *buf, ArrayRef<uint8_t> d) {
327 memcpy(dest: buf, src: d.data(), n: d.size());
328 // Fix the size field. -4 since size does not include the size field itself.
329 write32(ctx, p: buf, v: d.size() - 4);
330}
331
332void EhFrameSection::finalizeContents() {
333 assert(!this->size); // Not finalized.
334
335 switch (ctx.arg.ekind) {
336 case ELFNoneKind:
337 llvm_unreachable("invalid ekind");
338 case ELF32LEKind:
339 case ELF64LEKind:
340 for (EhInputSection *sec : sections)
341 if (sec->isLive())
342 addRecords<endianness::little>(sec);
343 break;
344 case ELF32BEKind:
345 case ELF64BEKind:
346 for (EhInputSection *sec : sections)
347 if (sec->isLive())
348 addRecords<endianness::big>(sec);
349 break;
350 }
351
352 size_t off = 0;
353 for (CieRecord *rec : cieRecords) {
354 rec->cie->outputOff = off;
355 off += rec->cie->size;
356
357 for (EhSectionPiece *fde : rec->fdes) {
358 fde->outputOff = off;
359 off += fde->size;
360 }
361 }
362
363 // The LSB standard does not allow a .eh_frame section with zero
364 // Call Frame Information records. glibc unwind-dw2-fde.c
365 // classify_object_over_fdes expects there is a CIE record length 0 as a
366 // terminator. Thus we add one unconditionally.
367 off += 4;
368
369 this->size = off;
370}
371
372void EhFrameSection::writeTo(uint8_t *buf) {
373 // Write CIE and FDE records.
374 for (CieRecord *rec : cieRecords) {
375 size_t cieOffset = rec->cie->outputOff;
376 writeCieFde(ctx, buf: buf + cieOffset, d: rec->cie->data());
377
378 for (EhSectionPiece *fde : rec->fdes) {
379 size_t off = fde->outputOff;
380 writeCieFde(ctx, buf: buf + off, d: fde->data());
381
382 // FDE's second word should have the offset to an associated CIE.
383 // Write it.
384 write32(ctx, p: buf + off + 4, v: off + 4 - cieOffset);
385 }
386 }
387
388 // Apply relocations to .eh_frame entries. This includes CIE personality
389 // pointers, FDE initial_location fields, and LSDA pointers.
390 for (EhInputSection *s : sections)
391 ctx.target->relocateEh(sec&: *s, buf);
392
393 EhFrameHeader *hdr = ctx.in.ehFrameHdr.get();
394 if (!hdr || !hdr->getParent())
395 return;
396
397 // Write the .eh_frame_hdr section using cached FDE data from updateAllocSize.
398 bool large = hdr->large;
399 int64_t ehFramePtr = getParent()->addr - hdr->getVA() - 4;
400 auto writeField = [&](uint8_t *buf, uint64_t val) {
401 large ? write64(ctx, p: buf, v: val) : write32(ctx, p: buf, v: val);
402 };
403
404 uint8_t *hdrBuf = ctx.bufferStart + hdr->getParent()->offset + hdr->outSecOff;
405 // version
406 hdrBuf[0] = 1;
407 // eh_frame_ptr_enc
408 hdrBuf[1] = DW_EH_PE_pcrel | (large ? DW_EH_PE_sdata8 : DW_EH_PE_sdata4);
409 // fde_count_enc
410 hdrBuf[2] = DW_EH_PE_udata4;
411 // table_enc
412 hdrBuf[3] = DW_EH_PE_datarel | (large ? DW_EH_PE_sdata8 : DW_EH_PE_sdata4);
413 hdrBuf += 4;
414 writeField(hdrBuf, ehFramePtr);
415 hdrBuf += large ? 8 : 4;
416 write32(ctx, p: hdrBuf, v: hdr->fdes.size());
417 hdrBuf += 4;
418 for (const FdeData &fde : hdr->fdes) {
419 writeField(hdrBuf, fde.pcRel);
420 writeField(hdrBuf + (large ? 8 : 4), fde.fdeVARel);
421 hdrBuf += large ? 16 : 8;
422 }
423}
424
425EhFrameHeader::EhFrameHeader(Ctx &ctx)
426 : SyntheticSection(ctx, ".eh_frame_hdr", SHT_PROGBITS, SHF_ALLOC, 4) {}
427
428void EhFrameHeader::writeTo(uint8_t *buf) {
429 // The section content is written during EhFrameSection::writeTo.
430}
431
432bool EhFrameHeader::isNeeded() const {
433 return isLive() && ctx.in.ehFrame->isNeeded();
434}
435
436void EhFrameHeader::finalizeContents() {
437 // Compute size: 4-byte header + eh_frame_ptr + fde_count + FDE table.
438 // Initially `large` is false; updateAllocSize may set it to true if addresses
439 // exceed the 32-bit range, then call finalizeContents again.
440 auto numFdes = ctx.in.ehFrame->numFdes;
441 size = 4 + (large ? 8 : 4) + 4 + numFdes * (large ? 16 : 8);
442}
443
444bool EhFrameHeader::updateAllocSize(Ctx &ctx) {
445 // This is called after `finalizeSynthetic`, so in the typical case without
446 // .relr.dyn, this function will not change the size and assignAddresses
447 // will not need another iteration.
448 EhFrameSection *ehFrame = ctx.in.ehFrame.get();
449 uint64_t hdrVA = getVA();
450 int64_t ehFramePtr = ehFrame->getParent()->addr - hdrVA - 4;
451 // Determine if 64-bit encodings are needed.
452 bool newLarge = !isInt<32>(x: ehFramePtr);
453
454 // Collect FDE entries. For each FDE, compute pcRel and fdeVARel relative to
455 // .eh_frame_hdr's VA.
456 fdes.clear();
457 for (CieRecord *rec : ehFrame->getCieRecords()) {
458 uint8_t enc = getFdeEncoding(p: rec->cie);
459 if ((enc & 0x70) != DW_EH_PE_absptr && (enc & 0x70) != DW_EH_PE_pcrel) {
460 Err(ctx) << "unknown FDE size encoding";
461 continue;
462 }
463 for (EhSectionPiece *fde : rec->fdes) {
464 // The FDE has passed `isFdeLive`, so the first relocation's symbol is a
465 // live Defined.
466 auto *isec = cast<EhInputSection>(Val: fde->sec);
467 auto &reloc = isec->rels[fde->firstRelocation];
468 assert(isa<Defined>(reloc.sym) && "isFdeLive should have checked this");
469 int64_t pcRel = reloc.sym->getVA(ctx) + reloc.addend - hdrVA;
470 int64_t fdeVARel = ehFrame->getParent()->addr + fde->outputOff - hdrVA;
471 fdes.push_back(Elt: {.pcRel: pcRel, .fdeVARel: fdeVARel});
472 newLarge |= !isInt<32>(x: pcRel) || !isInt<32>(x: fdeVARel);
473 }
474 }
475
476 // Sort the FDE list by their PC and uniquify. Usually there is only one FDE
477 // at an address, but there can be more than one FDEs pointing to the address.
478 llvm::stable_sort(
479 Range&: fdes, C: [](const EhFrameSection::FdeData &a,
480 const EhFrameSection::FdeData &b) { return a.pcRel < b.pcRel; });
481 fdes.erase(CS: llvm::unique(R&: fdes,
482 P: [](const EhFrameSection::FdeData &a,
483 const EhFrameSection::FdeData &b) {
484 return a.pcRel == b.pcRel;
485 }),
486 CE: fdes.end());
487 ehFrame->numFdes = fdes.size();
488
489 large = newLarge;
490
491 // Compute size.
492 size_t oldSize = size;
493 finalizeContents();
494
495 // Don't allow the section to shrink; otherwise the size of the section can
496 // oscillate infinitely.
497 if (size < oldSize)
498 size = oldSize;
499
500 return size != oldSize;
501}
502
503GotSection::GotSection(Ctx &ctx)
504 : SyntheticSection(ctx, ".got", SHT_PROGBITS, SHF_ALLOC | SHF_WRITE,
505 ctx.target->gotEntrySize) {
506 numEntries = ctx.target->gotHeaderEntriesNum;
507}
508
509void GotSection::addEntry(const Symbol &sym) {
510 assert(sym.auxIdx == ctx.symAux.size() - 1);
511 ctx.symAux.back().gotIdx = numEntries++;
512}
513
514void GotSection::addAuthEntry(const Symbol &sym) {
515 authEntries.push_back(
516 Elt: {/*offset=*/(numEntries - 1) * ctx.target->gotEntrySize,
517 /*isSymbolFunc=*/sym.isFunc(),
518 /*isUndefinedNonPreemptible=*/sym.isUndefined() && !sym.isPreemptible});
519}
520
521bool GotSection::addTlsDescEntry(const Symbol &sym) {
522 assert(sym.auxIdx == ctx.symAux.size() - 1);
523 ctx.symAux.back().tlsDescIdx = numEntries;
524 numEntries += 2;
525 return true;
526}
527
528void GotSection::addTlsDescAuthEntry(const Symbol &sym) {
529 authEntries.push_back(Elt: {/*offset=*/(numEntries - 2) * ctx.target->gotEntrySize,
530 /*isSymbolFunc=*/true,
531 /*isUndefinedNonPreemptible=*/false});
532 assert(!sym.isFunc());
533 addAuthEntry(sym);
534}
535
536bool GotSection::addDynTlsEntry(const Symbol &sym) {
537 assert(sym.auxIdx == ctx.symAux.size() - 1);
538 ctx.symAux.back().tlsGdIdx = numEntries;
539 // Global Dynamic TLS entries take two GOT slots.
540 numEntries += 2;
541 return true;
542}
543
544// Reserves TLS entries for a TLS module ID and a TLS block offset.
545// In total it takes two GOT slots.
546bool GotSection::addTlsIndex() {
547 if (tlsIndexOff != uint32_t(-1))
548 return false;
549 tlsIndexOff = numEntries * ctx.target->gotEntrySize;
550 numEntries += 2;
551 return true;
552}
553
554uint32_t GotSection::getTlsDescOffset(const Symbol &sym) const {
555 return sym.getTlsDescIdx(ctx) * ctx.target->gotEntrySize;
556}
557
558uint64_t GotSection::getTlsDescAddr(const Symbol &sym) const {
559 return getVA() + getTlsDescOffset(sym);
560}
561
562uint64_t GotSection::getGlobalDynAddr(const Symbol &b) const {
563 return this->getVA() + b.getTlsGdIdx(ctx) * ctx.target->gotEntrySize;
564}
565
566uint64_t GotSection::getGlobalDynOffset(const Symbol &b) const {
567 return b.getTlsGdIdx(ctx) * ctx.target->gotEntrySize;
568}
569
570void GotSection::finalizeContents() {
571 if (ctx.arg.emachine == EM_PPC64 &&
572 numEntries <= ctx.target->gotHeaderEntriesNum &&
573 !ctx.sym.globalOffsetTable)
574 size = 0;
575 else
576 size = numEntries * ctx.target->gotEntrySize;
577}
578
579bool GotSection::isNeeded() const {
580 // Needed if the GOT symbol is used or the number of entries is more than just
581 // the header. A GOT with just the header may not be needed.
582 return hasGotOffRel || hasDeferredEntries ||
583 numEntries > ctx.target->gotHeaderEntriesNum;
584}
585
586void GotSection::writeTo(uint8_t *buf) {
587 // On PPC64 .got may be needed but empty. Skip the write.
588 if (size == 0)
589 return;
590 ctx.target->writeGotHeader(buf);
591 ctx.target->relocateAlloc(sec&: *this, buf);
592 for (const AuthEntryInfo &authEntry : authEntries) {
593 uint8_t *dest = buf + authEntry.offset;
594
595 if (authEntry.isUndefinedNonPreemptible) {
596 write64(ctx, p: dest, v: 0);
597 continue;
598 }
599 // https://github.com/ARM-software/abi-aa/blob/2024Q3/pauthabielf64/pauthabielf64.rst#default-signing-schema
600 // Signed GOT entries use the IA key for symbols of type STT_FUNC and the
601 // DA key for all other symbol types, with the address of the GOT entry as
602 // the modifier. The static linker must encode the signing schema into the
603 // GOT slot.
604 //
605 // https://github.com/ARM-software/abi-aa/blob/2024Q3/pauthabielf64/pauthabielf64.rst#encoding-the-signing-schema
606 // If address diversity is set and the discriminator
607 // is 0 then modifier = Place
608 uint64_t key = authEntry.isSymbolFunc ? /*IA=*/0b00 : /*DA=*/0b10;
609 uint64_t addrDiversity = 1;
610 write64(ctx, p: dest, v: (addrDiversity << 63) | (key << 60));
611 }
612}
613
614static uint64_t getMipsPageCount(uint64_t size) {
615 return (size + 0xfffe) / 0xffff + 1;
616}
617
618MipsGotSection::MipsGotSection(Ctx &ctx)
619 : SyntheticSection(ctx, ".got", SHT_PROGBITS,
620 SHF_ALLOC | SHF_WRITE | SHF_MIPS_GPREL, 16) {}
621
622void MipsGotSection::addEntry(InputFile &file, Symbol &sym, int64_t addend,
623 RelExpr expr) {
624 FileGot &g = getGot(f&: file);
625 if (expr == RE_MIPS_GOT_LOCAL_PAGE) {
626 if (const OutputSection *os = sym.getOutputSection())
627 g.pagesMap.insert(KV: {os, {&sym}});
628 else
629 g.local16.insert(KV: {{nullptr, getMipsPageAddr(addr: sym.getVA(ctx, addend))}, 0});
630 } else if (sym.isTls())
631 g.tls.insert(KV: {&sym, 0});
632 else if (sym.isPreemptible && expr == R_ABS)
633 g.relocs.insert(KV: {&sym, 0});
634 else if (sym.isPreemptible)
635 g.global.insert(KV: {&sym, 0});
636 else if (expr == RE_MIPS_GOT_OFF32)
637 g.local32.insert(KV: {{&sym, addend}, 0});
638 else
639 g.local16.insert(KV: {{&sym, addend}, 0});
640}
641
642void MipsGotSection::addDynTlsEntry(InputFile &file, Symbol &sym) {
643 getGot(f&: file).dynTlsSymbols.insert(KV: {&sym, 0});
644}
645
646void MipsGotSection::addTlsIndex(InputFile &file) {
647 getGot(f&: file).dynTlsSymbols.insert(KV: {nullptr, 0});
648}
649
650size_t MipsGotSection::FileGot::getEntriesNum() const {
651 return getPageEntriesNum() + local16.size() + global.size() + relocs.size() +
652 tls.size() + dynTlsSymbols.size() * 2;
653}
654
655size_t MipsGotSection::FileGot::getPageEntriesNum() const {
656 size_t num = 0;
657 for (const std::pair<const OutputSection *, FileGot::PageBlock> &p : pagesMap)
658 num += p.second.count;
659 return num;
660}
661
662size_t MipsGotSection::FileGot::getIndexedEntriesNum() const {
663 size_t count = getPageEntriesNum() + local16.size() + global.size();
664 // If there are relocation-only entries in the GOT, TLS entries
665 // are allocated after them. TLS entries should be addressable
666 // by 16-bit index so count both reloc-only and TLS entries.
667 if (!tls.empty() || !dynTlsSymbols.empty())
668 count += relocs.size() + tls.size() + dynTlsSymbols.size() * 2;
669 return count;
670}
671
672MipsGotSection::FileGot &MipsGotSection::getGot(InputFile &f) {
673 if (f.mipsGotIndex == uint32_t(-1)) {
674 gots.emplace_back();
675 gots.back().file = &f;
676 f.mipsGotIndex = gots.size() - 1;
677 }
678 return gots[f.mipsGotIndex];
679}
680
681uint64_t MipsGotSection::getPageEntryOffset(const InputFile *f,
682 const Symbol &sym,
683 int64_t addend) const {
684 const FileGot &g = gots[f->mipsGotIndex];
685 uint64_t index = 0;
686 if (const OutputSection *outSec = sym.getOutputSection()) {
687 uint64_t secAddr = getMipsPageAddr(addr: outSec->addr);
688 uint64_t symAddr = getMipsPageAddr(addr: sym.getVA(ctx, addend));
689 index = g.pagesMap.lookup(Key: outSec).firstIndex + (symAddr - secAddr) / 0xffff;
690 } else {
691 index =
692 g.local16.lookup(Key: {nullptr, getMipsPageAddr(addr: sym.getVA(ctx, addend))});
693 }
694 return index * ctx.arg.wordsize;
695}
696
697uint64_t MipsGotSection::getSymEntryOffset(const InputFile *f, const Symbol &s,
698 int64_t addend) const {
699 const FileGot &g = gots[f->mipsGotIndex];
700 Symbol *sym = const_cast<Symbol *>(&s);
701 if (sym->isTls())
702 return g.tls.lookup(Key: sym) * ctx.arg.wordsize;
703 if (sym->isPreemptible)
704 return g.global.lookup(Key: sym) * ctx.arg.wordsize;
705 return g.local16.lookup(Key: {sym, addend}) * ctx.arg.wordsize;
706}
707
708uint64_t MipsGotSection::getTlsIndexOffset(const InputFile *f) const {
709 const FileGot &g = gots[f->mipsGotIndex];
710 return g.dynTlsSymbols.lookup(Key: nullptr) * ctx.arg.wordsize;
711}
712
713uint64_t MipsGotSection::getGlobalDynOffset(const InputFile *f,
714 const Symbol &s) const {
715 const FileGot &g = gots[f->mipsGotIndex];
716 Symbol *sym = const_cast<Symbol *>(&s);
717 return g.dynTlsSymbols.lookup(Key: sym) * ctx.arg.wordsize;
718}
719
720const Symbol *MipsGotSection::getFirstGlobalEntry() const {
721 if (gots.empty())
722 return nullptr;
723 const FileGot &primGot = gots.front();
724 if (!primGot.global.empty())
725 return primGot.global.front().first;
726 if (!primGot.relocs.empty())
727 return primGot.relocs.front().first;
728 return nullptr;
729}
730
731unsigned MipsGotSection::getLocalEntriesNum() const {
732 if (gots.empty())
733 return headerEntriesNum;
734 return headerEntriesNum + gots.front().getPageEntriesNum() +
735 gots.front().local16.size();
736}
737
738bool MipsGotSection::tryMergeGots(FileGot &dst, FileGot &src, bool isPrimary) {
739 FileGot tmp = dst;
740 set_union(S1&: tmp.pagesMap, S2: src.pagesMap);
741 set_union(S1&: tmp.local16, S2: src.local16);
742 set_union(S1&: tmp.global, S2: src.global);
743 set_union(S1&: tmp.relocs, S2: src.relocs);
744 set_union(S1&: tmp.tls, S2: src.tls);
745 set_union(S1&: tmp.dynTlsSymbols, S2: src.dynTlsSymbols);
746
747 size_t count = isPrimary ? headerEntriesNum : 0;
748 count += tmp.getIndexedEntriesNum();
749
750 if (count * ctx.arg.wordsize > ctx.arg.mipsGotSize)
751 return false;
752
753 std::swap(a&: tmp, b&: dst);
754 return true;
755}
756
757void MipsGotSection::finalizeContents() { updateAllocSize(ctx); }
758
759bool MipsGotSection::updateAllocSize(Ctx &ctx) {
760 size = headerEntriesNum * ctx.arg.wordsize;
761 for (const FileGot &g : gots)
762 size += g.getEntriesNum() * ctx.arg.wordsize;
763 return false;
764}
765
766void MipsGotSection::build() {
767 if (gots.empty())
768 return;
769
770 std::vector<FileGot> mergedGots(1);
771
772 // For each GOT move non-preemptible symbols from the `Global`
773 // to `Local16` list. Preemptible symbol might become non-preemptible
774 // one if, for example, it gets a related copy relocation.
775 for (FileGot &got : gots) {
776 for (auto &p: got.global)
777 if (!p.first->isPreemptible)
778 got.local16.insert(KV: {{p.first, 0}, 0});
779 got.global.remove_if(Pred: [&](const std::pair<Symbol *, size_t> &p) {
780 return !p.first->isPreemptible;
781 });
782 }
783
784 // For each GOT remove "reloc-only" entry if there is "global"
785 // entry for the same symbol. And add local entries which indexed
786 // using 32-bit value at the end of 16-bit entries.
787 for (FileGot &got : gots) {
788 got.relocs.remove_if(Pred: [&](const std::pair<Symbol *, size_t> &p) {
789 return got.global.contains(Key: p.first);
790 });
791 set_union(S1&: got.local16, S2: got.local32);
792 got.local32.clear();
793 }
794
795 // Evaluate number of "reloc-only" entries in the resulting GOT.
796 // To do that put all unique "reloc-only" and "global" entries
797 // from all GOTs to the future primary GOT.
798 FileGot *primGot = &mergedGots.front();
799 for (FileGot &got : gots) {
800 set_union(S1&: primGot->relocs, S2: got.global);
801 set_union(S1&: primGot->relocs, S2: got.relocs);
802 got.relocs.clear();
803 }
804
805 // Evaluate number of "page" entries in each GOT.
806 for (FileGot &got : gots) {
807 for (std::pair<const OutputSection *, FileGot::PageBlock> &p :
808 got.pagesMap) {
809 const OutputSection *os = p.first;
810 uint64_t secSize = 0;
811 for (SectionCommand *cmd : os->commands) {
812 if (auto *isd = dyn_cast<InputSectionDescription>(Val: cmd))
813 for (InputSection *isec : isd->sections) {
814 uint64_t off = alignToPowerOf2(Value: secSize, Align: isec->addralign);
815 secSize = off + isec->getSize();
816 }
817 }
818 p.second.count = getMipsPageCount(size: secSize);
819 }
820 }
821
822 // Merge GOTs. Try to join as much as possible GOTs but do not exceed
823 // maximum GOT size. At first, try to fill the primary GOT because
824 // the primary GOT can be accessed in the most effective way. If it
825 // is not possible, try to fill the last GOT in the list, and finally
826 // create a new GOT if both attempts failed.
827 for (FileGot &srcGot : gots) {
828 InputFile *file = srcGot.file;
829 if (tryMergeGots(dst&: mergedGots.front(), src&: srcGot, isPrimary: true)) {
830 file->mipsGotIndex = 0;
831 } else {
832 // If this is the first time we failed to merge with the primary GOT,
833 // MergedGots.back() will also be the primary GOT. We must make sure not
834 // to try to merge again with isPrimary=false, as otherwise, if the
835 // inputs are just right, we could allow the primary GOT to become 1 or 2
836 // words bigger due to ignoring the header size.
837 if (mergedGots.size() == 1 ||
838 !tryMergeGots(dst&: mergedGots.back(), src&: srcGot, isPrimary: false)) {
839 mergedGots.emplace_back();
840 std::swap(a&: mergedGots.back(), b&: srcGot);
841 }
842 file->mipsGotIndex = mergedGots.size() - 1;
843 }
844 }
845 std::swap(x&: gots, y&: mergedGots);
846
847 // Reduce number of "reloc-only" entries in the primary GOT
848 // by subtracting "global" entries in the primary GOT.
849 primGot = &gots.front();
850 primGot->relocs.remove_if(Pred: [&](const std::pair<Symbol *, size_t> &p) {
851 return primGot->global.contains(Key: p.first);
852 });
853
854 // Calculate indexes for each GOT entry.
855 size_t index = headerEntriesNum;
856 for (FileGot &got : gots) {
857 got.startIndex = &got == primGot ? 0 : index;
858 for (std::pair<const OutputSection *, FileGot::PageBlock> &p :
859 got.pagesMap) {
860 // For each output section referenced by GOT page relocations calculate
861 // and save into pagesMap an upper bound of MIPS GOT entries required
862 // to store page addresses of local symbols. We assume the worst case -
863 // each 64kb page of the output section has at least one GOT relocation
864 // against it. And take in account the case when the section intersects
865 // page boundaries.
866 p.second.firstIndex = index;
867 index += p.second.count;
868 }
869 for (auto &p: got.local16)
870 p.second = index++;
871 for (auto &p: got.global)
872 p.second = index++;
873 for (auto &p: got.relocs)
874 p.second = index++;
875 for (auto &p: got.tls)
876 p.second = index++;
877 for (auto &p: got.dynTlsSymbols) {
878 p.second = index;
879 index += 2;
880 }
881 }
882
883 // Update SymbolAux::gotIdx field to use this
884 // value later in the `sortMipsSymbols` function.
885 for (auto &p : primGot->global) {
886 if (p.first->auxIdx == 0)
887 p.first->allocateAux(ctx);
888 ctx.symAux.back().gotIdx = p.second;
889 }
890 for (auto &p : primGot->relocs) {
891 if (p.first->auxIdx == 0)
892 p.first->allocateAux(ctx);
893 ctx.symAux.back().gotIdx = p.second;
894 }
895
896 // Create relocations.
897 //
898 // Note the primary GOT's local and global relocations are implicit, and the
899 // MIPS ABI requires the VA be written even for the global entries, so we
900 // treat both as constants here.
901 for (FileGot &got : gots) {
902 // Create relocations for TLS entries.
903 for (std::pair<Symbol *, size_t> &p : got.tls) {
904 Symbol *s = p.first;
905 uint64_t offset = p.second * ctx.arg.wordsize;
906 // When building a shared library we still need a dynamic relocation
907 // for the TP-relative offset as we don't know how much other data will
908 // be allocated before us in the static TLS block.
909 if (!s->isPreemptible && !ctx.arg.shared)
910 addConstant(r: {.expr: R_TPREL, .type: ctx.target->symbolicRel, .offset: offset, .addend: 0, .sym: s});
911 else
912 ctx.in.relaDyn->addAddendOnlyRelocIfNonPreemptible(
913 dynType: ctx.target->tlsGotRel, isec&: *this, offsetInSec: offset, sym&: *s, addendRelType: ctx.target->symbolicRel);
914 }
915 for (std::pair<Symbol *, size_t> &p : got.dynTlsSymbols) {
916 Symbol *s = p.first;
917 uint64_t off = p.second * ctx.arg.wordsize;
918 if (s == nullptr) {
919 if (ctx.arg.shared)
920 ctx.in.relaDyn->addReloc(reloc: {ctx.target->tlsModuleIndexRel, this, off});
921 else
922 addConstant(
923 r: {.expr: R_ADDEND, .type: ctx.target->symbolicRel, .offset: off, .addend: 1, .sym: ctx.dummySym});
924 } else {
925 // When building a shared library we still need a dynamic relocation
926 // for the module index. Therefore only checking for
927 // S->isPreemptible is not sufficient (this happens e.g. for
928 // thread-locals that have been marked as local through a linker script)
929 // However, we can skip writing the TLS offset reloc for non-preemptible
930 // symbols since it is known even in shared libraries
931 uint64_t offsetOff = off + ctx.arg.wordsize;
932 if (s->isPreemptible) {
933 ctx.in.relaDyn->addSymbolReloc(dynType: ctx.target->tlsModuleIndexRel, isec&: *this,
934 offsetInSec: off, sym&: *s);
935 ctx.in.relaDyn->addSymbolReloc(dynType: ctx.target->tlsOffsetRel, isec&: *this,
936 offsetInSec: offsetOff, sym&: *s);
937 } else {
938 if (ctx.arg.shared)
939 ctx.in.relaDyn->addReloc(
940 reloc: {ctx.target->tlsModuleIndexRel, this, off});
941 else
942 // Write one to the GOT slot.
943 addConstant(r: {.expr: R_ADDEND, .type: ctx.target->symbolicRel, .offset: off, .addend: 1, .sym: s});
944 addConstant(r: {.expr: R_ABS, .type: ctx.target->tlsOffsetRel, .offset: offsetOff, .addend: 0, .sym: s});
945 }
946 }
947 }
948
949 // Relocations for "global" entries.
950 for (const std::pair<Symbol *, size_t> &p : got.global) {
951 uint64_t offset = p.second * ctx.arg.wordsize;
952 if (&got == primGot)
953 addConstant(r: {.expr: R_ABS, .type: ctx.target->relativeRel, .offset: offset, .addend: 0, .sym: p.first});
954 else
955 ctx.in.relaDyn->addSymbolReloc(dynType: ctx.target->relativeRel, isec&: *this, offsetInSec: offset,
956 sym&: *p.first);
957 }
958 // Relocation-only entries exist as dummy entries for dynamic symbols that
959 // aren't otherwise in the primary GOT, as the ABI requires an entry for
960 // each dynamic symbol. Secondary GOTs have no need for them.
961 assert((got.relocs.empty() || &got == primGot) &&
962 "Relocation-only entries should only be in the primary GOT");
963 for (const std::pair<Symbol *, size_t> &p : got.relocs) {
964 uint64_t offset = p.second * ctx.arg.wordsize;
965 addConstant(r: {.expr: R_ABS, .type: ctx.target->relativeRel, .offset: offset, .addend: 0, .sym: p.first});
966 }
967
968 // Relocations for "local" entries
969 for (const std::pair<const OutputSection *, FileGot::PageBlock> &l :
970 got.pagesMap) {
971 size_t pageCount = l.second.count;
972 for (size_t pi = 0; pi < pageCount; ++pi) {
973 uint64_t offset = (l.second.firstIndex + pi) * ctx.arg.wordsize;
974 int64_t addend = int64_t(pi * 0x10000);
975 if (!ctx.arg.isPic || &got == primGot)
976 addConstant(r: {.expr: RE_MIPS_OSEC_LOCAL_PAGE, .type: ctx.target->relativeRel, .offset: offset,
977 .addend: addend, .sym: l.second.repSym});
978 else
979 ctx.in.relaDyn->addRelativeReloc(
980 dynType: ctx.target->relativeRel, isec&: *this, offsetInSec: offset, sym&: *l.second.repSym, addend,
981 addendRelType: ctx.target->relativeRel, expr: RE_MIPS_OSEC_LOCAL_PAGE);
982 }
983 }
984 for (const std::pair<GotEntry, size_t> &p : got.local16) {
985 uint64_t offset = p.second * ctx.arg.wordsize;
986 if (p.first.first == nullptr)
987 addConstant(r: {.expr: R_ADDEND, .type: ctx.target->relativeRel, .offset: offset, .addend: p.first.second,
988 .sym: ctx.dummySym});
989 else if (!ctx.arg.isPic || &got == primGot)
990 addConstant(r: {.expr: R_ABS, .type: ctx.target->relativeRel, .offset: offset, .addend: p.first.second,
991 .sym: p.first.first});
992 else
993 ctx.in.relaDyn->addRelativeReloc(dynType: ctx.target->relativeRel, isec&: *this, offsetInSec: offset,
994 sym&: *p.first.first, addend: p.first.second,
995 addendRelType: ctx.target->relativeRel, expr: R_ABS);
996 }
997 }
998}
999
1000bool MipsGotSection::isNeeded() const {
1001 // We add the .got section to the result for dynamic MIPS target because
1002 // its address and properties are mentioned in the .dynamic section.
1003 return !ctx.arg.relocatable;
1004}
1005
1006uint64_t MipsGotSection::getGp(const InputFile *f) const {
1007 // For files without related GOT or files refer a primary GOT
1008 // returns "common" _gp value. For secondary GOTs calculate
1009 // individual _gp values.
1010 if (!f || f->mipsGotIndex == uint32_t(-1) || f->mipsGotIndex == 0)
1011 return ctx.sym.mipsGp->getVA(ctx, addend: 0);
1012 return getVA() + gots[f->mipsGotIndex].startIndex * ctx.arg.wordsize + 0x7ff0;
1013}
1014
1015void MipsGotSection::writeTo(uint8_t *buf) {
1016 // Set the MSB of the second GOT slot. This is not required by any
1017 // MIPS ABI documentation, though.
1018 //
1019 // There is a comment in glibc saying that "The MSB of got[1] of a
1020 // gnu object is set to identify gnu objects," and in GNU gold it
1021 // says "the second entry will be used by some runtime loaders".
1022 // But how this field is being used is unclear.
1023 //
1024 // We are not really willing to mimic other linkers behaviors
1025 // without understanding why they do that, but because all files
1026 // generated by GNU tools have this special GOT value, and because
1027 // we've been doing this for years, it is probably a safe bet to
1028 // keep doing this for now. We really need to revisit this to see
1029 // if we had to do this.
1030 writeUint(ctx, buf: buf + ctx.arg.wordsize,
1031 val: (uint64_t)1 << (ctx.arg.wordsize * 8 - 1));
1032 ctx.target->relocateAlloc(sec&: *this, buf);
1033}
1034
1035// On PowerPC the .plt section is used to hold the table of function addresses
1036// instead of the .got.plt, and the type is SHT_NOBITS similar to a .bss
1037// section. I don't know why we have a BSS style type for the section but it is
1038// consistent across both 64-bit PowerPC ABIs as well as the 32-bit PowerPC ABI.
1039GotPltSection::GotPltSection(Ctx &ctx)
1040 : SyntheticSection(ctx, ".got.plt", SHT_PROGBITS, SHF_ALLOC | SHF_WRITE,
1041 ctx.target->gotEntrySize) {
1042 if (ctx.arg.emachine == EM_PPC) {
1043 name = ".plt";
1044 } else if (ctx.arg.emachine == EM_PPC64) {
1045 type = SHT_NOBITS;
1046 name = ".plt";
1047 }
1048}
1049
1050void GotPltSection::addEntry(Symbol &sym) {
1051 assert(sym.auxIdx == ctx.symAux.size() - 1 &&
1052 ctx.symAux.back().pltIdx == entries.size());
1053 entries.push_back(Elt: &sym);
1054}
1055
1056size_t GotPltSection::getSize() const {
1057 return (ctx.target->gotPltHeaderEntriesNum + entries.size()) *
1058 ctx.target->gotEntrySize;
1059}
1060
1061void GotPltSection::writeTo(uint8_t *buf) {
1062 ctx.target->writeGotPltHeader(buf);
1063 buf += ctx.target->gotPltHeaderEntriesNum * ctx.target->gotEntrySize;
1064 for (const Symbol *b : entries) {
1065 ctx.target->writeGotPlt(buf, s: *b);
1066 buf += ctx.target->gotEntrySize;
1067 }
1068}
1069
1070bool GotPltSection::isNeeded() const {
1071 // We need to emit GOTPLT even if it's empty if there's a relocation relative
1072 // to it.
1073 return !entries.empty() || hasGotPltOffRel;
1074}
1075
1076static StringRef getIgotPltName(Ctx &ctx) {
1077 // On ARM the IgotPltSection is part of the GotSection.
1078 if (ctx.arg.emachine == EM_ARM)
1079 return ".got";
1080
1081 // On PowerPC64 the GotPltSection is renamed to '.plt' so the IgotPltSection
1082 // needs to be named the same.
1083 if (ctx.arg.emachine == EM_PPC64)
1084 return ".plt";
1085
1086 return ".got.plt";
1087}
1088
1089// On PowerPC64 the GotPltSection type is SHT_NOBITS so we have to follow suit
1090// with the IgotPltSection.
1091IgotPltSection::IgotPltSection(Ctx &ctx)
1092 : SyntheticSection(ctx, getIgotPltName(ctx),
1093 ctx.arg.emachine == EM_PPC64 ? SHT_NOBITS : SHT_PROGBITS,
1094 SHF_ALLOC | SHF_WRITE, ctx.target->gotEntrySize) {}
1095
1096void IgotPltSection::addEntry(Symbol &sym) {
1097 assert(ctx.symAux.back().pltIdx == entries.size());
1098 entries.push_back(Elt: &sym);
1099}
1100
1101size_t IgotPltSection::getSize() const {
1102 return entries.size() * ctx.target->gotEntrySize;
1103}
1104
1105void IgotPltSection::writeTo(uint8_t *buf) {
1106 for (const Symbol *b : entries) {
1107 ctx.target->writeIgotPlt(buf, s: *b);
1108 buf += ctx.target->gotEntrySize;
1109 }
1110}
1111
1112StringTableSection::StringTableSection(Ctx &ctx, StringRef name, bool dynamic)
1113 : SyntheticSection(ctx, name, SHT_STRTAB, dynamic ? (uint64_t)SHF_ALLOC : 0,
1114 1),
1115 dynamic(dynamic) {
1116 // ELF string tables start with a NUL byte.
1117 strings.push_back(Elt: "");
1118 stringMap.try_emplace(Key: CachedHashStringRef(""), Args: 0);
1119 size = 1;
1120}
1121
1122// Adds a string to the string table. If `hashIt` is true we hash and check for
1123// duplicates. It is optional because the name of global symbols are already
1124// uniqued and hashing them again has a big cost for a small value: uniquing
1125// them with some other string that happens to be the same.
1126unsigned StringTableSection::addString(StringRef s, bool hashIt) {
1127 if (hashIt) {
1128 auto r = stringMap.try_emplace(Key: CachedHashStringRef(s), Args&: size);
1129 if (!r.second)
1130 return r.first->second;
1131 }
1132 if (s.empty())
1133 return 0;
1134 unsigned ret = this->size;
1135 this->size = this->size + s.size() + 1;
1136 strings.push_back(Elt: s);
1137 return ret;
1138}
1139
1140void StringTableSection::writeTo(uint8_t *buf) {
1141 for (StringRef s : strings) {
1142 memcpy(dest: buf, src: s.data(), n: s.size());
1143 buf[s.size()] = '\0';
1144 buf += s.size() + 1;
1145 }
1146}
1147
1148// Returns the number of entries in .gnu.version_d: the number of
1149// non-VER_NDX_LOCAL-non-VER_NDX_GLOBAL definitions, plus 1.
1150// Note that we don't support vd_cnt > 1 yet.
1151static unsigned getVerDefNum(Ctx &ctx) {
1152 return namedVersionDefs(ctx).size() + 1;
1153}
1154
1155template <class ELFT>
1156DynamicSection<ELFT>::DynamicSection(Ctx &ctx)
1157 : SyntheticSection(ctx, ".dynamic", SHT_DYNAMIC, SHF_ALLOC | SHF_WRITE,
1158 ctx.arg.wordsize) {
1159 this->entsize = ELFT::Is64Bits ? 16 : 8;
1160
1161 // .dynamic section is not writable on MIPS and on Fuchsia OS
1162 // which passes -z rodynamic.
1163 // See "Special Section" in Chapter 4 in the following document:
1164 // ftp://www.linux-mips.org/pub/linux/mips/doc/ABI/mipsabi.pdf
1165 if (ctx.arg.emachine == EM_MIPS || ctx.arg.zRodynamic)
1166 this->flags = SHF_ALLOC;
1167}
1168
1169// The output section .rela.dyn may include these synthetic sections:
1170//
1171// - ctx.in.relaDyn
1172// - ctx.in.relaPlt: this is included if a linker script places .rela.plt inside
1173// .rela.dyn
1174//
1175// DT_RELASZ is the total size of the included sections.
1176static uint64_t addRelaSz(Ctx &ctx, const RelocationBaseSection &relaDyn) {
1177 size_t size = relaDyn.getSize();
1178 if (ctx.in.relaPlt->getParent() == relaDyn.getParent())
1179 size += ctx.in.relaPlt->getSize();
1180 return size;
1181}
1182
1183// A Linker script may assign the RELA relocation sections to the same
1184// output section. When this occurs we cannot just use the OutputSection
1185// Size. Moreover the [DT_JMPREL, DT_JMPREL + DT_PLTRELSZ) is permitted to
1186// overlap with the [DT_RELA, DT_RELA + DT_RELASZ).
1187static uint64_t addPltRelSz(Ctx &ctx) { return ctx.in.relaPlt->getSize(); }
1188
1189// Add remaining entries to complete .dynamic contents.
1190template <class ELFT>
1191std::vector<std::pair<int32_t, uint64_t>>
1192DynamicSection<ELFT>::computeContents() {
1193 std::vector<std::pair<int32_t, uint64_t>> entries;
1194
1195 auto addInt = [&](int32_t tag, uint64_t val) {
1196 entries.emplace_back(args&: tag, args&: val);
1197 };
1198 auto addInSec = [&](int32_t tag, const InputSection &sec) {
1199 entries.emplace_back(args&: tag, args: sec.getVA());
1200 };
1201
1202 for (StringRef s : ctx.arg.filterList)
1203 addInt(DT_FILTER, ctx.in.dynStrTab->addString(s));
1204 for (StringRef s : ctx.arg.auxiliaryList)
1205 addInt(DT_AUXILIARY, ctx.in.dynStrTab->addString(s));
1206
1207 if (!ctx.arg.rpath.empty())
1208 addInt(ctx.arg.enableNewDtags ? DT_RUNPATH : DT_RPATH,
1209 ctx.in.dynStrTab->addString(s: ctx.arg.rpath));
1210
1211 for (SharedFile *file : ctx.sharedFiles)
1212 if (file->isNeeded)
1213 addInt(DT_NEEDED, ctx.in.dynStrTab->addString(s: file->soName));
1214
1215 if (!ctx.arg.soName.empty())
1216 addInt(DT_SONAME, ctx.in.dynStrTab->addString(s: ctx.arg.soName));
1217
1218 // Set DT_FLAGS and DT_FLAGS_1.
1219 uint32_t dtFlags = 0;
1220 uint32_t dtFlags1 = 0;
1221 if (ctx.arg.bsymbolic == BsymbolicKind::All)
1222 dtFlags |= DF_SYMBOLIC;
1223 if (ctx.arg.zGlobal)
1224 dtFlags1 |= DF_1_GLOBAL;
1225 if (ctx.arg.zInitfirst)
1226 dtFlags1 |= DF_1_INITFIRST;
1227 if (ctx.arg.zInterpose)
1228 dtFlags1 |= DF_1_INTERPOSE;
1229 if (ctx.arg.zNodefaultlib)
1230 dtFlags1 |= DF_1_NODEFLIB;
1231 if (ctx.arg.zNodelete)
1232 dtFlags1 |= DF_1_NODELETE;
1233 if (ctx.arg.zNodlopen)
1234 dtFlags1 |= DF_1_NOOPEN;
1235 if (ctx.arg.pie)
1236 dtFlags1 |= DF_1_PIE;
1237 if (ctx.arg.zNow) {
1238 dtFlags |= DF_BIND_NOW;
1239 dtFlags1 |= DF_1_NOW;
1240 }
1241 if (ctx.arg.zOrigin) {
1242 dtFlags |= DF_ORIGIN;
1243 dtFlags1 |= DF_1_ORIGIN;
1244 }
1245 if (!ctx.arg.zText)
1246 dtFlags |= DF_TEXTREL;
1247 if (ctx.hasTlsIe && ctx.arg.shared)
1248 dtFlags |= DF_STATIC_TLS;
1249
1250 if (dtFlags)
1251 addInt(DT_FLAGS, dtFlags);
1252 if (dtFlags1)
1253 addInt(DT_FLAGS_1, dtFlags1);
1254
1255 // DT_DEBUG is a pointer to debug information used by debuggers at runtime. We
1256 // need it for each process, so we don't write it for DSOs. The loader writes
1257 // the pointer into this entry.
1258 //
1259 // DT_DEBUG is the only .dynamic entry that needs to be written to. Some
1260 // systems (currently only Fuchsia OS) provide other means to give the
1261 // debugger this information. Such systems may choose make .dynamic read-only.
1262 // If the target is such a system (used -z rodynamic) don't write DT_DEBUG.
1263 if (!ctx.arg.shared && !ctx.arg.relocatable && !ctx.arg.zRodynamic)
1264 addInt(DT_DEBUG, 0);
1265
1266 if (ctx.in.relaDyn->isNeeded()) {
1267 addInSec(ctx.in.relaDyn->dynamicTag, *ctx.in.relaDyn);
1268 entries.emplace_back(ctx.in.relaDyn->sizeDynamicTag,
1269 addRelaSz(ctx, *ctx.in.relaDyn));
1270
1271 bool isRela = ctx.arg.isRela;
1272 addInt(isRela ? DT_RELAENT : DT_RELENT,
1273 isRela ? sizeof(Elf_Rela) : sizeof(Elf_Rel));
1274
1275 // MIPS dynamic loader does not support RELCOUNT tag.
1276 // The problem is in the tight relation between dynamic
1277 // relocations and GOT. So do not emit this tag on MIPS.
1278 if (ctx.arg.emachine != EM_MIPS) {
1279 size_t numRelativeRels = ctx.in.relaDyn->getRelativeRelocCount();
1280 if (ctx.arg.zCombreloc && numRelativeRels)
1281 addInt(isRela ? DT_RELACOUNT : DT_RELCOUNT, numRelativeRels);
1282 }
1283 }
1284 if (ctx.in.relrDyn && ctx.in.relrDyn->getParent() &&
1285 !ctx.in.relrDyn->relocs.empty()) {
1286 addInSec(ctx.arg.useAndroidRelrTags ? DT_ANDROID_RELR : DT_RELR,
1287 *ctx.in.relrDyn);
1288 addInt(ctx.arg.useAndroidRelrTags ? DT_ANDROID_RELRSZ : DT_RELRSZ,
1289 ctx.in.relrDyn->getParent()->size);
1290 addInt(ctx.arg.useAndroidRelrTags ? DT_ANDROID_RELRENT : DT_RELRENT,
1291 sizeof(Elf_Relr));
1292 }
1293 if (ctx.in.relrAuthDyn && ctx.in.relrAuthDyn->getParent() &&
1294 !ctx.in.relrAuthDyn->relocs.empty()) {
1295 addInSec(DT_AARCH64_AUTH_RELR, *ctx.in.relrAuthDyn);
1296 addInt(DT_AARCH64_AUTH_RELRSZ, ctx.in.relrAuthDyn->getParent()->size);
1297 addInt(DT_AARCH64_AUTH_RELRENT, sizeof(Elf_Relr));
1298 }
1299 if (ctx.in.relaPlt->isNeeded()) {
1300 addInSec(DT_JMPREL, *ctx.in.relaPlt);
1301 entries.emplace_back(DT_PLTRELSZ, addPltRelSz(ctx));
1302 switch (ctx.arg.emachine) {
1303 case EM_MIPS:
1304 addInSec(DT_MIPS_PLTGOT, *ctx.in.gotPlt);
1305 break;
1306 case EM_S390:
1307 addInSec(DT_PLTGOT, *ctx.in.got);
1308 break;
1309 case EM_SPARCV9:
1310 addInSec(DT_PLTGOT, *ctx.in.plt);
1311 break;
1312 case EM_AARCH64:
1313 if (llvm::find_if(ctx.in.relaPlt->relocs, [&ctx = ctx](
1314 const DynamicReloc &r) {
1315 return r.type == ctx.target->pltRel &&
1316 r.sym->stOther & STO_AARCH64_VARIANT_PCS;
1317 }) != ctx.in.relaPlt->relocs.end())
1318 addInt(DT_AARCH64_VARIANT_PCS, 0);
1319 addInSec(DT_PLTGOT, *ctx.in.gotPlt);
1320 break;
1321 case EM_RISCV:
1322 if (llvm::any_of(ctx.in.relaPlt->relocs, [&ctx = ctx](
1323 const DynamicReloc &r) {
1324 return r.type == ctx.target->pltRel &&
1325 (r.sym->stOther & STO_RISCV_VARIANT_CC);
1326 }))
1327 addInt(DT_RISCV_VARIANT_CC, 0);
1328 [[fallthrough]];
1329 default:
1330 addInSec(DT_PLTGOT, *ctx.in.gotPlt);
1331 break;
1332 }
1333 addInt(DT_PLTREL, ctx.arg.isRela ? DT_RELA : DT_REL);
1334 }
1335
1336 if (ctx.arg.zMarkPlt && ctx.in.plt->isNeeded()) {
1337 addInSec(DT_X86_64_PLT, *ctx.in.plt);
1338 addInt(DT_X86_64_PLTSZ, ctx.in.plt->getSize());
1339 addInt(DT_X86_64_PLTENT, ctx.target->pltEntrySize);
1340 }
1341
1342 if (ctx.arg.emachine == EM_AARCH64) {
1343 if (ctx.arg.andFeatures & GNU_PROPERTY_AARCH64_FEATURE_1_BTI)
1344 addInt(DT_AARCH64_BTI_PLT, 0);
1345 if (ctx.arg.zPacPlt)
1346 addInt(DT_AARCH64_PAC_PLT, 0);
1347
1348 if (hasMemtag(ctx)) {
1349 addInt(DT_AARCH64_MEMTAG_MODE,
1350 ctx.arg.memtagMode == NT_MEMTAG_LEVEL_ASYNC);
1351 addInt(DT_AARCH64_MEMTAG_HEAP, ctx.arg.memtagHeap);
1352 addInt(DT_AARCH64_MEMTAG_STACK, ctx.arg.memtagStack);
1353 if (ctx.in.memtagGlobalDescriptors->isNeeded()) {
1354 addInSec(DT_AARCH64_MEMTAG_GLOBALS, *ctx.in.memtagGlobalDescriptors);
1355 addInt(DT_AARCH64_MEMTAG_GLOBALSSZ,
1356 ctx.in.memtagGlobalDescriptors->getSize());
1357 }
1358 }
1359 }
1360
1361 addInSec(DT_SYMTAB, *ctx.in.dynSymTab);
1362 addInt(DT_SYMENT, sizeof(Elf_Sym));
1363 addInSec(DT_STRTAB, *ctx.in.dynStrTab);
1364 addInt(DT_STRSZ, ctx.in.dynStrTab->getSize());
1365 if (!ctx.arg.zText)
1366 addInt(DT_TEXTREL, 0);
1367 if (ctx.in.gnuHashTab && ctx.in.gnuHashTab->getParent())
1368 addInSec(DT_GNU_HASH, *ctx.in.gnuHashTab);
1369 if (ctx.in.hashTab && ctx.in.hashTab->getParent())
1370 addInSec(DT_HASH, *ctx.in.hashTab);
1371
1372 if (ctx.out.preinitArray) {
1373 addInt(DT_PREINIT_ARRAY, ctx.out.preinitArray->addr);
1374 addInt(DT_PREINIT_ARRAYSZ, ctx.out.preinitArray->size);
1375 }
1376 if (ctx.out.initArray) {
1377 addInt(DT_INIT_ARRAY, ctx.out.initArray->addr);
1378 addInt(DT_INIT_ARRAYSZ, ctx.out.initArray->size);
1379 }
1380 if (ctx.out.finiArray) {
1381 addInt(DT_FINI_ARRAY, ctx.out.finiArray->addr);
1382 addInt(DT_FINI_ARRAYSZ, ctx.out.finiArray->size);
1383 }
1384
1385 if (Symbol *b = ctx.symtab->find(name: ctx.arg.init))
1386 if (b->isDefined())
1387 addInt(DT_INIT, b->getVA(ctx));
1388 if (Symbol *b = ctx.symtab->find(name: ctx.arg.fini))
1389 if (b->isDefined())
1390 addInt(DT_FINI, b->getVA(ctx));
1391
1392 if (ctx.in.verSym && ctx.in.verSym->isNeeded())
1393 addInSec(DT_VERSYM, *ctx.in.verSym);
1394 if (ctx.in.verDef && ctx.in.verDef->isLive()) {
1395 addInSec(DT_VERDEF, *ctx.in.verDef);
1396 addInt(DT_VERDEFNUM, getVerDefNum(ctx));
1397 }
1398 if (ctx.in.verNeed && ctx.in.verNeed->isNeeded()) {
1399 addInSec(DT_VERNEED, *ctx.in.verNeed);
1400 unsigned needNum = 0;
1401 for (SharedFile *f : ctx.sharedFiles)
1402 if (!f->verneedInfo.empty())
1403 ++needNum;
1404 addInt(DT_VERNEEDNUM, needNum);
1405 }
1406
1407 if (ctx.arg.emachine == EM_MIPS) {
1408 addInt(DT_MIPS_RLD_VERSION, 1);
1409 addInt(DT_MIPS_FLAGS, RHF_NOTPOT);
1410 addInt(DT_MIPS_BASE_ADDRESS, ctx.target->getImageBase());
1411 addInt(DT_MIPS_SYMTABNO, ctx.in.dynSymTab->getNumSymbols());
1412 addInt(DT_MIPS_LOCAL_GOTNO, ctx.in.mipsGot->getLocalEntriesNum());
1413
1414 if (const Symbol *b = ctx.in.mipsGot->getFirstGlobalEntry())
1415 addInt(DT_MIPS_GOTSYM, b->dynsymIndex);
1416 else
1417 addInt(DT_MIPS_GOTSYM, ctx.in.dynSymTab->getNumSymbols());
1418 addInSec(DT_PLTGOT, *ctx.in.mipsGot);
1419 if (ctx.in.mipsRldMap) {
1420 if (!ctx.arg.pie)
1421 addInSec(DT_MIPS_RLD_MAP, *ctx.in.mipsRldMap);
1422 // Store the offset to the .rld_map section
1423 // relative to the address of the tag.
1424 addInt(DT_MIPS_RLD_MAP_REL,
1425 ctx.in.mipsRldMap->getVA() - (getVA() + entries.size() * entsize));
1426 }
1427 }
1428
1429 // DT_PPC_GOT indicates to glibc Secure PLT is used. If DT_PPC_GOT is absent,
1430 // glibc assumes the old-style BSS PLT layout which we don't support.
1431 if (ctx.arg.emachine == EM_PPC)
1432 addInSec(DT_PPC_GOT, *ctx.in.got);
1433
1434 // Glink dynamic tag is required by the V2 abi if the plt section isn't empty.
1435 if (ctx.arg.emachine == EM_PPC64 && ctx.in.plt->isNeeded()) {
1436 // The Glink tag points to 32 bytes before the first lazy symbol resolution
1437 // stub, which starts directly after the header.
1438 addInt(DT_PPC64_GLINK,
1439 ctx.in.plt->getVA() + ctx.target->pltHeaderSize - 32);
1440 }
1441
1442 if (ctx.arg.emachine == EM_PPC64)
1443 addInt(DT_PPC64_OPT, ctx.target->ppc64DynamicSectionOpt);
1444
1445 addInt(DT_NULL, 0);
1446 return entries;
1447}
1448
1449template <class ELFT> void DynamicSection<ELFT>::finalizeContents() {
1450 if (OutputSection *sec = ctx.in.dynStrTab->getParent())
1451 getParent()->link = sec->sectionIndex;
1452 this->size = computeContents().size() * this->entsize;
1453}
1454
1455template <class ELFT> void DynamicSection<ELFT>::writeTo(uint8_t *buf) {
1456 auto *p = reinterpret_cast<Elf_Dyn *>(buf);
1457
1458 for (std::pair<int32_t, uint64_t> kv : computeContents()) {
1459 p->d_tag = kv.first;
1460 p->d_un.d_val = kv.second;
1461 ++p;
1462 }
1463}
1464
1465uint64_t DynamicReloc::getOffset() const {
1466 return inputSec->getRelocVA(offset: offsetInSec);
1467}
1468
1469int64_t DynamicReloc::computeAddend(Ctx &ctx) const {
1470 assert(!isFinal && "addend already computed");
1471 uint64_t ca = inputSec->getRelocTargetVA(
1472 ctx, r: Relocation{.expr: expr, .type: type, .offset: 0, .addend: addend, .sym: sym}, p: getOffset());
1473 return ctx.arg.is64 ? ca : SignExtend64<32>(x: ca);
1474}
1475
1476uint32_t DynamicReloc::getSymIndex(SymbolTableBaseSection *symTab) const {
1477 if (!needsDynSymIndex())
1478 return 0;
1479
1480 size_t index = symTab->getSymbolIndex(sym: *sym);
1481 assert((index != 0 ||
1482 (type != symTab->ctx.target->gotRel &&
1483 type != symTab->ctx.target->pltRel) ||
1484 !symTab->ctx.in.dynSymTab->getParent()) &&
1485 "GOT or PLT relocation must refer to symbol in dynamic symbol table");
1486 return index;
1487}
1488
1489RelocationBaseSection::RelocationBaseSection(Ctx &ctx, StringRef name,
1490 uint32_t type, int32_t dynamicTag,
1491 int32_t sizeDynamicTag,
1492 bool combreloc,
1493 unsigned concurrency)
1494 : SyntheticSection(ctx, name, type, SHF_ALLOC, ctx.arg.wordsize),
1495 dynamicTag(dynamicTag), sizeDynamicTag(sizeDynamicTag),
1496 relocsVec(concurrency), relativeRel(ctx.target->relativeRel),
1497 combreloc(combreloc) {}
1498
1499void RelocationBaseSection::addSymbolReloc(
1500 RelType dynType, InputSectionBase &isec, uint64_t offsetInSec, Symbol &sym,
1501 int64_t addend, std::optional<RelType> addendRelType) {
1502 addReloc(isAgainstSymbol: true, dynType, sec&: isec, offsetInSec, sym, addend, expr: R_ADDEND,
1503 addendRelType: addendRelType ? *addendRelType : ctx.target->noneRel);
1504}
1505
1506void RelocationBaseSection::addAddendOnlyRelocIfNonPreemptible(
1507 RelType dynType, InputSectionBase &isec, uint64_t offsetInSec, Symbol &sym,
1508 RelType addendRelType) {
1509 // No need to write an addend to the section for preemptible symbols.
1510 if (sym.isPreemptible)
1511 addReloc(reloc: {dynType, &isec, offsetInSec, true, sym, 0, R_ADDEND});
1512 else
1513 addReloc(isAgainstSymbol: false, dynType, sec&: isec, offsetInSec, sym, addend: 0, expr: R_ABS, addendRelType);
1514}
1515
1516void RelocationBaseSection::mergeRels() {
1517 size_t newSize = relativeRelocs.size();
1518 for (const auto &v : relocsVec)
1519 newSize += v.size();
1520 relativeRelocs.reserve(N: newSize);
1521 // Classify relocsVec entries into relativeRelocs or relocs. Note that
1522 // relocsVec may contain non-relative entries (e.g. R_AARCH64_AUTH_RELATIVE)
1523 // so we must check the type.
1524 for (const auto &v : relocsVec)
1525 for (const DynamicReloc &r : v)
1526 addReloc(reloc: r);
1527 relocsVec.clear();
1528}
1529
1530void RelocationBaseSection::finalizeContents() {
1531 mergeRels();
1532 // Cache the count for DT_RELACOUNT. DynamicSection<ELFT>::computeContents
1533 // uses ctx.arg.zCombreloc (not the per-section combreloc) to decide whether
1534 // to emit DT_RELACOUNT, so this must match.
1535 if (combreloc)
1536 numRelativeRelocs = relativeRelocs.size();
1537 SymbolTableBaseSection *symTab = ctx.in.dynSymTab.get();
1538
1539 // When linking glibc statically, .rel{,a}.plt contains R_*_IRELATIVE
1540 // relocations due to IFUNC (e.g. strcpy). sh_link will be set to 0 in that
1541 // case.
1542 if (symTab && symTab->getParent())
1543 getParent()->link = symTab->getParent()->sectionIndex;
1544 else
1545 getParent()->link = 0;
1546
1547 if (ctx.in.relaPlt.get() == this) {
1548 InputSection *sec = ctx.target->usesGotPlt
1549 ? static_cast<InputSection *>(ctx.in.gotPlt.get())
1550 : static_cast<InputSection *>(ctx.in.plt.get());
1551 if (sec->getParent()) {
1552 getParent()->flags |= ELF::SHF_INFO_LINK;
1553 getParent()->info = sec->getParent()->sectionIndex;
1554 }
1555 }
1556}
1557
1558void DynamicReloc::finalize(Ctx &ctx, SymbolTableBaseSection *symt) {
1559 r_offset = getOffset();
1560 r_sym = getSymIndex(symTab: symt);
1561 addend = computeAddend(ctx);
1562 isFinal = true; // Catch errors
1563}
1564
1565void RelocationBaseSection::computeRels() {
1566 SymbolTableBaseSection *symTab = ctx.in.dynSymTab.get();
1567 parallelForEach(R&: relativeRelocs, Fn: [&ctx = ctx, symTab](DynamicReloc &rel) {
1568 rel.finalize(ctx, symt: symTab);
1569 });
1570 parallelForEach(R&: relocs, Fn: [&ctx = ctx, symTab](DynamicReloc &rel) {
1571 rel.finalize(ctx, symt: symTab);
1572 });
1573
1574 // Place IRELATIVE relocations last so that other dynamic relocations are
1575 // applied before IFUNC resolvers run.
1576 auto irelative = std::stable_partition(
1577 first: relocs.begin(), last: relocs.end(),
1578 pred: [t = ctx.target->iRelativeRel](auto &r) { return r.type != t; });
1579
1580 // Sort by (!IsRelative,SymIndex,r_offset). DT_REL[A]COUNT requires us to
1581 // place R_*_RELATIVE first. SymIndex is to improve locality, while r_offset
1582 // is to make results easier to read.
1583 parallelSort(Start: relativeRelocs.begin(), End: relativeRelocs.end(),
1584 Comp: [](auto &a, auto &b) { return a.r_offset < b.r_offset; });
1585 // Non-relative relocations are few, so don't bother with parallelSort.
1586 if (combreloc)
1587 llvm::sort(Start: relocs.begin(), End: irelative, Comp: [](auto &a, auto &b) {
1588 return std::tie(a.r_sym, a.r_offset) < std::tie(b.r_sym, b.r_offset);
1589 });
1590}
1591
1592template <class ELFT>
1593RelocationSection<ELFT>::RelocationSection(Ctx &ctx, StringRef name,
1594 bool combreloc, unsigned concurrency)
1595 : RelocationBaseSection(ctx, name, ctx.arg.isRela ? SHT_RELA : SHT_REL,
1596 ctx.arg.isRela ? DT_RELA : DT_REL,
1597 ctx.arg.isRela ? DT_RELASZ : DT_RELSZ, combreloc,
1598 concurrency) {
1599 this->entsize = ctx.arg.isRela ? sizeof(Elf_Rela) : sizeof(Elf_Rel);
1600}
1601
1602template <class ELFT> void RelocationSection<ELFT>::writeTo(uint8_t *buf) {
1603 computeRels();
1604 // Write relative relocations first for DT_REL[A]COUNT.
1605 for (const DynamicReloc &rel :
1606 llvm::concat<const DynamicReloc>(relativeRelocs, relocs)) {
1607 auto *p = reinterpret_cast<Elf_Rela *>(buf);
1608 p->r_offset = rel.r_offset;
1609 p->setSymbolAndType(rel.r_sym, rel.type, ctx.arg.isMips64EL);
1610 if (ctx.arg.isRela)
1611 p->r_addend = rel.addend;
1612 buf += ctx.arg.isRela ? sizeof(Elf_Rela) : sizeof(Elf_Rel);
1613 }
1614}
1615
1616RelrBaseSection::RelrBaseSection(Ctx &ctx, unsigned concurrency,
1617 bool isAArch64Auth)
1618 : SyntheticSection(
1619 ctx, isAArch64Auth ? ".relr.auth.dyn" : ".relr.dyn",
1620 isAArch64Auth
1621 ? SHT_AARCH64_AUTH_RELR
1622 : (ctx.arg.useAndroidRelrTags ? SHT_ANDROID_RELR : SHT_RELR),
1623 SHF_ALLOC, ctx.arg.wordsize),
1624 relocsVec(concurrency) {}
1625
1626void RelrBaseSection::mergeRels() {
1627 size_t newSize = relocs.size();
1628 for (const auto &v : relocsVec)
1629 newSize += v.size();
1630 relocs.reserve(N: newSize);
1631 for (const auto &v : relocsVec)
1632 llvm::append_range(C&: relocs, R: v);
1633 relocsVec.clear();
1634}
1635
1636void RelrBaseSection::finalizeContents() { mergeRels(); }
1637
1638template <class ELFT>
1639AndroidPackedRelocationSection<ELFT>::AndroidPackedRelocationSection(
1640 Ctx &ctx, StringRef name, unsigned concurrency)
1641 : RelocationBaseSection(
1642 ctx, name, ctx.arg.isRela ? SHT_ANDROID_RELA : SHT_ANDROID_REL,
1643 ctx.arg.isRela ? DT_ANDROID_RELA : DT_ANDROID_REL,
1644 ctx.arg.isRela ? DT_ANDROID_RELASZ : DT_ANDROID_RELSZ,
1645 /*combreloc=*/false, concurrency) {
1646 this->entsize = 1;
1647}
1648
1649template <class ELFT>
1650bool AndroidPackedRelocationSection<ELFT>::updateAllocSize(Ctx &ctx) {
1651 // This function computes the contents of an Android-format packed relocation
1652 // section.
1653 //
1654 // This format compresses relocations by using relocation groups to factor out
1655 // fields that are common between relocations and storing deltas from previous
1656 // relocations in SLEB128 format (which has a short representation for small
1657 // numbers). A good example of a relocation type with common fields is
1658 // R_*_RELATIVE, which is normally used to represent function pointers in
1659 // vtables. In the REL format, each relative relocation has the same r_info
1660 // field, and is only different from other relative relocations in terms of
1661 // the r_offset field. By sorting relocations by offset, grouping them by
1662 // r_info and representing each relocation with only the delta from the
1663 // previous offset, each 8-byte relocation can be compressed to as little as 1
1664 // byte (or less with run-length encoding). This relocation packer was able to
1665 // reduce the size of the relocation section in an Android Chromium DSO from
1666 // 2,911,184 bytes to 174,693 bytes, or 6% of the original size.
1667 //
1668 // A relocation section consists of a header containing the literal bytes
1669 // 'APS2' followed by a sequence of SLEB128-encoded integers. The first two
1670 // elements are the total number of relocations in the section and an initial
1671 // r_offset value. The remaining elements define a sequence of relocation
1672 // groups. Each relocation group starts with a header consisting of the
1673 // following elements:
1674 //
1675 // - the number of relocations in the relocation group
1676 // - flags for the relocation group
1677 // - (if RELOCATION_GROUPED_BY_OFFSET_DELTA_FLAG is set) the r_offset delta
1678 // for each relocation in the group.
1679 // - (if RELOCATION_GROUPED_BY_INFO_FLAG is set) the value of the r_info
1680 // field for each relocation in the group.
1681 // - (if RELOCATION_GROUP_HAS_ADDEND_FLAG and
1682 // RELOCATION_GROUPED_BY_ADDEND_FLAG are set) the r_addend delta for
1683 // each relocation in the group.
1684 //
1685 // Following the relocation group header are descriptions of each of the
1686 // relocations in the group. They consist of the following elements:
1687 //
1688 // - (if RELOCATION_GROUPED_BY_OFFSET_DELTA_FLAG is not set) the r_offset
1689 // delta for this relocation.
1690 // - (if RELOCATION_GROUPED_BY_INFO_FLAG is not set) the value of the r_info
1691 // field for this relocation.
1692 // - (if RELOCATION_GROUP_HAS_ADDEND_FLAG is set and
1693 // RELOCATION_GROUPED_BY_ADDEND_FLAG is not set) the r_addend delta for
1694 // this relocation.
1695
1696 size_t oldSize = relocData.size();
1697
1698 relocData = {'A', 'P', 'S', '2'};
1699 raw_svector_ostream os(relocData);
1700 auto add = [&](int64_t v) { encodeSLEB128(Value: v, OS&: os); };
1701
1702 // The format header includes the number of relocations and the initial
1703 // offset (we set this to zero because the first relocation group will
1704 // perform the initial adjustment).
1705 add(relativeRelocs.size() + relocs.size());
1706 add(0);
1707
1708 SymbolTableBaseSection *symTab = ctx.in.dynSymTab.get();
1709 auto makeRela = [&](const DynamicReloc &rel) {
1710 Elf_Rela r;
1711 r.r_offset = rel.getOffset();
1712 r.setSymbolAndType(rel.getSymIndex(symTab), rel.type, false);
1713 r.r_addend = ctx.arg.isRela ? rel.computeAddend(ctx) : 0;
1714 return r;
1715 };
1716 std::vector<Elf_Rela> relatives, nonRelatives;
1717 for (const DynamicReloc &rel : relativeRelocs)
1718 relatives.push_back(makeRela(rel));
1719 for (const DynamicReloc &rel : relocs)
1720 nonRelatives.push_back(makeRela(rel));
1721
1722 llvm::sort(relatives, [](const Elf_Rel &a, const Elf_Rel &b) {
1723 return a.r_offset < b.r_offset;
1724 });
1725
1726 // Try to find groups of relative relocations which are spaced one word
1727 // apart from one another. These generally correspond to vtable entries. The
1728 // format allows these groups to be encoded using a sort of run-length
1729 // encoding, but each group will cost 7 bytes in addition to the offset from
1730 // the previous group, so it is only profitable to do this for groups of
1731 // size 8 or larger.
1732 std::vector<Elf_Rela> ungroupedRelatives;
1733 std::vector<std::vector<Elf_Rela>> relativeGroups;
1734 for (auto i = relatives.begin(), e = relatives.end(); i != e;) {
1735 std::vector<Elf_Rela> group;
1736 do {
1737 group.push_back(*i++);
1738 } while (i != e && (i - 1)->r_offset + ctx.arg.wordsize == i->r_offset);
1739
1740 if (group.size() < 8)
1741 ungroupedRelatives.insert(ungroupedRelatives.end(), group.begin(),
1742 group.end());
1743 else
1744 relativeGroups.emplace_back(std::move(group));
1745 }
1746
1747 // For non-relative relocations, we would like to:
1748 // 1. Have relocations with the same symbol offset to be consecutive, so
1749 // that the runtime linker can speed-up symbol lookup by implementing an
1750 // 1-entry cache.
1751 // 2. Group relocations by r_info to reduce the size of the relocation
1752 // section.
1753 // Since the symbol offset is the high bits in r_info, sorting by r_info
1754 // allows us to do both.
1755 //
1756 // For Rela, we also want to sort by r_addend when r_info is the same. This
1757 // enables us to group by r_addend as well.
1758 llvm::sort(nonRelatives, [](const Elf_Rela &a, const Elf_Rela &b) {
1759 return std::tie(a.r_info, a.r_addend, a.r_offset) <
1760 std::tie(b.r_info, b.r_addend, b.r_offset);
1761 });
1762
1763 // Group relocations with the same r_info. Note that each group emits a group
1764 // header and that may make the relocation section larger. It is hard to
1765 // estimate the size of a group header as the encoded size of that varies
1766 // based on r_info. However, we can approximate this trade-off by the number
1767 // of values encoded. Each group header contains 3 values, and each relocation
1768 // in a group encodes one less value, as compared to when it is not grouped.
1769 // Therefore, we only group relocations if there are 3 or more of them with
1770 // the same r_info.
1771 //
1772 // For Rela, the addend for most non-relative relocations is zero, and thus we
1773 // can usually get a smaller relocation section if we group relocations with 0
1774 // addend as well.
1775 std::vector<Elf_Rela> ungroupedNonRelatives;
1776 std::vector<std::vector<Elf_Rela>> nonRelativeGroups;
1777 for (auto i = nonRelatives.begin(), e = nonRelatives.end(); i != e;) {
1778 auto j = i + 1;
1779 while (j != e && i->r_info == j->r_info &&
1780 (!ctx.arg.isRela || i->r_addend == j->r_addend))
1781 ++j;
1782 if (j - i < 3 || (ctx.arg.isRela && i->r_addend != 0))
1783 ungroupedNonRelatives.insert(ungroupedNonRelatives.end(), i, j);
1784 else
1785 nonRelativeGroups.emplace_back(i, j);
1786 i = j;
1787 }
1788
1789 // Sort ungrouped relocations by offset to minimize the encoded length.
1790 llvm::sort(ungroupedNonRelatives, [](const Elf_Rela &a, const Elf_Rela &b) {
1791 return a.r_offset < b.r_offset;
1792 });
1793
1794 unsigned hasAddendIfRela =
1795 ctx.arg.isRela ? RELOCATION_GROUP_HAS_ADDEND_FLAG : 0;
1796
1797 uint64_t offset = 0;
1798 uint64_t addend = 0;
1799
1800 // Emit the run-length encoding for the groups of adjacent relative
1801 // relocations. Each group is represented using two groups in the packed
1802 // format. The first is used to set the current offset to the start of the
1803 // group (and also encodes the first relocation), and the second encodes the
1804 // remaining relocations.
1805 for (std::vector<Elf_Rela> &g : relativeGroups) {
1806 // The first relocation in the group.
1807 add(1);
1808 add(RELOCATION_GROUPED_BY_OFFSET_DELTA_FLAG |
1809 RELOCATION_GROUPED_BY_INFO_FLAG | hasAddendIfRela);
1810 add(g[0].r_offset - offset);
1811 add(ctx.target->relativeRel);
1812 if (ctx.arg.isRela) {
1813 add(g[0].r_addend - addend);
1814 addend = g[0].r_addend;
1815 }
1816
1817 // The remaining relocations.
1818 add(g.size() - 1);
1819 add(RELOCATION_GROUPED_BY_OFFSET_DELTA_FLAG |
1820 RELOCATION_GROUPED_BY_INFO_FLAG | hasAddendIfRela);
1821 add(ctx.arg.wordsize);
1822 add(ctx.target->relativeRel);
1823 if (ctx.arg.isRela) {
1824 for (const auto &i : llvm::drop_begin(g)) {
1825 add(i.r_addend - addend);
1826 addend = i.r_addend;
1827 }
1828 }
1829
1830 offset = g.back().r_offset;
1831 }
1832
1833 // Now the ungrouped relatives.
1834 if (!ungroupedRelatives.empty()) {
1835 add(ungroupedRelatives.size());
1836 add(RELOCATION_GROUPED_BY_INFO_FLAG | hasAddendIfRela);
1837 add(ctx.target->relativeRel);
1838 for (Elf_Rela &r : ungroupedRelatives) {
1839 add(r.r_offset - offset);
1840 offset = r.r_offset;
1841 if (ctx.arg.isRela) {
1842 add(r.r_addend - addend);
1843 addend = r.r_addend;
1844 }
1845 }
1846 }
1847
1848 // Grouped non-relatives.
1849 for (ArrayRef<Elf_Rela> g : nonRelativeGroups) {
1850 add(g.size());
1851 add(RELOCATION_GROUPED_BY_INFO_FLAG);
1852 add(g[0].r_info);
1853 for (const Elf_Rela &r : g) {
1854 add(r.r_offset - offset);
1855 offset = r.r_offset;
1856 }
1857 addend = 0;
1858 }
1859
1860 // Finally the ungrouped non-relative relocations.
1861 if (!ungroupedNonRelatives.empty()) {
1862 add(ungroupedNonRelatives.size());
1863 add(hasAddendIfRela);
1864 for (Elf_Rela &r : ungroupedNonRelatives) {
1865 add(r.r_offset - offset);
1866 offset = r.r_offset;
1867 add(r.r_info);
1868 if (ctx.arg.isRela) {
1869 add(r.r_addend - addend);
1870 addend = r.r_addend;
1871 }
1872 }
1873 }
1874
1875 // Don't allow the section to shrink; otherwise the size of the section can
1876 // oscillate infinitely.
1877 if (relocData.size() < oldSize)
1878 relocData.append(NumInputs: oldSize - relocData.size(), Elt: 0);
1879
1880 // Returns whether the section size changed. We need to keep recomputing both
1881 // section layout and the contents of this section until the size converges
1882 // because changing this section's size can affect section layout, which in
1883 // turn can affect the sizes of the LEB-encoded integers stored in this
1884 // section.
1885 return relocData.size() != oldSize;
1886}
1887
1888template <class ELFT>
1889RelrSection<ELFT>::RelrSection(Ctx &ctx, unsigned concurrency,
1890 bool isAArch64Auth)
1891 : RelrBaseSection(ctx, concurrency, isAArch64Auth) {
1892 this->entsize = ctx.arg.wordsize;
1893}
1894
1895template <class ELFT> bool RelrSection<ELFT>::updateAllocSize(Ctx &ctx) {
1896 // This function computes the contents of an SHT_RELR packed relocation
1897 // section.
1898 //
1899 // Proposal for adding SHT_RELR sections to generic-abi is here:
1900 // https://groups.google.com/forum/#!topic/generic-abi/bX460iggiKg
1901 //
1902 // The encoded sequence of Elf64_Relr entries in a SHT_RELR section looks
1903 // like [ AAAAAAAA BBBBBBB1 BBBBBBB1 ... AAAAAAAA BBBBBB1 ... ]
1904 //
1905 // i.e. start with an address, followed by any number of bitmaps. The address
1906 // entry encodes 1 relocation. The subsequent bitmap entries encode up to 63
1907 // relocations each, at subsequent offsets following the last address entry.
1908 //
1909 // The bitmap entries must have 1 in the least significant bit. The assumption
1910 // here is that an address cannot have 1 in lsb. Odd addresses are not
1911 // supported.
1912 //
1913 // Excluding the least significant bit in the bitmap, each non-zero bit in
1914 // the bitmap represents a relocation to be applied to a corresponding machine
1915 // word that follows the base address word. The second least significant bit
1916 // represents the machine word immediately following the initial address, and
1917 // each bit that follows represents the next word, in linear order. As such,
1918 // a single bitmap can encode up to 31 relocations in a 32-bit object, and
1919 // 63 relocations in a 64-bit object.
1920 //
1921 // This encoding has a couple of interesting properties:
1922 // 1. Looking at any entry, it is clear whether it's an address or a bitmap:
1923 // even means address, odd means bitmap.
1924 // 2. Just a simple list of addresses is a valid encoding.
1925
1926 size_t oldSize = relrRelocs.size();
1927 relrRelocs.clear();
1928
1929 const size_t wordsize = sizeof(typename ELFT::uint);
1930
1931 // Number of bits to use for the relocation offsets bitmap.
1932 // Must be either 63 or 31.
1933 const size_t nBits = wordsize * 8 - 1;
1934
1935 // Get offsets for all relative relocations and sort them.
1936 std::unique_ptr<uint64_t[]> offsets(new uint64_t[relocs.size()]);
1937 for (auto [i, r] : llvm::enumerate(relocs))
1938 offsets[i] = r.getOffset();
1939 llvm::sort(offsets.get(), offsets.get() + relocs.size());
1940
1941 // For each leading relocation, find following ones that can be folded
1942 // as a bitmap and fold them.
1943 for (size_t i = 0, e = relocs.size(); i != e;) {
1944 // Add a leading relocation.
1945 relrRelocs.push_back(Elf_Relr(offsets[i]));
1946 uint64_t base = offsets[i] + wordsize;
1947 ++i;
1948
1949 // Find foldable relocations to construct bitmaps.
1950 for (;;) {
1951 uint64_t bitmap = 0;
1952 for (; i != e; ++i) {
1953 uint64_t d = offsets[i] - base;
1954 if (d >= nBits * wordsize || d % wordsize)
1955 break;
1956 bitmap |= uint64_t(1) << (d / wordsize);
1957 }
1958 if (!bitmap)
1959 break;
1960 relrRelocs.push_back(Elf_Relr((bitmap << 1) | 1));
1961 base += nBits * wordsize;
1962 }
1963 }
1964
1965 // Don't allow the section to shrink; otherwise the size of the section can
1966 // oscillate infinitely. Trailing 1s do not decode to more relocations.
1967 if (relrRelocs.size() < oldSize) {
1968 Log(ctx) << ".relr.dyn needs " << (oldSize - relrRelocs.size())
1969 << " padding word(s)";
1970 relrRelocs.resize(oldSize, Elf_Relr(1));
1971 }
1972
1973 return relrRelocs.size() != oldSize;
1974}
1975
1976SymbolTableBaseSection::SymbolTableBaseSection(Ctx &ctx,
1977 StringTableSection &strTabSec)
1978 : SyntheticSection(ctx, strTabSec.isDynamic() ? ".dynsym" : ".symtab",
1979 strTabSec.isDynamic() ? SHT_DYNSYM : SHT_SYMTAB,
1980 strTabSec.isDynamic() ? (uint64_t)SHF_ALLOC : 0,
1981 ctx.arg.wordsize),
1982 strTabSec(strTabSec) {}
1983
1984// Orders symbols according to their positions in the GOT,
1985// in compliance with MIPS ABI rules.
1986// See "Global Offset Table" in Chapter 5 in the following document
1987// for detailed description:
1988// ftp://www.linux-mips.org/pub/linux/mips/doc/ABI/mipsabi.pdf
1989static void sortMipsSymbols(Ctx &ctx, SmallVector<SymbolTableEntry, 0> &syms) {
1990 llvm::stable_sort(Range&: syms,
1991 C: [&](const SymbolTableEntry &l, const SymbolTableEntry &r) {
1992 // Sort entries related to non-local preemptible symbols
1993 // by GOT indexes. All other entries go to the beginning
1994 // of a dynsym in arbitrary order.
1995 if (l.sym->isInGot(ctx) && r.sym->isInGot(ctx))
1996 return l.sym->getGotIdx(ctx) < r.sym->getGotIdx(ctx);
1997 if (!l.sym->isInGot(ctx) && !r.sym->isInGot(ctx))
1998 return false;
1999 return !l.sym->isInGot(ctx);
2000 });
2001}
2002
2003void SymbolTableBaseSection::finalizeContents() {
2004 if (OutputSection *sec = strTabSec.getParent())
2005 getParent()->link = sec->sectionIndex;
2006
2007 if (this->type != SHT_DYNSYM) {
2008 sortSymTabSymbols();
2009 return;
2010 }
2011
2012 // If it is a .dynsym, there should be no local symbols, but we need
2013 // to do a few things for the dynamic linker.
2014
2015 // Section's Info field has the index of the first non-local symbol.
2016 // Because the first symbol entry is a null entry, 1 is the first.
2017 getParent()->info = 1;
2018
2019 if (ctx.in.gnuHashTab) {
2020 // NB: It also sorts Symbols to meet the GNU hash table requirements.
2021 ctx.in.gnuHashTab->addSymbols(symbols);
2022 } else if (ctx.arg.emachine == EM_MIPS) {
2023 sortMipsSymbols(ctx, syms&: symbols);
2024 }
2025
2026 // The dynamic symbol table records each symbol's index in the symbol itself.
2027 // The static .symtab cannot (the slot is taken) and instead uses a lazy
2028 // lookup table; see getSymbolIndex.
2029 size_t i = 0;
2030 for (const SymbolTableEntry &s : symbols)
2031 s.sym->dynsymIndex = ++i;
2032}
2033
2034// The ELF spec requires local symbols to precede globals. We additionally group
2035// the locals by file, each led by its first STT_FILE.
2036//
2037// From firstGlobalIdx on, a local cannot be attributed to a file (a demoted
2038// global, or a thunk/errata patch added later). Move these after the per-file
2039// groups, behind the synthetic STT_FILE synthSttFileSym.
2040void SymbolTableBaseSection::sortSymTabSymbols() {
2041 MapVector<InputFile *, SmallVector<SymbolTableEntry, 0>> fileToLocals;
2042 SmallVector<SymbolTableEntry, 0> localized, globals;
2043 SymbolTableEntry fileEntry{};
2044 for (size_t i = 0, e = symbols.size(); i != e; ++i) {
2045 const SymbolTableEntry &s = symbols[i];
2046 if (!s.sym->isLocal())
2047 globals.push_back(Elt: s);
2048 else if (s.sym == synthSttFileSym)
2049 fileEntry = s;
2050 else if (synthSttFileSym && i >= firstGlobalIdx)
2051 localized.push_back(Elt: s);
2052 else
2053 fileToLocals[s.sym->file].push_back(Elt: s);
2054 }
2055
2056 auto i = symbols.begin();
2057 for (auto &p : fileToLocals)
2058 for (SymbolTableEntry &entry : p.second)
2059 *i++ = entry;
2060 if (synthSttFileSym) {
2061 *i++ = fileEntry;
2062 i = std::copy(first: localized.begin(), last: localized.end(), result: i);
2063 }
2064 getParent()->info = i - symbols.begin() + 1;
2065 std::copy(first: globals.begin(), last: globals.end(), result: i);
2066}
2067
2068// A symbol converted to STB_LOCAL cannot be reliably attributed to a file:
2069// within a file's group the wrong STT_FILE would claim it, as a file may hold
2070// several STT_FILE symbols (relocatable output) or none. Like GNU ld, when the
2071// output has an STT_FILE, add a synthetic empty-name STT_FILE.
2072void SymbolTableBaseSection::maybeAddSttFile() {
2073 ArrayRef<SymbolTableEntry> syms = symbols;
2074 if (llvm::none_of(Range: syms.take_front(N: firstGlobalIdx),
2075 P: [](const SymbolTableEntry &s) { return s.sym->isFile(); }))
2076 return;
2077 if (llvm::any_of(
2078 Range: syms.drop_front(N: firstGlobalIdx),
2079 P: [](const SymbolTableEntry &s) { return s.sym->isLocal(); })) {
2080 synthSttFileSym =
2081 makeDefined(args&: ctx, args&: ctx.internalFile, args: "", args: STB_LOCAL, args: STV_DEFAULT, args: STT_FILE,
2082 /*value=*/args: 0, /*size=*/args: 0, args: nullptr);
2083 addSymbol(sym: synthSttFileSym);
2084 }
2085}
2086
2087void SymbolTableBaseSection::addSymbol(Symbol *b) {
2088 // Adding a local symbol to a .dynsym is a bug.
2089 assert(this->type != SHT_DYNSYM || !b->isLocal());
2090 symbols.push_back(Elt: {.sym: b, .strTabOffset: strTabSec.addString(s: b->getName(), hashIt: false)});
2091}
2092
2093size_t SymbolTableBaseSection::getSymbolIndex(const Symbol &sym) {
2094 if (this == ctx.in.dynSymTab.get())
2095 return sym.dynsymIndex;
2096
2097 // Initialize the symbol lookup table lazily. This is used for the static
2098 // symbol table (.symtab), e.g. with -r or --emit-relocs.
2099 llvm::call_once(flag&: onceFlag, F: [&] {
2100 symbolIndexMap.reserve(NumEntries: symbols.size());
2101 size_t i = 0;
2102 for (const SymbolTableEntry &e : symbols) {
2103 if (e.sym->type == STT_SECTION)
2104 sectionIndexMap[e.sym->getOutputSection()] = ++i;
2105 else
2106 symbolIndexMap[e.sym] = ++i;
2107 }
2108 });
2109
2110 // Section symbols are mapped based on their output sections
2111 // to maintain their semantics.
2112 if (sym.type == STT_SECTION)
2113 return sectionIndexMap.lookup(Val: sym.getOutputSection());
2114 return symbolIndexMap.lookup(Val: &sym);
2115}
2116
2117template <class ELFT>
2118SymbolTableSection<ELFT>::SymbolTableSection(Ctx &ctx,
2119 StringTableSection &strTabSec)
2120 : SymbolTableBaseSection(ctx, strTabSec) {
2121 this->entsize = sizeof(Elf_Sym);
2122}
2123
2124static BssSection *getCommonSec(bool relocatable, Symbol *sym) {
2125 if (relocatable)
2126 if (auto *d = dyn_cast<Defined>(Val: sym))
2127 return dyn_cast_or_null<BssSection>(Val: d->section);
2128 return nullptr;
2129}
2130
2131static uint32_t getSymSectionIndex(Symbol *sym) {
2132 assert(!(sym->hasFlag(NEEDS_COPY) && sym->isObject()));
2133 if (!isa<Defined>(Val: sym) || sym->hasFlag(bit: NEEDS_COPY))
2134 return SHN_UNDEF;
2135 if (const OutputSection *os = sym->getOutputSection())
2136 return os->sectionIndex >= SHN_LORESERVE ? (uint32_t)SHN_XINDEX
2137 : os->sectionIndex;
2138 return SHN_ABS;
2139}
2140
2141// Write the internal symbol table contents to the output symbol table.
2142template <class ELFT> void SymbolTableSection<ELFT>::writeTo(uint8_t *buf) {
2143 // The first entry is a null entry as per the ELF spec.
2144 buf += sizeof(Elf_Sym);
2145
2146 auto *eSym = reinterpret_cast<Elf_Sym *>(buf);
2147 bool relocatable = ctx.arg.relocatable;
2148 for (SymbolTableEntry &ent : symbols) {
2149 Symbol *sym = ent.sym;
2150 // Set st_name, st_info and st_other.
2151 eSym->st_name = ent.strTabOffset;
2152 eSym->setBindingAndType(sym->binding, sym->type);
2153 eSym->st_other = sym->stOther;
2154
2155 if (BssSection *commonSec = getCommonSec(relocatable, sym)) {
2156 // When -r is specified, a COMMON symbol is not allocated. Its st_shndx
2157 // holds SHN_COMMON and st_value holds the alignment.
2158 eSym->st_shndx = SHN_COMMON;
2159 eSym->st_value = commonSec->addralign;
2160 eSym->st_size = cast<Defined>(Val: sym)->size;
2161 } else {
2162 const uint32_t shndx = getSymSectionIndex(sym);
2163 eSym->st_shndx = shndx;
2164 eSym->st_value = sym->getVA(ctx);
2165 // Copy symbol size if it is a defined symbol. st_size is not
2166 // significant for undefined symbols, so whether copying it or not is up
2167 // to us if that's the case. We'll leave it as zero because by not
2168 // setting a value, we can get the exact same outputs for two sets of
2169 // input files that differ only in undefined symbol size in DSOs.
2170 eSym->st_size = shndx != SHN_UNDEF ? cast<Defined>(Val: sym)->size : 0;
2171 }
2172
2173 ++eSym;
2174 }
2175
2176 // On MIPS we need to mark symbol which has a PLT entry and requires
2177 // pointer equality by STO_MIPS_PLT flag. That is necessary to help
2178 // dynamic linker distinguish such symbols and MIPS lazy-binding stubs.
2179 // https://sourceware.org/ml/binutils/2008-07/txt00000.txt
2180 if (ctx.arg.emachine == EM_MIPS) {
2181 auto *eSym = reinterpret_cast<Elf_Sym *>(buf);
2182
2183 for (SymbolTableEntry &ent : symbols) {
2184 Symbol *sym = ent.sym;
2185 if (sym->isInPlt(ctx) && sym->hasFlag(bit: NEEDS_COPY))
2186 eSym->st_other |= STO_MIPS_PLT;
2187 if (isMicroMips(ctx)) {
2188 // We already set the less-significant bit for symbols
2189 // marked by the `STO_MIPS_MICROMIPS` flag and for microMIPS PLT
2190 // records. That allows us to distinguish such symbols in
2191 // the `MIPS<ELFT>::relocate()` routine. Now we should
2192 // clear that bit for non-dynamic symbol table, so tools
2193 // like `objdump` will be able to deal with a correct
2194 // symbol position.
2195 if (sym->isDefined() &&
2196 ((sym->stOther & STO_MIPS_MICROMIPS) || sym->hasFlag(bit: NEEDS_COPY))) {
2197 if (!strTabSec.isDynamic())
2198 eSym->st_value &= ~1;
2199 eSym->st_other |= STO_MIPS_MICROMIPS;
2200 }
2201 }
2202 if (ctx.arg.relocatable)
2203 if (auto *d = dyn_cast<Defined>(Val: sym))
2204 if (isMipsPIC<ELFT>(d))
2205 eSym->st_other |= STO_MIPS_PIC;
2206 ++eSym;
2207 }
2208 }
2209}
2210
2211SymtabShndxSection::SymtabShndxSection(Ctx &ctx)
2212 : SyntheticSection(ctx, ".symtab_shndx", SHT_SYMTAB_SHNDX, 0, 4) {
2213 this->entsize = 4;
2214}
2215
2216void SymtabShndxSection::writeTo(uint8_t *buf) {
2217 // We write an array of 32 bit values, where each value has 1:1 association
2218 // with an entry in ctx.in.symTab if the corresponding entry contains
2219 // SHN_XINDEX, we need to write actual index, otherwise, we must write
2220 // SHN_UNDEF(0).
2221 buf += 4; // Ignore .symtab[0] entry.
2222 bool relocatable = ctx.arg.relocatable;
2223 for (const SymbolTableEntry &entry : ctx.in.symTab->getSymbols()) {
2224 if (!getCommonSec(relocatable, sym: entry.sym) &&
2225 getSymSectionIndex(sym: entry.sym) == SHN_XINDEX)
2226 write32(ctx, p: buf, v: entry.sym->getOutputSection()->sectionIndex);
2227 buf += 4;
2228 }
2229}
2230
2231bool SymtabShndxSection::isNeeded() const {
2232 // SHT_SYMTAB can hold symbols with section indices values up to
2233 // SHN_LORESERVE. If we need more, we want to use extension SHT_SYMTAB_SHNDX
2234 // section. Problem is that we reveal the final section indices a bit too
2235 // late, and we do not know them here. For simplicity, we just always create
2236 // a .symtab_shndx section when the amount of output sections is huge.
2237 size_t size = 0;
2238 for (SectionCommand *cmd : ctx.script->sectionCommands)
2239 if (isa<OutputDesc>(Val: cmd))
2240 ++size;
2241 return size >= SHN_LORESERVE;
2242}
2243
2244void SymtabShndxSection::finalizeContents() {
2245 getParent()->link = ctx.in.symTab->getParent()->sectionIndex;
2246}
2247
2248size_t SymtabShndxSection::getSize() const {
2249 return ctx.in.symTab->getNumSymbols() * 4;
2250}
2251
2252// .hash and .gnu.hash sections contain on-disk hash tables that map
2253// symbol names to their dynamic symbol table indices. Their purpose
2254// is to help the dynamic linker resolve symbols quickly. If ELF files
2255// don't have them, the dynamic linker has to do linear search on all
2256// dynamic symbols, which makes programs slower. Therefore, a .hash
2257// section is added to a DSO by default.
2258//
2259// The Unix semantics of resolving dynamic symbols is somewhat expensive.
2260// Each ELF file has a list of DSOs that the ELF file depends on and a
2261// list of dynamic symbols that need to be resolved from any of the
2262// DSOs. That means resolving all dynamic symbols takes O(m)*O(n)
2263// where m is the number of DSOs and n is the number of dynamic
2264// symbols. For modern large programs, both m and n are large. So
2265// making each step faster by using hash tables substantially
2266// improves time to load programs.
2267//
2268// (Note that this is not the only way to design the shared library.
2269// For instance, the Windows DLL takes a different approach. On
2270// Windows, each dynamic symbol has a name of DLL from which the symbol
2271// has to be resolved. That makes the cost of symbol resolution O(n).
2272// This disables some hacky techniques you can use on Unix such as
2273// LD_PRELOAD, but this is arguably better semantics than the Unix ones.)
2274//
2275// Due to historical reasons, we have two different hash tables, .hash
2276// and .gnu.hash. They are for the same purpose, and .gnu.hash is a new
2277// and better version of .hash. .hash is just an on-disk hash table, but
2278// .gnu.hash has a bloom filter in addition to a hash table to skip
2279// DSOs very quickly. If you are sure that your dynamic linker knows
2280// about .gnu.hash, you want to specify --hash-style=gnu. Otherwise, a
2281// safe bet is to specify --hash-style=both for backward compatibility.
2282GnuHashTableSection::GnuHashTableSection(Ctx &ctx)
2283 : SyntheticSection(ctx, ".gnu.hash", SHT_GNU_HASH, SHF_ALLOC,
2284 ctx.arg.wordsize) {}
2285
2286void GnuHashTableSection::finalizeContents() {
2287 if (OutputSection *sec = ctx.in.dynSymTab->getParent())
2288 getParent()->link = sec->sectionIndex;
2289
2290 // Computes bloom filter size in word size. We want to allocate 12
2291 // bits for each symbol. It must be a power of two.
2292 if (symbols.empty()) {
2293 maskWords = 1;
2294 } else {
2295 uint64_t numBits = symbols.size() * 12;
2296 maskWords = NextPowerOf2(A: numBits / (ctx.arg.wordsize * 8));
2297 }
2298
2299 size = 16; // Header
2300 size += ctx.arg.wordsize * maskWords; // Bloom filter
2301 size += nBuckets * 4; // Hash buckets
2302 size += symbols.size() * 4; // Hash values
2303}
2304
2305void GnuHashTableSection::writeTo(uint8_t *buf) {
2306 // Write a header.
2307 write32(ctx, p: buf, v: nBuckets);
2308 write32(ctx, p: buf + 4, v: ctx.in.dynSymTab->getNumSymbols() - symbols.size());
2309 write32(ctx, p: buf + 8, v: maskWords);
2310 write32(ctx, p: buf + 12, v: Shift2);
2311 buf += 16;
2312
2313 // Write the 2-bit bloom filter.
2314 const unsigned c = ctx.arg.is64 ? 64 : 32;
2315 for (const Entry &sym : symbols) {
2316 // When C = 64, we choose a word with bits [6:...] and set 1 to two bits in
2317 // the word using bits [0:5] and [26:31].
2318 size_t i = (sym.hash / c) & (maskWords - 1);
2319 uint64_t val = readUint(ctx, buf: buf + i * ctx.arg.wordsize);
2320 val |= uint64_t(1) << (sym.hash % c);
2321 val |= uint64_t(1) << ((sym.hash >> Shift2) % c);
2322 writeUint(ctx, buf: buf + i * ctx.arg.wordsize, val);
2323 }
2324 buf += ctx.arg.wordsize * maskWords;
2325
2326 // Write the hash table.
2327 uint32_t *buckets = reinterpret_cast<uint32_t *>(buf);
2328 uint32_t oldBucket = -1;
2329 uint32_t *values = buckets + nBuckets;
2330 for (auto i = symbols.begin(), e = symbols.end(); i != e; ++i) {
2331 // Write a hash value. It represents a sequence of chains that share the
2332 // same hash modulo value. The last element of each chain is terminated by
2333 // LSB 1.
2334 uint32_t hash = i->hash;
2335 bool isLastInChain = (i + 1) == e || i->bucketIdx != (i + 1)->bucketIdx;
2336 hash = isLastInChain ? hash | 1 : hash & ~1;
2337 write32(ctx, p: values++, v: hash);
2338
2339 if (i->bucketIdx == oldBucket)
2340 continue;
2341 // Write a hash bucket. Hash buckets contain indices in the following hash
2342 // value table.
2343 write32(ctx, p: buckets + i->bucketIdx,
2344 v: ctx.in.dynSymTab->getSymbolIndex(sym: *i->sym));
2345 oldBucket = i->bucketIdx;
2346 }
2347}
2348
2349// Add symbols to this symbol hash table. Note that this function
2350// destructively sort a given vector -- which is needed because
2351// GNU-style hash table places some sorting requirements.
2352void GnuHashTableSection::addSymbols(SmallVectorImpl<SymbolTableEntry> &v) {
2353 // We cannot use 'auto' for Mid because GCC 6.1 cannot deduce
2354 // its type correctly.
2355 auto mid =
2356 std::stable_partition(first: v.begin(), last: v.end(), pred: [&](const SymbolTableEntry &s) {
2357 return !s.sym->isDefined();
2358 });
2359
2360 // We chose load factor 4 for the on-disk hash table. For each hash
2361 // collision, the dynamic linker will compare a uint32_t hash value.
2362 // Since the integer comparison is quite fast, we believe we can
2363 // make the load factor even larger. 4 is just a conservative choice.
2364 //
2365 // Note that we don't want to create a zero-sized hash table because
2366 // Android loader as of 2018 doesn't like a .gnu.hash containing such
2367 // table. If that's the case, we create a hash table with one unused
2368 // dummy slot.
2369 nBuckets = std::max<size_t>(a: (v.end() - mid) / 4, b: 1);
2370
2371 if (mid == v.end())
2372 return;
2373
2374 for (SymbolTableEntry &ent : llvm::make_range(x: mid, y: v.end())) {
2375 Symbol *b = ent.sym;
2376 uint32_t hash = hashGnu(Name: b->getName());
2377 uint32_t bucketIdx = hash % nBuckets;
2378 symbols.push_back(Elt: {.sym: b, .strTabOffset: ent.strTabOffset, .hash: hash, .bucketIdx: bucketIdx});
2379 }
2380
2381 llvm::sort(C&: symbols, Comp: [](const Entry &l, const Entry &r) {
2382 return std::tie(args: l.bucketIdx, args: l.strTabOffset) <
2383 std::tie(args: r.bucketIdx, args: r.strTabOffset);
2384 });
2385
2386 v.erase(CS: mid, CE: v.end());
2387 for (const Entry &ent : symbols)
2388 v.push_back(Elt: {.sym: ent.sym, .strTabOffset: ent.strTabOffset});
2389}
2390
2391HashTableSection::HashTableSection(Ctx &ctx)
2392 : SyntheticSection(ctx, ".hash", SHT_HASH, SHF_ALLOC, 4) {
2393 this->entsize = 4;
2394}
2395
2396void HashTableSection::finalizeContents() {
2397 SymbolTableBaseSection *symTab = ctx.in.dynSymTab.get();
2398
2399 if (OutputSection *sec = symTab->getParent())
2400 getParent()->link = sec->sectionIndex;
2401
2402 unsigned numEntries = 2; // nbucket and nchain.
2403 numEntries += symTab->getNumSymbols(); // The chain entries.
2404
2405 // Create as many buckets as there are symbols.
2406 numEntries += symTab->getNumSymbols();
2407 this->size = numEntries * 4;
2408}
2409
2410void HashTableSection::writeTo(uint8_t *buf) {
2411 SymbolTableBaseSection *symTab = ctx.in.dynSymTab.get();
2412 unsigned numSymbols = symTab->getNumSymbols();
2413
2414 uint32_t *p = reinterpret_cast<uint32_t *>(buf);
2415 write32(ctx, p: p++, v: numSymbols); // nbucket
2416 write32(ctx, p: p++, v: numSymbols); // nchain
2417
2418 uint32_t *buckets = p;
2419 uint32_t *chains = p + numSymbols;
2420
2421 for (const SymbolTableEntry &s : symTab->getSymbols()) {
2422 Symbol *sym = s.sym;
2423 StringRef name = sym->getName();
2424 unsigned i = sym->dynsymIndex;
2425 uint32_t hash = hashSysV(SymbolName: name) % numSymbols;
2426 chains[i] = buckets[hash];
2427 write32(ctx, p: buckets + hash, v: i);
2428 }
2429}
2430
2431PltSection::PltSection(Ctx &ctx)
2432 : SyntheticSection(ctx, ".plt", SHT_PROGBITS, SHF_ALLOC | SHF_EXECINSTR,
2433 16),
2434 headerSize(ctx.target->pltHeaderSize) {
2435 // On AArch64, PLT entries only do loads from the .got.plt section, so the
2436 // .plt section can be marked with the SHF_AARCH64_PURECODE section flag.
2437 if (ctx.arg.emachine == EM_AARCH64)
2438 this->flags |= SHF_AARCH64_PURECODE;
2439
2440 // On PowerPC, this section contains lazy symbol resolvers.
2441 if (ctx.arg.emachine == EM_PPC64) {
2442 name = ".glink";
2443 addralign = 4;
2444 }
2445
2446 // On x86 when IBT is enabled, this section contains the second PLT (lazy
2447 // symbol resolvers).
2448 if ((ctx.arg.emachine == EM_386 || ctx.arg.emachine == EM_X86_64) &&
2449 (ctx.arg.andFeatures & GNU_PROPERTY_X86_FEATURE_1_IBT))
2450 name = ".plt.sec";
2451
2452 // The PLT needs to be writable on SPARC as the dynamic linker will
2453 // modify the instructions in the PLT entries.
2454 if (ctx.arg.emachine == EM_SPARCV9)
2455 this->flags |= SHF_WRITE;
2456}
2457
2458void PltSection::writeTo(uint8_t *buf) {
2459 // At beginning of PLT, we have code to call the dynamic
2460 // linker to resolve dynsyms at runtime. Write such code.
2461 ctx.target->writePltHeader(buf);
2462 size_t off = headerSize;
2463
2464 for (const Symbol *sym : entries) {
2465 ctx.target->writePlt(buf: buf + off, sym: *sym, pltEntryAddr: getVA() + off);
2466 off += ctx.target->pltEntrySize;
2467 }
2468}
2469
2470void PltSection::addEntry(Symbol &sym) {
2471 assert(sym.auxIdx == ctx.symAux.size() - 1);
2472 ctx.symAux.back().pltIdx = entries.size();
2473 entries.push_back(Elt: &sym);
2474}
2475
2476size_t PltSection::getSize() const {
2477 return headerSize + entries.size() * ctx.target->pltEntrySize;
2478}
2479
2480bool PltSection::isNeeded() const {
2481 // For -z retpolineplt, .iplt needs the .plt header.
2482 return !entries.empty() || (ctx.arg.zRetpolineplt && ctx.in.iplt->isNeeded());
2483}
2484
2485// Used by ARM to add mapping symbols in the PLT section, which aid
2486// disassembly.
2487void PltSection::addSymbols() {
2488 ctx.target->addPltHeaderSymbols(isec&: *this);
2489
2490 size_t off = headerSize;
2491 for (size_t i = 0; i < entries.size(); ++i) {
2492 ctx.target->addPltSymbols(isec&: *this, off);
2493 off += ctx.target->pltEntrySize;
2494 }
2495}
2496
2497IpltSection::IpltSection(Ctx &ctx)
2498 : SyntheticSection(ctx, ".iplt", SHT_PROGBITS, SHF_ALLOC | SHF_EXECINSTR,
2499 16) {
2500 // On AArch64, PLT entries only do loads from the .got.plt section, so the
2501 // .iplt section can be marked with the SHF_AARCH64_PURECODE section flag.
2502 if (ctx.arg.emachine == EM_AARCH64)
2503 this->flags |= SHF_AARCH64_PURECODE;
2504
2505 if (ctx.arg.emachine == EM_PPC || ctx.arg.emachine == EM_PPC64) {
2506 name = ".glink";
2507 addralign = 4;
2508 }
2509}
2510
2511void IpltSection::writeTo(uint8_t *buf) {
2512 uint32_t off = 0;
2513 for (const Symbol *sym : entries) {
2514 ctx.target->writeIplt(buf: buf + off, sym: *sym, pltEntryAddr: getVA() + off);
2515 off += ctx.target->ipltEntrySize;
2516 }
2517}
2518
2519size_t IpltSection::getSize() const {
2520 return entries.size() * ctx.target->ipltEntrySize;
2521}
2522
2523void IpltSection::addEntry(Symbol &sym) {
2524 assert(sym.auxIdx == ctx.symAux.size() - 1);
2525 ctx.symAux.back().pltIdx = entries.size();
2526 entries.push_back(Elt: &sym);
2527}
2528
2529// ARM uses mapping symbols to aid disassembly.
2530void IpltSection::addSymbols() {
2531 size_t off = 0;
2532 for (size_t i = 0, e = entries.size(); i != e; ++i) {
2533 ctx.target->addPltSymbols(isec&: *this, off);
2534 off += ctx.target->pltEntrySize;
2535 }
2536}
2537
2538PPC32GlinkSection::PPC32GlinkSection(Ctx &ctx) : PltSection(ctx) {
2539 name = ".glink";
2540 addralign = 4;
2541}
2542
2543void PPC32GlinkSection::writeTo(uint8_t *buf) {
2544 writePPC32GlinkSection(ctx, buf, numEntries: entries.size());
2545}
2546
2547size_t PPC32GlinkSection::getSize() const {
2548 return headerSize + entries.size() * ctx.target->pltEntrySize + footerSize;
2549}
2550
2551// This is an x86-only extra PLT section and used only when a security
2552// enhancement feature called CET is enabled. In this comment, I'll explain what
2553// the feature is and why we have two PLT sections if CET is enabled.
2554//
2555// So, what does CET do? CET introduces a new restriction to indirect jump
2556// instructions. CET works this way. Assume that CET is enabled. Then, if you
2557// execute an indirect jump instruction, the processor verifies that a special
2558// "landing pad" instruction (which is actually a repurposed NOP instruction and
2559// now called "endbr32" or "endbr64") is at the jump target. If the jump target
2560// does not start with that instruction, the processor raises an exception
2561// instead of continuing executing code.
2562//
2563// If CET is enabled, the compiler emits endbr to all locations where indirect
2564// jumps may jump to.
2565//
2566// This mechanism makes it extremely hard to transfer the control to a middle of
2567// a function that is not supporsed to be a indirect jump target, preventing
2568// certain types of attacks such as ROP or JOP.
2569//
2570// Note that the processors in the market as of 2019 don't actually support the
2571// feature. Only the spec is available at the moment.
2572//
2573// Now, I'll explain why we have this extra PLT section for CET.
2574//
2575// Since you can indirectly jump to a PLT entry, we have to make PLT entries
2576// start with endbr. The problem is there's no extra space for endbr (which is 4
2577// bytes long), as the PLT entry is only 16 bytes long and all bytes are already
2578// used.
2579//
2580// In order to deal with the issue, we split a PLT entry into two PLT entries.
2581// Remember that each PLT entry contains code to jump to an address read from
2582// .got.plt AND code to resolve a dynamic symbol lazily. With the 2-PLT scheme,
2583// the former code is written to .plt.sec, and the latter code is written to
2584// .plt.
2585//
2586// Lazy symbol resolution in the 2-PLT scheme works in the usual way, except
2587// that the regular .plt is now called .plt.sec and .plt is repurposed to
2588// contain only code for lazy symbol resolution.
2589//
2590// In other words, this is how the 2-PLT scheme works. Application code is
2591// supposed to jump to .plt.sec to call an external function. Each .plt.sec
2592// entry contains code to read an address from a corresponding .got.plt entry
2593// and jump to that address. Addresses in .got.plt initially point to .plt, so
2594// when an application calls an external function for the first time, the
2595// control is transferred to a function that resolves a symbol name from
2596// external shared object files. That function then rewrites a .got.plt entry
2597// with a resolved address, so that the subsequent function calls directly jump
2598// to a desired location from .plt.sec.
2599//
2600// There is an open question as to whether the 2-PLT scheme was desirable or
2601// not. We could have simply extended the PLT entry size to 32-bytes to
2602// accommodate endbr, and that scheme would have been much simpler than the
2603// 2-PLT scheme. One reason to split PLT was, by doing that, we could keep hot
2604// code (.plt.sec) from cold code (.plt). But as far as I know no one proved
2605// that the optimization actually makes a difference.
2606//
2607// That said, the 2-PLT scheme is a part of the ABI, debuggers and other tools
2608// depend on it, so we implement the ABI.
2609IBTPltSection::IBTPltSection(Ctx &ctx)
2610 : SyntheticSection(ctx, ".plt", SHT_PROGBITS, SHF_ALLOC | SHF_EXECINSTR,
2611 16) {}
2612
2613void IBTPltSection::writeTo(uint8_t *buf) {
2614 ctx.target->writeIBTPlt(buf, numEntries: ctx.in.plt->getNumEntries());
2615}
2616
2617size_t IBTPltSection::getSize() const {
2618 // 16 is the header size of .plt.
2619 return 16 + ctx.in.plt->getNumEntries() * ctx.target->pltEntrySize;
2620}
2621
2622bool IBTPltSection::isNeeded() const { return ctx.in.plt->getNumEntries() > 0; }
2623
2624RelroPaddingSection::RelroPaddingSection(Ctx &ctx)
2625 : SyntheticSection(ctx, ".relro_padding", SHT_NOBITS, SHF_ALLOC | SHF_WRITE,
2626 1) {}
2627
2628PaddingSection::PaddingSection(Ctx &ctx, uint64_t amount, OutputSection *parent)
2629 : SyntheticSection(ctx, ".padding", SHT_PROGBITS, SHF_ALLOC, 1) {
2630 size = amount;
2631 this->parent = parent;
2632}
2633
2634void PaddingSection::writeTo(uint8_t *buf) {
2635 std::array<uint8_t, 4> filler = getParent()->getFiller(ctx);
2636 uint8_t *end = buf + size;
2637 for (; buf + 4 <= end; buf += 4)
2638 memcpy(dest: buf, src: &filler[0], n: 4);
2639 memcpy(dest: buf, src: &filler[0], n: end - buf);
2640}
2641
2642// The string hash function for .gdb_index.
2643static uint32_t computeGdbHash(StringRef s) {
2644 uint32_t h = 0;
2645 for (uint8_t c : s)
2646 h = h * 67 + toLower(x: c) - 113;
2647 return h;
2648}
2649
2650// 4-byte alignment ensures that values in the hash lookup table and the name
2651// table are aligned.
2652DebugNamesBaseSection::DebugNamesBaseSection(Ctx &ctx)
2653 : SyntheticSection(ctx, ".debug_names", SHT_PROGBITS, 0, 4) {}
2654
2655// Get the size of the .debug_names section header in bytes for DWARF32:
2656static uint32_t getDebugNamesHeaderSize(uint32_t augmentationStringSize) {
2657 return /* unit length */ 4 +
2658 /* version */ 2 +
2659 /* padding */ 2 +
2660 /* CU count */ 4 +
2661 /* TU count */ 4 +
2662 /* Foreign TU count */ 4 +
2663 /* Bucket Count */ 4 +
2664 /* Name Count */ 4 +
2665 /* Abbrev table size */ 4 +
2666 /* Augmentation string size */ 4 +
2667 /* Augmentation string */ augmentationStringSize;
2668}
2669
2670static Expected<DebugNamesBaseSection::IndexEntry *>
2671readEntry(uint64_t &offset, const DWARFDebugNames::NameIndex &ni,
2672 uint64_t entriesBase, DWARFDataExtractor &namesExtractor,
2673 const LLDDWARFSection &namesSec) {
2674 auto ie = makeThreadLocal<DebugNamesBaseSection::IndexEntry>();
2675 ie->poolOffset = offset;
2676 Error err = Error::success();
2677 uint64_t ulebVal = namesExtractor.getULEB128(offset_ptr: &offset, Err: &err);
2678 if (err)
2679 return createStringError(EC: inconvertibleErrorCode(),
2680 Fmt: "invalid abbrev code: %s",
2681 Vals: llvm::toString(E: std::move(err)).c_str());
2682 if (!isUInt<32>(x: ulebVal))
2683 return createStringError(EC: inconvertibleErrorCode(),
2684 Fmt: "abbrev code too large for DWARF32: %" PRIu64,
2685 Vals: ulebVal);
2686 ie->abbrevCode = static_cast<uint32_t>(ulebVal);
2687 auto it = ni.getAbbrevs().find_as(Val: ie->abbrevCode);
2688 if (it == ni.getAbbrevs().end())
2689 return createStringError(EC: inconvertibleErrorCode(),
2690 Fmt: "abbrev code not found in abbrev table: %" PRIu32,
2691 Vals: ie->abbrevCode);
2692
2693 DebugNamesBaseSection::AttrValue attr, cuAttr = {.attrValue: 0, .attrSize: 0};
2694 for (DWARFDebugNames::AttributeEncoding a : it->Attributes) {
2695 if (a.Index == dwarf::DW_IDX_parent) {
2696 if (a.Form == dwarf::DW_FORM_ref4) {
2697 attr.attrValue = namesExtractor.getU32(offset_ptr: &offset, Err: &err);
2698 attr.attrSize = 4;
2699 ie->parentOffset = entriesBase + attr.attrValue;
2700 } else if (a.Form != DW_FORM_flag_present)
2701 return createStringError(EC: inconvertibleErrorCode(),
2702 S: "invalid form for DW_IDX_parent");
2703 } else {
2704 switch (a.Form) {
2705 case DW_FORM_data1:
2706 case DW_FORM_ref1: {
2707 attr.attrValue = namesExtractor.getU8(offset_ptr: &offset, Err: &err);
2708 attr.attrSize = 1;
2709 break;
2710 }
2711 case DW_FORM_data2:
2712 case DW_FORM_ref2: {
2713 attr.attrValue = namesExtractor.getU16(offset_ptr: &offset, Err: &err);
2714 attr.attrSize = 2;
2715 break;
2716 }
2717 case DW_FORM_data4:
2718 case DW_FORM_ref4: {
2719 attr.attrValue = namesExtractor.getU32(offset_ptr: &offset, Err: &err);
2720 attr.attrSize = 4;
2721 break;
2722 }
2723 default:
2724 return createStringError(
2725 EC: inconvertibleErrorCode(),
2726 Fmt: "unrecognized form encoding %d in abbrev table", Vals: a.Form);
2727 }
2728 }
2729 if (err)
2730 return createStringError(EC: inconvertibleErrorCode(),
2731 Fmt: "error while reading attributes: %s",
2732 Vals: llvm::toString(E: std::move(err)).c_str());
2733 if (a.Index == DW_IDX_compile_unit)
2734 cuAttr = attr;
2735 else if (a.Form != DW_FORM_flag_present)
2736 ie->attrValues.push_back(Elt: attr);
2737 }
2738 // Canonicalize abbrev by placing the CU/TU index at the end.
2739 ie->attrValues.push_back(Elt: cuAttr);
2740 return ie;
2741}
2742
2743void DebugNamesBaseSection::parseDebugNames(
2744 Ctx &ctx, InputChunk &inputChunk, OutputChunk &chunk,
2745 DWARFDataExtractor &namesExtractor, DataExtractor &strExtractor,
2746 function_ref<SmallVector<uint32_t, 0>(
2747 uint32_t numCus, const DWARFDebugNames::Header &,
2748 const DWARFDebugNames::DWARFDebugNamesOffsets &)>
2749 readOffsets) {
2750 const LLDDWARFSection &namesSec = inputChunk.section;
2751 DenseMap<uint32_t, IndexEntry *> offsetMap;
2752 // Number of CUs seen in previous NameIndex sections within current chunk.
2753 uint32_t numCus = 0;
2754 for (const DWARFDebugNames::NameIndex &ni : *inputChunk.llvmDebugNames) {
2755 NameData &nd = inputChunk.nameData.emplace_back();
2756 nd.hdr = ni.getHeader();
2757 if (nd.hdr.Format != DwarfFormat::DWARF32) {
2758 Err(ctx) << namesSec.sec
2759 << ": found DWARF64, which is currently unsupported";
2760 return;
2761 }
2762 if (nd.hdr.Version != 5) {
2763 Err(ctx) << namesSec.sec << ": unsupported version: " << nd.hdr.Version;
2764 return;
2765 }
2766 uint32_t dwarfSize = dwarf::getDwarfOffsetByteSize(Format: DwarfFormat::DWARF32);
2767 DWARFDebugNames::DWARFDebugNamesOffsets locs = ni.getOffsets();
2768 if (locs.EntriesBase > namesExtractor.getData().size()) {
2769 Err(ctx) << namesSec.sec << ": entry pool start is beyond end of section";
2770 return;
2771 }
2772
2773 SmallVector<uint32_t, 0> entryOffsets = readOffsets(numCus, nd.hdr, locs);
2774
2775 // Read the entry pool.
2776 offsetMap.clear();
2777 nd.nameEntries.resize(N: nd.hdr.NameCount);
2778 for (auto i : seq(Size: nd.hdr.NameCount)) {
2779 NameEntry &ne = nd.nameEntries[i];
2780 uint64_t strOffset = locs.StringOffsetsBase + i * dwarfSize;
2781 ne.stringOffset = strOffset;
2782 uint64_t strp = namesExtractor.getRelocatedValue(Size: dwarfSize, Off: &strOffset);
2783 StringRef name = strExtractor.getCStrRef(OffsetPtr: &strp);
2784 ne.name = name.data();
2785 ne.hashValue = caseFoldingDjbHash(Buffer: name);
2786
2787 // Read a series of index entries that end with abbreviation code 0.
2788 uint64_t offset = locs.EntriesBase + entryOffsets[i];
2789 while (offset < namesSec.Data.size() && namesSec.Data[offset] != 0) {
2790 // Read & store all entries (for the same string).
2791 Expected<IndexEntry *> ieOrErr =
2792 readEntry(offset, ni, entriesBase: locs.EntriesBase, namesExtractor, namesSec);
2793 if (!ieOrErr) {
2794 Err(ctx) << namesSec.sec << ": " << ieOrErr.takeError();
2795 return;
2796 }
2797 ne.indexEntries.push_back(Elt: std::move(*ieOrErr));
2798 }
2799 if (offset >= namesSec.Data.size())
2800 Err(ctx) << namesSec.sec << ": index entry is out of bounds";
2801
2802 for (IndexEntry &ie : ne.entries())
2803 offsetMap[ie.poolOffset] = &ie;
2804 }
2805
2806 // Assign parent pointers, which will be used to update DW_IDX_parent index
2807 // attributes. Note: offsetMap[0] does not exist, so parentOffset == 0 will
2808 // get parentEntry == null as well.
2809 for (NameEntry &ne : nd.nameEntries)
2810 for (IndexEntry &ie : ne.entries())
2811 ie.parentEntry = offsetMap.lookup(Val: ie.parentOffset);
2812 numCus += nd.hdr.CompUnitCount;
2813 }
2814}
2815
2816// Compute the form for output DW_IDX_compile_unit attributes, similar to
2817// DIEInteger::BestForm. The input form (often DW_FORM_data1) may not hold all
2818// the merged CU indices.
2819std::pair<uint8_t, dwarf::Form> static getMergedCuCountForm(
2820 uint32_t compUnitCount) {
2821 if (compUnitCount > UINT16_MAX)
2822 return {4, DW_FORM_data4};
2823 if (compUnitCount > UINT8_MAX)
2824 return {2, DW_FORM_data2};
2825 return {1, DW_FORM_data1};
2826}
2827
2828void DebugNamesBaseSection::computeHdrAndAbbrevTable(
2829 MutableArrayRef<InputChunk> inputChunks) {
2830 TimeTraceScope timeScope("Merge .debug_names", "hdr and abbrev table");
2831 size_t numCu = 0;
2832 hdr.Format = DwarfFormat::DWARF32;
2833 hdr.Version = 5;
2834 hdr.CompUnitCount = 0;
2835 hdr.LocalTypeUnitCount = 0;
2836 hdr.ForeignTypeUnitCount = 0;
2837 hdr.AugmentationStringSize = 0;
2838
2839 // Compute CU and TU counts.
2840 for (auto i : seq(Size: numChunks)) {
2841 InputChunk &inputChunk = inputChunks[i];
2842 inputChunk.baseCuIdx = numCu;
2843 numCu += chunks[i].compUnits.size();
2844 for (const NameData &nd : inputChunk.nameData) {
2845 hdr.CompUnitCount += nd.hdr.CompUnitCount;
2846 // TODO: We don't handle type units yet, so LocalTypeUnitCount &
2847 // ForeignTypeUnitCount are left as 0.
2848 if (nd.hdr.LocalTypeUnitCount || nd.hdr.ForeignTypeUnitCount)
2849 Warn(ctx) << inputChunk.section.sec
2850 << ": type units are not implemented";
2851 // If augmentation strings are not identical, use an empty string.
2852 if (i == 0) {
2853 hdr.AugmentationStringSize = nd.hdr.AugmentationStringSize;
2854 hdr.AugmentationString = nd.hdr.AugmentationString;
2855 } else if (hdr.AugmentationString != nd.hdr.AugmentationString) {
2856 // There are conflicting augmentation strings, so it's best for the
2857 // merged index to not use an augmentation string.
2858 hdr.AugmentationStringSize = 0;
2859 hdr.AugmentationString.clear();
2860 }
2861 }
2862 }
2863
2864 // Create the merged abbrev table, uniquifyinng the input abbrev tables and
2865 // computing mapping from old (per-cu) abbrev codes to new (merged) abbrev
2866 // codes.
2867 FoldingSet<Abbrev> abbrevSet;
2868 // Determine the form for the DW_IDX_compile_unit attributes in the merged
2869 // index. The input form may not be big enough for all CU indices.
2870 dwarf::Form cuAttrForm = getMergedCuCountForm(compUnitCount: hdr.CompUnitCount).second;
2871 for (InputChunk &inputChunk : inputChunks) {
2872 for (auto [i, ni] : enumerate(First&: *inputChunk.llvmDebugNames)) {
2873 for (const DWARFDebugNames::Abbrev &oldAbbrev : ni.getAbbrevs()) {
2874 // Canonicalize abbrev by placing the CU/TU index at the end,
2875 // similar to 'parseDebugNames'.
2876 Abbrev abbrev;
2877 DWARFDebugNames::AttributeEncoding cuAttr(DW_IDX_compile_unit,
2878 cuAttrForm);
2879 abbrev.code = oldAbbrev.Code;
2880 abbrev.tag = oldAbbrev.Tag;
2881 for (DWARFDebugNames::AttributeEncoding a : oldAbbrev.Attributes) {
2882 if (a.Index == DW_IDX_compile_unit)
2883 cuAttr.Index = a.Index;
2884 else
2885 abbrev.attributes.push_back(Elt: {a.Index, a.Form});
2886 }
2887 // Put the CU/TU index at the end of the attributes list.
2888 abbrev.attributes.push_back(Elt: cuAttr);
2889
2890 // Profile the abbrev, get or assign a new code, then record the abbrev
2891 // code mapping.
2892 FoldingSetNodeID id;
2893 abbrev.Profile(id);
2894 uint32_t newCode;
2895 FoldingSetInsertToken token;
2896 if (Abbrev *existing = abbrevSet.lookup(ID: id, Token&: token)) {
2897 // Found it; we've already seen an identical abbreviation.
2898 newCode = existing->code;
2899 } else {
2900 Abbrev *abbrev2 =
2901 new (abbrevAlloc.Allocate()) Abbrev(std::move(abbrev));
2902 abbrevSet.insert(N: abbrev2, Token: token);
2903 abbrevTable.push_back(Elt: abbrev2);
2904 newCode = abbrevTable.size();
2905 abbrev2->code = newCode;
2906 }
2907 inputChunk.nameData[i].abbrevCodeMap[oldAbbrev.Code] = newCode;
2908 }
2909 }
2910 }
2911
2912 // Compute the merged abbrev table.
2913 raw_svector_ostream os(abbrevTableBuf);
2914 for (Abbrev *abbrev : abbrevTable) {
2915 encodeULEB128(Value: abbrev->code, OS&: os);
2916 encodeULEB128(Value: abbrev->tag, OS&: os);
2917 for (DWARFDebugNames::AttributeEncoding a : abbrev->attributes) {
2918 encodeULEB128(Value: a.Index, OS&: os);
2919 encodeULEB128(Value: a.Form, OS&: os);
2920 }
2921 os.write(Ptr: "\0", Size: 2); // attribute specification end
2922 }
2923 os.write(C: 0); // abbrev table end
2924 hdr.AbbrevTableSize = abbrevTableBuf.size();
2925}
2926
2927void DebugNamesBaseSection::Abbrev::Profile(FoldingSetNodeID &id) const {
2928 id.AddInteger(I: tag);
2929 for (const DWARFDebugNames::AttributeEncoding &attr : attributes) {
2930 id.AddInteger(I: attr.Index);
2931 id.AddInteger(I: attr.Form);
2932 }
2933}
2934
2935std::pair<uint32_t, uint32_t> DebugNamesBaseSection::computeEntryPool(
2936 MutableArrayRef<InputChunk> inputChunks) {
2937 TimeTraceScope timeScope("Merge .debug_names", "entry pool");
2938 // Collect and de-duplicate all the names (preserving all the entries).
2939 // Speed it up using multithreading, as the number of symbols can be in the
2940 // order of millions.
2941 const size_t concurrency =
2942 bit_floor(Value: std::min<size_t>(a: ctx.arg.threadCount, b: numShards));
2943 const size_t shift = 32 - countr_zero(Val: numShards);
2944 const uint8_t cuAttrSize = getMergedCuCountForm(compUnitCount: hdr.CompUnitCount).first;
2945 DenseMap<CachedHashStringRef, size_t> maps[numShards];
2946
2947 parallelFor(Begin: 0, End: concurrency, Fn: [&](size_t threadId) {
2948 for (auto i : seq(Size: numChunks)) {
2949 InputChunk &inputChunk = inputChunks[i];
2950 for (auto j : seq(Size: inputChunk.nameData.size())) {
2951 NameData &nd = inputChunk.nameData[j];
2952 // Deduplicate the NameEntry records (based on the string/name),
2953 // appending all IndexEntries from duplicate NameEntry records to
2954 // the single preserved copy.
2955 for (NameEntry &ne : nd.nameEntries) {
2956 auto shardId = ne.hashValue >> shift;
2957 if ((shardId & (concurrency - 1)) != threadId)
2958 continue;
2959
2960 ne.chunkIdx = i;
2961 for (IndexEntry &ie : ne.entries()) {
2962 // Update the IndexEntry's abbrev code to match the merged
2963 // abbreviations.
2964 ie.abbrevCode = nd.abbrevCodeMap[ie.abbrevCode];
2965 // Update the DW_IDX_compile_unit attribute (the last one after
2966 // canonicalization) to have correct merged offset value and size.
2967 auto &back = ie.attrValues.back();
2968 back.attrValue += inputChunk.baseCuIdx + j;
2969 back.attrSize = cuAttrSize;
2970 }
2971
2972 auto &nameVec = nameVecs[shardId];
2973 auto [it, inserted] = maps[shardId].try_emplace(
2974 Key: CachedHashStringRef(ne.name, ne.hashValue), Args: nameVec.size());
2975 if (inserted)
2976 nameVec.push_back(Elt: std::move(ne));
2977 else
2978 nameVec[it->second].indexEntries.append(RHS: std::move(ne.indexEntries));
2979 }
2980 }
2981 }
2982 });
2983
2984 // Compute entry offsets in parallel. First, compute offsets relative to the
2985 // current shard.
2986 uint32_t offsets[numShards];
2987 parallelFor(Begin: 0, End: numShards, Fn: [&](size_t shard) {
2988 uint32_t offset = 0;
2989 for (NameEntry &ne : nameVecs[shard]) {
2990 ne.entryOffset = offset;
2991 for (IndexEntry &ie : ne.entries()) {
2992 ie.poolOffset = offset;
2993 offset += getULEB128Size(Value: ie.abbrevCode);
2994 for (AttrValue value : ie.attrValues)
2995 offset += value.attrSize;
2996 }
2997 ++offset; // index entry sentinel
2998 }
2999 offsets[shard] = offset;
3000 });
3001 // Then add shard offsets.
3002 std::partial_sum(first: offsets, last: std::end(arr&: offsets), result: offsets);
3003 parallelFor(Begin: 1, End: numShards, Fn: [&](size_t shard) {
3004 uint32_t offset = offsets[shard - 1];
3005 for (NameEntry &ne : nameVecs[shard]) {
3006 ne.entryOffset += offset;
3007 for (IndexEntry &ie : ne.entries())
3008 ie.poolOffset += offset;
3009 }
3010 });
3011
3012 // Update the DW_IDX_parent entries that refer to real parents (have
3013 // DW_FORM_ref4).
3014 parallelFor(Begin: 0, End: numShards, Fn: [&](size_t shard) {
3015 for (NameEntry &ne : nameVecs[shard]) {
3016 for (IndexEntry &ie : ne.entries()) {
3017 if (!ie.parentEntry)
3018 continue;
3019 // Abbrevs are indexed starting at 1; vector starts at 0. (abbrevCode
3020 // corresponds to position in the merged table vector).
3021 const Abbrev *abbrev = abbrevTable[ie.abbrevCode - 1];
3022 for (const auto &[a, v] : zip_equal(t: abbrev->attributes, u&: ie.attrValues))
3023 if (a.Index == DW_IDX_parent && a.Form == DW_FORM_ref4)
3024 v.attrValue = ie.parentEntry->poolOffset;
3025 }
3026 }
3027 });
3028
3029 // Return (entry pool size, number of entries).
3030 uint32_t num = 0;
3031 for (auto &map : maps)
3032 num += map.size();
3033 return {offsets[numShards - 1], num};
3034}
3035
3036void DebugNamesBaseSection::init(
3037 function_ref<void(InputFile *, InputChunk &, OutputChunk &)> parseFile) {
3038 TimeTraceScope timeScope("Merge .debug_names");
3039 // Collect and remove input .debug_names sections. Save InputSection pointers
3040 // to relocate string offsets in `writeTo`.
3041 SetVector<InputFile *> files;
3042 for (InputSectionBase *s : ctx.inputSections) {
3043 InputSection *isec = dyn_cast<InputSection>(Val: s);
3044 if (!isec)
3045 continue;
3046 if (!(s->flags & SHF_ALLOC) && s->name == ".debug_names") {
3047 s->markDead();
3048 inputSections.push_back(Elt: isec);
3049 files.insert(X: isec->file);
3050 }
3051 }
3052
3053 // Parse input .debug_names sections and extract InputChunk and OutputChunk
3054 // data. OutputChunk contains CU information, which will be needed by
3055 // `writeTo`.
3056 auto inputChunksPtr = std::make_unique<InputChunk[]>(num: files.size());
3057 MutableArrayRef<InputChunk> inputChunks(inputChunksPtr.get(), files.size());
3058 numChunks = files.size();
3059 chunks = std::make_unique<OutputChunk[]>(num: files.size());
3060 {
3061 TimeTraceScope timeScope("Merge .debug_names", "parse");
3062 parallelFor(Begin: 0, End: files.size(), Fn: [&](size_t i) {
3063 parseFile(files[i], inputChunks[i], chunks[i]);
3064 });
3065 }
3066
3067 // Compute section header (except unit_length), abbrev table, and entry pool.
3068 computeHdrAndAbbrevTable(inputChunks);
3069 uint32_t entryPoolSize;
3070 std::tie(args&: entryPoolSize, args&: hdr.NameCount) = computeEntryPool(inputChunks);
3071 hdr.BucketCount = dwarf::getDebugNamesBucketCount(UniqueHashCount: hdr.NameCount);
3072
3073 // Compute the section size. Subtract 4 to get the unit_length for DWARF32.
3074 uint32_t hdrSize = getDebugNamesHeaderSize(augmentationStringSize: hdr.AugmentationStringSize);
3075 size = findDebugNamesOffsets(EndOfHeaderOffset: hdrSize, Hdr: hdr).EntriesBase + entryPoolSize;
3076 hdr.UnitLength = size - 4;
3077}
3078
3079template <class ELFT>
3080DebugNamesSection<ELFT>::DebugNamesSection(Ctx &ctx)
3081 : DebugNamesBaseSection(ctx) {
3082 init(parseFile: [&](InputFile *f, InputChunk &inputChunk, OutputChunk &chunk) {
3083 auto *file = cast<ObjFile<ELFT>>(f);
3084 DWARFContext dwarf(std::make_unique<LLDDwarfObj<ELFT>>(file));
3085 auto &dobj = static_cast<const LLDDwarfObj<ELFT> &>(dwarf.getDWARFObj());
3086 chunk.infoSec = dobj.getInfoSection();
3087 DWARFDataExtractor namesExtractor(dobj, dobj.getNamesSection(),
3088 ELFT::Endianness == endianness::little,
3089 ELFT::Is64Bits ? 8 : 4);
3090 // .debug_str is needed to get symbol names from string offsets.
3091 DataExtractor strExtractor(dobj.getStrSection(),
3092 ELFT::Endianness == endianness::little);
3093 inputChunk.section = dobj.getNamesSection();
3094
3095 inputChunk.llvmDebugNames.emplace(args&: namesExtractor, args&: strExtractor);
3096 if (Error e = inputChunk.llvmDebugNames->extract()) {
3097 Err(ctx) << dobj.getNamesSection().sec << ": " << std::move(e);
3098 }
3099 parseDebugNames(
3100 ctx, inputChunk, chunk, namesExtractor, strExtractor,
3101 readOffsets: [&chunk, namesData = dobj.getNamesSection().Data.data()](
3102 uint32_t numCus, const DWARFDebugNames::Header &hdr,
3103 const DWARFDebugNames::DWARFDebugNamesOffsets &locs) {
3104 // Read CU offsets, which are relocated by .debug_info + X
3105 // relocations. Record the section offset to be relocated by
3106 // `finalizeContents`.
3107 chunk.compUnits.resize_for_overwrite(N: numCus + hdr.CompUnitCount);
3108 for (auto i : seq(Size: hdr.CompUnitCount))
3109 chunk.compUnits[numCus + i] = locs.CUsBase + i * 4;
3110
3111 // Read entry offsets.
3112 const char *p = namesData + locs.EntryOffsetsBase;
3113 SmallVector<uint32_t, 0> entryOffsets;
3114 entryOffsets.resize_for_overwrite(N: hdr.NameCount);
3115 for (uint32_t &offset : entryOffsets)
3116 offset = endian::readNext<uint32_t, ELFT::Endianness, unaligned>(p);
3117 return entryOffsets;
3118 });
3119 });
3120}
3121
3122template <class ELFT>
3123template <class RelTy>
3124void DebugNamesSection<ELFT>::getNameRelocs(
3125 const InputFile &file, DenseMap<uint32_t, uint32_t> &relocs,
3126 Relocs<RelTy> rels) {
3127 for (const RelTy &rel : rels) {
3128 Symbol &sym = file.getRelocTargetSym(rel);
3129 relocs[rel.r_offset] = sym.getVA(ctx, addend: getAddend<ELFT>(rel));
3130 }
3131}
3132
3133template <class ELFT> void DebugNamesSection<ELFT>::finalizeContents() {
3134 // Get relocations of .debug_names sections.
3135 auto relocs = std::make_unique<DenseMap<uint32_t, uint32_t>[]>(numChunks);
3136 parallelFor(0, numChunks, [&](size_t i) {
3137 InputSection *sec = inputSections[i];
3138 invokeOnRelocs(*sec, getNameRelocs, *sec->file, relocs.get()[i]);
3139
3140 // Relocate CU offsets with .debug_info + X relocations.
3141 OutputChunk &chunk = chunks.get()[i];
3142 for (auto [j, cuOffset] : enumerate(First&: chunk.compUnits))
3143 cuOffset = relocs.get()[i].lookup(cuOffset);
3144 });
3145
3146 // Relocate string offsets in the name table with .debug_str + X relocations.
3147 parallelForEach(nameVecs, [&](auto &nameVec) {
3148 for (NameEntry &ne : nameVec)
3149 ne.stringOffset = relocs.get()[ne.chunkIdx].lookup(ne.stringOffset);
3150 });
3151}
3152
3153template <class ELFT> void DebugNamesSection<ELFT>::writeTo(uint8_t *buf) {
3154 [[maybe_unused]] const uint8_t *const beginBuf = buf;
3155 // Write the header.
3156 endian::writeNext<uint32_t, ELFT::Endianness>(buf, hdr.UnitLength);
3157 endian::writeNext<uint16_t, ELFT::Endianness>(buf, hdr.Version);
3158 buf += 2; // padding
3159 endian::writeNext<uint32_t, ELFT::Endianness>(buf, hdr.CompUnitCount);
3160 endian::writeNext<uint32_t, ELFT::Endianness>(buf, hdr.LocalTypeUnitCount);
3161 endian::writeNext<uint32_t, ELFT::Endianness>(buf, hdr.ForeignTypeUnitCount);
3162 endian::writeNext<uint32_t, ELFT::Endianness>(buf, hdr.BucketCount);
3163 endian::writeNext<uint32_t, ELFT::Endianness>(buf, hdr.NameCount);
3164 endian::writeNext<uint32_t, ELFT::Endianness>(buf, hdr.AbbrevTableSize);
3165 endian::writeNext<uint32_t, ELFT::Endianness>(buf,
3166 hdr.AugmentationStringSize);
3167 memcpy(buf, hdr.AugmentationString.c_str(), hdr.AugmentationString.size());
3168 buf += hdr.AugmentationStringSize;
3169
3170 // Write the CU list.
3171 for (auto &chunk : getChunks())
3172 for (uint32_t cuOffset : chunk.compUnits)
3173 endian::writeNext<uint32_t, ELFT::Endianness>(buf, cuOffset);
3174
3175 // TODO: Write the local TU list, then the foreign TU list..
3176
3177 // Write the hash lookup table.
3178 SmallVector<SmallVector<NameEntry *, 0>, 0> buckets(hdr.BucketCount);
3179 // Symbols enter into a bucket whose index is the hash modulo bucket_count.
3180 for (auto &nameVec : nameVecs)
3181 for (NameEntry &ne : nameVec)
3182 buckets[ne.hashValue % hdr.BucketCount].push_back(&ne);
3183
3184 // Write buckets (accumulated bucket counts).
3185 uint32_t bucketIdx = 1;
3186 for (const SmallVector<NameEntry *, 0> &bucket : buckets) {
3187 if (!bucket.empty())
3188 endian::write32<ELFT::Endianness>(buf, bucketIdx);
3189 buf += 4;
3190 bucketIdx += bucket.size();
3191 }
3192 // Write the hashes.
3193 for (const SmallVector<NameEntry *, 0> &bucket : buckets)
3194 for (const NameEntry *e : bucket)
3195 endian::writeNext<uint32_t, ELFT::Endianness>(buf, e->hashValue);
3196
3197 // Write the name table. The name entries are ordered by bucket_idx and
3198 // correspond one-to-one with the hash lookup table.
3199 //
3200 // First, write the relocated string offsets.
3201 for (const SmallVector<NameEntry *, 0> &bucket : buckets)
3202 for (const NameEntry *ne : bucket)
3203 endian::writeNext<uint32_t, ELFT::Endianness>(buf, ne->stringOffset);
3204
3205 // Then write the entry offsets.
3206 for (const SmallVector<NameEntry *, 0> &bucket : buckets)
3207 for (const NameEntry *ne : bucket)
3208 endian::writeNext<uint32_t, ELFT::Endianness>(buf, ne->entryOffset);
3209
3210 // Write the abbrev table.
3211 buf = llvm::copy(abbrevTableBuf, buf);
3212
3213 // Write the entry pool. Unlike the name table, the name entries follow the
3214 // nameVecs order computed by `computeEntryPool`.
3215 for (auto &nameVec : nameVecs) {
3216 for (NameEntry &ne : nameVec) {
3217 // Write all the entries for the string.
3218 for (const IndexEntry &ie : ne.entries()) {
3219 buf += encodeULEB128(Value: ie.abbrevCode, p: buf);
3220 for (AttrValue value : ie.attrValues) {
3221 switch (value.attrSize) {
3222 case 1:
3223 *buf++ = value.attrValue;
3224 break;
3225 case 2:
3226 endian::writeNext<uint16_t, ELFT::Endianness>(buf, value.attrValue);
3227 break;
3228 case 4:
3229 endian::writeNext<uint32_t, ELFT::Endianness>(buf, value.attrValue);
3230 break;
3231 default:
3232 llvm_unreachable("invalid attrSize");
3233 }
3234 }
3235 }
3236 ++buf; // index entry sentinel
3237 }
3238 }
3239 assert(uint64_t(buf - beginBuf) == size);
3240}
3241
3242GdbIndexSection::GdbIndexSection(Ctx &ctx)
3243 : SyntheticSection(ctx, ".gdb_index", SHT_PROGBITS, 0, 1) {}
3244
3245// Returns the desired size of an on-disk hash table for a .gdb_index section.
3246// There's a tradeoff between size and collision rate. We aim 75% utilization.
3247size_t GdbIndexSection::computeSymtabSize() const {
3248 return std::max<size_t>(a: NextPowerOf2(A: symbols.size() * 4 / 3), b: 1024);
3249}
3250
3251static SmallVector<GdbIndexSection::CuEntry, 0>
3252readCuList(DWARFContext &dwarf) {
3253 SmallVector<GdbIndexSection::CuEntry, 0> ret;
3254 for (std::unique_ptr<DWARFUnit> &cu : dwarf.compile_units())
3255 ret.push_back(Elt: {.cuOffset: cu->getOffset(), .cuLength: cu->getLength() + 4});
3256 return ret;
3257}
3258
3259static SmallVector<GdbIndexSection::AddressEntry, 0>
3260readAddressAreas(Ctx &ctx, DWARFContext &dwarf, InputSection *sec) {
3261 SmallVector<GdbIndexSection::AddressEntry, 0> ret;
3262
3263 uint32_t cuIdx = 0;
3264 for (std::unique_ptr<DWARFUnit> &cu : dwarf.compile_units()) {
3265 if (Error e = cu->tryExtractDIEsIfNeeded(CUDieOnly: false)) {
3266 Warn(ctx) << sec << ": " << std::move(e);
3267 return {};
3268 }
3269 Expected<DWARFAddressRangesVector> ranges = cu->collectAddressRanges();
3270 if (!ranges) {
3271 Warn(ctx) << sec << ": " << ranges.takeError();
3272 return {};
3273 }
3274
3275 ArrayRef<InputSectionBase *> sections = sec->file->getSections();
3276 for (DWARFAddressRange &r : *ranges) {
3277 if (r.SectionIndex == -1ULL)
3278 continue;
3279 // Range list with zero size has no effect.
3280 InputSectionBase *s = sections[r.SectionIndex];
3281 if (s && s != &InputSection::discarded && s->isLive())
3282 if (r.LowPC != r.HighPC)
3283 ret.push_back(Elt: {.section: cast<InputSection>(Val: s), .lowAddress: r.LowPC, .highAddress: r.HighPC, .cuIndex: cuIdx});
3284 }
3285 ++cuIdx;
3286 }
3287
3288 return ret;
3289}
3290
3291template <class ELFT>
3292static SmallVector<GdbIndexSection::NameAttrEntry, 0>
3293readPubNamesAndTypes(Ctx &ctx, const LLDDwarfObj<ELFT> &obj,
3294 const SmallVectorImpl<GdbIndexSection::CuEntry> &cus) {
3295 const LLDDWARFSection &pubNames = obj.getGnuPubnamesSection();
3296 const LLDDWARFSection &pubTypes = obj.getGnuPubtypesSection();
3297
3298 SmallVector<GdbIndexSection::NameAttrEntry, 0> ret;
3299 for (const LLDDWARFSection *pub : {&pubNames, &pubTypes}) {
3300 DWARFDataExtractor data(obj, *pub, ELFT::Endianness == endianness::little,
3301 ELFT::Is64Bits ? 8 : 4);
3302 DWARFDebugPubTable table;
3303 table.extract(Data: data, /*GnuStyle=*/true, RecoverableErrorHandler: [&](Error e) {
3304 Warn(ctx) << pub->sec << ": " << std::move(e);
3305 });
3306 for (const DWARFDebugPubTable::Set &set : table.getData()) {
3307 // The value written into the constant pool is kind << 24 | cuIndex. As we
3308 // don't know how many compilation units precede this object to compute
3309 // cuIndex, we compute (kind << 24 | cuIndexInThisObject) instead, and add
3310 // the number of preceding compilation units later.
3311 uint32_t i = llvm::partition_point(cus,
3312 [&](GdbIndexSection::CuEntry cu) {
3313 return cu.cuOffset < set.Offset;
3314 }) -
3315 cus.begin();
3316 for (const DWARFDebugPubTable::Entry &ent : set.Entries)
3317 ret.push_back(Elt: {.name: {ent.Name, computeGdbHash(s: ent.Name)},
3318 .cuIndexAndAttrs: (ent.Descriptor.toBits() << 24) | i});
3319 }
3320 }
3321 return ret;
3322}
3323
3324// Create a list of symbols from a given list of symbol names and types
3325// by uniquifying them by name.
3326static std::pair<SmallVector<GdbIndexSection::GdbSymbol, 0>, size_t>
3327createSymbols(
3328 Ctx &ctx,
3329 ArrayRef<SmallVector<GdbIndexSection::NameAttrEntry, 0>> nameAttrs,
3330 const SmallVector<GdbIndexSection::GdbChunk, 0> &chunks) {
3331 using GdbSymbol = GdbIndexSection::GdbSymbol;
3332 using NameAttrEntry = GdbIndexSection::NameAttrEntry;
3333
3334 // For each chunk, compute the number of compilation units preceding it.
3335 uint32_t cuIdx = 0;
3336 std::unique_ptr<uint32_t[]> cuIdxs(new uint32_t[chunks.size()]);
3337 for (uint32_t i = 0, e = chunks.size(); i != e; ++i) {
3338 cuIdxs[i] = cuIdx;
3339 cuIdx += chunks[i].compilationUnits.size();
3340 }
3341
3342 // Collect the compilation unitss for each unique name. Speed it up using
3343 // multi-threading as the number of symbols can be in the order of millions.
3344 // Shard GdbSymbols by hash's high bits.
3345 constexpr size_t numShards = 32;
3346 const size_t concurrency =
3347 llvm::bit_floor(Value: std::min<size_t>(a: ctx.arg.threadCount, b: numShards));
3348 const size_t shift = 32 - llvm::countr_zero(Val: numShards);
3349 auto map =
3350 std::make_unique<DenseMap<CachedHashStringRef, size_t>[]>(num: numShards);
3351 auto symbols = std::make_unique<SmallVector<GdbSymbol, 0>[]>(num: numShards);
3352 parallelFor(Begin: 0, End: concurrency, Fn: [&](size_t threadId) {
3353 uint32_t i = 0;
3354 for (ArrayRef<NameAttrEntry> entries : nameAttrs) {
3355 for (const NameAttrEntry &ent : entries) {
3356 size_t shardId = ent.name.hash() >> shift;
3357 if ((shardId & (concurrency - 1)) != threadId)
3358 continue;
3359
3360 uint32_t v = ent.cuIndexAndAttrs + cuIdxs[i];
3361 auto [it, inserted] =
3362 map[shardId].try_emplace(Key: ent.name, Args: symbols[shardId].size());
3363 if (inserted)
3364 symbols[shardId].push_back(Elt: {.name: ent.name, .cuVector: {v}, .nameOff: 0, .cuVectorOff: 0});
3365 else
3366 symbols[shardId][it->second].cuVector.push_back(Elt: v);
3367 }
3368 ++i;
3369 }
3370 });
3371
3372 size_t numSymbols = 0;
3373 for (ArrayRef<GdbSymbol> v : ArrayRef(symbols.get(), numShards))
3374 numSymbols += v.size();
3375
3376 // The return type is a flattened vector, so we'll copy each vector
3377 // contents to Ret.
3378 SmallVector<GdbSymbol, 0> ret;
3379 ret.reserve(N: numSymbols);
3380 for (SmallVector<GdbSymbol, 0> &vec :
3381 MutableArrayRef(symbols.get(), numShards))
3382 for (GdbSymbol &sym : vec)
3383 ret.push_back(Elt: std::move(sym));
3384
3385 // CU vectors and symbol names are adjacent in the output file.
3386 // We can compute their offsets in the output file now.
3387 size_t off = 0;
3388 for (GdbSymbol &sym : ret) {
3389 sym.cuVectorOff = off;
3390 off += (sym.cuVector.size() + 1) * 4;
3391 }
3392 for (GdbSymbol &sym : ret) {
3393 sym.nameOff = off;
3394 off += sym.name.size() + 1;
3395 }
3396 // If off overflows, the last symbol's nameOff likely overflows.
3397 if (!isUInt<32>(x: off))
3398 Err(ctx) << "--gdb-index: constant pool size (" << off
3399 << ") exceeds UINT32_MAX";
3400
3401 return {ret, off};
3402}
3403
3404// Returns a newly-created .gdb_index section.
3405template <class ELFT>
3406std::unique_ptr<GdbIndexSection> GdbIndexSection::create(Ctx &ctx) {
3407 llvm::TimeTraceScope timeScope("Create gdb index");
3408
3409 // Collect InputFiles with .debug_info. See the comment in
3410 // LLDDwarfObj<ELFT>::LLDDwarfObj. If we do lightweight parsing in the future,
3411 // note that isec->data() may uncompress the full content, which should be
3412 // parallelized.
3413 SetVector<InputFile *> files;
3414 for (InputSectionBase *s : ctx.inputSections) {
3415 InputSection *isec = dyn_cast<InputSection>(Val: s);
3416 if (!isec)
3417 continue;
3418 // .debug_gnu_pub{names,types} are useless in executables.
3419 // They are present in input object files solely for creating
3420 // a .gdb_index. So we can remove them from the output.
3421 if (s->name == ".debug_gnu_pubnames" || s->name == ".debug_gnu_pubtypes")
3422 s->markDead();
3423 else if (isec->name == ".debug_info")
3424 files.insert(X: isec->file);
3425 }
3426 // Drop .rel[a].debug_gnu_pub{names,types} for --emit-relocs.
3427 llvm::erase_if(ctx.inputSections, [](InputSectionBase *s) {
3428 if (auto *isec = dyn_cast<InputSection>(Val: s))
3429 if (InputSectionBase *rel = isec->getRelocatedSection())
3430 return !rel->isLive();
3431 return !s->isLive();
3432 });
3433
3434 SmallVector<GdbChunk, 0> chunks(files.size());
3435 SmallVector<SmallVector<NameAttrEntry, 0>, 0> nameAttrs(files.size());
3436
3437 parallelFor(0, files.size(), [&](size_t i) {
3438 // To keep memory usage low, we don't want to keep cached DWARFContext, so
3439 // avoid getDwarf() here.
3440 ObjFile<ELFT> *file = cast<ObjFile<ELFT>>(files[i]);
3441 DWARFContext dwarf(std::make_unique<LLDDwarfObj<ELFT>>(file));
3442 auto &dobj = static_cast<const LLDDwarfObj<ELFT> &>(dwarf.getDWARFObj());
3443
3444 // If the are multiple compile units .debug_info (very rare ld -r --unique),
3445 // this only picks the last one. Other address ranges are lost.
3446 chunks[i].sec = dobj.getInfoSection();
3447 chunks[i].compilationUnits = readCuList(dwarf);
3448 chunks[i].addressAreas = readAddressAreas(ctx, dwarf, sec: chunks[i].sec);
3449 nameAttrs[i] =
3450 readPubNamesAndTypes<ELFT>(ctx, dobj, chunks[i].compilationUnits);
3451 });
3452
3453 auto ret = std::make_unique<GdbIndexSection>(args&: ctx);
3454 ret->chunks = std::move(chunks);
3455 std::tie(args&: ret->symbols, args&: ret->size) =
3456 createSymbols(ctx, nameAttrs, chunks: ret->chunks);
3457
3458 // Count the areas other than the constant pool.
3459 ret->size += sizeof(GdbIndexHeader) + ret->computeSymtabSize() * 8;
3460 for (GdbChunk &chunk : ret->chunks)
3461 ret->size +=
3462 chunk.compilationUnits.size() * 16 + chunk.addressAreas.size() * 20;
3463
3464 return ret;
3465}
3466
3467void GdbIndexSection::writeTo(uint8_t *buf) {
3468 // Write the header.
3469 auto *hdr = reinterpret_cast<GdbIndexHeader *>(buf);
3470 uint8_t *start = buf;
3471 hdr->version = 7;
3472 buf += sizeof(*hdr);
3473
3474 // Write the CU list.
3475 hdr->cuListOff = buf - start;
3476 for (GdbChunk &chunk : chunks) {
3477 for (CuEntry &cu : chunk.compilationUnits) {
3478 write64le(P: buf, V: chunk.sec->outSecOff + cu.cuOffset);
3479 write64le(P: buf + 8, V: cu.cuLength);
3480 buf += 16;
3481 }
3482 }
3483
3484 // Write the address area.
3485 hdr->cuTypesOff = buf - start;
3486 hdr->addressAreaOff = buf - start;
3487 uint32_t cuOff = 0;
3488 for (GdbChunk &chunk : chunks) {
3489 for (AddressEntry &e : chunk.addressAreas) {
3490 // In the case of ICF there may be duplicate address range entries.
3491 const uint64_t baseAddr = e.section->repl->getVA(offset: 0);
3492 write64le(P: buf, V: baseAddr + e.lowAddress);
3493 write64le(P: buf + 8, V: baseAddr + e.highAddress);
3494 write32le(P: buf + 16, V: e.cuIndex + cuOff);
3495 buf += 20;
3496 }
3497 cuOff += chunk.compilationUnits.size();
3498 }
3499
3500 // Write the on-disk open-addressing hash table containing symbols.
3501 hdr->symtabOff = buf - start;
3502 size_t symtabSize = computeSymtabSize();
3503 uint32_t mask = symtabSize - 1;
3504
3505 for (GdbSymbol &sym : symbols) {
3506 uint32_t h = sym.name.hash();
3507 uint32_t i = h & mask;
3508 uint32_t step = ((h * 17) & mask) | 1;
3509
3510 while (read32le(P: buf + i * 8))
3511 i = (i + step) & mask;
3512
3513 write32le(P: buf + i * 8, V: sym.nameOff);
3514 write32le(P: buf + i * 8 + 4, V: sym.cuVectorOff);
3515 }
3516
3517 buf += symtabSize * 8;
3518
3519 // Write the string pool.
3520 hdr->constantPoolOff = buf - start;
3521 parallelForEach(R&: symbols, Fn: [&](GdbSymbol &sym) {
3522 memcpy(dest: buf + sym.nameOff, src: sym.name.data(), n: sym.name.size());
3523 });
3524
3525 // Write the CU vectors.
3526 for (GdbSymbol &sym : symbols) {
3527 write32le(P: buf, V: sym.cuVector.size());
3528 buf += 4;
3529 for (uint32_t val : sym.cuVector) {
3530 write32le(P: buf, V: val);
3531 buf += 4;
3532 }
3533 }
3534}
3535
3536bool GdbIndexSection::isNeeded() const { return !chunks.empty(); }
3537
3538VersionDefinitionSection::VersionDefinitionSection(Ctx &ctx)
3539 : SyntheticSection(ctx, ".gnu.version_d", SHT_GNU_verdef, SHF_ALLOC,
3540 sizeof(uint32_t)) {}
3541
3542StringRef VersionDefinitionSection::getFileDefName() {
3543 if (!ctx.arg.soName.empty())
3544 return ctx.arg.soName;
3545 return ctx.arg.outputFile;
3546}
3547
3548void VersionDefinitionSection::finalizeContents() {
3549 fileDefNameOff = ctx.in.dynStrTab->addString(s: getFileDefName());
3550 for (const VersionDefinition &v : namedVersionDefs(ctx))
3551 verDefNameOffs.push_back(Elt: ctx.in.dynStrTab->addString(s: v.name));
3552
3553 if (OutputSection *sec = ctx.in.dynStrTab->getParent())
3554 getParent()->link = sec->sectionIndex;
3555
3556 // sh_info should be set to the number of definitions. This fact is missed in
3557 // documentation, but confirmed by binutils community:
3558 // https://sourceware.org/ml/binutils/2014-11/msg00355.html
3559 getParent()->info = getVerDefNum(ctx);
3560}
3561
3562void VersionDefinitionSection::writeOne(uint8_t *buf, uint32_t index,
3563 StringRef name, size_t nameOff) {
3564 uint16_t flags = index == 1 ? VER_FLG_BASE : 0;
3565
3566 // Write a verdef.
3567 write16(ctx, p: buf, v: 1); // vd_version
3568 write16(ctx, p: buf + 2, v: flags); // vd_flags
3569 write16(ctx, p: buf + 4, v: index); // vd_ndx
3570 write16(ctx, p: buf + 6, v: 1); // vd_cnt
3571 write32(ctx, p: buf + 8, v: hashSysV(SymbolName: name)); // vd_hash
3572 write32(ctx, p: buf + 12, v: 20); // vd_aux
3573 write32(ctx, p: buf + 16, v: 28); // vd_next
3574
3575 // Write a veraux.
3576 write32(ctx, p: buf + 20, v: nameOff); // vda_name
3577 write32(ctx, p: buf + 24, v: 0); // vda_next
3578}
3579
3580void VersionDefinitionSection::writeTo(uint8_t *buf) {
3581 writeOne(buf, index: 1, name: getFileDefName(), nameOff: fileDefNameOff);
3582
3583 auto nameOffIt = verDefNameOffs.begin();
3584 for (const VersionDefinition &v : namedVersionDefs(ctx)) {
3585 buf += EntrySize;
3586 writeOne(buf, index: v.id, name: v.name, nameOff: *nameOffIt++);
3587 }
3588
3589 // Need to terminate the last version definition.
3590 write32(ctx, p: buf + 16, v: 0); // vd_next
3591}
3592
3593size_t VersionDefinitionSection::getSize() const {
3594 return EntrySize * getVerDefNum(ctx);
3595}
3596
3597// .gnu.version is a table where each entry is 2 byte long.
3598VersionTableSection::VersionTableSection(Ctx &ctx)
3599 : SyntheticSection(ctx, ".gnu.version", SHT_GNU_versym, SHF_ALLOC,
3600 sizeof(uint16_t)) {
3601 this->entsize = 2;
3602}
3603
3604void VersionTableSection::finalizeContents() {
3605 if (OutputSection *osec = ctx.in.dynSymTab->getParent())
3606 getParent()->link = osec->sectionIndex;
3607}
3608
3609size_t VersionTableSection::getSize() const {
3610 return (ctx.in.dynSymTab->getSymbols().size() + 1) * 2;
3611}
3612
3613void VersionTableSection::writeTo(uint8_t *buf) {
3614 buf += 2;
3615 for (const SymbolTableEntry &s : ctx.in.dynSymTab->getSymbols()) {
3616 // For an unextracted lazy symbol (undefined weak), it must have been
3617 // converted to Undefined.
3618 assert(!s.sym->isLazy());
3619 // Undefined symbols should use index 0 when unversioned.
3620 write16(ctx, p: buf, v: s.sym->isUndefined() ? 0 : s.sym->versionId);
3621 buf += 2;
3622 }
3623}
3624
3625bool VersionTableSection::isNeeded() const {
3626 return isLive() && (ctx.in.verDef || ctx.in.verNeed->isNeeded());
3627}
3628
3629void elf::addVerneed(Ctx &ctx, Symbol &ss) {
3630 auto &file = cast<SharedFile>(Val&: *ss.file);
3631 if (ss.versionId == VER_NDX_GLOBAL)
3632 return;
3633
3634 if (file.verneedInfo.empty())
3635 file.verneedInfo.resize(N: file.verdefs.size());
3636
3637 // Select a version identifier for the vernaux data structure, if we haven't
3638 // already allocated one. The verdef identifiers cover the range
3639 // [1..getVerDefNum(ctx)]; this causes the vernaux identifiers to start from
3640 // getVerDefNum(ctx)+1.
3641 if (file.verneedInfo[ss.versionId].id == 0)
3642 file.verneedInfo[ss.versionId].id = ++ctx.vernauxNum + getVerDefNum(ctx);
3643 file.verneedInfo[ss.versionId].weak &= ss.isWeak();
3644
3645 ss.versionId = file.verneedInfo[ss.versionId].id;
3646}
3647
3648template <class ELFT>
3649VersionNeedSection<ELFT>::VersionNeedSection(Ctx &ctx)
3650 : SyntheticSection(ctx, ".gnu.version_r", SHT_GNU_verneed, SHF_ALLOC,
3651 sizeof(uint32_t)) {}
3652
3653template <class ELFT> void VersionNeedSection<ELFT>::finalizeContents() {
3654 for (SharedFile *f : ctx.sharedFiles) {
3655 if (f->verneedInfo.empty())
3656 continue;
3657 verneeds.emplace_back();
3658 Verneed &vn = verneeds.back();
3659 vn.nameStrTab = ctx.in.dynStrTab->addString(s: f->soName);
3660 bool isLibc = ctx.arg.relrGlibc && f->soName.starts_with(Prefix: "libc.so.");
3661 bool isGlibc2 = false;
3662 for (unsigned i = 0; i != f->verneedInfo.size(); ++i) {
3663 if (f->verneedInfo[i].id == 0)
3664 continue;
3665 // Each Verdef has one or more Verdaux entries. The first Verdaux gives
3666 // the version name; subsequent entries (if any) are parent versions
3667 // (e.g., v2 {} v1;). We only use the first one, as parent versions have
3668 // no rtld behavior difference in practice.
3669 auto *verdef =
3670 reinterpret_cast<const typename ELFT::Verdef *>(f->verdefs[i]);
3671 StringRef ver(f->getStringTable().data() + verdef->getAux()->vda_name);
3672 if (isLibc && ver.starts_with(Prefix: "GLIBC_2."))
3673 isGlibc2 = true;
3674 vn.vernauxs.push_back({verdef->vd_hash, f->verneedInfo[i],
3675 ctx.in.dynStrTab->addString(s: ver)});
3676 }
3677 if (isGlibc2) {
3678 const char *ver = "GLIBC_ABI_DT_RELR";
3679 vn.vernauxs.push_back(
3680 {hashSysV(SymbolName: ver),
3681 {uint16_t(++ctx.vernauxNum + getVerDefNum(ctx)), false},
3682 ctx.in.dynStrTab->addString(s: ver)});
3683 }
3684 }
3685
3686 if (OutputSection *sec = ctx.in.dynStrTab->getParent())
3687 getParent()->link = sec->sectionIndex;
3688 getParent()->info = verneeds.size();
3689}
3690
3691template <class ELFT> void VersionNeedSection<ELFT>::writeTo(uint8_t *buf) {
3692 // The Elf_Verneeds need to appear first, followed by the Elf_Vernauxs.
3693 auto *verneed = reinterpret_cast<Elf_Verneed *>(buf);
3694 auto *vernaux = reinterpret_cast<Elf_Vernaux *>(verneed + verneeds.size());
3695
3696 for (auto &vn : verneeds) {
3697 // Create an Elf_Verneed for this DSO.
3698 verneed->vn_version = 1;
3699 verneed->vn_cnt = vn.vernauxs.size();
3700 verneed->vn_file = vn.nameStrTab;
3701 verneed->vn_aux =
3702 reinterpret_cast<char *>(vernaux) - reinterpret_cast<char *>(verneed);
3703 verneed->vn_next = sizeof(Elf_Verneed);
3704 ++verneed;
3705
3706 // Create the Elf_Vernauxs for this Elf_Verneed.
3707 for (auto &vna : vn.vernauxs) {
3708 vernaux->vna_hash = vna.hash;
3709 vernaux->vna_flags = vna.verneedInfo.weak ? VER_FLG_WEAK : 0;
3710 vernaux->vna_other = vna.verneedInfo.id;
3711 vernaux->vna_name = vna.nameStrTab;
3712 vernaux->vna_next = sizeof(Elf_Vernaux);
3713 ++vernaux;
3714 }
3715
3716 vernaux[-1].vna_next = 0;
3717 }
3718 verneed[-1].vn_next = 0;
3719}
3720
3721template <class ELFT> size_t VersionNeedSection<ELFT>::getSize() const {
3722 return verneeds.size() * sizeof(Elf_Verneed) +
3723 ctx.vernauxNum * sizeof(Elf_Vernaux);
3724}
3725
3726template <class ELFT> bool VersionNeedSection<ELFT>::isNeeded() const {
3727 return isLive() && ctx.vernauxNum != 0;
3728}
3729
3730void MergeSyntheticSection::addSection(MergeInputSection *ms) {
3731 ms->parent = this;
3732 sections.push_back(Elt: ms);
3733 assert(addralign == ms->addralign || !(ms->flags & SHF_STRINGS));
3734 addralign = std::max(a: addralign, b: ms->addralign);
3735}
3736
3737MergeTailSection::MergeTailSection(Ctx &ctx, StringRef name, uint32_t type,
3738 uint64_t flags, uint32_t alignment)
3739 : MergeSyntheticSection(ctx, name, type, flags, alignment),
3740 builder(StringTableBuilder::RAW, llvm::Align(alignment)) {}
3741
3742size_t MergeTailSection::getSize() const { return builder.getSize(); }
3743
3744void MergeTailSection::writeTo(uint8_t *buf) { builder.write(Buf: buf); }
3745
3746void MergeTailSection::finalizeContents() {
3747 // Add all string pieces to the string table builder to create section
3748 // contents.
3749 for (MergeInputSection *sec : sections)
3750 for (size_t i = 0, e = sec->pieces.size(); i != e; ++i)
3751 if (sec->pieces[i].live)
3752 builder.add(S: sec->getData(i));
3753
3754 // Fix the string table content. After this, the contents will never change.
3755 builder.finalize();
3756
3757 // finalize() fixed tail-optimized strings, so we can now get
3758 // offsets of strings. Get an offset for each string and save it
3759 // to a corresponding SectionPiece for easy access.
3760 for (MergeInputSection *sec : sections)
3761 for (size_t i = 0, e = sec->pieces.size(); i != e; ++i)
3762 if (sec->pieces[i].live)
3763 sec->pieces[i].outputOff = builder.getOffset(S: sec->getData(i));
3764}
3765
3766void MergeNoTailSection::writeTo(uint8_t *buf) {
3767 parallelFor(Begin: 0, End: numShards,
3768 Fn: [&](size_t i) { shards[i].write(Buf: buf + shardOffsets[i]); });
3769}
3770
3771// This function is very hot (i.e. it can take several seconds to finish)
3772// because sometimes the number of inputs is in an order of magnitude of
3773// millions. So, we use multi-threading.
3774//
3775// For any strings S and T, we know S is not mergeable with T if S's hash
3776// value is different from T's. If that's the case, we can safely put S and
3777// T into different string builders without worrying about merge misses.
3778// We do it in parallel.
3779void MergeNoTailSection::finalizeContents() {
3780 // Initializes string table builders.
3781 for (size_t i = 0; i < numShards; ++i)
3782 shards.emplace_back(Args: StringTableBuilder::RAW, Args: llvm::Align(addralign));
3783
3784 // Concurrency level. Must be a power of 2 to avoid expensive modulo
3785 // operations in the following tight loop.
3786 const size_t concurrency =
3787 llvm::bit_floor(Value: std::min<size_t>(a: ctx.arg.threadCount, b: numShards));
3788
3789 // Add section pieces to the builders.
3790 parallelFor(Begin: 0, End: concurrency, Fn: [&](size_t threadId) {
3791 for (MergeInputSection *sec : sections) {
3792 for (size_t i = 0, e = sec->pieces.size(); i != e; ++i) {
3793 if (!sec->pieces[i].live)
3794 continue;
3795 size_t shardId = getShardId(hash: sec->pieces[i].hash);
3796 if ((shardId & (concurrency - 1)) == threadId)
3797 sec->pieces[i].outputOff = shards[shardId].add(S: sec->getData(i));
3798 }
3799 }
3800 });
3801
3802 // Compute an in-section offset for each shard.
3803 size_t off = 0;
3804 for (size_t i = 0; i < numShards; ++i) {
3805 shards[i].finalizeInOrder();
3806 if (shards[i].getSize() > 0)
3807 off = alignToPowerOf2(Value: off, Align: addralign);
3808 shardOffsets[i] = off;
3809 off += shards[i].getSize();
3810 }
3811 size = off;
3812
3813 // So far, section pieces have offsets from beginning of shards, but
3814 // we want offsets from beginning of the whole section. Fix them.
3815 parallelForEach(R&: sections, Fn: [&](MergeInputSection *sec) {
3816 for (SectionPiece &piece : sec->pieces)
3817 if (piece.live)
3818 piece.outputOff += shardOffsets[getShardId(hash: piece.hash)];
3819 });
3820}
3821
3822template <class ELFT> void elf::splitSections(Ctx &ctx) {
3823 llvm::TimeTraceScope timeScope("Split sections");
3824 // splitIntoPieces needs to be called on each MergeInputSection
3825 // before calling finalizeContents().
3826 parallelForEach(ctx.objectFiles, [](ELFFileBase *file) {
3827 for (InputSectionBase *sec : file->getSections()) {
3828 if (!sec)
3829 continue;
3830 if (auto *s = dyn_cast<MergeInputSection>(Val: sec))
3831 s->splitIntoPieces();
3832 else if (auto *eh = dyn_cast<EhInputSection>(Val: sec))
3833 eh->split<ELFT>();
3834 }
3835
3836 // For non-section Defined symbols in merge sections, pre-resolve the piece
3837 // index to avoid potentially repeated binary search (MarkLive, RelocScan,
3838 // includeInSymtab). Encode each non-section Defined symbol's value as
3839 // ((pieceIdx + 1) << mergeValueShift) | intraPieceOffset. A one-past-end
3840 // label is anchored on the last piece.
3841 auto resolve = [](Defined *d) {
3842 auto *ms = dyn_cast_or_null<MergeInputSection>(Val: d->section);
3843 if (!ms || d->isSection())
3844 return;
3845 uint64_t v = d->value;
3846 SectionPiece &piece = v >= ms->content().size() ? ms->pieces.back()
3847 : ms->getSectionPiece(offset: v);
3848 uint32_t idx = &piece - ms->pieces.data();
3849 uint64_t off = v - piece.inputOff;
3850 d->value = ((uint64_t)(idx + 1) << mergeValueShift) | off;
3851 };
3852 for (Symbol *sym : file->getLocalSymbols())
3853 if (auto *d = dyn_cast<Defined>(Val: sym))
3854 resolve(d);
3855 for (Symbol *sym : file->getGlobalSymbols())
3856 if (auto *d = dyn_cast<Defined>(Val: sym); d && d->file == file)
3857 resolve(d);
3858 });
3859}
3860
3861void elf::combineEhSections(Ctx &ctx) {
3862 llvm::TimeTraceScope timeScope("Combine EH sections");
3863 for (EhInputSection *sec : ctx.ehInputSections) {
3864 EhFrameSection &eh = *ctx.in.ehFrame;
3865 sec->parent = &eh;
3866 eh.addralign = std::max(a: eh.addralign, b: sec->addralign);
3867 eh.sections.push_back(Elt: sec);
3868 llvm::append_range(C&: eh.dependentSections, R&: sec->dependentSections);
3869 }
3870
3871 if (!ctx.in.armExidx)
3872 return;
3873 llvm::erase_if(C&: ctx.inputSections, P: [&](InputSectionBase *s) {
3874 if (!s->isLive())
3875 return false;
3876 return s->kind() == SectionBase::Regular &&
3877 ctx.in.armExidx->addSection(isec: cast<InputSection>(Val: s));
3878 });
3879}
3880
3881ARMExidxSyntheticSection::ARMExidxSyntheticSection(Ctx &ctx)
3882 : SyntheticSection(ctx, ".ARM.exidx", SHT_ARM_EXIDX,
3883 SHF_ALLOC | SHF_LINK_ORDER, ctx.arg.wordsize) {}
3884
3885static InputSection *findExidxSection(InputSection *isec) {
3886 for (InputSection *d : isec->dependentSections)
3887 if (d->type == SHT_ARM_EXIDX && d->isLive())
3888 return d;
3889 return nullptr;
3890}
3891
3892static bool isValidExidxSectionDep(InputSection *isec) {
3893 return (isec->flags & SHF_ALLOC) && (isec->flags & SHF_EXECINSTR) &&
3894 isec->getSize() > 0;
3895}
3896
3897bool ARMExidxSyntheticSection::addSection(InputSection *isec) {
3898 if (isec->type == SHT_ARM_EXIDX) {
3899 if (InputSection *dep = isec->getLinkOrderDep())
3900 if (isValidExidxSectionDep(isec: dep)) {
3901 exidxSections.push_back(Elt: isec);
3902 // Every exidxSection is 8 bytes, we need an estimate of
3903 // size before assignAddresses can be called. Final size
3904 // will only be known after finalize is called.
3905 size += 8;
3906 }
3907 return true;
3908 }
3909
3910 if (isValidExidxSectionDep(isec)) {
3911 executableSections.push_back(Elt: isec);
3912 return false;
3913 }
3914
3915 // FIXME: we do not output a relocation section when --emit-relocs is used
3916 // as we do not have relocation sections for linker generated table entries
3917 // and we would have to erase at a late stage relocations from merged entries.
3918 // Given that exception tables are already position independent and a binary
3919 // analyzer could derive the relocations we choose to erase the relocations.
3920 if (ctx.arg.emitRelocs && isec->type == SHT_REL)
3921 if (InputSectionBase *ex = isec->getRelocatedSection())
3922 if (isa<InputSection>(Val: ex) && ex->type == SHT_ARM_EXIDX)
3923 return true;
3924
3925 return false;
3926}
3927
3928// References to .ARM.Extab Sections have bit 31 clear and are not the
3929// special EXIDX_CANTUNWIND bit-pattern.
3930static bool isExtabRef(uint32_t unwind) {
3931 return (unwind & 0x80000000) == 0 && unwind != 0x1;
3932}
3933
3934// Return true if the .ARM.exidx section Cur can be merged into the .ARM.exidx
3935// section Prev, where Cur follows Prev in the table. This can be done if the
3936// unwinding instructions in Cur are identical to Prev. Linker generated
3937// EXIDX_CANTUNWIND entries are represented by nullptr as they do not have an
3938// InputSection.
3939static bool isDuplicateArmExidxSec(Ctx &ctx, InputSection *prev,
3940 InputSection *cur) {
3941 // Get the last table Entry from the previous .ARM.exidx section. If Prev is
3942 // nullptr then it will be a synthesized EXIDX_CANTUNWIND entry.
3943 uint32_t prevUnwind = 1;
3944 if (prev)
3945 prevUnwind =
3946 read32(ctx, p: prev->content().data() + prev->content().size() - 4);
3947 if (isExtabRef(unwind: prevUnwind))
3948 return false;
3949
3950 // We consider the unwind instructions of an .ARM.exidx table entry
3951 // a duplicate if the previous unwind instructions if:
3952 // - Both are the special EXIDX_CANTUNWIND.
3953 // - Both are the same inline unwind instructions.
3954 // We do not attempt to follow and check links into .ARM.extab tables as
3955 // consecutive identical entries are rare and the effort to check that they
3956 // are identical is high.
3957
3958 // If Cur is nullptr then this is synthesized EXIDX_CANTUNWIND entry.
3959 if (cur == nullptr)
3960 return prevUnwind == 1;
3961
3962 for (uint32_t offset = 4; offset < (uint32_t)cur->content().size(); offset +=8) {
3963 uint32_t curUnwind = read32(ctx, p: cur->content().data() + offset);
3964 if (isExtabRef(unwind: curUnwind) || curUnwind != prevUnwind)
3965 return false;
3966 }
3967 // All table entries in this .ARM.exidx Section can be merged into the
3968 // previous Section.
3969 return true;
3970}
3971
3972// The .ARM.exidx table must be sorted in ascending order of the address of the
3973// functions the table describes. std::optionally duplicate adjacent table
3974// entries can be removed. At the end of the function the executableSections
3975// must be sorted in ascending order of address, Sentinel is set to the
3976// InputSection with the highest address and any InputSections that have
3977// mergeable .ARM.exidx table entries are removed from it.
3978void ARMExidxSyntheticSection::finalizeContents() {
3979 // Ensure that any fixed-point iterations after the first see the original set
3980 // of sections.
3981 if (!originalExecutableSections.empty())
3982 executableSections = originalExecutableSections;
3983 else if (ctx.arg.enableNonContiguousRegions)
3984 originalExecutableSections = executableSections;
3985
3986 // The executableSections and exidxSections that we use to derive the final
3987 // contents of this SyntheticSection are populated before
3988 // processSectionCommands() and ICF. A /DISCARD/ entry in SECTIONS command or
3989 // ICF may remove executable InputSections and their dependent .ARM.exidx
3990 // section that we recorded earlier.
3991 auto isDiscarded = [](const InputSection *isec) { return !isec->isLive(); };
3992 llvm::erase_if(C&: exidxSections, P: isDiscarded);
3993 // We need to remove discarded InputSections and InputSections without
3994 // .ARM.exidx sections that if we generated the .ARM.exidx it would be out
3995 // of range.
3996 auto isDiscardedOrOutOfRange = [this](InputSection *isec) {
3997 if (!isec->isLive())
3998 return true;
3999 if (findExidxSection(isec))
4000 return false;
4001 int64_t off = static_cast<int64_t>(isec->getVA() - getVA());
4002 return off != llvm::SignExtend64(X: off, B: 31);
4003 };
4004 llvm::erase_if(C&: executableSections, P: isDiscardedOrOutOfRange);
4005
4006 // Sort the executable sections that may or may not have associated
4007 // .ARM.exidx sections by order of ascending address. This requires the
4008 // relative positions of InputSections and OutputSections to be known.
4009 auto compareByFilePosition = [](const InputSection *a,
4010 const InputSection *b) {
4011 OutputSection *aOut = a->getParent();
4012 OutputSection *bOut = b->getParent();
4013
4014 if (aOut != bOut)
4015 return aOut->addr < bOut->addr;
4016 return a->outSecOff < b->outSecOff;
4017 };
4018 llvm::stable_sort(Range&: executableSections, C: compareByFilePosition);
4019 sentinel = executableSections.back();
4020 // std::optionally merge adjacent duplicate entries.
4021 if (ctx.arg.mergeArmExidx) {
4022 SmallVector<InputSection *, 0> selectedSections;
4023 selectedSections.reserve(N: executableSections.size());
4024 selectedSections.push_back(Elt: executableSections[0]);
4025 size_t prev = 0;
4026 for (size_t i = 1; i < executableSections.size(); ++i) {
4027 InputSection *ex1 = findExidxSection(isec: executableSections[prev]);
4028 InputSection *ex2 = findExidxSection(isec: executableSections[i]);
4029 if (!isDuplicateArmExidxSec(ctx, prev: ex1, cur: ex2)) {
4030 selectedSections.push_back(Elt: executableSections[i]);
4031 prev = i;
4032 }
4033 }
4034 executableSections = std::move(selectedSections);
4035 }
4036 // offset is within the SyntheticSection.
4037 size_t offset = 0;
4038 size = 0;
4039 for (InputSection *isec : executableSections) {
4040 if (InputSection *d = findExidxSection(isec)) {
4041 d->outSecOff = offset;
4042 d->parent = getParent();
4043 offset += d->getSize();
4044 } else {
4045 offset += 8;
4046 }
4047 }
4048 // Size includes Sentinel.
4049 size = offset + 8;
4050}
4051
4052InputSection *ARMExidxSyntheticSection::getLinkOrderDep() const {
4053 return executableSections.front();
4054}
4055
4056// To write the .ARM.exidx table from the ExecutableSections we have three cases
4057// 1.) The InputSection has a .ARM.exidx InputSection in its dependent sections.
4058// We write the .ARM.exidx section contents and apply its relocations.
4059// 2.) The InputSection does not have a dependent .ARM.exidx InputSection. We
4060// must write the contents of an EXIDX_CANTUNWIND directly. We use the
4061// start of the InputSection as the purpose of the linker generated
4062// section is to terminate the address range of the previous entry.
4063// 3.) A trailing EXIDX_CANTUNWIND sentinel section is required at the end of
4064// the table to terminate the address range of the final entry.
4065void ARMExidxSyntheticSection::writeTo(uint8_t *buf) {
4066
4067 // A linker generated CANTUNWIND entry is made up of two words:
4068 // 0x0 with R_ARM_PREL31 relocation to target.
4069 // 0x1 with EXIDX_CANTUNWIND.
4070 uint64_t offset = 0;
4071 for (InputSection *isec : executableSections) {
4072 assert(isec->getParent() != nullptr);
4073 if (InputSection *d = findExidxSection(isec)) {
4074 for (int dataOffset = 0; dataOffset != (int)d->content().size();
4075 dataOffset += 4)
4076 write32(ctx, p: buf + offset + dataOffset,
4077 v: read32(ctx, p: d->content().data() + dataOffset));
4078 // Recalculate outSecOff as finalizeAddressDependentContent()
4079 // may have altered syntheticSection outSecOff.
4080 d->outSecOff = offset + outSecOff;
4081 ctx.target->relocateAlloc(sec&: *d, buf: buf + offset);
4082 offset += d->getSize();
4083 } else {
4084 // A Linker generated CANTUNWIND section.
4085 write32(ctx, p: buf + offset + 0, v: 0x0);
4086 write32(ctx, p: buf + offset + 4, v: 0x1);
4087 uint64_t s = isec->getVA();
4088 uint64_t p = getVA() + offset;
4089 ctx.target->relocateNoSym(loc: buf + offset, type: R_ARM_PREL31, val: s - p);
4090 offset += 8;
4091 }
4092 }
4093 // Write Sentinel CANTUNWIND entry.
4094 write32(ctx, p: buf + offset + 0, v: 0x0);
4095 write32(ctx, p: buf + offset + 4, v: 0x1);
4096 uint64_t s = sentinel->getVA(offset: sentinel->getSize());
4097 uint64_t p = getVA() + offset;
4098 ctx.target->relocateNoSym(loc: buf + offset, type: R_ARM_PREL31, val: s - p);
4099 assert(size == offset + 8);
4100}
4101
4102bool ARMExidxSyntheticSection::isNeeded() const {
4103 return llvm::any_of(Range: exidxSections,
4104 P: [](InputSection *isec) { return isec->isLive(); });
4105}
4106
4107ThunkSection::ThunkSection(Ctx &ctx, OutputSection *os, uint64_t off)
4108 : SyntheticSection(ctx, ".text.thunk", SHT_PROGBITS,
4109 SHF_ALLOC | SHF_EXECINSTR,
4110 ctx.arg.emachine == EM_PPC64 ? 16 : 4) {
4111 this->parent = os;
4112 this->outSecOff = off;
4113}
4114
4115size_t ThunkSection::getSize() const {
4116 if (roundUpSizeForErrata)
4117 return alignTo(Value: size, Align: 4096);
4118 return size;
4119}
4120
4121void ThunkSection::addThunk(Thunk *t) {
4122 thunks.push_back(Elt: t);
4123 t->addSymbols(isec&: *this);
4124}
4125
4126void ThunkSection::writeTo(uint8_t *buf) {
4127 for (Thunk *t : thunks)
4128 t->writeTo(buf: buf + t->offset);
4129}
4130
4131InputSection *ThunkSection::getTargetInputSection() const {
4132 if (thunks.empty())
4133 return nullptr;
4134 const Thunk *t = thunks.front();
4135 return t->getTargetInputSection();
4136}
4137
4138// Move forward thunks to the right half and sort them by destination VA:
4139//
4140// dstA, dstB, [backward A, B], [forward D, C], dstC, dstD
4141//
4142// A forward thunk's distance grows when a thunk after it grows. Ordering
4143// forward thunks by descending destination keeps the most promotable ones
4144// lowest, where their growth stays below the rest. A backward thunk's distance
4145// grows only with a promotion before it, already applied by
4146// ThunkSection::assignOffsets when we reach it, so backward thunks need no
4147// ordering and stay ahead of forward thunks in creation order.
4148void ThunkSection::sortByDestination() {
4149 uint64_t base = getVA();
4150 SmallVector<std::pair<uint64_t, Thunk *>, 0> keys;
4151 keys.resize_for_overwrite(N: thunks.size());
4152 for (auto [i, t] : enumerate(First&: thunks))
4153 keys[i] = {t->getDestVA(), t};
4154 auto *forward =
4155 std::stable_partition(first: keys.begin(), last: keys.end(),
4156 pred: [base](const auto &k) { return k.first <= base; });
4157 std::stable_sort(first: forward, last: keys.end(), comp: [](const auto &a, const auto &b) {
4158 return a.first > b.first;
4159 });
4160 for (auto [i, p] : llvm::enumerate(First&: keys))
4161 thunks[i] = p.second;
4162}
4163
4164bool ThunkSection::assignOffsets() {
4165 uint64_t off = 0;
4166 bool changed = false;
4167 for (Thunk *t : thunks) {
4168 if (t->alignment > addralign) {
4169 addralign = t->alignment;
4170 changed = true;
4171 }
4172 off = alignToPowerOf2(Value: off, Align: t->alignment);
4173 t->setOffset(off);
4174 uint32_t size = t->size();
4175 t->getThunkTargetSym()->size = size;
4176 off += size;
4177 }
4178 if (off != size)
4179 changed = true;
4180 size = off;
4181 return changed;
4182}
4183
4184// If linking position-dependent code then the table will store the addresses
4185// directly in the binary so the section has type SHT_PROGBITS. If linking
4186// position-independent code the section has type SHT_NOBITS since it will be
4187// allocated and filled in by the dynamic linker.
4188PPC64LongBranchTargetSection::PPC64LongBranchTargetSection(Ctx &ctx)
4189 : SyntheticSection(ctx, ".branch_lt",
4190 ctx.arg.isPic ? SHT_NOBITS : SHT_PROGBITS,
4191 SHF_ALLOC | SHF_WRITE, 8) {}
4192
4193uint64_t PPC64LongBranchTargetSection::getEntryVA(const Symbol *sym,
4194 int64_t addend) {
4195 return getVA() + entry_index.find(Val: {sym, addend})->second * 8;
4196}
4197
4198std::optional<uint32_t>
4199PPC64LongBranchTargetSection::addEntry(const Symbol *sym, int64_t addend) {
4200 auto res =
4201 entry_index.try_emplace(Key: std::make_pair(x&: sym, y&: addend), Args: entries.size());
4202 if (!res.second)
4203 return std::nullopt;
4204 entries.emplace_back(Args&: sym, Args&: addend);
4205 return res.first->second;
4206}
4207
4208size_t PPC64LongBranchTargetSection::getSize() const {
4209 return entries.size() * 8;
4210}
4211
4212void PPC64LongBranchTargetSection::writeTo(uint8_t *buf) {
4213 // If linking non-pic we have the final addresses of the targets and they get
4214 // written to the table directly. For pic the dynamic linker will allocate
4215 // the section and fill it.
4216 if (ctx.arg.isPic)
4217 return;
4218
4219 for (auto entry : entries) {
4220 const Symbol *sym = entry.first;
4221 int64_t addend = entry.second;
4222 assert(sym->getVA(ctx));
4223 // Need calls to branch to the local entry-point since a long-branch
4224 // must be a local-call.
4225 write64(ctx, p: buf,
4226 v: sym->getVA(ctx, addend) +
4227 getPPC64GlobalEntryToLocalEntryOffset(ctx, stOther: sym->stOther));
4228 buf += 8;
4229 }
4230}
4231
4232bool PPC64LongBranchTargetSection::isNeeded() const {
4233 // `removeUnusedSyntheticSections()` is called before thunk allocation which
4234 // is too early to determine if this section will be empty or not. We need
4235 // Finalized to keep the section alive until after thunk creation. Finalized
4236 // only gets set to true once `finalizeSections()` is called after thunk
4237 // creation. Because of this, if we don't create any long-branch thunks we end
4238 // up with an empty .branch_lt section in the binary.
4239 return !finalized || !entries.empty();
4240}
4241
4242static uint8_t getAbiVersion(Ctx &ctx) {
4243 // MIPS non-PIC executable gets ABI version 1.
4244 if (ctx.arg.emachine == EM_MIPS) {
4245 if (!ctx.arg.isPic && !ctx.arg.relocatable &&
4246 (ctx.arg.eflags & (EF_MIPS_PIC | EF_MIPS_CPIC)) == EF_MIPS_CPIC)
4247 return 1;
4248 return 0;
4249 }
4250
4251 if (ctx.arg.emachine == EM_AMDGPU && !ctx.objectFiles.empty()) {
4252 uint8_t ver = ctx.objectFiles[0]->abiVersion;
4253 for (InputFile *file : ArrayRef(ctx.objectFiles).slice(N: 1))
4254 if (file->abiVersion != ver)
4255 Err(ctx) << "incompatible ABI version: " << file;
4256 return ver;
4257 }
4258
4259 return 0;
4260}
4261
4262template <typename ELFT> void elf::writeEhdr(Ctx &ctx, uint8_t *buf) {
4263 memcpy(dest: buf, src: "\177ELF", n: 4);
4264
4265 auto *eHdr = reinterpret_cast<typename ELFT::Ehdr *>(buf);
4266 eHdr->e_ident[EI_CLASS] = ELFT::Is64Bits ? ELFCLASS64 : ELFCLASS32;
4267 eHdr->e_ident[EI_DATA] =
4268 ELFT::Endianness == endianness::little ? ELFDATA2LSB : ELFDATA2MSB;
4269 eHdr->e_ident[EI_VERSION] = EV_CURRENT;
4270 eHdr->e_ident[EI_OSABI] = ctx.arg.osabi;
4271 eHdr->e_ident[EI_ABIVERSION] = getAbiVersion(ctx);
4272 eHdr->e_machine = ctx.arg.emachine;
4273 eHdr->e_version = EV_CURRENT;
4274 eHdr->e_flags = ctx.arg.eflags;
4275 eHdr->e_ehsize = sizeof(typename ELFT::Ehdr);
4276 eHdr->e_phnum = ctx.phdrs.size();
4277 eHdr->e_shentsize = sizeof(typename ELFT::Shdr);
4278
4279 if (!ctx.arg.relocatable) {
4280 eHdr->e_phoff = sizeof(typename ELFT::Ehdr);
4281 eHdr->e_phentsize = sizeof(typename ELFT::Phdr);
4282 }
4283}
4284
4285template <typename ELFT> void elf::writePhdrs(Ctx &ctx, uint8_t *buf) {
4286 // Write the program header table.
4287 auto *hBuf = reinterpret_cast<typename ELFT::Phdr *>(buf);
4288 for (std::unique_ptr<PhdrEntry> &p : ctx.phdrs) {
4289 hBuf->p_type = p->p_type;
4290 hBuf->p_flags = p->p_flags;
4291 hBuf->p_offset = p->p_offset;
4292 hBuf->p_vaddr = p->p_vaddr;
4293 hBuf->p_paddr = p->p_paddr;
4294 hBuf->p_filesz = p->p_filesz;
4295 hBuf->p_memsz = p->p_memsz;
4296 hBuf->p_align = p->p_align;
4297 ++hBuf;
4298 }
4299}
4300
4301static bool needsInterpSection(Ctx &ctx) {
4302 return !ctx.arg.relocatable && !ctx.arg.shared &&
4303 !ctx.arg.dynamicLinker.empty() && ctx.script->needsInterpSection();
4304}
4305
4306bool elf::hasMemtag(Ctx &ctx) {
4307 return ctx.arg.emachine == EM_AARCH64 &&
4308 ctx.arg.memtagMode != ELF::NT_MEMTAG_LEVEL_NONE;
4309}
4310
4311// Fully static executables don't support MTE globals at this point in time, as
4312// we currently rely on:
4313// - A dynamic loader to process relocations, and
4314// - Dynamic entries.
4315// This restriction could be removed in future by re-using some of the ideas
4316// that ifuncs use in fully static executables.
4317bool elf::canHaveMemtagGlobals(Ctx &ctx) {
4318 return hasMemtag(ctx) &&
4319 (ctx.arg.relocatable || ctx.arg.shared || needsInterpSection(ctx));
4320}
4321
4322constexpr char kMemtagAndroidNoteName[] = "Android";
4323void MemtagAndroidNote::writeTo(uint8_t *buf) {
4324 static_assert(
4325 sizeof(kMemtagAndroidNoteName) == 8,
4326 "Android 11 & 12 have an ABI that the note name is 8 bytes long. Keep it "
4327 "that way for backwards compatibility.");
4328
4329 write32(ctx, p: buf, v: sizeof(kMemtagAndroidNoteName));
4330 write32(ctx, p: buf + 4, v: sizeof(uint32_t));
4331 write32(ctx, p: buf + 8, v: ELF::NT_ANDROID_TYPE_MEMTAG);
4332 memcpy(dest: buf + 12, src: kMemtagAndroidNoteName, n: sizeof(kMemtagAndroidNoteName));
4333 buf += 12 + alignTo(Value: sizeof(kMemtagAndroidNoteName), Align: 4);
4334
4335 uint32_t value = 0;
4336 value |= ctx.arg.memtagMode;
4337 if (ctx.arg.memtagHeap)
4338 value |= ELF::NT_MEMTAG_HEAP;
4339 // Note, MTE stack is an ABI break. Attempting to run an MTE stack-enabled
4340 // binary on Android 11 or 12 will result in a checkfail in the loader.
4341 if (ctx.arg.memtagStack)
4342 value |= ELF::NT_MEMTAG_STACK;
4343 write32(ctx, p: buf, v: value); // note value
4344}
4345
4346size_t MemtagAndroidNote::getSize() const {
4347 return sizeof(llvm::ELF::Elf64_Nhdr) +
4348 /*namesz=*/alignTo(Value: sizeof(kMemtagAndroidNoteName), Align: 4) +
4349 /*descsz=*/sizeof(uint32_t);
4350}
4351
4352void PackageMetadataNote::writeTo(uint8_t *buf) {
4353 write32(ctx, p: buf, v: 4);
4354 write32(ctx, p: buf + 4, v: ctx.arg.packageMetadata.size() + 1);
4355 write32(ctx, p: buf + 8, v: FDO_PACKAGING_METADATA);
4356 memcpy(dest: buf + 12, src: "FDO", n: 4);
4357 memcpy(dest: buf + 16, src: ctx.arg.packageMetadata.data(),
4358 n: ctx.arg.packageMetadata.size());
4359}
4360
4361size_t PackageMetadataNote::getSize() const {
4362 return sizeof(llvm::ELF::Elf64_Nhdr) + 4 +
4363 alignTo(Value: ctx.arg.packageMetadata.size() + 1, Align: 4);
4364}
4365
4366// Helper function, return the size of the ULEB128 for 'v', optionally writing
4367// it to `*(buf + offset)` if `buf` is non-null.
4368static size_t computeOrWriteULEB128(uint64_t v, uint8_t *buf, size_t offset) {
4369 if (buf)
4370 return encodeULEB128(Value: v, p: buf + offset);
4371 return getULEB128Size(Value: v);
4372}
4373
4374// https://github.com/ARM-software/abi-aa/blob/main/memtagabielf64/memtagabielf64.rst#83encoding-of-sht_aarch64_memtag_globals_dynamic
4375constexpr uint64_t kMemtagStepSizeBits = 3;
4376constexpr uint64_t kMemtagGranuleSize = 16;
4377static size_t
4378createMemtagGlobalDescriptors(Ctx &ctx,
4379 const SmallVector<const Symbol *, 0> &symbols,
4380 uint8_t *buf = nullptr) {
4381 size_t sectionSize = 0;
4382 uint64_t lastGlobalEnd = 0;
4383
4384 for (const Symbol *sym : symbols) {
4385 if (!includeInSymtab(ctx, *sym))
4386 continue;
4387 const uint64_t addr = sym->getVA(ctx);
4388 const uint64_t size = sym->getSize();
4389
4390 if (addr <= kMemtagGranuleSize && buf != nullptr)
4391 Err(ctx) << "address of the tagged symbol \"" << sym->getName()
4392 << "\" falls in the ELF header. This is indicative of a "
4393 "compiler/linker bug";
4394 if (addr % kMemtagGranuleSize != 0)
4395 Err(ctx) << "address of the tagged symbol \"" << sym->getName()
4396 << "\" at 0x" << Twine::utohexstr(Val: addr)
4397 << "\" is not granule (16-byte) aligned";
4398 if (size == 0)
4399 Err(ctx) << "size of the tagged symbol \"" << sym->getName()
4400 << "\" is not allowed to be zero";
4401 if (size % kMemtagGranuleSize != 0)
4402 Err(ctx) << "size of the tagged symbol \"" << sym->getName()
4403 << "\" (size 0x" << Twine::utohexstr(Val: size)
4404 << ") is not granule (16-byte) aligned";
4405
4406 const uint64_t sizeToEncode = size / kMemtagGranuleSize;
4407 const uint64_t stepToEncode = ((addr - lastGlobalEnd) / kMemtagGranuleSize)
4408 << kMemtagStepSizeBits;
4409 if (sizeToEncode < (1 << kMemtagStepSizeBits)) {
4410 sectionSize += computeOrWriteULEB128(v: stepToEncode | sizeToEncode, buf, offset: sectionSize);
4411 } else {
4412 sectionSize += computeOrWriteULEB128(v: stepToEncode, buf, offset: sectionSize);
4413 sectionSize += computeOrWriteULEB128(v: sizeToEncode - 1, buf, offset: sectionSize);
4414 }
4415 lastGlobalEnd = addr + size;
4416 }
4417
4418 return sectionSize;
4419}
4420
4421bool MemtagGlobalDescriptors::updateAllocSize(Ctx &ctx) {
4422 size_t oldSize = getSize();
4423 llvm::stable_sort(Range&: symbols, C: [&ctx = ctx](const Symbol *s1, const Symbol *s2) {
4424 return s1->getVA(ctx) < s2->getVA(ctx);
4425 });
4426 return oldSize != getSize();
4427}
4428
4429void MemtagGlobalDescriptors::writeTo(uint8_t *buf) {
4430 createMemtagGlobalDescriptors(ctx, symbols, buf);
4431}
4432
4433size_t MemtagGlobalDescriptors::getSize() const {
4434 return createMemtagGlobalDescriptors(ctx, symbols);
4435}
4436
4437DynamicDebugSection::DynamicDebugSection(Ctx &ctx)
4438 : SyntheticSection(ctx, dynDbgSecName, SHT_LLVM_DYNDBG_ELF, 0, 8) {
4439 assert(ctx.dynDbgOutput);
4440}
4441
4442size_t DynamicDebugSection::getSize() const {
4443 return ctx.dynDbgOutput->getBufferSize();
4444}
4445
4446void DynamicDebugSection::writeTo(uint8_t *buf) {
4447 memcpy(dest: buf, src: ctx.dynDbgOutput->getBufferStart(),
4448 n: ctx.dynDbgOutput->getBufferSize());
4449}
4450
4451constexpr char dynDbgNoteName[] = "LLVM";
4452
4453DynamicDebugNote::DynamicDebugNote(Ctx &ctx)
4454 : SyntheticSection(ctx, ".note.llvm.dyndbg", SHT_NOTE, 0, 4) {}
4455
4456size_t DynamicDebugNote::getSize() const {
4457 return sizeof(llvm::ELF::Elf64_Nhdr) + alignTo(Value: sizeof(dynDbgNoteName), Align: 4) +
4458 /*descsz=*/sizeof(uint32_t);
4459}
4460
4461void DynamicDebugNote::writeTo(uint8_t *buf) {
4462 write32(ctx, p: buf, v: sizeof(dynDbgNoteName)); // Name size
4463 write32(ctx, p: buf + 4, v: sizeof(uint32_t)); // Content size
4464 write32(ctx, p: buf + 8, v: NT_LLVM_DYNAMIC_DEBUGGING); // Type
4465 memcpy(dest: buf + 12, src: dynDbgNoteName, n: sizeof(dynDbgNoteName));
4466 write32(ctx, p: buf + 12 + alignTo(Value: sizeof(dynDbgNoteName), Align: 4), v: 0); // Version
4467}
4468
4469static OutputSection *findSection(Ctx &ctx, StringRef name) {
4470 for (SectionCommand *cmd : ctx.script->sectionCommands)
4471 if (auto *osd = dyn_cast<OutputDesc>(Val: cmd))
4472 if (osd->osec.name == name)
4473 return &osd->osec;
4474 return nullptr;
4475}
4476
4477template <class ELFT> void elf::createSyntheticSections(Ctx &ctx) {
4478 // Add the .interp section first because it is not a SyntheticSection.
4479 // The removeUnusedSyntheticSections() function relies on the
4480 // SyntheticSections coming last.
4481 if (needsInterpSection(ctx)) {
4482 InputSection *sec = createInterpSection(ctx);
4483 sec->partition = 1;
4484 ctx.inputSections.push_back(Elt: sec);
4485 }
4486
4487 auto add = [&](SyntheticSection &sec) { ctx.inputSections.push_back(Elt: &sec); };
4488
4489 if (ctx.arg.zSectionHeader)
4490 ctx.in.shStrTab =
4491 std::make_unique<StringTableSection>(args&: ctx, args: ".shstrtab", args: false);
4492
4493 ctx.out.programHeaders =
4494 std::make_unique<OutputSection>(args&: ctx, args: "", args: 0, args: SHF_ALLOC);
4495 ctx.out.programHeaders->addralign = ctx.arg.wordsize;
4496
4497 if (ctx.arg.strip != StripPolicy::All) {
4498 ctx.in.strTab = std::make_unique<StringTableSection>(args&: ctx, args: ".strtab", args: false);
4499 ctx.in.symTab =
4500 std::make_unique<SymbolTableSection<ELFT>>(ctx, *ctx.in.strTab);
4501 ctx.in.symTabShndx = std::make_unique<SymtabShndxSection>(args&: ctx);
4502 }
4503
4504 ctx.in.bss = std::make_unique<BssSection>(args&: ctx, args: ".bss", args: 0, args: 1);
4505 add(*ctx.in.bss);
4506
4507 // If there is a SECTIONS command and a .data.rel.ro section name use name
4508 // .data.rel.ro.bss so that we match in the .data.rel.ro output section.
4509 // This makes sure our relro is contiguous.
4510 bool hasDataRelRo =
4511 ctx.script->hasSectionsCommand && findSection(ctx, name: ".data.rel.ro");
4512 ctx.in.bssRelRo = std::make_unique<BssSection>(
4513 args&: ctx, args: hasDataRelRo ? ".data.rel.ro.bss" : ".bss.rel.ro", args: 0, args: 1);
4514 add(*ctx.in.bssRelRo);
4515
4516 ctx.target->initTargetSpecificSections();
4517
4518 StringRef relaDynName = ctx.arg.isRela ? ".rela.dyn" : ".rel.dyn";
4519
4520 const unsigned threadCount = ctx.arg.threadCount;
4521 do {
4522 if (ctx.arg.buildId != BuildIdKind::None) {
4523 ctx.in.buildId = std::make_unique<BuildIdSection>(args&: ctx);
4524 add(*ctx.in.buildId);
4525 }
4526
4527 // dynSymTab is always present to simplify several finalizeSections
4528 // functions.
4529 ctx.in.dynStrTab =
4530 std::make_unique<StringTableSection>(args&: ctx, args: ".dynstr", args: true);
4531 ctx.in.dynSymTab =
4532 std::make_unique<SymbolTableSection<ELFT>>(ctx, *ctx.in.dynStrTab);
4533
4534 if (ctx.arg.relocatable)
4535 break;
4536 ctx.in.dynamic = std::make_unique<DynamicSection<ELFT>>(ctx);
4537
4538 if (hasMemtag(ctx)) {
4539 if (ctx.arg.memtagAndroidNote) {
4540 ctx.in.memtagAndroidNote = std::make_unique<MemtagAndroidNote>(args&: ctx);
4541 add(*ctx.in.memtagAndroidNote);
4542 }
4543 if (canHaveMemtagGlobals(ctx)) {
4544 ctx.in.memtagGlobalDescriptors =
4545 std::make_unique<MemtagGlobalDescriptors>(args&: ctx);
4546 add(*ctx.in.memtagGlobalDescriptors);
4547 }
4548 }
4549
4550 if (ctx.arg.androidPackDynRelocs)
4551 ctx.in.relaDyn = std::make_unique<AndroidPackedRelocationSection<ELFT>>(
4552 ctx, relaDynName, threadCount);
4553 else
4554 ctx.in.relaDyn = std::make_unique<RelocationSection<ELFT>>(
4555 ctx, relaDynName, /*combreloc=*/true, threadCount);
4556
4557 if (ctx.hasDynsym) {
4558 add(*ctx.in.dynSymTab);
4559
4560 ctx.in.verSym = std::make_unique<VersionTableSection>(args&: ctx);
4561 add(*ctx.in.verSym);
4562
4563 if (!namedVersionDefs(ctx).empty()) {
4564 ctx.in.verDef = std::make_unique<VersionDefinitionSection>(args&: ctx);
4565 add(*ctx.in.verDef);
4566 }
4567
4568 ctx.in.verNeed = std::make_unique<VersionNeedSection<ELFT>>(ctx);
4569 add(*ctx.in.verNeed);
4570
4571 if (ctx.arg.gnuHash) {
4572 ctx.in.gnuHashTab = std::make_unique<GnuHashTableSection>(args&: ctx);
4573 add(*ctx.in.gnuHashTab);
4574 }
4575
4576 if (ctx.arg.sysvHash) {
4577 ctx.in.hashTab = std::make_unique<HashTableSection>(args&: ctx);
4578 add(*ctx.in.hashTab);
4579 }
4580
4581 add(*ctx.in.dynamic);
4582 add(*ctx.in.dynStrTab);
4583 }
4584 add(*ctx.in.relaDyn);
4585
4586 if (ctx.arg.relrPackDynRelocs) {
4587 ctx.in.relrDyn = std::make_unique<RelrSection<ELFT>>(ctx, threadCount);
4588 add(*ctx.in.relrDyn);
4589 ctx.in.relrAuthDyn = std::make_unique<RelrSection<ELFT>>(
4590 ctx, threadCount, /*isAArch64Auth=*/true);
4591 add(*ctx.in.relrAuthDyn);
4592 }
4593
4594 if (ctx.arg.ehFrameHdr) {
4595 ctx.in.ehFrameHdr = std::make_unique<EhFrameHeader>(args&: ctx);
4596 add(*ctx.in.ehFrameHdr);
4597 }
4598 ctx.in.ehFrame = std::make_unique<EhFrameSection>(args&: ctx);
4599 add(*ctx.in.ehFrame);
4600
4601 if (ctx.arg.emachine == EM_ARM) {
4602 // This section replaces all the individual .ARM.exidx InputSections.
4603 ctx.in.armExidx = std::make_unique<ARMExidxSyntheticSection>(args&: ctx);
4604 add(*ctx.in.armExidx);
4605 }
4606
4607 if (!ctx.arg.packageMetadata.empty()) {
4608 ctx.in.packageMetadataNote = std::make_unique<PackageMetadataNote>(args&: ctx);
4609 add(*ctx.in.packageMetadataNote);
4610 }
4611 } while (0);
4612
4613 // Add .got. MIPS' .got is so different from the other archs,
4614 // it has its own class.
4615 if (ctx.arg.emachine == EM_MIPS) {
4616 ctx.in.mipsGot = std::make_unique<MipsGotSection>(args&: ctx);
4617 add(*ctx.in.mipsGot);
4618 } else {
4619 ctx.in.got = std::make_unique<GotSection>(args&: ctx);
4620 add(*ctx.in.got);
4621 }
4622
4623 ctx.in.gotPlt = std::make_unique<GotPltSection>(args&: ctx);
4624 add(*ctx.in.gotPlt);
4625 ctx.in.igotPlt = std::make_unique<IgotPltSection>(args&: ctx);
4626 add(*ctx.in.igotPlt);
4627 // Add .relro_padding if DATA_SEGMENT_RELRO_END is used; otherwise, add the
4628 // section in the absence of PHDRS/SECTIONS commands.
4629 if (ctx.arg.zRelro &&
4630 ((!ctx.script->hasPhdrsCommands() && !ctx.script->hasSectionsCommand) ||
4631 ctx.script->seenRelroEnd)) {
4632 ctx.in.relroPadding = std::make_unique<RelroPaddingSection>(args&: ctx);
4633 add(*ctx.in.relroPadding);
4634 }
4635
4636 // _GLOBAL_OFFSET_TABLE_ is defined relative to either .got.plt or .got. Treat
4637 // it as a relocation and ensure the referenced section is created.
4638 if (ctx.sym.globalOffsetTable && ctx.arg.emachine != EM_MIPS) {
4639 if (ctx.target->gotBaseSymInGotPlt)
4640 ctx.in.gotPlt->hasGotPltOffRel = true;
4641 else
4642 ctx.in.got->hasGotOffRel = true;
4643 }
4644
4645 // We always need to add rel[a].plt to output if it has entries.
4646 // Even for static linking it can contain R_[*]_IRELATIVE relocations.
4647 ctx.in.relaPlt = std::make_unique<RelocationSection<ELFT>>(
4648 ctx, ctx.arg.isRela ? ".rela.plt" : ".rel.plt", /*sort=*/false,
4649 /*threadCount=*/1);
4650 add(*ctx.in.relaPlt);
4651
4652 if (ctx.arg.emachine == EM_PPC)
4653 ctx.in.plt = std::make_unique<PPC32GlinkSection>(args&: ctx);
4654 else
4655 ctx.in.plt = std::make_unique<PltSection>(args&: ctx);
4656 add(*ctx.in.plt);
4657 ctx.in.iplt = std::make_unique<IpltSection>(args&: ctx);
4658 add(*ctx.in.iplt);
4659
4660 if (ctx.arg.andFeatures || ctx.aarch64PauthAbiCoreInfo) {
4661 ctx.in.gnuProperty = std::make_unique<GnuPropertySection>(args&: ctx);
4662 add(*ctx.in.gnuProperty);
4663 }
4664
4665 if (ctx.arg.debugNames) {
4666 ctx.in.debugNames = std::make_unique<DebugNamesSection<ELFT>>(ctx);
4667 add(*ctx.in.debugNames);
4668 }
4669
4670 if (ctx.arg.gdbIndex) {
4671 ctx.in.gdbIndex = GdbIndexSection::create<ELFT>(ctx);
4672 add(*ctx.in.gdbIndex);
4673 }
4674
4675 // .note.GNU-stack is always added when we are creating a re-linkable
4676 // object file. Other linkers are using the presence of this marker
4677 // section to control the executable-ness of the stack area, but that
4678 // is irrelevant these days. Stack area should always be non-executable
4679 // by default. So we emit this section unconditionally.
4680 if (ctx.arg.relocatable) {
4681 ctx.in.gnuStack = std::make_unique<GnuStackSection>(args&: ctx);
4682 add(*ctx.in.gnuStack);
4683 }
4684
4685 if (ctx.in.symTab)
4686 add(*ctx.in.symTab);
4687 if (ctx.in.symTabShndx)
4688 add(*ctx.in.symTabShndx);
4689 if (ctx.in.shStrTab)
4690 add(*ctx.in.shStrTab);
4691 if (ctx.in.strTab)
4692 add(*ctx.in.strTab);
4693
4694 if (ctx.dynDbgOutput) {
4695 ctx.in.dynDbg = std::make_unique<DynamicDebugSection>(args&: ctx);
4696 add(*ctx.in.dynDbg);
4697 if (!ctx.arg.relocatable) {
4698 ctx.in.dynDbgNote = std::make_unique<DynamicDebugNote>(args&: ctx);
4699 add(*ctx.in.dynDbgNote);
4700 }
4701 }
4702}
4703
4704template void elf::splitSections<ELF32LE>(Ctx &);
4705template void elf::splitSections<ELF32BE>(Ctx &);
4706template void elf::splitSections<ELF64LE>(Ctx &);
4707template void elf::splitSections<ELF64BE>(Ctx &);
4708
4709template void EhFrameSection::iterateFDEWithLSDA<ELF32LE>(
4710 function_ref<void(InputSection &)>);
4711template void EhFrameSection::iterateFDEWithLSDA<ELF32BE>(
4712 function_ref<void(InputSection &)>);
4713template void EhFrameSection::iterateFDEWithLSDA<ELF64LE>(
4714 function_ref<void(InputSection &)>);
4715template void EhFrameSection::iterateFDEWithLSDA<ELF64BE>(
4716 function_ref<void(InputSection &)>);
4717
4718template class elf::SymbolTableSection<ELF32LE>;
4719template class elf::SymbolTableSection<ELF32BE>;
4720template class elf::SymbolTableSection<ELF64LE>;
4721template class elf::SymbolTableSection<ELF64BE>;
4722
4723template void elf::writeEhdr<ELF32LE>(Ctx &, uint8_t *Buf);
4724template void elf::writeEhdr<ELF32BE>(Ctx &, uint8_t *Buf);
4725template void elf::writeEhdr<ELF64LE>(Ctx &, uint8_t *Buf);
4726template void elf::writeEhdr<ELF64BE>(Ctx &, uint8_t *Buf);
4727
4728template void elf::writePhdrs<ELF32LE>(Ctx &, uint8_t *Buf);
4729template void elf::writePhdrs<ELF32BE>(Ctx &, uint8_t *Buf);
4730template void elf::writePhdrs<ELF64LE>(Ctx &, uint8_t *Buf);
4731template void elf::writePhdrs<ELF64BE>(Ctx &, uint8_t *Buf);
4732
4733template void elf::createSyntheticSections<ELF32LE>(Ctx &);
4734template void elf::createSyntheticSections<ELF32BE>(Ctx &);
4735template void elf::createSyntheticSections<ELF64LE>(Ctx &);
4736template void elf::createSyntheticSections<ELF64BE>(Ctx &);
4737