1 //===- LinkerScript.cpp ---------------------------------------------------===// 2 // 3 // The LLVM Linker 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //===----------------------------------------------------------------------===// 9 // 10 // This file contains the parser/evaluator of the linker script. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "LinkerScript.h" 15 #include "Config.h" 16 #include "Driver.h" 17 #include "InputSection.h" 18 #include "Memory.h" 19 #include "OutputSections.h" 20 #include "ScriptLexer.h" 21 #include "Strings.h" 22 #include "SymbolTable.h" 23 #include "Symbols.h" 24 #include "SyntheticSections.h" 25 #include "Target.h" 26 #include "Writer.h" 27 #include "llvm/ADT/STLExtras.h" 28 #include "llvm/ADT/SmallString.h" 29 #include "llvm/ADT/StringRef.h" 30 #include "llvm/ADT/StringSwitch.h" 31 #include "llvm/Support/Casting.h" 32 #include "llvm/Support/ELF.h" 33 #include "llvm/Support/Endian.h" 34 #include "llvm/Support/ErrorHandling.h" 35 #include "llvm/Support/FileSystem.h" 36 #include "llvm/Support/MathExtras.h" 37 #include "llvm/Support/Path.h" 38 #include <algorithm> 39 #include <cassert> 40 #include <cstddef> 41 #include <cstdint> 42 #include <iterator> 43 #include <limits> 44 #include <memory> 45 #include <string> 46 #include <tuple> 47 #include <vector> 48 49 using namespace llvm; 50 using namespace llvm::ELF; 51 using namespace llvm::object; 52 using namespace llvm::support::endian; 53 using namespace lld; 54 using namespace lld::elf; 55 56 LinkerScriptBase *elf::ScriptBase; 57 ScriptConfiguration *elf::ScriptConfig; 58 59 template <class ELFT> static SymbolBody *addRegular(SymbolAssignment *Cmd) { 60 Symbol *Sym; 61 uint8_t Visibility = Cmd->Hidden ? STV_HIDDEN : STV_DEFAULT; 62 std::tie(Sym, std::ignore) = Symtab<ELFT>::X->insert( 63 Cmd->Name, /*Type*/ 0, Visibility, /*CanOmitFromDynSym*/ false, 64 /*File*/ nullptr); 65 Sym->Binding = STB_GLOBAL; 66 replaceBody<DefinedRegular<ELFT>>(Sym, Cmd->Name, /*IsLocal=*/false, 67 Visibility, STT_NOTYPE, 0, 0, nullptr, 68 nullptr); 69 return Sym->body(); 70 } 71 72 template <class ELFT> static SymbolBody *addSynthetic(SymbolAssignment *Cmd) { 73 Symbol *Sym; 74 uint8_t Visibility = Cmd->Hidden ? STV_HIDDEN : STV_DEFAULT; 75 const OutputSectionBase *Sec = 76 ScriptConfig->HasSections ? nullptr : Cmd->Expression.Section(); 77 std::tie(Sym, std::ignore) = Symtab<ELFT>::X->insert( 78 Cmd->Name, /*Type*/ 0, Visibility, /*CanOmitFromDynSym*/ false, 79 /*File*/ nullptr); 80 Sym->Binding = STB_GLOBAL; 81 replaceBody<DefinedSynthetic>(Sym, Cmd->Name, 0, Sec); 82 return Sym->body(); 83 } 84 85 static bool isUnderSysroot(StringRef Path) { 86 if (Config->Sysroot == "") 87 return false; 88 for (; !Path.empty(); Path = sys::path::parent_path(Path)) 89 if (sys::fs::equivalent(Config->Sysroot, Path)) 90 return true; 91 return false; 92 } 93 94 template <class ELFT> 95 void LinkerScript<ELFT>::setDot(Expr E, const Twine &Loc, bool InSec) { 96 uintX_t Val = E(Dot); 97 if (Val < Dot) { 98 if (InSec) 99 error(Loc + ": unable to move location counter backward for: " + 100 CurOutSec->Name); 101 else 102 error(Loc + ": unable to move location counter backward"); 103 } 104 Dot = Val; 105 // Update to location counter means update to section size. 106 if (InSec) 107 CurOutSec->Size = Dot - CurOutSec->Addr; 108 } 109 110 // Sets value of a symbol. Two kinds of symbols are processed: synthetic 111 // symbols, whose value is an offset from beginning of section and regular 112 // symbols whose value is absolute. 113 template <class ELFT> 114 void LinkerScript<ELFT>::assignSymbol(SymbolAssignment *Cmd, bool InSec) { 115 if (Cmd->Name == ".") { 116 setDot(Cmd->Expression, Cmd->Location, InSec); 117 return; 118 } 119 120 if (!Cmd->Sym) 121 return; 122 123 if (auto *Body = dyn_cast<DefinedSynthetic>(Cmd->Sym)) { 124 Body->Section = Cmd->Expression.Section(); 125 if (Body->Section) { 126 uint64_t VA = 0; 127 if (Body->Section->Flags & SHF_ALLOC) 128 VA = Body->Section->Addr; 129 Body->Value = Cmd->Expression(Dot) - VA; 130 } 131 return; 132 } 133 134 cast<DefinedRegular<ELFT>>(Cmd->Sym)->Value = Cmd->Expression(Dot); 135 } 136 137 template <class ELFT> 138 void LinkerScript<ELFT>::addSymbol(SymbolAssignment *Cmd) { 139 if (Cmd->Name == ".") 140 return; 141 142 // If a symbol was in PROVIDE(), we need to define it only when 143 // it is a referenced undefined symbol. 144 SymbolBody *B = Symtab<ELFT>::X->find(Cmd->Name); 145 if (Cmd->Provide && (!B || B->isDefined())) 146 return; 147 148 // Otherwise, create a new symbol if one does not exist or an 149 // undefined one does exist. 150 if (Cmd->Expression.IsAbsolute()) 151 Cmd->Sym = addRegular<ELFT>(Cmd); 152 else 153 Cmd->Sym = addSynthetic<ELFT>(Cmd); 154 155 // If there are sections, then let the value be assigned later in 156 // `assignAddresses`. 157 if (!ScriptConfig->HasSections) 158 assignSymbol(Cmd); 159 } 160 161 bool SymbolAssignment::classof(const BaseCommand *C) { 162 return C->Kind == AssignmentKind; 163 } 164 165 bool OutputSectionCommand::classof(const BaseCommand *C) { 166 return C->Kind == OutputSectionKind; 167 } 168 169 bool InputSectionDescription::classof(const BaseCommand *C) { 170 return C->Kind == InputSectionKind; 171 } 172 173 bool AssertCommand::classof(const BaseCommand *C) { 174 return C->Kind == AssertKind; 175 } 176 177 bool BytesDataCommand::classof(const BaseCommand *C) { 178 return C->Kind == BytesDataKind; 179 } 180 181 template <class ELFT> LinkerScript<ELFT>::LinkerScript() = default; 182 template <class ELFT> LinkerScript<ELFT>::~LinkerScript() = default; 183 184 static StringRef basename(InputSectionBase *S) { 185 if (S->File) 186 return sys::path::filename(S->File->getName()); 187 return ""; 188 } 189 190 template <class ELFT> bool LinkerScript<ELFT>::shouldKeep(InputSectionBase *S) { 191 for (InputSectionDescription *ID : Opt.KeptSections) 192 if (ID->FilePat.match(basename(S))) 193 for (SectionPattern &P : ID->SectionPatterns) 194 if (P.SectionPat.match(S->Name)) 195 return true; 196 return false; 197 } 198 199 static bool comparePriority(InputSectionBase *A, InputSectionBase *B) { 200 return getPriority(A->Name) < getPriority(B->Name); 201 } 202 203 static bool compareName(InputSectionBase *A, InputSectionBase *B) { 204 return A->Name < B->Name; 205 } 206 207 static bool compareAlignment(InputSectionBase *A, InputSectionBase *B) { 208 // ">" is not a mistake. Larger alignments are placed before smaller 209 // alignments in order to reduce the amount of padding necessary. 210 // This is compatible with GNU. 211 return A->Alignment > B->Alignment; 212 } 213 214 static std::function<bool(InputSectionBase *, InputSectionBase *)> 215 getComparator(SortSectionPolicy K) { 216 switch (K) { 217 case SortSectionPolicy::Alignment: 218 return compareAlignment; 219 case SortSectionPolicy::Name: 220 return compareName; 221 case SortSectionPolicy::Priority: 222 return comparePriority; 223 default: 224 llvm_unreachable("unknown sort policy"); 225 } 226 } 227 228 template <class ELFT> 229 static bool matchConstraints(ArrayRef<InputSectionBase *> Sections, 230 ConstraintKind Kind) { 231 if (Kind == ConstraintKind::NoConstraint) 232 return true; 233 bool IsRW = llvm::any_of(Sections, [=](InputSectionBase *Sec2) { 234 auto *Sec = static_cast<InputSectionBase *>(Sec2); 235 return Sec->Flags & SHF_WRITE; 236 }); 237 return (IsRW && Kind == ConstraintKind::ReadWrite) || 238 (!IsRW && Kind == ConstraintKind::ReadOnly); 239 } 240 241 static void sortSections(InputSectionBase **Begin, InputSectionBase **End, 242 SortSectionPolicy K) { 243 if (K != SortSectionPolicy::Default && K != SortSectionPolicy::None) 244 std::stable_sort(Begin, End, getComparator(K)); 245 } 246 247 // Compute and remember which sections the InputSectionDescription matches. 248 template <class ELFT> 249 void LinkerScript<ELFT>::computeInputSections(InputSectionDescription *I) { 250 // Collects all sections that satisfy constraints of I 251 // and attach them to I. 252 for (SectionPattern &Pat : I->SectionPatterns) { 253 size_t SizeBefore = I->Sections.size(); 254 255 for (InputSectionBase *S : Symtab<ELFT>::X->Sections) { 256 if (S->Assigned) 257 continue; 258 // For -emit-relocs we have to ignore entries like 259 // .rela.dyn : { *(.rela.data) } 260 // which are common because they are in the default bfd script. 261 if (S->Type == SHT_REL || S->Type == SHT_RELA) 262 continue; 263 264 StringRef Filename = basename(S); 265 if (!I->FilePat.match(Filename) || Pat.ExcludedFilePat.match(Filename)) 266 continue; 267 if (!Pat.SectionPat.match(S->Name)) 268 continue; 269 I->Sections.push_back(S); 270 S->Assigned = true; 271 } 272 273 // Sort sections as instructed by SORT-family commands and --sort-section 274 // option. Because SORT-family commands can be nested at most two depth 275 // (e.g. SORT_BY_NAME(SORT_BY_ALIGNMENT(.text.*))) and because the command 276 // line option is respected even if a SORT command is given, the exact 277 // behavior we have here is a bit complicated. Here are the rules. 278 // 279 // 1. If two SORT commands are given, --sort-section is ignored. 280 // 2. If one SORT command is given, and if it is not SORT_NONE, 281 // --sort-section is handled as an inner SORT command. 282 // 3. If one SORT command is given, and if it is SORT_NONE, don't sort. 283 // 4. If no SORT command is given, sort according to --sort-section. 284 InputSectionBase **Begin = I->Sections.data() + SizeBefore; 285 InputSectionBase **End = I->Sections.data() + I->Sections.size(); 286 if (Pat.SortOuter != SortSectionPolicy::None) { 287 if (Pat.SortInner == SortSectionPolicy::Default) 288 sortSections(Begin, End, Config->SortSection); 289 else 290 sortSections(Begin, End, Pat.SortInner); 291 sortSections(Begin, End, Pat.SortOuter); 292 } 293 } 294 } 295 296 template <class ELFT> 297 void LinkerScript<ELFT>::discard(ArrayRef<InputSectionBase *> V) { 298 for (InputSectionBase *S : V) { 299 S->Live = false; 300 if (S == In<ELFT>::ShStrTab) 301 error("discarding .shstrtab section is not allowed"); 302 discard(S->DependentSections); 303 } 304 } 305 306 template <class ELFT> 307 std::vector<InputSectionBase *> 308 LinkerScript<ELFT>::createInputSectionList(OutputSectionCommand &OutCmd) { 309 std::vector<InputSectionBase *> Ret; 310 311 for (const std::unique_ptr<BaseCommand> &Base : OutCmd.Commands) { 312 auto *Cmd = dyn_cast<InputSectionDescription>(Base.get()); 313 if (!Cmd) 314 continue; 315 computeInputSections(Cmd); 316 for (InputSectionBase *S : Cmd->Sections) 317 Ret.push_back(static_cast<InputSectionBase *>(S)); 318 } 319 320 return Ret; 321 } 322 323 template <class ELFT> 324 void LinkerScript<ELFT>::processCommands(OutputSectionFactory<ELFT> &Factory) { 325 for (unsigned I = 0; I < Opt.Commands.size(); ++I) { 326 auto Iter = Opt.Commands.begin() + I; 327 const std::unique_ptr<BaseCommand> &Base1 = *Iter; 328 329 // Handle symbol assignments outside of any output section. 330 if (auto *Cmd = dyn_cast<SymbolAssignment>(Base1.get())) { 331 addSymbol(Cmd); 332 continue; 333 } 334 335 if (auto *Cmd = dyn_cast<AssertCommand>(Base1.get())) { 336 // If we don't have SECTIONS then output sections have already been 337 // created by Writer<ELFT>. The LinkerScript<ELFT>::assignAddresses 338 // will not be called, so ASSERT should be evaluated now. 339 if (!Opt.HasSections) 340 Cmd->Expression(0); 341 continue; 342 } 343 344 if (auto *Cmd = dyn_cast<OutputSectionCommand>(Base1.get())) { 345 std::vector<InputSectionBase *> V = createInputSectionList(*Cmd); 346 347 // The output section name `/DISCARD/' is special. 348 // Any input section assigned to it is discarded. 349 if (Cmd->Name == "/DISCARD/") { 350 discard(V); 351 continue; 352 } 353 354 // This is for ONLY_IF_RO and ONLY_IF_RW. An output section directive 355 // ".foo : ONLY_IF_R[OW] { ... }" is handled only if all member input 356 // sections satisfy a given constraint. If not, a directive is handled 357 // as if it wasn't present from the beginning. 358 // 359 // Because we'll iterate over Commands many more times, the easiest 360 // way to "make it as if it wasn't present" is to just remove it. 361 if (!matchConstraints<ELFT>(V, Cmd->Constraint)) { 362 for (InputSectionBase *S : V) 363 S->Assigned = false; 364 Opt.Commands.erase(Iter); 365 --I; 366 continue; 367 } 368 369 // A directive may contain symbol definitions like this: 370 // ".foo : { ...; bar = .; }". Handle them. 371 for (const std::unique_ptr<BaseCommand> &Base : Cmd->Commands) 372 if (auto *OutCmd = dyn_cast<SymbolAssignment>(Base.get())) 373 addSymbol(OutCmd); 374 375 // Handle subalign (e.g. ".foo : SUBALIGN(32) { ... }"). If subalign 376 // is given, input sections are aligned to that value, whether the 377 // given value is larger or smaller than the original section alignment. 378 if (Cmd->SubalignExpr) { 379 uint32_t Subalign = Cmd->SubalignExpr(0); 380 for (InputSectionBase *S : V) 381 S->Alignment = Subalign; 382 } 383 384 // Add input sections to an output section. 385 for (InputSectionBase *S : V) 386 Factory.addInputSec(S, Cmd->Name); 387 } 388 } 389 } 390 391 // Add sections that didn't match any sections command. 392 template <class ELFT> 393 void LinkerScript<ELFT>::addOrphanSections( 394 OutputSectionFactory<ELFT> &Factory) { 395 for (InputSectionBase *S : Symtab<ELFT>::X->Sections) 396 if (S->Live && !S->OutSec) 397 Factory.addInputSec(S, getOutputSectionName(S->Name)); 398 } 399 400 template <class ELFT> static bool isTbss(OutputSectionBase *Sec) { 401 return (Sec->Flags & SHF_TLS) && Sec->Type == SHT_NOBITS; 402 } 403 404 template <class ELFT> void LinkerScript<ELFT>::output(InputSection *S) { 405 if (!AlreadyOutputIS.insert(S).second) 406 return; 407 bool IsTbss = isTbss<ELFT>(CurOutSec); 408 409 uintX_t Pos = IsTbss ? Dot + ThreadBssOffset : Dot; 410 Pos = alignTo(Pos, S->Alignment); 411 S->OutSecOff = Pos - CurOutSec->Addr; 412 Pos += S->template getSize<ELFT>(); 413 414 // Update output section size after adding each section. This is so that 415 // SIZEOF works correctly in the case below: 416 // .foo { *(.aaa) a = SIZEOF(.foo); *(.bbb) } 417 CurOutSec->Size = Pos - CurOutSec->Addr; 418 419 // If there is a memory region associated with this input section, then 420 // place the section in that region and update the region index. 421 if (CurMemRegion) { 422 CurMemRegion->Offset += CurOutSec->Size; 423 uint64_t CurSize = CurMemRegion->Offset - CurMemRegion->Origin; 424 if (CurSize > CurMemRegion->Length) { 425 uint64_t OverflowAmt = CurSize - CurMemRegion->Length; 426 error("section '" + CurOutSec->Name + "' will not fit in region '" + 427 CurMemRegion->Name + "': overflowed by " + Twine(OverflowAmt) + 428 " bytes"); 429 } 430 } 431 432 if (IsTbss) 433 ThreadBssOffset = Pos - Dot; 434 else 435 Dot = Pos; 436 } 437 438 template <class ELFT> void LinkerScript<ELFT>::flush() { 439 if (!CurOutSec || !AlreadyOutputOS.insert(CurOutSec).second) 440 return; 441 if (auto *OutSec = dyn_cast<OutputSection<ELFT>>(CurOutSec)) { 442 for (InputSection *I : OutSec->Sections) 443 output(I); 444 } else { 445 Dot += CurOutSec->Size; 446 } 447 } 448 449 template <class ELFT> 450 void LinkerScript<ELFT>::switchTo(OutputSectionBase *Sec) { 451 if (CurOutSec == Sec) 452 return; 453 if (AlreadyOutputOS.count(Sec)) 454 return; 455 456 flush(); 457 CurOutSec = Sec; 458 459 Dot = alignTo(Dot, CurOutSec->Addralign); 460 CurOutSec->Addr = isTbss<ELFT>(CurOutSec) ? Dot + ThreadBssOffset : Dot; 461 462 // If neither AT nor AT> is specified for an allocatable section, the linker 463 // will set the LMA such that the difference between VMA and LMA for the 464 // section is the same as the preceding output section in the same region 465 // https://sourceware.org/binutils/docs-2.20/ld/Output-Section-LMA.html 466 if (LMAOffset) 467 CurOutSec->setLMAOffset(LMAOffset()); 468 } 469 470 template <class ELFT> void LinkerScript<ELFT>::process(BaseCommand &Base) { 471 // This handles the assignments to symbol or to a location counter (.) 472 if (auto *AssignCmd = dyn_cast<SymbolAssignment>(&Base)) { 473 assignSymbol(AssignCmd, true); 474 return; 475 } 476 477 // Handle BYTE(), SHORT(), LONG(), or QUAD(). 478 if (auto *DataCmd = dyn_cast<BytesDataCommand>(&Base)) { 479 DataCmd->Offset = Dot - CurOutSec->Addr; 480 Dot += DataCmd->Size; 481 CurOutSec->Size = Dot - CurOutSec->Addr; 482 return; 483 } 484 485 if (auto *AssertCmd = dyn_cast<AssertCommand>(&Base)) { 486 AssertCmd->Expression(Dot); 487 return; 488 } 489 490 // It handles single input section description command, 491 // calculates and assigns the offsets for each section and also 492 // updates the output section size. 493 auto &ICmd = cast<InputSectionDescription>(Base); 494 for (InputSectionBase *ID : ICmd.Sections) { 495 // We tentatively added all synthetic sections at the beginning and removed 496 // empty ones afterwards (because there is no way to know whether they were 497 // going be empty or not other than actually running linker scripts.) 498 // We need to ignore remains of empty sections. 499 if (auto *Sec = dyn_cast<SyntheticSection<ELFT>>(ID)) 500 if (Sec->empty()) 501 continue; 502 503 auto *IB = static_cast<InputSectionBase *>(ID); 504 if (!IB->Live) 505 continue; 506 switchTo(IB->OutSec); 507 if (auto *I = dyn_cast<InputSection>(IB)) 508 output(I); 509 else 510 flush(); 511 } 512 } 513 514 template <class ELFT> 515 static OutputSectionBase * 516 findSection(StringRef Name, const std::vector<OutputSectionBase *> &Sections) { 517 auto End = Sections.end(); 518 auto HasName = [=](OutputSectionBase *Sec) { return Sec->getName() == Name; }; 519 auto I = std::find_if(Sections.begin(), End, HasName); 520 std::vector<OutputSectionBase *> Ret; 521 if (I == End) 522 return nullptr; 523 assert(std::find_if(I + 1, End, HasName) == End); 524 return *I; 525 } 526 527 // This function searches for a memory region to place the given output 528 // section in. If found, a pointer to the appropriate memory region is 529 // returned. Otherwise, a nullptr is returned. 530 template <class ELFT> 531 MemoryRegion *LinkerScript<ELFT>::findMemoryRegion(OutputSectionCommand *Cmd, 532 OutputSectionBase *Sec) { 533 // If a memory region name was specified in the output section command, 534 // then try to find that region first. 535 if (!Cmd->MemoryRegionName.empty()) { 536 auto It = Opt.MemoryRegions.find(Cmd->MemoryRegionName); 537 if (It != Opt.MemoryRegions.end()) 538 return &It->second; 539 error("memory region '" + Cmd->MemoryRegionName + "' not declared"); 540 return nullptr; 541 } 542 543 // The memory region name is empty, thus a suitable region must be 544 // searched for in the region map. If the region map is empty, just 545 // return. Note that this check doesn't happen at the very beginning 546 // so that uses of undeclared regions can be caught. 547 if (!Opt.MemoryRegions.size()) 548 return nullptr; 549 550 // See if a region can be found by matching section flags. 551 for (auto &MRI : Opt.MemoryRegions) { 552 MemoryRegion &MR = MRI.second; 553 if ((MR.Flags & Sec->Flags) != 0 && (MR.NegFlags & Sec->Flags) == 0) 554 return &MR; 555 } 556 557 // Otherwise, no suitable region was found. 558 if (Sec->Flags & SHF_ALLOC) 559 error("no memory region specified for section '" + Sec->Name + "'"); 560 return nullptr; 561 } 562 563 // This function assigns offsets to input sections and an output section 564 // for a single sections command (e.g. ".text { *(.text); }"). 565 template <class ELFT> 566 void LinkerScript<ELFT>::assignOffsets(OutputSectionCommand *Cmd) { 567 if (Cmd->LMAExpr) { 568 uintX_t D = Dot; 569 LMAOffset = [=] { return Cmd->LMAExpr(D) - D; }; 570 } 571 OutputSectionBase *Sec = findSection<ELFT>(Cmd->Name, *OutputSections); 572 if (!Sec) 573 return; 574 575 if (Cmd->AddrExpr && Sec->Flags & SHF_ALLOC) 576 setDot(Cmd->AddrExpr, Cmd->Location); 577 578 // Handle align (e.g. ".foo : ALIGN(16) { ... }"). 579 if (Cmd->AlignExpr) 580 Sec->updateAlignment(Cmd->AlignExpr(0)); 581 582 // Try and find an appropriate memory region to assign offsets in. 583 CurMemRegion = findMemoryRegion(Cmd, Sec); 584 if (CurMemRegion) 585 Dot = CurMemRegion->Offset; 586 switchTo(Sec); 587 588 // Find the last section output location. We will output orphan sections 589 // there so that end symbols point to the correct location. 590 auto E = std::find_if(Cmd->Commands.rbegin(), Cmd->Commands.rend(), 591 [](const std::unique_ptr<BaseCommand> &Cmd) { 592 return !isa<SymbolAssignment>(*Cmd); 593 }) 594 .base(); 595 for (auto I = Cmd->Commands.begin(); I != E; ++I) 596 process(**I); 597 flush(); 598 std::for_each(E, Cmd->Commands.end(), 599 [this](std::unique_ptr<BaseCommand> &B) { process(*B.get()); }); 600 } 601 602 template <class ELFT> void LinkerScript<ELFT>::removeEmptyCommands() { 603 // It is common practice to use very generic linker scripts. So for any 604 // given run some of the output sections in the script will be empty. 605 // We could create corresponding empty output sections, but that would 606 // clutter the output. 607 // We instead remove trivially empty sections. The bfd linker seems even 608 // more aggressive at removing them. 609 auto Pos = std::remove_if( 610 Opt.Commands.begin(), Opt.Commands.end(), 611 [&](const std::unique_ptr<BaseCommand> &Base) { 612 if (auto *Cmd = dyn_cast<OutputSectionCommand>(Base.get())) 613 return !findSection<ELFT>(Cmd->Name, *OutputSections); 614 return false; 615 }); 616 Opt.Commands.erase(Pos, Opt.Commands.end()); 617 } 618 619 static bool isAllSectionDescription(const OutputSectionCommand &Cmd) { 620 for (const std::unique_ptr<BaseCommand> &I : Cmd.Commands) 621 if (!isa<InputSectionDescription>(*I)) 622 return false; 623 return true; 624 } 625 626 template <class ELFT> void LinkerScript<ELFT>::adjustSectionsBeforeSorting() { 627 // If the output section contains only symbol assignments, create a 628 // corresponding output section. The bfd linker seems to only create them if 629 // '.' is assigned to, but creating these section should not have any bad 630 // consequeces and gives us a section to put the symbol in. 631 uintX_t Flags = SHF_ALLOC; 632 uint32_t Type = SHT_NOBITS; 633 for (const std::unique_ptr<BaseCommand> &Base : Opt.Commands) { 634 auto *Cmd = dyn_cast<OutputSectionCommand>(Base.get()); 635 if (!Cmd) 636 continue; 637 if (OutputSectionBase *Sec = 638 findSection<ELFT>(Cmd->Name, *OutputSections)) { 639 Flags = Sec->Flags; 640 Type = Sec->Type; 641 continue; 642 } 643 644 if (isAllSectionDescription(*Cmd)) 645 continue; 646 647 auto *OutSec = make<OutputSection<ELFT>>(Cmd->Name, Type, Flags); 648 OutputSections->push_back(OutSec); 649 } 650 } 651 652 template <class ELFT> void LinkerScript<ELFT>::adjustSectionsAfterSorting() { 653 placeOrphanSections(); 654 655 // If output section command doesn't specify any segments, 656 // and we haven't previously assigned any section to segment, 657 // then we simply assign section to the very first load segment. 658 // Below is an example of such linker script: 659 // PHDRS { seg PT_LOAD; } 660 // SECTIONS { .aaa : { *(.aaa) } } 661 std::vector<StringRef> DefPhdrs; 662 auto FirstPtLoad = 663 std::find_if(Opt.PhdrsCommands.begin(), Opt.PhdrsCommands.end(), 664 [](const PhdrsCommand &Cmd) { return Cmd.Type == PT_LOAD; }); 665 if (FirstPtLoad != Opt.PhdrsCommands.end()) 666 DefPhdrs.push_back(FirstPtLoad->Name); 667 668 // Walk the commands and propagate the program headers to commands that don't 669 // explicitly specify them. 670 for (const std::unique_ptr<BaseCommand> &Base : Opt.Commands) { 671 auto *Cmd = dyn_cast<OutputSectionCommand>(Base.get()); 672 if (!Cmd) 673 continue; 674 if (Cmd->Phdrs.empty()) 675 Cmd->Phdrs = DefPhdrs; 676 else 677 DefPhdrs = Cmd->Phdrs; 678 } 679 680 removeEmptyCommands(); 681 } 682 683 // When placing orphan sections, we want to place them after symbol assignments 684 // so that an orphan after 685 // begin_foo = .; 686 // foo : { *(foo) } 687 // end_foo = .; 688 // doesn't break the intended meaning of the begin/end symbols. 689 // We don't want to go over sections since Writer<ELFT>::sortSections is the 690 // one in charge of deciding the order of the sections. 691 // We don't want to go over alignments, since doing so in 692 // rx_sec : { *(rx_sec) } 693 // . = ALIGN(0x1000); 694 // /* The RW PT_LOAD starts here*/ 695 // rw_sec : { *(rw_sec) } 696 // would mean that the RW PT_LOAD would become unaligned. 697 static bool shouldSkip(const BaseCommand &Cmd) { 698 if (isa<OutputSectionCommand>(Cmd)) 699 return false; 700 const auto *Assign = dyn_cast<SymbolAssignment>(&Cmd); 701 if (!Assign) 702 return true; 703 return Assign->Name != "."; 704 } 705 706 // Orphan sections are sections present in the input files which are 707 // not explicitly placed into the output file by the linker script. 708 // 709 // When the control reaches this function, Opt.Commands contains 710 // output section commands for non-orphan sections only. This function 711 // adds new elements for orphan sections to Opt.Commands so that all 712 // sections are explicitly handled by Opt.Commands. 713 // 714 // Writer<ELFT>::sortSections has already sorted output sections. 715 // What we need to do is to scan OutputSections vector and 716 // Opt.Commands in parallel to find orphan sections. If there is an 717 // output section that doesn't have a corresponding entry in 718 // Opt.Commands, we will insert a new entry to Opt.Commands. 719 // 720 // There is some ambiguity as to where exactly a new entry should be 721 // inserted, because Opt.Commands contains not only output section 722 // commands but other types of commands such as symbol assignment 723 // expressions. There's no correct answer here due to the lack of the 724 // formal specification of the linker script. We use heuristics to 725 // determine whether a new output command should be added before or 726 // after another commands. For the details, look at shouldSkip 727 // function. 728 template <class ELFT> void LinkerScript<ELFT>::placeOrphanSections() { 729 // The OutputSections are already in the correct order. 730 // This loops creates or moves commands as needed so that they are in the 731 // correct order. 732 int CmdIndex = 0; 733 734 // As a horrible special case, skip the first . assignment if it is before any 735 // section. We do this because it is common to set a load address by starting 736 // the script with ". = 0xabcd" and the expectation is that every section is 737 // after that. 738 auto FirstSectionOrDotAssignment = 739 std::find_if(Opt.Commands.begin(), Opt.Commands.end(), 740 [](const std::unique_ptr<BaseCommand> &Cmd) { 741 if (isa<OutputSectionCommand>(*Cmd)) 742 return true; 743 const auto *Assign = dyn_cast<SymbolAssignment>(Cmd.get()); 744 if (!Assign) 745 return false; 746 return Assign->Name == "."; 747 }); 748 if (FirstSectionOrDotAssignment != Opt.Commands.end()) { 749 CmdIndex = FirstSectionOrDotAssignment - Opt.Commands.begin(); 750 if (isa<SymbolAssignment>(**FirstSectionOrDotAssignment)) 751 ++CmdIndex; 752 } 753 754 for (OutputSectionBase *Sec : *OutputSections) { 755 StringRef Name = Sec->getName(); 756 757 // Find the last spot where we can insert a command and still get the 758 // correct result. 759 auto CmdIter = Opt.Commands.begin() + CmdIndex; 760 auto E = Opt.Commands.end(); 761 while (CmdIter != E && shouldSkip(**CmdIter)) { 762 ++CmdIter; 763 ++CmdIndex; 764 } 765 766 auto Pos = 767 std::find_if(CmdIter, E, [&](const std::unique_ptr<BaseCommand> &Base) { 768 auto *Cmd = dyn_cast<OutputSectionCommand>(Base.get()); 769 return Cmd && Cmd->Name == Name; 770 }); 771 if (Pos == E) { 772 Opt.Commands.insert(CmdIter, 773 llvm::make_unique<OutputSectionCommand>(Name)); 774 ++CmdIndex; 775 continue; 776 } 777 778 // Continue from where we found it. 779 CmdIndex = (Pos - Opt.Commands.begin()) + 1; 780 } 781 } 782 783 template <class ELFT> 784 void LinkerScript<ELFT>::assignAddresses(std::vector<PhdrEntry> &Phdrs) { 785 // Assign addresses as instructed by linker script SECTIONS sub-commands. 786 Dot = 0; 787 788 // A symbol can be assigned before any section is mentioned in the linker 789 // script. In an DSO, the symbol values are addresses, so the only important 790 // section values are: 791 // * SHN_UNDEF 792 // * SHN_ABS 793 // * Any value meaning a regular section. 794 // To handle that, create a dummy aether section that fills the void before 795 // the linker scripts switches to another section. It has an index of one 796 // which will map to whatever the first actual section is. 797 auto *Aether = make<OutputSectionBase>("", 0, SHF_ALLOC); 798 Aether->SectionIndex = 1; 799 switchTo(Aether); 800 801 for (const std::unique_ptr<BaseCommand> &Base : Opt.Commands) { 802 if (auto *Cmd = dyn_cast<SymbolAssignment>(Base.get())) { 803 assignSymbol(Cmd); 804 continue; 805 } 806 807 if (auto *Cmd = dyn_cast<AssertCommand>(Base.get())) { 808 Cmd->Expression(Dot); 809 continue; 810 } 811 812 auto *Cmd = cast<OutputSectionCommand>(Base.get()); 813 assignOffsets(Cmd); 814 } 815 816 uintX_t MinVA = std::numeric_limits<uintX_t>::max(); 817 for (OutputSectionBase *Sec : *OutputSections) { 818 if (Sec->Flags & SHF_ALLOC) 819 MinVA = std::min<uint64_t>(MinVA, Sec->Addr); 820 else 821 Sec->Addr = 0; 822 } 823 824 allocateHeaders<ELFT>(Phdrs, *OutputSections, MinVA); 825 } 826 827 // Creates program headers as instructed by PHDRS linker script command. 828 template <class ELFT> std::vector<PhdrEntry> LinkerScript<ELFT>::createPhdrs() { 829 std::vector<PhdrEntry> Ret; 830 831 // Process PHDRS and FILEHDR keywords because they are not 832 // real output sections and cannot be added in the following loop. 833 for (const PhdrsCommand &Cmd : Opt.PhdrsCommands) { 834 Ret.emplace_back(Cmd.Type, Cmd.Flags == UINT_MAX ? PF_R : Cmd.Flags); 835 PhdrEntry &Phdr = Ret.back(); 836 837 if (Cmd.HasFilehdr) 838 Phdr.add(Out<ELFT>::ElfHeader); 839 if (Cmd.HasPhdrs) 840 Phdr.add(Out<ELFT>::ProgramHeaders); 841 842 if (Cmd.LMAExpr) { 843 Phdr.p_paddr = Cmd.LMAExpr(0); 844 Phdr.HasLMA = true; 845 } 846 } 847 848 // Add output sections to program headers. 849 for (OutputSectionBase *Sec : *OutputSections) { 850 if (!(Sec->Flags & SHF_ALLOC)) 851 break; 852 853 // Assign headers specified by linker script 854 for (size_t Id : getPhdrIndices(Sec->getName())) { 855 Ret[Id].add(Sec); 856 if (Opt.PhdrsCommands[Id].Flags == UINT_MAX) 857 Ret[Id].p_flags |= Sec->getPhdrFlags(); 858 } 859 } 860 return Ret; 861 } 862 863 template <class ELFT> bool LinkerScript<ELFT>::ignoreInterpSection() { 864 // Ignore .interp section in case we have PHDRS specification 865 // and PT_INTERP isn't listed. 866 return !Opt.PhdrsCommands.empty() && 867 llvm::find_if(Opt.PhdrsCommands, [](const PhdrsCommand &Cmd) { 868 return Cmd.Type == PT_INTERP; 869 }) == Opt.PhdrsCommands.end(); 870 } 871 872 template <class ELFT> uint32_t LinkerScript<ELFT>::getFiller(StringRef Name) { 873 for (const std::unique_ptr<BaseCommand> &Base : Opt.Commands) 874 if (auto *Cmd = dyn_cast<OutputSectionCommand>(Base.get())) 875 if (Cmd->Name == Name) 876 return Cmd->Filler; 877 return 0; 878 } 879 880 template <class ELFT> 881 static void writeInt(uint8_t *Buf, uint64_t Data, uint64_t Size) { 882 const endianness E = ELFT::TargetEndianness; 883 884 switch (Size) { 885 case 1: 886 *Buf = (uint8_t)Data; 887 break; 888 case 2: 889 write16<E>(Buf, Data); 890 break; 891 case 4: 892 write32<E>(Buf, Data); 893 break; 894 case 8: 895 write64<E>(Buf, Data); 896 break; 897 default: 898 llvm_unreachable("unsupported Size argument"); 899 } 900 } 901 902 template <class ELFT> 903 void LinkerScript<ELFT>::writeDataBytes(StringRef Name, uint8_t *Buf) { 904 int I = getSectionIndex(Name); 905 if (I == INT_MAX) 906 return; 907 908 auto *Cmd = dyn_cast<OutputSectionCommand>(Opt.Commands[I].get()); 909 for (const std::unique_ptr<BaseCommand> &Base : Cmd->Commands) 910 if (auto *Data = dyn_cast<BytesDataCommand>(Base.get())) 911 writeInt<ELFT>(Buf + Data->Offset, Data->Expression(0), Data->Size); 912 } 913 914 template <class ELFT> bool LinkerScript<ELFT>::hasLMA(StringRef Name) { 915 for (const std::unique_ptr<BaseCommand> &Base : Opt.Commands) 916 if (auto *Cmd = dyn_cast<OutputSectionCommand>(Base.get())) 917 if (Cmd->LMAExpr && Cmd->Name == Name) 918 return true; 919 return false; 920 } 921 922 // Returns the index of the given section name in linker script 923 // SECTIONS commands. Sections are laid out as the same order as they 924 // were in the script. If a given name did not appear in the script, 925 // it returns INT_MAX, so that it will be laid out at end of file. 926 template <class ELFT> int LinkerScript<ELFT>::getSectionIndex(StringRef Name) { 927 for (int I = 0, E = Opt.Commands.size(); I != E; ++I) 928 if (auto *Cmd = dyn_cast<OutputSectionCommand>(Opt.Commands[I].get())) 929 if (Cmd->Name == Name) 930 return I; 931 return INT_MAX; 932 } 933 934 template <class ELFT> bool LinkerScript<ELFT>::hasPhdrsCommands() { 935 return !Opt.PhdrsCommands.empty(); 936 } 937 938 template <class ELFT> 939 const OutputSectionBase *LinkerScript<ELFT>::getOutputSection(const Twine &Loc, 940 StringRef Name) { 941 static OutputSectionBase FakeSec("", 0, 0); 942 943 for (OutputSectionBase *Sec : *OutputSections) 944 if (Sec->getName() == Name) 945 return Sec; 946 947 error(Loc + ": undefined section " + Name); 948 return &FakeSec; 949 } 950 951 // This function is essentially the same as getOutputSection(Name)->Size, 952 // but it won't print out an error message if a given section is not found. 953 // 954 // Linker script does not create an output section if its content is empty. 955 // We want to allow SIZEOF(.foo) where .foo is a section which happened to 956 // be empty. That is why this function is different from getOutputSection(). 957 template <class ELFT> 958 uint64_t LinkerScript<ELFT>::getOutputSectionSize(StringRef Name) { 959 for (OutputSectionBase *Sec : *OutputSections) 960 if (Sec->getName() == Name) 961 return Sec->Size; 962 return 0; 963 } 964 965 template <class ELFT> uint64_t LinkerScript<ELFT>::getHeaderSize() { 966 return elf::getHeaderSize<ELFT>(); 967 } 968 969 template <class ELFT> 970 uint64_t LinkerScript<ELFT>::getSymbolValue(const Twine &Loc, StringRef S) { 971 if (SymbolBody *B = Symtab<ELFT>::X->find(S)) 972 return B->getVA<ELFT>(); 973 error(Loc + ": symbol not found: " + S); 974 return 0; 975 } 976 977 template <class ELFT> bool LinkerScript<ELFT>::isDefined(StringRef S) { 978 return Symtab<ELFT>::X->find(S) != nullptr; 979 } 980 981 template <class ELFT> bool LinkerScript<ELFT>::isAbsolute(StringRef S) { 982 SymbolBody *Sym = Symtab<ELFT>::X->find(S); 983 auto *DR = dyn_cast_or_null<DefinedRegular<ELFT>>(Sym); 984 return DR && !DR->Section; 985 } 986 987 // Gets section symbol belongs to. Symbol "." doesn't belong to any 988 // specific section but isn't absolute at the same time, so we try 989 // to find suitable section for it as well. 990 template <class ELFT> 991 const OutputSectionBase *LinkerScript<ELFT>::getSymbolSection(StringRef S) { 992 if (SymbolBody *Sym = Symtab<ELFT>::X->find(S)) 993 return SymbolTableSection<ELFT>::getOutputSection(Sym); 994 return CurOutSec; 995 } 996 997 // Returns indices of ELF headers containing specific section, identified 998 // by Name. Each index is a zero based number of ELF header listed within 999 // PHDRS {} script block. 1000 template <class ELFT> 1001 std::vector<size_t> LinkerScript<ELFT>::getPhdrIndices(StringRef SectionName) { 1002 for (const std::unique_ptr<BaseCommand> &Base : Opt.Commands) { 1003 auto *Cmd = dyn_cast<OutputSectionCommand>(Base.get()); 1004 if (!Cmd || Cmd->Name != SectionName) 1005 continue; 1006 1007 std::vector<size_t> Ret; 1008 for (StringRef PhdrName : Cmd->Phdrs) 1009 Ret.push_back(getPhdrIndex(Cmd->Location, PhdrName)); 1010 return Ret; 1011 } 1012 return {}; 1013 } 1014 1015 template <class ELFT> 1016 size_t LinkerScript<ELFT>::getPhdrIndex(const Twine &Loc, StringRef PhdrName) { 1017 size_t I = 0; 1018 for (PhdrsCommand &Cmd : Opt.PhdrsCommands) { 1019 if (Cmd.Name == PhdrName) 1020 return I; 1021 ++I; 1022 } 1023 error(Loc + ": section header '" + PhdrName + "' is not listed in PHDRS"); 1024 return 0; 1025 } 1026 1027 class elf::ScriptParser final : public ScriptLexer { 1028 typedef void (ScriptParser::*Handler)(); 1029 1030 public: 1031 ScriptParser(MemoryBufferRef MB) 1032 : ScriptLexer(MB), 1033 IsUnderSysroot(isUnderSysroot(MB.getBufferIdentifier())) {} 1034 1035 void readLinkerScript(); 1036 void readVersionScript(); 1037 void readDynamicList(); 1038 1039 private: 1040 void addFile(StringRef Path); 1041 1042 void readAsNeeded(); 1043 void readEntry(); 1044 void readExtern(); 1045 void readGroup(); 1046 void readInclude(); 1047 void readMemory(); 1048 void readOutput(); 1049 void readOutputArch(); 1050 void readOutputFormat(); 1051 void readPhdrs(); 1052 void readSearchDir(); 1053 void readSections(); 1054 void readVersion(); 1055 void readVersionScriptCommand(); 1056 1057 SymbolAssignment *readAssignment(StringRef Name); 1058 BytesDataCommand *readBytesDataCommand(StringRef Tok); 1059 uint32_t readFill(); 1060 OutputSectionCommand *readOutputSectionDescription(StringRef OutSec); 1061 uint32_t readOutputSectionFiller(StringRef Tok); 1062 std::vector<StringRef> readOutputSectionPhdrs(); 1063 InputSectionDescription *readInputSectionDescription(StringRef Tok); 1064 StringMatcher readFilePatterns(); 1065 std::vector<SectionPattern> readInputSectionsList(); 1066 InputSectionDescription *readInputSectionRules(StringRef FilePattern); 1067 unsigned readPhdrType(); 1068 SortSectionPolicy readSortKind(); 1069 SymbolAssignment *readProvideHidden(bool Provide, bool Hidden); 1070 SymbolAssignment *readProvideOrAssignment(StringRef Tok); 1071 void readSort(); 1072 Expr readAssert(); 1073 1074 uint64_t readMemoryAssignment(StringRef, StringRef, StringRef); 1075 std::pair<uint32_t, uint32_t> readMemoryAttributes(); 1076 1077 Expr readExpr(); 1078 Expr readExpr1(Expr Lhs, int MinPrec); 1079 StringRef readParenLiteral(); 1080 Expr readPrimary(); 1081 Expr readTernary(Expr Cond); 1082 Expr readParenExpr(); 1083 1084 // For parsing version script. 1085 std::vector<SymbolVersion> readVersionExtern(); 1086 void readAnonymousDeclaration(); 1087 void readVersionDeclaration(StringRef VerStr); 1088 std::vector<SymbolVersion> readSymbols(); 1089 void readLocals(); 1090 1091 ScriptConfiguration &Opt = *ScriptConfig; 1092 bool IsUnderSysroot; 1093 }; 1094 1095 void ScriptParser::readDynamicList() { 1096 expect("{"); 1097 readAnonymousDeclaration(); 1098 if (!atEOF()) 1099 setError("EOF expected, but got " + next()); 1100 } 1101 1102 void ScriptParser::readVersionScript() { 1103 readVersionScriptCommand(); 1104 if (!atEOF()) 1105 setError("EOF expected, but got " + next()); 1106 } 1107 1108 void ScriptParser::readVersionScriptCommand() { 1109 if (consume("{")) { 1110 readAnonymousDeclaration(); 1111 return; 1112 } 1113 1114 while (!atEOF() && !Error && peek() != "}") { 1115 StringRef VerStr = next(); 1116 if (VerStr == "{") { 1117 setError("anonymous version definition is used in " 1118 "combination with other version definitions"); 1119 return; 1120 } 1121 expect("{"); 1122 readVersionDeclaration(VerStr); 1123 } 1124 } 1125 1126 void ScriptParser::readVersion() { 1127 expect("{"); 1128 readVersionScriptCommand(); 1129 expect("}"); 1130 } 1131 1132 void ScriptParser::readLinkerScript() { 1133 while (!atEOF()) { 1134 StringRef Tok = next(); 1135 if (Tok == ";") 1136 continue; 1137 1138 if (Tok == "ASSERT") { 1139 Opt.Commands.emplace_back(new AssertCommand(readAssert())); 1140 } else if (Tok == "ENTRY") { 1141 readEntry(); 1142 } else if (Tok == "EXTERN") { 1143 readExtern(); 1144 } else if (Tok == "GROUP" || Tok == "INPUT") { 1145 readGroup(); 1146 } else if (Tok == "INCLUDE") { 1147 readInclude(); 1148 } else if (Tok == "MEMORY") { 1149 readMemory(); 1150 } else if (Tok == "OUTPUT") { 1151 readOutput(); 1152 } else if (Tok == "OUTPUT_ARCH") { 1153 readOutputArch(); 1154 } else if (Tok == "OUTPUT_FORMAT") { 1155 readOutputFormat(); 1156 } else if (Tok == "PHDRS") { 1157 readPhdrs(); 1158 } else if (Tok == "SEARCH_DIR") { 1159 readSearchDir(); 1160 } else if (Tok == "SECTIONS") { 1161 readSections(); 1162 } else if (Tok == "VERSION") { 1163 readVersion(); 1164 } else if (SymbolAssignment *Cmd = readProvideOrAssignment(Tok)) { 1165 Opt.Commands.emplace_back(Cmd); 1166 } else { 1167 setError("unknown directive: " + Tok); 1168 } 1169 } 1170 } 1171 1172 void ScriptParser::addFile(StringRef S) { 1173 if (IsUnderSysroot && S.startswith("/")) { 1174 SmallString<128> PathData; 1175 StringRef Path = (Config->Sysroot + S).toStringRef(PathData); 1176 if (sys::fs::exists(Path)) { 1177 Driver->addFile(Saver.save(Path)); 1178 return; 1179 } 1180 } 1181 1182 if (sys::path::is_absolute(S)) { 1183 Driver->addFile(S); 1184 } else if (S.startswith("=")) { 1185 if (Config->Sysroot.empty()) 1186 Driver->addFile(S.substr(1)); 1187 else 1188 Driver->addFile(Saver.save(Config->Sysroot + "/" + S.substr(1))); 1189 } else if (S.startswith("-l")) { 1190 Driver->addLibrary(S.substr(2)); 1191 } else if (sys::fs::exists(S)) { 1192 Driver->addFile(S); 1193 } else { 1194 if (Optional<std::string> Path = findFromSearchPaths(S)) 1195 Driver->addFile(Saver.save(*Path)); 1196 else 1197 setError("unable to find " + S); 1198 } 1199 } 1200 1201 void ScriptParser::readAsNeeded() { 1202 expect("("); 1203 bool Orig = Config->AsNeeded; 1204 Config->AsNeeded = true; 1205 while (!Error && !consume(")")) 1206 addFile(unquote(next())); 1207 Config->AsNeeded = Orig; 1208 } 1209 1210 void ScriptParser::readEntry() { 1211 // -e <symbol> takes predecence over ENTRY(<symbol>). 1212 expect("("); 1213 StringRef Tok = next(); 1214 if (Config->Entry.empty()) 1215 Config->Entry = Tok; 1216 expect(")"); 1217 } 1218 1219 void ScriptParser::readExtern() { 1220 expect("("); 1221 while (!Error && !consume(")")) 1222 Config->Undefined.push_back(next()); 1223 } 1224 1225 void ScriptParser::readGroup() { 1226 expect("("); 1227 while (!Error && !consume(")")) { 1228 StringRef Tok = next(); 1229 if (Tok == "AS_NEEDED") 1230 readAsNeeded(); 1231 else 1232 addFile(unquote(Tok)); 1233 } 1234 } 1235 1236 void ScriptParser::readInclude() { 1237 StringRef Tok = unquote(next()); 1238 1239 // https://sourceware.org/binutils/docs/ld/File-Commands.html: 1240 // The file will be searched for in the current directory, and in any 1241 // directory specified with the -L option. 1242 if (sys::fs::exists(Tok)) { 1243 if (Optional<MemoryBufferRef> MB = readFile(Tok)) 1244 tokenize(*MB); 1245 return; 1246 } 1247 if (Optional<std::string> Path = findFromSearchPaths(Tok)) { 1248 if (Optional<MemoryBufferRef> MB = readFile(*Path)) 1249 tokenize(*MB); 1250 return; 1251 } 1252 setError("cannot open " + Tok); 1253 } 1254 1255 void ScriptParser::readOutput() { 1256 // -o <file> takes predecence over OUTPUT(<file>). 1257 expect("("); 1258 StringRef Tok = next(); 1259 if (Config->OutputFile.empty()) 1260 Config->OutputFile = unquote(Tok); 1261 expect(")"); 1262 } 1263 1264 void ScriptParser::readOutputArch() { 1265 // OUTPUT_ARCH is ignored for now. 1266 expect("("); 1267 while (!Error && !consume(")")) 1268 skip(); 1269 } 1270 1271 void ScriptParser::readOutputFormat() { 1272 // Error checking only for now. 1273 expect("("); 1274 skip(); 1275 StringRef Tok = next(); 1276 if (Tok == ")") 1277 return; 1278 if (Tok != ",") { 1279 setError("unexpected token: " + Tok); 1280 return; 1281 } 1282 skip(); 1283 expect(","); 1284 skip(); 1285 expect(")"); 1286 } 1287 1288 void ScriptParser::readPhdrs() { 1289 expect("{"); 1290 while (!Error && !consume("}")) { 1291 StringRef Tok = next(); 1292 Opt.PhdrsCommands.push_back( 1293 {Tok, PT_NULL, false, false, UINT_MAX, nullptr}); 1294 PhdrsCommand &PhdrCmd = Opt.PhdrsCommands.back(); 1295 1296 PhdrCmd.Type = readPhdrType(); 1297 do { 1298 Tok = next(); 1299 if (Tok == ";") 1300 break; 1301 if (Tok == "FILEHDR") 1302 PhdrCmd.HasFilehdr = true; 1303 else if (Tok == "PHDRS") 1304 PhdrCmd.HasPhdrs = true; 1305 else if (Tok == "AT") 1306 PhdrCmd.LMAExpr = readParenExpr(); 1307 else if (Tok == "FLAGS") { 1308 expect("("); 1309 // Passing 0 for the value of dot is a bit of a hack. It means that 1310 // we accept expressions like ".|1". 1311 PhdrCmd.Flags = readExpr()(0); 1312 expect(")"); 1313 } else 1314 setError("unexpected header attribute: " + Tok); 1315 } while (!Error); 1316 } 1317 } 1318 1319 void ScriptParser::readSearchDir() { 1320 expect("("); 1321 StringRef Tok = next(); 1322 if (!Config->Nostdlib) 1323 Config->SearchPaths.push_back(unquote(Tok)); 1324 expect(")"); 1325 } 1326 1327 void ScriptParser::readSections() { 1328 Opt.HasSections = true; 1329 // -no-rosegment is used to avoid placing read only non-executable sections in 1330 // their own segment. We do the same if SECTIONS command is present in linker 1331 // script. See comment for computeFlags(). 1332 Config->SingleRoRx = true; 1333 1334 expect("{"); 1335 while (!Error && !consume("}")) { 1336 StringRef Tok = next(); 1337 BaseCommand *Cmd = readProvideOrAssignment(Tok); 1338 if (!Cmd) { 1339 if (Tok == "ASSERT") 1340 Cmd = new AssertCommand(readAssert()); 1341 else 1342 Cmd = readOutputSectionDescription(Tok); 1343 } 1344 Opt.Commands.emplace_back(Cmd); 1345 } 1346 } 1347 1348 static int precedence(StringRef Op) { 1349 return StringSwitch<int>(Op) 1350 .Cases("*", "/", 5) 1351 .Cases("+", "-", 4) 1352 .Cases("<<", ">>", 3) 1353 .Cases("<", "<=", ">", ">=", "==", "!=", 2) 1354 .Cases("&", "|", 1) 1355 .Default(-1); 1356 } 1357 1358 StringMatcher ScriptParser::readFilePatterns() { 1359 std::vector<StringRef> V; 1360 while (!Error && !consume(")")) 1361 V.push_back(next()); 1362 return StringMatcher(V); 1363 } 1364 1365 SortSectionPolicy ScriptParser::readSortKind() { 1366 if (consume("SORT") || consume("SORT_BY_NAME")) 1367 return SortSectionPolicy::Name; 1368 if (consume("SORT_BY_ALIGNMENT")) 1369 return SortSectionPolicy::Alignment; 1370 if (consume("SORT_BY_INIT_PRIORITY")) 1371 return SortSectionPolicy::Priority; 1372 if (consume("SORT_NONE")) 1373 return SortSectionPolicy::None; 1374 return SortSectionPolicy::Default; 1375 } 1376 1377 // Method reads a list of sequence of excluded files and section globs given in 1378 // a following form: ((EXCLUDE_FILE(file_pattern+))? section_pattern+)+ 1379 // Example: *(.foo.1 EXCLUDE_FILE (*a.o) .foo.2 EXCLUDE_FILE (*b.o) .foo.3) 1380 // The semantics of that is next: 1381 // * Include .foo.1 from every file. 1382 // * Include .foo.2 from every file but a.o 1383 // * Include .foo.3 from every file but b.o 1384 std::vector<SectionPattern> ScriptParser::readInputSectionsList() { 1385 std::vector<SectionPattern> Ret; 1386 while (!Error && peek() != ")") { 1387 StringMatcher ExcludeFilePat; 1388 if (consume("EXCLUDE_FILE")) { 1389 expect("("); 1390 ExcludeFilePat = readFilePatterns(); 1391 } 1392 1393 std::vector<StringRef> V; 1394 while (!Error && peek() != ")" && peek() != "EXCLUDE_FILE") 1395 V.push_back(next()); 1396 1397 if (!V.empty()) 1398 Ret.push_back({std::move(ExcludeFilePat), StringMatcher(V)}); 1399 else 1400 setError("section pattern is expected"); 1401 } 1402 return Ret; 1403 } 1404 1405 // Reads contents of "SECTIONS" directive. That directive contains a 1406 // list of glob patterns for input sections. The grammar is as follows. 1407 // 1408 // <patterns> ::= <section-list> 1409 // | <sort> "(" <section-list> ")" 1410 // | <sort> "(" <sort> "(" <section-list> ")" ")" 1411 // 1412 // <sort> ::= "SORT" | "SORT_BY_NAME" | "SORT_BY_ALIGNMENT" 1413 // | "SORT_BY_INIT_PRIORITY" | "SORT_NONE" 1414 // 1415 // <section-list> is parsed by readInputSectionsList(). 1416 InputSectionDescription * 1417 ScriptParser::readInputSectionRules(StringRef FilePattern) { 1418 auto *Cmd = new InputSectionDescription(FilePattern); 1419 expect("("); 1420 while (!Error && !consume(")")) { 1421 SortSectionPolicy Outer = readSortKind(); 1422 SortSectionPolicy Inner = SortSectionPolicy::Default; 1423 std::vector<SectionPattern> V; 1424 if (Outer != SortSectionPolicy::Default) { 1425 expect("("); 1426 Inner = readSortKind(); 1427 if (Inner != SortSectionPolicy::Default) { 1428 expect("("); 1429 V = readInputSectionsList(); 1430 expect(")"); 1431 } else { 1432 V = readInputSectionsList(); 1433 } 1434 expect(")"); 1435 } else { 1436 V = readInputSectionsList(); 1437 } 1438 1439 for (SectionPattern &Pat : V) { 1440 Pat.SortInner = Inner; 1441 Pat.SortOuter = Outer; 1442 } 1443 1444 std::move(V.begin(), V.end(), std::back_inserter(Cmd->SectionPatterns)); 1445 } 1446 return Cmd; 1447 } 1448 1449 InputSectionDescription * 1450 ScriptParser::readInputSectionDescription(StringRef Tok) { 1451 // Input section wildcard can be surrounded by KEEP. 1452 // https://sourceware.org/binutils/docs/ld/Input-Section-Keep.html#Input-Section-Keep 1453 if (Tok == "KEEP") { 1454 expect("("); 1455 StringRef FilePattern = next(); 1456 InputSectionDescription *Cmd = readInputSectionRules(FilePattern); 1457 expect(")"); 1458 Opt.KeptSections.push_back(Cmd); 1459 return Cmd; 1460 } 1461 return readInputSectionRules(Tok); 1462 } 1463 1464 void ScriptParser::readSort() { 1465 expect("("); 1466 expect("CONSTRUCTORS"); 1467 expect(")"); 1468 } 1469 1470 Expr ScriptParser::readAssert() { 1471 expect("("); 1472 Expr E = readExpr(); 1473 expect(","); 1474 StringRef Msg = unquote(next()); 1475 expect(")"); 1476 return [=](uint64_t Dot) { 1477 if (!E(Dot)) 1478 error(Msg); 1479 return Dot; 1480 }; 1481 } 1482 1483 // Reads a FILL(expr) command. We handle the FILL command as an 1484 // alias for =fillexp section attribute, which is different from 1485 // what GNU linkers do. 1486 // https://sourceware.org/binutils/docs/ld/Output-Section-Data.html 1487 uint32_t ScriptParser::readFill() { 1488 expect("("); 1489 uint32_t V = readOutputSectionFiller(next()); 1490 expect(")"); 1491 expect(";"); 1492 return V; 1493 } 1494 1495 OutputSectionCommand * 1496 ScriptParser::readOutputSectionDescription(StringRef OutSec) { 1497 OutputSectionCommand *Cmd = new OutputSectionCommand(OutSec); 1498 Cmd->Location = getCurrentLocation(); 1499 1500 // Read an address expression. 1501 // https://sourceware.org/binutils/docs/ld/Output-Section-Address.html#Output-Section-Address 1502 if (peek() != ":") 1503 Cmd->AddrExpr = readExpr(); 1504 1505 expect(":"); 1506 1507 if (consume("AT")) 1508 Cmd->LMAExpr = readParenExpr(); 1509 if (consume("ALIGN")) 1510 Cmd->AlignExpr = readParenExpr(); 1511 if (consume("SUBALIGN")) 1512 Cmd->SubalignExpr = readParenExpr(); 1513 1514 // Parse constraints. 1515 if (consume("ONLY_IF_RO")) 1516 Cmd->Constraint = ConstraintKind::ReadOnly; 1517 if (consume("ONLY_IF_RW")) 1518 Cmd->Constraint = ConstraintKind::ReadWrite; 1519 expect("{"); 1520 1521 while (!Error && !consume("}")) { 1522 StringRef Tok = next(); 1523 if (Tok == ";") { 1524 // Empty commands are allowed. Do nothing here. 1525 } else if (SymbolAssignment *Assignment = readProvideOrAssignment(Tok)) { 1526 Cmd->Commands.emplace_back(Assignment); 1527 } else if (BytesDataCommand *Data = readBytesDataCommand(Tok)) { 1528 Cmd->Commands.emplace_back(Data); 1529 } else if (Tok == "ASSERT") { 1530 Cmd->Commands.emplace_back(new AssertCommand(readAssert())); 1531 expect(";"); 1532 } else if (Tok == "CONSTRUCTORS") { 1533 // CONSTRUCTORS is a keyword to make the linker recognize C++ ctors/dtors 1534 // by name. This is for very old file formats such as ECOFF/XCOFF. 1535 // For ELF, we should ignore. 1536 } else if (Tok == "FILL") { 1537 Cmd->Filler = readFill(); 1538 } else if (Tok == "SORT") { 1539 readSort(); 1540 } else if (peek() == "(") { 1541 Cmd->Commands.emplace_back(readInputSectionDescription(Tok)); 1542 } else { 1543 setError("unknown command " + Tok); 1544 } 1545 } 1546 1547 if (consume(">")) 1548 Cmd->MemoryRegionName = next(); 1549 1550 Cmd->Phdrs = readOutputSectionPhdrs(); 1551 1552 if (consume("=")) 1553 Cmd->Filler = readOutputSectionFiller(next()); 1554 else if (peek().startswith("=")) 1555 Cmd->Filler = readOutputSectionFiller(next().drop_front()); 1556 1557 // Consume optional comma following output section command. 1558 consume(","); 1559 1560 return Cmd; 1561 } 1562 1563 // Read "=<number>" where <number> is an octal/decimal/hexadecimal number. 1564 // https://sourceware.org/binutils/docs/ld/Output-Section-Fill.html 1565 // 1566 // ld.gold is not fully compatible with ld.bfd. ld.bfd handles 1567 // hexstrings as blobs of arbitrary sizes, while ld.gold handles them 1568 // as 32-bit big-endian values. We will do the same as ld.gold does 1569 // because it's simpler than what ld.bfd does. 1570 uint32_t ScriptParser::readOutputSectionFiller(StringRef Tok) { 1571 uint32_t V; 1572 if (!Tok.getAsInteger(0, V)) 1573 return V; 1574 setError("invalid filler expression: " + Tok); 1575 return 0; 1576 } 1577 1578 SymbolAssignment *ScriptParser::readProvideHidden(bool Provide, bool Hidden) { 1579 expect("("); 1580 SymbolAssignment *Cmd = readAssignment(next()); 1581 Cmd->Provide = Provide; 1582 Cmd->Hidden = Hidden; 1583 expect(")"); 1584 expect(";"); 1585 return Cmd; 1586 } 1587 1588 SymbolAssignment *ScriptParser::readProvideOrAssignment(StringRef Tok) { 1589 SymbolAssignment *Cmd = nullptr; 1590 if (peek() == "=" || peek() == "+=") { 1591 Cmd = readAssignment(Tok); 1592 expect(";"); 1593 } else if (Tok == "PROVIDE") { 1594 Cmd = readProvideHidden(true, false); 1595 } else if (Tok == "HIDDEN") { 1596 Cmd = readProvideHidden(false, true); 1597 } else if (Tok == "PROVIDE_HIDDEN") { 1598 Cmd = readProvideHidden(true, true); 1599 } 1600 return Cmd; 1601 } 1602 1603 static uint64_t getSymbolValue(const Twine &Loc, StringRef S, uint64_t Dot) { 1604 if (S == ".") 1605 return Dot; 1606 return ScriptBase->getSymbolValue(Loc, S); 1607 } 1608 1609 static bool isAbsolute(StringRef S) { 1610 if (S == ".") 1611 return false; 1612 return ScriptBase->isAbsolute(S); 1613 } 1614 1615 SymbolAssignment *ScriptParser::readAssignment(StringRef Name) { 1616 StringRef Op = next(); 1617 Expr E; 1618 assert(Op == "=" || Op == "+="); 1619 if (consume("ABSOLUTE")) { 1620 E = readExpr(); 1621 E.IsAbsolute = [] { return true; }; 1622 } else { 1623 E = readExpr(); 1624 } 1625 if (Op == "+=") { 1626 std::string Loc = getCurrentLocation(); 1627 E = [=](uint64_t Dot) { 1628 return getSymbolValue(Loc, Name, Dot) + E(Dot); 1629 }; 1630 } 1631 return new SymbolAssignment(Name, E, getCurrentLocation()); 1632 } 1633 1634 // This is an operator-precedence parser to parse a linker 1635 // script expression. 1636 Expr ScriptParser::readExpr() { 1637 // Our lexer is context-aware. Set the in-expression bit so that 1638 // they apply different tokenization rules. 1639 bool Orig = InExpr; 1640 InExpr = true; 1641 Expr E = readExpr1(readPrimary(), 0); 1642 InExpr = Orig; 1643 return E; 1644 } 1645 1646 static Expr combine(StringRef Op, Expr L, Expr R) { 1647 auto IsAbs = [=] { return L.IsAbsolute() && R.IsAbsolute(); }; 1648 auto GetOutSec = [=] { 1649 const OutputSectionBase *S = L.Section(); 1650 return S ? S : R.Section(); 1651 }; 1652 1653 if (Op == "*") 1654 return [=](uint64_t Dot) { return L(Dot) * R(Dot); }; 1655 if (Op == "/") { 1656 return [=](uint64_t Dot) -> uint64_t { 1657 uint64_t RHS = R(Dot); 1658 if (RHS == 0) { 1659 error("division by zero"); 1660 return 0; 1661 } 1662 return L(Dot) / RHS; 1663 }; 1664 } 1665 if (Op == "+") 1666 return {[=](uint64_t Dot) { return L(Dot) + R(Dot); }, IsAbs, GetOutSec}; 1667 if (Op == "-") 1668 return {[=](uint64_t Dot) { return L(Dot) - R(Dot); }, IsAbs, GetOutSec}; 1669 if (Op == "<<") 1670 return [=](uint64_t Dot) { return L(Dot) << R(Dot); }; 1671 if (Op == ">>") 1672 return [=](uint64_t Dot) { return L(Dot) >> R(Dot); }; 1673 if (Op == "<") 1674 return [=](uint64_t Dot) { return L(Dot) < R(Dot); }; 1675 if (Op == ">") 1676 return [=](uint64_t Dot) { return L(Dot) > R(Dot); }; 1677 if (Op == ">=") 1678 return [=](uint64_t Dot) { return L(Dot) >= R(Dot); }; 1679 if (Op == "<=") 1680 return [=](uint64_t Dot) { return L(Dot) <= R(Dot); }; 1681 if (Op == "==") 1682 return [=](uint64_t Dot) { return L(Dot) == R(Dot); }; 1683 if (Op == "!=") 1684 return [=](uint64_t Dot) { return L(Dot) != R(Dot); }; 1685 if (Op == "&") 1686 return [=](uint64_t Dot) { return L(Dot) & R(Dot); }; 1687 if (Op == "|") 1688 return [=](uint64_t Dot) { return L(Dot) | R(Dot); }; 1689 llvm_unreachable("invalid operator"); 1690 } 1691 1692 // This is a part of the operator-precedence parser. This function 1693 // assumes that the remaining token stream starts with an operator. 1694 Expr ScriptParser::readExpr1(Expr Lhs, int MinPrec) { 1695 while (!atEOF() && !Error) { 1696 // Read an operator and an expression. 1697 if (consume("?")) 1698 return readTernary(Lhs); 1699 StringRef Op1 = peek(); 1700 if (precedence(Op1) < MinPrec) 1701 break; 1702 skip(); 1703 Expr Rhs = readPrimary(); 1704 1705 // Evaluate the remaining part of the expression first if the 1706 // next operator has greater precedence than the previous one. 1707 // For example, if we have read "+" and "3", and if the next 1708 // operator is "*", then we'll evaluate 3 * ... part first. 1709 while (!atEOF()) { 1710 StringRef Op2 = peek(); 1711 if (precedence(Op2) <= precedence(Op1)) 1712 break; 1713 Rhs = readExpr1(Rhs, precedence(Op2)); 1714 } 1715 1716 Lhs = combine(Op1, Lhs, Rhs); 1717 } 1718 return Lhs; 1719 } 1720 1721 uint64_t static getConstant(StringRef S) { 1722 if (S == "COMMONPAGESIZE") 1723 return Target->PageSize; 1724 if (S == "MAXPAGESIZE") 1725 return Config->MaxPageSize; 1726 error("unknown constant: " + S); 1727 return 0; 1728 } 1729 1730 // Parses Tok as an integer. Returns true if successful. 1731 // It recognizes hexadecimal (prefixed with "0x" or suffixed with "H") 1732 // and decimal numbers. Decimal numbers may have "K" (kilo) or 1733 // "M" (mega) prefixes. 1734 static bool readInteger(StringRef Tok, uint64_t &Result) { 1735 // Negative number 1736 if (Tok.startswith("-")) { 1737 if (!readInteger(Tok.substr(1), Result)) 1738 return false; 1739 Result = -Result; 1740 return true; 1741 } 1742 1743 // Hexadecimal 1744 if (Tok.startswith_lower("0x")) 1745 return !Tok.substr(2).getAsInteger(16, Result); 1746 if (Tok.endswith_lower("H")) 1747 return !Tok.drop_back().getAsInteger(16, Result); 1748 1749 // Decimal 1750 int Suffix = 1; 1751 if (Tok.endswith_lower("K")) { 1752 Suffix = 1024; 1753 Tok = Tok.drop_back(); 1754 } else if (Tok.endswith_lower("M")) { 1755 Suffix = 1024 * 1024; 1756 Tok = Tok.drop_back(); 1757 } 1758 if (Tok.getAsInteger(10, Result)) 1759 return false; 1760 Result *= Suffix; 1761 return true; 1762 } 1763 1764 BytesDataCommand *ScriptParser::readBytesDataCommand(StringRef Tok) { 1765 int Size = StringSwitch<unsigned>(Tok) 1766 .Case("BYTE", 1) 1767 .Case("SHORT", 2) 1768 .Case("LONG", 4) 1769 .Case("QUAD", 8) 1770 .Default(-1); 1771 if (Size == -1) 1772 return nullptr; 1773 1774 return new BytesDataCommand(readParenExpr(), Size); 1775 } 1776 1777 StringRef ScriptParser::readParenLiteral() { 1778 expect("("); 1779 StringRef Tok = next(); 1780 expect(")"); 1781 return Tok; 1782 } 1783 1784 Expr ScriptParser::readPrimary() { 1785 if (peek() == "(") 1786 return readParenExpr(); 1787 1788 StringRef Tok = next(); 1789 std::string Location = getCurrentLocation(); 1790 1791 if (Tok == "~") { 1792 Expr E = readPrimary(); 1793 return [=](uint64_t Dot) { return ~E(Dot); }; 1794 } 1795 if (Tok == "-") { 1796 Expr E = readPrimary(); 1797 return [=](uint64_t Dot) { return -E(Dot); }; 1798 } 1799 1800 // Built-in functions are parsed here. 1801 // https://sourceware.org/binutils/docs/ld/Builtin-Functions.html. 1802 if (Tok == "ADDR") { 1803 StringRef Name = readParenLiteral(); 1804 return {[=](uint64_t Dot) { 1805 return ScriptBase->getOutputSection(Location, Name)->Addr; 1806 }, 1807 [=] { return false; }, 1808 [=] { return ScriptBase->getOutputSection(Location, Name); }}; 1809 } 1810 if (Tok == "LOADADDR") { 1811 StringRef Name = readParenLiteral(); 1812 return [=](uint64_t Dot) { 1813 return ScriptBase->getOutputSection(Location, Name)->getLMA(); 1814 }; 1815 } 1816 if (Tok == "ASSERT") 1817 return readAssert(); 1818 if (Tok == "ALIGN") { 1819 expect("("); 1820 Expr E = readExpr(); 1821 if (consume(",")) { 1822 Expr E2 = readExpr(); 1823 expect(")"); 1824 return [=](uint64_t Dot) { return alignTo(E(Dot), E2(Dot)); }; 1825 } 1826 expect(")"); 1827 return [=](uint64_t Dot) { return alignTo(Dot, E(Dot)); }; 1828 } 1829 if (Tok == "CONSTANT") { 1830 StringRef Name = readParenLiteral(); 1831 return [=](uint64_t Dot) { return getConstant(Name); }; 1832 } 1833 if (Tok == "DEFINED") { 1834 StringRef Name = readParenLiteral(); 1835 return [=](uint64_t Dot) { return ScriptBase->isDefined(Name) ? 1 : 0; }; 1836 } 1837 if (Tok == "SEGMENT_START") { 1838 expect("("); 1839 skip(); 1840 expect(","); 1841 Expr E = readExpr(); 1842 expect(")"); 1843 return [=](uint64_t Dot) { return E(Dot); }; 1844 } 1845 if (Tok == "DATA_SEGMENT_ALIGN") { 1846 expect("("); 1847 Expr E = readExpr(); 1848 expect(","); 1849 readExpr(); 1850 expect(")"); 1851 return [=](uint64_t Dot) { return alignTo(Dot, E(Dot)); }; 1852 } 1853 if (Tok == "DATA_SEGMENT_END") { 1854 expect("("); 1855 expect("."); 1856 expect(")"); 1857 return [](uint64_t Dot) { return Dot; }; 1858 } 1859 // GNU linkers implements more complicated logic to handle 1860 // DATA_SEGMENT_RELRO_END. We instead ignore the arguments and just align to 1861 // the next page boundary for simplicity. 1862 if (Tok == "DATA_SEGMENT_RELRO_END") { 1863 expect("("); 1864 readExpr(); 1865 expect(","); 1866 readExpr(); 1867 expect(")"); 1868 return [](uint64_t Dot) { return alignTo(Dot, Target->PageSize); }; 1869 } 1870 if (Tok == "SIZEOF") { 1871 StringRef Name = readParenLiteral(); 1872 return [=](uint64_t Dot) { return ScriptBase->getOutputSectionSize(Name); }; 1873 } 1874 if (Tok == "ALIGNOF") { 1875 StringRef Name = readParenLiteral(); 1876 return [=](uint64_t Dot) { 1877 return ScriptBase->getOutputSection(Location, Name)->Addralign; 1878 }; 1879 } 1880 if (Tok == "SIZEOF_HEADERS") 1881 return [=](uint64_t Dot) { return ScriptBase->getHeaderSize(); }; 1882 1883 // Tok is a literal number. 1884 uint64_t V; 1885 if (readInteger(Tok, V)) 1886 return [=](uint64_t Dot) { return V; }; 1887 1888 // Tok is a symbol name. 1889 if (Tok != "." && !isValidCIdentifier(Tok)) 1890 setError("malformed number: " + Tok); 1891 return {[=](uint64_t Dot) { return getSymbolValue(Location, Tok, Dot); }, 1892 [=] { return isAbsolute(Tok); }, 1893 [=] { return ScriptBase->getSymbolSection(Tok); }}; 1894 } 1895 1896 Expr ScriptParser::readTernary(Expr Cond) { 1897 Expr L = readExpr(); 1898 expect(":"); 1899 Expr R = readExpr(); 1900 return [=](uint64_t Dot) { return Cond(Dot) ? L(Dot) : R(Dot); }; 1901 } 1902 1903 Expr ScriptParser::readParenExpr() { 1904 expect("("); 1905 Expr E = readExpr(); 1906 expect(")"); 1907 return E; 1908 } 1909 1910 std::vector<StringRef> ScriptParser::readOutputSectionPhdrs() { 1911 std::vector<StringRef> Phdrs; 1912 while (!Error && peek().startswith(":")) { 1913 StringRef Tok = next(); 1914 Phdrs.push_back((Tok.size() == 1) ? next() : Tok.substr(1)); 1915 } 1916 return Phdrs; 1917 } 1918 1919 // Read a program header type name. The next token must be a 1920 // name of a program header type or a constant (e.g. "0x3"). 1921 unsigned ScriptParser::readPhdrType() { 1922 StringRef Tok = next(); 1923 uint64_t Val; 1924 if (readInteger(Tok, Val)) 1925 return Val; 1926 1927 unsigned Ret = StringSwitch<unsigned>(Tok) 1928 .Case("PT_NULL", PT_NULL) 1929 .Case("PT_LOAD", PT_LOAD) 1930 .Case("PT_DYNAMIC", PT_DYNAMIC) 1931 .Case("PT_INTERP", PT_INTERP) 1932 .Case("PT_NOTE", PT_NOTE) 1933 .Case("PT_SHLIB", PT_SHLIB) 1934 .Case("PT_PHDR", PT_PHDR) 1935 .Case("PT_TLS", PT_TLS) 1936 .Case("PT_GNU_EH_FRAME", PT_GNU_EH_FRAME) 1937 .Case("PT_GNU_STACK", PT_GNU_STACK) 1938 .Case("PT_GNU_RELRO", PT_GNU_RELRO) 1939 .Case("PT_OPENBSD_RANDOMIZE", PT_OPENBSD_RANDOMIZE) 1940 .Case("PT_OPENBSD_WXNEEDED", PT_OPENBSD_WXNEEDED) 1941 .Case("PT_OPENBSD_BOOTDATA", PT_OPENBSD_BOOTDATA) 1942 .Default(-1); 1943 1944 if (Ret == (unsigned)-1) { 1945 setError("invalid program header type: " + Tok); 1946 return PT_NULL; 1947 } 1948 return Ret; 1949 } 1950 1951 // Reads a list of symbols, e.g. "{ global: foo; bar; local: *; };". 1952 void ScriptParser::readAnonymousDeclaration() { 1953 // Read global symbols first. "global:" is default, so if there's 1954 // no label, we assume global symbols. 1955 if (peek() != "local") { 1956 if (consume("global")) 1957 expect(":"); 1958 for (SymbolVersion V : readSymbols()) 1959 Config->VersionScriptGlobals.push_back(V); 1960 } 1961 readLocals(); 1962 expect("}"); 1963 expect(";"); 1964 } 1965 1966 void ScriptParser::readLocals() { 1967 if (!consume("local")) 1968 return; 1969 expect(":"); 1970 std::vector<SymbolVersion> Locals = readSymbols(); 1971 for (SymbolVersion V : Locals) { 1972 if (V.Name == "*") { 1973 Config->DefaultSymbolVersion = VER_NDX_LOCAL; 1974 continue; 1975 } 1976 Config->VersionScriptLocals.push_back(V); 1977 } 1978 } 1979 1980 // Reads a list of symbols, e.g. "VerStr { global: foo; bar; local: *; };". 1981 void ScriptParser::readVersionDeclaration(StringRef VerStr) { 1982 // Identifiers start at 2 because 0 and 1 are reserved 1983 // for VER_NDX_LOCAL and VER_NDX_GLOBAL constants. 1984 uint16_t VersionId = Config->VersionDefinitions.size() + 2; 1985 Config->VersionDefinitions.push_back({VerStr, VersionId}); 1986 1987 // Read global symbols. 1988 if (peek() != "local") { 1989 if (consume("global")) 1990 expect(":"); 1991 Config->VersionDefinitions.back().Globals = readSymbols(); 1992 } 1993 readLocals(); 1994 expect("}"); 1995 1996 // Each version may have a parent version. For example, "Ver2" 1997 // defined as "Ver2 { global: foo; local: *; } Ver1;" has "Ver1" 1998 // as a parent. This version hierarchy is, probably against your 1999 // instinct, purely for hint; the runtime doesn't care about it 2000 // at all. In LLD, we simply ignore it. 2001 if (peek() != ";") 2002 skip(); 2003 expect(";"); 2004 } 2005 2006 // Reads a list of symbols for a versions cript. 2007 std::vector<SymbolVersion> ScriptParser::readSymbols() { 2008 std::vector<SymbolVersion> Ret; 2009 for (;;) { 2010 if (consume("extern")) { 2011 for (SymbolVersion V : readVersionExtern()) 2012 Ret.push_back(V); 2013 continue; 2014 } 2015 2016 if (peek() == "}" || (peek() == "local" && peek(1) == ":") || Error) 2017 break; 2018 StringRef Tok = next(); 2019 Ret.push_back({unquote(Tok), false, hasWildcard(Tok)}); 2020 expect(";"); 2021 } 2022 return Ret; 2023 } 2024 2025 // Reads an "extern C++" directive, e.g., 2026 // "extern "C++" { ns::*; "f(int, double)"; };" 2027 std::vector<SymbolVersion> ScriptParser::readVersionExtern() { 2028 StringRef Tok = next(); 2029 bool IsCXX = Tok == "\"C++\""; 2030 if (!IsCXX && Tok != "\"C\"") 2031 setError("Unknown language"); 2032 expect("{"); 2033 2034 std::vector<SymbolVersion> Ret; 2035 while (!Error && peek() != "}") { 2036 StringRef Tok = next(); 2037 bool HasWildcard = !Tok.startswith("\"") && hasWildcard(Tok); 2038 Ret.push_back({unquote(Tok), IsCXX, HasWildcard}); 2039 expect(";"); 2040 } 2041 2042 expect("}"); 2043 expect(";"); 2044 return Ret; 2045 } 2046 2047 uint64_t ScriptParser::readMemoryAssignment( 2048 StringRef S1, StringRef S2, StringRef S3) { 2049 if (!(consume(S1) || consume(S2) || consume(S3))) { 2050 setError("expected one of: " + S1 + ", " + S2 + ", or " + S3); 2051 return 0; 2052 } 2053 expect("="); 2054 2055 // TODO: Fully support constant expressions. 2056 uint64_t Val; 2057 if (!readInteger(next(), Val)) 2058 setError("nonconstant expression for "+ S1); 2059 return Val; 2060 } 2061 2062 // Parse the MEMORY command as specified in: 2063 // https://sourceware.org/binutils/docs/ld/MEMORY.html 2064 // 2065 // MEMORY { name [(attr)] : ORIGIN = origin, LENGTH = len ... } 2066 void ScriptParser::readMemory() { 2067 expect("{"); 2068 while (!Error && !consume("}")) { 2069 StringRef Name = next(); 2070 2071 uint32_t Flags = 0; 2072 uint32_t NegFlags = 0; 2073 if (consume("(")) { 2074 std::tie(Flags, NegFlags) = readMemoryAttributes(); 2075 expect(")"); 2076 } 2077 expect(":"); 2078 2079 uint64_t Origin = readMemoryAssignment("ORIGIN", "org", "o"); 2080 expect(","); 2081 uint64_t Length = readMemoryAssignment("LENGTH", "len", "l"); 2082 2083 // Add the memory region to the region map (if it doesn't already exist). 2084 auto It = Opt.MemoryRegions.find(Name); 2085 if (It != Opt.MemoryRegions.end()) 2086 setError("region '" + Name + "' already defined"); 2087 else 2088 Opt.MemoryRegions[Name] = {Name, Origin, Length, Origin, Flags, NegFlags}; 2089 } 2090 } 2091 2092 // This function parses the attributes used to match against section 2093 // flags when placing output sections in a memory region. These flags 2094 // are only used when an explicit memory region name is not used. 2095 std::pair<uint32_t, uint32_t> ScriptParser::readMemoryAttributes() { 2096 uint32_t Flags = 0; 2097 uint32_t NegFlags = 0; 2098 bool Invert = false; 2099 2100 for (char C : next().lower()) { 2101 uint32_t Flag = 0; 2102 if (C == '!') 2103 Invert = !Invert; 2104 else if (C == 'w') 2105 Flag = SHF_WRITE; 2106 else if (C == 'x') 2107 Flag = SHF_EXECINSTR; 2108 else if (C == 'a') 2109 Flag = SHF_ALLOC; 2110 else if (C != 'r') 2111 setError("invalid memory region attribute"); 2112 2113 if (Invert) 2114 NegFlags |= Flag; 2115 else 2116 Flags |= Flag; 2117 } 2118 return {Flags, NegFlags}; 2119 } 2120 2121 void elf::readLinkerScript(MemoryBufferRef MB) { 2122 ScriptParser(MB).readLinkerScript(); 2123 } 2124 2125 void elf::readVersionScript(MemoryBufferRef MB) { 2126 ScriptParser(MB).readVersionScript(); 2127 } 2128 2129 void elf::readDynamicList(MemoryBufferRef MB) { 2130 ScriptParser(MB).readDynamicList(); 2131 } 2132 2133 template class elf::LinkerScript<ELF32LE>; 2134 template class elf::LinkerScript<ELF32BE>; 2135 template class elf::LinkerScript<ELF64LE>; 2136 template class elf::LinkerScript<ELF64BE>; 2137