1 //===-- X86AsmBackend.cpp - X86 Assembler Backend -------------------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 9 #include "MCTargetDesc/X86BaseInfo.h" 10 #include "MCTargetDesc/X86FixupKinds.h" 11 #include "llvm/ADT/StringSwitch.h" 12 #include "llvm/BinaryFormat/ELF.h" 13 #include "llvm/BinaryFormat/MachO.h" 14 #include "llvm/MC/MCAsmBackend.h" 15 #include "llvm/MC/MCAsmLayout.h" 16 #include "llvm/MC/MCAssembler.h" 17 #include "llvm/MC/MCCodeEmitter.h" 18 #include "llvm/MC/MCContext.h" 19 #include "llvm/MC/MCDwarf.h" 20 #include "llvm/MC/MCELFObjectWriter.h" 21 #include "llvm/MC/MCExpr.h" 22 #include "llvm/MC/MCFixupKindInfo.h" 23 #include "llvm/MC/MCInst.h" 24 #include "llvm/MC/MCInstrInfo.h" 25 #include "llvm/MC/MCMachObjectWriter.h" 26 #include "llvm/MC/MCObjectStreamer.h" 27 #include "llvm/MC/MCObjectWriter.h" 28 #include "llvm/MC/MCRegisterInfo.h" 29 #include "llvm/MC/MCSectionMachO.h" 30 #include "llvm/MC/MCSubtargetInfo.h" 31 #include "llvm/MC/MCValue.h" 32 #include "llvm/Support/CommandLine.h" 33 #include "llvm/Support/ErrorHandling.h" 34 #include "llvm/Support/TargetRegistry.h" 35 #include "llvm/Support/raw_ostream.h" 36 37 using namespace llvm; 38 39 namespace { 40 /// A wrapper for holding a mask of the values from X86::AlignBranchBoundaryKind 41 class X86AlignBranchKind { 42 private: 43 uint8_t AlignBranchKind = 0; 44 45 public: 46 void operator=(const std::string &Val) { 47 if (Val.empty()) 48 return; 49 SmallVector<StringRef, 6> BranchTypes; 50 StringRef(Val).split(BranchTypes, '+', -1, false); 51 for (auto BranchType : BranchTypes) { 52 if (BranchType == "fused") 53 addKind(X86::AlignBranchFused); 54 else if (BranchType == "jcc") 55 addKind(X86::AlignBranchJcc); 56 else if (BranchType == "jmp") 57 addKind(X86::AlignBranchJmp); 58 else if (BranchType == "call") 59 addKind(X86::AlignBranchCall); 60 else if (BranchType == "ret") 61 addKind(X86::AlignBranchRet); 62 else if (BranchType == "indirect") 63 addKind(X86::AlignBranchIndirect); 64 else { 65 report_fatal_error( 66 "'-x86-align-branch 'The branches's type is combination of jcc, " 67 "fused, jmp, call, ret, indirect.(plus separated)", 68 false); 69 } 70 } 71 } 72 73 operator uint8_t() const { return AlignBranchKind; } 74 void addKind(X86::AlignBranchBoundaryKind Value) { AlignBranchKind |= Value; } 75 }; 76 77 X86AlignBranchKind X86AlignBranchKindLoc; 78 79 cl::opt<unsigned> X86AlignBranchBoundary( 80 "x86-align-branch-boundary", cl::init(0), 81 cl::desc( 82 "Control how the assembler should align branches with NOP. If the " 83 "boundary's size is not 0, it should be a power of 2 and no less " 84 "than 32. Branches will be aligned to prevent from being across or " 85 "against the boundary of specified size. The default value 0 does not " 86 "align branches.")); 87 88 cl::opt<X86AlignBranchKind, true, cl::parser<std::string>> X86AlignBranch( 89 "x86-align-branch", 90 cl::desc( 91 "Specify types of branches to align (plus separated list of types):" 92 "\njcc indicates conditional jumps" 93 "\nfused indicates fused conditional jumps" 94 "\njmp indicates direct unconditional jumps" 95 "\ncall indicates direct and indirect calls" 96 "\nret indicates rets" 97 "\nindirect indicates indirect unconditional jumps"), 98 cl::location(X86AlignBranchKindLoc)); 99 100 cl::opt<bool> X86AlignBranchWithin32BBoundaries( 101 "x86-branches-within-32B-boundaries", cl::init(false), 102 cl::desc( 103 "Align selected instructions to mitigate negative performance impact " 104 "of Intel's micro code update for errata skx102. May break " 105 "assumptions about labels corresponding to particular instructions, " 106 "and should be used with caution.")); 107 108 cl::opt<unsigned> X86PadMaxPrefixSize( 109 "x86-pad-max-prefix-size", cl::init(0), 110 cl::desc("Maximum number of prefixes to use for padding")); 111 112 cl::opt<bool> X86PadForAlign( 113 "x86-pad-for-align", cl::init(true), cl::Hidden, 114 cl::desc("Pad previous instructions to implement align directives")); 115 116 cl::opt<bool> X86PadForBranchAlign( 117 "x86-pad-for-branch-align", cl::init(true), cl::Hidden, 118 cl::desc("Pad previous instructions to implement branch alignment")); 119 120 class X86ELFObjectWriter : public MCELFObjectTargetWriter { 121 public: 122 X86ELFObjectWriter(bool is64Bit, uint8_t OSABI, uint16_t EMachine, 123 bool HasRelocationAddend, bool foobar) 124 : MCELFObjectTargetWriter(is64Bit, OSABI, EMachine, HasRelocationAddend) {} 125 }; 126 127 class X86AsmBackend : public MCAsmBackend { 128 const MCSubtargetInfo &STI; 129 std::unique_ptr<const MCInstrInfo> MCII; 130 X86AlignBranchKind AlignBranchType; 131 Align AlignBoundary; 132 133 uint8_t determinePaddingPrefix(const MCInst &Inst) const; 134 135 bool isMacroFused(const MCInst &Cmp, const MCInst &Jcc) const; 136 137 bool needAlign(MCObjectStreamer &OS) const; 138 bool needAlignInst(const MCInst &Inst) const; 139 MCInst PrevInst; 140 MCBoundaryAlignFragment *PendingBoundaryAlign = nullptr; 141 std::pair<MCFragment *, size_t> PrevInstPosition; 142 143 public: 144 X86AsmBackend(const Target &T, const MCSubtargetInfo &STI) 145 : MCAsmBackend(support::little), STI(STI), 146 MCII(T.createMCInstrInfo()) { 147 if (X86AlignBranchWithin32BBoundaries) { 148 // At the moment, this defaults to aligning fused branches, unconditional 149 // jumps, and (unfused) conditional jumps with nops. Both the 150 // instructions aligned and the alignment method (nop vs prefix) may 151 // change in the future. 152 AlignBoundary = assumeAligned(32);; 153 AlignBranchType.addKind(X86::AlignBranchFused); 154 AlignBranchType.addKind(X86::AlignBranchJcc); 155 AlignBranchType.addKind(X86::AlignBranchJmp); 156 } 157 // Allow overriding defaults set by master flag 158 if (X86AlignBranchBoundary.getNumOccurrences()) 159 AlignBoundary = assumeAligned(X86AlignBranchBoundary); 160 if (X86AlignBranch.getNumOccurrences()) 161 AlignBranchType = X86AlignBranchKindLoc; 162 } 163 164 bool allowAutoPadding() const override; 165 void emitInstructionBegin(MCObjectStreamer &OS, const MCInst &Inst) override; 166 void emitInstructionEnd(MCObjectStreamer &OS, const MCInst &Inst) override; 167 168 unsigned getNumFixupKinds() const override { 169 return X86::NumTargetFixupKinds; 170 } 171 172 Optional<MCFixupKind> getFixupKind(StringRef Name) const override; 173 174 const MCFixupKindInfo &getFixupKindInfo(MCFixupKind Kind) const override; 175 176 bool shouldForceRelocation(const MCAssembler &Asm, const MCFixup &Fixup, 177 const MCValue &Target) override; 178 179 void applyFixup(const MCAssembler &Asm, const MCFixup &Fixup, 180 const MCValue &Target, MutableArrayRef<char> Data, 181 uint64_t Value, bool IsResolved, 182 const MCSubtargetInfo *STI) const override; 183 184 bool mayNeedRelaxation(const MCInst &Inst, 185 const MCSubtargetInfo &STI) const override; 186 187 bool fixupNeedsRelaxation(const MCFixup &Fixup, uint64_t Value, 188 const MCRelaxableFragment *DF, 189 const MCAsmLayout &Layout) const override; 190 191 void relaxInstruction(const MCInst &Inst, const MCSubtargetInfo &STI, 192 MCInst &Res) const override; 193 194 bool padInstructionViaRelaxation(MCRelaxableFragment &RF, 195 MCCodeEmitter &Emitter, 196 unsigned &RemainingSize) const; 197 198 bool padInstructionViaPrefix(MCRelaxableFragment &RF, MCCodeEmitter &Emitter, 199 unsigned &RemainingSize) const; 200 201 bool padInstructionEncoding(MCRelaxableFragment &RF, MCCodeEmitter &Emitter, 202 unsigned &RemainingSize) const; 203 204 void finishLayout(MCAssembler const &Asm, MCAsmLayout &Layout) const override; 205 206 bool writeNopData(raw_ostream &OS, uint64_t Count) const override; 207 }; 208 } // end anonymous namespace 209 210 static unsigned getRelaxedOpcodeBranch(const MCInst &Inst, bool Is16BitMode) { 211 unsigned Op = Inst.getOpcode(); 212 switch (Op) { 213 default: 214 return Op; 215 case X86::JCC_1: 216 return (Is16BitMode) ? X86::JCC_2 : X86::JCC_4; 217 case X86::JMP_1: 218 return (Is16BitMode) ? X86::JMP_2 : X86::JMP_4; 219 } 220 } 221 222 static unsigned getRelaxedOpcodeArith(const MCInst &Inst) { 223 unsigned Op = Inst.getOpcode(); 224 switch (Op) { 225 default: 226 return Op; 227 228 // IMUL 229 case X86::IMUL16rri8: return X86::IMUL16rri; 230 case X86::IMUL16rmi8: return X86::IMUL16rmi; 231 case X86::IMUL32rri8: return X86::IMUL32rri; 232 case X86::IMUL32rmi8: return X86::IMUL32rmi; 233 case X86::IMUL64rri8: return X86::IMUL64rri32; 234 case X86::IMUL64rmi8: return X86::IMUL64rmi32; 235 236 // AND 237 case X86::AND16ri8: return X86::AND16ri; 238 case X86::AND16mi8: return X86::AND16mi; 239 case X86::AND32ri8: return X86::AND32ri; 240 case X86::AND32mi8: return X86::AND32mi; 241 case X86::AND64ri8: return X86::AND64ri32; 242 case X86::AND64mi8: return X86::AND64mi32; 243 244 // OR 245 case X86::OR16ri8: return X86::OR16ri; 246 case X86::OR16mi8: return X86::OR16mi; 247 case X86::OR32ri8: return X86::OR32ri; 248 case X86::OR32mi8: return X86::OR32mi; 249 case X86::OR64ri8: return X86::OR64ri32; 250 case X86::OR64mi8: return X86::OR64mi32; 251 252 // XOR 253 case X86::XOR16ri8: return X86::XOR16ri; 254 case X86::XOR16mi8: return X86::XOR16mi; 255 case X86::XOR32ri8: return X86::XOR32ri; 256 case X86::XOR32mi8: return X86::XOR32mi; 257 case X86::XOR64ri8: return X86::XOR64ri32; 258 case X86::XOR64mi8: return X86::XOR64mi32; 259 260 // ADD 261 case X86::ADD16ri8: return X86::ADD16ri; 262 case X86::ADD16mi8: return X86::ADD16mi; 263 case X86::ADD32ri8: return X86::ADD32ri; 264 case X86::ADD32mi8: return X86::ADD32mi; 265 case X86::ADD64ri8: return X86::ADD64ri32; 266 case X86::ADD64mi8: return X86::ADD64mi32; 267 268 // ADC 269 case X86::ADC16ri8: return X86::ADC16ri; 270 case X86::ADC16mi8: return X86::ADC16mi; 271 case X86::ADC32ri8: return X86::ADC32ri; 272 case X86::ADC32mi8: return X86::ADC32mi; 273 case X86::ADC64ri8: return X86::ADC64ri32; 274 case X86::ADC64mi8: return X86::ADC64mi32; 275 276 // SUB 277 case X86::SUB16ri8: return X86::SUB16ri; 278 case X86::SUB16mi8: return X86::SUB16mi; 279 case X86::SUB32ri8: return X86::SUB32ri; 280 case X86::SUB32mi8: return X86::SUB32mi; 281 case X86::SUB64ri8: return X86::SUB64ri32; 282 case X86::SUB64mi8: return X86::SUB64mi32; 283 284 // SBB 285 case X86::SBB16ri8: return X86::SBB16ri; 286 case X86::SBB16mi8: return X86::SBB16mi; 287 case X86::SBB32ri8: return X86::SBB32ri; 288 case X86::SBB32mi8: return X86::SBB32mi; 289 case X86::SBB64ri8: return X86::SBB64ri32; 290 case X86::SBB64mi8: return X86::SBB64mi32; 291 292 // CMP 293 case X86::CMP16ri8: return X86::CMP16ri; 294 case X86::CMP16mi8: return X86::CMP16mi; 295 case X86::CMP32ri8: return X86::CMP32ri; 296 case X86::CMP32mi8: return X86::CMP32mi; 297 case X86::CMP64ri8: return X86::CMP64ri32; 298 case X86::CMP64mi8: return X86::CMP64mi32; 299 300 // PUSH 301 case X86::PUSH32i8: return X86::PUSHi32; 302 case X86::PUSH16i8: return X86::PUSHi16; 303 case X86::PUSH64i8: return X86::PUSH64i32; 304 } 305 } 306 307 static unsigned getRelaxedOpcode(const MCInst &Inst, bool Is16BitMode) { 308 unsigned R = getRelaxedOpcodeArith(Inst); 309 if (R != Inst.getOpcode()) 310 return R; 311 return getRelaxedOpcodeBranch(Inst, Is16BitMode); 312 } 313 314 static X86::CondCode getCondFromBranch(const MCInst &MI, 315 const MCInstrInfo &MCII) { 316 unsigned Opcode = MI.getOpcode(); 317 switch (Opcode) { 318 default: 319 return X86::COND_INVALID; 320 case X86::JCC_1: { 321 const MCInstrDesc &Desc = MCII.get(Opcode); 322 return static_cast<X86::CondCode>( 323 MI.getOperand(Desc.getNumOperands() - 1).getImm()); 324 } 325 } 326 } 327 328 static X86::SecondMacroFusionInstKind 329 classifySecondInstInMacroFusion(const MCInst &MI, const MCInstrInfo &MCII) { 330 X86::CondCode CC = getCondFromBranch(MI, MCII); 331 return classifySecondCondCodeInMacroFusion(CC); 332 } 333 334 /// Check if the instruction uses RIP relative addressing. 335 static bool isRIPRelative(const MCInst &MI, const MCInstrInfo &MCII) { 336 unsigned Opcode = MI.getOpcode(); 337 const MCInstrDesc &Desc = MCII.get(Opcode); 338 uint64_t TSFlags = Desc.TSFlags; 339 unsigned CurOp = X86II::getOperandBias(Desc); 340 int MemoryOperand = X86II::getMemoryOperandNo(TSFlags); 341 if (MemoryOperand < 0) 342 return false; 343 unsigned BaseRegNum = MemoryOperand + CurOp + X86::AddrBaseReg; 344 unsigned BaseReg = MI.getOperand(BaseRegNum).getReg(); 345 return (BaseReg == X86::RIP); 346 } 347 348 /// Check if the instruction is a prefix. 349 static bool isPrefix(const MCInst &MI, const MCInstrInfo &MCII) { 350 return X86II::isPrefix(MCII.get(MI.getOpcode()).TSFlags); 351 } 352 353 /// Check if the instruction is valid as the first instruction in macro fusion. 354 static bool isFirstMacroFusibleInst(const MCInst &Inst, 355 const MCInstrInfo &MCII) { 356 // An Intel instruction with RIP relative addressing is not macro fusible. 357 if (isRIPRelative(Inst, MCII)) 358 return false; 359 X86::FirstMacroFusionInstKind FIK = 360 X86::classifyFirstOpcodeInMacroFusion(Inst.getOpcode()); 361 return FIK != X86::FirstMacroFusionInstKind::Invalid; 362 } 363 364 /// X86 can reduce the bytes of NOP by padding instructions with prefixes to 365 /// get a better peformance in some cases. Here, we determine which prefix is 366 /// the most suitable. 367 /// 368 /// If the instruction has a segment override prefix, use the existing one. 369 /// If the target is 64-bit, use the CS. 370 /// If the target is 32-bit, 371 /// - If the instruction has a ESP/EBP base register, use SS. 372 /// - Otherwise use DS. 373 uint8_t X86AsmBackend::determinePaddingPrefix(const MCInst &Inst) const { 374 assert((STI.hasFeature(X86::Mode32Bit) || STI.hasFeature(X86::Mode64Bit)) && 375 "Prefixes can be added only in 32-bit or 64-bit mode."); 376 const MCInstrDesc &Desc = MCII->get(Inst.getOpcode()); 377 uint64_t TSFlags = Desc.TSFlags; 378 379 // Determine where the memory operand starts, if present. 380 int MemoryOperand = X86II::getMemoryOperandNo(TSFlags); 381 if (MemoryOperand != -1) 382 MemoryOperand += X86II::getOperandBias(Desc); 383 384 unsigned SegmentReg = 0; 385 if (MemoryOperand >= 0) { 386 // Check for explicit segment override on memory operand. 387 SegmentReg = Inst.getOperand(MemoryOperand + X86::AddrSegmentReg).getReg(); 388 } 389 390 switch (TSFlags & X86II::FormMask) { 391 default: 392 break; 393 case X86II::RawFrmDstSrc: { 394 // Check segment override opcode prefix as needed (not for %ds). 395 if (Inst.getOperand(2).getReg() != X86::DS) 396 SegmentReg = Inst.getOperand(2).getReg(); 397 break; 398 } 399 case X86II::RawFrmSrc: { 400 // Check segment override opcode prefix as needed (not for %ds). 401 if (Inst.getOperand(1).getReg() != X86::DS) 402 SegmentReg = Inst.getOperand(1).getReg(); 403 break; 404 } 405 case X86II::RawFrmMemOffs: { 406 // Check segment override opcode prefix as needed. 407 SegmentReg = Inst.getOperand(1).getReg(); 408 break; 409 } 410 } 411 412 if (SegmentReg != 0) 413 return X86::getSegmentOverridePrefixForReg(SegmentReg); 414 415 if (STI.hasFeature(X86::Mode64Bit)) 416 return X86::CS_Encoding; 417 418 if (MemoryOperand >= 0) { 419 unsigned BaseRegNum = MemoryOperand + X86::AddrBaseReg; 420 unsigned BaseReg = Inst.getOperand(BaseRegNum).getReg(); 421 if (BaseReg == X86::ESP || BaseReg == X86::EBP) 422 return X86::SS_Encoding; 423 } 424 return X86::DS_Encoding; 425 } 426 427 /// Check if the two instructions will be macro-fused on the target cpu. 428 bool X86AsmBackend::isMacroFused(const MCInst &Cmp, const MCInst &Jcc) const { 429 const MCInstrDesc &InstDesc = MCII->get(Jcc.getOpcode()); 430 if (!InstDesc.isConditionalBranch()) 431 return false; 432 if (!isFirstMacroFusibleInst(Cmp, *MCII)) 433 return false; 434 const X86::FirstMacroFusionInstKind CmpKind = 435 X86::classifyFirstOpcodeInMacroFusion(Cmp.getOpcode()); 436 const X86::SecondMacroFusionInstKind BranchKind = 437 classifySecondInstInMacroFusion(Jcc, *MCII); 438 return X86::isMacroFused(CmpKind, BranchKind); 439 } 440 441 /// Check if the instruction has a variant symbol operand. 442 static bool hasVariantSymbol(const MCInst &MI) { 443 for (auto &Operand : MI) { 444 if (!Operand.isExpr()) 445 continue; 446 const MCExpr &Expr = *Operand.getExpr(); 447 if (Expr.getKind() == MCExpr::SymbolRef && 448 cast<MCSymbolRefExpr>(Expr).getKind() != MCSymbolRefExpr::VK_None) 449 return true; 450 } 451 return false; 452 } 453 454 bool X86AsmBackend::allowAutoPadding() const { 455 return (AlignBoundary != Align(1) && AlignBranchType != X86::AlignBranchNone); 456 } 457 458 bool X86AsmBackend::needAlign(MCObjectStreamer &OS) const { 459 if (!OS.getAllowAutoPadding()) 460 return false; 461 assert(allowAutoPadding() && "incorrect initialization!"); 462 463 // To be Done: Currently don't deal with Bundle cases. 464 if (OS.getAssembler().isBundlingEnabled()) 465 return false; 466 467 // Branches only need to be aligned in 32-bit or 64-bit mode. 468 if (!(STI.hasFeature(X86::Mode64Bit) || STI.hasFeature(X86::Mode32Bit))) 469 return false; 470 471 return true; 472 } 473 474 /// X86 has certain instructions which enable interrupts exactly one 475 /// instruction *after* the instruction which stores to SS. Return true if the 476 /// given instruction has such an interrupt delay slot. 477 static bool hasInterruptDelaySlot(const MCInst &Inst) { 478 switch (Inst.getOpcode()) { 479 case X86::POPSS16: 480 case X86::POPSS32: 481 case X86::STI: 482 return true; 483 484 case X86::MOV16sr: 485 case X86::MOV32sr: 486 case X86::MOV64sr: 487 case X86::MOV16sm: 488 if (Inst.getOperand(0).getReg() == X86::SS) 489 return true; 490 break; 491 } 492 return false; 493 } 494 495 /// Check if the instruction to be emitted is right after any data. 496 static bool 497 isRightAfterData(MCFragment *CurrentFragment, 498 const std::pair<MCFragment *, size_t> &PrevInstPosition) { 499 MCFragment *F = CurrentFragment; 500 // Empty data fragments may be created to prevent further data being 501 // added into the previous fragment, we need to skip them since they 502 // have no contents. 503 for (; isa_and_nonnull<MCDataFragment>(F); F = F->getPrevNode()) 504 if (cast<MCDataFragment>(F)->getContents().size() != 0) 505 break; 506 507 // Since data is always emitted into a DataFragment, our check strategy is 508 // simple here. 509 // - If the fragment is a DataFragment 510 // - If it's not the fragment where the previous instruction is, 511 // returns true. 512 // - If it's the fragment holding the previous instruction but its 513 // size changed since the the previous instruction was emitted into 514 // it, returns true. 515 // - Otherwise returns false. 516 // - If the fragment is not a DataFragment, returns false. 517 if (auto *DF = dyn_cast_or_null<MCDataFragment>(F)) 518 return DF != PrevInstPosition.first || 519 DF->getContents().size() != PrevInstPosition.second; 520 521 return false; 522 } 523 524 /// \returns the fragment size if it has instructions, otherwise returns 0. 525 static size_t getSizeForInstFragment(const MCFragment *F) { 526 if (!F || !F->hasInstructions()) 527 return 0; 528 // MCEncodedFragmentWithContents being templated makes this tricky. 529 switch (F->getKind()) { 530 default: 531 llvm_unreachable("Unknown fragment with instructions!"); 532 case MCFragment::FT_Data: 533 return cast<MCDataFragment>(*F).getContents().size(); 534 case MCFragment::FT_Relaxable: 535 return cast<MCRelaxableFragment>(*F).getContents().size(); 536 case MCFragment::FT_CompactEncodedInst: 537 return cast<MCCompactEncodedInstFragment>(*F).getContents().size(); 538 } 539 } 540 541 /// Check if the instruction operand needs to be aligned. Padding is disabled 542 /// before intruction which may be rewritten by linker(e.g. TLSCALL). 543 bool X86AsmBackend::needAlignInst(const MCInst &Inst) const { 544 // Linker may rewrite the instruction with variant symbol operand. 545 if (hasVariantSymbol(Inst)) 546 return false; 547 548 const MCInstrDesc &InstDesc = MCII->get(Inst.getOpcode()); 549 return (InstDesc.isConditionalBranch() && 550 (AlignBranchType & X86::AlignBranchJcc)) || 551 (InstDesc.isUnconditionalBranch() && 552 (AlignBranchType & X86::AlignBranchJmp)) || 553 (InstDesc.isCall() && 554 (AlignBranchType & X86::AlignBranchCall)) || 555 (InstDesc.isReturn() && 556 (AlignBranchType & X86::AlignBranchRet)) || 557 (InstDesc.isIndirectBranch() && 558 (AlignBranchType & X86::AlignBranchIndirect)); 559 } 560 561 /// Insert BoundaryAlignFragment before instructions to align branches. 562 void X86AsmBackend::emitInstructionBegin(MCObjectStreamer &OS, 563 const MCInst &Inst) { 564 if (!needAlign(OS)) 565 return; 566 567 if (hasInterruptDelaySlot(PrevInst)) 568 // If this instruction follows an interrupt enabling instruction with a one 569 // instruction delay, inserting a nop would change behavior. 570 return; 571 572 if (isPrefix(PrevInst, *MCII)) 573 // If this instruction follows a prefix, inserting a nop would change 574 // semantic. 575 return; 576 577 if (isRightAfterData(OS.getCurrentFragment(), PrevInstPosition)) 578 // If this instruction follows any data, there is no clear 579 // instruction boundary, inserting a nop would change semantic. 580 return; 581 582 if (!isMacroFused(PrevInst, Inst)) 583 // Macro fusion doesn't happen indeed, clear the pending. 584 PendingBoundaryAlign = nullptr; 585 586 if (PendingBoundaryAlign && 587 OS.getCurrentFragment()->getPrevNode() == PendingBoundaryAlign) { 588 // Macro fusion actually happens and there is no other fragment inserted 589 // after the previous instruction. 590 // 591 // Do nothing here since we already inserted a BoudaryAlign fragment when 592 // we met the first instruction in the fused pair and we'll tie them 593 // together in emitInstructionEnd. 594 // 595 // Note: When there is at least one fragment, such as MCAlignFragment, 596 // inserted after the previous instruction, e.g. 597 // 598 // \code 599 // cmp %rax %rcx 600 // .align 16 601 // je .Label0 602 // \ endcode 603 // 604 // We will treat the JCC as a unfused branch although it may be fused 605 // with the CMP. 606 return; 607 } 608 609 if (needAlignInst(Inst) || ((AlignBranchType & X86::AlignBranchFused) && 610 isFirstMacroFusibleInst(Inst, *MCII))) { 611 // If we meet a unfused branch or the first instuction in a fusiable pair, 612 // insert a BoundaryAlign fragment. 613 OS.insert(PendingBoundaryAlign = 614 new MCBoundaryAlignFragment(AlignBoundary)); 615 } 616 } 617 618 /// Set the last fragment to be aligned for the BoundaryAlignFragment. 619 void X86AsmBackend::emitInstructionEnd(MCObjectStreamer &OS, const MCInst &Inst) { 620 if (!needAlign(OS)) 621 return; 622 623 PrevInst = Inst; 624 MCFragment *CF = OS.getCurrentFragment(); 625 PrevInstPosition = std::make_pair(CF, getSizeForInstFragment(CF)); 626 627 if (!needAlignInst(Inst) || !PendingBoundaryAlign) 628 return; 629 630 // Tie the aligned instructions into a a pending BoundaryAlign. 631 PendingBoundaryAlign->setLastFragment(CF); 632 PendingBoundaryAlign = nullptr; 633 634 // We need to ensure that further data isn't added to the current 635 // DataFragment, so that we can get the size of instructions later in 636 // MCAssembler::relaxBoundaryAlign. The easiest way is to insert a new empty 637 // DataFragment. 638 if (isa_and_nonnull<MCDataFragment>(CF)) 639 OS.insert(new MCDataFragment()); 640 641 // Update the maximum alignment on the current section if necessary. 642 MCSection *Sec = OS.getCurrentSectionOnly(); 643 if (AlignBoundary.value() > Sec->getAlignment()) 644 Sec->setAlignment(AlignBoundary); 645 } 646 647 Optional<MCFixupKind> X86AsmBackend::getFixupKind(StringRef Name) const { 648 if (STI.getTargetTriple().isOSBinFormatELF()) { 649 unsigned Type; 650 if (STI.getTargetTriple().getArch() == Triple::x86_64) { 651 Type = llvm::StringSwitch<unsigned>(Name) 652 #define ELF_RELOC(X, Y) .Case(#X, Y) 653 #include "llvm/BinaryFormat/ELFRelocs/x86_64.def" 654 #undef ELF_RELOC 655 .Default(-1u); 656 } else { 657 Type = llvm::StringSwitch<unsigned>(Name) 658 #define ELF_RELOC(X, Y) .Case(#X, Y) 659 #include "llvm/BinaryFormat/ELFRelocs/i386.def" 660 #undef ELF_RELOC 661 .Default(-1u); 662 } 663 if (Type == -1u) 664 return None; 665 return static_cast<MCFixupKind>(FirstLiteralRelocationKind + Type); 666 } 667 return MCAsmBackend::getFixupKind(Name); 668 } 669 670 const MCFixupKindInfo &X86AsmBackend::getFixupKindInfo(MCFixupKind Kind) const { 671 const static MCFixupKindInfo Infos[X86::NumTargetFixupKinds] = { 672 {"reloc_riprel_4byte", 0, 32, MCFixupKindInfo::FKF_IsPCRel}, 673 {"reloc_riprel_4byte_movq_load", 0, 32, MCFixupKindInfo::FKF_IsPCRel}, 674 {"reloc_riprel_4byte_relax", 0, 32, MCFixupKindInfo::FKF_IsPCRel}, 675 {"reloc_riprel_4byte_relax_rex", 0, 32, MCFixupKindInfo::FKF_IsPCRel}, 676 {"reloc_signed_4byte", 0, 32, 0}, 677 {"reloc_signed_4byte_relax", 0, 32, 0}, 678 {"reloc_global_offset_table", 0, 32, 0}, 679 {"reloc_global_offset_table8", 0, 64, 0}, 680 {"reloc_branch_4byte_pcrel", 0, 32, MCFixupKindInfo::FKF_IsPCRel}, 681 }; 682 683 // Fixup kinds from .reloc directive are like R_386_NONE/R_X86_64_NONE. They 684 // do not require any extra processing. 685 if (Kind >= FirstLiteralRelocationKind) 686 return MCAsmBackend::getFixupKindInfo(FK_NONE); 687 688 if (Kind < FirstTargetFixupKind) 689 return MCAsmBackend::getFixupKindInfo(Kind); 690 691 assert(unsigned(Kind - FirstTargetFixupKind) < getNumFixupKinds() && 692 "Invalid kind!"); 693 assert(Infos[Kind - FirstTargetFixupKind].Name && "Empty fixup name!"); 694 return Infos[Kind - FirstTargetFixupKind]; 695 } 696 697 bool X86AsmBackend::shouldForceRelocation(const MCAssembler &, 698 const MCFixup &Fixup, 699 const MCValue &) { 700 return Fixup.getKind() >= FirstLiteralRelocationKind; 701 } 702 703 static unsigned getFixupKindSize(unsigned Kind) { 704 switch (Kind) { 705 default: 706 llvm_unreachable("invalid fixup kind!"); 707 case FK_NONE: 708 return 0; 709 case FK_PCRel_1: 710 case FK_SecRel_1: 711 case FK_Data_1: 712 return 1; 713 case FK_PCRel_2: 714 case FK_SecRel_2: 715 case FK_Data_2: 716 return 2; 717 case FK_PCRel_4: 718 case X86::reloc_riprel_4byte: 719 case X86::reloc_riprel_4byte_relax: 720 case X86::reloc_riprel_4byte_relax_rex: 721 case X86::reloc_riprel_4byte_movq_load: 722 case X86::reloc_signed_4byte: 723 case X86::reloc_signed_4byte_relax: 724 case X86::reloc_global_offset_table: 725 case X86::reloc_branch_4byte_pcrel: 726 case FK_SecRel_4: 727 case FK_Data_4: 728 return 4; 729 case FK_PCRel_8: 730 case FK_SecRel_8: 731 case FK_Data_8: 732 case X86::reloc_global_offset_table8: 733 return 8; 734 } 735 } 736 737 void X86AsmBackend::applyFixup(const MCAssembler &Asm, const MCFixup &Fixup, 738 const MCValue &Target, 739 MutableArrayRef<char> Data, 740 uint64_t Value, bool IsResolved, 741 const MCSubtargetInfo *STI) const { 742 unsigned Kind = Fixup.getKind(); 743 if (Kind >= FirstLiteralRelocationKind) 744 return; 745 unsigned Size = getFixupKindSize(Kind); 746 747 assert(Fixup.getOffset() + Size <= Data.size() && "Invalid fixup offset!"); 748 749 int64_t SignedValue = static_cast<int64_t>(Value); 750 if ((Target.isAbsolute() || IsResolved) && 751 getFixupKindInfo(Fixup.getKind()).Flags & 752 MCFixupKindInfo::FKF_IsPCRel) { 753 // check that PC relative fixup fits into the fixup size. 754 if (Size > 0 && !isIntN(Size * 8, SignedValue)) 755 Asm.getContext().reportError( 756 Fixup.getLoc(), "value of " + Twine(SignedValue) + 757 " is too large for field of " + Twine(Size) + 758 ((Size == 1) ? " byte." : " bytes.")); 759 } else { 760 // Check that uppper bits are either all zeros or all ones. 761 // Specifically ignore overflow/underflow as long as the leakage is 762 // limited to the lower bits. This is to remain compatible with 763 // other assemblers. 764 assert((Size == 0 || isIntN(Size * 8 + 1, SignedValue)) && 765 "Value does not fit in the Fixup field"); 766 } 767 768 for (unsigned i = 0; i != Size; ++i) 769 Data[Fixup.getOffset() + i] = uint8_t(Value >> (i * 8)); 770 } 771 772 bool X86AsmBackend::mayNeedRelaxation(const MCInst &Inst, 773 const MCSubtargetInfo &STI) const { 774 // Branches can always be relaxed in either mode. 775 if (getRelaxedOpcodeBranch(Inst, false) != Inst.getOpcode()) 776 return true; 777 778 // Check if this instruction is ever relaxable. 779 if (getRelaxedOpcodeArith(Inst) == Inst.getOpcode()) 780 return false; 781 782 783 // Check if the relaxable operand has an expression. For the current set of 784 // relaxable instructions, the relaxable operand is always the last operand. 785 unsigned RelaxableOp = Inst.getNumOperands() - 1; 786 if (Inst.getOperand(RelaxableOp).isExpr()) 787 return true; 788 789 return false; 790 } 791 792 bool X86AsmBackend::fixupNeedsRelaxation(const MCFixup &Fixup, 793 uint64_t Value, 794 const MCRelaxableFragment *DF, 795 const MCAsmLayout &Layout) const { 796 // Relax if the value is too big for a (signed) i8. 797 return !isInt<8>(Value); 798 } 799 800 // FIXME: Can tblgen help at all here to verify there aren't other instructions 801 // we can relax? 802 void X86AsmBackend::relaxInstruction(const MCInst &Inst, 803 const MCSubtargetInfo &STI, 804 MCInst &Res) const { 805 // The only relaxations X86 does is from a 1byte pcrel to a 4byte pcrel. 806 bool Is16BitMode = STI.getFeatureBits()[X86::Mode16Bit]; 807 unsigned RelaxedOp = getRelaxedOpcode(Inst, Is16BitMode); 808 809 if (RelaxedOp == Inst.getOpcode()) { 810 SmallString<256> Tmp; 811 raw_svector_ostream OS(Tmp); 812 Inst.dump_pretty(OS); 813 OS << "\n"; 814 report_fatal_error("unexpected instruction to relax: " + OS.str()); 815 } 816 817 Res = Inst; 818 Res.setOpcode(RelaxedOp); 819 } 820 821 /// Return true if this instruction has been fully relaxed into it's most 822 /// general available form. 823 static bool isFullyRelaxed(const MCRelaxableFragment &RF) { 824 auto &Inst = RF.getInst(); 825 auto &STI = *RF.getSubtargetInfo(); 826 bool Is16BitMode = STI.getFeatureBits()[X86::Mode16Bit]; 827 return getRelaxedOpcode(Inst, Is16BitMode) == Inst.getOpcode(); 828 } 829 830 831 static bool shouldAddPrefix(const MCInst &Inst, const MCInstrInfo &MCII) { 832 // Linker may rewrite the instruction with variant symbol operand. 833 return !hasVariantSymbol(Inst); 834 } 835 836 static unsigned getRemainingPrefixSize(const MCInst &Inst, 837 const MCSubtargetInfo &STI, 838 MCCodeEmitter &Emitter) { 839 SmallString<256> Code; 840 raw_svector_ostream VecOS(Code); 841 Emitter.emitPrefix(Inst, VecOS, STI); 842 assert(Code.size() < 15 && "The number of prefixes must be less than 15."); 843 844 // TODO: It turns out we need a decent amount of plumbing for the target 845 // specific bits to determine number of prefixes its safe to add. Various 846 // targets (older chips mostly, but also Atom family) encounter decoder 847 // stalls with too many prefixes. For testing purposes, we set the value 848 // externally for the moment. 849 unsigned ExistingPrefixSize = Code.size(); 850 unsigned TargetPrefixMax = X86PadMaxPrefixSize; 851 if (TargetPrefixMax <= ExistingPrefixSize) 852 return 0; 853 return TargetPrefixMax - ExistingPrefixSize; 854 } 855 856 bool X86AsmBackend::padInstructionViaPrefix(MCRelaxableFragment &RF, 857 MCCodeEmitter &Emitter, 858 unsigned &RemainingSize) const { 859 if (!shouldAddPrefix(RF.getInst(), *MCII)) 860 return false; 861 // If the instruction isn't fully relaxed, shifting it around might require a 862 // larger value for one of the fixups then can be encoded. The outer loop 863 // will also catch this before moving to the next instruction, but we need to 864 // prevent padding this single instruction as well. 865 if (!isFullyRelaxed(RF)) 866 return false; 867 868 const unsigned OldSize = RF.getContents().size(); 869 if (OldSize == 15) 870 return false; 871 872 const unsigned MaxPossiblePad = std::min(15 - OldSize, RemainingSize); 873 const unsigned PrefixBytesToAdd = 874 std::min(MaxPossiblePad, 875 getRemainingPrefixSize(RF.getInst(), STI, Emitter)); 876 if (PrefixBytesToAdd == 0) 877 return false; 878 879 const uint8_t Prefix = determinePaddingPrefix(RF.getInst()); 880 881 SmallString<256> Code; 882 Code.append(PrefixBytesToAdd, Prefix); 883 Code.append(RF.getContents().begin(), RF.getContents().end()); 884 RF.getContents() = Code; 885 886 // Adjust the fixups for the change in offsets 887 for (auto &F : RF.getFixups()) { 888 F.setOffset(F.getOffset() + PrefixBytesToAdd); 889 } 890 891 RemainingSize -= PrefixBytesToAdd; 892 return true; 893 } 894 895 bool X86AsmBackend::padInstructionViaRelaxation(MCRelaxableFragment &RF, 896 MCCodeEmitter &Emitter, 897 unsigned &RemainingSize) const { 898 if (isFullyRelaxed(RF)) 899 // TODO: There are lots of other tricks we could apply for increasing 900 // encoding size without impacting performance. 901 return false; 902 903 MCInst Relaxed; 904 relaxInstruction(RF.getInst(), *RF.getSubtargetInfo(), Relaxed); 905 906 SmallVector<MCFixup, 4> Fixups; 907 SmallString<15> Code; 908 raw_svector_ostream VecOS(Code); 909 Emitter.encodeInstruction(Relaxed, VecOS, Fixups, *RF.getSubtargetInfo()); 910 const unsigned OldSize = RF.getContents().size(); 911 const unsigned NewSize = Code.size(); 912 assert(NewSize >= OldSize && "size decrease during relaxation?"); 913 unsigned Delta = NewSize - OldSize; 914 if (Delta > RemainingSize) 915 return false; 916 RF.setInst(Relaxed); 917 RF.getContents() = Code; 918 RF.getFixups() = Fixups; 919 RemainingSize -= Delta; 920 return true; 921 } 922 923 bool X86AsmBackend::padInstructionEncoding(MCRelaxableFragment &RF, 924 MCCodeEmitter &Emitter, 925 unsigned &RemainingSize) const { 926 bool Changed = false; 927 if (RemainingSize != 0) 928 Changed |= padInstructionViaRelaxation(RF, Emitter, RemainingSize); 929 if (RemainingSize != 0) 930 Changed |= padInstructionViaPrefix(RF, Emitter, RemainingSize); 931 return Changed; 932 } 933 934 void X86AsmBackend::finishLayout(MCAssembler const &Asm, 935 MCAsmLayout &Layout) const { 936 // See if we can further relax some instructions to cut down on the number of 937 // nop bytes required for code alignment. The actual win is in reducing 938 // instruction count, not number of bytes. Modern X86-64 can easily end up 939 // decode limited. It is often better to reduce the number of instructions 940 // (i.e. eliminate nops) even at the cost of increasing the size and 941 // complexity of others. 942 if (!X86PadForAlign && !X86PadForBranchAlign) 943 return; 944 945 DenseSet<MCFragment *> LabeledFragments; 946 for (const MCSymbol &S : Asm.symbols()) 947 LabeledFragments.insert(S.getFragment(false)); 948 949 for (MCSection &Sec : Asm) { 950 if (!Sec.getKind().isText()) 951 continue; 952 953 SmallVector<MCRelaxableFragment *, 4> Relaxable; 954 for (MCSection::iterator I = Sec.begin(), IE = Sec.end(); I != IE; ++I) { 955 MCFragment &F = *I; 956 957 if (LabeledFragments.count(&F)) 958 Relaxable.clear(); 959 960 if (F.getKind() == MCFragment::FT_Data || 961 F.getKind() == MCFragment::FT_CompactEncodedInst) 962 // Skip and ignore 963 continue; 964 965 if (F.getKind() == MCFragment::FT_Relaxable) { 966 auto &RF = cast<MCRelaxableFragment>(*I); 967 Relaxable.push_back(&RF); 968 continue; 969 } 970 971 auto canHandle = [](MCFragment &F) -> bool { 972 switch (F.getKind()) { 973 default: 974 return false; 975 case MCFragment::FT_Align: 976 return X86PadForAlign; 977 case MCFragment::FT_BoundaryAlign: 978 return X86PadForBranchAlign; 979 } 980 }; 981 // For any unhandled kind, assume we can't change layout. 982 if (!canHandle(F)) { 983 Relaxable.clear(); 984 continue; 985 } 986 987 #ifndef NDEBUG 988 const uint64_t OrigOffset = Layout.getFragmentOffset(&F); 989 #endif 990 const uint64_t OrigSize = Asm.computeFragmentSize(Layout, F); 991 992 // To keep the effects local, prefer to relax instructions closest to 993 // the align directive. This is purely about human understandability 994 // of the resulting code. If we later find a reason to expand 995 // particular instructions over others, we can adjust. 996 MCFragment *FirstChangedFragment = nullptr; 997 unsigned RemainingSize = OrigSize; 998 while (!Relaxable.empty() && RemainingSize != 0) { 999 auto &RF = *Relaxable.pop_back_val(); 1000 // Give the backend a chance to play any tricks it wishes to increase 1001 // the encoding size of the given instruction. Target independent code 1002 // will try further relaxation, but target's may play further tricks. 1003 if (padInstructionEncoding(RF, Asm.getEmitter(), RemainingSize)) 1004 FirstChangedFragment = &RF; 1005 1006 // If we have an instruction which hasn't been fully relaxed, we can't 1007 // skip past it and insert bytes before it. Changing its starting 1008 // offset might require a larger negative offset than it can encode. 1009 // We don't need to worry about larger positive offsets as none of the 1010 // possible offsets between this and our align are visible, and the 1011 // ones afterwards aren't changing. 1012 if (!isFullyRelaxed(RF)) 1013 break; 1014 } 1015 Relaxable.clear(); 1016 1017 if (FirstChangedFragment) { 1018 // Make sure the offsets for any fragments in the effected range get 1019 // updated. Note that this (conservatively) invalidates the offsets of 1020 // those following, but this is not required. 1021 Layout.invalidateFragmentsFrom(FirstChangedFragment); 1022 } 1023 1024 // BoundaryAlign explicitly tracks it's size (unlike align) 1025 if (F.getKind() == MCFragment::FT_BoundaryAlign) 1026 cast<MCBoundaryAlignFragment>(F).setSize(RemainingSize); 1027 1028 #ifndef NDEBUG 1029 const uint64_t FinalOffset = Layout.getFragmentOffset(&F); 1030 const uint64_t FinalSize = Asm.computeFragmentSize(Layout, F); 1031 assert(OrigOffset + OrigSize == FinalOffset + FinalSize && 1032 "can't move start of next fragment!"); 1033 assert(FinalSize == RemainingSize && "inconsistent size computation?"); 1034 #endif 1035 1036 // If we're looking at a boundary align, make sure we don't try to pad 1037 // its target instructions for some following directive. Doing so would 1038 // break the alignment of the current boundary align. 1039 if (auto *BF = dyn_cast<MCBoundaryAlignFragment>(&F)) { 1040 const MCFragment *LastFragment = BF->getLastFragment(); 1041 if (!LastFragment) 1042 continue; 1043 while (&*I != LastFragment) 1044 ++I; 1045 } 1046 } 1047 } 1048 1049 // The layout is done. Mark every fragment as valid. 1050 for (unsigned int i = 0, n = Layout.getSectionOrder().size(); i != n; ++i) { 1051 MCSection &Section = *Layout.getSectionOrder()[i]; 1052 Layout.getFragmentOffset(&*Section.getFragmentList().rbegin()); 1053 Asm.computeFragmentSize(Layout, *Section.getFragmentList().rbegin()); 1054 } 1055 } 1056 1057 /// Write a sequence of optimal nops to the output, covering \p Count 1058 /// bytes. 1059 /// \return - true on success, false on failure 1060 bool X86AsmBackend::writeNopData(raw_ostream &OS, uint64_t Count) const { 1061 static const char Nops[10][11] = { 1062 // nop 1063 "\x90", 1064 // xchg %ax,%ax 1065 "\x66\x90", 1066 // nopl (%[re]ax) 1067 "\x0f\x1f\x00", 1068 // nopl 0(%[re]ax) 1069 "\x0f\x1f\x40\x00", 1070 // nopl 0(%[re]ax,%[re]ax,1) 1071 "\x0f\x1f\x44\x00\x00", 1072 // nopw 0(%[re]ax,%[re]ax,1) 1073 "\x66\x0f\x1f\x44\x00\x00", 1074 // nopl 0L(%[re]ax) 1075 "\x0f\x1f\x80\x00\x00\x00\x00", 1076 // nopl 0L(%[re]ax,%[re]ax,1) 1077 "\x0f\x1f\x84\x00\x00\x00\x00\x00", 1078 // nopw 0L(%[re]ax,%[re]ax,1) 1079 "\x66\x0f\x1f\x84\x00\x00\x00\x00\x00", 1080 // nopw %cs:0L(%[re]ax,%[re]ax,1) 1081 "\x66\x2e\x0f\x1f\x84\x00\x00\x00\x00\x00", 1082 }; 1083 1084 // This CPU doesn't support long nops. If needed add more. 1085 // FIXME: We could generated something better than plain 0x90. 1086 if (!STI.getFeatureBits()[X86::FeatureNOPL]) { 1087 for (uint64_t i = 0; i < Count; ++i) 1088 OS << '\x90'; 1089 return true; 1090 } 1091 1092 // 15-bytes is the longest single NOP instruction, but 10-bytes is 1093 // commonly the longest that can be efficiently decoded. 1094 uint64_t MaxNopLength = 10; 1095 if (STI.getFeatureBits()[X86::FeatureFast7ByteNOP]) 1096 MaxNopLength = 7; 1097 else if (STI.getFeatureBits()[X86::FeatureFast15ByteNOP]) 1098 MaxNopLength = 15; 1099 else if (STI.getFeatureBits()[X86::FeatureFast11ByteNOP]) 1100 MaxNopLength = 11; 1101 1102 // Emit as many MaxNopLength NOPs as needed, then emit a NOP of the remaining 1103 // length. 1104 do { 1105 const uint8_t ThisNopLength = (uint8_t) std::min(Count, MaxNopLength); 1106 const uint8_t Prefixes = ThisNopLength <= 10 ? 0 : ThisNopLength - 10; 1107 for (uint8_t i = 0; i < Prefixes; i++) 1108 OS << '\x66'; 1109 const uint8_t Rest = ThisNopLength - Prefixes; 1110 if (Rest != 0) 1111 OS.write(Nops[Rest - 1], Rest); 1112 Count -= ThisNopLength; 1113 } while (Count != 0); 1114 1115 return true; 1116 } 1117 1118 /* *** */ 1119 1120 namespace { 1121 1122 class ELFX86AsmBackend : public X86AsmBackend { 1123 public: 1124 uint8_t OSABI; 1125 ELFX86AsmBackend(const Target &T, uint8_t OSABI, const MCSubtargetInfo &STI) 1126 : X86AsmBackend(T, STI), OSABI(OSABI) {} 1127 }; 1128 1129 class ELFX86_32AsmBackend : public ELFX86AsmBackend { 1130 public: 1131 ELFX86_32AsmBackend(const Target &T, uint8_t OSABI, 1132 const MCSubtargetInfo &STI) 1133 : ELFX86AsmBackend(T, OSABI, STI) {} 1134 1135 std::unique_ptr<MCObjectTargetWriter> 1136 createObjectTargetWriter() const override { 1137 return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI, ELF::EM_386); 1138 } 1139 }; 1140 1141 class ELFX86_X32AsmBackend : public ELFX86AsmBackend { 1142 public: 1143 ELFX86_X32AsmBackend(const Target &T, uint8_t OSABI, 1144 const MCSubtargetInfo &STI) 1145 : ELFX86AsmBackend(T, OSABI, STI) {} 1146 1147 std::unique_ptr<MCObjectTargetWriter> 1148 createObjectTargetWriter() const override { 1149 return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI, 1150 ELF::EM_X86_64); 1151 } 1152 }; 1153 1154 class ELFX86_IAMCUAsmBackend : public ELFX86AsmBackend { 1155 public: 1156 ELFX86_IAMCUAsmBackend(const Target &T, uint8_t OSABI, 1157 const MCSubtargetInfo &STI) 1158 : ELFX86AsmBackend(T, OSABI, STI) {} 1159 1160 std::unique_ptr<MCObjectTargetWriter> 1161 createObjectTargetWriter() const override { 1162 return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI, 1163 ELF::EM_IAMCU); 1164 } 1165 }; 1166 1167 class ELFX86_64AsmBackend : public ELFX86AsmBackend { 1168 public: 1169 ELFX86_64AsmBackend(const Target &T, uint8_t OSABI, 1170 const MCSubtargetInfo &STI) 1171 : ELFX86AsmBackend(T, OSABI, STI) {} 1172 1173 std::unique_ptr<MCObjectTargetWriter> 1174 createObjectTargetWriter() const override { 1175 return createX86ELFObjectWriter(/*IsELF64*/ true, OSABI, ELF::EM_X86_64); 1176 } 1177 }; 1178 1179 class WindowsX86AsmBackend : public X86AsmBackend { 1180 bool Is64Bit; 1181 1182 public: 1183 WindowsX86AsmBackend(const Target &T, bool is64Bit, 1184 const MCSubtargetInfo &STI) 1185 : X86AsmBackend(T, STI) 1186 , Is64Bit(is64Bit) { 1187 } 1188 1189 Optional<MCFixupKind> getFixupKind(StringRef Name) const override { 1190 return StringSwitch<Optional<MCFixupKind>>(Name) 1191 .Case("dir32", FK_Data_4) 1192 .Case("secrel32", FK_SecRel_4) 1193 .Case("secidx", FK_SecRel_2) 1194 .Default(MCAsmBackend::getFixupKind(Name)); 1195 } 1196 1197 std::unique_ptr<MCObjectTargetWriter> 1198 createObjectTargetWriter() const override { 1199 return createX86WinCOFFObjectWriter(Is64Bit); 1200 } 1201 }; 1202 1203 namespace CU { 1204 1205 /// Compact unwind encoding values. 1206 enum CompactUnwindEncodings { 1207 /// [RE]BP based frame where [RE]BP is pused on the stack immediately after 1208 /// the return address, then [RE]SP is moved to [RE]BP. 1209 UNWIND_MODE_BP_FRAME = 0x01000000, 1210 1211 /// A frameless function with a small constant stack size. 1212 UNWIND_MODE_STACK_IMMD = 0x02000000, 1213 1214 /// A frameless function with a large constant stack size. 1215 UNWIND_MODE_STACK_IND = 0x03000000, 1216 1217 /// No compact unwind encoding is available. 1218 UNWIND_MODE_DWARF = 0x04000000, 1219 1220 /// Mask for encoding the frame registers. 1221 UNWIND_BP_FRAME_REGISTERS = 0x00007FFF, 1222 1223 /// Mask for encoding the frameless registers. 1224 UNWIND_FRAMELESS_STACK_REG_PERMUTATION = 0x000003FF 1225 }; 1226 1227 } // end CU namespace 1228 1229 class DarwinX86AsmBackend : public X86AsmBackend { 1230 const MCRegisterInfo &MRI; 1231 1232 /// Number of registers that can be saved in a compact unwind encoding. 1233 enum { CU_NUM_SAVED_REGS = 6 }; 1234 1235 mutable unsigned SavedRegs[CU_NUM_SAVED_REGS]; 1236 Triple TT; 1237 bool Is64Bit; 1238 1239 unsigned OffsetSize; ///< Offset of a "push" instruction. 1240 unsigned MoveInstrSize; ///< Size of a "move" instruction. 1241 unsigned StackDivide; ///< Amount to adjust stack size by. 1242 protected: 1243 /// Size of a "push" instruction for the given register. 1244 unsigned PushInstrSize(unsigned Reg) const { 1245 switch (Reg) { 1246 case X86::EBX: 1247 case X86::ECX: 1248 case X86::EDX: 1249 case X86::EDI: 1250 case X86::ESI: 1251 case X86::EBP: 1252 case X86::RBX: 1253 case X86::RBP: 1254 return 1; 1255 case X86::R12: 1256 case X86::R13: 1257 case X86::R14: 1258 case X86::R15: 1259 return 2; 1260 } 1261 return 1; 1262 } 1263 1264 private: 1265 /// Get the compact unwind number for a given register. The number 1266 /// corresponds to the enum lists in compact_unwind_encoding.h. 1267 int getCompactUnwindRegNum(unsigned Reg) const { 1268 static const MCPhysReg CU32BitRegs[7] = { 1269 X86::EBX, X86::ECX, X86::EDX, X86::EDI, X86::ESI, X86::EBP, 0 1270 }; 1271 static const MCPhysReg CU64BitRegs[] = { 1272 X86::RBX, X86::R12, X86::R13, X86::R14, X86::R15, X86::RBP, 0 1273 }; 1274 const MCPhysReg *CURegs = Is64Bit ? CU64BitRegs : CU32BitRegs; 1275 for (int Idx = 1; *CURegs; ++CURegs, ++Idx) 1276 if (*CURegs == Reg) 1277 return Idx; 1278 1279 return -1; 1280 } 1281 1282 /// Return the registers encoded for a compact encoding with a frame 1283 /// pointer. 1284 uint32_t encodeCompactUnwindRegistersWithFrame() const { 1285 // Encode the registers in the order they were saved --- 3-bits per 1286 // register. The list of saved registers is assumed to be in reverse 1287 // order. The registers are numbered from 1 to CU_NUM_SAVED_REGS. 1288 uint32_t RegEnc = 0; 1289 for (int i = 0, Idx = 0; i != CU_NUM_SAVED_REGS; ++i) { 1290 unsigned Reg = SavedRegs[i]; 1291 if (Reg == 0) break; 1292 1293 int CURegNum = getCompactUnwindRegNum(Reg); 1294 if (CURegNum == -1) return ~0U; 1295 1296 // Encode the 3-bit register number in order, skipping over 3-bits for 1297 // each register. 1298 RegEnc |= (CURegNum & 0x7) << (Idx++ * 3); 1299 } 1300 1301 assert((RegEnc & 0x3FFFF) == RegEnc && 1302 "Invalid compact register encoding!"); 1303 return RegEnc; 1304 } 1305 1306 /// Create the permutation encoding used with frameless stacks. It is 1307 /// passed the number of registers to be saved and an array of the registers 1308 /// saved. 1309 uint32_t encodeCompactUnwindRegistersWithoutFrame(unsigned RegCount) const { 1310 // The saved registers are numbered from 1 to 6. In order to encode the 1311 // order in which they were saved, we re-number them according to their 1312 // place in the register order. The re-numbering is relative to the last 1313 // re-numbered register. E.g., if we have registers {6, 2, 4, 5} saved in 1314 // that order: 1315 // 1316 // Orig Re-Num 1317 // ---- ------ 1318 // 6 6 1319 // 2 2 1320 // 4 3 1321 // 5 3 1322 // 1323 for (unsigned i = 0; i < RegCount; ++i) { 1324 int CUReg = getCompactUnwindRegNum(SavedRegs[i]); 1325 if (CUReg == -1) return ~0U; 1326 SavedRegs[i] = CUReg; 1327 } 1328 1329 // Reverse the list. 1330 std::reverse(&SavedRegs[0], &SavedRegs[CU_NUM_SAVED_REGS]); 1331 1332 uint32_t RenumRegs[CU_NUM_SAVED_REGS]; 1333 for (unsigned i = CU_NUM_SAVED_REGS - RegCount; i < CU_NUM_SAVED_REGS; ++i){ 1334 unsigned Countless = 0; 1335 for (unsigned j = CU_NUM_SAVED_REGS - RegCount; j < i; ++j) 1336 if (SavedRegs[j] < SavedRegs[i]) 1337 ++Countless; 1338 1339 RenumRegs[i] = SavedRegs[i] - Countless - 1; 1340 } 1341 1342 // Take the renumbered values and encode them into a 10-bit number. 1343 uint32_t permutationEncoding = 0; 1344 switch (RegCount) { 1345 case 6: 1346 permutationEncoding |= 120 * RenumRegs[0] + 24 * RenumRegs[1] 1347 + 6 * RenumRegs[2] + 2 * RenumRegs[3] 1348 + RenumRegs[4]; 1349 break; 1350 case 5: 1351 permutationEncoding |= 120 * RenumRegs[1] + 24 * RenumRegs[2] 1352 + 6 * RenumRegs[3] + 2 * RenumRegs[4] 1353 + RenumRegs[5]; 1354 break; 1355 case 4: 1356 permutationEncoding |= 60 * RenumRegs[2] + 12 * RenumRegs[3] 1357 + 3 * RenumRegs[4] + RenumRegs[5]; 1358 break; 1359 case 3: 1360 permutationEncoding |= 20 * RenumRegs[3] + 4 * RenumRegs[4] 1361 + RenumRegs[5]; 1362 break; 1363 case 2: 1364 permutationEncoding |= 5 * RenumRegs[4] + RenumRegs[5]; 1365 break; 1366 case 1: 1367 permutationEncoding |= RenumRegs[5]; 1368 break; 1369 } 1370 1371 assert((permutationEncoding & 0x3FF) == permutationEncoding && 1372 "Invalid compact register encoding!"); 1373 return permutationEncoding; 1374 } 1375 1376 public: 1377 DarwinX86AsmBackend(const Target &T, const MCRegisterInfo &MRI, 1378 const MCSubtargetInfo &STI) 1379 : X86AsmBackend(T, STI), MRI(MRI), TT(STI.getTargetTriple()), 1380 Is64Bit(TT.isArch64Bit()) { 1381 memset(SavedRegs, 0, sizeof(SavedRegs)); 1382 OffsetSize = Is64Bit ? 8 : 4; 1383 MoveInstrSize = Is64Bit ? 3 : 2; 1384 StackDivide = Is64Bit ? 8 : 4; 1385 } 1386 1387 std::unique_ptr<MCObjectTargetWriter> 1388 createObjectTargetWriter() const override { 1389 uint32_t CPUType = cantFail(MachO::getCPUType(TT)); 1390 uint32_t CPUSubType = cantFail(MachO::getCPUSubType(TT)); 1391 return createX86MachObjectWriter(Is64Bit, CPUType, CPUSubType); 1392 } 1393 1394 /// Implementation of algorithm to generate the compact unwind encoding 1395 /// for the CFI instructions. 1396 uint32_t 1397 generateCompactUnwindEncoding(ArrayRef<MCCFIInstruction> Instrs) const override { 1398 if (Instrs.empty()) return 0; 1399 1400 // Reset the saved registers. 1401 unsigned SavedRegIdx = 0; 1402 memset(SavedRegs, 0, sizeof(SavedRegs)); 1403 1404 bool HasFP = false; 1405 1406 // Encode that we are using EBP/RBP as the frame pointer. 1407 uint32_t CompactUnwindEncoding = 0; 1408 1409 unsigned SubtractInstrIdx = Is64Bit ? 3 : 2; 1410 unsigned InstrOffset = 0; 1411 unsigned StackAdjust = 0; 1412 unsigned StackSize = 0; 1413 unsigned NumDefCFAOffsets = 0; 1414 1415 for (unsigned i = 0, e = Instrs.size(); i != e; ++i) { 1416 const MCCFIInstruction &Inst = Instrs[i]; 1417 1418 switch (Inst.getOperation()) { 1419 default: 1420 // Any other CFI directives indicate a frame that we aren't prepared 1421 // to represent via compact unwind, so just bail out. 1422 return 0; 1423 case MCCFIInstruction::OpDefCfaRegister: { 1424 // Defines a frame pointer. E.g. 1425 // 1426 // movq %rsp, %rbp 1427 // L0: 1428 // .cfi_def_cfa_register %rbp 1429 // 1430 HasFP = true; 1431 1432 // If the frame pointer is other than esp/rsp, we do not have a way to 1433 // generate a compact unwinding representation, so bail out. 1434 if (*MRI.getLLVMRegNum(Inst.getRegister(), true) != 1435 (Is64Bit ? X86::RBP : X86::EBP)) 1436 return 0; 1437 1438 // Reset the counts. 1439 memset(SavedRegs, 0, sizeof(SavedRegs)); 1440 StackAdjust = 0; 1441 SavedRegIdx = 0; 1442 InstrOffset += MoveInstrSize; 1443 break; 1444 } 1445 case MCCFIInstruction::OpDefCfaOffset: { 1446 // Defines a new offset for the CFA. E.g. 1447 // 1448 // With frame: 1449 // 1450 // pushq %rbp 1451 // L0: 1452 // .cfi_def_cfa_offset 16 1453 // 1454 // Without frame: 1455 // 1456 // subq $72, %rsp 1457 // L0: 1458 // .cfi_def_cfa_offset 80 1459 // 1460 StackSize = std::abs(Inst.getOffset()) / StackDivide; 1461 ++NumDefCFAOffsets; 1462 break; 1463 } 1464 case MCCFIInstruction::OpOffset: { 1465 // Defines a "push" of a callee-saved register. E.g. 1466 // 1467 // pushq %r15 1468 // pushq %r14 1469 // pushq %rbx 1470 // L0: 1471 // subq $120, %rsp 1472 // L1: 1473 // .cfi_offset %rbx, -40 1474 // .cfi_offset %r14, -32 1475 // .cfi_offset %r15, -24 1476 // 1477 if (SavedRegIdx == CU_NUM_SAVED_REGS) 1478 // If there are too many saved registers, we cannot use a compact 1479 // unwind encoding. 1480 return CU::UNWIND_MODE_DWARF; 1481 1482 unsigned Reg = *MRI.getLLVMRegNum(Inst.getRegister(), true); 1483 SavedRegs[SavedRegIdx++] = Reg; 1484 StackAdjust += OffsetSize; 1485 InstrOffset += PushInstrSize(Reg); 1486 break; 1487 } 1488 } 1489 } 1490 1491 StackAdjust /= StackDivide; 1492 1493 if (HasFP) { 1494 if ((StackAdjust & 0xFF) != StackAdjust) 1495 // Offset was too big for a compact unwind encoding. 1496 return CU::UNWIND_MODE_DWARF; 1497 1498 // Get the encoding of the saved registers when we have a frame pointer. 1499 uint32_t RegEnc = encodeCompactUnwindRegistersWithFrame(); 1500 if (RegEnc == ~0U) return CU::UNWIND_MODE_DWARF; 1501 1502 CompactUnwindEncoding |= CU::UNWIND_MODE_BP_FRAME; 1503 CompactUnwindEncoding |= (StackAdjust & 0xFF) << 16; 1504 CompactUnwindEncoding |= RegEnc & CU::UNWIND_BP_FRAME_REGISTERS; 1505 } else { 1506 SubtractInstrIdx += InstrOffset; 1507 ++StackAdjust; 1508 1509 if ((StackSize & 0xFF) == StackSize) { 1510 // Frameless stack with a small stack size. 1511 CompactUnwindEncoding |= CU::UNWIND_MODE_STACK_IMMD; 1512 1513 // Encode the stack size. 1514 CompactUnwindEncoding |= (StackSize & 0xFF) << 16; 1515 } else { 1516 if ((StackAdjust & 0x7) != StackAdjust) 1517 // The extra stack adjustments are too big for us to handle. 1518 return CU::UNWIND_MODE_DWARF; 1519 1520 // Frameless stack with an offset too large for us to encode compactly. 1521 CompactUnwindEncoding |= CU::UNWIND_MODE_STACK_IND; 1522 1523 // Encode the offset to the nnnnnn value in the 'subl $nnnnnn, ESP' 1524 // instruction. 1525 CompactUnwindEncoding |= (SubtractInstrIdx & 0xFF) << 16; 1526 1527 // Encode any extra stack adjustments (done via push instructions). 1528 CompactUnwindEncoding |= (StackAdjust & 0x7) << 13; 1529 } 1530 1531 // Encode the number of registers saved. (Reverse the list first.) 1532 std::reverse(&SavedRegs[0], &SavedRegs[SavedRegIdx]); 1533 CompactUnwindEncoding |= (SavedRegIdx & 0x7) << 10; 1534 1535 // Get the encoding of the saved registers when we don't have a frame 1536 // pointer. 1537 uint32_t RegEnc = encodeCompactUnwindRegistersWithoutFrame(SavedRegIdx); 1538 if (RegEnc == ~0U) return CU::UNWIND_MODE_DWARF; 1539 1540 // Encode the register encoding. 1541 CompactUnwindEncoding |= 1542 RegEnc & CU::UNWIND_FRAMELESS_STACK_REG_PERMUTATION; 1543 } 1544 1545 return CompactUnwindEncoding; 1546 } 1547 }; 1548 1549 } // end anonymous namespace 1550 1551 MCAsmBackend *llvm::createX86_32AsmBackend(const Target &T, 1552 const MCSubtargetInfo &STI, 1553 const MCRegisterInfo &MRI, 1554 const MCTargetOptions &Options) { 1555 const Triple &TheTriple = STI.getTargetTriple(); 1556 if (TheTriple.isOSBinFormatMachO()) 1557 return new DarwinX86AsmBackend(T, MRI, STI); 1558 1559 if (TheTriple.isOSWindows() && TheTriple.isOSBinFormatCOFF()) 1560 return new WindowsX86AsmBackend(T, false, STI); 1561 1562 uint8_t OSABI = MCELFObjectTargetWriter::getOSABI(TheTriple.getOS()); 1563 1564 if (TheTriple.isOSIAMCU()) 1565 return new ELFX86_IAMCUAsmBackend(T, OSABI, STI); 1566 1567 return new ELFX86_32AsmBackend(T, OSABI, STI); 1568 } 1569 1570 MCAsmBackend *llvm::createX86_64AsmBackend(const Target &T, 1571 const MCSubtargetInfo &STI, 1572 const MCRegisterInfo &MRI, 1573 const MCTargetOptions &Options) { 1574 const Triple &TheTriple = STI.getTargetTriple(); 1575 if (TheTriple.isOSBinFormatMachO()) 1576 return new DarwinX86AsmBackend(T, MRI, STI); 1577 1578 if (TheTriple.isOSWindows() && TheTriple.isOSBinFormatCOFF()) 1579 return new WindowsX86AsmBackend(T, true, STI); 1580 1581 uint8_t OSABI = MCELFObjectTargetWriter::getOSABI(TheTriple.getOS()); 1582 1583 if (TheTriple.getEnvironment() == Triple::GNUX32) 1584 return new ELFX86_X32AsmBackend(T, OSABI, STI); 1585 return new ELFX86_64AsmBackend(T, OSABI, STI); 1586 } 1587