1 //===-- X86AsmBackend.cpp - X86 Assembler Backend -------------------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 
9 #include "MCTargetDesc/X86BaseInfo.h"
10 #include "MCTargetDesc/X86FixupKinds.h"
11 #include "llvm/ADT/StringSwitch.h"
12 #include "llvm/BinaryFormat/ELF.h"
13 #include "llvm/BinaryFormat/MachO.h"
14 #include "llvm/MC/MCAsmBackend.h"
15 #include "llvm/MC/MCAsmLayout.h"
16 #include "llvm/MC/MCAssembler.h"
17 #include "llvm/MC/MCCodeEmitter.h"
18 #include "llvm/MC/MCContext.h"
19 #include "llvm/MC/MCDwarf.h"
20 #include "llvm/MC/MCELFObjectWriter.h"
21 #include "llvm/MC/MCExpr.h"
22 #include "llvm/MC/MCFixupKindInfo.h"
23 #include "llvm/MC/MCInst.h"
24 #include "llvm/MC/MCInstrInfo.h"
25 #include "llvm/MC/MCMachObjectWriter.h"
26 #include "llvm/MC/MCObjectStreamer.h"
27 #include "llvm/MC/MCObjectWriter.h"
28 #include "llvm/MC/MCRegisterInfo.h"
29 #include "llvm/MC/MCSectionMachO.h"
30 #include "llvm/MC/MCSubtargetInfo.h"
31 #include "llvm/MC/MCValue.h"
32 #include "llvm/Support/CommandLine.h"
33 #include "llvm/Support/ErrorHandling.h"
34 #include "llvm/Support/TargetRegistry.h"
35 #include "llvm/Support/raw_ostream.h"
36 
37 using namespace llvm;
38 
39 namespace {
40 /// A wrapper for holding a mask of the values from X86::AlignBranchBoundaryKind
41 class X86AlignBranchKind {
42 private:
43   uint8_t AlignBranchKind = 0;
44 
45 public:
46   void operator=(const std::string &Val) {
47     if (Val.empty())
48       return;
49     SmallVector<StringRef, 6> BranchTypes;
50     StringRef(Val).split(BranchTypes, '+', -1, false);
51     for (auto BranchType : BranchTypes) {
52       if (BranchType == "fused")
53         addKind(X86::AlignBranchFused);
54       else if (BranchType == "jcc")
55         addKind(X86::AlignBranchJcc);
56       else if (BranchType == "jmp")
57         addKind(X86::AlignBranchJmp);
58       else if (BranchType == "call")
59         addKind(X86::AlignBranchCall);
60       else if (BranchType == "ret")
61         addKind(X86::AlignBranchRet);
62       else if (BranchType == "indirect")
63         addKind(X86::AlignBranchIndirect);
64       else {
65         report_fatal_error(
66             "'-x86-align-branch 'The branches's type is combination of jcc, "
67             "fused, jmp, call, ret, indirect.(plus separated)",
68             false);
69       }
70     }
71   }
72 
73   operator uint8_t() const { return AlignBranchKind; }
74   void addKind(X86::AlignBranchBoundaryKind Value) { AlignBranchKind |= Value; }
75 };
76 
77 X86AlignBranchKind X86AlignBranchKindLoc;
78 
79 cl::opt<unsigned> X86AlignBranchBoundary(
80     "x86-align-branch-boundary", cl::init(0),
81     cl::desc(
82         "Control how the assembler should align branches with NOP. If the "
83         "boundary's size is not 0, it should be a power of 2 and no less "
84         "than 32. Branches will be aligned to prevent from being across or "
85         "against the boundary of specified size. The default value 0 does not "
86         "align branches."));
87 
88 cl::opt<X86AlignBranchKind, true, cl::parser<std::string>> X86AlignBranch(
89     "x86-align-branch",
90     cl::desc(
91         "Specify types of branches to align (plus separated list of types):"
92              "\njcc      indicates conditional jumps"
93              "\nfused    indicates fused conditional jumps"
94              "\njmp      indicates direct unconditional jumps"
95              "\ncall     indicates direct and indirect calls"
96              "\nret      indicates rets"
97              "\nindirect indicates indirect unconditional jumps"),
98     cl::location(X86AlignBranchKindLoc));
99 
100 cl::opt<bool> X86AlignBranchWithin32BBoundaries(
101     "x86-branches-within-32B-boundaries", cl::init(false),
102     cl::desc(
103         "Align selected instructions to mitigate negative performance impact "
104         "of Intel's micro code update for errata skx102.  May break "
105         "assumptions about labels corresponding to particular instructions, "
106         "and should be used with caution."));
107 
108 cl::opt<unsigned> X86PadMaxPrefixSize(
109     "x86-pad-max-prefix-size", cl::init(0),
110     cl::desc("Maximum number of prefixes to use for padding"));
111 
112 cl::opt<bool> X86PadForAlign(
113     "x86-pad-for-align", cl::init(true), cl::Hidden,
114     cl::desc("Pad previous instructions to implement align directives"));
115 
116 cl::opt<bool> X86PadForBranchAlign(
117     "x86-pad-for-branch-align", cl::init(true), cl::Hidden,
118     cl::desc("Pad previous instructions to implement branch alignment"));
119 
120 class X86ELFObjectWriter : public MCELFObjectTargetWriter {
121 public:
122   X86ELFObjectWriter(bool is64Bit, uint8_t OSABI, uint16_t EMachine,
123                      bool HasRelocationAddend, bool foobar)
124     : MCELFObjectTargetWriter(is64Bit, OSABI, EMachine, HasRelocationAddend) {}
125 };
126 
127 class X86AsmBackend : public MCAsmBackend {
128   const MCSubtargetInfo &STI;
129   std::unique_ptr<const MCInstrInfo> MCII;
130   X86AlignBranchKind AlignBranchType;
131   Align AlignBoundary;
132 
133   uint8_t determinePaddingPrefix(const MCInst &Inst) const;
134 
135   bool isMacroFused(const MCInst &Cmp, const MCInst &Jcc) const;
136 
137   bool needAlign(MCObjectStreamer &OS) const;
138   bool needAlignInst(const MCInst &Inst) const;
139   MCInst PrevInst;
140   MCBoundaryAlignFragment *PendingBoundaryAlign = nullptr;
141   std::pair<MCFragment *, size_t> PrevInstPosition;
142 
143 public:
144   X86AsmBackend(const Target &T, const MCSubtargetInfo &STI)
145       : MCAsmBackend(support::little), STI(STI),
146         MCII(T.createMCInstrInfo()) {
147     if (X86AlignBranchWithin32BBoundaries) {
148       // At the moment, this defaults to aligning fused branches, unconditional
149       // jumps, and (unfused) conditional jumps with nops.  Both the
150       // instructions aligned and the alignment method (nop vs prefix) may
151       // change in the future.
152       AlignBoundary = assumeAligned(32);;
153       AlignBranchType.addKind(X86::AlignBranchFused);
154       AlignBranchType.addKind(X86::AlignBranchJcc);
155       AlignBranchType.addKind(X86::AlignBranchJmp);
156     }
157     // Allow overriding defaults set by master flag
158     if (X86AlignBranchBoundary.getNumOccurrences())
159       AlignBoundary = assumeAligned(X86AlignBranchBoundary);
160     if (X86AlignBranch.getNumOccurrences())
161       AlignBranchType = X86AlignBranchKindLoc;
162   }
163 
164   bool allowAutoPadding() const override;
165   void emitInstructionBegin(MCObjectStreamer &OS, const MCInst &Inst) override;
166   void emitInstructionEnd(MCObjectStreamer &OS, const MCInst &Inst) override;
167 
168   unsigned getNumFixupKinds() const override {
169     return X86::NumTargetFixupKinds;
170   }
171 
172   Optional<MCFixupKind> getFixupKind(StringRef Name) const override;
173 
174   const MCFixupKindInfo &getFixupKindInfo(MCFixupKind Kind) const override;
175 
176   bool shouldForceRelocation(const MCAssembler &Asm, const MCFixup &Fixup,
177                              const MCValue &Target) override;
178 
179   void applyFixup(const MCAssembler &Asm, const MCFixup &Fixup,
180                   const MCValue &Target, MutableArrayRef<char> Data,
181                   uint64_t Value, bool IsResolved,
182                   const MCSubtargetInfo *STI) const override;
183 
184   bool mayNeedRelaxation(const MCInst &Inst,
185                          const MCSubtargetInfo &STI) const override;
186 
187   bool fixupNeedsRelaxation(const MCFixup &Fixup, uint64_t Value,
188                             const MCRelaxableFragment *DF,
189                             const MCAsmLayout &Layout) const override;
190 
191   void relaxInstruction(const MCInst &Inst, const MCSubtargetInfo &STI,
192                         MCInst &Res) const override;
193 
194   bool padInstructionViaRelaxation(MCRelaxableFragment &RF,
195                                    MCCodeEmitter &Emitter,
196                                    unsigned &RemainingSize) const;
197 
198   bool padInstructionViaPrefix(MCRelaxableFragment &RF, MCCodeEmitter &Emitter,
199                                unsigned &RemainingSize) const;
200 
201   bool padInstructionEncoding(MCRelaxableFragment &RF, MCCodeEmitter &Emitter,
202                               unsigned &RemainingSize) const;
203 
204   void finishLayout(MCAssembler const &Asm, MCAsmLayout &Layout) const override;
205 
206   bool writeNopData(raw_ostream &OS, uint64_t Count) const override;
207 };
208 } // end anonymous namespace
209 
210 static unsigned getRelaxedOpcodeBranch(const MCInst &Inst, bool Is16BitMode) {
211   unsigned Op = Inst.getOpcode();
212   switch (Op) {
213   default:
214     return Op;
215   case X86::JCC_1:
216     return (Is16BitMode) ? X86::JCC_2 : X86::JCC_4;
217   case X86::JMP_1:
218     return (Is16BitMode) ? X86::JMP_2 : X86::JMP_4;
219   }
220 }
221 
222 static unsigned getRelaxedOpcodeArith(const MCInst &Inst) {
223   unsigned Op = Inst.getOpcode();
224   switch (Op) {
225   default:
226     return Op;
227 
228     // IMUL
229   case X86::IMUL16rri8: return X86::IMUL16rri;
230   case X86::IMUL16rmi8: return X86::IMUL16rmi;
231   case X86::IMUL32rri8: return X86::IMUL32rri;
232   case X86::IMUL32rmi8: return X86::IMUL32rmi;
233   case X86::IMUL64rri8: return X86::IMUL64rri32;
234   case X86::IMUL64rmi8: return X86::IMUL64rmi32;
235 
236     // AND
237   case X86::AND16ri8: return X86::AND16ri;
238   case X86::AND16mi8: return X86::AND16mi;
239   case X86::AND32ri8: return X86::AND32ri;
240   case X86::AND32mi8: return X86::AND32mi;
241   case X86::AND64ri8: return X86::AND64ri32;
242   case X86::AND64mi8: return X86::AND64mi32;
243 
244     // OR
245   case X86::OR16ri8: return X86::OR16ri;
246   case X86::OR16mi8: return X86::OR16mi;
247   case X86::OR32ri8: return X86::OR32ri;
248   case X86::OR32mi8: return X86::OR32mi;
249   case X86::OR64ri8: return X86::OR64ri32;
250   case X86::OR64mi8: return X86::OR64mi32;
251 
252     // XOR
253   case X86::XOR16ri8: return X86::XOR16ri;
254   case X86::XOR16mi8: return X86::XOR16mi;
255   case X86::XOR32ri8: return X86::XOR32ri;
256   case X86::XOR32mi8: return X86::XOR32mi;
257   case X86::XOR64ri8: return X86::XOR64ri32;
258   case X86::XOR64mi8: return X86::XOR64mi32;
259 
260     // ADD
261   case X86::ADD16ri8: return X86::ADD16ri;
262   case X86::ADD16mi8: return X86::ADD16mi;
263   case X86::ADD32ri8: return X86::ADD32ri;
264   case X86::ADD32mi8: return X86::ADD32mi;
265   case X86::ADD64ri8: return X86::ADD64ri32;
266   case X86::ADD64mi8: return X86::ADD64mi32;
267 
268    // ADC
269   case X86::ADC16ri8: return X86::ADC16ri;
270   case X86::ADC16mi8: return X86::ADC16mi;
271   case X86::ADC32ri8: return X86::ADC32ri;
272   case X86::ADC32mi8: return X86::ADC32mi;
273   case X86::ADC64ri8: return X86::ADC64ri32;
274   case X86::ADC64mi8: return X86::ADC64mi32;
275 
276     // SUB
277   case X86::SUB16ri8: return X86::SUB16ri;
278   case X86::SUB16mi8: return X86::SUB16mi;
279   case X86::SUB32ri8: return X86::SUB32ri;
280   case X86::SUB32mi8: return X86::SUB32mi;
281   case X86::SUB64ri8: return X86::SUB64ri32;
282   case X86::SUB64mi8: return X86::SUB64mi32;
283 
284    // SBB
285   case X86::SBB16ri8: return X86::SBB16ri;
286   case X86::SBB16mi8: return X86::SBB16mi;
287   case X86::SBB32ri8: return X86::SBB32ri;
288   case X86::SBB32mi8: return X86::SBB32mi;
289   case X86::SBB64ri8: return X86::SBB64ri32;
290   case X86::SBB64mi8: return X86::SBB64mi32;
291 
292     // CMP
293   case X86::CMP16ri8: return X86::CMP16ri;
294   case X86::CMP16mi8: return X86::CMP16mi;
295   case X86::CMP32ri8: return X86::CMP32ri;
296   case X86::CMP32mi8: return X86::CMP32mi;
297   case X86::CMP64ri8: return X86::CMP64ri32;
298   case X86::CMP64mi8: return X86::CMP64mi32;
299 
300     // PUSH
301   case X86::PUSH32i8:  return X86::PUSHi32;
302   case X86::PUSH16i8:  return X86::PUSHi16;
303   case X86::PUSH64i8:  return X86::PUSH64i32;
304   }
305 }
306 
307 static unsigned getRelaxedOpcode(const MCInst &Inst, bool Is16BitMode) {
308   unsigned R = getRelaxedOpcodeArith(Inst);
309   if (R != Inst.getOpcode())
310     return R;
311   return getRelaxedOpcodeBranch(Inst, Is16BitMode);
312 }
313 
314 static X86::CondCode getCondFromBranch(const MCInst &MI,
315                                        const MCInstrInfo &MCII) {
316   unsigned Opcode = MI.getOpcode();
317   switch (Opcode) {
318   default:
319     return X86::COND_INVALID;
320   case X86::JCC_1: {
321     const MCInstrDesc &Desc = MCII.get(Opcode);
322     return static_cast<X86::CondCode>(
323         MI.getOperand(Desc.getNumOperands() - 1).getImm());
324   }
325   }
326 }
327 
328 static X86::SecondMacroFusionInstKind
329 classifySecondInstInMacroFusion(const MCInst &MI, const MCInstrInfo &MCII) {
330   X86::CondCode CC = getCondFromBranch(MI, MCII);
331   return classifySecondCondCodeInMacroFusion(CC);
332 }
333 
334 /// Check if the instruction uses RIP relative addressing.
335 static bool isRIPRelative(const MCInst &MI, const MCInstrInfo &MCII) {
336   unsigned Opcode = MI.getOpcode();
337   const MCInstrDesc &Desc = MCII.get(Opcode);
338   uint64_t TSFlags = Desc.TSFlags;
339   unsigned CurOp = X86II::getOperandBias(Desc);
340   int MemoryOperand = X86II::getMemoryOperandNo(TSFlags);
341   if (MemoryOperand < 0)
342     return false;
343   unsigned BaseRegNum = MemoryOperand + CurOp + X86::AddrBaseReg;
344   unsigned BaseReg = MI.getOperand(BaseRegNum).getReg();
345   return (BaseReg == X86::RIP);
346 }
347 
348 /// Check if the instruction is a prefix.
349 static bool isPrefix(const MCInst &MI, const MCInstrInfo &MCII) {
350   return X86II::isPrefix(MCII.get(MI.getOpcode()).TSFlags);
351 }
352 
353 /// Check if the instruction is valid as the first instruction in macro fusion.
354 static bool isFirstMacroFusibleInst(const MCInst &Inst,
355                                     const MCInstrInfo &MCII) {
356   // An Intel instruction with RIP relative addressing is not macro fusible.
357   if (isRIPRelative(Inst, MCII))
358     return false;
359   X86::FirstMacroFusionInstKind FIK =
360       X86::classifyFirstOpcodeInMacroFusion(Inst.getOpcode());
361   return FIK != X86::FirstMacroFusionInstKind::Invalid;
362 }
363 
364 /// X86 can reduce the bytes of NOP by padding instructions with prefixes to
365 /// get a better peformance in some cases. Here, we determine which prefix is
366 /// the most suitable.
367 ///
368 /// If the instruction has a segment override prefix, use the existing one.
369 /// If the target is 64-bit, use the CS.
370 /// If the target is 32-bit,
371 ///   - If the instruction has a ESP/EBP base register, use SS.
372 ///   - Otherwise use DS.
373 uint8_t X86AsmBackend::determinePaddingPrefix(const MCInst &Inst) const {
374   assert((STI.hasFeature(X86::Mode32Bit) || STI.hasFeature(X86::Mode64Bit)) &&
375          "Prefixes can be added only in 32-bit or 64-bit mode.");
376   const MCInstrDesc &Desc = MCII->get(Inst.getOpcode());
377   uint64_t TSFlags = Desc.TSFlags;
378 
379   // Determine where the memory operand starts, if present.
380   int MemoryOperand = X86II::getMemoryOperandNo(TSFlags);
381   if (MemoryOperand != -1)
382     MemoryOperand += X86II::getOperandBias(Desc);
383 
384   unsigned SegmentReg = 0;
385   if (MemoryOperand >= 0) {
386     // Check for explicit segment override on memory operand.
387     SegmentReg = Inst.getOperand(MemoryOperand + X86::AddrSegmentReg).getReg();
388   }
389 
390   switch (TSFlags & X86II::FormMask) {
391   default:
392     break;
393   case X86II::RawFrmDstSrc: {
394     // Check segment override opcode prefix as needed (not for %ds).
395     if (Inst.getOperand(2).getReg() != X86::DS)
396       SegmentReg = Inst.getOperand(2).getReg();
397     break;
398   }
399   case X86II::RawFrmSrc: {
400     // Check segment override opcode prefix as needed (not for %ds).
401     if (Inst.getOperand(1).getReg() != X86::DS)
402       SegmentReg = Inst.getOperand(1).getReg();
403     break;
404   }
405   case X86II::RawFrmMemOffs: {
406     // Check segment override opcode prefix as needed.
407     SegmentReg = Inst.getOperand(1).getReg();
408     break;
409   }
410   }
411 
412   if (SegmentReg != 0)
413     return X86::getSegmentOverridePrefixForReg(SegmentReg);
414 
415   if (STI.hasFeature(X86::Mode64Bit))
416     return X86::CS_Encoding;
417 
418   if (MemoryOperand >= 0) {
419     unsigned BaseRegNum = MemoryOperand + X86::AddrBaseReg;
420     unsigned BaseReg = Inst.getOperand(BaseRegNum).getReg();
421     if (BaseReg == X86::ESP || BaseReg == X86::EBP)
422       return X86::SS_Encoding;
423   }
424   return X86::DS_Encoding;
425 }
426 
427 /// Check if the two instructions will be macro-fused on the target cpu.
428 bool X86AsmBackend::isMacroFused(const MCInst &Cmp, const MCInst &Jcc) const {
429   const MCInstrDesc &InstDesc = MCII->get(Jcc.getOpcode());
430   if (!InstDesc.isConditionalBranch())
431     return false;
432   if (!isFirstMacroFusibleInst(Cmp, *MCII))
433     return false;
434   const X86::FirstMacroFusionInstKind CmpKind =
435       X86::classifyFirstOpcodeInMacroFusion(Cmp.getOpcode());
436   const X86::SecondMacroFusionInstKind BranchKind =
437       classifySecondInstInMacroFusion(Jcc, *MCII);
438   return X86::isMacroFused(CmpKind, BranchKind);
439 }
440 
441 /// Check if the instruction has a variant symbol operand.
442 static bool hasVariantSymbol(const MCInst &MI) {
443   for (auto &Operand : MI) {
444     if (!Operand.isExpr())
445       continue;
446     const MCExpr &Expr = *Operand.getExpr();
447     if (Expr.getKind() == MCExpr::SymbolRef &&
448         cast<MCSymbolRefExpr>(Expr).getKind() != MCSymbolRefExpr::VK_None)
449       return true;
450   }
451   return false;
452 }
453 
454 bool X86AsmBackend::allowAutoPadding() const {
455   return (AlignBoundary != Align(1) && AlignBranchType != X86::AlignBranchNone);
456 }
457 
458 bool X86AsmBackend::needAlign(MCObjectStreamer &OS) const {
459   if (!OS.getAllowAutoPadding())
460     return false;
461   assert(allowAutoPadding() && "incorrect initialization!");
462 
463   // To be Done: Currently don't deal with Bundle cases.
464   if (OS.getAssembler().isBundlingEnabled())
465     return false;
466 
467   // Branches only need to be aligned in 32-bit or 64-bit mode.
468   if (!(STI.hasFeature(X86::Mode64Bit) || STI.hasFeature(X86::Mode32Bit)))
469     return false;
470 
471   return true;
472 }
473 
474 /// X86 has certain instructions which enable interrupts exactly one
475 /// instruction *after* the instruction which stores to SS.  Return true if the
476 /// given instruction has such an interrupt delay slot.
477 static bool hasInterruptDelaySlot(const MCInst &Inst) {
478   switch (Inst.getOpcode()) {
479   case X86::POPSS16:
480   case X86::POPSS32:
481   case X86::STI:
482     return true;
483 
484   case X86::MOV16sr:
485   case X86::MOV32sr:
486   case X86::MOV64sr:
487   case X86::MOV16sm:
488     if (Inst.getOperand(0).getReg() == X86::SS)
489       return true;
490     break;
491   }
492   return false;
493 }
494 
495 /// Check if the instruction to be emitted is right after any data.
496 static bool
497 isRightAfterData(MCFragment *CurrentFragment,
498                  const std::pair<MCFragment *, size_t> &PrevInstPosition) {
499   MCFragment *F = CurrentFragment;
500   // Empty data fragments may be created to prevent further data being
501   // added into the previous fragment, we need to skip them since they
502   // have no contents.
503   for (; isa_and_nonnull<MCDataFragment>(F); F = F->getPrevNode())
504     if (cast<MCDataFragment>(F)->getContents().size() != 0)
505       break;
506 
507   // Since data is always emitted into a DataFragment, our check strategy is
508   // simple here.
509   //   - If the fragment is a DataFragment
510   //     - If it's not the fragment where the previous instruction is,
511   //       returns true.
512   //     - If it's the fragment holding the previous instruction but its
513   //       size changed since the the previous instruction was emitted into
514   //       it, returns true.
515   //     - Otherwise returns false.
516   //   - If the fragment is not a DataFragment, returns false.
517   if (auto *DF = dyn_cast_or_null<MCDataFragment>(F))
518     return DF != PrevInstPosition.first ||
519            DF->getContents().size() != PrevInstPosition.second;
520 
521   return false;
522 }
523 
524 /// \returns the fragment size if it has instructions, otherwise returns 0.
525 static size_t getSizeForInstFragment(const MCFragment *F) {
526   if (!F || !F->hasInstructions())
527     return 0;
528   // MCEncodedFragmentWithContents being templated makes this tricky.
529   switch (F->getKind()) {
530   default:
531     llvm_unreachable("Unknown fragment with instructions!");
532   case MCFragment::FT_Data:
533     return cast<MCDataFragment>(*F).getContents().size();
534   case MCFragment::FT_Relaxable:
535     return cast<MCRelaxableFragment>(*F).getContents().size();
536   case MCFragment::FT_CompactEncodedInst:
537     return cast<MCCompactEncodedInstFragment>(*F).getContents().size();
538   }
539 }
540 
541 /// Check if the instruction operand needs to be aligned. Padding is disabled
542 /// before intruction which may be rewritten by linker(e.g. TLSCALL).
543 bool X86AsmBackend::needAlignInst(const MCInst &Inst) const {
544   // Linker may rewrite the instruction with variant symbol operand.
545   if (hasVariantSymbol(Inst))
546     return false;
547 
548   const MCInstrDesc &InstDesc = MCII->get(Inst.getOpcode());
549   return (InstDesc.isConditionalBranch() &&
550           (AlignBranchType & X86::AlignBranchJcc)) ||
551          (InstDesc.isUnconditionalBranch() &&
552           (AlignBranchType & X86::AlignBranchJmp)) ||
553          (InstDesc.isCall() &&
554           (AlignBranchType & X86::AlignBranchCall)) ||
555          (InstDesc.isReturn() &&
556           (AlignBranchType & X86::AlignBranchRet)) ||
557          (InstDesc.isIndirectBranch() &&
558           (AlignBranchType & X86::AlignBranchIndirect));
559 }
560 
561 /// Insert BoundaryAlignFragment before instructions to align branches.
562 void X86AsmBackend::emitInstructionBegin(MCObjectStreamer &OS,
563                                        const MCInst &Inst) {
564   if (!needAlign(OS))
565     return;
566 
567   if (hasInterruptDelaySlot(PrevInst))
568     // If this instruction follows an interrupt enabling instruction with a one
569     // instruction delay, inserting a nop would change behavior.
570     return;
571 
572   if (isPrefix(PrevInst, *MCII))
573     // If this instruction follows a prefix, inserting a nop would change
574     // semantic.
575     return;
576 
577   if (isRightAfterData(OS.getCurrentFragment(), PrevInstPosition))
578     // If this instruction follows any data, there is no clear
579     // instruction boundary, inserting a nop would change semantic.
580     return;
581 
582   if (!isMacroFused(PrevInst, Inst))
583     // Macro fusion doesn't happen indeed, clear the pending.
584     PendingBoundaryAlign = nullptr;
585 
586   if (PendingBoundaryAlign &&
587       OS.getCurrentFragment()->getPrevNode() == PendingBoundaryAlign) {
588     // Macro fusion actually happens and there is no other fragment inserted
589     // after the previous instruction.
590     //
591     // Do nothing here since we already inserted a BoudaryAlign fragment when
592     // we met the first instruction in the fused pair and we'll tie them
593     // together in emitInstructionEnd.
594     //
595     // Note: When there is at least one fragment, such as MCAlignFragment,
596     // inserted after the previous instruction, e.g.
597     //
598     // \code
599     //   cmp %rax %rcx
600     //   .align 16
601     //   je .Label0
602     // \ endcode
603     //
604     // We will treat the JCC as a unfused branch although it may be fused
605     // with the CMP.
606     return;
607   }
608 
609   if (needAlignInst(Inst) || ((AlignBranchType & X86::AlignBranchFused) &&
610                               isFirstMacroFusibleInst(Inst, *MCII))) {
611     // If we meet a unfused branch or the first instuction in a fusiable pair,
612     // insert a BoundaryAlign fragment.
613     OS.insert(PendingBoundaryAlign =
614                   new MCBoundaryAlignFragment(AlignBoundary));
615   }
616 }
617 
618 /// Set the last fragment to be aligned for the BoundaryAlignFragment.
619 void X86AsmBackend::emitInstructionEnd(MCObjectStreamer &OS, const MCInst &Inst) {
620   if (!needAlign(OS))
621     return;
622 
623   PrevInst = Inst;
624   MCFragment *CF = OS.getCurrentFragment();
625   PrevInstPosition = std::make_pair(CF, getSizeForInstFragment(CF));
626 
627   if (!needAlignInst(Inst) || !PendingBoundaryAlign)
628     return;
629 
630   // Tie the aligned instructions into a a pending BoundaryAlign.
631   PendingBoundaryAlign->setLastFragment(CF);
632   PendingBoundaryAlign = nullptr;
633 
634   // We need to ensure that further data isn't added to the current
635   // DataFragment, so that we can get the size of instructions later in
636   // MCAssembler::relaxBoundaryAlign. The easiest way is to insert a new empty
637   // DataFragment.
638   if (isa_and_nonnull<MCDataFragment>(CF))
639     OS.insert(new MCDataFragment());
640 
641   // Update the maximum alignment on the current section if necessary.
642   MCSection *Sec = OS.getCurrentSectionOnly();
643   if (AlignBoundary.value() > Sec->getAlignment())
644     Sec->setAlignment(AlignBoundary);
645 }
646 
647 Optional<MCFixupKind> X86AsmBackend::getFixupKind(StringRef Name) const {
648   if (STI.getTargetTriple().isOSBinFormatELF()) {
649     unsigned Type;
650     if (STI.getTargetTriple().getArch() == Triple::x86_64) {
651       Type = llvm::StringSwitch<unsigned>(Name)
652 #define ELF_RELOC(X, Y) .Case(#X, Y)
653 #include "llvm/BinaryFormat/ELFRelocs/x86_64.def"
654 #undef ELF_RELOC
655                  .Default(-1u);
656     } else {
657       Type = llvm::StringSwitch<unsigned>(Name)
658 #define ELF_RELOC(X, Y) .Case(#X, Y)
659 #include "llvm/BinaryFormat/ELFRelocs/i386.def"
660 #undef ELF_RELOC
661                  .Default(-1u);
662     }
663     if (Type == -1u)
664       return None;
665     return static_cast<MCFixupKind>(FirstLiteralRelocationKind + Type);
666   }
667   return MCAsmBackend::getFixupKind(Name);
668 }
669 
670 const MCFixupKindInfo &X86AsmBackend::getFixupKindInfo(MCFixupKind Kind) const {
671   const static MCFixupKindInfo Infos[X86::NumTargetFixupKinds] = {
672       {"reloc_riprel_4byte", 0, 32, MCFixupKindInfo::FKF_IsPCRel},
673       {"reloc_riprel_4byte_movq_load", 0, 32, MCFixupKindInfo::FKF_IsPCRel},
674       {"reloc_riprel_4byte_relax", 0, 32, MCFixupKindInfo::FKF_IsPCRel},
675       {"reloc_riprel_4byte_relax_rex", 0, 32, MCFixupKindInfo::FKF_IsPCRel},
676       {"reloc_signed_4byte", 0, 32, 0},
677       {"reloc_signed_4byte_relax", 0, 32, 0},
678       {"reloc_global_offset_table", 0, 32, 0},
679       {"reloc_global_offset_table8", 0, 64, 0},
680       {"reloc_branch_4byte_pcrel", 0, 32, MCFixupKindInfo::FKF_IsPCRel},
681   };
682 
683   // Fixup kinds from .reloc directive are like R_386_NONE/R_X86_64_NONE. They
684   // do not require any extra processing.
685   if (Kind >= FirstLiteralRelocationKind)
686     return MCAsmBackend::getFixupKindInfo(FK_NONE);
687 
688   if (Kind < FirstTargetFixupKind)
689     return MCAsmBackend::getFixupKindInfo(Kind);
690 
691   assert(unsigned(Kind - FirstTargetFixupKind) < getNumFixupKinds() &&
692          "Invalid kind!");
693   assert(Infos[Kind - FirstTargetFixupKind].Name && "Empty fixup name!");
694   return Infos[Kind - FirstTargetFixupKind];
695 }
696 
697 bool X86AsmBackend::shouldForceRelocation(const MCAssembler &,
698                                           const MCFixup &Fixup,
699                                           const MCValue &) {
700   return Fixup.getKind() >= FirstLiteralRelocationKind;
701 }
702 
703 static unsigned getFixupKindSize(unsigned Kind) {
704   switch (Kind) {
705   default:
706     llvm_unreachable("invalid fixup kind!");
707   case FK_NONE:
708     return 0;
709   case FK_PCRel_1:
710   case FK_SecRel_1:
711   case FK_Data_1:
712     return 1;
713   case FK_PCRel_2:
714   case FK_SecRel_2:
715   case FK_Data_2:
716     return 2;
717   case FK_PCRel_4:
718   case X86::reloc_riprel_4byte:
719   case X86::reloc_riprel_4byte_relax:
720   case X86::reloc_riprel_4byte_relax_rex:
721   case X86::reloc_riprel_4byte_movq_load:
722   case X86::reloc_signed_4byte:
723   case X86::reloc_signed_4byte_relax:
724   case X86::reloc_global_offset_table:
725   case X86::reloc_branch_4byte_pcrel:
726   case FK_SecRel_4:
727   case FK_Data_4:
728     return 4;
729   case FK_PCRel_8:
730   case FK_SecRel_8:
731   case FK_Data_8:
732   case X86::reloc_global_offset_table8:
733     return 8;
734   }
735 }
736 
737 void X86AsmBackend::applyFixup(const MCAssembler &Asm, const MCFixup &Fixup,
738                                const MCValue &Target,
739                                MutableArrayRef<char> Data,
740                                uint64_t Value, bool IsResolved,
741                                const MCSubtargetInfo *STI) const {
742   unsigned Kind = Fixup.getKind();
743   if (Kind >= FirstLiteralRelocationKind)
744     return;
745   unsigned Size = getFixupKindSize(Kind);
746 
747   assert(Fixup.getOffset() + Size <= Data.size() && "Invalid fixup offset!");
748 
749   int64_t SignedValue = static_cast<int64_t>(Value);
750   if ((Target.isAbsolute() || IsResolved) &&
751       getFixupKindInfo(Fixup.getKind()).Flags &
752       MCFixupKindInfo::FKF_IsPCRel) {
753     // check that PC relative fixup fits into the fixup size.
754     if (Size > 0 && !isIntN(Size * 8, SignedValue))
755       Asm.getContext().reportError(
756                                    Fixup.getLoc(), "value of " + Twine(SignedValue) +
757                                    " is too large for field of " + Twine(Size) +
758                                    ((Size == 1) ? " byte." : " bytes."));
759   } else {
760     // Check that uppper bits are either all zeros or all ones.
761     // Specifically ignore overflow/underflow as long as the leakage is
762     // limited to the lower bits. This is to remain compatible with
763     // other assemblers.
764     assert((Size == 0 || isIntN(Size * 8 + 1, SignedValue)) &&
765            "Value does not fit in the Fixup field");
766   }
767 
768   for (unsigned i = 0; i != Size; ++i)
769     Data[Fixup.getOffset() + i] = uint8_t(Value >> (i * 8));
770 }
771 
772 bool X86AsmBackend::mayNeedRelaxation(const MCInst &Inst,
773                                       const MCSubtargetInfo &STI) const {
774   // Branches can always be relaxed in either mode.
775   if (getRelaxedOpcodeBranch(Inst, false) != Inst.getOpcode())
776     return true;
777 
778   // Check if this instruction is ever relaxable.
779   if (getRelaxedOpcodeArith(Inst) == Inst.getOpcode())
780     return false;
781 
782 
783   // Check if the relaxable operand has an expression. For the current set of
784   // relaxable instructions, the relaxable operand is always the last operand.
785   unsigned RelaxableOp = Inst.getNumOperands() - 1;
786   if (Inst.getOperand(RelaxableOp).isExpr())
787     return true;
788 
789   return false;
790 }
791 
792 bool X86AsmBackend::fixupNeedsRelaxation(const MCFixup &Fixup,
793                                          uint64_t Value,
794                                          const MCRelaxableFragment *DF,
795                                          const MCAsmLayout &Layout) const {
796   // Relax if the value is too big for a (signed) i8.
797   return !isInt<8>(Value);
798 }
799 
800 // FIXME: Can tblgen help at all here to verify there aren't other instructions
801 // we can relax?
802 void X86AsmBackend::relaxInstruction(const MCInst &Inst,
803                                      const MCSubtargetInfo &STI,
804                                      MCInst &Res) const {
805   // The only relaxations X86 does is from a 1byte pcrel to a 4byte pcrel.
806   bool Is16BitMode = STI.getFeatureBits()[X86::Mode16Bit];
807   unsigned RelaxedOp = getRelaxedOpcode(Inst, Is16BitMode);
808 
809   if (RelaxedOp == Inst.getOpcode()) {
810     SmallString<256> Tmp;
811     raw_svector_ostream OS(Tmp);
812     Inst.dump_pretty(OS);
813     OS << "\n";
814     report_fatal_error("unexpected instruction to relax: " + OS.str());
815   }
816 
817   Res = Inst;
818   Res.setOpcode(RelaxedOp);
819 }
820 
821 /// Return true if this instruction has been fully relaxed into it's most
822 /// general available form.
823 static bool isFullyRelaxed(const MCRelaxableFragment &RF) {
824   auto &Inst = RF.getInst();
825   auto &STI = *RF.getSubtargetInfo();
826   bool Is16BitMode = STI.getFeatureBits()[X86::Mode16Bit];
827   return getRelaxedOpcode(Inst, Is16BitMode) == Inst.getOpcode();
828 }
829 
830 
831 static bool shouldAddPrefix(const MCInst &Inst, const MCInstrInfo &MCII) {
832   // Linker may rewrite the instruction with variant symbol operand.
833   return !hasVariantSymbol(Inst);
834 }
835 
836 static unsigned getRemainingPrefixSize(const MCInst &Inst,
837                                        const MCSubtargetInfo &STI,
838                                        MCCodeEmitter &Emitter) {
839   SmallString<256> Code;
840   raw_svector_ostream VecOS(Code);
841   Emitter.emitPrefix(Inst, VecOS, STI);
842   assert(Code.size() < 15 && "The number of prefixes must be less than 15.");
843 
844   // TODO: It turns out we need a decent amount of plumbing for the target
845   // specific bits to determine number of prefixes its safe to add.  Various
846   // targets (older chips mostly, but also Atom family) encounter decoder
847   // stalls with too many prefixes.  For testing purposes, we set the value
848   // externally for the moment.
849   unsigned ExistingPrefixSize = Code.size();
850   unsigned TargetPrefixMax = X86PadMaxPrefixSize;
851   if (TargetPrefixMax <= ExistingPrefixSize)
852     return 0;
853   return TargetPrefixMax - ExistingPrefixSize;
854 }
855 
856 bool X86AsmBackend::padInstructionViaPrefix(MCRelaxableFragment &RF,
857                                             MCCodeEmitter &Emitter,
858                                             unsigned &RemainingSize) const {
859   if (!shouldAddPrefix(RF.getInst(), *MCII))
860     return false;
861   // If the instruction isn't fully relaxed, shifting it around might require a
862   // larger value for one of the fixups then can be encoded.  The outer loop
863   // will also catch this before moving to the next instruction, but we need to
864   // prevent padding this single instruction as well.
865   if (!isFullyRelaxed(RF))
866     return false;
867 
868   const unsigned OldSize = RF.getContents().size();
869   if (OldSize == 15)
870     return false;
871 
872   const unsigned MaxPossiblePad = std::min(15 - OldSize, RemainingSize);
873   const unsigned PrefixBytesToAdd =
874     std::min(MaxPossiblePad,
875              getRemainingPrefixSize(RF.getInst(), STI, Emitter));
876   if (PrefixBytesToAdd == 0)
877     return false;
878 
879   const uint8_t Prefix = determinePaddingPrefix(RF.getInst());
880 
881   SmallString<256> Code;
882   Code.append(PrefixBytesToAdd, Prefix);
883   Code.append(RF.getContents().begin(), RF.getContents().end());
884   RF.getContents() = Code;
885 
886   // Adjust the fixups for the change in offsets
887   for (auto &F : RF.getFixups()) {
888     F.setOffset(F.getOffset() + PrefixBytesToAdd);
889   }
890 
891   RemainingSize -= PrefixBytesToAdd;
892   return true;
893 }
894 
895 bool X86AsmBackend::padInstructionViaRelaxation(MCRelaxableFragment &RF,
896                                                 MCCodeEmitter &Emitter,
897                                                 unsigned &RemainingSize) const {
898   if (isFullyRelaxed(RF))
899     // TODO: There are lots of other tricks we could apply for increasing
900     // encoding size without impacting performance.
901     return false;
902 
903   MCInst Relaxed;
904   relaxInstruction(RF.getInst(), *RF.getSubtargetInfo(), Relaxed);
905 
906   SmallVector<MCFixup, 4> Fixups;
907   SmallString<15> Code;
908   raw_svector_ostream VecOS(Code);
909   Emitter.encodeInstruction(Relaxed, VecOS, Fixups, *RF.getSubtargetInfo());
910   const unsigned OldSize = RF.getContents().size();
911   const unsigned NewSize = Code.size();
912   assert(NewSize >= OldSize && "size decrease during relaxation?");
913   unsigned Delta = NewSize - OldSize;
914   if (Delta > RemainingSize)
915     return false;
916   RF.setInst(Relaxed);
917   RF.getContents() = Code;
918   RF.getFixups() = Fixups;
919   RemainingSize -= Delta;
920   return true;
921 }
922 
923 bool X86AsmBackend::padInstructionEncoding(MCRelaxableFragment &RF,
924                                            MCCodeEmitter &Emitter,
925                                            unsigned &RemainingSize) const {
926   bool Changed = false;
927   if (RemainingSize != 0)
928     Changed |= padInstructionViaRelaxation(RF, Emitter, RemainingSize);
929   if (RemainingSize != 0)
930     Changed |= padInstructionViaPrefix(RF, Emitter, RemainingSize);
931   return Changed;
932 }
933 
934 void X86AsmBackend::finishLayout(MCAssembler const &Asm,
935                                  MCAsmLayout &Layout) const {
936   // See if we can further relax some instructions to cut down on the number of
937   // nop bytes required for code alignment.  The actual win is in reducing
938   // instruction count, not number of bytes.  Modern X86-64 can easily end up
939   // decode limited.  It is often better to reduce the number of instructions
940   // (i.e. eliminate nops) even at the cost of increasing the size and
941   // complexity of others.
942   if (!X86PadForAlign && !X86PadForBranchAlign)
943     return;
944 
945   DenseSet<MCFragment *> LabeledFragments;
946   for (const MCSymbol &S : Asm.symbols())
947     LabeledFragments.insert(S.getFragment(false));
948 
949   for (MCSection &Sec : Asm) {
950     if (!Sec.getKind().isText())
951       continue;
952 
953     SmallVector<MCRelaxableFragment *, 4> Relaxable;
954     for (MCSection::iterator I = Sec.begin(), IE = Sec.end(); I != IE; ++I) {
955       MCFragment &F = *I;
956 
957       if (LabeledFragments.count(&F))
958         Relaxable.clear();
959 
960       if (F.getKind() == MCFragment::FT_Data ||
961           F.getKind() == MCFragment::FT_CompactEncodedInst)
962         // Skip and ignore
963         continue;
964 
965       if (F.getKind() == MCFragment::FT_Relaxable) {
966         auto &RF = cast<MCRelaxableFragment>(*I);
967         Relaxable.push_back(&RF);
968         continue;
969       }
970 
971       auto canHandle = [](MCFragment &F) -> bool {
972         switch (F.getKind()) {
973         default:
974           return false;
975         case MCFragment::FT_Align:
976           return X86PadForAlign;
977         case MCFragment::FT_BoundaryAlign:
978           return X86PadForBranchAlign;
979         }
980       };
981       // For any unhandled kind, assume we can't change layout.
982       if (!canHandle(F)) {
983         Relaxable.clear();
984         continue;
985       }
986 
987 #ifndef NDEBUG
988       const uint64_t OrigOffset = Layout.getFragmentOffset(&F);
989 #endif
990       const uint64_t OrigSize = Asm.computeFragmentSize(Layout, F);
991 
992       // To keep the effects local, prefer to relax instructions closest to
993       // the align directive.  This is purely about human understandability
994       // of the resulting code.  If we later find a reason to expand
995       // particular instructions over others, we can adjust.
996       MCFragment *FirstChangedFragment = nullptr;
997       unsigned RemainingSize = OrigSize;
998       while (!Relaxable.empty() && RemainingSize != 0) {
999         auto &RF = *Relaxable.pop_back_val();
1000         // Give the backend a chance to play any tricks it wishes to increase
1001         // the encoding size of the given instruction.  Target independent code
1002         // will try further relaxation, but target's may play further tricks.
1003         if (padInstructionEncoding(RF, Asm.getEmitter(), RemainingSize))
1004           FirstChangedFragment = &RF;
1005 
1006         // If we have an instruction which hasn't been fully relaxed, we can't
1007         // skip past it and insert bytes before it.  Changing its starting
1008         // offset might require a larger negative offset than it can encode.
1009         // We don't need to worry about larger positive offsets as none of the
1010         // possible offsets between this and our align are visible, and the
1011         // ones afterwards aren't changing.
1012         if (!isFullyRelaxed(RF))
1013           break;
1014       }
1015       Relaxable.clear();
1016 
1017       if (FirstChangedFragment) {
1018         // Make sure the offsets for any fragments in the effected range get
1019         // updated.  Note that this (conservatively) invalidates the offsets of
1020         // those following, but this is not required.
1021         Layout.invalidateFragmentsFrom(FirstChangedFragment);
1022       }
1023 
1024       // BoundaryAlign explicitly tracks it's size (unlike align)
1025       if (F.getKind() == MCFragment::FT_BoundaryAlign)
1026         cast<MCBoundaryAlignFragment>(F).setSize(RemainingSize);
1027 
1028 #ifndef NDEBUG
1029       const uint64_t FinalOffset = Layout.getFragmentOffset(&F);
1030       const uint64_t FinalSize = Asm.computeFragmentSize(Layout, F);
1031       assert(OrigOffset + OrigSize == FinalOffset + FinalSize &&
1032              "can't move start of next fragment!");
1033       assert(FinalSize == RemainingSize && "inconsistent size computation?");
1034 #endif
1035 
1036       // If we're looking at a boundary align, make sure we don't try to pad
1037       // its target instructions for some following directive.  Doing so would
1038       // break the alignment of the current boundary align.
1039       if (auto *BF = dyn_cast<MCBoundaryAlignFragment>(&F)) {
1040         const MCFragment *LastFragment = BF->getLastFragment();
1041         if (!LastFragment)
1042           continue;
1043         while (&*I != LastFragment)
1044           ++I;
1045       }
1046     }
1047   }
1048 
1049   // The layout is done. Mark every fragment as valid.
1050   for (unsigned int i = 0, n = Layout.getSectionOrder().size(); i != n; ++i) {
1051     MCSection &Section = *Layout.getSectionOrder()[i];
1052     Layout.getFragmentOffset(&*Section.getFragmentList().rbegin());
1053     Asm.computeFragmentSize(Layout, *Section.getFragmentList().rbegin());
1054   }
1055 }
1056 
1057 /// Write a sequence of optimal nops to the output, covering \p Count
1058 /// bytes.
1059 /// \return - true on success, false on failure
1060 bool X86AsmBackend::writeNopData(raw_ostream &OS, uint64_t Count) const {
1061   static const char Nops[10][11] = {
1062     // nop
1063     "\x90",
1064     // xchg %ax,%ax
1065     "\x66\x90",
1066     // nopl (%[re]ax)
1067     "\x0f\x1f\x00",
1068     // nopl 0(%[re]ax)
1069     "\x0f\x1f\x40\x00",
1070     // nopl 0(%[re]ax,%[re]ax,1)
1071     "\x0f\x1f\x44\x00\x00",
1072     // nopw 0(%[re]ax,%[re]ax,1)
1073     "\x66\x0f\x1f\x44\x00\x00",
1074     // nopl 0L(%[re]ax)
1075     "\x0f\x1f\x80\x00\x00\x00\x00",
1076     // nopl 0L(%[re]ax,%[re]ax,1)
1077     "\x0f\x1f\x84\x00\x00\x00\x00\x00",
1078     // nopw 0L(%[re]ax,%[re]ax,1)
1079     "\x66\x0f\x1f\x84\x00\x00\x00\x00\x00",
1080     // nopw %cs:0L(%[re]ax,%[re]ax,1)
1081     "\x66\x2e\x0f\x1f\x84\x00\x00\x00\x00\x00",
1082   };
1083 
1084   // This CPU doesn't support long nops. If needed add more.
1085   // FIXME: We could generated something better than plain 0x90.
1086   if (!STI.getFeatureBits()[X86::FeatureNOPL]) {
1087     for (uint64_t i = 0; i < Count; ++i)
1088       OS << '\x90';
1089     return true;
1090   }
1091 
1092   // 15-bytes is the longest single NOP instruction, but 10-bytes is
1093   // commonly the longest that can be efficiently decoded.
1094   uint64_t MaxNopLength = 10;
1095   if (STI.getFeatureBits()[X86::FeatureFast7ByteNOP])
1096     MaxNopLength = 7;
1097   else if (STI.getFeatureBits()[X86::FeatureFast15ByteNOP])
1098     MaxNopLength = 15;
1099   else if (STI.getFeatureBits()[X86::FeatureFast11ByteNOP])
1100     MaxNopLength = 11;
1101 
1102   // Emit as many MaxNopLength NOPs as needed, then emit a NOP of the remaining
1103   // length.
1104   do {
1105     const uint8_t ThisNopLength = (uint8_t) std::min(Count, MaxNopLength);
1106     const uint8_t Prefixes = ThisNopLength <= 10 ? 0 : ThisNopLength - 10;
1107     for (uint8_t i = 0; i < Prefixes; i++)
1108       OS << '\x66';
1109     const uint8_t Rest = ThisNopLength - Prefixes;
1110     if (Rest != 0)
1111       OS.write(Nops[Rest - 1], Rest);
1112     Count -= ThisNopLength;
1113   } while (Count != 0);
1114 
1115   return true;
1116 }
1117 
1118 /* *** */
1119 
1120 namespace {
1121 
1122 class ELFX86AsmBackend : public X86AsmBackend {
1123 public:
1124   uint8_t OSABI;
1125   ELFX86AsmBackend(const Target &T, uint8_t OSABI, const MCSubtargetInfo &STI)
1126       : X86AsmBackend(T, STI), OSABI(OSABI) {}
1127 };
1128 
1129 class ELFX86_32AsmBackend : public ELFX86AsmBackend {
1130 public:
1131   ELFX86_32AsmBackend(const Target &T, uint8_t OSABI,
1132                       const MCSubtargetInfo &STI)
1133     : ELFX86AsmBackend(T, OSABI, STI) {}
1134 
1135   std::unique_ptr<MCObjectTargetWriter>
1136   createObjectTargetWriter() const override {
1137     return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI, ELF::EM_386);
1138   }
1139 };
1140 
1141 class ELFX86_X32AsmBackend : public ELFX86AsmBackend {
1142 public:
1143   ELFX86_X32AsmBackend(const Target &T, uint8_t OSABI,
1144                        const MCSubtargetInfo &STI)
1145       : ELFX86AsmBackend(T, OSABI, STI) {}
1146 
1147   std::unique_ptr<MCObjectTargetWriter>
1148   createObjectTargetWriter() const override {
1149     return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI,
1150                                     ELF::EM_X86_64);
1151   }
1152 };
1153 
1154 class ELFX86_IAMCUAsmBackend : public ELFX86AsmBackend {
1155 public:
1156   ELFX86_IAMCUAsmBackend(const Target &T, uint8_t OSABI,
1157                          const MCSubtargetInfo &STI)
1158       : ELFX86AsmBackend(T, OSABI, STI) {}
1159 
1160   std::unique_ptr<MCObjectTargetWriter>
1161   createObjectTargetWriter() const override {
1162     return createX86ELFObjectWriter(/*IsELF64*/ false, OSABI,
1163                                     ELF::EM_IAMCU);
1164   }
1165 };
1166 
1167 class ELFX86_64AsmBackend : public ELFX86AsmBackend {
1168 public:
1169   ELFX86_64AsmBackend(const Target &T, uint8_t OSABI,
1170                       const MCSubtargetInfo &STI)
1171     : ELFX86AsmBackend(T, OSABI, STI) {}
1172 
1173   std::unique_ptr<MCObjectTargetWriter>
1174   createObjectTargetWriter() const override {
1175     return createX86ELFObjectWriter(/*IsELF64*/ true, OSABI, ELF::EM_X86_64);
1176   }
1177 };
1178 
1179 class WindowsX86AsmBackend : public X86AsmBackend {
1180   bool Is64Bit;
1181 
1182 public:
1183   WindowsX86AsmBackend(const Target &T, bool is64Bit,
1184                        const MCSubtargetInfo &STI)
1185     : X86AsmBackend(T, STI)
1186     , Is64Bit(is64Bit) {
1187   }
1188 
1189   Optional<MCFixupKind> getFixupKind(StringRef Name) const override {
1190     return StringSwitch<Optional<MCFixupKind>>(Name)
1191         .Case("dir32", FK_Data_4)
1192         .Case("secrel32", FK_SecRel_4)
1193         .Case("secidx", FK_SecRel_2)
1194         .Default(MCAsmBackend::getFixupKind(Name));
1195   }
1196 
1197   std::unique_ptr<MCObjectTargetWriter>
1198   createObjectTargetWriter() const override {
1199     return createX86WinCOFFObjectWriter(Is64Bit);
1200   }
1201 };
1202 
1203 namespace CU {
1204 
1205   /// Compact unwind encoding values.
1206   enum CompactUnwindEncodings {
1207     /// [RE]BP based frame where [RE]BP is pused on the stack immediately after
1208     /// the return address, then [RE]SP is moved to [RE]BP.
1209     UNWIND_MODE_BP_FRAME                   = 0x01000000,
1210 
1211     /// A frameless function with a small constant stack size.
1212     UNWIND_MODE_STACK_IMMD                 = 0x02000000,
1213 
1214     /// A frameless function with a large constant stack size.
1215     UNWIND_MODE_STACK_IND                  = 0x03000000,
1216 
1217     /// No compact unwind encoding is available.
1218     UNWIND_MODE_DWARF                      = 0x04000000,
1219 
1220     /// Mask for encoding the frame registers.
1221     UNWIND_BP_FRAME_REGISTERS              = 0x00007FFF,
1222 
1223     /// Mask for encoding the frameless registers.
1224     UNWIND_FRAMELESS_STACK_REG_PERMUTATION = 0x000003FF
1225   };
1226 
1227 } // end CU namespace
1228 
1229 class DarwinX86AsmBackend : public X86AsmBackend {
1230   const MCRegisterInfo &MRI;
1231 
1232   /// Number of registers that can be saved in a compact unwind encoding.
1233   enum { CU_NUM_SAVED_REGS = 6 };
1234 
1235   mutable unsigned SavedRegs[CU_NUM_SAVED_REGS];
1236   Triple TT;
1237   bool Is64Bit;
1238 
1239   unsigned OffsetSize;                   ///< Offset of a "push" instruction.
1240   unsigned MoveInstrSize;                ///< Size of a "move" instruction.
1241   unsigned StackDivide;                  ///< Amount to adjust stack size by.
1242 protected:
1243   /// Size of a "push" instruction for the given register.
1244   unsigned PushInstrSize(unsigned Reg) const {
1245     switch (Reg) {
1246       case X86::EBX:
1247       case X86::ECX:
1248       case X86::EDX:
1249       case X86::EDI:
1250       case X86::ESI:
1251       case X86::EBP:
1252       case X86::RBX:
1253       case X86::RBP:
1254         return 1;
1255       case X86::R12:
1256       case X86::R13:
1257       case X86::R14:
1258       case X86::R15:
1259         return 2;
1260     }
1261     return 1;
1262   }
1263 
1264 private:
1265   /// Get the compact unwind number for a given register. The number
1266   /// corresponds to the enum lists in compact_unwind_encoding.h.
1267   int getCompactUnwindRegNum(unsigned Reg) const {
1268     static const MCPhysReg CU32BitRegs[7] = {
1269       X86::EBX, X86::ECX, X86::EDX, X86::EDI, X86::ESI, X86::EBP, 0
1270     };
1271     static const MCPhysReg CU64BitRegs[] = {
1272       X86::RBX, X86::R12, X86::R13, X86::R14, X86::R15, X86::RBP, 0
1273     };
1274     const MCPhysReg *CURegs = Is64Bit ? CU64BitRegs : CU32BitRegs;
1275     for (int Idx = 1; *CURegs; ++CURegs, ++Idx)
1276       if (*CURegs == Reg)
1277         return Idx;
1278 
1279     return -1;
1280   }
1281 
1282   /// Return the registers encoded for a compact encoding with a frame
1283   /// pointer.
1284   uint32_t encodeCompactUnwindRegistersWithFrame() const {
1285     // Encode the registers in the order they were saved --- 3-bits per
1286     // register. The list of saved registers is assumed to be in reverse
1287     // order. The registers are numbered from 1 to CU_NUM_SAVED_REGS.
1288     uint32_t RegEnc = 0;
1289     for (int i = 0, Idx = 0; i != CU_NUM_SAVED_REGS; ++i) {
1290       unsigned Reg = SavedRegs[i];
1291       if (Reg == 0) break;
1292 
1293       int CURegNum = getCompactUnwindRegNum(Reg);
1294       if (CURegNum == -1) return ~0U;
1295 
1296       // Encode the 3-bit register number in order, skipping over 3-bits for
1297       // each register.
1298       RegEnc |= (CURegNum & 0x7) << (Idx++ * 3);
1299     }
1300 
1301     assert((RegEnc & 0x3FFFF) == RegEnc &&
1302            "Invalid compact register encoding!");
1303     return RegEnc;
1304   }
1305 
1306   /// Create the permutation encoding used with frameless stacks. It is
1307   /// passed the number of registers to be saved and an array of the registers
1308   /// saved.
1309   uint32_t encodeCompactUnwindRegistersWithoutFrame(unsigned RegCount) const {
1310     // The saved registers are numbered from 1 to 6. In order to encode the
1311     // order in which they were saved, we re-number them according to their
1312     // place in the register order. The re-numbering is relative to the last
1313     // re-numbered register. E.g., if we have registers {6, 2, 4, 5} saved in
1314     // that order:
1315     //
1316     //    Orig  Re-Num
1317     //    ----  ------
1318     //     6       6
1319     //     2       2
1320     //     4       3
1321     //     5       3
1322     //
1323     for (unsigned i = 0; i < RegCount; ++i) {
1324       int CUReg = getCompactUnwindRegNum(SavedRegs[i]);
1325       if (CUReg == -1) return ~0U;
1326       SavedRegs[i] = CUReg;
1327     }
1328 
1329     // Reverse the list.
1330     std::reverse(&SavedRegs[0], &SavedRegs[CU_NUM_SAVED_REGS]);
1331 
1332     uint32_t RenumRegs[CU_NUM_SAVED_REGS];
1333     for (unsigned i = CU_NUM_SAVED_REGS - RegCount; i < CU_NUM_SAVED_REGS; ++i){
1334       unsigned Countless = 0;
1335       for (unsigned j = CU_NUM_SAVED_REGS - RegCount; j < i; ++j)
1336         if (SavedRegs[j] < SavedRegs[i])
1337           ++Countless;
1338 
1339       RenumRegs[i] = SavedRegs[i] - Countless - 1;
1340     }
1341 
1342     // Take the renumbered values and encode them into a 10-bit number.
1343     uint32_t permutationEncoding = 0;
1344     switch (RegCount) {
1345     case 6:
1346       permutationEncoding |= 120 * RenumRegs[0] + 24 * RenumRegs[1]
1347                              + 6 * RenumRegs[2] +  2 * RenumRegs[3]
1348                              +     RenumRegs[4];
1349       break;
1350     case 5:
1351       permutationEncoding |= 120 * RenumRegs[1] + 24 * RenumRegs[2]
1352                              + 6 * RenumRegs[3] +  2 * RenumRegs[4]
1353                              +     RenumRegs[5];
1354       break;
1355     case 4:
1356       permutationEncoding |=  60 * RenumRegs[2] + 12 * RenumRegs[3]
1357                              + 3 * RenumRegs[4] +      RenumRegs[5];
1358       break;
1359     case 3:
1360       permutationEncoding |=  20 * RenumRegs[3] +  4 * RenumRegs[4]
1361                              +     RenumRegs[5];
1362       break;
1363     case 2:
1364       permutationEncoding |=   5 * RenumRegs[4] +      RenumRegs[5];
1365       break;
1366     case 1:
1367       permutationEncoding |=       RenumRegs[5];
1368       break;
1369     }
1370 
1371     assert((permutationEncoding & 0x3FF) == permutationEncoding &&
1372            "Invalid compact register encoding!");
1373     return permutationEncoding;
1374   }
1375 
1376 public:
1377   DarwinX86AsmBackend(const Target &T, const MCRegisterInfo &MRI,
1378                       const MCSubtargetInfo &STI)
1379       : X86AsmBackend(T, STI), MRI(MRI), TT(STI.getTargetTriple()),
1380         Is64Bit(TT.isArch64Bit()) {
1381     memset(SavedRegs, 0, sizeof(SavedRegs));
1382     OffsetSize = Is64Bit ? 8 : 4;
1383     MoveInstrSize = Is64Bit ? 3 : 2;
1384     StackDivide = Is64Bit ? 8 : 4;
1385   }
1386 
1387   std::unique_ptr<MCObjectTargetWriter>
1388   createObjectTargetWriter() const override {
1389     uint32_t CPUType = cantFail(MachO::getCPUType(TT));
1390     uint32_t CPUSubType = cantFail(MachO::getCPUSubType(TT));
1391     return createX86MachObjectWriter(Is64Bit, CPUType, CPUSubType);
1392   }
1393 
1394   /// Implementation of algorithm to generate the compact unwind encoding
1395   /// for the CFI instructions.
1396   uint32_t
1397   generateCompactUnwindEncoding(ArrayRef<MCCFIInstruction> Instrs) const override {
1398     if (Instrs.empty()) return 0;
1399 
1400     // Reset the saved registers.
1401     unsigned SavedRegIdx = 0;
1402     memset(SavedRegs, 0, sizeof(SavedRegs));
1403 
1404     bool HasFP = false;
1405 
1406     // Encode that we are using EBP/RBP as the frame pointer.
1407     uint32_t CompactUnwindEncoding = 0;
1408 
1409     unsigned SubtractInstrIdx = Is64Bit ? 3 : 2;
1410     unsigned InstrOffset = 0;
1411     unsigned StackAdjust = 0;
1412     unsigned StackSize = 0;
1413     unsigned NumDefCFAOffsets = 0;
1414 
1415     for (unsigned i = 0, e = Instrs.size(); i != e; ++i) {
1416       const MCCFIInstruction &Inst = Instrs[i];
1417 
1418       switch (Inst.getOperation()) {
1419       default:
1420         // Any other CFI directives indicate a frame that we aren't prepared
1421         // to represent via compact unwind, so just bail out.
1422         return 0;
1423       case MCCFIInstruction::OpDefCfaRegister: {
1424         // Defines a frame pointer. E.g.
1425         //
1426         //     movq %rsp, %rbp
1427         //  L0:
1428         //     .cfi_def_cfa_register %rbp
1429         //
1430         HasFP = true;
1431 
1432         // If the frame pointer is other than esp/rsp, we do not have a way to
1433         // generate a compact unwinding representation, so bail out.
1434         if (*MRI.getLLVMRegNum(Inst.getRegister(), true) !=
1435             (Is64Bit ? X86::RBP : X86::EBP))
1436           return 0;
1437 
1438         // Reset the counts.
1439         memset(SavedRegs, 0, sizeof(SavedRegs));
1440         StackAdjust = 0;
1441         SavedRegIdx = 0;
1442         InstrOffset += MoveInstrSize;
1443         break;
1444       }
1445       case MCCFIInstruction::OpDefCfaOffset: {
1446         // Defines a new offset for the CFA. E.g.
1447         //
1448         //  With frame:
1449         //
1450         //     pushq %rbp
1451         //  L0:
1452         //     .cfi_def_cfa_offset 16
1453         //
1454         //  Without frame:
1455         //
1456         //     subq $72, %rsp
1457         //  L0:
1458         //     .cfi_def_cfa_offset 80
1459         //
1460         StackSize = std::abs(Inst.getOffset()) / StackDivide;
1461         ++NumDefCFAOffsets;
1462         break;
1463       }
1464       case MCCFIInstruction::OpOffset: {
1465         // Defines a "push" of a callee-saved register. E.g.
1466         //
1467         //     pushq %r15
1468         //     pushq %r14
1469         //     pushq %rbx
1470         //  L0:
1471         //     subq $120, %rsp
1472         //  L1:
1473         //     .cfi_offset %rbx, -40
1474         //     .cfi_offset %r14, -32
1475         //     .cfi_offset %r15, -24
1476         //
1477         if (SavedRegIdx == CU_NUM_SAVED_REGS)
1478           // If there are too many saved registers, we cannot use a compact
1479           // unwind encoding.
1480           return CU::UNWIND_MODE_DWARF;
1481 
1482         unsigned Reg = *MRI.getLLVMRegNum(Inst.getRegister(), true);
1483         SavedRegs[SavedRegIdx++] = Reg;
1484         StackAdjust += OffsetSize;
1485         InstrOffset += PushInstrSize(Reg);
1486         break;
1487       }
1488       }
1489     }
1490 
1491     StackAdjust /= StackDivide;
1492 
1493     if (HasFP) {
1494       if ((StackAdjust & 0xFF) != StackAdjust)
1495         // Offset was too big for a compact unwind encoding.
1496         return CU::UNWIND_MODE_DWARF;
1497 
1498       // Get the encoding of the saved registers when we have a frame pointer.
1499       uint32_t RegEnc = encodeCompactUnwindRegistersWithFrame();
1500       if (RegEnc == ~0U) return CU::UNWIND_MODE_DWARF;
1501 
1502       CompactUnwindEncoding |= CU::UNWIND_MODE_BP_FRAME;
1503       CompactUnwindEncoding |= (StackAdjust & 0xFF) << 16;
1504       CompactUnwindEncoding |= RegEnc & CU::UNWIND_BP_FRAME_REGISTERS;
1505     } else {
1506       SubtractInstrIdx += InstrOffset;
1507       ++StackAdjust;
1508 
1509       if ((StackSize & 0xFF) == StackSize) {
1510         // Frameless stack with a small stack size.
1511         CompactUnwindEncoding |= CU::UNWIND_MODE_STACK_IMMD;
1512 
1513         // Encode the stack size.
1514         CompactUnwindEncoding |= (StackSize & 0xFF) << 16;
1515       } else {
1516         if ((StackAdjust & 0x7) != StackAdjust)
1517           // The extra stack adjustments are too big for us to handle.
1518           return CU::UNWIND_MODE_DWARF;
1519 
1520         // Frameless stack with an offset too large for us to encode compactly.
1521         CompactUnwindEncoding |= CU::UNWIND_MODE_STACK_IND;
1522 
1523         // Encode the offset to the nnnnnn value in the 'subl $nnnnnn, ESP'
1524         // instruction.
1525         CompactUnwindEncoding |= (SubtractInstrIdx & 0xFF) << 16;
1526 
1527         // Encode any extra stack adjustments (done via push instructions).
1528         CompactUnwindEncoding |= (StackAdjust & 0x7) << 13;
1529       }
1530 
1531       // Encode the number of registers saved. (Reverse the list first.)
1532       std::reverse(&SavedRegs[0], &SavedRegs[SavedRegIdx]);
1533       CompactUnwindEncoding |= (SavedRegIdx & 0x7) << 10;
1534 
1535       // Get the encoding of the saved registers when we don't have a frame
1536       // pointer.
1537       uint32_t RegEnc = encodeCompactUnwindRegistersWithoutFrame(SavedRegIdx);
1538       if (RegEnc == ~0U) return CU::UNWIND_MODE_DWARF;
1539 
1540       // Encode the register encoding.
1541       CompactUnwindEncoding |=
1542         RegEnc & CU::UNWIND_FRAMELESS_STACK_REG_PERMUTATION;
1543     }
1544 
1545     return CompactUnwindEncoding;
1546   }
1547 };
1548 
1549 } // end anonymous namespace
1550 
1551 MCAsmBackend *llvm::createX86_32AsmBackend(const Target &T,
1552                                            const MCSubtargetInfo &STI,
1553                                            const MCRegisterInfo &MRI,
1554                                            const MCTargetOptions &Options) {
1555   const Triple &TheTriple = STI.getTargetTriple();
1556   if (TheTriple.isOSBinFormatMachO())
1557     return new DarwinX86AsmBackend(T, MRI, STI);
1558 
1559   if (TheTriple.isOSWindows() && TheTriple.isOSBinFormatCOFF())
1560     return new WindowsX86AsmBackend(T, false, STI);
1561 
1562   uint8_t OSABI = MCELFObjectTargetWriter::getOSABI(TheTriple.getOS());
1563 
1564   if (TheTriple.isOSIAMCU())
1565     return new ELFX86_IAMCUAsmBackend(T, OSABI, STI);
1566 
1567   return new ELFX86_32AsmBackend(T, OSABI, STI);
1568 }
1569 
1570 MCAsmBackend *llvm::createX86_64AsmBackend(const Target &T,
1571                                            const MCSubtargetInfo &STI,
1572                                            const MCRegisterInfo &MRI,
1573                                            const MCTargetOptions &Options) {
1574   const Triple &TheTriple = STI.getTargetTriple();
1575   if (TheTriple.isOSBinFormatMachO())
1576     return new DarwinX86AsmBackend(T, MRI, STI);
1577 
1578   if (TheTriple.isOSWindows() && TheTriple.isOSBinFormatCOFF())
1579     return new WindowsX86AsmBackend(T, true, STI);
1580 
1581   uint8_t OSABI = MCELFObjectTargetWriter::getOSABI(TheTriple.getOS());
1582 
1583   if (TheTriple.getEnvironment() == Triple::GNUX32)
1584     return new ELFX86_X32AsmBackend(T, OSABI, STI);
1585   return new ELFX86_64AsmBackend(T, OSABI, STI);
1586 }
1587