1 //===- AArch64InstrInfo.cpp - AArch64 Instruction Information -------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This file contains the AArch64 implementation of the TargetInstrInfo class.
10 //
11 //===----------------------------------------------------------------------===//
12 
13 #include "AArch64InstrInfo.h"
14 #include "AArch64MachineFunctionInfo.h"
15 #include "AArch64Subtarget.h"
16 #include "MCTargetDesc/AArch64AddressingModes.h"
17 #include "Utils/AArch64BaseInfo.h"
18 #include "llvm/ADT/ArrayRef.h"
19 #include "llvm/ADT/STLExtras.h"
20 #include "llvm/ADT/SmallVector.h"
21 #include "llvm/CodeGen/MachineBasicBlock.h"
22 #include "llvm/CodeGen/MachineFrameInfo.h"
23 #include "llvm/CodeGen/MachineFunction.h"
24 #include "llvm/CodeGen/MachineInstr.h"
25 #include "llvm/CodeGen/MachineInstrBuilder.h"
26 #include "llvm/CodeGen/MachineMemOperand.h"
27 #include "llvm/CodeGen/MachineOperand.h"
28 #include "llvm/CodeGen/MachineRegisterInfo.h"
29 #include "llvm/CodeGen/MachineModuleInfo.h"
30 #include "llvm/CodeGen/StackMaps.h"
31 #include "llvm/CodeGen/TargetRegisterInfo.h"
32 #include "llvm/CodeGen/TargetSubtargetInfo.h"
33 #include "llvm/IR/DebugLoc.h"
34 #include "llvm/IR/GlobalValue.h"
35 #include "llvm/MC/MCInst.h"
36 #include "llvm/MC/MCInstrDesc.h"
37 #include "llvm/Support/Casting.h"
38 #include "llvm/Support/CodeGen.h"
39 #include "llvm/Support/CommandLine.h"
40 #include "llvm/Support/Compiler.h"
41 #include "llvm/Support/ErrorHandling.h"
42 #include "llvm/Support/MathExtras.h"
43 #include "llvm/Target/TargetMachine.h"
44 #include "llvm/Target/TargetOptions.h"
45 #include <cassert>
46 #include <cstdint>
47 #include <iterator>
48 #include <utility>
49 
50 using namespace llvm;
51 
52 #define GET_INSTRINFO_CTOR_DTOR
53 #include "AArch64GenInstrInfo.inc"
54 
55 static cl::opt<unsigned> TBZDisplacementBits(
56     "aarch64-tbz-offset-bits", cl::Hidden, cl::init(14),
57     cl::desc("Restrict range of TB[N]Z instructions (DEBUG)"));
58 
59 static cl::opt<unsigned> CBZDisplacementBits(
60     "aarch64-cbz-offset-bits", cl::Hidden, cl::init(19),
61     cl::desc("Restrict range of CB[N]Z instructions (DEBUG)"));
62 
63 static cl::opt<unsigned>
64     BCCDisplacementBits("aarch64-bcc-offset-bits", cl::Hidden, cl::init(19),
65                         cl::desc("Restrict range of Bcc instructions (DEBUG)"));
66 
67 AArch64InstrInfo::AArch64InstrInfo(const AArch64Subtarget &STI)
68     : AArch64GenInstrInfo(AArch64::ADJCALLSTACKDOWN, AArch64::ADJCALLSTACKUP,
69                           AArch64::CATCHRET),
70       RI(STI.getTargetTriple()), Subtarget(STI) {}
71 
72 /// GetInstSize - Return the number of bytes of code the specified
73 /// instruction may be.  This returns the maximum number of bytes.
74 unsigned AArch64InstrInfo::getInstSizeInBytes(const MachineInstr &MI) const {
75   const MachineBasicBlock &MBB = *MI.getParent();
76   const MachineFunction *MF = MBB.getParent();
77   const MCAsmInfo *MAI = MF->getTarget().getMCAsmInfo();
78 
79   if (MI.getOpcode() == AArch64::INLINEASM)
80     return getInlineAsmLength(MI.getOperand(0).getSymbolName(), *MAI);
81 
82   // FIXME: We currently only handle pseudoinstructions that don't get expanded
83   //        before the assembly printer.
84   unsigned NumBytes = 0;
85   const MCInstrDesc &Desc = MI.getDesc();
86   switch (Desc.getOpcode()) {
87   default:
88     // Anything not explicitly designated otherwise is a normal 4-byte insn.
89     NumBytes = 4;
90     break;
91   case TargetOpcode::DBG_VALUE:
92   case TargetOpcode::EH_LABEL:
93   case TargetOpcode::IMPLICIT_DEF:
94   case TargetOpcode::KILL:
95     NumBytes = 0;
96     break;
97   case TargetOpcode::STACKMAP:
98     // The upper bound for a stackmap intrinsic is the full length of its shadow
99     NumBytes = StackMapOpers(&MI).getNumPatchBytes();
100     assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
101     break;
102   case TargetOpcode::PATCHPOINT:
103     // The size of the patchpoint intrinsic is the number of bytes requested
104     NumBytes = PatchPointOpers(&MI).getNumPatchBytes();
105     assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
106     break;
107   case AArch64::TLSDESC_CALLSEQ:
108     // This gets lowered to an instruction sequence which takes 16 bytes
109     NumBytes = 16;
110     break;
111   case AArch64::JumpTableDest32:
112   case AArch64::JumpTableDest16:
113   case AArch64::JumpTableDest8:
114     NumBytes = 12;
115     break;
116   case AArch64::SPACE:
117     NumBytes = MI.getOperand(1).getImm();
118     break;
119   }
120 
121   return NumBytes;
122 }
123 
124 static void parseCondBranch(MachineInstr *LastInst, MachineBasicBlock *&Target,
125                             SmallVectorImpl<MachineOperand> &Cond) {
126   // Block ends with fall-through condbranch.
127   switch (LastInst->getOpcode()) {
128   default:
129     llvm_unreachable("Unknown branch instruction?");
130   case AArch64::Bcc:
131     Target = LastInst->getOperand(1).getMBB();
132     Cond.push_back(LastInst->getOperand(0));
133     break;
134   case AArch64::CBZW:
135   case AArch64::CBZX:
136   case AArch64::CBNZW:
137   case AArch64::CBNZX:
138     Target = LastInst->getOperand(1).getMBB();
139     Cond.push_back(MachineOperand::CreateImm(-1));
140     Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode()));
141     Cond.push_back(LastInst->getOperand(0));
142     break;
143   case AArch64::TBZW:
144   case AArch64::TBZX:
145   case AArch64::TBNZW:
146   case AArch64::TBNZX:
147     Target = LastInst->getOperand(2).getMBB();
148     Cond.push_back(MachineOperand::CreateImm(-1));
149     Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode()));
150     Cond.push_back(LastInst->getOperand(0));
151     Cond.push_back(LastInst->getOperand(1));
152   }
153 }
154 
155 static unsigned getBranchDisplacementBits(unsigned Opc) {
156   switch (Opc) {
157   default:
158     llvm_unreachable("unexpected opcode!");
159   case AArch64::B:
160     return 64;
161   case AArch64::TBNZW:
162   case AArch64::TBZW:
163   case AArch64::TBNZX:
164   case AArch64::TBZX:
165     return TBZDisplacementBits;
166   case AArch64::CBNZW:
167   case AArch64::CBZW:
168   case AArch64::CBNZX:
169   case AArch64::CBZX:
170     return CBZDisplacementBits;
171   case AArch64::Bcc:
172     return BCCDisplacementBits;
173   }
174 }
175 
176 bool AArch64InstrInfo::isBranchOffsetInRange(unsigned BranchOp,
177                                              int64_t BrOffset) const {
178   unsigned Bits = getBranchDisplacementBits(BranchOp);
179   assert(Bits >= 3 && "max branch displacement must be enough to jump"
180                       "over conditional branch expansion");
181   return isIntN(Bits, BrOffset / 4);
182 }
183 
184 MachineBasicBlock *
185 AArch64InstrInfo::getBranchDestBlock(const MachineInstr &MI) const {
186   switch (MI.getOpcode()) {
187   default:
188     llvm_unreachable("unexpected opcode!");
189   case AArch64::B:
190     return MI.getOperand(0).getMBB();
191   case AArch64::TBZW:
192   case AArch64::TBNZW:
193   case AArch64::TBZX:
194   case AArch64::TBNZX:
195     return MI.getOperand(2).getMBB();
196   case AArch64::CBZW:
197   case AArch64::CBNZW:
198   case AArch64::CBZX:
199   case AArch64::CBNZX:
200   case AArch64::Bcc:
201     return MI.getOperand(1).getMBB();
202   }
203 }
204 
205 // Branch analysis.
206 bool AArch64InstrInfo::analyzeBranch(MachineBasicBlock &MBB,
207                                      MachineBasicBlock *&TBB,
208                                      MachineBasicBlock *&FBB,
209                                      SmallVectorImpl<MachineOperand> &Cond,
210                                      bool AllowModify) const {
211   // If the block has no terminators, it just falls into the block after it.
212   MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
213   if (I == MBB.end())
214     return false;
215 
216   if (!isUnpredicatedTerminator(*I))
217     return false;
218 
219   // Get the last instruction in the block.
220   MachineInstr *LastInst = &*I;
221 
222   // If there is only one terminator instruction, process it.
223   unsigned LastOpc = LastInst->getOpcode();
224   if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
225     if (isUncondBranchOpcode(LastOpc)) {
226       TBB = LastInst->getOperand(0).getMBB();
227       return false;
228     }
229     if (isCondBranchOpcode(LastOpc)) {
230       // Block ends with fall-through condbranch.
231       parseCondBranch(LastInst, TBB, Cond);
232       return false;
233     }
234     return true; // Can't handle indirect branch.
235   }
236 
237   // Get the instruction before it if it is a terminator.
238   MachineInstr *SecondLastInst = &*I;
239   unsigned SecondLastOpc = SecondLastInst->getOpcode();
240 
241   // If AllowModify is true and the block ends with two or more unconditional
242   // branches, delete all but the first unconditional branch.
243   if (AllowModify && isUncondBranchOpcode(LastOpc)) {
244     while (isUncondBranchOpcode(SecondLastOpc)) {
245       LastInst->eraseFromParent();
246       LastInst = SecondLastInst;
247       LastOpc = LastInst->getOpcode();
248       if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
249         // Return now the only terminator is an unconditional branch.
250         TBB = LastInst->getOperand(0).getMBB();
251         return false;
252       } else {
253         SecondLastInst = &*I;
254         SecondLastOpc = SecondLastInst->getOpcode();
255       }
256     }
257   }
258 
259   // If there are three terminators, we don't know what sort of block this is.
260   if (SecondLastInst && I != MBB.begin() && isUnpredicatedTerminator(*--I))
261     return true;
262 
263   // If the block ends with a B and a Bcc, handle it.
264   if (isCondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
265     parseCondBranch(SecondLastInst, TBB, Cond);
266     FBB = LastInst->getOperand(0).getMBB();
267     return false;
268   }
269 
270   // If the block ends with two unconditional branches, handle it.  The second
271   // one is not executed, so remove it.
272   if (isUncondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
273     TBB = SecondLastInst->getOperand(0).getMBB();
274     I = LastInst;
275     if (AllowModify)
276       I->eraseFromParent();
277     return false;
278   }
279 
280   // ...likewise if it ends with an indirect branch followed by an unconditional
281   // branch.
282   if (isIndirectBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
283     I = LastInst;
284     if (AllowModify)
285       I->eraseFromParent();
286     return true;
287   }
288 
289   // Otherwise, can't handle this.
290   return true;
291 }
292 
293 bool AArch64InstrInfo::reverseBranchCondition(
294     SmallVectorImpl<MachineOperand> &Cond) const {
295   if (Cond[0].getImm() != -1) {
296     // Regular Bcc
297     AArch64CC::CondCode CC = (AArch64CC::CondCode)(int)Cond[0].getImm();
298     Cond[0].setImm(AArch64CC::getInvertedCondCode(CC));
299   } else {
300     // Folded compare-and-branch
301     switch (Cond[1].getImm()) {
302     default:
303       llvm_unreachable("Unknown conditional branch!");
304     case AArch64::CBZW:
305       Cond[1].setImm(AArch64::CBNZW);
306       break;
307     case AArch64::CBNZW:
308       Cond[1].setImm(AArch64::CBZW);
309       break;
310     case AArch64::CBZX:
311       Cond[1].setImm(AArch64::CBNZX);
312       break;
313     case AArch64::CBNZX:
314       Cond[1].setImm(AArch64::CBZX);
315       break;
316     case AArch64::TBZW:
317       Cond[1].setImm(AArch64::TBNZW);
318       break;
319     case AArch64::TBNZW:
320       Cond[1].setImm(AArch64::TBZW);
321       break;
322     case AArch64::TBZX:
323       Cond[1].setImm(AArch64::TBNZX);
324       break;
325     case AArch64::TBNZX:
326       Cond[1].setImm(AArch64::TBZX);
327       break;
328     }
329   }
330 
331   return false;
332 }
333 
334 unsigned AArch64InstrInfo::removeBranch(MachineBasicBlock &MBB,
335                                         int *BytesRemoved) const {
336   MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
337   if (I == MBB.end())
338     return 0;
339 
340   if (!isUncondBranchOpcode(I->getOpcode()) &&
341       !isCondBranchOpcode(I->getOpcode()))
342     return 0;
343 
344   // Remove the branch.
345   I->eraseFromParent();
346 
347   I = MBB.end();
348 
349   if (I == MBB.begin()) {
350     if (BytesRemoved)
351       *BytesRemoved = 4;
352     return 1;
353   }
354   --I;
355   if (!isCondBranchOpcode(I->getOpcode())) {
356     if (BytesRemoved)
357       *BytesRemoved = 4;
358     return 1;
359   }
360 
361   // Remove the branch.
362   I->eraseFromParent();
363   if (BytesRemoved)
364     *BytesRemoved = 8;
365 
366   return 2;
367 }
368 
369 void AArch64InstrInfo::instantiateCondBranch(
370     MachineBasicBlock &MBB, const DebugLoc &DL, MachineBasicBlock *TBB,
371     ArrayRef<MachineOperand> Cond) const {
372   if (Cond[0].getImm() != -1) {
373     // Regular Bcc
374     BuildMI(&MBB, DL, get(AArch64::Bcc)).addImm(Cond[0].getImm()).addMBB(TBB);
375   } else {
376     // Folded compare-and-branch
377     // Note that we use addOperand instead of addReg to keep the flags.
378     const MachineInstrBuilder MIB =
379         BuildMI(&MBB, DL, get(Cond[1].getImm())).add(Cond[2]);
380     if (Cond.size() > 3)
381       MIB.addImm(Cond[3].getImm());
382     MIB.addMBB(TBB);
383   }
384 }
385 
386 unsigned AArch64InstrInfo::insertBranch(
387     MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB,
388     ArrayRef<MachineOperand> Cond, const DebugLoc &DL, int *BytesAdded) const {
389   // Shouldn't be a fall through.
390   assert(TBB && "insertBranch must not be told to insert a fallthrough");
391 
392   if (!FBB) {
393     if (Cond.empty()) // Unconditional branch?
394       BuildMI(&MBB, DL, get(AArch64::B)).addMBB(TBB);
395     else
396       instantiateCondBranch(MBB, DL, TBB, Cond);
397 
398     if (BytesAdded)
399       *BytesAdded = 4;
400 
401     return 1;
402   }
403 
404   // Two-way conditional branch.
405   instantiateCondBranch(MBB, DL, TBB, Cond);
406   BuildMI(&MBB, DL, get(AArch64::B)).addMBB(FBB);
407 
408   if (BytesAdded)
409     *BytesAdded = 8;
410 
411   return 2;
412 }
413 
414 // Find the original register that VReg is copied from.
415 static unsigned removeCopies(const MachineRegisterInfo &MRI, unsigned VReg) {
416   while (TargetRegisterInfo::isVirtualRegister(VReg)) {
417     const MachineInstr *DefMI = MRI.getVRegDef(VReg);
418     if (!DefMI->isFullCopy())
419       return VReg;
420     VReg = DefMI->getOperand(1).getReg();
421   }
422   return VReg;
423 }
424 
425 // Determine if VReg is defined by an instruction that can be folded into a
426 // csel instruction. If so, return the folded opcode, and the replacement
427 // register.
428 static unsigned canFoldIntoCSel(const MachineRegisterInfo &MRI, unsigned VReg,
429                                 unsigned *NewVReg = nullptr) {
430   VReg = removeCopies(MRI, VReg);
431   if (!TargetRegisterInfo::isVirtualRegister(VReg))
432     return 0;
433 
434   bool Is64Bit = AArch64::GPR64allRegClass.hasSubClassEq(MRI.getRegClass(VReg));
435   const MachineInstr *DefMI = MRI.getVRegDef(VReg);
436   unsigned Opc = 0;
437   unsigned SrcOpNum = 0;
438   switch (DefMI->getOpcode()) {
439   case AArch64::ADDSXri:
440   case AArch64::ADDSWri:
441     // if NZCV is used, do not fold.
442     if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1)
443       return 0;
444     // fall-through to ADDXri and ADDWri.
445     LLVM_FALLTHROUGH;
446   case AArch64::ADDXri:
447   case AArch64::ADDWri:
448     // add x, 1 -> csinc.
449     if (!DefMI->getOperand(2).isImm() || DefMI->getOperand(2).getImm() != 1 ||
450         DefMI->getOperand(3).getImm() != 0)
451       return 0;
452     SrcOpNum = 1;
453     Opc = Is64Bit ? AArch64::CSINCXr : AArch64::CSINCWr;
454     break;
455 
456   case AArch64::ORNXrr:
457   case AArch64::ORNWrr: {
458     // not x -> csinv, represented as orn dst, xzr, src.
459     unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg());
460     if (ZReg != AArch64::XZR && ZReg != AArch64::WZR)
461       return 0;
462     SrcOpNum = 2;
463     Opc = Is64Bit ? AArch64::CSINVXr : AArch64::CSINVWr;
464     break;
465   }
466 
467   case AArch64::SUBSXrr:
468   case AArch64::SUBSWrr:
469     // if NZCV is used, do not fold.
470     if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1)
471       return 0;
472     // fall-through to SUBXrr and SUBWrr.
473     LLVM_FALLTHROUGH;
474   case AArch64::SUBXrr:
475   case AArch64::SUBWrr: {
476     // neg x -> csneg, represented as sub dst, xzr, src.
477     unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg());
478     if (ZReg != AArch64::XZR && ZReg != AArch64::WZR)
479       return 0;
480     SrcOpNum = 2;
481     Opc = Is64Bit ? AArch64::CSNEGXr : AArch64::CSNEGWr;
482     break;
483   }
484   default:
485     return 0;
486   }
487   assert(Opc && SrcOpNum && "Missing parameters");
488 
489   if (NewVReg)
490     *NewVReg = DefMI->getOperand(SrcOpNum).getReg();
491   return Opc;
492 }
493 
494 bool AArch64InstrInfo::canInsertSelect(const MachineBasicBlock &MBB,
495                                        ArrayRef<MachineOperand> Cond,
496                                        unsigned TrueReg, unsigned FalseReg,
497                                        int &CondCycles, int &TrueCycles,
498                                        int &FalseCycles) const {
499   // Check register classes.
500   const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
501   const TargetRegisterClass *RC =
502       RI.getCommonSubClass(MRI.getRegClass(TrueReg), MRI.getRegClass(FalseReg));
503   if (!RC)
504     return false;
505 
506   // Expanding cbz/tbz requires an extra cycle of latency on the condition.
507   unsigned ExtraCondLat = Cond.size() != 1;
508 
509   // GPRs are handled by csel.
510   // FIXME: Fold in x+1, -x, and ~x when applicable.
511   if (AArch64::GPR64allRegClass.hasSubClassEq(RC) ||
512       AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
513     // Single-cycle csel, csinc, csinv, and csneg.
514     CondCycles = 1 + ExtraCondLat;
515     TrueCycles = FalseCycles = 1;
516     if (canFoldIntoCSel(MRI, TrueReg))
517       TrueCycles = 0;
518     else if (canFoldIntoCSel(MRI, FalseReg))
519       FalseCycles = 0;
520     return true;
521   }
522 
523   // Scalar floating point is handled by fcsel.
524   // FIXME: Form fabs, fmin, and fmax when applicable.
525   if (AArch64::FPR64RegClass.hasSubClassEq(RC) ||
526       AArch64::FPR32RegClass.hasSubClassEq(RC)) {
527     CondCycles = 5 + ExtraCondLat;
528     TrueCycles = FalseCycles = 2;
529     return true;
530   }
531 
532   // Can't do vectors.
533   return false;
534 }
535 
536 void AArch64InstrInfo::insertSelect(MachineBasicBlock &MBB,
537                                     MachineBasicBlock::iterator I,
538                                     const DebugLoc &DL, unsigned DstReg,
539                                     ArrayRef<MachineOperand> Cond,
540                                     unsigned TrueReg, unsigned FalseReg) const {
541   MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
542 
543   // Parse the condition code, see parseCondBranch() above.
544   AArch64CC::CondCode CC;
545   switch (Cond.size()) {
546   default:
547     llvm_unreachable("Unknown condition opcode in Cond");
548   case 1: // b.cc
549     CC = AArch64CC::CondCode(Cond[0].getImm());
550     break;
551   case 3: { // cbz/cbnz
552     // We must insert a compare against 0.
553     bool Is64Bit;
554     switch (Cond[1].getImm()) {
555     default:
556       llvm_unreachable("Unknown branch opcode in Cond");
557     case AArch64::CBZW:
558       Is64Bit = false;
559       CC = AArch64CC::EQ;
560       break;
561     case AArch64::CBZX:
562       Is64Bit = true;
563       CC = AArch64CC::EQ;
564       break;
565     case AArch64::CBNZW:
566       Is64Bit = false;
567       CC = AArch64CC::NE;
568       break;
569     case AArch64::CBNZX:
570       Is64Bit = true;
571       CC = AArch64CC::NE;
572       break;
573     }
574     unsigned SrcReg = Cond[2].getReg();
575     if (Is64Bit) {
576       // cmp reg, #0 is actually subs xzr, reg, #0.
577       MRI.constrainRegClass(SrcReg, &AArch64::GPR64spRegClass);
578       BuildMI(MBB, I, DL, get(AArch64::SUBSXri), AArch64::XZR)
579           .addReg(SrcReg)
580           .addImm(0)
581           .addImm(0);
582     } else {
583       MRI.constrainRegClass(SrcReg, &AArch64::GPR32spRegClass);
584       BuildMI(MBB, I, DL, get(AArch64::SUBSWri), AArch64::WZR)
585           .addReg(SrcReg)
586           .addImm(0)
587           .addImm(0);
588     }
589     break;
590   }
591   case 4: { // tbz/tbnz
592     // We must insert a tst instruction.
593     switch (Cond[1].getImm()) {
594     default:
595       llvm_unreachable("Unknown branch opcode in Cond");
596     case AArch64::TBZW:
597     case AArch64::TBZX:
598       CC = AArch64CC::EQ;
599       break;
600     case AArch64::TBNZW:
601     case AArch64::TBNZX:
602       CC = AArch64CC::NE;
603       break;
604     }
605     // cmp reg, #foo is actually ands xzr, reg, #1<<foo.
606     if (Cond[1].getImm() == AArch64::TBZW || Cond[1].getImm() == AArch64::TBNZW)
607       BuildMI(MBB, I, DL, get(AArch64::ANDSWri), AArch64::WZR)
608           .addReg(Cond[2].getReg())
609           .addImm(
610               AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 32));
611     else
612       BuildMI(MBB, I, DL, get(AArch64::ANDSXri), AArch64::XZR)
613           .addReg(Cond[2].getReg())
614           .addImm(
615               AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 64));
616     break;
617   }
618   }
619 
620   unsigned Opc = 0;
621   const TargetRegisterClass *RC = nullptr;
622   bool TryFold = false;
623   if (MRI.constrainRegClass(DstReg, &AArch64::GPR64RegClass)) {
624     RC = &AArch64::GPR64RegClass;
625     Opc = AArch64::CSELXr;
626     TryFold = true;
627   } else if (MRI.constrainRegClass(DstReg, &AArch64::GPR32RegClass)) {
628     RC = &AArch64::GPR32RegClass;
629     Opc = AArch64::CSELWr;
630     TryFold = true;
631   } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR64RegClass)) {
632     RC = &AArch64::FPR64RegClass;
633     Opc = AArch64::FCSELDrrr;
634   } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR32RegClass)) {
635     RC = &AArch64::FPR32RegClass;
636     Opc = AArch64::FCSELSrrr;
637   }
638   assert(RC && "Unsupported regclass");
639 
640   // Try folding simple instructions into the csel.
641   if (TryFold) {
642     unsigned NewVReg = 0;
643     unsigned FoldedOpc = canFoldIntoCSel(MRI, TrueReg, &NewVReg);
644     if (FoldedOpc) {
645       // The folded opcodes csinc, csinc and csneg apply the operation to
646       // FalseReg, so we need to invert the condition.
647       CC = AArch64CC::getInvertedCondCode(CC);
648       TrueReg = FalseReg;
649     } else
650       FoldedOpc = canFoldIntoCSel(MRI, FalseReg, &NewVReg);
651 
652     // Fold the operation. Leave any dead instructions for DCE to clean up.
653     if (FoldedOpc) {
654       FalseReg = NewVReg;
655       Opc = FoldedOpc;
656       // The extends the live range of NewVReg.
657       MRI.clearKillFlags(NewVReg);
658     }
659   }
660 
661   // Pull all virtual register into the appropriate class.
662   MRI.constrainRegClass(TrueReg, RC);
663   MRI.constrainRegClass(FalseReg, RC);
664 
665   // Insert the csel.
666   BuildMI(MBB, I, DL, get(Opc), DstReg)
667       .addReg(TrueReg)
668       .addReg(FalseReg)
669       .addImm(CC);
670 }
671 
672 /// Returns true if a MOVi32imm or MOVi64imm can be expanded to an  ORRxx.
673 static bool canBeExpandedToORR(const MachineInstr &MI, unsigned BitSize) {
674   uint64_t Imm = MI.getOperand(1).getImm();
675   uint64_t UImm = Imm << (64 - BitSize) >> (64 - BitSize);
676   uint64_t Encoding;
677   return AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding);
678 }
679 
680 // FIXME: this implementation should be micro-architecture dependent, so a
681 // micro-architecture target hook should be introduced here in future.
682 bool AArch64InstrInfo::isAsCheapAsAMove(const MachineInstr &MI) const {
683   if (!Subtarget.hasCustomCheapAsMoveHandling())
684     return MI.isAsCheapAsAMove();
685 
686   const unsigned Opcode = MI.getOpcode();
687 
688   // Firstly, check cases gated by features.
689 
690   if (Subtarget.hasZeroCycleZeroingFP()) {
691     if (Opcode == AArch64::FMOVH0 ||
692         Opcode == AArch64::FMOVS0 ||
693         Opcode == AArch64::FMOVD0)
694       return true;
695   }
696 
697   if (Subtarget.hasZeroCycleZeroingGP()) {
698     if (Opcode == TargetOpcode::COPY &&
699         (MI.getOperand(1).getReg() == AArch64::WZR ||
700          MI.getOperand(1).getReg() == AArch64::XZR))
701       return true;
702   }
703 
704   // Secondly, check cases specific to sub-targets.
705 
706   if (Subtarget.hasExynosCheapAsMoveHandling()) {
707     if (isExynosCheapAsMove(MI))
708       return true;
709 
710     return MI.isAsCheapAsAMove();
711   }
712 
713   // Finally, check generic cases.
714 
715   switch (Opcode) {
716   default:
717     return false;
718 
719   // add/sub on register without shift
720   case AArch64::ADDWri:
721   case AArch64::ADDXri:
722   case AArch64::SUBWri:
723   case AArch64::SUBXri:
724     return (MI.getOperand(3).getImm() == 0);
725 
726   // logical ops on immediate
727   case AArch64::ANDWri:
728   case AArch64::ANDXri:
729   case AArch64::EORWri:
730   case AArch64::EORXri:
731   case AArch64::ORRWri:
732   case AArch64::ORRXri:
733     return true;
734 
735   // logical ops on register without shift
736   case AArch64::ANDWrr:
737   case AArch64::ANDXrr:
738   case AArch64::BICWrr:
739   case AArch64::BICXrr:
740   case AArch64::EONWrr:
741   case AArch64::EONXrr:
742   case AArch64::EORWrr:
743   case AArch64::EORXrr:
744   case AArch64::ORNWrr:
745   case AArch64::ORNXrr:
746   case AArch64::ORRWrr:
747   case AArch64::ORRXrr:
748     return true;
749 
750   // If MOVi32imm or MOVi64imm can be expanded into ORRWri or
751   // ORRXri, it is as cheap as MOV
752   case AArch64::MOVi32imm:
753     return canBeExpandedToORR(MI, 32);
754   case AArch64::MOVi64imm:
755     return canBeExpandedToORR(MI, 64);
756   }
757 
758   llvm_unreachable("Unknown opcode to check as cheap as a move!");
759 }
760 
761 bool AArch64InstrInfo::isFalkorShiftExtFast(const MachineInstr &MI) {
762   switch (MI.getOpcode()) {
763   default:
764     return false;
765 
766   case AArch64::ADDWrs:
767   case AArch64::ADDXrs:
768   case AArch64::ADDSWrs:
769   case AArch64::ADDSXrs: {
770     unsigned Imm = MI.getOperand(3).getImm();
771     unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
772     if (ShiftVal == 0)
773       return true;
774     return AArch64_AM::getShiftType(Imm) == AArch64_AM::LSL && ShiftVal <= 5;
775   }
776 
777   case AArch64::ADDWrx:
778   case AArch64::ADDXrx:
779   case AArch64::ADDXrx64:
780   case AArch64::ADDSWrx:
781   case AArch64::ADDSXrx:
782   case AArch64::ADDSXrx64: {
783     unsigned Imm = MI.getOperand(3).getImm();
784     switch (AArch64_AM::getArithExtendType(Imm)) {
785     default:
786       return false;
787     case AArch64_AM::UXTB:
788     case AArch64_AM::UXTH:
789     case AArch64_AM::UXTW:
790     case AArch64_AM::UXTX:
791       return AArch64_AM::getArithShiftValue(Imm) <= 4;
792     }
793   }
794 
795   case AArch64::SUBWrs:
796   case AArch64::SUBSWrs: {
797     unsigned Imm = MI.getOperand(3).getImm();
798     unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
799     return ShiftVal == 0 ||
800            (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 31);
801   }
802 
803   case AArch64::SUBXrs:
804   case AArch64::SUBSXrs: {
805     unsigned Imm = MI.getOperand(3).getImm();
806     unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
807     return ShiftVal == 0 ||
808            (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 63);
809   }
810 
811   case AArch64::SUBWrx:
812   case AArch64::SUBXrx:
813   case AArch64::SUBXrx64:
814   case AArch64::SUBSWrx:
815   case AArch64::SUBSXrx:
816   case AArch64::SUBSXrx64: {
817     unsigned Imm = MI.getOperand(3).getImm();
818     switch (AArch64_AM::getArithExtendType(Imm)) {
819     default:
820       return false;
821     case AArch64_AM::UXTB:
822     case AArch64_AM::UXTH:
823     case AArch64_AM::UXTW:
824     case AArch64_AM::UXTX:
825       return AArch64_AM::getArithShiftValue(Imm) == 0;
826     }
827   }
828 
829   case AArch64::LDRBBroW:
830   case AArch64::LDRBBroX:
831   case AArch64::LDRBroW:
832   case AArch64::LDRBroX:
833   case AArch64::LDRDroW:
834   case AArch64::LDRDroX:
835   case AArch64::LDRHHroW:
836   case AArch64::LDRHHroX:
837   case AArch64::LDRHroW:
838   case AArch64::LDRHroX:
839   case AArch64::LDRQroW:
840   case AArch64::LDRQroX:
841   case AArch64::LDRSBWroW:
842   case AArch64::LDRSBWroX:
843   case AArch64::LDRSBXroW:
844   case AArch64::LDRSBXroX:
845   case AArch64::LDRSHWroW:
846   case AArch64::LDRSHWroX:
847   case AArch64::LDRSHXroW:
848   case AArch64::LDRSHXroX:
849   case AArch64::LDRSWroW:
850   case AArch64::LDRSWroX:
851   case AArch64::LDRSroW:
852   case AArch64::LDRSroX:
853   case AArch64::LDRWroW:
854   case AArch64::LDRWroX:
855   case AArch64::LDRXroW:
856   case AArch64::LDRXroX:
857   case AArch64::PRFMroW:
858   case AArch64::PRFMroX:
859   case AArch64::STRBBroW:
860   case AArch64::STRBBroX:
861   case AArch64::STRBroW:
862   case AArch64::STRBroX:
863   case AArch64::STRDroW:
864   case AArch64::STRDroX:
865   case AArch64::STRHHroW:
866   case AArch64::STRHHroX:
867   case AArch64::STRHroW:
868   case AArch64::STRHroX:
869   case AArch64::STRQroW:
870   case AArch64::STRQroX:
871   case AArch64::STRSroW:
872   case AArch64::STRSroX:
873   case AArch64::STRWroW:
874   case AArch64::STRWroX:
875   case AArch64::STRXroW:
876   case AArch64::STRXroX: {
877     unsigned IsSigned = MI.getOperand(3).getImm();
878     return !IsSigned;
879   }
880   }
881 }
882 
883 bool AArch64InstrInfo::isSEHInstruction(const MachineInstr &MI) {
884   unsigned Opc = MI.getOpcode();
885   switch (Opc) {
886     default:
887       return false;
888     case AArch64::SEH_StackAlloc:
889     case AArch64::SEH_SaveFPLR:
890     case AArch64::SEH_SaveFPLR_X:
891     case AArch64::SEH_SaveReg:
892     case AArch64::SEH_SaveReg_X:
893     case AArch64::SEH_SaveRegP:
894     case AArch64::SEH_SaveRegP_X:
895     case AArch64::SEH_SaveFReg:
896     case AArch64::SEH_SaveFReg_X:
897     case AArch64::SEH_SaveFRegP:
898     case AArch64::SEH_SaveFRegP_X:
899     case AArch64::SEH_SetFP:
900     case AArch64::SEH_AddFP:
901     case AArch64::SEH_Nop:
902     case AArch64::SEH_PrologEnd:
903     case AArch64::SEH_EpilogStart:
904     case AArch64::SEH_EpilogEnd:
905       return true;
906   }
907 }
908 
909 bool AArch64InstrInfo::isCoalescableExtInstr(const MachineInstr &MI,
910                                              unsigned &SrcReg, unsigned &DstReg,
911                                              unsigned &SubIdx) const {
912   switch (MI.getOpcode()) {
913   default:
914     return false;
915   case AArch64::SBFMXri: // aka sxtw
916   case AArch64::UBFMXri: // aka uxtw
917     // Check for the 32 -> 64 bit extension case, these instructions can do
918     // much more.
919     if (MI.getOperand(2).getImm() != 0 || MI.getOperand(3).getImm() != 31)
920       return false;
921     // This is a signed or unsigned 32 -> 64 bit extension.
922     SrcReg = MI.getOperand(1).getReg();
923     DstReg = MI.getOperand(0).getReg();
924     SubIdx = AArch64::sub_32;
925     return true;
926   }
927 }
928 
929 bool AArch64InstrInfo::areMemAccessesTriviallyDisjoint(
930     const MachineInstr &MIa, const MachineInstr &MIb, AliasAnalysis *AA) const {
931   const TargetRegisterInfo *TRI = &getRegisterInfo();
932   const MachineOperand *BaseOpA = nullptr, *BaseOpB = nullptr;
933   int64_t OffsetA = 0, OffsetB = 0;
934   unsigned WidthA = 0, WidthB = 0;
935 
936   assert(MIa.mayLoadOrStore() && "MIa must be a load or store.");
937   assert(MIb.mayLoadOrStore() && "MIb must be a load or store.");
938 
939   if (MIa.hasUnmodeledSideEffects() || MIb.hasUnmodeledSideEffects() ||
940       MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef())
941     return false;
942 
943   // Retrieve the base, offset from the base and width. Width
944   // is the size of memory that is being loaded/stored (e.g. 1, 2, 4, 8).  If
945   // base are identical, and the offset of a lower memory access +
946   // the width doesn't overlap the offset of a higher memory access,
947   // then the memory accesses are different.
948   if (getMemOperandWithOffsetWidth(MIa, BaseOpA, OffsetA, WidthA, TRI) &&
949       getMemOperandWithOffsetWidth(MIb, BaseOpB, OffsetB, WidthB, TRI)) {
950     if (BaseOpA->isIdenticalTo(*BaseOpB)) {
951       int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB;
952       int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA;
953       int LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB;
954       if (LowOffset + LowWidth <= HighOffset)
955         return true;
956     }
957   }
958   return false;
959 }
960 
961 bool AArch64InstrInfo::isSchedulingBoundary(const MachineInstr &MI,
962                                             const MachineBasicBlock *MBB,
963                                             const MachineFunction &MF) const {
964   if (TargetInstrInfo::isSchedulingBoundary(MI, MBB, MF))
965     return true;
966   switch (MI.getOpcode()) {
967   case AArch64::HINT:
968     // CSDB hints are scheduling barriers.
969     if (MI.getOperand(0).getImm() == 0x14)
970       return true;
971     break;
972   case AArch64::DSB:
973   case AArch64::ISB:
974     // DSB and ISB also are scheduling barriers.
975     return true;
976   default:;
977   }
978   return isSEHInstruction(MI);
979 }
980 
981 /// analyzeCompare - For a comparison instruction, return the source registers
982 /// in SrcReg and SrcReg2, and the value it compares against in CmpValue.
983 /// Return true if the comparison instruction can be analyzed.
984 bool AArch64InstrInfo::analyzeCompare(const MachineInstr &MI, unsigned &SrcReg,
985                                       unsigned &SrcReg2, int &CmpMask,
986                                       int &CmpValue) const {
987   // The first operand can be a frame index where we'd normally expect a
988   // register.
989   assert(MI.getNumOperands() >= 2 && "All AArch64 cmps should have 2 operands");
990   if (!MI.getOperand(1).isReg())
991     return false;
992 
993   switch (MI.getOpcode()) {
994   default:
995     break;
996   case AArch64::SUBSWrr:
997   case AArch64::SUBSWrs:
998   case AArch64::SUBSWrx:
999   case AArch64::SUBSXrr:
1000   case AArch64::SUBSXrs:
1001   case AArch64::SUBSXrx:
1002   case AArch64::ADDSWrr:
1003   case AArch64::ADDSWrs:
1004   case AArch64::ADDSWrx:
1005   case AArch64::ADDSXrr:
1006   case AArch64::ADDSXrs:
1007   case AArch64::ADDSXrx:
1008     // Replace SUBSWrr with SUBWrr if NZCV is not used.
1009     SrcReg = MI.getOperand(1).getReg();
1010     SrcReg2 = MI.getOperand(2).getReg();
1011     CmpMask = ~0;
1012     CmpValue = 0;
1013     return true;
1014   case AArch64::SUBSWri:
1015   case AArch64::ADDSWri:
1016   case AArch64::SUBSXri:
1017   case AArch64::ADDSXri:
1018     SrcReg = MI.getOperand(1).getReg();
1019     SrcReg2 = 0;
1020     CmpMask = ~0;
1021     // FIXME: In order to convert CmpValue to 0 or 1
1022     CmpValue = MI.getOperand(2).getImm() != 0;
1023     return true;
1024   case AArch64::ANDSWri:
1025   case AArch64::ANDSXri:
1026     // ANDS does not use the same encoding scheme as the others xxxS
1027     // instructions.
1028     SrcReg = MI.getOperand(1).getReg();
1029     SrcReg2 = 0;
1030     CmpMask = ~0;
1031     // FIXME:The return val type of decodeLogicalImmediate is uint64_t,
1032     // while the type of CmpValue is int. When converting uint64_t to int,
1033     // the high 32 bits of uint64_t will be lost.
1034     // In fact it causes a bug in spec2006-483.xalancbmk
1035     // CmpValue is only used to compare with zero in OptimizeCompareInstr
1036     CmpValue = AArch64_AM::decodeLogicalImmediate(
1037                    MI.getOperand(2).getImm(),
1038                    MI.getOpcode() == AArch64::ANDSWri ? 32 : 64) != 0;
1039     return true;
1040   }
1041 
1042   return false;
1043 }
1044 
1045 static bool UpdateOperandRegClass(MachineInstr &Instr) {
1046   MachineBasicBlock *MBB = Instr.getParent();
1047   assert(MBB && "Can't get MachineBasicBlock here");
1048   MachineFunction *MF = MBB->getParent();
1049   assert(MF && "Can't get MachineFunction here");
1050   const TargetInstrInfo *TII = MF->getSubtarget().getInstrInfo();
1051   const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo();
1052   MachineRegisterInfo *MRI = &MF->getRegInfo();
1053 
1054   for (unsigned OpIdx = 0, EndIdx = Instr.getNumOperands(); OpIdx < EndIdx;
1055        ++OpIdx) {
1056     MachineOperand &MO = Instr.getOperand(OpIdx);
1057     const TargetRegisterClass *OpRegCstraints =
1058         Instr.getRegClassConstraint(OpIdx, TII, TRI);
1059 
1060     // If there's no constraint, there's nothing to do.
1061     if (!OpRegCstraints)
1062       continue;
1063     // If the operand is a frame index, there's nothing to do here.
1064     // A frame index operand will resolve correctly during PEI.
1065     if (MO.isFI())
1066       continue;
1067 
1068     assert(MO.isReg() &&
1069            "Operand has register constraints without being a register!");
1070 
1071     unsigned Reg = MO.getReg();
1072     if (TargetRegisterInfo::isPhysicalRegister(Reg)) {
1073       if (!OpRegCstraints->contains(Reg))
1074         return false;
1075     } else if (!OpRegCstraints->hasSubClassEq(MRI->getRegClass(Reg)) &&
1076                !MRI->constrainRegClass(Reg, OpRegCstraints))
1077       return false;
1078   }
1079 
1080   return true;
1081 }
1082 
1083 /// Return the opcode that does not set flags when possible - otherwise
1084 /// return the original opcode. The caller is responsible to do the actual
1085 /// substitution and legality checking.
1086 static unsigned convertToNonFlagSettingOpc(const MachineInstr &MI) {
1087   // Don't convert all compare instructions, because for some the zero register
1088   // encoding becomes the sp register.
1089   bool MIDefinesZeroReg = false;
1090   if (MI.definesRegister(AArch64::WZR) || MI.definesRegister(AArch64::XZR))
1091     MIDefinesZeroReg = true;
1092 
1093   switch (MI.getOpcode()) {
1094   default:
1095     return MI.getOpcode();
1096   case AArch64::ADDSWrr:
1097     return AArch64::ADDWrr;
1098   case AArch64::ADDSWri:
1099     return MIDefinesZeroReg ? AArch64::ADDSWri : AArch64::ADDWri;
1100   case AArch64::ADDSWrs:
1101     return MIDefinesZeroReg ? AArch64::ADDSWrs : AArch64::ADDWrs;
1102   case AArch64::ADDSWrx:
1103     return AArch64::ADDWrx;
1104   case AArch64::ADDSXrr:
1105     return AArch64::ADDXrr;
1106   case AArch64::ADDSXri:
1107     return MIDefinesZeroReg ? AArch64::ADDSXri : AArch64::ADDXri;
1108   case AArch64::ADDSXrs:
1109     return MIDefinesZeroReg ? AArch64::ADDSXrs : AArch64::ADDXrs;
1110   case AArch64::ADDSXrx:
1111     return AArch64::ADDXrx;
1112   case AArch64::SUBSWrr:
1113     return AArch64::SUBWrr;
1114   case AArch64::SUBSWri:
1115     return MIDefinesZeroReg ? AArch64::SUBSWri : AArch64::SUBWri;
1116   case AArch64::SUBSWrs:
1117     return MIDefinesZeroReg ? AArch64::SUBSWrs : AArch64::SUBWrs;
1118   case AArch64::SUBSWrx:
1119     return AArch64::SUBWrx;
1120   case AArch64::SUBSXrr:
1121     return AArch64::SUBXrr;
1122   case AArch64::SUBSXri:
1123     return MIDefinesZeroReg ? AArch64::SUBSXri : AArch64::SUBXri;
1124   case AArch64::SUBSXrs:
1125     return MIDefinesZeroReg ? AArch64::SUBSXrs : AArch64::SUBXrs;
1126   case AArch64::SUBSXrx:
1127     return AArch64::SUBXrx;
1128   }
1129 }
1130 
1131 enum AccessKind { AK_Write = 0x01, AK_Read = 0x10, AK_All = 0x11 };
1132 
1133 /// True when condition flags are accessed (either by writing or reading)
1134 /// on the instruction trace starting at From and ending at To.
1135 ///
1136 /// Note: If From and To are from different blocks it's assumed CC are accessed
1137 ///       on the path.
1138 static bool areCFlagsAccessedBetweenInstrs(
1139     MachineBasicBlock::iterator From, MachineBasicBlock::iterator To,
1140     const TargetRegisterInfo *TRI, const AccessKind AccessToCheck = AK_All) {
1141   // Early exit if To is at the beginning of the BB.
1142   if (To == To->getParent()->begin())
1143     return true;
1144 
1145   // Check whether the instructions are in the same basic block
1146   // If not, assume the condition flags might get modified somewhere.
1147   if (To->getParent() != From->getParent())
1148     return true;
1149 
1150   // From must be above To.
1151   assert(std::find_if(++To.getReverse(), To->getParent()->rend(),
1152                       [From](MachineInstr &MI) {
1153                         return MI.getIterator() == From;
1154                       }) != To->getParent()->rend());
1155 
1156   // We iterate backward starting \p To until we hit \p From.
1157   for (--To; To != From; --To) {
1158     const MachineInstr &Instr = *To;
1159 
1160     if (((AccessToCheck & AK_Write) &&
1161          Instr.modifiesRegister(AArch64::NZCV, TRI)) ||
1162         ((AccessToCheck & AK_Read) && Instr.readsRegister(AArch64::NZCV, TRI)))
1163       return true;
1164   }
1165   return false;
1166 }
1167 
1168 /// Try to optimize a compare instruction. A compare instruction is an
1169 /// instruction which produces AArch64::NZCV. It can be truly compare
1170 /// instruction
1171 /// when there are no uses of its destination register.
1172 ///
1173 /// The following steps are tried in order:
1174 /// 1. Convert CmpInstr into an unconditional version.
1175 /// 2. Remove CmpInstr if above there is an instruction producing a needed
1176 ///    condition code or an instruction which can be converted into such an
1177 ///    instruction.
1178 ///    Only comparison with zero is supported.
1179 bool AArch64InstrInfo::optimizeCompareInstr(
1180     MachineInstr &CmpInstr, unsigned SrcReg, unsigned SrcReg2, int CmpMask,
1181     int CmpValue, const MachineRegisterInfo *MRI) const {
1182   assert(CmpInstr.getParent());
1183   assert(MRI);
1184 
1185   // Replace SUBSWrr with SUBWrr if NZCV is not used.
1186   int DeadNZCVIdx = CmpInstr.findRegisterDefOperandIdx(AArch64::NZCV, true);
1187   if (DeadNZCVIdx != -1) {
1188     if (CmpInstr.definesRegister(AArch64::WZR) ||
1189         CmpInstr.definesRegister(AArch64::XZR)) {
1190       CmpInstr.eraseFromParent();
1191       return true;
1192     }
1193     unsigned Opc = CmpInstr.getOpcode();
1194     unsigned NewOpc = convertToNonFlagSettingOpc(CmpInstr);
1195     if (NewOpc == Opc)
1196       return false;
1197     const MCInstrDesc &MCID = get(NewOpc);
1198     CmpInstr.setDesc(MCID);
1199     CmpInstr.RemoveOperand(DeadNZCVIdx);
1200     bool succeeded = UpdateOperandRegClass(CmpInstr);
1201     (void)succeeded;
1202     assert(succeeded && "Some operands reg class are incompatible!");
1203     return true;
1204   }
1205 
1206   // Continue only if we have a "ri" where immediate is zero.
1207   // FIXME:CmpValue has already been converted to 0 or 1 in analyzeCompare
1208   // function.
1209   assert((CmpValue == 0 || CmpValue == 1) && "CmpValue must be 0 or 1!");
1210   if (CmpValue != 0 || SrcReg2 != 0)
1211     return false;
1212 
1213   // CmpInstr is a Compare instruction if destination register is not used.
1214   if (!MRI->use_nodbg_empty(CmpInstr.getOperand(0).getReg()))
1215     return false;
1216 
1217   return substituteCmpToZero(CmpInstr, SrcReg, MRI);
1218 }
1219 
1220 /// Get opcode of S version of Instr.
1221 /// If Instr is S version its opcode is returned.
1222 /// AArch64::INSTRUCTION_LIST_END is returned if Instr does not have S version
1223 /// or we are not interested in it.
1224 static unsigned sForm(MachineInstr &Instr) {
1225   switch (Instr.getOpcode()) {
1226   default:
1227     return AArch64::INSTRUCTION_LIST_END;
1228 
1229   case AArch64::ADDSWrr:
1230   case AArch64::ADDSWri:
1231   case AArch64::ADDSXrr:
1232   case AArch64::ADDSXri:
1233   case AArch64::SUBSWrr:
1234   case AArch64::SUBSWri:
1235   case AArch64::SUBSXrr:
1236   case AArch64::SUBSXri:
1237     return Instr.getOpcode();
1238 
1239   case AArch64::ADDWrr:
1240     return AArch64::ADDSWrr;
1241   case AArch64::ADDWri:
1242     return AArch64::ADDSWri;
1243   case AArch64::ADDXrr:
1244     return AArch64::ADDSXrr;
1245   case AArch64::ADDXri:
1246     return AArch64::ADDSXri;
1247   case AArch64::ADCWr:
1248     return AArch64::ADCSWr;
1249   case AArch64::ADCXr:
1250     return AArch64::ADCSXr;
1251   case AArch64::SUBWrr:
1252     return AArch64::SUBSWrr;
1253   case AArch64::SUBWri:
1254     return AArch64::SUBSWri;
1255   case AArch64::SUBXrr:
1256     return AArch64::SUBSXrr;
1257   case AArch64::SUBXri:
1258     return AArch64::SUBSXri;
1259   case AArch64::SBCWr:
1260     return AArch64::SBCSWr;
1261   case AArch64::SBCXr:
1262     return AArch64::SBCSXr;
1263   case AArch64::ANDWri:
1264     return AArch64::ANDSWri;
1265   case AArch64::ANDXri:
1266     return AArch64::ANDSXri;
1267   }
1268 }
1269 
1270 /// Check if AArch64::NZCV should be alive in successors of MBB.
1271 static bool areCFlagsAliveInSuccessors(MachineBasicBlock *MBB) {
1272   for (auto *BB : MBB->successors())
1273     if (BB->isLiveIn(AArch64::NZCV))
1274       return true;
1275   return false;
1276 }
1277 
1278 namespace {
1279 
1280 struct UsedNZCV {
1281   bool N = false;
1282   bool Z = false;
1283   bool C = false;
1284   bool V = false;
1285 
1286   UsedNZCV() = default;
1287 
1288   UsedNZCV &operator|=(const UsedNZCV &UsedFlags) {
1289     this->N |= UsedFlags.N;
1290     this->Z |= UsedFlags.Z;
1291     this->C |= UsedFlags.C;
1292     this->V |= UsedFlags.V;
1293     return *this;
1294   }
1295 };
1296 
1297 } // end anonymous namespace
1298 
1299 /// Find a condition code used by the instruction.
1300 /// Returns AArch64CC::Invalid if either the instruction does not use condition
1301 /// codes or we don't optimize CmpInstr in the presence of such instructions.
1302 static AArch64CC::CondCode findCondCodeUsedByInstr(const MachineInstr &Instr) {
1303   switch (Instr.getOpcode()) {
1304   default:
1305     return AArch64CC::Invalid;
1306 
1307   case AArch64::Bcc: {
1308     int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV);
1309     assert(Idx >= 2);
1310     return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 2).getImm());
1311   }
1312 
1313   case AArch64::CSINVWr:
1314   case AArch64::CSINVXr:
1315   case AArch64::CSINCWr:
1316   case AArch64::CSINCXr:
1317   case AArch64::CSELWr:
1318   case AArch64::CSELXr:
1319   case AArch64::CSNEGWr:
1320   case AArch64::CSNEGXr:
1321   case AArch64::FCSELSrrr:
1322   case AArch64::FCSELDrrr: {
1323     int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV);
1324     assert(Idx >= 1);
1325     return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 1).getImm());
1326   }
1327   }
1328 }
1329 
1330 static UsedNZCV getUsedNZCV(AArch64CC::CondCode CC) {
1331   assert(CC != AArch64CC::Invalid);
1332   UsedNZCV UsedFlags;
1333   switch (CC) {
1334   default:
1335     break;
1336 
1337   case AArch64CC::EQ: // Z set
1338   case AArch64CC::NE: // Z clear
1339     UsedFlags.Z = true;
1340     break;
1341 
1342   case AArch64CC::HI: // Z clear and C set
1343   case AArch64CC::LS: // Z set   or  C clear
1344     UsedFlags.Z = true;
1345     LLVM_FALLTHROUGH;
1346   case AArch64CC::HS: // C set
1347   case AArch64CC::LO: // C clear
1348     UsedFlags.C = true;
1349     break;
1350 
1351   case AArch64CC::MI: // N set
1352   case AArch64CC::PL: // N clear
1353     UsedFlags.N = true;
1354     break;
1355 
1356   case AArch64CC::VS: // V set
1357   case AArch64CC::VC: // V clear
1358     UsedFlags.V = true;
1359     break;
1360 
1361   case AArch64CC::GT: // Z clear, N and V the same
1362   case AArch64CC::LE: // Z set,   N and V differ
1363     UsedFlags.Z = true;
1364     LLVM_FALLTHROUGH;
1365   case AArch64CC::GE: // N and V the same
1366   case AArch64CC::LT: // N and V differ
1367     UsedFlags.N = true;
1368     UsedFlags.V = true;
1369     break;
1370   }
1371   return UsedFlags;
1372 }
1373 
1374 static bool isADDSRegImm(unsigned Opcode) {
1375   return Opcode == AArch64::ADDSWri || Opcode == AArch64::ADDSXri;
1376 }
1377 
1378 static bool isSUBSRegImm(unsigned Opcode) {
1379   return Opcode == AArch64::SUBSWri || Opcode == AArch64::SUBSXri;
1380 }
1381 
1382 /// Check if CmpInstr can be substituted by MI.
1383 ///
1384 /// CmpInstr can be substituted:
1385 /// - CmpInstr is either 'ADDS %vreg, 0' or 'SUBS %vreg, 0'
1386 /// - and, MI and CmpInstr are from the same MachineBB
1387 /// - and, condition flags are not alive in successors of the CmpInstr parent
1388 /// - and, if MI opcode is the S form there must be no defs of flags between
1389 ///        MI and CmpInstr
1390 ///        or if MI opcode is not the S form there must be neither defs of flags
1391 ///        nor uses of flags between MI and CmpInstr.
1392 /// - and  C/V flags are not used after CmpInstr
1393 static bool canInstrSubstituteCmpInstr(MachineInstr *MI, MachineInstr *CmpInstr,
1394                                        const TargetRegisterInfo *TRI) {
1395   assert(MI);
1396   assert(sForm(*MI) != AArch64::INSTRUCTION_LIST_END);
1397   assert(CmpInstr);
1398 
1399   const unsigned CmpOpcode = CmpInstr->getOpcode();
1400   if (!isADDSRegImm(CmpOpcode) && !isSUBSRegImm(CmpOpcode))
1401     return false;
1402 
1403   if (MI->getParent() != CmpInstr->getParent())
1404     return false;
1405 
1406   if (areCFlagsAliveInSuccessors(CmpInstr->getParent()))
1407     return false;
1408 
1409   AccessKind AccessToCheck = AK_Write;
1410   if (sForm(*MI) != MI->getOpcode())
1411     AccessToCheck = AK_All;
1412   if (areCFlagsAccessedBetweenInstrs(MI, CmpInstr, TRI, AccessToCheck))
1413     return false;
1414 
1415   UsedNZCV NZCVUsedAfterCmp;
1416   for (auto I = std::next(CmpInstr->getIterator()),
1417             E = CmpInstr->getParent()->instr_end();
1418        I != E; ++I) {
1419     const MachineInstr &Instr = *I;
1420     if (Instr.readsRegister(AArch64::NZCV, TRI)) {
1421       AArch64CC::CondCode CC = findCondCodeUsedByInstr(Instr);
1422       if (CC == AArch64CC::Invalid) // Unsupported conditional instruction
1423         return false;
1424       NZCVUsedAfterCmp |= getUsedNZCV(CC);
1425     }
1426 
1427     if (Instr.modifiesRegister(AArch64::NZCV, TRI))
1428       break;
1429   }
1430 
1431   return !NZCVUsedAfterCmp.C && !NZCVUsedAfterCmp.V;
1432 }
1433 
1434 /// Substitute an instruction comparing to zero with another instruction
1435 /// which produces needed condition flags.
1436 ///
1437 /// Return true on success.
1438 bool AArch64InstrInfo::substituteCmpToZero(
1439     MachineInstr &CmpInstr, unsigned SrcReg,
1440     const MachineRegisterInfo *MRI) const {
1441   assert(MRI);
1442   // Get the unique definition of SrcReg.
1443   MachineInstr *MI = MRI->getUniqueVRegDef(SrcReg);
1444   if (!MI)
1445     return false;
1446 
1447   const TargetRegisterInfo *TRI = &getRegisterInfo();
1448 
1449   unsigned NewOpc = sForm(*MI);
1450   if (NewOpc == AArch64::INSTRUCTION_LIST_END)
1451     return false;
1452 
1453   if (!canInstrSubstituteCmpInstr(MI, &CmpInstr, TRI))
1454     return false;
1455 
1456   // Update the instruction to set NZCV.
1457   MI->setDesc(get(NewOpc));
1458   CmpInstr.eraseFromParent();
1459   bool succeeded = UpdateOperandRegClass(*MI);
1460   (void)succeeded;
1461   assert(succeeded && "Some operands reg class are incompatible!");
1462   MI->addRegisterDefined(AArch64::NZCV, TRI);
1463   return true;
1464 }
1465 
1466 bool AArch64InstrInfo::expandPostRAPseudo(MachineInstr &MI) const {
1467   if (MI.getOpcode() != TargetOpcode::LOAD_STACK_GUARD &&
1468       MI.getOpcode() != AArch64::CATCHRET)
1469     return false;
1470 
1471   MachineBasicBlock &MBB = *MI.getParent();
1472   DebugLoc DL = MI.getDebugLoc();
1473 
1474   if (MI.getOpcode() == AArch64::CATCHRET) {
1475     // Skip to the first instruction before the epilog.
1476     const TargetInstrInfo *TII =
1477       MBB.getParent()->getSubtarget().getInstrInfo();
1478     MachineBasicBlock *TargetMBB = MI.getOperand(0).getMBB();
1479     auto MBBI = MachineBasicBlock::iterator(MI);
1480     MachineBasicBlock::iterator FirstEpilogSEH = std::prev(MBBI);
1481     while (FirstEpilogSEH->getFlag(MachineInstr::FrameDestroy) &&
1482            FirstEpilogSEH != MBB.begin())
1483       FirstEpilogSEH = std::prev(FirstEpilogSEH);
1484     if (FirstEpilogSEH != MBB.begin())
1485       FirstEpilogSEH = std::next(FirstEpilogSEH);
1486     BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADRP))
1487         .addReg(AArch64::X0, RegState::Define)
1488         .addMBB(TargetMBB);
1489     BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADDXri))
1490         .addReg(AArch64::X0, RegState::Define)
1491         .addReg(AArch64::X0)
1492         .addMBB(TargetMBB)
1493         .addImm(0);
1494     return true;
1495   }
1496 
1497   unsigned Reg = MI.getOperand(0).getReg();
1498   const GlobalValue *GV =
1499       cast<GlobalValue>((*MI.memoperands_begin())->getValue());
1500   const TargetMachine &TM = MBB.getParent()->getTarget();
1501   unsigned char OpFlags = Subtarget.ClassifyGlobalReference(GV, TM);
1502   const unsigned char MO_NC = AArch64II::MO_NC;
1503 
1504   if ((OpFlags & AArch64II::MO_GOT) != 0) {
1505     BuildMI(MBB, MI, DL, get(AArch64::LOADgot), Reg)
1506         .addGlobalAddress(GV, 0, OpFlags);
1507     BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
1508         .addReg(Reg, RegState::Kill)
1509         .addImm(0)
1510         .addMemOperand(*MI.memoperands_begin());
1511   } else if (TM.getCodeModel() == CodeModel::Large) {
1512     BuildMI(MBB, MI, DL, get(AArch64::MOVZXi), Reg)
1513         .addGlobalAddress(GV, 0, AArch64II::MO_G0 | MO_NC)
1514         .addImm(0);
1515     BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
1516         .addReg(Reg, RegState::Kill)
1517         .addGlobalAddress(GV, 0, AArch64II::MO_G1 | MO_NC)
1518         .addImm(16);
1519     BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
1520         .addReg(Reg, RegState::Kill)
1521         .addGlobalAddress(GV, 0, AArch64II::MO_G2 | MO_NC)
1522         .addImm(32);
1523     BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
1524         .addReg(Reg, RegState::Kill)
1525         .addGlobalAddress(GV, 0, AArch64II::MO_G3)
1526         .addImm(48);
1527     BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
1528         .addReg(Reg, RegState::Kill)
1529         .addImm(0)
1530         .addMemOperand(*MI.memoperands_begin());
1531   } else if (TM.getCodeModel() == CodeModel::Tiny) {
1532     BuildMI(MBB, MI, DL, get(AArch64::ADR), Reg)
1533         .addGlobalAddress(GV, 0, OpFlags);
1534   } else {
1535     BuildMI(MBB, MI, DL, get(AArch64::ADRP), Reg)
1536         .addGlobalAddress(GV, 0, OpFlags | AArch64II::MO_PAGE);
1537     unsigned char LoFlags = OpFlags | AArch64II::MO_PAGEOFF | MO_NC;
1538     BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
1539         .addReg(Reg, RegState::Kill)
1540         .addGlobalAddress(GV, 0, LoFlags)
1541         .addMemOperand(*MI.memoperands_begin());
1542   }
1543 
1544   MBB.erase(MI);
1545 
1546   return true;
1547 }
1548 
1549 // Return true if this instruction simply sets its single destination register
1550 // to zero. This is equivalent to a register rename of the zero-register.
1551 bool AArch64InstrInfo::isGPRZero(const MachineInstr &MI) {
1552   switch (MI.getOpcode()) {
1553   default:
1554     break;
1555   case AArch64::MOVZWi:
1556   case AArch64::MOVZXi: // movz Rd, #0 (LSL #0)
1557     if (MI.getOperand(1).isImm() && MI.getOperand(1).getImm() == 0) {
1558       assert(MI.getDesc().getNumOperands() == 3 &&
1559              MI.getOperand(2).getImm() == 0 && "invalid MOVZi operands");
1560       return true;
1561     }
1562     break;
1563   case AArch64::ANDWri: // and Rd, Rzr, #imm
1564     return MI.getOperand(1).getReg() == AArch64::WZR;
1565   case AArch64::ANDXri:
1566     return MI.getOperand(1).getReg() == AArch64::XZR;
1567   case TargetOpcode::COPY:
1568     return MI.getOperand(1).getReg() == AArch64::WZR;
1569   }
1570   return false;
1571 }
1572 
1573 // Return true if this instruction simply renames a general register without
1574 // modifying bits.
1575 bool AArch64InstrInfo::isGPRCopy(const MachineInstr &MI) {
1576   switch (MI.getOpcode()) {
1577   default:
1578     break;
1579   case TargetOpcode::COPY: {
1580     // GPR32 copies will by lowered to ORRXrs
1581     unsigned DstReg = MI.getOperand(0).getReg();
1582     return (AArch64::GPR32RegClass.contains(DstReg) ||
1583             AArch64::GPR64RegClass.contains(DstReg));
1584   }
1585   case AArch64::ORRXrs: // orr Xd, Xzr, Xm (LSL #0)
1586     if (MI.getOperand(1).getReg() == AArch64::XZR) {
1587       assert(MI.getDesc().getNumOperands() == 4 &&
1588              MI.getOperand(3).getImm() == 0 && "invalid ORRrs operands");
1589       return true;
1590     }
1591     break;
1592   case AArch64::ADDXri: // add Xd, Xn, #0 (LSL #0)
1593     if (MI.getOperand(2).getImm() == 0) {
1594       assert(MI.getDesc().getNumOperands() == 4 &&
1595              MI.getOperand(3).getImm() == 0 && "invalid ADDXri operands");
1596       return true;
1597     }
1598     break;
1599   }
1600   return false;
1601 }
1602 
1603 // Return true if this instruction simply renames a general register without
1604 // modifying bits.
1605 bool AArch64InstrInfo::isFPRCopy(const MachineInstr &MI) {
1606   switch (MI.getOpcode()) {
1607   default:
1608     break;
1609   case TargetOpcode::COPY: {
1610     // FPR64 copies will by lowered to ORR.16b
1611     unsigned DstReg = MI.getOperand(0).getReg();
1612     return (AArch64::FPR64RegClass.contains(DstReg) ||
1613             AArch64::FPR128RegClass.contains(DstReg));
1614   }
1615   case AArch64::ORRv16i8:
1616     if (MI.getOperand(1).getReg() == MI.getOperand(2).getReg()) {
1617       assert(MI.getDesc().getNumOperands() == 3 && MI.getOperand(0).isReg() &&
1618              "invalid ORRv16i8 operands");
1619       return true;
1620     }
1621     break;
1622   }
1623   return false;
1624 }
1625 
1626 unsigned AArch64InstrInfo::isLoadFromStackSlot(const MachineInstr &MI,
1627                                                int &FrameIndex) const {
1628   switch (MI.getOpcode()) {
1629   default:
1630     break;
1631   case AArch64::LDRWui:
1632   case AArch64::LDRXui:
1633   case AArch64::LDRBui:
1634   case AArch64::LDRHui:
1635   case AArch64::LDRSui:
1636   case AArch64::LDRDui:
1637   case AArch64::LDRQui:
1638     if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() &&
1639         MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) {
1640       FrameIndex = MI.getOperand(1).getIndex();
1641       return MI.getOperand(0).getReg();
1642     }
1643     break;
1644   }
1645 
1646   return 0;
1647 }
1648 
1649 unsigned AArch64InstrInfo::isStoreToStackSlot(const MachineInstr &MI,
1650                                               int &FrameIndex) const {
1651   switch (MI.getOpcode()) {
1652   default:
1653     break;
1654   case AArch64::STRWui:
1655   case AArch64::STRXui:
1656   case AArch64::STRBui:
1657   case AArch64::STRHui:
1658   case AArch64::STRSui:
1659   case AArch64::STRDui:
1660   case AArch64::STRQui:
1661     if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() &&
1662         MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) {
1663       FrameIndex = MI.getOperand(1).getIndex();
1664       return MI.getOperand(0).getReg();
1665     }
1666     break;
1667   }
1668   return 0;
1669 }
1670 
1671 /// Check all MachineMemOperands for a hint to suppress pairing.
1672 bool AArch64InstrInfo::isLdStPairSuppressed(const MachineInstr &MI) {
1673   return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) {
1674     return MMO->getFlags() & MOSuppressPair;
1675   });
1676 }
1677 
1678 /// Set a flag on the first MachineMemOperand to suppress pairing.
1679 void AArch64InstrInfo::suppressLdStPair(MachineInstr &MI) {
1680   if (MI.memoperands_empty())
1681     return;
1682   (*MI.memoperands_begin())->setFlags(MOSuppressPair);
1683 }
1684 
1685 /// Check all MachineMemOperands for a hint that the load/store is strided.
1686 bool AArch64InstrInfo::isStridedAccess(const MachineInstr &MI) {
1687   return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) {
1688     return MMO->getFlags() & MOStridedAccess;
1689   });
1690 }
1691 
1692 bool AArch64InstrInfo::isUnscaledLdSt(unsigned Opc) {
1693   switch (Opc) {
1694   default:
1695     return false;
1696   case AArch64::STURSi:
1697   case AArch64::STURDi:
1698   case AArch64::STURQi:
1699   case AArch64::STURBBi:
1700   case AArch64::STURHHi:
1701   case AArch64::STURWi:
1702   case AArch64::STURXi:
1703   case AArch64::LDURSi:
1704   case AArch64::LDURDi:
1705   case AArch64::LDURQi:
1706   case AArch64::LDURWi:
1707   case AArch64::LDURXi:
1708   case AArch64::LDURSWi:
1709   case AArch64::LDURHHi:
1710   case AArch64::LDURBBi:
1711   case AArch64::LDURSBWi:
1712   case AArch64::LDURSHWi:
1713     return true;
1714   }
1715 }
1716 
1717 Optional<unsigned> AArch64InstrInfo::getUnscaledLdSt(unsigned Opc) {
1718   switch (Opc) {
1719   default: return {};
1720   case AArch64::PRFMui: return AArch64::PRFUMi;
1721   case AArch64::LDRXui: return AArch64::LDURXi;
1722   case AArch64::LDRWui: return AArch64::LDURWi;
1723   case AArch64::LDRBui: return AArch64::LDURBi;
1724   case AArch64::LDRHui: return AArch64::LDURHi;
1725   case AArch64::LDRSui: return AArch64::LDURSi;
1726   case AArch64::LDRDui: return AArch64::LDURDi;
1727   case AArch64::LDRQui: return AArch64::LDURQi;
1728   case AArch64::LDRBBui: return AArch64::LDURBBi;
1729   case AArch64::LDRHHui: return AArch64::LDURHHi;
1730   case AArch64::LDRSBXui: return AArch64::LDURSBXi;
1731   case AArch64::LDRSBWui: return AArch64::LDURSBWi;
1732   case AArch64::LDRSHXui: return AArch64::LDURSHXi;
1733   case AArch64::LDRSHWui: return AArch64::LDURSHWi;
1734   case AArch64::LDRSWui: return AArch64::LDURSWi;
1735   case AArch64::STRXui: return AArch64::STURXi;
1736   case AArch64::STRWui: return AArch64::STURWi;
1737   case AArch64::STRBui: return AArch64::STURBi;
1738   case AArch64::STRHui: return AArch64::STURHi;
1739   case AArch64::STRSui: return AArch64::STURSi;
1740   case AArch64::STRDui: return AArch64::STURDi;
1741   case AArch64::STRQui: return AArch64::STURQi;
1742   case AArch64::STRBBui: return AArch64::STURBBi;
1743   case AArch64::STRHHui: return AArch64::STURHHi;
1744   }
1745 }
1746 
1747 unsigned AArch64InstrInfo::getLoadStoreImmIdx(unsigned Opc) {
1748   switch (Opc) {
1749   default:
1750     return 2;
1751   case AArch64::LDPXi:
1752   case AArch64::LDPDi:
1753   case AArch64::STPXi:
1754   case AArch64::STPDi:
1755   case AArch64::LDNPXi:
1756   case AArch64::LDNPDi:
1757   case AArch64::STNPXi:
1758   case AArch64::STNPDi:
1759   case AArch64::LDPQi:
1760   case AArch64::STPQi:
1761   case AArch64::LDNPQi:
1762   case AArch64::STNPQi:
1763   case AArch64::LDPWi:
1764   case AArch64::LDPSi:
1765   case AArch64::STPWi:
1766   case AArch64::STPSi:
1767   case AArch64::LDNPWi:
1768   case AArch64::LDNPSi:
1769   case AArch64::STNPWi:
1770   case AArch64::STNPSi:
1771   case AArch64::LDG:
1772     return 3;
1773   case AArch64::ADDG:
1774   case AArch64::STGOffset:
1775     return 2;
1776   }
1777 }
1778 
1779 bool AArch64InstrInfo::isPairableLdStInst(const MachineInstr &MI) {
1780   switch (MI.getOpcode()) {
1781   default:
1782     return false;
1783   // Scaled instructions.
1784   case AArch64::STRSui:
1785   case AArch64::STRDui:
1786   case AArch64::STRQui:
1787   case AArch64::STRXui:
1788   case AArch64::STRWui:
1789   case AArch64::LDRSui:
1790   case AArch64::LDRDui:
1791   case AArch64::LDRQui:
1792   case AArch64::LDRXui:
1793   case AArch64::LDRWui:
1794   case AArch64::LDRSWui:
1795   // Unscaled instructions.
1796   case AArch64::STURSi:
1797   case AArch64::STURDi:
1798   case AArch64::STURQi:
1799   case AArch64::STURWi:
1800   case AArch64::STURXi:
1801   case AArch64::LDURSi:
1802   case AArch64::LDURDi:
1803   case AArch64::LDURQi:
1804   case AArch64::LDURWi:
1805   case AArch64::LDURXi:
1806   case AArch64::LDURSWi:
1807     return true;
1808   }
1809 }
1810 
1811 unsigned AArch64InstrInfo::convertToFlagSettingOpc(unsigned Opc,
1812                                                    bool &Is64Bit) {
1813   switch (Opc) {
1814   default:
1815     llvm_unreachable("Opcode has no flag setting equivalent!");
1816   // 32-bit cases:
1817   case AArch64::ADDWri:
1818     Is64Bit = false;
1819     return AArch64::ADDSWri;
1820   case AArch64::ADDWrr:
1821     Is64Bit = false;
1822     return AArch64::ADDSWrr;
1823   case AArch64::ADDWrs:
1824     Is64Bit = false;
1825     return AArch64::ADDSWrs;
1826   case AArch64::ADDWrx:
1827     Is64Bit = false;
1828     return AArch64::ADDSWrx;
1829   case AArch64::ANDWri:
1830     Is64Bit = false;
1831     return AArch64::ANDSWri;
1832   case AArch64::ANDWrr:
1833     Is64Bit = false;
1834     return AArch64::ANDSWrr;
1835   case AArch64::ANDWrs:
1836     Is64Bit = false;
1837     return AArch64::ANDSWrs;
1838   case AArch64::BICWrr:
1839     Is64Bit = false;
1840     return AArch64::BICSWrr;
1841   case AArch64::BICWrs:
1842     Is64Bit = false;
1843     return AArch64::BICSWrs;
1844   case AArch64::SUBWri:
1845     Is64Bit = false;
1846     return AArch64::SUBSWri;
1847   case AArch64::SUBWrr:
1848     Is64Bit = false;
1849     return AArch64::SUBSWrr;
1850   case AArch64::SUBWrs:
1851     Is64Bit = false;
1852     return AArch64::SUBSWrs;
1853   case AArch64::SUBWrx:
1854     Is64Bit = false;
1855     return AArch64::SUBSWrx;
1856   // 64-bit cases:
1857   case AArch64::ADDXri:
1858     Is64Bit = true;
1859     return AArch64::ADDSXri;
1860   case AArch64::ADDXrr:
1861     Is64Bit = true;
1862     return AArch64::ADDSXrr;
1863   case AArch64::ADDXrs:
1864     Is64Bit = true;
1865     return AArch64::ADDSXrs;
1866   case AArch64::ADDXrx:
1867     Is64Bit = true;
1868     return AArch64::ADDSXrx;
1869   case AArch64::ANDXri:
1870     Is64Bit = true;
1871     return AArch64::ANDSXri;
1872   case AArch64::ANDXrr:
1873     Is64Bit = true;
1874     return AArch64::ANDSXrr;
1875   case AArch64::ANDXrs:
1876     Is64Bit = true;
1877     return AArch64::ANDSXrs;
1878   case AArch64::BICXrr:
1879     Is64Bit = true;
1880     return AArch64::BICSXrr;
1881   case AArch64::BICXrs:
1882     Is64Bit = true;
1883     return AArch64::BICSXrs;
1884   case AArch64::SUBXri:
1885     Is64Bit = true;
1886     return AArch64::SUBSXri;
1887   case AArch64::SUBXrr:
1888     Is64Bit = true;
1889     return AArch64::SUBSXrr;
1890   case AArch64::SUBXrs:
1891     Is64Bit = true;
1892     return AArch64::SUBSXrs;
1893   case AArch64::SUBXrx:
1894     Is64Bit = true;
1895     return AArch64::SUBSXrx;
1896   }
1897 }
1898 
1899 // Is this a candidate for ld/st merging or pairing?  For example, we don't
1900 // touch volatiles or load/stores that have a hint to avoid pair formation.
1901 bool AArch64InstrInfo::isCandidateToMergeOrPair(const MachineInstr &MI) const {
1902   // If this is a volatile load/store, don't mess with it.
1903   if (MI.hasOrderedMemoryRef())
1904     return false;
1905 
1906   // Make sure this is a reg/fi+imm (as opposed to an address reloc).
1907   assert((MI.getOperand(1).isReg() || MI.getOperand(1).isFI()) &&
1908          "Expected a reg or frame index operand.");
1909   if (!MI.getOperand(2).isImm())
1910     return false;
1911 
1912   // Can't merge/pair if the instruction modifies the base register.
1913   // e.g., ldr x0, [x0]
1914   // This case will never occur with an FI base.
1915   if (MI.getOperand(1).isReg()) {
1916     unsigned BaseReg = MI.getOperand(1).getReg();
1917     const TargetRegisterInfo *TRI = &getRegisterInfo();
1918     if (MI.modifiesRegister(BaseReg, TRI))
1919       return false;
1920   }
1921 
1922   // Check if this load/store has a hint to avoid pair formation.
1923   // MachineMemOperands hints are set by the AArch64StorePairSuppress pass.
1924   if (isLdStPairSuppressed(MI))
1925     return false;
1926 
1927   // On some CPUs quad load/store pairs are slower than two single load/stores.
1928   if (Subtarget.isPaired128Slow()) {
1929     switch (MI.getOpcode()) {
1930     default:
1931       break;
1932     case AArch64::LDURQi:
1933     case AArch64::STURQi:
1934     case AArch64::LDRQui:
1935     case AArch64::STRQui:
1936       return false;
1937     }
1938   }
1939 
1940   return true;
1941 }
1942 
1943 bool AArch64InstrInfo::getMemOperandWithOffset(const MachineInstr &LdSt,
1944                                           const MachineOperand *&BaseOp,
1945                                           int64_t &Offset,
1946                                           const TargetRegisterInfo *TRI) const {
1947   unsigned Width;
1948   return getMemOperandWithOffsetWidth(LdSt, BaseOp, Offset, Width, TRI);
1949 }
1950 
1951 bool AArch64InstrInfo::getMemOperandWithOffsetWidth(
1952     const MachineInstr &LdSt, const MachineOperand *&BaseOp, int64_t &Offset,
1953     unsigned &Width, const TargetRegisterInfo *TRI) const {
1954   assert(LdSt.mayLoadOrStore() && "Expected a memory operation.");
1955   // Handle only loads/stores with base register followed by immediate offset.
1956   if (LdSt.getNumExplicitOperands() == 3) {
1957     // Non-paired instruction (e.g., ldr x1, [x0, #8]).
1958     if ((!LdSt.getOperand(1).isReg() && !LdSt.getOperand(1).isFI()) ||
1959         !LdSt.getOperand(2).isImm())
1960       return false;
1961   } else if (LdSt.getNumExplicitOperands() == 4) {
1962     // Paired instruction (e.g., ldp x1, x2, [x0, #8]).
1963     if (!LdSt.getOperand(1).isReg() ||
1964         (!LdSt.getOperand(2).isReg() && !LdSt.getOperand(2).isFI()) ||
1965         !LdSt.getOperand(3).isImm())
1966       return false;
1967   } else
1968     return false;
1969 
1970   // Get the scaling factor for the instruction and set the width for the
1971   // instruction.
1972   unsigned Scale = 0;
1973   int64_t Dummy1, Dummy2;
1974 
1975   // If this returns false, then it's an instruction we don't want to handle.
1976   if (!getMemOpInfo(LdSt.getOpcode(), Scale, Width, Dummy1, Dummy2))
1977     return false;
1978 
1979   // Compute the offset. Offset is calculated as the immediate operand
1980   // multiplied by the scaling factor. Unscaled instructions have scaling factor
1981   // set to 1.
1982   if (LdSt.getNumExplicitOperands() == 3) {
1983     BaseOp = &LdSt.getOperand(1);
1984     Offset = LdSt.getOperand(2).getImm() * Scale;
1985   } else {
1986     assert(LdSt.getNumExplicitOperands() == 4 && "invalid number of operands");
1987     BaseOp = &LdSt.getOperand(2);
1988     Offset = LdSt.getOperand(3).getImm() * Scale;
1989   }
1990 
1991   assert((BaseOp->isReg() || BaseOp->isFI()) &&
1992          "getMemOperandWithOffset only supports base "
1993          "operands of type register or frame index.");
1994 
1995   return true;
1996 }
1997 
1998 MachineOperand &
1999 AArch64InstrInfo::getMemOpBaseRegImmOfsOffsetOperand(MachineInstr &LdSt) const {
2000   assert(LdSt.mayLoadOrStore() && "Expected a memory operation.");
2001   MachineOperand &OfsOp = LdSt.getOperand(LdSt.getNumExplicitOperands() - 1);
2002   assert(OfsOp.isImm() && "Offset operand wasn't immediate.");
2003   return OfsOp;
2004 }
2005 
2006 bool AArch64InstrInfo::getMemOpInfo(unsigned Opcode, unsigned &Scale,
2007                                     unsigned &Width, int64_t &MinOffset,
2008                                     int64_t &MaxOffset) {
2009   switch (Opcode) {
2010   // Not a memory operation or something we want to handle.
2011   default:
2012     Scale = Width = 0;
2013     MinOffset = MaxOffset = 0;
2014     return false;
2015   case AArch64::STRWpost:
2016   case AArch64::LDRWpost:
2017     Width = 32;
2018     Scale = 4;
2019     MinOffset = -256;
2020     MaxOffset = 255;
2021     break;
2022   case AArch64::LDURQi:
2023   case AArch64::STURQi:
2024     Width = 16;
2025     Scale = 1;
2026     MinOffset = -256;
2027     MaxOffset = 255;
2028     break;
2029   case AArch64::PRFUMi:
2030   case AArch64::LDURXi:
2031   case AArch64::LDURDi:
2032   case AArch64::STURXi:
2033   case AArch64::STURDi:
2034     Width = 8;
2035     Scale = 1;
2036     MinOffset = -256;
2037     MaxOffset = 255;
2038     break;
2039   case AArch64::LDURWi:
2040   case AArch64::LDURSi:
2041   case AArch64::LDURSWi:
2042   case AArch64::STURWi:
2043   case AArch64::STURSi:
2044     Width = 4;
2045     Scale = 1;
2046     MinOffset = -256;
2047     MaxOffset = 255;
2048     break;
2049   case AArch64::LDURHi:
2050   case AArch64::LDURHHi:
2051   case AArch64::LDURSHXi:
2052   case AArch64::LDURSHWi:
2053   case AArch64::STURHi:
2054   case AArch64::STURHHi:
2055     Width = 2;
2056     Scale = 1;
2057     MinOffset = -256;
2058     MaxOffset = 255;
2059     break;
2060   case AArch64::LDURBi:
2061   case AArch64::LDURBBi:
2062   case AArch64::LDURSBXi:
2063   case AArch64::LDURSBWi:
2064   case AArch64::STURBi:
2065   case AArch64::STURBBi:
2066     Width = 1;
2067     Scale = 1;
2068     MinOffset = -256;
2069     MaxOffset = 255;
2070     break;
2071   case AArch64::LDPQi:
2072   case AArch64::LDNPQi:
2073   case AArch64::STPQi:
2074   case AArch64::STNPQi:
2075     Scale = 16;
2076     Width = 32;
2077     MinOffset = -64;
2078     MaxOffset = 63;
2079     break;
2080   case AArch64::LDRQui:
2081   case AArch64::STRQui:
2082     Scale = Width = 16;
2083     MinOffset = 0;
2084     MaxOffset = 4095;
2085     break;
2086   case AArch64::LDPXi:
2087   case AArch64::LDPDi:
2088   case AArch64::LDNPXi:
2089   case AArch64::LDNPDi:
2090   case AArch64::STPXi:
2091   case AArch64::STPDi:
2092   case AArch64::STNPXi:
2093   case AArch64::STNPDi:
2094     Scale = 8;
2095     Width = 16;
2096     MinOffset = -64;
2097     MaxOffset = 63;
2098     break;
2099   case AArch64::PRFMui:
2100   case AArch64::LDRXui:
2101   case AArch64::LDRDui:
2102   case AArch64::STRXui:
2103   case AArch64::STRDui:
2104     Scale = Width = 8;
2105     MinOffset = 0;
2106     MaxOffset = 4095;
2107     break;
2108   case AArch64::LDPWi:
2109   case AArch64::LDPSi:
2110   case AArch64::LDNPWi:
2111   case AArch64::LDNPSi:
2112   case AArch64::STPWi:
2113   case AArch64::STPSi:
2114   case AArch64::STNPWi:
2115   case AArch64::STNPSi:
2116     Scale = 4;
2117     Width = 8;
2118     MinOffset = -64;
2119     MaxOffset = 63;
2120     break;
2121   case AArch64::LDRWui:
2122   case AArch64::LDRSui:
2123   case AArch64::LDRSWui:
2124   case AArch64::STRWui:
2125   case AArch64::STRSui:
2126     Scale = Width = 4;
2127     MinOffset = 0;
2128     MaxOffset = 4095;
2129     break;
2130   case AArch64::LDRHui:
2131   case AArch64::LDRHHui:
2132   case AArch64::LDRSHWui:
2133   case AArch64::LDRSHXui:
2134   case AArch64::STRHui:
2135   case AArch64::STRHHui:
2136     Scale = Width = 2;
2137     MinOffset = 0;
2138     MaxOffset = 4095;
2139     break;
2140   case AArch64::LDRBui:
2141   case AArch64::LDRBBui:
2142   case AArch64::LDRSBWui:
2143   case AArch64::LDRSBXui:
2144   case AArch64::STRBui:
2145   case AArch64::STRBBui:
2146     Scale = Width = 1;
2147     MinOffset = 0;
2148     MaxOffset = 4095;
2149     break;
2150   case AArch64::ADDG:
2151     Scale = 16;
2152     Width = 0;
2153     MinOffset = 0;
2154     MaxOffset = 63;
2155     break;
2156   case AArch64::LDG:
2157   case AArch64::STGOffset:
2158     Scale = Width = 16;
2159     MinOffset = -256;
2160     MaxOffset = 255;
2161     break;
2162   }
2163 
2164   return true;
2165 }
2166 
2167 static unsigned getOffsetStride(unsigned Opc) {
2168   switch (Opc) {
2169   default:
2170     return 0;
2171   case AArch64::LDURQi:
2172   case AArch64::STURQi:
2173     return 16;
2174   case AArch64::LDURXi:
2175   case AArch64::LDURDi:
2176   case AArch64::STURXi:
2177   case AArch64::STURDi:
2178     return 8;
2179   case AArch64::LDURWi:
2180   case AArch64::LDURSi:
2181   case AArch64::LDURSWi:
2182   case AArch64::STURWi:
2183   case AArch64::STURSi:
2184     return 4;
2185   }
2186 }
2187 
2188 // Scale the unscaled offsets.  Returns false if the unscaled offset can't be
2189 // scaled.
2190 static bool scaleOffset(unsigned Opc, int64_t &Offset) {
2191   unsigned OffsetStride = getOffsetStride(Opc);
2192   if (OffsetStride == 0)
2193     return false;
2194   // If the byte-offset isn't a multiple of the stride, we can't scale this
2195   // offset.
2196   if (Offset % OffsetStride != 0)
2197     return false;
2198 
2199   // Convert the byte-offset used by unscaled into an "element" offset used
2200   // by the scaled pair load/store instructions.
2201   Offset /= OffsetStride;
2202   return true;
2203 }
2204 
2205 // Unscale the scaled offsets. Returns false if the scaled offset can't be
2206 // unscaled.
2207 static bool unscaleOffset(unsigned Opc, int64_t &Offset) {
2208   unsigned OffsetStride = getOffsetStride(Opc);
2209   if (OffsetStride == 0)
2210     return false;
2211 
2212   // Convert the "element" offset used by scaled pair load/store instructions
2213   // into the byte-offset used by unscaled.
2214   Offset *= OffsetStride;
2215   return true;
2216 }
2217 
2218 static bool canPairLdStOpc(unsigned FirstOpc, unsigned SecondOpc) {
2219   if (FirstOpc == SecondOpc)
2220     return true;
2221   // We can also pair sign-ext and zero-ext instructions.
2222   switch (FirstOpc) {
2223   default:
2224     return false;
2225   case AArch64::LDRWui:
2226   case AArch64::LDURWi:
2227     return SecondOpc == AArch64::LDRSWui || SecondOpc == AArch64::LDURSWi;
2228   case AArch64::LDRSWui:
2229   case AArch64::LDURSWi:
2230     return SecondOpc == AArch64::LDRWui || SecondOpc == AArch64::LDURWi;
2231   }
2232   // These instructions can't be paired based on their opcodes.
2233   return false;
2234 }
2235 
2236 static bool shouldClusterFI(const MachineFrameInfo &MFI, int FI1,
2237                             int64_t Offset1, unsigned Opcode1, int FI2,
2238                             int64_t Offset2, unsigned Opcode2) {
2239   // Accesses through fixed stack object frame indices may access a different
2240   // fixed stack slot. Check that the object offsets + offsets match.
2241   if (MFI.isFixedObjectIndex(FI1) && MFI.isFixedObjectIndex(FI2)) {
2242     int64_t ObjectOffset1 = MFI.getObjectOffset(FI1);
2243     int64_t ObjectOffset2 = MFI.getObjectOffset(FI2);
2244     assert(ObjectOffset1 <= ObjectOffset2 && "Object offsets are not ordered.");
2245     // Get the byte-offset from the object offset.
2246     if (!unscaleOffset(Opcode1, Offset1) || !unscaleOffset(Opcode2, Offset2))
2247       return false;
2248     ObjectOffset1 += Offset1;
2249     ObjectOffset2 += Offset2;
2250     // Get the "element" index in the object.
2251     if (!scaleOffset(Opcode1, ObjectOffset1) ||
2252         !scaleOffset(Opcode2, ObjectOffset2))
2253       return false;
2254     return ObjectOffset1 + 1 == ObjectOffset2;
2255   }
2256 
2257   return FI1 == FI2;
2258 }
2259 
2260 /// Detect opportunities for ldp/stp formation.
2261 ///
2262 /// Only called for LdSt for which getMemOperandWithOffset returns true.
2263 bool AArch64InstrInfo::shouldClusterMemOps(const MachineOperand &BaseOp1,
2264                                            const MachineOperand &BaseOp2,
2265                                            unsigned NumLoads) const {
2266   const MachineInstr &FirstLdSt = *BaseOp1.getParent();
2267   const MachineInstr &SecondLdSt = *BaseOp2.getParent();
2268   if (BaseOp1.getType() != BaseOp2.getType())
2269     return false;
2270 
2271   assert((BaseOp1.isReg() || BaseOp1.isFI()) &&
2272          "Only base registers and frame indices are supported.");
2273 
2274   // Check for both base regs and base FI.
2275   if (BaseOp1.isReg() && BaseOp1.getReg() != BaseOp2.getReg())
2276     return false;
2277 
2278   // Only cluster up to a single pair.
2279   if (NumLoads > 1)
2280     return false;
2281 
2282   if (!isPairableLdStInst(FirstLdSt) || !isPairableLdStInst(SecondLdSt))
2283     return false;
2284 
2285   // Can we pair these instructions based on their opcodes?
2286   unsigned FirstOpc = FirstLdSt.getOpcode();
2287   unsigned SecondOpc = SecondLdSt.getOpcode();
2288   if (!canPairLdStOpc(FirstOpc, SecondOpc))
2289     return false;
2290 
2291   // Can't merge volatiles or load/stores that have a hint to avoid pair
2292   // formation, for example.
2293   if (!isCandidateToMergeOrPair(FirstLdSt) ||
2294       !isCandidateToMergeOrPair(SecondLdSt))
2295     return false;
2296 
2297   // isCandidateToMergeOrPair guarantees that operand 2 is an immediate.
2298   int64_t Offset1 = FirstLdSt.getOperand(2).getImm();
2299   if (isUnscaledLdSt(FirstOpc) && !scaleOffset(FirstOpc, Offset1))
2300     return false;
2301 
2302   int64_t Offset2 = SecondLdSt.getOperand(2).getImm();
2303   if (isUnscaledLdSt(SecondOpc) && !scaleOffset(SecondOpc, Offset2))
2304     return false;
2305 
2306   // Pairwise instructions have a 7-bit signed offset field.
2307   if (Offset1 > 63 || Offset1 < -64)
2308     return false;
2309 
2310   // The caller should already have ordered First/SecondLdSt by offset.
2311   // Note: except for non-equal frame index bases
2312   if (BaseOp1.isFI()) {
2313     assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 >= Offset2) &&
2314            "Caller should have ordered offsets.");
2315 
2316     const MachineFrameInfo &MFI =
2317         FirstLdSt.getParent()->getParent()->getFrameInfo();
2318     return shouldClusterFI(MFI, BaseOp1.getIndex(), Offset1, FirstOpc,
2319                            BaseOp2.getIndex(), Offset2, SecondOpc);
2320   }
2321 
2322   assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 <= Offset2) &&
2323          "Caller should have ordered offsets.");
2324 
2325   return Offset1 + 1 == Offset2;
2326 }
2327 
2328 static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB,
2329                                             unsigned Reg, unsigned SubIdx,
2330                                             unsigned State,
2331                                             const TargetRegisterInfo *TRI) {
2332   if (!SubIdx)
2333     return MIB.addReg(Reg, State);
2334 
2335   if (TargetRegisterInfo::isPhysicalRegister(Reg))
2336     return MIB.addReg(TRI->getSubReg(Reg, SubIdx), State);
2337   return MIB.addReg(Reg, State, SubIdx);
2338 }
2339 
2340 static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg,
2341                                         unsigned NumRegs) {
2342   // We really want the positive remainder mod 32 here, that happens to be
2343   // easily obtainable with a mask.
2344   return ((DestReg - SrcReg) & 0x1f) < NumRegs;
2345 }
2346 
2347 void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
2348                                         MachineBasicBlock::iterator I,
2349                                         const DebugLoc &DL, unsigned DestReg,
2350                                         unsigned SrcReg, bool KillSrc,
2351                                         unsigned Opcode,
2352                                         ArrayRef<unsigned> Indices) const {
2353   assert(Subtarget.hasNEON() && "Unexpected register copy without NEON");
2354   const TargetRegisterInfo *TRI = &getRegisterInfo();
2355   uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
2356   uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
2357   unsigned NumRegs = Indices.size();
2358 
2359   int SubReg = 0, End = NumRegs, Incr = 1;
2360   if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs)) {
2361     SubReg = NumRegs - 1;
2362     End = -1;
2363     Incr = -1;
2364   }
2365 
2366   for (; SubReg != End; SubReg += Incr) {
2367     const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
2368     AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
2369     AddSubReg(MIB, SrcReg, Indices[SubReg], 0, TRI);
2370     AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
2371   }
2372 }
2373 
2374 void AArch64InstrInfo::copyGPRRegTuple(MachineBasicBlock &MBB,
2375                                        MachineBasicBlock::iterator I,
2376                                        DebugLoc DL, unsigned DestReg,
2377                                        unsigned SrcReg, bool KillSrc,
2378                                        unsigned Opcode, unsigned ZeroReg,
2379                                        llvm::ArrayRef<unsigned> Indices) const {
2380   const TargetRegisterInfo *TRI = &getRegisterInfo();
2381   unsigned NumRegs = Indices.size();
2382 
2383 #ifndef NDEBUG
2384   uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
2385   uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
2386   assert(DestEncoding % NumRegs == 0 && SrcEncoding % NumRegs == 0 &&
2387          "GPR reg sequences should not be able to overlap");
2388 #endif
2389 
2390   for (unsigned SubReg = 0; SubReg != NumRegs; ++SubReg) {
2391     const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
2392     AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
2393     MIB.addReg(ZeroReg);
2394     AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
2395     MIB.addImm(0);
2396   }
2397 }
2398 
2399 void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
2400                                    MachineBasicBlock::iterator I,
2401                                    const DebugLoc &DL, unsigned DestReg,
2402                                    unsigned SrcReg, bool KillSrc) const {
2403   if (AArch64::GPR32spRegClass.contains(DestReg) &&
2404       (AArch64::GPR32spRegClass.contains(SrcReg) || SrcReg == AArch64::WZR)) {
2405     const TargetRegisterInfo *TRI = &getRegisterInfo();
2406 
2407     if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) {
2408       // If either operand is WSP, expand to ADD #0.
2409       if (Subtarget.hasZeroCycleRegMove()) {
2410         // Cyclone recognizes "ADD Xd, Xn, #0" as a zero-cycle register move.
2411         unsigned DestRegX = TRI->getMatchingSuperReg(DestReg, AArch64::sub_32,
2412                                                      &AArch64::GPR64spRegClass);
2413         unsigned SrcRegX = TRI->getMatchingSuperReg(SrcReg, AArch64::sub_32,
2414                                                     &AArch64::GPR64spRegClass);
2415         // This instruction is reading and writing X registers.  This may upset
2416         // the register scavenger and machine verifier, so we need to indicate
2417         // that we are reading an undefined value from SrcRegX, but a proper
2418         // value from SrcReg.
2419         BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestRegX)
2420             .addReg(SrcRegX, RegState::Undef)
2421             .addImm(0)
2422             .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0))
2423             .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
2424       } else {
2425         BuildMI(MBB, I, DL, get(AArch64::ADDWri), DestReg)
2426             .addReg(SrcReg, getKillRegState(KillSrc))
2427             .addImm(0)
2428             .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2429       }
2430     } else if (SrcReg == AArch64::WZR && Subtarget.hasZeroCycleZeroingGP()) {
2431       BuildMI(MBB, I, DL, get(AArch64::MOVZWi), DestReg)
2432           .addImm(0)
2433           .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2434     } else {
2435       if (Subtarget.hasZeroCycleRegMove()) {
2436         // Cyclone recognizes "ORR Xd, XZR, Xm" as a zero-cycle register move.
2437         unsigned DestRegX = TRI->getMatchingSuperReg(DestReg, AArch64::sub_32,
2438                                                      &AArch64::GPR64spRegClass);
2439         unsigned SrcRegX = TRI->getMatchingSuperReg(SrcReg, AArch64::sub_32,
2440                                                     &AArch64::GPR64spRegClass);
2441         // This instruction is reading and writing X registers.  This may upset
2442         // the register scavenger and machine verifier, so we need to indicate
2443         // that we are reading an undefined value from SrcRegX, but a proper
2444         // value from SrcReg.
2445         BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestRegX)
2446             .addReg(AArch64::XZR)
2447             .addReg(SrcRegX, RegState::Undef)
2448             .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
2449       } else {
2450         // Otherwise, expand to ORR WZR.
2451         BuildMI(MBB, I, DL, get(AArch64::ORRWrr), DestReg)
2452             .addReg(AArch64::WZR)
2453             .addReg(SrcReg, getKillRegState(KillSrc));
2454       }
2455     }
2456     return;
2457   }
2458 
2459   if (AArch64::GPR64spRegClass.contains(DestReg) &&
2460       (AArch64::GPR64spRegClass.contains(SrcReg) || SrcReg == AArch64::XZR)) {
2461     if (DestReg == AArch64::SP || SrcReg == AArch64::SP) {
2462       // If either operand is SP, expand to ADD #0.
2463       BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestReg)
2464           .addReg(SrcReg, getKillRegState(KillSrc))
2465           .addImm(0)
2466           .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2467     } else if (SrcReg == AArch64::XZR && Subtarget.hasZeroCycleZeroingGP()) {
2468       BuildMI(MBB, I, DL, get(AArch64::MOVZXi), DestReg)
2469           .addImm(0)
2470           .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2471     } else {
2472       // Otherwise, expand to ORR XZR.
2473       BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestReg)
2474           .addReg(AArch64::XZR)
2475           .addReg(SrcReg, getKillRegState(KillSrc));
2476     }
2477     return;
2478   }
2479 
2480   // Copy a DDDD register quad by copying the individual sub-registers.
2481   if (AArch64::DDDDRegClass.contains(DestReg) &&
2482       AArch64::DDDDRegClass.contains(SrcReg)) {
2483     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
2484                                        AArch64::dsub2, AArch64::dsub3};
2485     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
2486                      Indices);
2487     return;
2488   }
2489 
2490   // Copy a DDD register triple by copying the individual sub-registers.
2491   if (AArch64::DDDRegClass.contains(DestReg) &&
2492       AArch64::DDDRegClass.contains(SrcReg)) {
2493     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
2494                                        AArch64::dsub2};
2495     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
2496                      Indices);
2497     return;
2498   }
2499 
2500   // Copy a DD register pair by copying the individual sub-registers.
2501   if (AArch64::DDRegClass.contains(DestReg) &&
2502       AArch64::DDRegClass.contains(SrcReg)) {
2503     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1};
2504     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
2505                      Indices);
2506     return;
2507   }
2508 
2509   // Copy a QQQQ register quad by copying the individual sub-registers.
2510   if (AArch64::QQQQRegClass.contains(DestReg) &&
2511       AArch64::QQQQRegClass.contains(SrcReg)) {
2512     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
2513                                        AArch64::qsub2, AArch64::qsub3};
2514     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
2515                      Indices);
2516     return;
2517   }
2518 
2519   // Copy a QQQ register triple by copying the individual sub-registers.
2520   if (AArch64::QQQRegClass.contains(DestReg) &&
2521       AArch64::QQQRegClass.contains(SrcReg)) {
2522     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
2523                                        AArch64::qsub2};
2524     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
2525                      Indices);
2526     return;
2527   }
2528 
2529   // Copy a QQ register pair by copying the individual sub-registers.
2530   if (AArch64::QQRegClass.contains(DestReg) &&
2531       AArch64::QQRegClass.contains(SrcReg)) {
2532     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1};
2533     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
2534                      Indices);
2535     return;
2536   }
2537 
2538   if (AArch64::XSeqPairsClassRegClass.contains(DestReg) &&
2539       AArch64::XSeqPairsClassRegClass.contains(SrcReg)) {
2540     static const unsigned Indices[] = {AArch64::sube64, AArch64::subo64};
2541     copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRXrs,
2542                     AArch64::XZR, Indices);
2543     return;
2544   }
2545 
2546   if (AArch64::WSeqPairsClassRegClass.contains(DestReg) &&
2547       AArch64::WSeqPairsClassRegClass.contains(SrcReg)) {
2548     static const unsigned Indices[] = {AArch64::sube32, AArch64::subo32};
2549     copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRWrs,
2550                     AArch64::WZR, Indices);
2551     return;
2552   }
2553 
2554   if (AArch64::FPR128RegClass.contains(DestReg) &&
2555       AArch64::FPR128RegClass.contains(SrcReg)) {
2556     if (Subtarget.hasNEON()) {
2557       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2558           .addReg(SrcReg)
2559           .addReg(SrcReg, getKillRegState(KillSrc));
2560     } else {
2561       BuildMI(MBB, I, DL, get(AArch64::STRQpre))
2562           .addReg(AArch64::SP, RegState::Define)
2563           .addReg(SrcReg, getKillRegState(KillSrc))
2564           .addReg(AArch64::SP)
2565           .addImm(-16);
2566       BuildMI(MBB, I, DL, get(AArch64::LDRQpre))
2567           .addReg(AArch64::SP, RegState::Define)
2568           .addReg(DestReg, RegState::Define)
2569           .addReg(AArch64::SP)
2570           .addImm(16);
2571     }
2572     return;
2573   }
2574 
2575   if (AArch64::FPR64RegClass.contains(DestReg) &&
2576       AArch64::FPR64RegClass.contains(SrcReg)) {
2577     if (Subtarget.hasNEON()) {
2578       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::dsub,
2579                                        &AArch64::FPR128RegClass);
2580       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::dsub,
2581                                       &AArch64::FPR128RegClass);
2582       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2583           .addReg(SrcReg)
2584           .addReg(SrcReg, getKillRegState(KillSrc));
2585     } else {
2586       BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestReg)
2587           .addReg(SrcReg, getKillRegState(KillSrc));
2588     }
2589     return;
2590   }
2591 
2592   if (AArch64::FPR32RegClass.contains(DestReg) &&
2593       AArch64::FPR32RegClass.contains(SrcReg)) {
2594     if (Subtarget.hasNEON()) {
2595       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::ssub,
2596                                        &AArch64::FPR128RegClass);
2597       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::ssub,
2598                                       &AArch64::FPR128RegClass);
2599       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2600           .addReg(SrcReg)
2601           .addReg(SrcReg, getKillRegState(KillSrc));
2602     } else {
2603       BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
2604           .addReg(SrcReg, getKillRegState(KillSrc));
2605     }
2606     return;
2607   }
2608 
2609   if (AArch64::FPR16RegClass.contains(DestReg) &&
2610       AArch64::FPR16RegClass.contains(SrcReg)) {
2611     if (Subtarget.hasNEON()) {
2612       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub,
2613                                        &AArch64::FPR128RegClass);
2614       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub,
2615                                       &AArch64::FPR128RegClass);
2616       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2617           .addReg(SrcReg)
2618           .addReg(SrcReg, getKillRegState(KillSrc));
2619     } else {
2620       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub,
2621                                        &AArch64::FPR32RegClass);
2622       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub,
2623                                       &AArch64::FPR32RegClass);
2624       BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
2625           .addReg(SrcReg, getKillRegState(KillSrc));
2626     }
2627     return;
2628   }
2629 
2630   if (AArch64::FPR8RegClass.contains(DestReg) &&
2631       AArch64::FPR8RegClass.contains(SrcReg)) {
2632     if (Subtarget.hasNEON()) {
2633       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub,
2634                                        &AArch64::FPR128RegClass);
2635       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub,
2636                                       &AArch64::FPR128RegClass);
2637       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2638           .addReg(SrcReg)
2639           .addReg(SrcReg, getKillRegState(KillSrc));
2640     } else {
2641       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub,
2642                                        &AArch64::FPR32RegClass);
2643       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub,
2644                                       &AArch64::FPR32RegClass);
2645       BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
2646           .addReg(SrcReg, getKillRegState(KillSrc));
2647     }
2648     return;
2649   }
2650 
2651   // Copies between GPR64 and FPR64.
2652   if (AArch64::FPR64RegClass.contains(DestReg) &&
2653       AArch64::GPR64RegClass.contains(SrcReg)) {
2654     BuildMI(MBB, I, DL, get(AArch64::FMOVXDr), DestReg)
2655         .addReg(SrcReg, getKillRegState(KillSrc));
2656     return;
2657   }
2658   if (AArch64::GPR64RegClass.contains(DestReg) &&
2659       AArch64::FPR64RegClass.contains(SrcReg)) {
2660     BuildMI(MBB, I, DL, get(AArch64::FMOVDXr), DestReg)
2661         .addReg(SrcReg, getKillRegState(KillSrc));
2662     return;
2663   }
2664   // Copies between GPR32 and FPR32.
2665   if (AArch64::FPR32RegClass.contains(DestReg) &&
2666       AArch64::GPR32RegClass.contains(SrcReg)) {
2667     BuildMI(MBB, I, DL, get(AArch64::FMOVWSr), DestReg)
2668         .addReg(SrcReg, getKillRegState(KillSrc));
2669     return;
2670   }
2671   if (AArch64::GPR32RegClass.contains(DestReg) &&
2672       AArch64::FPR32RegClass.contains(SrcReg)) {
2673     BuildMI(MBB, I, DL, get(AArch64::FMOVSWr), DestReg)
2674         .addReg(SrcReg, getKillRegState(KillSrc));
2675     return;
2676   }
2677 
2678   if (DestReg == AArch64::NZCV) {
2679     assert(AArch64::GPR64RegClass.contains(SrcReg) && "Invalid NZCV copy");
2680     BuildMI(MBB, I, DL, get(AArch64::MSR))
2681         .addImm(AArch64SysReg::NZCV)
2682         .addReg(SrcReg, getKillRegState(KillSrc))
2683         .addReg(AArch64::NZCV, RegState::Implicit | RegState::Define);
2684     return;
2685   }
2686 
2687   if (SrcReg == AArch64::NZCV) {
2688     assert(AArch64::GPR64RegClass.contains(DestReg) && "Invalid NZCV copy");
2689     BuildMI(MBB, I, DL, get(AArch64::MRS), DestReg)
2690         .addImm(AArch64SysReg::NZCV)
2691         .addReg(AArch64::NZCV, RegState::Implicit | getKillRegState(KillSrc));
2692     return;
2693   }
2694 
2695   llvm_unreachable("unimplemented reg-to-reg copy");
2696 }
2697 
2698 static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI,
2699                                     MachineBasicBlock &MBB,
2700                                     MachineBasicBlock::iterator InsertBefore,
2701                                     const MCInstrDesc &MCID,
2702                                     unsigned SrcReg, bool IsKill,
2703                                     unsigned SubIdx0, unsigned SubIdx1, int FI,
2704                                     MachineMemOperand *MMO) {
2705   unsigned SrcReg0 = SrcReg;
2706   unsigned SrcReg1 = SrcReg;
2707   if (TargetRegisterInfo::isPhysicalRegister(SrcReg)) {
2708     SrcReg0 = TRI.getSubReg(SrcReg, SubIdx0);
2709     SubIdx0 = 0;
2710     SrcReg1 = TRI.getSubReg(SrcReg, SubIdx1);
2711     SubIdx1 = 0;
2712   }
2713   BuildMI(MBB, InsertBefore, DebugLoc(), MCID)
2714       .addReg(SrcReg0, getKillRegState(IsKill), SubIdx0)
2715       .addReg(SrcReg1, getKillRegState(IsKill), SubIdx1)
2716       .addFrameIndex(FI)
2717       .addImm(0)
2718       .addMemOperand(MMO);
2719 }
2720 
2721 void AArch64InstrInfo::storeRegToStackSlot(
2722     MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned SrcReg,
2723     bool isKill, int FI, const TargetRegisterClass *RC,
2724     const TargetRegisterInfo *TRI) const {
2725   MachineFunction &MF = *MBB.getParent();
2726   MachineFrameInfo &MFI = MF.getFrameInfo();
2727   unsigned Align = MFI.getObjectAlignment(FI);
2728 
2729   MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI);
2730   MachineMemOperand *MMO = MF.getMachineMemOperand(
2731       PtrInfo, MachineMemOperand::MOStore, MFI.getObjectSize(FI), Align);
2732   unsigned Opc = 0;
2733   bool Offset = true;
2734   switch (TRI->getSpillSize(*RC)) {
2735   case 1:
2736     if (AArch64::FPR8RegClass.hasSubClassEq(RC))
2737       Opc = AArch64::STRBui;
2738     break;
2739   case 2:
2740     if (AArch64::FPR16RegClass.hasSubClassEq(RC))
2741       Opc = AArch64::STRHui;
2742     break;
2743   case 4:
2744     if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
2745       Opc = AArch64::STRWui;
2746       if (TargetRegisterInfo::isVirtualRegister(SrcReg))
2747         MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR32RegClass);
2748       else
2749         assert(SrcReg != AArch64::WSP);
2750     } else if (AArch64::FPR32RegClass.hasSubClassEq(RC))
2751       Opc = AArch64::STRSui;
2752     break;
2753   case 8:
2754     if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) {
2755       Opc = AArch64::STRXui;
2756       if (TargetRegisterInfo::isVirtualRegister(SrcReg))
2757         MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass);
2758       else
2759         assert(SrcReg != AArch64::SP);
2760     } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) {
2761       Opc = AArch64::STRDui;
2762     } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) {
2763       storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI,
2764                               get(AArch64::STPWi), SrcReg, isKill,
2765                               AArch64::sube32, AArch64::subo32, FI, MMO);
2766       return;
2767     }
2768     break;
2769   case 16:
2770     if (AArch64::FPR128RegClass.hasSubClassEq(RC))
2771       Opc = AArch64::STRQui;
2772     else if (AArch64::DDRegClass.hasSubClassEq(RC)) {
2773       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
2774       Opc = AArch64::ST1Twov1d;
2775       Offset = false;
2776     } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) {
2777       storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI,
2778                               get(AArch64::STPXi), SrcReg, isKill,
2779                               AArch64::sube64, AArch64::subo64, FI, MMO);
2780       return;
2781     }
2782     break;
2783   case 24:
2784     if (AArch64::DDDRegClass.hasSubClassEq(RC)) {
2785       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
2786       Opc = AArch64::ST1Threev1d;
2787       Offset = false;
2788     }
2789     break;
2790   case 32:
2791     if (AArch64::DDDDRegClass.hasSubClassEq(RC)) {
2792       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
2793       Opc = AArch64::ST1Fourv1d;
2794       Offset = false;
2795     } else if (AArch64::QQRegClass.hasSubClassEq(RC)) {
2796       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
2797       Opc = AArch64::ST1Twov2d;
2798       Offset = false;
2799     }
2800     break;
2801   case 48:
2802     if (AArch64::QQQRegClass.hasSubClassEq(RC)) {
2803       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
2804       Opc = AArch64::ST1Threev2d;
2805       Offset = false;
2806     }
2807     break;
2808   case 64:
2809     if (AArch64::QQQQRegClass.hasSubClassEq(RC)) {
2810       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
2811       Opc = AArch64::ST1Fourv2d;
2812       Offset = false;
2813     }
2814     break;
2815   }
2816   assert(Opc && "Unknown register class");
2817 
2818   const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc))
2819                                      .addReg(SrcReg, getKillRegState(isKill))
2820                                      .addFrameIndex(FI);
2821 
2822   if (Offset)
2823     MI.addImm(0);
2824   MI.addMemOperand(MMO);
2825 }
2826 
2827 static void loadRegPairFromStackSlot(const TargetRegisterInfo &TRI,
2828                                      MachineBasicBlock &MBB,
2829                                      MachineBasicBlock::iterator InsertBefore,
2830                                      const MCInstrDesc &MCID,
2831                                      unsigned DestReg, unsigned SubIdx0,
2832                                      unsigned SubIdx1, int FI,
2833                                      MachineMemOperand *MMO) {
2834   unsigned DestReg0 = DestReg;
2835   unsigned DestReg1 = DestReg;
2836   bool IsUndef = true;
2837   if (TargetRegisterInfo::isPhysicalRegister(DestReg)) {
2838     DestReg0 = TRI.getSubReg(DestReg, SubIdx0);
2839     SubIdx0 = 0;
2840     DestReg1 = TRI.getSubReg(DestReg, SubIdx1);
2841     SubIdx1 = 0;
2842     IsUndef = false;
2843   }
2844   BuildMI(MBB, InsertBefore, DebugLoc(), MCID)
2845       .addReg(DestReg0, RegState::Define | getUndefRegState(IsUndef), SubIdx0)
2846       .addReg(DestReg1, RegState::Define | getUndefRegState(IsUndef), SubIdx1)
2847       .addFrameIndex(FI)
2848       .addImm(0)
2849       .addMemOperand(MMO);
2850 }
2851 
2852 void AArch64InstrInfo::loadRegFromStackSlot(
2853     MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned DestReg,
2854     int FI, const TargetRegisterClass *RC,
2855     const TargetRegisterInfo *TRI) const {
2856   MachineFunction &MF = *MBB.getParent();
2857   MachineFrameInfo &MFI = MF.getFrameInfo();
2858   unsigned Align = MFI.getObjectAlignment(FI);
2859   MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI);
2860   MachineMemOperand *MMO = MF.getMachineMemOperand(
2861       PtrInfo, MachineMemOperand::MOLoad, MFI.getObjectSize(FI), Align);
2862 
2863   unsigned Opc = 0;
2864   bool Offset = true;
2865   switch (TRI->getSpillSize(*RC)) {
2866   case 1:
2867     if (AArch64::FPR8RegClass.hasSubClassEq(RC))
2868       Opc = AArch64::LDRBui;
2869     break;
2870   case 2:
2871     if (AArch64::FPR16RegClass.hasSubClassEq(RC))
2872       Opc = AArch64::LDRHui;
2873     break;
2874   case 4:
2875     if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
2876       Opc = AArch64::LDRWui;
2877       if (TargetRegisterInfo::isVirtualRegister(DestReg))
2878         MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR32RegClass);
2879       else
2880         assert(DestReg != AArch64::WSP);
2881     } else if (AArch64::FPR32RegClass.hasSubClassEq(RC))
2882       Opc = AArch64::LDRSui;
2883     break;
2884   case 8:
2885     if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) {
2886       Opc = AArch64::LDRXui;
2887       if (TargetRegisterInfo::isVirtualRegister(DestReg))
2888         MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR64RegClass);
2889       else
2890         assert(DestReg != AArch64::SP);
2891     } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) {
2892       Opc = AArch64::LDRDui;
2893     } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) {
2894       loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI,
2895                                get(AArch64::LDPWi), DestReg, AArch64::sube32,
2896                                AArch64::subo32, FI, MMO);
2897       return;
2898     }
2899     break;
2900   case 16:
2901     if (AArch64::FPR128RegClass.hasSubClassEq(RC))
2902       Opc = AArch64::LDRQui;
2903     else if (AArch64::DDRegClass.hasSubClassEq(RC)) {
2904       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
2905       Opc = AArch64::LD1Twov1d;
2906       Offset = false;
2907     } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) {
2908       loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI,
2909                                get(AArch64::LDPXi), DestReg, AArch64::sube64,
2910                                AArch64::subo64, FI, MMO);
2911       return;
2912     }
2913     break;
2914   case 24:
2915     if (AArch64::DDDRegClass.hasSubClassEq(RC)) {
2916       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
2917       Opc = AArch64::LD1Threev1d;
2918       Offset = false;
2919     }
2920     break;
2921   case 32:
2922     if (AArch64::DDDDRegClass.hasSubClassEq(RC)) {
2923       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
2924       Opc = AArch64::LD1Fourv1d;
2925       Offset = false;
2926     } else if (AArch64::QQRegClass.hasSubClassEq(RC)) {
2927       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
2928       Opc = AArch64::LD1Twov2d;
2929       Offset = false;
2930     }
2931     break;
2932   case 48:
2933     if (AArch64::QQQRegClass.hasSubClassEq(RC)) {
2934       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
2935       Opc = AArch64::LD1Threev2d;
2936       Offset = false;
2937     }
2938     break;
2939   case 64:
2940     if (AArch64::QQQQRegClass.hasSubClassEq(RC)) {
2941       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
2942       Opc = AArch64::LD1Fourv2d;
2943       Offset = false;
2944     }
2945     break;
2946   }
2947   assert(Opc && "Unknown register class");
2948 
2949   const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc))
2950                                      .addReg(DestReg, getDefRegState(true))
2951                                      .addFrameIndex(FI);
2952   if (Offset)
2953     MI.addImm(0);
2954   MI.addMemOperand(MMO);
2955 }
2956 
2957 void llvm::emitFrameOffset(MachineBasicBlock &MBB,
2958                            MachineBasicBlock::iterator MBBI, const DebugLoc &DL,
2959                            unsigned DestReg, unsigned SrcReg, int Offset,
2960                            const TargetInstrInfo *TII,
2961                            MachineInstr::MIFlag Flag, bool SetNZCV,
2962                            bool NeedsWinCFI) {
2963   if (DestReg == SrcReg && Offset == 0)
2964     return;
2965 
2966   assert((DestReg != AArch64::SP || Offset % 16 == 0) &&
2967          "SP increment/decrement not 16-byte aligned");
2968 
2969   bool isSub = Offset < 0;
2970   if (isSub)
2971     Offset = -Offset;
2972 
2973   // FIXME: If the offset won't fit in 24-bits, compute the offset into a
2974   // scratch register.  If DestReg is a virtual register, use it as the
2975   // scratch register; otherwise, create a new virtual register (to be
2976   // replaced by the scavenger at the end of PEI).  That case can be optimized
2977   // slightly if DestReg is SP which is always 16-byte aligned, so the scratch
2978   // register can be loaded with offset%8 and the add/sub can use an extending
2979   // instruction with LSL#3.
2980   // Currently the function handles any offsets but generates a poor sequence
2981   // of code.
2982   //  assert(Offset < (1 << 24) && "unimplemented reg plus immediate");
2983 
2984   unsigned Opc;
2985   if (SetNZCV)
2986     Opc = isSub ? AArch64::SUBSXri : AArch64::ADDSXri;
2987   else
2988     Opc = isSub ? AArch64::SUBXri : AArch64::ADDXri;
2989   const unsigned MaxEncoding = 0xfff;
2990   const unsigned ShiftSize = 12;
2991   const unsigned MaxEncodableValue = MaxEncoding << ShiftSize;
2992   while (((unsigned)Offset) >= (1 << ShiftSize)) {
2993     unsigned ThisVal;
2994     if (((unsigned)Offset) > MaxEncodableValue) {
2995       ThisVal = MaxEncodableValue;
2996     } else {
2997       ThisVal = Offset & MaxEncodableValue;
2998     }
2999     assert((ThisVal >> ShiftSize) <= MaxEncoding &&
3000            "Encoding cannot handle value that big");
3001     BuildMI(MBB, MBBI, DL, TII->get(Opc), DestReg)
3002         .addReg(SrcReg)
3003         .addImm(ThisVal >> ShiftSize)
3004         .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, ShiftSize))
3005         .setMIFlag(Flag);
3006 
3007    if (NeedsWinCFI && SrcReg == AArch64::SP && DestReg == AArch64::SP)
3008      BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_StackAlloc))
3009          .addImm(ThisVal)
3010          .setMIFlag(Flag);
3011 
3012     SrcReg = DestReg;
3013     Offset -= ThisVal;
3014     if (Offset == 0)
3015       return;
3016   }
3017   BuildMI(MBB, MBBI, DL, TII->get(Opc), DestReg)
3018       .addReg(SrcReg)
3019       .addImm(Offset)
3020       .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0))
3021       .setMIFlag(Flag);
3022 
3023   if (NeedsWinCFI) {
3024     if ((DestReg == AArch64::FP && SrcReg == AArch64::SP) ||
3025         (SrcReg == AArch64::FP && DestReg == AArch64::SP)) {
3026       if (Offset == 0)
3027         BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_SetFP)).
3028                 setMIFlag(Flag);
3029       else
3030         BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_AddFP)).
3031                 addImm(Offset).setMIFlag(Flag);
3032     } else if (DestReg == AArch64::SP) {
3033       BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_StackAlloc)).
3034               addImm(Offset).setMIFlag(Flag);
3035     }
3036   }
3037 }
3038 
3039 MachineInstr *AArch64InstrInfo::foldMemoryOperandImpl(
3040     MachineFunction &MF, MachineInstr &MI, ArrayRef<unsigned> Ops,
3041     MachineBasicBlock::iterator InsertPt, int FrameIndex,
3042     LiveIntervals *LIS) const {
3043   // This is a bit of a hack. Consider this instruction:
3044   //
3045   //   %0 = COPY %sp; GPR64all:%0
3046   //
3047   // We explicitly chose GPR64all for the virtual register so such a copy might
3048   // be eliminated by RegisterCoalescer. However, that may not be possible, and
3049   // %0 may even spill. We can't spill %sp, and since it is in the GPR64all
3050   // register class, TargetInstrInfo::foldMemoryOperand() is going to try.
3051   //
3052   // To prevent that, we are going to constrain the %0 register class here.
3053   //
3054   // <rdar://problem/11522048>
3055   //
3056   if (MI.isFullCopy()) {
3057     unsigned DstReg = MI.getOperand(0).getReg();
3058     unsigned SrcReg = MI.getOperand(1).getReg();
3059     if (SrcReg == AArch64::SP &&
3060         TargetRegisterInfo::isVirtualRegister(DstReg)) {
3061       MF.getRegInfo().constrainRegClass(DstReg, &AArch64::GPR64RegClass);
3062       return nullptr;
3063     }
3064     if (DstReg == AArch64::SP &&
3065         TargetRegisterInfo::isVirtualRegister(SrcReg)) {
3066       MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass);
3067       return nullptr;
3068     }
3069   }
3070 
3071   // Handle the case where a copy is being spilled or filled but the source
3072   // and destination register class don't match.  For example:
3073   //
3074   //   %0 = COPY %xzr; GPR64common:%0
3075   //
3076   // In this case we can still safely fold away the COPY and generate the
3077   // following spill code:
3078   //
3079   //   STRXui %xzr, %stack.0
3080   //
3081   // This also eliminates spilled cross register class COPYs (e.g. between x and
3082   // d regs) of the same size.  For example:
3083   //
3084   //   %0 = COPY %1; GPR64:%0, FPR64:%1
3085   //
3086   // will be filled as
3087   //
3088   //   LDRDui %0, fi<#0>
3089   //
3090   // instead of
3091   //
3092   //   LDRXui %Temp, fi<#0>
3093   //   %0 = FMOV %Temp
3094   //
3095   if (MI.isCopy() && Ops.size() == 1 &&
3096       // Make sure we're only folding the explicit COPY defs/uses.
3097       (Ops[0] == 0 || Ops[0] == 1)) {
3098     bool IsSpill = Ops[0] == 0;
3099     bool IsFill = !IsSpill;
3100     const TargetRegisterInfo &TRI = *MF.getSubtarget().getRegisterInfo();
3101     const MachineRegisterInfo &MRI = MF.getRegInfo();
3102     MachineBasicBlock &MBB = *MI.getParent();
3103     const MachineOperand &DstMO = MI.getOperand(0);
3104     const MachineOperand &SrcMO = MI.getOperand(1);
3105     unsigned DstReg = DstMO.getReg();
3106     unsigned SrcReg = SrcMO.getReg();
3107     // This is slightly expensive to compute for physical regs since
3108     // getMinimalPhysRegClass is slow.
3109     auto getRegClass = [&](unsigned Reg) {
3110       return TargetRegisterInfo::isVirtualRegister(Reg)
3111                  ? MRI.getRegClass(Reg)
3112                  : TRI.getMinimalPhysRegClass(Reg);
3113     };
3114 
3115     if (DstMO.getSubReg() == 0 && SrcMO.getSubReg() == 0) {
3116       assert(TRI.getRegSizeInBits(*getRegClass(DstReg)) ==
3117                  TRI.getRegSizeInBits(*getRegClass(SrcReg)) &&
3118              "Mismatched register size in non subreg COPY");
3119       if (IsSpill)
3120         storeRegToStackSlot(MBB, InsertPt, SrcReg, SrcMO.isKill(), FrameIndex,
3121                             getRegClass(SrcReg), &TRI);
3122       else
3123         loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex,
3124                              getRegClass(DstReg), &TRI);
3125       return &*--InsertPt;
3126     }
3127 
3128     // Handle cases like spilling def of:
3129     //
3130     //   %0:sub_32<def,read-undef> = COPY %wzr; GPR64common:%0
3131     //
3132     // where the physical register source can be widened and stored to the full
3133     // virtual reg destination stack slot, in this case producing:
3134     //
3135     //   STRXui %xzr, %stack.0
3136     //
3137     if (IsSpill && DstMO.isUndef() &&
3138         TargetRegisterInfo::isPhysicalRegister(SrcReg)) {
3139       assert(SrcMO.getSubReg() == 0 &&
3140              "Unexpected subreg on physical register");
3141       const TargetRegisterClass *SpillRC;
3142       unsigned SpillSubreg;
3143       switch (DstMO.getSubReg()) {
3144       default:
3145         SpillRC = nullptr;
3146         break;
3147       case AArch64::sub_32:
3148       case AArch64::ssub:
3149         if (AArch64::GPR32RegClass.contains(SrcReg)) {
3150           SpillRC = &AArch64::GPR64RegClass;
3151           SpillSubreg = AArch64::sub_32;
3152         } else if (AArch64::FPR32RegClass.contains(SrcReg)) {
3153           SpillRC = &AArch64::FPR64RegClass;
3154           SpillSubreg = AArch64::ssub;
3155         } else
3156           SpillRC = nullptr;
3157         break;
3158       case AArch64::dsub:
3159         if (AArch64::FPR64RegClass.contains(SrcReg)) {
3160           SpillRC = &AArch64::FPR128RegClass;
3161           SpillSubreg = AArch64::dsub;
3162         } else
3163           SpillRC = nullptr;
3164         break;
3165       }
3166 
3167       if (SpillRC)
3168         if (unsigned WidenedSrcReg =
3169                 TRI.getMatchingSuperReg(SrcReg, SpillSubreg, SpillRC)) {
3170           storeRegToStackSlot(MBB, InsertPt, WidenedSrcReg, SrcMO.isKill(),
3171                               FrameIndex, SpillRC, &TRI);
3172           return &*--InsertPt;
3173         }
3174     }
3175 
3176     // Handle cases like filling use of:
3177     //
3178     //   %0:sub_32<def,read-undef> = COPY %1; GPR64:%0, GPR32:%1
3179     //
3180     // where we can load the full virtual reg source stack slot, into the subreg
3181     // destination, in this case producing:
3182     //
3183     //   LDRWui %0:sub_32<def,read-undef>, %stack.0
3184     //
3185     if (IsFill && SrcMO.getSubReg() == 0 && DstMO.isUndef()) {
3186       const TargetRegisterClass *FillRC;
3187       switch (DstMO.getSubReg()) {
3188       default:
3189         FillRC = nullptr;
3190         break;
3191       case AArch64::sub_32:
3192         FillRC = &AArch64::GPR32RegClass;
3193         break;
3194       case AArch64::ssub:
3195         FillRC = &AArch64::FPR32RegClass;
3196         break;
3197       case AArch64::dsub:
3198         FillRC = &AArch64::FPR64RegClass;
3199         break;
3200       }
3201 
3202       if (FillRC) {
3203         assert(TRI.getRegSizeInBits(*getRegClass(SrcReg)) ==
3204                    TRI.getRegSizeInBits(*FillRC) &&
3205                "Mismatched regclass size on folded subreg COPY");
3206         loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex, FillRC, &TRI);
3207         MachineInstr &LoadMI = *--InsertPt;
3208         MachineOperand &LoadDst = LoadMI.getOperand(0);
3209         assert(LoadDst.getSubReg() == 0 && "unexpected subreg on fill load");
3210         LoadDst.setSubReg(DstMO.getSubReg());
3211         LoadDst.setIsUndef();
3212         return &LoadMI;
3213       }
3214     }
3215   }
3216 
3217   // Cannot fold.
3218   return nullptr;
3219 }
3220 
3221 int llvm::isAArch64FrameOffsetLegal(const MachineInstr &MI, int &Offset,
3222                                     bool *OutUseUnscaledOp,
3223                                     unsigned *OutUnscaledOp,
3224                                     int *EmittableOffset) {
3225   // Set output values in case of early exit.
3226   if (EmittableOffset)
3227     *EmittableOffset = 0;
3228   if (OutUseUnscaledOp)
3229     *OutUseUnscaledOp = false;
3230   if (OutUnscaledOp)
3231     *OutUnscaledOp = 0;
3232 
3233   // Exit early for structured vector spills/fills as they can't take an
3234   // immediate offset.
3235   switch (MI.getOpcode()) {
3236   default:
3237     break;
3238   case AArch64::LD1Twov2d:
3239   case AArch64::LD1Threev2d:
3240   case AArch64::LD1Fourv2d:
3241   case AArch64::LD1Twov1d:
3242   case AArch64::LD1Threev1d:
3243   case AArch64::LD1Fourv1d:
3244   case AArch64::ST1Twov2d:
3245   case AArch64::ST1Threev2d:
3246   case AArch64::ST1Fourv2d:
3247   case AArch64::ST1Twov1d:
3248   case AArch64::ST1Threev1d:
3249   case AArch64::ST1Fourv1d:
3250     return AArch64FrameOffsetCannotUpdate;
3251   }
3252 
3253   // Get the min/max offset and the scale.
3254   unsigned Scale, Width;
3255   int64_t MinOff, MaxOff;
3256   if (!AArch64InstrInfo::getMemOpInfo(MI.getOpcode(), Scale, Width, MinOff,
3257                                       MaxOff))
3258     llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal");
3259 
3260   // Construct the complete offset.
3261   const MachineOperand &ImmOpnd =
3262       MI.getOperand(AArch64InstrInfo::getLoadStoreImmIdx(MI.getOpcode()));
3263   Offset += ImmOpnd.getImm() * Scale;
3264 
3265   // If the offset doesn't match the scale, we rewrite the instruction to
3266   // use the unscaled instruction instead. Likewise, if we have a negative
3267   // offset and there is an unscaled op to use.
3268   Optional<unsigned> UnscaledOp =
3269       AArch64InstrInfo::getUnscaledLdSt(MI.getOpcode());
3270   bool useUnscaledOp = UnscaledOp && (Offset % Scale || Offset < 0);
3271   if (useUnscaledOp &&
3272       !AArch64InstrInfo::getMemOpInfo(*UnscaledOp, Scale, Width, MinOff, MaxOff))
3273     llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal");
3274 
3275   int64_t Remainder = Offset % Scale;
3276   assert(!(Remainder && useUnscaledOp) &&
3277          "Cannot have remainder when using unscaled op");
3278 
3279   assert(MinOff < MaxOff && "Unexpected Min/Max offsets");
3280   int NewOffset = Offset / Scale;
3281   if (MinOff <= NewOffset && NewOffset <= MaxOff)
3282     Offset = Remainder;
3283   else {
3284     NewOffset = NewOffset < 0 ? MinOff : MaxOff;
3285     Offset = Offset - NewOffset * Scale + Remainder;
3286   }
3287 
3288   if (EmittableOffset)
3289     *EmittableOffset = NewOffset;
3290   if (OutUseUnscaledOp)
3291     *OutUseUnscaledOp = useUnscaledOp;
3292   if (OutUnscaledOp && UnscaledOp)
3293     *OutUnscaledOp = *UnscaledOp;
3294 
3295   return AArch64FrameOffsetCanUpdate |
3296          (Offset == 0 ? AArch64FrameOffsetIsLegal : 0);
3297 }
3298 
3299 bool llvm::rewriteAArch64FrameIndex(MachineInstr &MI, unsigned FrameRegIdx,
3300                                     unsigned FrameReg, int &Offset,
3301                                     const AArch64InstrInfo *TII) {
3302   unsigned Opcode = MI.getOpcode();
3303   unsigned ImmIdx = FrameRegIdx + 1;
3304 
3305   if (Opcode == AArch64::ADDSXri || Opcode == AArch64::ADDXri) {
3306     Offset += MI.getOperand(ImmIdx).getImm();
3307     emitFrameOffset(*MI.getParent(), MI, MI.getDebugLoc(),
3308                     MI.getOperand(0).getReg(), FrameReg, Offset, TII,
3309                     MachineInstr::NoFlags, (Opcode == AArch64::ADDSXri));
3310     MI.eraseFromParent();
3311     Offset = 0;
3312     return true;
3313   }
3314 
3315   int NewOffset;
3316   unsigned UnscaledOp;
3317   bool UseUnscaledOp;
3318   int Status = isAArch64FrameOffsetLegal(MI, Offset, &UseUnscaledOp,
3319                                          &UnscaledOp, &NewOffset);
3320   if (Status & AArch64FrameOffsetCanUpdate) {
3321     if (Status & AArch64FrameOffsetIsLegal)
3322       // Replace the FrameIndex with FrameReg.
3323       MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false);
3324     if (UseUnscaledOp)
3325       MI.setDesc(TII->get(UnscaledOp));
3326 
3327     MI.getOperand(ImmIdx).ChangeToImmediate(NewOffset);
3328     return Offset == 0;
3329   }
3330 
3331   return false;
3332 }
3333 
3334 void AArch64InstrInfo::getNoop(MCInst &NopInst) const {
3335   NopInst.setOpcode(AArch64::HINT);
3336   NopInst.addOperand(MCOperand::createImm(0));
3337 }
3338 
3339 // AArch64 supports MachineCombiner.
3340 bool AArch64InstrInfo::useMachineCombiner() const { return true; }
3341 
3342 // True when Opc sets flag
3343 static bool isCombineInstrSettingFlag(unsigned Opc) {
3344   switch (Opc) {
3345   case AArch64::ADDSWrr:
3346   case AArch64::ADDSWri:
3347   case AArch64::ADDSXrr:
3348   case AArch64::ADDSXri:
3349   case AArch64::SUBSWrr:
3350   case AArch64::SUBSXrr:
3351   // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
3352   case AArch64::SUBSWri:
3353   case AArch64::SUBSXri:
3354     return true;
3355   default:
3356     break;
3357   }
3358   return false;
3359 }
3360 
3361 // 32b Opcodes that can be combined with a MUL
3362 static bool isCombineInstrCandidate32(unsigned Opc) {
3363   switch (Opc) {
3364   case AArch64::ADDWrr:
3365   case AArch64::ADDWri:
3366   case AArch64::SUBWrr:
3367   case AArch64::ADDSWrr:
3368   case AArch64::ADDSWri:
3369   case AArch64::SUBSWrr:
3370   // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
3371   case AArch64::SUBWri:
3372   case AArch64::SUBSWri:
3373     return true;
3374   default:
3375     break;
3376   }
3377   return false;
3378 }
3379 
3380 // 64b Opcodes that can be combined with a MUL
3381 static bool isCombineInstrCandidate64(unsigned Opc) {
3382   switch (Opc) {
3383   case AArch64::ADDXrr:
3384   case AArch64::ADDXri:
3385   case AArch64::SUBXrr:
3386   case AArch64::ADDSXrr:
3387   case AArch64::ADDSXri:
3388   case AArch64::SUBSXrr:
3389   // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
3390   case AArch64::SUBXri:
3391   case AArch64::SUBSXri:
3392     return true;
3393   default:
3394     break;
3395   }
3396   return false;
3397 }
3398 
3399 // FP Opcodes that can be combined with a FMUL
3400 static bool isCombineInstrCandidateFP(const MachineInstr &Inst) {
3401   switch (Inst.getOpcode()) {
3402   default:
3403     break;
3404   case AArch64::FADDSrr:
3405   case AArch64::FADDDrr:
3406   case AArch64::FADDv2f32:
3407   case AArch64::FADDv2f64:
3408   case AArch64::FADDv4f32:
3409   case AArch64::FSUBSrr:
3410   case AArch64::FSUBDrr:
3411   case AArch64::FSUBv2f32:
3412   case AArch64::FSUBv2f64:
3413   case AArch64::FSUBv4f32:
3414     TargetOptions Options = Inst.getParent()->getParent()->getTarget().Options;
3415     return (Options.UnsafeFPMath ||
3416             Options.AllowFPOpFusion == FPOpFusion::Fast);
3417   }
3418   return false;
3419 }
3420 
3421 // Opcodes that can be combined with a MUL
3422 static bool isCombineInstrCandidate(unsigned Opc) {
3423   return (isCombineInstrCandidate32(Opc) || isCombineInstrCandidate64(Opc));
3424 }
3425 
3426 //
3427 // Utility routine that checks if \param MO is defined by an
3428 // \param CombineOpc instruction in the basic block \param MBB
3429 static bool canCombine(MachineBasicBlock &MBB, MachineOperand &MO,
3430                        unsigned CombineOpc, unsigned ZeroReg = 0,
3431                        bool CheckZeroReg = false) {
3432   MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3433   MachineInstr *MI = nullptr;
3434 
3435   if (MO.isReg() && TargetRegisterInfo::isVirtualRegister(MO.getReg()))
3436     MI = MRI.getUniqueVRegDef(MO.getReg());
3437   // And it needs to be in the trace (otherwise, it won't have a depth).
3438   if (!MI || MI->getParent() != &MBB || (unsigned)MI->getOpcode() != CombineOpc)
3439     return false;
3440   // Must only used by the user we combine with.
3441   if (!MRI.hasOneNonDBGUse(MI->getOperand(0).getReg()))
3442     return false;
3443 
3444   if (CheckZeroReg) {
3445     assert(MI->getNumOperands() >= 4 && MI->getOperand(0).isReg() &&
3446            MI->getOperand(1).isReg() && MI->getOperand(2).isReg() &&
3447            MI->getOperand(3).isReg() && "MAdd/MSub must have a least 4 regs");
3448     // The third input reg must be zero.
3449     if (MI->getOperand(3).getReg() != ZeroReg)
3450       return false;
3451   }
3452 
3453   return true;
3454 }
3455 
3456 //
3457 // Is \param MO defined by an integer multiply and can be combined?
3458 static bool canCombineWithMUL(MachineBasicBlock &MBB, MachineOperand &MO,
3459                               unsigned MulOpc, unsigned ZeroReg) {
3460   return canCombine(MBB, MO, MulOpc, ZeroReg, true);
3461 }
3462 
3463 //
3464 // Is \param MO defined by a floating-point multiply and can be combined?
3465 static bool canCombineWithFMUL(MachineBasicBlock &MBB, MachineOperand &MO,
3466                                unsigned MulOpc) {
3467   return canCombine(MBB, MO, MulOpc);
3468 }
3469 
3470 // TODO: There are many more machine instruction opcodes to match:
3471 //       1. Other data types (integer, vectors)
3472 //       2. Other math / logic operations (xor, or)
3473 //       3. Other forms of the same operation (intrinsics and other variants)
3474 bool AArch64InstrInfo::isAssociativeAndCommutative(
3475     const MachineInstr &Inst) const {
3476   switch (Inst.getOpcode()) {
3477   case AArch64::FADDDrr:
3478   case AArch64::FADDSrr:
3479   case AArch64::FADDv2f32:
3480   case AArch64::FADDv2f64:
3481   case AArch64::FADDv4f32:
3482   case AArch64::FMULDrr:
3483   case AArch64::FMULSrr:
3484   case AArch64::FMULX32:
3485   case AArch64::FMULX64:
3486   case AArch64::FMULXv2f32:
3487   case AArch64::FMULXv2f64:
3488   case AArch64::FMULXv4f32:
3489   case AArch64::FMULv2f32:
3490   case AArch64::FMULv2f64:
3491   case AArch64::FMULv4f32:
3492     return Inst.getParent()->getParent()->getTarget().Options.UnsafeFPMath;
3493   default:
3494     return false;
3495   }
3496 }
3497 
3498 /// Find instructions that can be turned into madd.
3499 static bool getMaddPatterns(MachineInstr &Root,
3500                             SmallVectorImpl<MachineCombinerPattern> &Patterns) {
3501   unsigned Opc = Root.getOpcode();
3502   MachineBasicBlock &MBB = *Root.getParent();
3503   bool Found = false;
3504 
3505   if (!isCombineInstrCandidate(Opc))
3506     return false;
3507   if (isCombineInstrSettingFlag(Opc)) {
3508     int Cmp_NZCV = Root.findRegisterDefOperandIdx(AArch64::NZCV, true);
3509     // When NZCV is live bail out.
3510     if (Cmp_NZCV == -1)
3511       return false;
3512     unsigned NewOpc = convertToNonFlagSettingOpc(Root);
3513     // When opcode can't change bail out.
3514     // CHECKME: do we miss any cases for opcode conversion?
3515     if (NewOpc == Opc)
3516       return false;
3517     Opc = NewOpc;
3518   }
3519 
3520   switch (Opc) {
3521   default:
3522     break;
3523   case AArch64::ADDWrr:
3524     assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
3525            "ADDWrr does not have register operands");
3526     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr,
3527                           AArch64::WZR)) {
3528       Patterns.push_back(MachineCombinerPattern::MULADDW_OP1);
3529       Found = true;
3530     }
3531     if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDWrrr,
3532                           AArch64::WZR)) {
3533       Patterns.push_back(MachineCombinerPattern::MULADDW_OP2);
3534       Found = true;
3535     }
3536     break;
3537   case AArch64::ADDXrr:
3538     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr,
3539                           AArch64::XZR)) {
3540       Patterns.push_back(MachineCombinerPattern::MULADDX_OP1);
3541       Found = true;
3542     }
3543     if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDXrrr,
3544                           AArch64::XZR)) {
3545       Patterns.push_back(MachineCombinerPattern::MULADDX_OP2);
3546       Found = true;
3547     }
3548     break;
3549   case AArch64::SUBWrr:
3550     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr,
3551                           AArch64::WZR)) {
3552       Patterns.push_back(MachineCombinerPattern::MULSUBW_OP1);
3553       Found = true;
3554     }
3555     if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDWrrr,
3556                           AArch64::WZR)) {
3557       Patterns.push_back(MachineCombinerPattern::MULSUBW_OP2);
3558       Found = true;
3559     }
3560     break;
3561   case AArch64::SUBXrr:
3562     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr,
3563                           AArch64::XZR)) {
3564       Patterns.push_back(MachineCombinerPattern::MULSUBX_OP1);
3565       Found = true;
3566     }
3567     if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDXrrr,
3568                           AArch64::XZR)) {
3569       Patterns.push_back(MachineCombinerPattern::MULSUBX_OP2);
3570       Found = true;
3571     }
3572     break;
3573   case AArch64::ADDWri:
3574     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr,
3575                           AArch64::WZR)) {
3576       Patterns.push_back(MachineCombinerPattern::MULADDWI_OP1);
3577       Found = true;
3578     }
3579     break;
3580   case AArch64::ADDXri:
3581     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr,
3582                           AArch64::XZR)) {
3583       Patterns.push_back(MachineCombinerPattern::MULADDXI_OP1);
3584       Found = true;
3585     }
3586     break;
3587   case AArch64::SUBWri:
3588     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr,
3589                           AArch64::WZR)) {
3590       Patterns.push_back(MachineCombinerPattern::MULSUBWI_OP1);
3591       Found = true;
3592     }
3593     break;
3594   case AArch64::SUBXri:
3595     if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr,
3596                           AArch64::XZR)) {
3597       Patterns.push_back(MachineCombinerPattern::MULSUBXI_OP1);
3598       Found = true;
3599     }
3600     break;
3601   }
3602   return Found;
3603 }
3604 /// Floating-Point Support
3605 
3606 /// Find instructions that can be turned into madd.
3607 static bool getFMAPatterns(MachineInstr &Root,
3608                            SmallVectorImpl<MachineCombinerPattern> &Patterns) {
3609 
3610   if (!isCombineInstrCandidateFP(Root))
3611     return false;
3612 
3613   MachineBasicBlock &MBB = *Root.getParent();
3614   bool Found = false;
3615 
3616   switch (Root.getOpcode()) {
3617   default:
3618     assert(false && "Unsupported FP instruction in combiner\n");
3619     break;
3620   case AArch64::FADDSrr:
3621     assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
3622            "FADDWrr does not have register operands");
3623     if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULSrr)) {
3624       Patterns.push_back(MachineCombinerPattern::FMULADDS_OP1);
3625       Found = true;
3626     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3627                                   AArch64::FMULv1i32_indexed)) {
3628       Patterns.push_back(MachineCombinerPattern::FMLAv1i32_indexed_OP1);
3629       Found = true;
3630     }
3631     if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULSrr)) {
3632       Patterns.push_back(MachineCombinerPattern::FMULADDS_OP2);
3633       Found = true;
3634     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3635                                   AArch64::FMULv1i32_indexed)) {
3636       Patterns.push_back(MachineCombinerPattern::FMLAv1i32_indexed_OP2);
3637       Found = true;
3638     }
3639     break;
3640   case AArch64::FADDDrr:
3641     if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULDrr)) {
3642       Patterns.push_back(MachineCombinerPattern::FMULADDD_OP1);
3643       Found = true;
3644     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3645                                   AArch64::FMULv1i64_indexed)) {
3646       Patterns.push_back(MachineCombinerPattern::FMLAv1i64_indexed_OP1);
3647       Found = true;
3648     }
3649     if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULDrr)) {
3650       Patterns.push_back(MachineCombinerPattern::FMULADDD_OP2);
3651       Found = true;
3652     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3653                                   AArch64::FMULv1i64_indexed)) {
3654       Patterns.push_back(MachineCombinerPattern::FMLAv1i64_indexed_OP2);
3655       Found = true;
3656     }
3657     break;
3658   case AArch64::FADDv2f32:
3659     if (canCombineWithFMUL(MBB, Root.getOperand(1),
3660                            AArch64::FMULv2i32_indexed)) {
3661       Patterns.push_back(MachineCombinerPattern::FMLAv2i32_indexed_OP1);
3662       Found = true;
3663     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3664                                   AArch64::FMULv2f32)) {
3665       Patterns.push_back(MachineCombinerPattern::FMLAv2f32_OP1);
3666       Found = true;
3667     }
3668     if (canCombineWithFMUL(MBB, Root.getOperand(2),
3669                            AArch64::FMULv2i32_indexed)) {
3670       Patterns.push_back(MachineCombinerPattern::FMLAv2i32_indexed_OP2);
3671       Found = true;
3672     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3673                                   AArch64::FMULv2f32)) {
3674       Patterns.push_back(MachineCombinerPattern::FMLAv2f32_OP2);
3675       Found = true;
3676     }
3677     break;
3678   case AArch64::FADDv2f64:
3679     if (canCombineWithFMUL(MBB, Root.getOperand(1),
3680                            AArch64::FMULv2i64_indexed)) {
3681       Patterns.push_back(MachineCombinerPattern::FMLAv2i64_indexed_OP1);
3682       Found = true;
3683     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3684                                   AArch64::FMULv2f64)) {
3685       Patterns.push_back(MachineCombinerPattern::FMLAv2f64_OP1);
3686       Found = true;
3687     }
3688     if (canCombineWithFMUL(MBB, Root.getOperand(2),
3689                            AArch64::FMULv2i64_indexed)) {
3690       Patterns.push_back(MachineCombinerPattern::FMLAv2i64_indexed_OP2);
3691       Found = true;
3692     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3693                                   AArch64::FMULv2f64)) {
3694       Patterns.push_back(MachineCombinerPattern::FMLAv2f64_OP2);
3695       Found = true;
3696     }
3697     break;
3698   case AArch64::FADDv4f32:
3699     if (canCombineWithFMUL(MBB, Root.getOperand(1),
3700                            AArch64::FMULv4i32_indexed)) {
3701       Patterns.push_back(MachineCombinerPattern::FMLAv4i32_indexed_OP1);
3702       Found = true;
3703     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3704                                   AArch64::FMULv4f32)) {
3705       Patterns.push_back(MachineCombinerPattern::FMLAv4f32_OP1);
3706       Found = true;
3707     }
3708     if (canCombineWithFMUL(MBB, Root.getOperand(2),
3709                            AArch64::FMULv4i32_indexed)) {
3710       Patterns.push_back(MachineCombinerPattern::FMLAv4i32_indexed_OP2);
3711       Found = true;
3712     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3713                                   AArch64::FMULv4f32)) {
3714       Patterns.push_back(MachineCombinerPattern::FMLAv4f32_OP2);
3715       Found = true;
3716     }
3717     break;
3718 
3719   case AArch64::FSUBSrr:
3720     if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULSrr)) {
3721       Patterns.push_back(MachineCombinerPattern::FMULSUBS_OP1);
3722       Found = true;
3723     }
3724     if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULSrr)) {
3725       Patterns.push_back(MachineCombinerPattern::FMULSUBS_OP2);
3726       Found = true;
3727     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3728                                   AArch64::FMULv1i32_indexed)) {
3729       Patterns.push_back(MachineCombinerPattern::FMLSv1i32_indexed_OP2);
3730       Found = true;
3731     }
3732     if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FNMULSrr)) {
3733       Patterns.push_back(MachineCombinerPattern::FNMULSUBS_OP1);
3734       Found = true;
3735     }
3736     break;
3737   case AArch64::FSUBDrr:
3738     if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULDrr)) {
3739       Patterns.push_back(MachineCombinerPattern::FMULSUBD_OP1);
3740       Found = true;
3741     }
3742     if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULDrr)) {
3743       Patterns.push_back(MachineCombinerPattern::FMULSUBD_OP2);
3744       Found = true;
3745     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3746                                   AArch64::FMULv1i64_indexed)) {
3747       Patterns.push_back(MachineCombinerPattern::FMLSv1i64_indexed_OP2);
3748       Found = true;
3749     }
3750     if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FNMULDrr)) {
3751       Patterns.push_back(MachineCombinerPattern::FNMULSUBD_OP1);
3752       Found = true;
3753     }
3754     break;
3755   case AArch64::FSUBv2f32:
3756     if (canCombineWithFMUL(MBB, Root.getOperand(2),
3757                            AArch64::FMULv2i32_indexed)) {
3758       Patterns.push_back(MachineCombinerPattern::FMLSv2i32_indexed_OP2);
3759       Found = true;
3760     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3761                                   AArch64::FMULv2f32)) {
3762       Patterns.push_back(MachineCombinerPattern::FMLSv2f32_OP2);
3763       Found = true;
3764     }
3765     if (canCombineWithFMUL(MBB, Root.getOperand(1),
3766                            AArch64::FMULv2i32_indexed)) {
3767       Patterns.push_back(MachineCombinerPattern::FMLSv2i32_indexed_OP1);
3768       Found = true;
3769     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3770                                   AArch64::FMULv2f32)) {
3771       Patterns.push_back(MachineCombinerPattern::FMLSv2f32_OP1);
3772       Found = true;
3773     }
3774     break;
3775   case AArch64::FSUBv2f64:
3776     if (canCombineWithFMUL(MBB, Root.getOperand(2),
3777                            AArch64::FMULv2i64_indexed)) {
3778       Patterns.push_back(MachineCombinerPattern::FMLSv2i64_indexed_OP2);
3779       Found = true;
3780     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3781                                   AArch64::FMULv2f64)) {
3782       Patterns.push_back(MachineCombinerPattern::FMLSv2f64_OP2);
3783       Found = true;
3784     }
3785     if (canCombineWithFMUL(MBB, Root.getOperand(1),
3786                            AArch64::FMULv2i64_indexed)) {
3787       Patterns.push_back(MachineCombinerPattern::FMLSv2i64_indexed_OP1);
3788       Found = true;
3789     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3790                                   AArch64::FMULv2f64)) {
3791       Patterns.push_back(MachineCombinerPattern::FMLSv2f64_OP1);
3792       Found = true;
3793     }
3794     break;
3795   case AArch64::FSUBv4f32:
3796     if (canCombineWithFMUL(MBB, Root.getOperand(2),
3797                            AArch64::FMULv4i32_indexed)) {
3798       Patterns.push_back(MachineCombinerPattern::FMLSv4i32_indexed_OP2);
3799       Found = true;
3800     } else if (canCombineWithFMUL(MBB, Root.getOperand(2),
3801                                   AArch64::FMULv4f32)) {
3802       Patterns.push_back(MachineCombinerPattern::FMLSv4f32_OP2);
3803       Found = true;
3804     }
3805     if (canCombineWithFMUL(MBB, Root.getOperand(1),
3806                            AArch64::FMULv4i32_indexed)) {
3807       Patterns.push_back(MachineCombinerPattern::FMLSv4i32_indexed_OP1);
3808       Found = true;
3809     } else if (canCombineWithFMUL(MBB, Root.getOperand(1),
3810                                   AArch64::FMULv4f32)) {
3811       Patterns.push_back(MachineCombinerPattern::FMLSv4f32_OP1);
3812       Found = true;
3813     }
3814     break;
3815   }
3816   return Found;
3817 }
3818 
3819 /// Return true when a code sequence can improve throughput. It
3820 /// should be called only for instructions in loops.
3821 /// \param Pattern - combiner pattern
3822 bool AArch64InstrInfo::isThroughputPattern(
3823     MachineCombinerPattern Pattern) const {
3824   switch (Pattern) {
3825   default:
3826     break;
3827   case MachineCombinerPattern::FMULADDS_OP1:
3828   case MachineCombinerPattern::FMULADDS_OP2:
3829   case MachineCombinerPattern::FMULSUBS_OP1:
3830   case MachineCombinerPattern::FMULSUBS_OP2:
3831   case MachineCombinerPattern::FMULADDD_OP1:
3832   case MachineCombinerPattern::FMULADDD_OP2:
3833   case MachineCombinerPattern::FMULSUBD_OP1:
3834   case MachineCombinerPattern::FMULSUBD_OP2:
3835   case MachineCombinerPattern::FNMULSUBS_OP1:
3836   case MachineCombinerPattern::FNMULSUBD_OP1:
3837   case MachineCombinerPattern::FMLAv1i32_indexed_OP1:
3838   case MachineCombinerPattern::FMLAv1i32_indexed_OP2:
3839   case MachineCombinerPattern::FMLAv1i64_indexed_OP1:
3840   case MachineCombinerPattern::FMLAv1i64_indexed_OP2:
3841   case MachineCombinerPattern::FMLAv2f32_OP2:
3842   case MachineCombinerPattern::FMLAv2f32_OP1:
3843   case MachineCombinerPattern::FMLAv2f64_OP1:
3844   case MachineCombinerPattern::FMLAv2f64_OP2:
3845   case MachineCombinerPattern::FMLAv2i32_indexed_OP1:
3846   case MachineCombinerPattern::FMLAv2i32_indexed_OP2:
3847   case MachineCombinerPattern::FMLAv2i64_indexed_OP1:
3848   case MachineCombinerPattern::FMLAv2i64_indexed_OP2:
3849   case MachineCombinerPattern::FMLAv4f32_OP1:
3850   case MachineCombinerPattern::FMLAv4f32_OP2:
3851   case MachineCombinerPattern::FMLAv4i32_indexed_OP1:
3852   case MachineCombinerPattern::FMLAv4i32_indexed_OP2:
3853   case MachineCombinerPattern::FMLSv1i32_indexed_OP2:
3854   case MachineCombinerPattern::FMLSv1i64_indexed_OP2:
3855   case MachineCombinerPattern::FMLSv2i32_indexed_OP2:
3856   case MachineCombinerPattern::FMLSv2i64_indexed_OP2:
3857   case MachineCombinerPattern::FMLSv2f32_OP2:
3858   case MachineCombinerPattern::FMLSv2f64_OP2:
3859   case MachineCombinerPattern::FMLSv4i32_indexed_OP2:
3860   case MachineCombinerPattern::FMLSv4f32_OP2:
3861     return true;
3862   } // end switch (Pattern)
3863   return false;
3864 }
3865 /// Return true when there is potentially a faster code sequence for an
3866 /// instruction chain ending in \p Root. All potential patterns are listed in
3867 /// the \p Pattern vector. Pattern should be sorted in priority order since the
3868 /// pattern evaluator stops checking as soon as it finds a faster sequence.
3869 
3870 bool AArch64InstrInfo::getMachineCombinerPatterns(
3871     MachineInstr &Root,
3872     SmallVectorImpl<MachineCombinerPattern> &Patterns) const {
3873   // Integer patterns
3874   if (getMaddPatterns(Root, Patterns))
3875     return true;
3876   // Floating point patterns
3877   if (getFMAPatterns(Root, Patterns))
3878     return true;
3879 
3880   return TargetInstrInfo::getMachineCombinerPatterns(Root, Patterns);
3881 }
3882 
3883 enum class FMAInstKind { Default, Indexed, Accumulator };
3884 /// genFusedMultiply - Generate fused multiply instructions.
3885 /// This function supports both integer and floating point instructions.
3886 /// A typical example:
3887 ///  F|MUL I=A,B,0
3888 ///  F|ADD R,I,C
3889 ///  ==> F|MADD R,A,B,C
3890 /// \param MF Containing MachineFunction
3891 /// \param MRI Register information
3892 /// \param TII Target information
3893 /// \param Root is the F|ADD instruction
3894 /// \param [out] InsInstrs is a vector of machine instructions and will
3895 /// contain the generated madd instruction
3896 /// \param IdxMulOpd is index of operand in Root that is the result of
3897 /// the F|MUL. In the example above IdxMulOpd is 1.
3898 /// \param MaddOpc the opcode fo the f|madd instruction
3899 /// \param RC Register class of operands
3900 /// \param kind of fma instruction (addressing mode) to be generated
3901 /// \param ReplacedAddend is the result register from the instruction
3902 /// replacing the non-combined operand, if any.
3903 static MachineInstr *
3904 genFusedMultiply(MachineFunction &MF, MachineRegisterInfo &MRI,
3905                  const TargetInstrInfo *TII, MachineInstr &Root,
3906                  SmallVectorImpl<MachineInstr *> &InsInstrs, unsigned IdxMulOpd,
3907                  unsigned MaddOpc, const TargetRegisterClass *RC,
3908                  FMAInstKind kind = FMAInstKind::Default,
3909                  const unsigned *ReplacedAddend = nullptr) {
3910   assert(IdxMulOpd == 1 || IdxMulOpd == 2);
3911 
3912   unsigned IdxOtherOpd = IdxMulOpd == 1 ? 2 : 1;
3913   MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg());
3914   unsigned ResultReg = Root.getOperand(0).getReg();
3915   unsigned SrcReg0 = MUL->getOperand(1).getReg();
3916   bool Src0IsKill = MUL->getOperand(1).isKill();
3917   unsigned SrcReg1 = MUL->getOperand(2).getReg();
3918   bool Src1IsKill = MUL->getOperand(2).isKill();
3919 
3920   unsigned SrcReg2;
3921   bool Src2IsKill;
3922   if (ReplacedAddend) {
3923     // If we just generated a new addend, we must be it's only use.
3924     SrcReg2 = *ReplacedAddend;
3925     Src2IsKill = true;
3926   } else {
3927     SrcReg2 = Root.getOperand(IdxOtherOpd).getReg();
3928     Src2IsKill = Root.getOperand(IdxOtherOpd).isKill();
3929   }
3930 
3931   if (TargetRegisterInfo::isVirtualRegister(ResultReg))
3932     MRI.constrainRegClass(ResultReg, RC);
3933   if (TargetRegisterInfo::isVirtualRegister(SrcReg0))
3934     MRI.constrainRegClass(SrcReg0, RC);
3935   if (TargetRegisterInfo::isVirtualRegister(SrcReg1))
3936     MRI.constrainRegClass(SrcReg1, RC);
3937   if (TargetRegisterInfo::isVirtualRegister(SrcReg2))
3938     MRI.constrainRegClass(SrcReg2, RC);
3939 
3940   MachineInstrBuilder MIB;
3941   if (kind == FMAInstKind::Default)
3942     MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
3943               .addReg(SrcReg0, getKillRegState(Src0IsKill))
3944               .addReg(SrcReg1, getKillRegState(Src1IsKill))
3945               .addReg(SrcReg2, getKillRegState(Src2IsKill));
3946   else if (kind == FMAInstKind::Indexed)
3947     MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
3948               .addReg(SrcReg2, getKillRegState(Src2IsKill))
3949               .addReg(SrcReg0, getKillRegState(Src0IsKill))
3950               .addReg(SrcReg1, getKillRegState(Src1IsKill))
3951               .addImm(MUL->getOperand(3).getImm());
3952   else if (kind == FMAInstKind::Accumulator)
3953     MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
3954               .addReg(SrcReg2, getKillRegState(Src2IsKill))
3955               .addReg(SrcReg0, getKillRegState(Src0IsKill))
3956               .addReg(SrcReg1, getKillRegState(Src1IsKill));
3957   else
3958     assert(false && "Invalid FMA instruction kind \n");
3959   // Insert the MADD (MADD, FMA, FMS, FMLA, FMSL)
3960   InsInstrs.push_back(MIB);
3961   return MUL;
3962 }
3963 
3964 /// genMaddR - Generate madd instruction and combine mul and add using
3965 /// an extra virtual register
3966 /// Example - an ADD intermediate needs to be stored in a register:
3967 ///   MUL I=A,B,0
3968 ///   ADD R,I,Imm
3969 ///   ==> ORR  V, ZR, Imm
3970 ///   ==> MADD R,A,B,V
3971 /// \param MF Containing MachineFunction
3972 /// \param MRI Register information
3973 /// \param TII Target information
3974 /// \param Root is the ADD instruction
3975 /// \param [out] InsInstrs is a vector of machine instructions and will
3976 /// contain the generated madd instruction
3977 /// \param IdxMulOpd is index of operand in Root that is the result of
3978 /// the MUL. In the example above IdxMulOpd is 1.
3979 /// \param MaddOpc the opcode fo the madd instruction
3980 /// \param VR is a virtual register that holds the value of an ADD operand
3981 /// (V in the example above).
3982 /// \param RC Register class of operands
3983 static MachineInstr *genMaddR(MachineFunction &MF, MachineRegisterInfo &MRI,
3984                               const TargetInstrInfo *TII, MachineInstr &Root,
3985                               SmallVectorImpl<MachineInstr *> &InsInstrs,
3986                               unsigned IdxMulOpd, unsigned MaddOpc, unsigned VR,
3987                               const TargetRegisterClass *RC) {
3988   assert(IdxMulOpd == 1 || IdxMulOpd == 2);
3989 
3990   MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg());
3991   unsigned ResultReg = Root.getOperand(0).getReg();
3992   unsigned SrcReg0 = MUL->getOperand(1).getReg();
3993   bool Src0IsKill = MUL->getOperand(1).isKill();
3994   unsigned SrcReg1 = MUL->getOperand(2).getReg();
3995   bool Src1IsKill = MUL->getOperand(2).isKill();
3996 
3997   if (TargetRegisterInfo::isVirtualRegister(ResultReg))
3998     MRI.constrainRegClass(ResultReg, RC);
3999   if (TargetRegisterInfo::isVirtualRegister(SrcReg0))
4000     MRI.constrainRegClass(SrcReg0, RC);
4001   if (TargetRegisterInfo::isVirtualRegister(SrcReg1))
4002     MRI.constrainRegClass(SrcReg1, RC);
4003   if (TargetRegisterInfo::isVirtualRegister(VR))
4004     MRI.constrainRegClass(VR, RC);
4005 
4006   MachineInstrBuilder MIB =
4007       BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
4008           .addReg(SrcReg0, getKillRegState(Src0IsKill))
4009           .addReg(SrcReg1, getKillRegState(Src1IsKill))
4010           .addReg(VR);
4011   // Insert the MADD
4012   InsInstrs.push_back(MIB);
4013   return MUL;
4014 }
4015 
4016 /// When getMachineCombinerPatterns() finds potential patterns,
4017 /// this function generates the instructions that could replace the
4018 /// original code sequence
4019 void AArch64InstrInfo::genAlternativeCodeSequence(
4020     MachineInstr &Root, MachineCombinerPattern Pattern,
4021     SmallVectorImpl<MachineInstr *> &InsInstrs,
4022     SmallVectorImpl<MachineInstr *> &DelInstrs,
4023     DenseMap<unsigned, unsigned> &InstrIdxForVirtReg) const {
4024   MachineBasicBlock &MBB = *Root.getParent();
4025   MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
4026   MachineFunction &MF = *MBB.getParent();
4027   const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo();
4028 
4029   MachineInstr *MUL;
4030   const TargetRegisterClass *RC;
4031   unsigned Opc;
4032   switch (Pattern) {
4033   default:
4034     // Reassociate instructions.
4035     TargetInstrInfo::genAlternativeCodeSequence(Root, Pattern, InsInstrs,
4036                                                 DelInstrs, InstrIdxForVirtReg);
4037     return;
4038   case MachineCombinerPattern::MULADDW_OP1:
4039   case MachineCombinerPattern::MULADDX_OP1:
4040     // MUL I=A,B,0
4041     // ADD R,I,C
4042     // ==> MADD R,A,B,C
4043     // --- Create(MADD);
4044     if (Pattern == MachineCombinerPattern::MULADDW_OP1) {
4045       Opc = AArch64::MADDWrrr;
4046       RC = &AArch64::GPR32RegClass;
4047     } else {
4048       Opc = AArch64::MADDXrrr;
4049       RC = &AArch64::GPR64RegClass;
4050     }
4051     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4052     break;
4053   case MachineCombinerPattern::MULADDW_OP2:
4054   case MachineCombinerPattern::MULADDX_OP2:
4055     // MUL I=A,B,0
4056     // ADD R,C,I
4057     // ==> MADD R,A,B,C
4058     // --- Create(MADD);
4059     if (Pattern == MachineCombinerPattern::MULADDW_OP2) {
4060       Opc = AArch64::MADDWrrr;
4061       RC = &AArch64::GPR32RegClass;
4062     } else {
4063       Opc = AArch64::MADDXrrr;
4064       RC = &AArch64::GPR64RegClass;
4065     }
4066     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4067     break;
4068   case MachineCombinerPattern::MULADDWI_OP1:
4069   case MachineCombinerPattern::MULADDXI_OP1: {
4070     // MUL I=A,B,0
4071     // ADD R,I,Imm
4072     // ==> ORR  V, ZR, Imm
4073     // ==> MADD R,A,B,V
4074     // --- Create(MADD);
4075     const TargetRegisterClass *OrrRC;
4076     unsigned BitSize, OrrOpc, ZeroReg;
4077     if (Pattern == MachineCombinerPattern::MULADDWI_OP1) {
4078       OrrOpc = AArch64::ORRWri;
4079       OrrRC = &AArch64::GPR32spRegClass;
4080       BitSize = 32;
4081       ZeroReg = AArch64::WZR;
4082       Opc = AArch64::MADDWrrr;
4083       RC = &AArch64::GPR32RegClass;
4084     } else {
4085       OrrOpc = AArch64::ORRXri;
4086       OrrRC = &AArch64::GPR64spRegClass;
4087       BitSize = 64;
4088       ZeroReg = AArch64::XZR;
4089       Opc = AArch64::MADDXrrr;
4090       RC = &AArch64::GPR64RegClass;
4091     }
4092     unsigned NewVR = MRI.createVirtualRegister(OrrRC);
4093     uint64_t Imm = Root.getOperand(2).getImm();
4094 
4095     if (Root.getOperand(3).isImm()) {
4096       unsigned Val = Root.getOperand(3).getImm();
4097       Imm = Imm << Val;
4098     }
4099     uint64_t UImm = SignExtend64(Imm, BitSize);
4100     uint64_t Encoding;
4101     if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) {
4102       MachineInstrBuilder MIB1 =
4103           BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR)
4104               .addReg(ZeroReg)
4105               .addImm(Encoding);
4106       InsInstrs.push_back(MIB1);
4107       InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4108       MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
4109     }
4110     break;
4111   }
4112   case MachineCombinerPattern::MULSUBW_OP1:
4113   case MachineCombinerPattern::MULSUBX_OP1: {
4114     // MUL I=A,B,0
4115     // SUB R,I, C
4116     // ==> SUB  V, 0, C
4117     // ==> MADD R,A,B,V // = -C + A*B
4118     // --- Create(MADD);
4119     const TargetRegisterClass *SubRC;
4120     unsigned SubOpc, ZeroReg;
4121     if (Pattern == MachineCombinerPattern::MULSUBW_OP1) {
4122       SubOpc = AArch64::SUBWrr;
4123       SubRC = &AArch64::GPR32spRegClass;
4124       ZeroReg = AArch64::WZR;
4125       Opc = AArch64::MADDWrrr;
4126       RC = &AArch64::GPR32RegClass;
4127     } else {
4128       SubOpc = AArch64::SUBXrr;
4129       SubRC = &AArch64::GPR64spRegClass;
4130       ZeroReg = AArch64::XZR;
4131       Opc = AArch64::MADDXrrr;
4132       RC = &AArch64::GPR64RegClass;
4133     }
4134     unsigned NewVR = MRI.createVirtualRegister(SubRC);
4135     // SUB NewVR, 0, C
4136     MachineInstrBuilder MIB1 =
4137         BuildMI(MF, Root.getDebugLoc(), TII->get(SubOpc), NewVR)
4138             .addReg(ZeroReg)
4139             .add(Root.getOperand(2));
4140     InsInstrs.push_back(MIB1);
4141     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4142     MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
4143     break;
4144   }
4145   case MachineCombinerPattern::MULSUBW_OP2:
4146   case MachineCombinerPattern::MULSUBX_OP2:
4147     // MUL I=A,B,0
4148     // SUB R,C,I
4149     // ==> MSUB R,A,B,C (computes C - A*B)
4150     // --- Create(MSUB);
4151     if (Pattern == MachineCombinerPattern::MULSUBW_OP2) {
4152       Opc = AArch64::MSUBWrrr;
4153       RC = &AArch64::GPR32RegClass;
4154     } else {
4155       Opc = AArch64::MSUBXrrr;
4156       RC = &AArch64::GPR64RegClass;
4157     }
4158     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4159     break;
4160   case MachineCombinerPattern::MULSUBWI_OP1:
4161   case MachineCombinerPattern::MULSUBXI_OP1: {
4162     // MUL I=A,B,0
4163     // SUB R,I, Imm
4164     // ==> ORR  V, ZR, -Imm
4165     // ==> MADD R,A,B,V // = -Imm + A*B
4166     // --- Create(MADD);
4167     const TargetRegisterClass *OrrRC;
4168     unsigned BitSize, OrrOpc, ZeroReg;
4169     if (Pattern == MachineCombinerPattern::MULSUBWI_OP1) {
4170       OrrOpc = AArch64::ORRWri;
4171       OrrRC = &AArch64::GPR32spRegClass;
4172       BitSize = 32;
4173       ZeroReg = AArch64::WZR;
4174       Opc = AArch64::MADDWrrr;
4175       RC = &AArch64::GPR32RegClass;
4176     } else {
4177       OrrOpc = AArch64::ORRXri;
4178       OrrRC = &AArch64::GPR64spRegClass;
4179       BitSize = 64;
4180       ZeroReg = AArch64::XZR;
4181       Opc = AArch64::MADDXrrr;
4182       RC = &AArch64::GPR64RegClass;
4183     }
4184     unsigned NewVR = MRI.createVirtualRegister(OrrRC);
4185     uint64_t Imm = Root.getOperand(2).getImm();
4186     if (Root.getOperand(3).isImm()) {
4187       unsigned Val = Root.getOperand(3).getImm();
4188       Imm = Imm << Val;
4189     }
4190     uint64_t UImm = SignExtend64(-Imm, BitSize);
4191     uint64_t Encoding;
4192     if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) {
4193       MachineInstrBuilder MIB1 =
4194           BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR)
4195               .addReg(ZeroReg)
4196               .addImm(Encoding);
4197       InsInstrs.push_back(MIB1);
4198       InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4199       MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
4200     }
4201     break;
4202   }
4203   // Floating Point Support
4204   case MachineCombinerPattern::FMULADDS_OP1:
4205   case MachineCombinerPattern::FMULADDD_OP1:
4206     // MUL I=A,B,0
4207     // ADD R,I,C
4208     // ==> MADD R,A,B,C
4209     // --- Create(MADD);
4210     if (Pattern == MachineCombinerPattern::FMULADDS_OP1) {
4211       Opc = AArch64::FMADDSrrr;
4212       RC = &AArch64::FPR32RegClass;
4213     } else {
4214       Opc = AArch64::FMADDDrrr;
4215       RC = &AArch64::FPR64RegClass;
4216     }
4217     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4218     break;
4219   case MachineCombinerPattern::FMULADDS_OP2:
4220   case MachineCombinerPattern::FMULADDD_OP2:
4221     // FMUL I=A,B,0
4222     // FADD R,C,I
4223     // ==> FMADD R,A,B,C
4224     // --- Create(FMADD);
4225     if (Pattern == MachineCombinerPattern::FMULADDS_OP2) {
4226       Opc = AArch64::FMADDSrrr;
4227       RC = &AArch64::FPR32RegClass;
4228     } else {
4229       Opc = AArch64::FMADDDrrr;
4230       RC = &AArch64::FPR64RegClass;
4231     }
4232     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4233     break;
4234 
4235   case MachineCombinerPattern::FMLAv1i32_indexed_OP1:
4236     Opc = AArch64::FMLAv1i32_indexed;
4237     RC = &AArch64::FPR32RegClass;
4238     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4239                            FMAInstKind::Indexed);
4240     break;
4241   case MachineCombinerPattern::FMLAv1i32_indexed_OP2:
4242     Opc = AArch64::FMLAv1i32_indexed;
4243     RC = &AArch64::FPR32RegClass;
4244     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4245                            FMAInstKind::Indexed);
4246     break;
4247 
4248   case MachineCombinerPattern::FMLAv1i64_indexed_OP1:
4249     Opc = AArch64::FMLAv1i64_indexed;
4250     RC = &AArch64::FPR64RegClass;
4251     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4252                            FMAInstKind::Indexed);
4253     break;
4254   case MachineCombinerPattern::FMLAv1i64_indexed_OP2:
4255     Opc = AArch64::FMLAv1i64_indexed;
4256     RC = &AArch64::FPR64RegClass;
4257     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4258                            FMAInstKind::Indexed);
4259     break;
4260 
4261   case MachineCombinerPattern::FMLAv2i32_indexed_OP1:
4262   case MachineCombinerPattern::FMLAv2f32_OP1:
4263     RC = &AArch64::FPR64RegClass;
4264     if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP1) {
4265       Opc = AArch64::FMLAv2i32_indexed;
4266       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4267                              FMAInstKind::Indexed);
4268     } else {
4269       Opc = AArch64::FMLAv2f32;
4270       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4271                              FMAInstKind::Accumulator);
4272     }
4273     break;
4274   case MachineCombinerPattern::FMLAv2i32_indexed_OP2:
4275   case MachineCombinerPattern::FMLAv2f32_OP2:
4276     RC = &AArch64::FPR64RegClass;
4277     if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP2) {
4278       Opc = AArch64::FMLAv2i32_indexed;
4279       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4280                              FMAInstKind::Indexed);
4281     } else {
4282       Opc = AArch64::FMLAv2f32;
4283       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4284                              FMAInstKind::Accumulator);
4285     }
4286     break;
4287 
4288   case MachineCombinerPattern::FMLAv2i64_indexed_OP1:
4289   case MachineCombinerPattern::FMLAv2f64_OP1:
4290     RC = &AArch64::FPR128RegClass;
4291     if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP1) {
4292       Opc = AArch64::FMLAv2i64_indexed;
4293       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4294                              FMAInstKind::Indexed);
4295     } else {
4296       Opc = AArch64::FMLAv2f64;
4297       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4298                              FMAInstKind::Accumulator);
4299     }
4300     break;
4301   case MachineCombinerPattern::FMLAv2i64_indexed_OP2:
4302   case MachineCombinerPattern::FMLAv2f64_OP2:
4303     RC = &AArch64::FPR128RegClass;
4304     if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP2) {
4305       Opc = AArch64::FMLAv2i64_indexed;
4306       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4307                              FMAInstKind::Indexed);
4308     } else {
4309       Opc = AArch64::FMLAv2f64;
4310       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4311                              FMAInstKind::Accumulator);
4312     }
4313     break;
4314 
4315   case MachineCombinerPattern::FMLAv4i32_indexed_OP1:
4316   case MachineCombinerPattern::FMLAv4f32_OP1:
4317     RC = &AArch64::FPR128RegClass;
4318     if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP1) {
4319       Opc = AArch64::FMLAv4i32_indexed;
4320       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4321                              FMAInstKind::Indexed);
4322     } else {
4323       Opc = AArch64::FMLAv4f32;
4324       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4325                              FMAInstKind::Accumulator);
4326     }
4327     break;
4328 
4329   case MachineCombinerPattern::FMLAv4i32_indexed_OP2:
4330   case MachineCombinerPattern::FMLAv4f32_OP2:
4331     RC = &AArch64::FPR128RegClass;
4332     if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP2) {
4333       Opc = AArch64::FMLAv4i32_indexed;
4334       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4335                              FMAInstKind::Indexed);
4336     } else {
4337       Opc = AArch64::FMLAv4f32;
4338       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4339                              FMAInstKind::Accumulator);
4340     }
4341     break;
4342 
4343   case MachineCombinerPattern::FMULSUBS_OP1:
4344   case MachineCombinerPattern::FMULSUBD_OP1: {
4345     // FMUL I=A,B,0
4346     // FSUB R,I,C
4347     // ==> FNMSUB R,A,B,C // = -C + A*B
4348     // --- Create(FNMSUB);
4349     if (Pattern == MachineCombinerPattern::FMULSUBS_OP1) {
4350       Opc = AArch64::FNMSUBSrrr;
4351       RC = &AArch64::FPR32RegClass;
4352     } else {
4353       Opc = AArch64::FNMSUBDrrr;
4354       RC = &AArch64::FPR64RegClass;
4355     }
4356     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4357     break;
4358   }
4359 
4360   case MachineCombinerPattern::FNMULSUBS_OP1:
4361   case MachineCombinerPattern::FNMULSUBD_OP1: {
4362     // FNMUL I=A,B,0
4363     // FSUB R,I,C
4364     // ==> FNMADD R,A,B,C // = -A*B - C
4365     // --- Create(FNMADD);
4366     if (Pattern == MachineCombinerPattern::FNMULSUBS_OP1) {
4367       Opc = AArch64::FNMADDSrrr;
4368       RC = &AArch64::FPR32RegClass;
4369     } else {
4370       Opc = AArch64::FNMADDDrrr;
4371       RC = &AArch64::FPR64RegClass;
4372     }
4373     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4374     break;
4375   }
4376 
4377   case MachineCombinerPattern::FMULSUBS_OP2:
4378   case MachineCombinerPattern::FMULSUBD_OP2: {
4379     // FMUL I=A,B,0
4380     // FSUB R,C,I
4381     // ==> FMSUB R,A,B,C (computes C - A*B)
4382     // --- Create(FMSUB);
4383     if (Pattern == MachineCombinerPattern::FMULSUBS_OP2) {
4384       Opc = AArch64::FMSUBSrrr;
4385       RC = &AArch64::FPR32RegClass;
4386     } else {
4387       Opc = AArch64::FMSUBDrrr;
4388       RC = &AArch64::FPR64RegClass;
4389     }
4390     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4391     break;
4392   }
4393 
4394   case MachineCombinerPattern::FMLSv1i32_indexed_OP2:
4395     Opc = AArch64::FMLSv1i32_indexed;
4396     RC = &AArch64::FPR32RegClass;
4397     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4398                            FMAInstKind::Indexed);
4399     break;
4400 
4401   case MachineCombinerPattern::FMLSv1i64_indexed_OP2:
4402     Opc = AArch64::FMLSv1i64_indexed;
4403     RC = &AArch64::FPR64RegClass;
4404     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4405                            FMAInstKind::Indexed);
4406     break;
4407 
4408   case MachineCombinerPattern::FMLSv2f32_OP2:
4409   case MachineCombinerPattern::FMLSv2i32_indexed_OP2:
4410     RC = &AArch64::FPR64RegClass;
4411     if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP2) {
4412       Opc = AArch64::FMLSv2i32_indexed;
4413       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4414                              FMAInstKind::Indexed);
4415     } else {
4416       Opc = AArch64::FMLSv2f32;
4417       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4418                              FMAInstKind::Accumulator);
4419     }
4420     break;
4421 
4422   case MachineCombinerPattern::FMLSv2f64_OP2:
4423   case MachineCombinerPattern::FMLSv2i64_indexed_OP2:
4424     RC = &AArch64::FPR128RegClass;
4425     if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP2) {
4426       Opc = AArch64::FMLSv2i64_indexed;
4427       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4428                              FMAInstKind::Indexed);
4429     } else {
4430       Opc = AArch64::FMLSv2f64;
4431       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4432                              FMAInstKind::Accumulator);
4433     }
4434     break;
4435 
4436   case MachineCombinerPattern::FMLSv4f32_OP2:
4437   case MachineCombinerPattern::FMLSv4i32_indexed_OP2:
4438     RC = &AArch64::FPR128RegClass;
4439     if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP2) {
4440       Opc = AArch64::FMLSv4i32_indexed;
4441       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4442                              FMAInstKind::Indexed);
4443     } else {
4444       Opc = AArch64::FMLSv4f32;
4445       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
4446                              FMAInstKind::Accumulator);
4447     }
4448     break;
4449   case MachineCombinerPattern::FMLSv2f32_OP1:
4450   case MachineCombinerPattern::FMLSv2i32_indexed_OP1: {
4451     RC = &AArch64::FPR64RegClass;
4452     unsigned NewVR = MRI.createVirtualRegister(RC);
4453     MachineInstrBuilder MIB1 =
4454         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f32), NewVR)
4455             .add(Root.getOperand(2));
4456     InsInstrs.push_back(MIB1);
4457     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4458     if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP1) {
4459       Opc = AArch64::FMLAv2i32_indexed;
4460       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4461                              FMAInstKind::Indexed, &NewVR);
4462     } else {
4463       Opc = AArch64::FMLAv2f32;
4464       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4465                              FMAInstKind::Accumulator, &NewVR);
4466     }
4467     break;
4468   }
4469   case MachineCombinerPattern::FMLSv4f32_OP1:
4470   case MachineCombinerPattern::FMLSv4i32_indexed_OP1: {
4471     RC = &AArch64::FPR128RegClass;
4472     unsigned NewVR = MRI.createVirtualRegister(RC);
4473     MachineInstrBuilder MIB1 =
4474         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv4f32), NewVR)
4475             .add(Root.getOperand(2));
4476     InsInstrs.push_back(MIB1);
4477     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4478     if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP1) {
4479       Opc = AArch64::FMLAv4i32_indexed;
4480       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4481                              FMAInstKind::Indexed, &NewVR);
4482     } else {
4483       Opc = AArch64::FMLAv4f32;
4484       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4485                              FMAInstKind::Accumulator, &NewVR);
4486     }
4487     break;
4488   }
4489   case MachineCombinerPattern::FMLSv2f64_OP1:
4490   case MachineCombinerPattern::FMLSv2i64_indexed_OP1: {
4491     RC = &AArch64::FPR128RegClass;
4492     unsigned NewVR = MRI.createVirtualRegister(RC);
4493     MachineInstrBuilder MIB1 =
4494         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f64), NewVR)
4495             .add(Root.getOperand(2));
4496     InsInstrs.push_back(MIB1);
4497     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4498     if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP1) {
4499       Opc = AArch64::FMLAv2i64_indexed;
4500       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4501                              FMAInstKind::Indexed, &NewVR);
4502     } else {
4503       Opc = AArch64::FMLAv2f64;
4504       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
4505                              FMAInstKind::Accumulator, &NewVR);
4506     }
4507     break;
4508   }
4509   } // end switch (Pattern)
4510   // Record MUL and ADD/SUB for deletion
4511   DelInstrs.push_back(MUL);
4512   DelInstrs.push_back(&Root);
4513 }
4514 
4515 /// Replace csincr-branch sequence by simple conditional branch
4516 ///
4517 /// Examples:
4518 /// 1. \code
4519 ///   csinc  w9, wzr, wzr, <condition code>
4520 ///   tbnz   w9, #0, 0x44
4521 ///    \endcode
4522 /// to
4523 ///    \code
4524 ///   b.<inverted condition code>
4525 ///    \endcode
4526 ///
4527 /// 2. \code
4528 ///   csinc w9, wzr, wzr, <condition code>
4529 ///   tbz   w9, #0, 0x44
4530 ///    \endcode
4531 /// to
4532 ///    \code
4533 ///   b.<condition code>
4534 ///    \endcode
4535 ///
4536 /// Replace compare and branch sequence by TBZ/TBNZ instruction when the
4537 /// compare's constant operand is power of 2.
4538 ///
4539 /// Examples:
4540 ///    \code
4541 ///   and  w8, w8, #0x400
4542 ///   cbnz w8, L1
4543 ///    \endcode
4544 /// to
4545 ///    \code
4546 ///   tbnz w8, #10, L1
4547 ///    \endcode
4548 ///
4549 /// \param  MI Conditional Branch
4550 /// \return True when the simple conditional branch is generated
4551 ///
4552 bool AArch64InstrInfo::optimizeCondBranch(MachineInstr &MI) const {
4553   bool IsNegativeBranch = false;
4554   bool IsTestAndBranch = false;
4555   unsigned TargetBBInMI = 0;
4556   switch (MI.getOpcode()) {
4557   default:
4558     llvm_unreachable("Unknown branch instruction?");
4559   case AArch64::Bcc:
4560     return false;
4561   case AArch64::CBZW:
4562   case AArch64::CBZX:
4563     TargetBBInMI = 1;
4564     break;
4565   case AArch64::CBNZW:
4566   case AArch64::CBNZX:
4567     TargetBBInMI = 1;
4568     IsNegativeBranch = true;
4569     break;
4570   case AArch64::TBZW:
4571   case AArch64::TBZX:
4572     TargetBBInMI = 2;
4573     IsTestAndBranch = true;
4574     break;
4575   case AArch64::TBNZW:
4576   case AArch64::TBNZX:
4577     TargetBBInMI = 2;
4578     IsNegativeBranch = true;
4579     IsTestAndBranch = true;
4580     break;
4581   }
4582   // So we increment a zero register and test for bits other
4583   // than bit 0? Conservatively bail out in case the verifier
4584   // missed this case.
4585   if (IsTestAndBranch && MI.getOperand(1).getImm())
4586     return false;
4587 
4588   // Find Definition.
4589   assert(MI.getParent() && "Incomplete machine instruciton\n");
4590   MachineBasicBlock *MBB = MI.getParent();
4591   MachineFunction *MF = MBB->getParent();
4592   MachineRegisterInfo *MRI = &MF->getRegInfo();
4593   unsigned VReg = MI.getOperand(0).getReg();
4594   if (!TargetRegisterInfo::isVirtualRegister(VReg))
4595     return false;
4596 
4597   MachineInstr *DefMI = MRI->getVRegDef(VReg);
4598 
4599   // Look through COPY instructions to find definition.
4600   while (DefMI->isCopy()) {
4601     unsigned CopyVReg = DefMI->getOperand(1).getReg();
4602     if (!MRI->hasOneNonDBGUse(CopyVReg))
4603       return false;
4604     if (!MRI->hasOneDef(CopyVReg))
4605       return false;
4606     DefMI = MRI->getVRegDef(CopyVReg);
4607   }
4608 
4609   switch (DefMI->getOpcode()) {
4610   default:
4611     return false;
4612   // Fold AND into a TBZ/TBNZ if constant operand is power of 2.
4613   case AArch64::ANDWri:
4614   case AArch64::ANDXri: {
4615     if (IsTestAndBranch)
4616       return false;
4617     if (DefMI->getParent() != MBB)
4618       return false;
4619     if (!MRI->hasOneNonDBGUse(VReg))
4620       return false;
4621 
4622     bool Is32Bit = (DefMI->getOpcode() == AArch64::ANDWri);
4623     uint64_t Mask = AArch64_AM::decodeLogicalImmediate(
4624         DefMI->getOperand(2).getImm(), Is32Bit ? 32 : 64);
4625     if (!isPowerOf2_64(Mask))
4626       return false;
4627 
4628     MachineOperand &MO = DefMI->getOperand(1);
4629     unsigned NewReg = MO.getReg();
4630     if (!TargetRegisterInfo::isVirtualRegister(NewReg))
4631       return false;
4632 
4633     assert(!MRI->def_empty(NewReg) && "Register must be defined.");
4634 
4635     MachineBasicBlock &RefToMBB = *MBB;
4636     MachineBasicBlock *TBB = MI.getOperand(1).getMBB();
4637     DebugLoc DL = MI.getDebugLoc();
4638     unsigned Imm = Log2_64(Mask);
4639     unsigned Opc = (Imm < 32)
4640                        ? (IsNegativeBranch ? AArch64::TBNZW : AArch64::TBZW)
4641                        : (IsNegativeBranch ? AArch64::TBNZX : AArch64::TBZX);
4642     MachineInstr *NewMI = BuildMI(RefToMBB, MI, DL, get(Opc))
4643                               .addReg(NewReg)
4644                               .addImm(Imm)
4645                               .addMBB(TBB);
4646     // Register lives on to the CBZ now.
4647     MO.setIsKill(false);
4648 
4649     // For immediate smaller than 32, we need to use the 32-bit
4650     // variant (W) in all cases. Indeed the 64-bit variant does not
4651     // allow to encode them.
4652     // Therefore, if the input register is 64-bit, we need to take the
4653     // 32-bit sub-part.
4654     if (!Is32Bit && Imm < 32)
4655       NewMI->getOperand(0).setSubReg(AArch64::sub_32);
4656     MI.eraseFromParent();
4657     return true;
4658   }
4659   // Look for CSINC
4660   case AArch64::CSINCWr:
4661   case AArch64::CSINCXr: {
4662     if (!(DefMI->getOperand(1).getReg() == AArch64::WZR &&
4663           DefMI->getOperand(2).getReg() == AArch64::WZR) &&
4664         !(DefMI->getOperand(1).getReg() == AArch64::XZR &&
4665           DefMI->getOperand(2).getReg() == AArch64::XZR))
4666       return false;
4667 
4668     if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) != -1)
4669       return false;
4670 
4671     AArch64CC::CondCode CC = (AArch64CC::CondCode)DefMI->getOperand(3).getImm();
4672     // Convert only when the condition code is not modified between
4673     // the CSINC and the branch. The CC may be used by other
4674     // instructions in between.
4675     if (areCFlagsAccessedBetweenInstrs(DefMI, MI, &getRegisterInfo(), AK_Write))
4676       return false;
4677     MachineBasicBlock &RefToMBB = *MBB;
4678     MachineBasicBlock *TBB = MI.getOperand(TargetBBInMI).getMBB();
4679     DebugLoc DL = MI.getDebugLoc();
4680     if (IsNegativeBranch)
4681       CC = AArch64CC::getInvertedCondCode(CC);
4682     BuildMI(RefToMBB, MI, DL, get(AArch64::Bcc)).addImm(CC).addMBB(TBB);
4683     MI.eraseFromParent();
4684     return true;
4685   }
4686   }
4687 }
4688 
4689 std::pair<unsigned, unsigned>
4690 AArch64InstrInfo::decomposeMachineOperandsTargetFlags(unsigned TF) const {
4691   const unsigned Mask = AArch64II::MO_FRAGMENT;
4692   return std::make_pair(TF & Mask, TF & ~Mask);
4693 }
4694 
4695 ArrayRef<std::pair<unsigned, const char *>>
4696 AArch64InstrInfo::getSerializableDirectMachineOperandTargetFlags() const {
4697   using namespace AArch64II;
4698 
4699   static const std::pair<unsigned, const char *> TargetFlags[] = {
4700       {MO_PAGE, "aarch64-page"}, {MO_PAGEOFF, "aarch64-pageoff"},
4701       {MO_G3, "aarch64-g3"},     {MO_G2, "aarch64-g2"},
4702       {MO_G1, "aarch64-g1"},     {MO_G0, "aarch64-g0"},
4703       {MO_HI12, "aarch64-hi12"}};
4704   return makeArrayRef(TargetFlags);
4705 }
4706 
4707 ArrayRef<std::pair<unsigned, const char *>>
4708 AArch64InstrInfo::getSerializableBitmaskMachineOperandTargetFlags() const {
4709   using namespace AArch64II;
4710 
4711   static const std::pair<unsigned, const char *> TargetFlags[] = {
4712       {MO_COFFSTUB, "aarch64-coffstub"},
4713       {MO_GOT, "aarch64-got"},   {MO_NC, "aarch64-nc"},
4714       {MO_S, "aarch64-s"},       {MO_TLS, "aarch64-tls"},
4715       {MO_DLLIMPORT, "aarch64-dllimport"}};
4716   return makeArrayRef(TargetFlags);
4717 }
4718 
4719 ArrayRef<std::pair<MachineMemOperand::Flags, const char *>>
4720 AArch64InstrInfo::getSerializableMachineMemOperandTargetFlags() const {
4721   static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] =
4722       {{MOSuppressPair, "aarch64-suppress-pair"},
4723        {MOStridedAccess, "aarch64-strided-access"}};
4724   return makeArrayRef(TargetFlags);
4725 }
4726 
4727 /// Constants defining how certain sequences should be outlined.
4728 /// This encompasses how an outlined function should be called, and what kind of
4729 /// frame should be emitted for that outlined function.
4730 ///
4731 /// \p MachineOutlinerDefault implies that the function should be called with
4732 /// a save and restore of LR to the stack.
4733 ///
4734 /// That is,
4735 ///
4736 /// I1     Save LR                    OUTLINED_FUNCTION:
4737 /// I2 --> BL OUTLINED_FUNCTION       I1
4738 /// I3     Restore LR                 I2
4739 ///                                   I3
4740 ///                                   RET
4741 ///
4742 /// * Call construction overhead: 3 (save + BL + restore)
4743 /// * Frame construction overhead: 1 (ret)
4744 /// * Requires stack fixups? Yes
4745 ///
4746 /// \p MachineOutlinerTailCall implies that the function is being created from
4747 /// a sequence of instructions ending in a return.
4748 ///
4749 /// That is,
4750 ///
4751 /// I1                             OUTLINED_FUNCTION:
4752 /// I2 --> B OUTLINED_FUNCTION     I1
4753 /// RET                            I2
4754 ///                                RET
4755 ///
4756 /// * Call construction overhead: 1 (B)
4757 /// * Frame construction overhead: 0 (Return included in sequence)
4758 /// * Requires stack fixups? No
4759 ///
4760 /// \p MachineOutlinerNoLRSave implies that the function should be called using
4761 /// a BL instruction, but doesn't require LR to be saved and restored. This
4762 /// happens when LR is known to be dead.
4763 ///
4764 /// That is,
4765 ///
4766 /// I1                                OUTLINED_FUNCTION:
4767 /// I2 --> BL OUTLINED_FUNCTION       I1
4768 /// I3                                I2
4769 ///                                   I3
4770 ///                                   RET
4771 ///
4772 /// * Call construction overhead: 1 (BL)
4773 /// * Frame construction overhead: 1 (RET)
4774 /// * Requires stack fixups? No
4775 ///
4776 /// \p MachineOutlinerThunk implies that the function is being created from
4777 /// a sequence of instructions ending in a call. The outlined function is
4778 /// called with a BL instruction, and the outlined function tail-calls the
4779 /// original call destination.
4780 ///
4781 /// That is,
4782 ///
4783 /// I1                                OUTLINED_FUNCTION:
4784 /// I2 --> BL OUTLINED_FUNCTION       I1
4785 /// BL f                              I2
4786 ///                                   B f
4787 /// * Call construction overhead: 1 (BL)
4788 /// * Frame construction overhead: 0
4789 /// * Requires stack fixups? No
4790 ///
4791 /// \p MachineOutlinerRegSave implies that the function should be called with a
4792 /// save and restore of LR to an available register. This allows us to avoid
4793 /// stack fixups. Note that this outlining variant is compatible with the
4794 /// NoLRSave case.
4795 ///
4796 /// That is,
4797 ///
4798 /// I1     Save LR                    OUTLINED_FUNCTION:
4799 /// I2 --> BL OUTLINED_FUNCTION       I1
4800 /// I3     Restore LR                 I2
4801 ///                                   I3
4802 ///                                   RET
4803 ///
4804 /// * Call construction overhead: 3 (save + BL + restore)
4805 /// * Frame construction overhead: 1 (ret)
4806 /// * Requires stack fixups? No
4807 enum MachineOutlinerClass {
4808   MachineOutlinerDefault,  /// Emit a save, restore, call, and return.
4809   MachineOutlinerTailCall, /// Only emit a branch.
4810   MachineOutlinerNoLRSave, /// Emit a call and return.
4811   MachineOutlinerThunk,    /// Emit a call and tail-call.
4812   MachineOutlinerRegSave   /// Same as default, but save to a register.
4813 };
4814 
4815 enum MachineOutlinerMBBFlags {
4816   LRUnavailableSomewhere = 0x2,
4817   HasCalls = 0x4,
4818   UnsafeRegsDead = 0x8
4819 };
4820 
4821 unsigned
4822 AArch64InstrInfo::findRegisterToSaveLRTo(const outliner::Candidate &C) const {
4823   assert(C.LRUWasSet && "LRU wasn't set?");
4824   MachineFunction *MF = C.getMF();
4825   const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>(
4826       MF->getSubtarget().getRegisterInfo());
4827 
4828   // Check if there is an available register across the sequence that we can
4829   // use.
4830   for (unsigned Reg : AArch64::GPR64RegClass) {
4831     if (!ARI->isReservedReg(*MF, Reg) &&
4832         Reg != AArch64::LR &&  // LR is not reserved, but don't use it.
4833         Reg != AArch64::X16 && // X16 is not guaranteed to be preserved.
4834         Reg != AArch64::X17 && // Ditto for X17.
4835         C.LRU.available(Reg) && C.UsedInSequence.available(Reg))
4836       return Reg;
4837   }
4838 
4839   // No suitable register. Return 0.
4840   return 0u;
4841 }
4842 
4843 outliner::OutlinedFunction
4844 AArch64InstrInfo::getOutliningCandidateInfo(
4845     std::vector<outliner::Candidate> &RepeatedSequenceLocs) const {
4846   outliner::Candidate &FirstCand = RepeatedSequenceLocs[0];
4847   unsigned SequenceSize =
4848       std::accumulate(FirstCand.front(), std::next(FirstCand.back()), 0,
4849                       [this](unsigned Sum, const MachineInstr &MI) {
4850                         return Sum + getInstSizeInBytes(MI);
4851                       });
4852 
4853   // Properties about candidate MBBs that hold for all of them.
4854   unsigned FlagsSetInAll = 0xF;
4855 
4856   // Compute liveness information for each candidate, and set FlagsSetInAll.
4857   const TargetRegisterInfo &TRI = getRegisterInfo();
4858   std::for_each(RepeatedSequenceLocs.begin(), RepeatedSequenceLocs.end(),
4859                 [&FlagsSetInAll](outliner::Candidate &C) {
4860                   FlagsSetInAll &= C.Flags;
4861                 });
4862 
4863   // According to the AArch64 Procedure Call Standard, the following are
4864   // undefined on entry/exit from a function call:
4865   //
4866   // * Registers x16, x17, (and thus w16, w17)
4867   // * Condition codes (and thus the NZCV register)
4868   //
4869   // Because if this, we can't outline any sequence of instructions where
4870   // one
4871   // of these registers is live into/across it. Thus, we need to delete
4872   // those
4873   // candidates.
4874   auto CantGuaranteeValueAcrossCall = [&TRI](outliner::Candidate &C) {
4875     // If the unsafe registers in this block are all dead, then we don't need
4876     // to compute liveness here.
4877     if (C.Flags & UnsafeRegsDead)
4878       return false;
4879     C.initLRU(TRI);
4880     LiveRegUnits LRU = C.LRU;
4881     return (!LRU.available(AArch64::W16) || !LRU.available(AArch64::W17) ||
4882             !LRU.available(AArch64::NZCV));
4883   };
4884 
4885   // Are there any candidates where those registers are live?
4886   if (!(FlagsSetInAll & UnsafeRegsDead)) {
4887     // Erase every candidate that violates the restrictions above. (It could be
4888     // true that we have viable candidates, so it's not worth bailing out in
4889     // the case that, say, 1 out of 20 candidates violate the restructions.)
4890     RepeatedSequenceLocs.erase(std::remove_if(RepeatedSequenceLocs.begin(),
4891                                               RepeatedSequenceLocs.end(),
4892                                               CantGuaranteeValueAcrossCall),
4893                                RepeatedSequenceLocs.end());
4894 
4895     // If the sequence doesn't have enough candidates left, then we're done.
4896     if (RepeatedSequenceLocs.size() < 2)
4897       return outliner::OutlinedFunction();
4898   }
4899 
4900   // At this point, we have only "safe" candidates to outline. Figure out
4901   // frame + call instruction information.
4902 
4903   unsigned LastInstrOpcode = RepeatedSequenceLocs[0].back()->getOpcode();
4904 
4905   // Helper lambda which sets call information for every candidate.
4906   auto SetCandidateCallInfo =
4907       [&RepeatedSequenceLocs](unsigned CallID, unsigned NumBytesForCall) {
4908         for (outliner::Candidate &C : RepeatedSequenceLocs)
4909           C.setCallInfo(CallID, NumBytesForCall);
4910       };
4911 
4912   unsigned FrameID = MachineOutlinerDefault;
4913   unsigned NumBytesToCreateFrame = 4;
4914 
4915   bool HasBTI = any_of(RepeatedSequenceLocs, [](outliner::Candidate &C) {
4916     return C.getMF()->getFunction().hasFnAttribute("branch-target-enforcement");
4917   });
4918 
4919   // Returns true if an instructions is safe to fix up, false otherwise.
4920   auto IsSafeToFixup = [this, &TRI](MachineInstr &MI) {
4921     if (MI.isCall())
4922       return true;
4923 
4924     if (!MI.modifiesRegister(AArch64::SP, &TRI) &&
4925         !MI.readsRegister(AArch64::SP, &TRI))
4926       return true;
4927 
4928     // Any modification of SP will break our code to save/restore LR.
4929     // FIXME: We could handle some instructions which add a constant
4930     // offset to SP, with a bit more work.
4931     if (MI.modifiesRegister(AArch64::SP, &TRI))
4932       return false;
4933 
4934     // At this point, we have a stack instruction that we might need to
4935     // fix up. We'll handle it if it's a load or store.
4936     if (MI.mayLoadOrStore()) {
4937       const MachineOperand *Base; // Filled with the base operand of MI.
4938       int64_t Offset;             // Filled with the offset of MI.
4939 
4940       // Does it allow us to offset the base operand and is the base the
4941       // register SP?
4942       if (!getMemOperandWithOffset(MI, Base, Offset, &TRI) || !Base->isReg() ||
4943           Base->getReg() != AArch64::SP)
4944         return false;
4945 
4946       // Find the minimum/maximum offset for this instruction and check
4947       // if fixing it up would be in range.
4948       int64_t MinOffset,
4949           MaxOffset;  // Unscaled offsets for the instruction.
4950       unsigned Scale; // The scale to multiply the offsets by.
4951       unsigned DummyWidth;
4952       getMemOpInfo(MI.getOpcode(), Scale, DummyWidth, MinOffset, MaxOffset);
4953 
4954       Offset += 16; // Update the offset to what it would be if we outlined.
4955       if (Offset < MinOffset * Scale || Offset > MaxOffset * Scale)
4956         return false;
4957 
4958       // It's in range, so we can outline it.
4959       return true;
4960     }
4961 
4962     // FIXME: Add handling for instructions like "add x0, sp, #8".
4963 
4964     // We can't fix it up, so don't outline it.
4965     return false;
4966   };
4967 
4968   // True if it's possible to fix up each stack instruction in this sequence.
4969   // Important for frames/call variants that modify the stack.
4970   bool AllStackInstrsSafe = std::all_of(
4971       FirstCand.front(), std::next(FirstCand.back()), IsSafeToFixup);
4972 
4973   // If the last instruction in any candidate is a terminator, then we should
4974   // tail call all of the candidates.
4975   if (RepeatedSequenceLocs[0].back()->isTerminator()) {
4976     FrameID = MachineOutlinerTailCall;
4977     NumBytesToCreateFrame = 0;
4978     SetCandidateCallInfo(MachineOutlinerTailCall, 4);
4979   }
4980 
4981   else if (LastInstrOpcode == AArch64::BL ||
4982            (LastInstrOpcode == AArch64::BLR && !HasBTI)) {
4983     // FIXME: Do we need to check if the code after this uses the value of LR?
4984     FrameID = MachineOutlinerThunk;
4985     NumBytesToCreateFrame = 0;
4986     SetCandidateCallInfo(MachineOutlinerThunk, 4);
4987   }
4988 
4989   else {
4990     // We need to decide how to emit calls + frames. We can always emit the same
4991     // frame if we don't need to save to the stack. If we have to save to the
4992     // stack, then we need a different frame.
4993     unsigned NumBytesNoStackCalls = 0;
4994     std::vector<outliner::Candidate> CandidatesWithoutStackFixups;
4995 
4996     for (outliner::Candidate &C : RepeatedSequenceLocs) {
4997       C.initLRU(TRI);
4998 
4999       // Is LR available? If so, we don't need a save.
5000       if (C.LRU.available(AArch64::LR)) {
5001         NumBytesNoStackCalls += 4;
5002         C.setCallInfo(MachineOutlinerNoLRSave, 4);
5003         CandidatesWithoutStackFixups.push_back(C);
5004       }
5005 
5006       // Is an unused register available? If so, we won't modify the stack, so
5007       // we can outline with the same frame type as those that don't save LR.
5008       else if (findRegisterToSaveLRTo(C)) {
5009         NumBytesNoStackCalls += 12;
5010         C.setCallInfo(MachineOutlinerRegSave, 12);
5011         CandidatesWithoutStackFixups.push_back(C);
5012       }
5013 
5014       // Is SP used in the sequence at all? If not, we don't have to modify
5015       // the stack, so we are guaranteed to get the same frame.
5016       else if (C.UsedInSequence.available(AArch64::SP)) {
5017         NumBytesNoStackCalls += 12;
5018         C.setCallInfo(MachineOutlinerDefault, 12);
5019         CandidatesWithoutStackFixups.push_back(C);
5020       }
5021 
5022       // If we outline this, we need to modify the stack. Pretend we don't
5023       // outline this by saving all of its bytes.
5024       else {
5025         NumBytesNoStackCalls += SequenceSize;
5026       }
5027     }
5028 
5029     // If there are no places where we have to save LR, then note that we
5030     // don't have to update the stack. Otherwise, give every candidate the
5031     // default call type, as long as it's safe to do so.
5032     if (!AllStackInstrsSafe ||
5033         NumBytesNoStackCalls <= RepeatedSequenceLocs.size() * 12) {
5034       RepeatedSequenceLocs = CandidatesWithoutStackFixups;
5035       FrameID = MachineOutlinerNoLRSave;
5036     } else {
5037       SetCandidateCallInfo(MachineOutlinerDefault, 12);
5038     }
5039 
5040     // If we dropped all of the candidates, bail out here.
5041     if (RepeatedSequenceLocs.size() < 2) {
5042       RepeatedSequenceLocs.clear();
5043       return outliner::OutlinedFunction();
5044     }
5045   }
5046 
5047   // Does every candidate's MBB contain a call? If so, then we might have a call
5048   // in the range.
5049   if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) {
5050     // Check if the range contains a call. These require a save + restore of the
5051     // link register.
5052     bool ModStackToSaveLR = false;
5053     if (std::any_of(FirstCand.front(), FirstCand.back(),
5054                     [](const MachineInstr &MI) { return MI.isCall(); }))
5055       ModStackToSaveLR = true;
5056 
5057     // Handle the last instruction separately. If this is a tail call, then the
5058     // last instruction is a call. We don't want to save + restore in this case.
5059     // However, it could be possible that the last instruction is a call without
5060     // it being valid to tail call this sequence. We should consider this as
5061     // well.
5062     else if (FrameID != MachineOutlinerThunk &&
5063              FrameID != MachineOutlinerTailCall && FirstCand.back()->isCall())
5064       ModStackToSaveLR = true;
5065 
5066     if (ModStackToSaveLR) {
5067       // We can't fix up the stack. Bail out.
5068       if (!AllStackInstrsSafe) {
5069         RepeatedSequenceLocs.clear();
5070         return outliner::OutlinedFunction();
5071       }
5072 
5073       // Save + restore LR.
5074       NumBytesToCreateFrame += 8;
5075     }
5076   }
5077 
5078   return outliner::OutlinedFunction(RepeatedSequenceLocs, SequenceSize,
5079                                     NumBytesToCreateFrame, FrameID);
5080 }
5081 
5082 bool AArch64InstrInfo::isFunctionSafeToOutlineFrom(
5083     MachineFunction &MF, bool OutlineFromLinkOnceODRs) const {
5084   const Function &F = MF.getFunction();
5085 
5086   // Can F be deduplicated by the linker? If it can, don't outline from it.
5087   if (!OutlineFromLinkOnceODRs && F.hasLinkOnceODRLinkage())
5088     return false;
5089 
5090   // Don't outline from functions with section markings; the program could
5091   // expect that all the code is in the named section.
5092   // FIXME: Allow outlining from multiple functions with the same section
5093   // marking.
5094   if (F.hasSection())
5095     return false;
5096 
5097   // Outlining from functions with redzones is unsafe since the outliner may
5098   // modify the stack. Check if hasRedZone is true or unknown; if yes, don't
5099   // outline from it.
5100   AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>();
5101   if (!AFI || AFI->hasRedZone().getValueOr(true))
5102     return false;
5103 
5104   // It's safe to outline from MF.
5105   return true;
5106 }
5107 
5108 bool AArch64InstrInfo::isMBBSafeToOutlineFrom(MachineBasicBlock &MBB,
5109                                               unsigned &Flags) const {
5110   // Check if LR is available through all of the MBB. If it's not, then set
5111   // a flag.
5112   assert(MBB.getParent()->getRegInfo().tracksLiveness() &&
5113          "Suitable Machine Function for outlining must track liveness");
5114   LiveRegUnits LRU(getRegisterInfo());
5115 
5116   std::for_each(MBB.rbegin(), MBB.rend(),
5117                 [&LRU](MachineInstr &MI) { LRU.accumulate(MI); });
5118 
5119   // Check if each of the unsafe registers are available...
5120   bool W16AvailableInBlock = LRU.available(AArch64::W16);
5121   bool W17AvailableInBlock = LRU.available(AArch64::W17);
5122   bool NZCVAvailableInBlock = LRU.available(AArch64::NZCV);
5123 
5124   // If all of these are dead (and not live out), we know we don't have to check
5125   // them later.
5126   if (W16AvailableInBlock && W17AvailableInBlock && NZCVAvailableInBlock)
5127     Flags |= MachineOutlinerMBBFlags::UnsafeRegsDead;
5128 
5129   // Now, add the live outs to the set.
5130   LRU.addLiveOuts(MBB);
5131 
5132   // If any of these registers is available in the MBB, but also a live out of
5133   // the block, then we know outlining is unsafe.
5134   if (W16AvailableInBlock && !LRU.available(AArch64::W16))
5135     return false;
5136   if (W17AvailableInBlock && !LRU.available(AArch64::W17))
5137     return false;
5138   if (NZCVAvailableInBlock && !LRU.available(AArch64::NZCV))
5139     return false;
5140 
5141   // Check if there's a call inside this MachineBasicBlock. If there is, then
5142   // set a flag.
5143   if (any_of(MBB, [](MachineInstr &MI) { return MI.isCall(); }))
5144     Flags |= MachineOutlinerMBBFlags::HasCalls;
5145 
5146   MachineFunction *MF = MBB.getParent();
5147 
5148   // In the event that we outline, we may have to save LR. If there is an
5149   // available register in the MBB, then we'll always save LR there. Check if
5150   // this is true.
5151   bool CanSaveLR = false;
5152   const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>(
5153       MF->getSubtarget().getRegisterInfo());
5154 
5155   // Check if there is an available register across the sequence that we can
5156   // use.
5157   for (unsigned Reg : AArch64::GPR64RegClass) {
5158     if (!ARI->isReservedReg(*MF, Reg) && Reg != AArch64::LR &&
5159         Reg != AArch64::X16 && Reg != AArch64::X17 && LRU.available(Reg)) {
5160       CanSaveLR = true;
5161       break;
5162     }
5163   }
5164 
5165   // Check if we have a register we can save LR to, and if LR was used
5166   // somewhere. If both of those things are true, then we need to evaluate the
5167   // safety of outlining stack instructions later.
5168   if (!CanSaveLR && !LRU.available(AArch64::LR))
5169     Flags |= MachineOutlinerMBBFlags::LRUnavailableSomewhere;
5170 
5171   return true;
5172 }
5173 
5174 outliner::InstrType
5175 AArch64InstrInfo::getOutliningType(MachineBasicBlock::iterator &MIT,
5176                                    unsigned Flags) const {
5177   MachineInstr &MI = *MIT;
5178   MachineBasicBlock *MBB = MI.getParent();
5179   MachineFunction *MF = MBB->getParent();
5180   AArch64FunctionInfo *FuncInfo = MF->getInfo<AArch64FunctionInfo>();
5181 
5182   // Don't outline LOHs.
5183   if (FuncInfo->getLOHRelated().count(&MI))
5184     return outliner::InstrType::Illegal;
5185 
5186   // Don't allow debug values to impact outlining type.
5187   if (MI.isDebugInstr() || MI.isIndirectDebugValue())
5188     return outliner::InstrType::Invisible;
5189 
5190   // At this point, KILL instructions don't really tell us much so we can go
5191   // ahead and skip over them.
5192   if (MI.isKill())
5193     return outliner::InstrType::Invisible;
5194 
5195   // Is this a terminator for a basic block?
5196   if (MI.isTerminator()) {
5197 
5198     // Is this the end of a function?
5199     if (MI.getParent()->succ_empty())
5200       return outliner::InstrType::Legal;
5201 
5202     // It's not, so don't outline it.
5203     return outliner::InstrType::Illegal;
5204   }
5205 
5206   // Make sure none of the operands are un-outlinable.
5207   for (const MachineOperand &MOP : MI.operands()) {
5208     if (MOP.isCPI() || MOP.isJTI() || MOP.isCFIIndex() || MOP.isFI() ||
5209         MOP.isTargetIndex())
5210       return outliner::InstrType::Illegal;
5211 
5212     // If it uses LR or W30 explicitly, then don't touch it.
5213     if (MOP.isReg() && !MOP.isImplicit() &&
5214         (MOP.getReg() == AArch64::LR || MOP.getReg() == AArch64::W30))
5215       return outliner::InstrType::Illegal;
5216   }
5217 
5218   // Special cases for instructions that can always be outlined, but will fail
5219   // the later tests. e.g, ADRPs, which are PC-relative use LR, but can always
5220   // be outlined because they don't require a *specific* value to be in LR.
5221   if (MI.getOpcode() == AArch64::ADRP)
5222     return outliner::InstrType::Legal;
5223 
5224   // If MI is a call we might be able to outline it. We don't want to outline
5225   // any calls that rely on the position of items on the stack. When we outline
5226   // something containing a call, we have to emit a save and restore of LR in
5227   // the outlined function. Currently, this always happens by saving LR to the
5228   // stack. Thus, if we outline, say, half the parameters for a function call
5229   // plus the call, then we'll break the callee's expectations for the layout
5230   // of the stack.
5231   //
5232   // FIXME: Allow calls to functions which construct a stack frame, as long
5233   // as they don't access arguments on the stack.
5234   // FIXME: Figure out some way to analyze functions defined in other modules.
5235   // We should be able to compute the memory usage based on the IR calling
5236   // convention, even if we can't see the definition.
5237   if (MI.isCall()) {
5238     // Get the function associated with the call. Look at each operand and find
5239     // the one that represents the callee and get its name.
5240     const Function *Callee = nullptr;
5241     for (const MachineOperand &MOP : MI.operands()) {
5242       if (MOP.isGlobal()) {
5243         Callee = dyn_cast<Function>(MOP.getGlobal());
5244         break;
5245       }
5246     }
5247 
5248     // Never outline calls to mcount.  There isn't any rule that would require
5249     // this, but the Linux kernel's "ftrace" feature depends on it.
5250     if (Callee && Callee->getName() == "\01_mcount")
5251       return outliner::InstrType::Illegal;
5252 
5253     // If we don't know anything about the callee, assume it depends on the
5254     // stack layout of the caller. In that case, it's only legal to outline
5255     // as a tail-call.  Whitelist the call instructions we know about so we
5256     // don't get unexpected results with call pseudo-instructions.
5257     auto UnknownCallOutlineType = outliner::InstrType::Illegal;
5258     if (MI.getOpcode() == AArch64::BLR || MI.getOpcode() == AArch64::BL)
5259       UnknownCallOutlineType = outliner::InstrType::LegalTerminator;
5260 
5261     if (!Callee)
5262       return UnknownCallOutlineType;
5263 
5264     // We have a function we have information about. Check it if it's something
5265     // can safely outline.
5266     MachineFunction *CalleeMF = MF->getMMI().getMachineFunction(*Callee);
5267 
5268     // We don't know what's going on with the callee at all. Don't touch it.
5269     if (!CalleeMF)
5270       return UnknownCallOutlineType;
5271 
5272     // Check if we know anything about the callee saves on the function. If we
5273     // don't, then don't touch it, since that implies that we haven't
5274     // computed anything about its stack frame yet.
5275     MachineFrameInfo &MFI = CalleeMF->getFrameInfo();
5276     if (!MFI.isCalleeSavedInfoValid() || MFI.getStackSize() > 0 ||
5277         MFI.getNumObjects() > 0)
5278       return UnknownCallOutlineType;
5279 
5280     // At this point, we can say that CalleeMF ought to not pass anything on the
5281     // stack. Therefore, we can outline it.
5282     return outliner::InstrType::Legal;
5283   }
5284 
5285   // Don't outline positions.
5286   if (MI.isPosition())
5287     return outliner::InstrType::Illegal;
5288 
5289   // Don't touch the link register or W30.
5290   if (MI.readsRegister(AArch64::W30, &getRegisterInfo()) ||
5291       MI.modifiesRegister(AArch64::W30, &getRegisterInfo()))
5292     return outliner::InstrType::Illegal;
5293 
5294   // Don't outline BTI instructions, because that will prevent the outlining
5295   // site from being indirectly callable.
5296   if (MI.getOpcode() == AArch64::HINT) {
5297     int64_t Imm = MI.getOperand(0).getImm();
5298     if (Imm == 32 || Imm == 34 || Imm == 36 || Imm == 38)
5299       return outliner::InstrType::Illegal;
5300   }
5301 
5302   return outliner::InstrType::Legal;
5303 }
5304 
5305 void AArch64InstrInfo::fixupPostOutline(MachineBasicBlock &MBB) const {
5306   for (MachineInstr &MI : MBB) {
5307     const MachineOperand *Base;
5308     unsigned Width;
5309     int64_t Offset;
5310 
5311     // Is this a load or store with an immediate offset with SP as the base?
5312     if (!MI.mayLoadOrStore() ||
5313         !getMemOperandWithOffsetWidth(MI, Base, Offset, Width, &RI) ||
5314         (Base->isReg() && Base->getReg() != AArch64::SP))
5315       continue;
5316 
5317     // It is, so we have to fix it up.
5318     unsigned Scale;
5319     int64_t Dummy1, Dummy2;
5320 
5321     MachineOperand &StackOffsetOperand = getMemOpBaseRegImmOfsOffsetOperand(MI);
5322     assert(StackOffsetOperand.isImm() && "Stack offset wasn't immediate!");
5323     getMemOpInfo(MI.getOpcode(), Scale, Width, Dummy1, Dummy2);
5324     assert(Scale != 0 && "Unexpected opcode!");
5325 
5326     // We've pushed the return address to the stack, so add 16 to the offset.
5327     // This is safe, since we already checked if it would overflow when we
5328     // checked if this instruction was legal to outline.
5329     int64_t NewImm = (Offset + 16) / Scale;
5330     StackOffsetOperand.setImm(NewImm);
5331   }
5332 }
5333 
5334 void AArch64InstrInfo::buildOutlinedFrame(
5335     MachineBasicBlock &MBB, MachineFunction &MF,
5336     const outliner::OutlinedFunction &OF) const {
5337   // For thunk outlining, rewrite the last instruction from a call to a
5338   // tail-call.
5339   if (OF.FrameConstructionID == MachineOutlinerThunk) {
5340     MachineInstr *Call = &*--MBB.instr_end();
5341     unsigned TailOpcode;
5342     if (Call->getOpcode() == AArch64::BL) {
5343       TailOpcode = AArch64::TCRETURNdi;
5344     } else {
5345       assert(Call->getOpcode() == AArch64::BLR);
5346       TailOpcode = AArch64::TCRETURNriALL;
5347     }
5348     MachineInstr *TC = BuildMI(MF, DebugLoc(), get(TailOpcode))
5349                             .add(Call->getOperand(0))
5350                             .addImm(0);
5351     MBB.insert(MBB.end(), TC);
5352     Call->eraseFromParent();
5353   }
5354 
5355   // Is there a call in the outlined range?
5356   auto IsNonTailCall = [](MachineInstr &MI) {
5357     return MI.isCall() && !MI.isReturn();
5358   };
5359   if (std::any_of(MBB.instr_begin(), MBB.instr_end(), IsNonTailCall)) {
5360     // Fix up the instructions in the range, since we're going to modify the
5361     // stack.
5362     assert(OF.FrameConstructionID != MachineOutlinerDefault &&
5363            "Can only fix up stack references once");
5364     fixupPostOutline(MBB);
5365 
5366     // LR has to be a live in so that we can save it.
5367     MBB.addLiveIn(AArch64::LR);
5368 
5369     MachineBasicBlock::iterator It = MBB.begin();
5370     MachineBasicBlock::iterator Et = MBB.end();
5371 
5372     if (OF.FrameConstructionID == MachineOutlinerTailCall ||
5373         OF.FrameConstructionID == MachineOutlinerThunk)
5374       Et = std::prev(MBB.end());
5375 
5376     // Insert a save before the outlined region
5377     MachineInstr *STRXpre = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre))
5378                                 .addReg(AArch64::SP, RegState::Define)
5379                                 .addReg(AArch64::LR)
5380                                 .addReg(AArch64::SP)
5381                                 .addImm(-16);
5382     It = MBB.insert(It, STRXpre);
5383 
5384     const TargetSubtargetInfo &STI = MF.getSubtarget();
5385     const MCRegisterInfo *MRI = STI.getRegisterInfo();
5386     unsigned DwarfReg = MRI->getDwarfRegNum(AArch64::LR, true);
5387 
5388     // Add a CFI saying the stack was moved 16 B down.
5389     int64_t StackPosEntry =
5390         MF.addFrameInst(MCCFIInstruction::createDefCfaOffset(nullptr, 16));
5391     BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION))
5392         .addCFIIndex(StackPosEntry)
5393         .setMIFlags(MachineInstr::FrameSetup);
5394 
5395     // Add a CFI saying that the LR that we want to find is now 16 B higher than
5396     // before.
5397     int64_t LRPosEntry =
5398         MF.addFrameInst(MCCFIInstruction::createOffset(nullptr, DwarfReg, 16));
5399     BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION))
5400         .addCFIIndex(LRPosEntry)
5401         .setMIFlags(MachineInstr::FrameSetup);
5402 
5403     // Insert a restore before the terminator for the function.
5404     MachineInstr *LDRXpost = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost))
5405                                  .addReg(AArch64::SP, RegState::Define)
5406                                  .addReg(AArch64::LR, RegState::Define)
5407                                  .addReg(AArch64::SP)
5408                                  .addImm(16);
5409     Et = MBB.insert(Et, LDRXpost);
5410   }
5411 
5412   // If this is a tail call outlined function, then there's already a return.
5413   if (OF.FrameConstructionID == MachineOutlinerTailCall ||
5414       OF.FrameConstructionID == MachineOutlinerThunk)
5415     return;
5416 
5417   // It's not a tail call, so we have to insert the return ourselves.
5418   MachineInstr *ret = BuildMI(MF, DebugLoc(), get(AArch64::RET))
5419                           .addReg(AArch64::LR, RegState::Undef);
5420   MBB.insert(MBB.end(), ret);
5421 
5422   // Did we have to modify the stack by saving the link register?
5423   if (OF.FrameConstructionID != MachineOutlinerDefault)
5424     return;
5425 
5426   // We modified the stack.
5427   // Walk over the basic block and fix up all the stack accesses.
5428   fixupPostOutline(MBB);
5429 }
5430 
5431 MachineBasicBlock::iterator AArch64InstrInfo::insertOutlinedCall(
5432     Module &M, MachineBasicBlock &MBB, MachineBasicBlock::iterator &It,
5433     MachineFunction &MF, const outliner::Candidate &C) const {
5434 
5435   // Are we tail calling?
5436   if (C.CallConstructionID == MachineOutlinerTailCall) {
5437     // If yes, then we can just branch to the label.
5438     It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::TCRETURNdi))
5439                             .addGlobalAddress(M.getNamedValue(MF.getName()))
5440                             .addImm(0));
5441     return It;
5442   }
5443 
5444   // Are we saving the link register?
5445   if (C.CallConstructionID == MachineOutlinerNoLRSave ||
5446       C.CallConstructionID == MachineOutlinerThunk) {
5447     // No, so just insert the call.
5448     It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL))
5449                             .addGlobalAddress(M.getNamedValue(MF.getName())));
5450     return It;
5451   }
5452 
5453   // We want to return the spot where we inserted the call.
5454   MachineBasicBlock::iterator CallPt;
5455 
5456   // Instructions for saving and restoring LR around the call instruction we're
5457   // going to insert.
5458   MachineInstr *Save;
5459   MachineInstr *Restore;
5460   // Can we save to a register?
5461   if (C.CallConstructionID == MachineOutlinerRegSave) {
5462     // FIXME: This logic should be sunk into a target-specific interface so that
5463     // we don't have to recompute the register.
5464     unsigned Reg = findRegisterToSaveLRTo(C);
5465     assert(Reg != 0 && "No callee-saved register available?");
5466 
5467     // Save and restore LR from that register.
5468     Save = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), Reg)
5469                .addReg(AArch64::XZR)
5470                .addReg(AArch64::LR)
5471                .addImm(0);
5472     Restore = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), AArch64::LR)
5473                 .addReg(AArch64::XZR)
5474                 .addReg(Reg)
5475                 .addImm(0);
5476   } else {
5477     // We have the default case. Save and restore from SP.
5478     Save = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre))
5479                .addReg(AArch64::SP, RegState::Define)
5480                .addReg(AArch64::LR)
5481                .addReg(AArch64::SP)
5482                .addImm(-16);
5483     Restore = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost))
5484                   .addReg(AArch64::SP, RegState::Define)
5485                   .addReg(AArch64::LR, RegState::Define)
5486                   .addReg(AArch64::SP)
5487                   .addImm(16);
5488   }
5489 
5490   It = MBB.insert(It, Save);
5491   It++;
5492 
5493   // Insert the call.
5494   It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL))
5495                           .addGlobalAddress(M.getNamedValue(MF.getName())));
5496   CallPt = It;
5497   It++;
5498 
5499   It = MBB.insert(It, Restore);
5500   return CallPt;
5501 }
5502 
5503 bool AArch64InstrInfo::shouldOutlineFromFunctionByDefault(
5504   MachineFunction &MF) const {
5505   return MF.getFunction().hasMinSize();
5506 }
5507 
5508 #define GET_INSTRINFO_HELPERS
5509 #include "AArch64GenInstrInfo.inc"
5510