1 //===- AArch64InstrInfo.cpp - AArch64 Instruction Information -------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This file contains the AArch64 implementation of the TargetInstrInfo class.
10 //
11 //===----------------------------------------------------------------------===//
12 
13 #include "AArch64InstrInfo.h"
14 #include "AArch64MachineFunctionInfo.h"
15 #include "AArch64Subtarget.h"
16 #include "MCTargetDesc/AArch64AddressingModes.h"
17 #include "Utils/AArch64BaseInfo.h"
18 #include "llvm/ADT/ArrayRef.h"
19 #include "llvm/ADT/STLExtras.h"
20 #include "llvm/ADT/SmallVector.h"
21 #include "llvm/CodeGen/MachineBasicBlock.h"
22 #include "llvm/CodeGen/MachineFrameInfo.h"
23 #include "llvm/CodeGen/MachineFunction.h"
24 #include "llvm/CodeGen/MachineInstr.h"
25 #include "llvm/CodeGen/MachineInstrBuilder.h"
26 #include "llvm/CodeGen/MachineMemOperand.h"
27 #include "llvm/CodeGen/MachineModuleInfo.h"
28 #include "llvm/CodeGen/MachineOperand.h"
29 #include "llvm/CodeGen/MachineRegisterInfo.h"
30 #include "llvm/CodeGen/StackMaps.h"
31 #include "llvm/CodeGen/TargetRegisterInfo.h"
32 #include "llvm/CodeGen/TargetSubtargetInfo.h"
33 #include "llvm/IR/DebugInfoMetadata.h"
34 #include "llvm/IR/DebugLoc.h"
35 #include "llvm/IR/GlobalValue.h"
36 #include "llvm/MC/MCAsmInfo.h"
37 #include "llvm/MC/MCInst.h"
38 #include "llvm/MC/MCInstrDesc.h"
39 #include "llvm/Support/Casting.h"
40 #include "llvm/Support/CodeGen.h"
41 #include "llvm/Support/CommandLine.h"
42 #include "llvm/Support/Compiler.h"
43 #include "llvm/Support/ErrorHandling.h"
44 #include "llvm/Support/MathExtras.h"
45 #include "llvm/Target/TargetMachine.h"
46 #include "llvm/Target/TargetOptions.h"
47 #include <cassert>
48 #include <cstdint>
49 #include <iterator>
50 #include <utility>
51 
52 using namespace llvm;
53 
54 #define GET_INSTRINFO_CTOR_DTOR
55 #include "AArch64GenInstrInfo.inc"
56 
57 static cl::opt<unsigned> TBZDisplacementBits(
58     "aarch64-tbz-offset-bits", cl::Hidden, cl::init(14),
59     cl::desc("Restrict range of TB[N]Z instructions (DEBUG)"));
60 
61 static cl::opt<unsigned> CBZDisplacementBits(
62     "aarch64-cbz-offset-bits", cl::Hidden, cl::init(19),
63     cl::desc("Restrict range of CB[N]Z instructions (DEBUG)"));
64 
65 static cl::opt<unsigned>
66     BCCDisplacementBits("aarch64-bcc-offset-bits", cl::Hidden, cl::init(19),
67                         cl::desc("Restrict range of Bcc instructions (DEBUG)"));
68 
69 AArch64InstrInfo::AArch64InstrInfo(const AArch64Subtarget &STI)
70     : AArch64GenInstrInfo(AArch64::ADJCALLSTACKDOWN, AArch64::ADJCALLSTACKUP,
71                           AArch64::CATCHRET),
72       RI(STI.getTargetTriple()), Subtarget(STI) {}
73 
74 /// GetInstSize - Return the number of bytes of code the specified
75 /// instruction may be.  This returns the maximum number of bytes.
76 unsigned AArch64InstrInfo::getInstSizeInBytes(const MachineInstr &MI) const {
77   const MachineBasicBlock &MBB = *MI.getParent();
78   const MachineFunction *MF = MBB.getParent();
79   const MCAsmInfo *MAI = MF->getTarget().getMCAsmInfo();
80 
81   {
82     auto Op = MI.getOpcode();
83     if (Op == AArch64::INLINEASM || Op == AArch64::INLINEASM_BR)
84       return getInlineAsmLength(MI.getOperand(0).getSymbolName(), *MAI);
85   }
86 
87   // Meta-instructions emit no code.
88   if (MI.isMetaInstruction())
89     return 0;
90 
91   // FIXME: We currently only handle pseudoinstructions that don't get expanded
92   //        before the assembly printer.
93   unsigned NumBytes = 0;
94   const MCInstrDesc &Desc = MI.getDesc();
95   switch (Desc.getOpcode()) {
96   default:
97     // Anything not explicitly designated otherwise is a normal 4-byte insn.
98     NumBytes = 4;
99     break;
100   case TargetOpcode::STACKMAP:
101     // The upper bound for a stackmap intrinsic is the full length of its shadow
102     NumBytes = StackMapOpers(&MI).getNumPatchBytes();
103     assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
104     break;
105   case TargetOpcode::PATCHPOINT:
106     // The size of the patchpoint intrinsic is the number of bytes requested
107     NumBytes = PatchPointOpers(&MI).getNumPatchBytes();
108     assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
109     break;
110   case TargetOpcode::STATEPOINT:
111     NumBytes = StatepointOpers(&MI).getNumPatchBytes();
112     assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
113     // No patch bytes means a normal call inst is emitted
114     if (NumBytes == 0)
115       NumBytes = 4;
116     break;
117   case AArch64::TLSDESC_CALLSEQ:
118     // This gets lowered to an instruction sequence which takes 16 bytes
119     NumBytes = 16;
120     break;
121   case AArch64::SpeculationBarrierISBDSBEndBB:
122     // This gets lowered to 2 4-byte instructions.
123     NumBytes = 8;
124     break;
125   case AArch64::SpeculationBarrierSBEndBB:
126     // This gets lowered to 1 4-byte instructions.
127     NumBytes = 4;
128     break;
129   case AArch64::JumpTableDest32:
130   case AArch64::JumpTableDest16:
131   case AArch64::JumpTableDest8:
132     NumBytes = 12;
133     break;
134   case AArch64::SPACE:
135     NumBytes = MI.getOperand(1).getImm();
136     break;
137   case TargetOpcode::BUNDLE:
138     NumBytes = getInstBundleLength(MI);
139     break;
140   }
141 
142   return NumBytes;
143 }
144 
145 unsigned AArch64InstrInfo::getInstBundleLength(const MachineInstr &MI) const {
146   unsigned Size = 0;
147   MachineBasicBlock::const_instr_iterator I = MI.getIterator();
148   MachineBasicBlock::const_instr_iterator E = MI.getParent()->instr_end();
149   while (++I != E && I->isInsideBundle()) {
150     assert(!I->isBundle() && "No nested bundle!");
151     Size += getInstSizeInBytes(*I);
152   }
153   return Size;
154 }
155 
156 static void parseCondBranch(MachineInstr *LastInst, MachineBasicBlock *&Target,
157                             SmallVectorImpl<MachineOperand> &Cond) {
158   // Block ends with fall-through condbranch.
159   switch (LastInst->getOpcode()) {
160   default:
161     llvm_unreachable("Unknown branch instruction?");
162   case AArch64::Bcc:
163     Target = LastInst->getOperand(1).getMBB();
164     Cond.push_back(LastInst->getOperand(0));
165     break;
166   case AArch64::CBZW:
167   case AArch64::CBZX:
168   case AArch64::CBNZW:
169   case AArch64::CBNZX:
170     Target = LastInst->getOperand(1).getMBB();
171     Cond.push_back(MachineOperand::CreateImm(-1));
172     Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode()));
173     Cond.push_back(LastInst->getOperand(0));
174     break;
175   case AArch64::TBZW:
176   case AArch64::TBZX:
177   case AArch64::TBNZW:
178   case AArch64::TBNZX:
179     Target = LastInst->getOperand(2).getMBB();
180     Cond.push_back(MachineOperand::CreateImm(-1));
181     Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode()));
182     Cond.push_back(LastInst->getOperand(0));
183     Cond.push_back(LastInst->getOperand(1));
184   }
185 }
186 
187 static unsigned getBranchDisplacementBits(unsigned Opc) {
188   switch (Opc) {
189   default:
190     llvm_unreachable("unexpected opcode!");
191   case AArch64::B:
192     return 64;
193   case AArch64::TBNZW:
194   case AArch64::TBZW:
195   case AArch64::TBNZX:
196   case AArch64::TBZX:
197     return TBZDisplacementBits;
198   case AArch64::CBNZW:
199   case AArch64::CBZW:
200   case AArch64::CBNZX:
201   case AArch64::CBZX:
202     return CBZDisplacementBits;
203   case AArch64::Bcc:
204     return BCCDisplacementBits;
205   }
206 }
207 
208 bool AArch64InstrInfo::isBranchOffsetInRange(unsigned BranchOp,
209                                              int64_t BrOffset) const {
210   unsigned Bits = getBranchDisplacementBits(BranchOp);
211   assert(Bits >= 3 && "max branch displacement must be enough to jump"
212                       "over conditional branch expansion");
213   return isIntN(Bits, BrOffset / 4);
214 }
215 
216 MachineBasicBlock *
217 AArch64InstrInfo::getBranchDestBlock(const MachineInstr &MI) const {
218   switch (MI.getOpcode()) {
219   default:
220     llvm_unreachable("unexpected opcode!");
221   case AArch64::B:
222     return MI.getOperand(0).getMBB();
223   case AArch64::TBZW:
224   case AArch64::TBNZW:
225   case AArch64::TBZX:
226   case AArch64::TBNZX:
227     return MI.getOperand(2).getMBB();
228   case AArch64::CBZW:
229   case AArch64::CBNZW:
230   case AArch64::CBZX:
231   case AArch64::CBNZX:
232   case AArch64::Bcc:
233     return MI.getOperand(1).getMBB();
234   }
235 }
236 
237 // Branch analysis.
238 bool AArch64InstrInfo::analyzeBranch(MachineBasicBlock &MBB,
239                                      MachineBasicBlock *&TBB,
240                                      MachineBasicBlock *&FBB,
241                                      SmallVectorImpl<MachineOperand> &Cond,
242                                      bool AllowModify) const {
243   // If the block has no terminators, it just falls into the block after it.
244   MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
245   if (I == MBB.end())
246     return false;
247 
248   // Skip over SpeculationBarrierEndBB terminators
249   if (I->getOpcode() == AArch64::SpeculationBarrierISBDSBEndBB ||
250       I->getOpcode() == AArch64::SpeculationBarrierSBEndBB) {
251     --I;
252   }
253 
254   if (!isUnpredicatedTerminator(*I))
255     return false;
256 
257   // Get the last instruction in the block.
258   MachineInstr *LastInst = &*I;
259 
260   // If there is only one terminator instruction, process it.
261   unsigned LastOpc = LastInst->getOpcode();
262   if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
263     if (isUncondBranchOpcode(LastOpc)) {
264       TBB = LastInst->getOperand(0).getMBB();
265       return false;
266     }
267     if (isCondBranchOpcode(LastOpc)) {
268       // Block ends with fall-through condbranch.
269       parseCondBranch(LastInst, TBB, Cond);
270       return false;
271     }
272     return true; // Can't handle indirect branch.
273   }
274 
275   // Get the instruction before it if it is a terminator.
276   MachineInstr *SecondLastInst = &*I;
277   unsigned SecondLastOpc = SecondLastInst->getOpcode();
278 
279   // If AllowModify is true and the block ends with two or more unconditional
280   // branches, delete all but the first unconditional branch.
281   if (AllowModify && isUncondBranchOpcode(LastOpc)) {
282     while (isUncondBranchOpcode(SecondLastOpc)) {
283       LastInst->eraseFromParent();
284       LastInst = SecondLastInst;
285       LastOpc = LastInst->getOpcode();
286       if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
287         // Return now the only terminator is an unconditional branch.
288         TBB = LastInst->getOperand(0).getMBB();
289         return false;
290       } else {
291         SecondLastInst = &*I;
292         SecondLastOpc = SecondLastInst->getOpcode();
293       }
294     }
295   }
296 
297   // If we're allowed to modify and the block ends in a unconditional branch
298   // which could simply fallthrough, remove the branch.  (Note: This case only
299   // matters when we can't understand the whole sequence, otherwise it's also
300   // handled by BranchFolding.cpp.)
301   if (AllowModify && isUncondBranchOpcode(LastOpc) &&
302       MBB.isLayoutSuccessor(getBranchDestBlock(*LastInst))) {
303     LastInst->eraseFromParent();
304     LastInst = SecondLastInst;
305     LastOpc = LastInst->getOpcode();
306     if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
307       assert(!isUncondBranchOpcode(LastOpc) &&
308              "unreachable unconditional branches removed above");
309 
310       if (isCondBranchOpcode(LastOpc)) {
311         // Block ends with fall-through condbranch.
312         parseCondBranch(LastInst, TBB, Cond);
313         return false;
314       }
315       return true; // Can't handle indirect branch.
316     } else {
317       SecondLastInst = &*I;
318       SecondLastOpc = SecondLastInst->getOpcode();
319     }
320   }
321 
322   // If there are three terminators, we don't know what sort of block this is.
323   if (SecondLastInst && I != MBB.begin() && isUnpredicatedTerminator(*--I))
324     return true;
325 
326   // If the block ends with a B and a Bcc, handle it.
327   if (isCondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
328     parseCondBranch(SecondLastInst, TBB, Cond);
329     FBB = LastInst->getOperand(0).getMBB();
330     return false;
331   }
332 
333   // If the block ends with two unconditional branches, handle it.  The second
334   // one is not executed, so remove it.
335   if (isUncondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
336     TBB = SecondLastInst->getOperand(0).getMBB();
337     I = LastInst;
338     if (AllowModify)
339       I->eraseFromParent();
340     return false;
341   }
342 
343   // ...likewise if it ends with an indirect branch followed by an unconditional
344   // branch.
345   if (isIndirectBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
346     I = LastInst;
347     if (AllowModify)
348       I->eraseFromParent();
349     return true;
350   }
351 
352   // Otherwise, can't handle this.
353   return true;
354 }
355 
356 bool AArch64InstrInfo::analyzeBranchPredicate(MachineBasicBlock &MBB,
357                                               MachineBranchPredicate &MBP,
358                                               bool AllowModify) const {
359   // For the moment, handle only a block which ends with a cb(n)zx followed by
360   // a fallthrough.  Why this?  Because it is a common form.
361   // TODO: Should we handle b.cc?
362 
363   MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
364   if (I == MBB.end())
365     return true;
366 
367   // Skip over SpeculationBarrierEndBB terminators
368   if (I->getOpcode() == AArch64::SpeculationBarrierISBDSBEndBB ||
369       I->getOpcode() == AArch64::SpeculationBarrierSBEndBB) {
370     --I;
371   }
372 
373   if (!isUnpredicatedTerminator(*I))
374     return true;
375 
376   // Get the last instruction in the block.
377   MachineInstr *LastInst = &*I;
378   unsigned LastOpc = LastInst->getOpcode();
379   if (!isCondBranchOpcode(LastOpc))
380     return true;
381 
382   switch (LastOpc) {
383   default:
384     return true;
385   case AArch64::CBZW:
386   case AArch64::CBZX:
387   case AArch64::CBNZW:
388   case AArch64::CBNZX:
389     break;
390   };
391 
392   MBP.TrueDest = LastInst->getOperand(1).getMBB();
393   assert(MBP.TrueDest && "expected!");
394   MBP.FalseDest = MBB.getNextNode();
395 
396   MBP.ConditionDef = nullptr;
397   MBP.SingleUseCondition = false;
398 
399   MBP.LHS = LastInst->getOperand(0);
400   MBP.RHS = MachineOperand::CreateImm(0);
401   MBP.Predicate = LastOpc == AArch64::CBNZX ? MachineBranchPredicate::PRED_NE
402                                             : MachineBranchPredicate::PRED_EQ;
403   return false;
404 }
405 
406 bool AArch64InstrInfo::reverseBranchCondition(
407     SmallVectorImpl<MachineOperand> &Cond) const {
408   if (Cond[0].getImm() != -1) {
409     // Regular Bcc
410     AArch64CC::CondCode CC = (AArch64CC::CondCode)(int)Cond[0].getImm();
411     Cond[0].setImm(AArch64CC::getInvertedCondCode(CC));
412   } else {
413     // Folded compare-and-branch
414     switch (Cond[1].getImm()) {
415     default:
416       llvm_unreachable("Unknown conditional branch!");
417     case AArch64::CBZW:
418       Cond[1].setImm(AArch64::CBNZW);
419       break;
420     case AArch64::CBNZW:
421       Cond[1].setImm(AArch64::CBZW);
422       break;
423     case AArch64::CBZX:
424       Cond[1].setImm(AArch64::CBNZX);
425       break;
426     case AArch64::CBNZX:
427       Cond[1].setImm(AArch64::CBZX);
428       break;
429     case AArch64::TBZW:
430       Cond[1].setImm(AArch64::TBNZW);
431       break;
432     case AArch64::TBNZW:
433       Cond[1].setImm(AArch64::TBZW);
434       break;
435     case AArch64::TBZX:
436       Cond[1].setImm(AArch64::TBNZX);
437       break;
438     case AArch64::TBNZX:
439       Cond[1].setImm(AArch64::TBZX);
440       break;
441     }
442   }
443 
444   return false;
445 }
446 
447 unsigned AArch64InstrInfo::removeBranch(MachineBasicBlock &MBB,
448                                         int *BytesRemoved) const {
449   MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
450   if (I == MBB.end())
451     return 0;
452 
453   if (!isUncondBranchOpcode(I->getOpcode()) &&
454       !isCondBranchOpcode(I->getOpcode()))
455     return 0;
456 
457   // Remove the branch.
458   I->eraseFromParent();
459 
460   I = MBB.end();
461 
462   if (I == MBB.begin()) {
463     if (BytesRemoved)
464       *BytesRemoved = 4;
465     return 1;
466   }
467   --I;
468   if (!isCondBranchOpcode(I->getOpcode())) {
469     if (BytesRemoved)
470       *BytesRemoved = 4;
471     return 1;
472   }
473 
474   // Remove the branch.
475   I->eraseFromParent();
476   if (BytesRemoved)
477     *BytesRemoved = 8;
478 
479   return 2;
480 }
481 
482 void AArch64InstrInfo::instantiateCondBranch(
483     MachineBasicBlock &MBB, const DebugLoc &DL, MachineBasicBlock *TBB,
484     ArrayRef<MachineOperand> Cond) const {
485   if (Cond[0].getImm() != -1) {
486     // Regular Bcc
487     BuildMI(&MBB, DL, get(AArch64::Bcc)).addImm(Cond[0].getImm()).addMBB(TBB);
488   } else {
489     // Folded compare-and-branch
490     // Note that we use addOperand instead of addReg to keep the flags.
491     const MachineInstrBuilder MIB =
492         BuildMI(&MBB, DL, get(Cond[1].getImm())).add(Cond[2]);
493     if (Cond.size() > 3)
494       MIB.addImm(Cond[3].getImm());
495     MIB.addMBB(TBB);
496   }
497 }
498 
499 unsigned AArch64InstrInfo::insertBranch(
500     MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB,
501     ArrayRef<MachineOperand> Cond, const DebugLoc &DL, int *BytesAdded) const {
502   // Shouldn't be a fall through.
503   assert(TBB && "insertBranch must not be told to insert a fallthrough");
504 
505   if (!FBB) {
506     if (Cond.empty()) // Unconditional branch?
507       BuildMI(&MBB, DL, get(AArch64::B)).addMBB(TBB);
508     else
509       instantiateCondBranch(MBB, DL, TBB, Cond);
510 
511     if (BytesAdded)
512       *BytesAdded = 4;
513 
514     return 1;
515   }
516 
517   // Two-way conditional branch.
518   instantiateCondBranch(MBB, DL, TBB, Cond);
519   BuildMI(&MBB, DL, get(AArch64::B)).addMBB(FBB);
520 
521   if (BytesAdded)
522     *BytesAdded = 8;
523 
524   return 2;
525 }
526 
527 // Find the original register that VReg is copied from.
528 static unsigned removeCopies(const MachineRegisterInfo &MRI, unsigned VReg) {
529   while (Register::isVirtualRegister(VReg)) {
530     const MachineInstr *DefMI = MRI.getVRegDef(VReg);
531     if (!DefMI->isFullCopy())
532       return VReg;
533     VReg = DefMI->getOperand(1).getReg();
534   }
535   return VReg;
536 }
537 
538 // Determine if VReg is defined by an instruction that can be folded into a
539 // csel instruction. If so, return the folded opcode, and the replacement
540 // register.
541 static unsigned canFoldIntoCSel(const MachineRegisterInfo &MRI, unsigned VReg,
542                                 unsigned *NewVReg = nullptr) {
543   VReg = removeCopies(MRI, VReg);
544   if (!Register::isVirtualRegister(VReg))
545     return 0;
546 
547   bool Is64Bit = AArch64::GPR64allRegClass.hasSubClassEq(MRI.getRegClass(VReg));
548   const MachineInstr *DefMI = MRI.getVRegDef(VReg);
549   unsigned Opc = 0;
550   unsigned SrcOpNum = 0;
551   switch (DefMI->getOpcode()) {
552   case AArch64::ADDSXri:
553   case AArch64::ADDSWri:
554     // if NZCV is used, do not fold.
555     if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1)
556       return 0;
557     // fall-through to ADDXri and ADDWri.
558     LLVM_FALLTHROUGH;
559   case AArch64::ADDXri:
560   case AArch64::ADDWri:
561     // add x, 1 -> csinc.
562     if (!DefMI->getOperand(2).isImm() || DefMI->getOperand(2).getImm() != 1 ||
563         DefMI->getOperand(3).getImm() != 0)
564       return 0;
565     SrcOpNum = 1;
566     Opc = Is64Bit ? AArch64::CSINCXr : AArch64::CSINCWr;
567     break;
568 
569   case AArch64::ORNXrr:
570   case AArch64::ORNWrr: {
571     // not x -> csinv, represented as orn dst, xzr, src.
572     unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg());
573     if (ZReg != AArch64::XZR && ZReg != AArch64::WZR)
574       return 0;
575     SrcOpNum = 2;
576     Opc = Is64Bit ? AArch64::CSINVXr : AArch64::CSINVWr;
577     break;
578   }
579 
580   case AArch64::SUBSXrr:
581   case AArch64::SUBSWrr:
582     // if NZCV is used, do not fold.
583     if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1)
584       return 0;
585     // fall-through to SUBXrr and SUBWrr.
586     LLVM_FALLTHROUGH;
587   case AArch64::SUBXrr:
588   case AArch64::SUBWrr: {
589     // neg x -> csneg, represented as sub dst, xzr, src.
590     unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg());
591     if (ZReg != AArch64::XZR && ZReg != AArch64::WZR)
592       return 0;
593     SrcOpNum = 2;
594     Opc = Is64Bit ? AArch64::CSNEGXr : AArch64::CSNEGWr;
595     break;
596   }
597   default:
598     return 0;
599   }
600   assert(Opc && SrcOpNum && "Missing parameters");
601 
602   if (NewVReg)
603     *NewVReg = DefMI->getOperand(SrcOpNum).getReg();
604   return Opc;
605 }
606 
607 bool AArch64InstrInfo::canInsertSelect(const MachineBasicBlock &MBB,
608                                        ArrayRef<MachineOperand> Cond,
609                                        Register DstReg, Register TrueReg,
610                                        Register FalseReg, int &CondCycles,
611                                        int &TrueCycles,
612                                        int &FalseCycles) const {
613   // Check register classes.
614   const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
615   const TargetRegisterClass *RC =
616       RI.getCommonSubClass(MRI.getRegClass(TrueReg), MRI.getRegClass(FalseReg));
617   if (!RC)
618     return false;
619 
620   // Also need to check the dest regclass, in case we're trying to optimize
621   // something like:
622   // %1(gpr) = PHI %2(fpr), bb1, %(fpr), bb2
623   if (!RI.getCommonSubClass(RC, MRI.getRegClass(DstReg)))
624     return false;
625 
626   // Expanding cbz/tbz requires an extra cycle of latency on the condition.
627   unsigned ExtraCondLat = Cond.size() != 1;
628 
629   // GPRs are handled by csel.
630   // FIXME: Fold in x+1, -x, and ~x when applicable.
631   if (AArch64::GPR64allRegClass.hasSubClassEq(RC) ||
632       AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
633     // Single-cycle csel, csinc, csinv, and csneg.
634     CondCycles = 1 + ExtraCondLat;
635     TrueCycles = FalseCycles = 1;
636     if (canFoldIntoCSel(MRI, TrueReg))
637       TrueCycles = 0;
638     else if (canFoldIntoCSel(MRI, FalseReg))
639       FalseCycles = 0;
640     return true;
641   }
642 
643   // Scalar floating point is handled by fcsel.
644   // FIXME: Form fabs, fmin, and fmax when applicable.
645   if (AArch64::FPR64RegClass.hasSubClassEq(RC) ||
646       AArch64::FPR32RegClass.hasSubClassEq(RC)) {
647     CondCycles = 5 + ExtraCondLat;
648     TrueCycles = FalseCycles = 2;
649     return true;
650   }
651 
652   // Can't do vectors.
653   return false;
654 }
655 
656 void AArch64InstrInfo::insertSelect(MachineBasicBlock &MBB,
657                                     MachineBasicBlock::iterator I,
658                                     const DebugLoc &DL, Register DstReg,
659                                     ArrayRef<MachineOperand> Cond,
660                                     Register TrueReg, Register FalseReg) const {
661   MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
662 
663   // Parse the condition code, see parseCondBranch() above.
664   AArch64CC::CondCode CC;
665   switch (Cond.size()) {
666   default:
667     llvm_unreachable("Unknown condition opcode in Cond");
668   case 1: // b.cc
669     CC = AArch64CC::CondCode(Cond[0].getImm());
670     break;
671   case 3: { // cbz/cbnz
672     // We must insert a compare against 0.
673     bool Is64Bit;
674     switch (Cond[1].getImm()) {
675     default:
676       llvm_unreachable("Unknown branch opcode in Cond");
677     case AArch64::CBZW:
678       Is64Bit = false;
679       CC = AArch64CC::EQ;
680       break;
681     case AArch64::CBZX:
682       Is64Bit = true;
683       CC = AArch64CC::EQ;
684       break;
685     case AArch64::CBNZW:
686       Is64Bit = false;
687       CC = AArch64CC::NE;
688       break;
689     case AArch64::CBNZX:
690       Is64Bit = true;
691       CC = AArch64CC::NE;
692       break;
693     }
694     Register SrcReg = Cond[2].getReg();
695     if (Is64Bit) {
696       // cmp reg, #0 is actually subs xzr, reg, #0.
697       MRI.constrainRegClass(SrcReg, &AArch64::GPR64spRegClass);
698       BuildMI(MBB, I, DL, get(AArch64::SUBSXri), AArch64::XZR)
699           .addReg(SrcReg)
700           .addImm(0)
701           .addImm(0);
702     } else {
703       MRI.constrainRegClass(SrcReg, &AArch64::GPR32spRegClass);
704       BuildMI(MBB, I, DL, get(AArch64::SUBSWri), AArch64::WZR)
705           .addReg(SrcReg)
706           .addImm(0)
707           .addImm(0);
708     }
709     break;
710   }
711   case 4: { // tbz/tbnz
712     // We must insert a tst instruction.
713     switch (Cond[1].getImm()) {
714     default:
715       llvm_unreachable("Unknown branch opcode in Cond");
716     case AArch64::TBZW:
717     case AArch64::TBZX:
718       CC = AArch64CC::EQ;
719       break;
720     case AArch64::TBNZW:
721     case AArch64::TBNZX:
722       CC = AArch64CC::NE;
723       break;
724     }
725     // cmp reg, #foo is actually ands xzr, reg, #1<<foo.
726     if (Cond[1].getImm() == AArch64::TBZW || Cond[1].getImm() == AArch64::TBNZW)
727       BuildMI(MBB, I, DL, get(AArch64::ANDSWri), AArch64::WZR)
728           .addReg(Cond[2].getReg())
729           .addImm(
730               AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 32));
731     else
732       BuildMI(MBB, I, DL, get(AArch64::ANDSXri), AArch64::XZR)
733           .addReg(Cond[2].getReg())
734           .addImm(
735               AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 64));
736     break;
737   }
738   }
739 
740   unsigned Opc = 0;
741   const TargetRegisterClass *RC = nullptr;
742   bool TryFold = false;
743   if (MRI.constrainRegClass(DstReg, &AArch64::GPR64RegClass)) {
744     RC = &AArch64::GPR64RegClass;
745     Opc = AArch64::CSELXr;
746     TryFold = true;
747   } else if (MRI.constrainRegClass(DstReg, &AArch64::GPR32RegClass)) {
748     RC = &AArch64::GPR32RegClass;
749     Opc = AArch64::CSELWr;
750     TryFold = true;
751   } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR64RegClass)) {
752     RC = &AArch64::FPR64RegClass;
753     Opc = AArch64::FCSELDrrr;
754   } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR32RegClass)) {
755     RC = &AArch64::FPR32RegClass;
756     Opc = AArch64::FCSELSrrr;
757   }
758   assert(RC && "Unsupported regclass");
759 
760   // Try folding simple instructions into the csel.
761   if (TryFold) {
762     unsigned NewVReg = 0;
763     unsigned FoldedOpc = canFoldIntoCSel(MRI, TrueReg, &NewVReg);
764     if (FoldedOpc) {
765       // The folded opcodes csinc, csinc and csneg apply the operation to
766       // FalseReg, so we need to invert the condition.
767       CC = AArch64CC::getInvertedCondCode(CC);
768       TrueReg = FalseReg;
769     } else
770       FoldedOpc = canFoldIntoCSel(MRI, FalseReg, &NewVReg);
771 
772     // Fold the operation. Leave any dead instructions for DCE to clean up.
773     if (FoldedOpc) {
774       FalseReg = NewVReg;
775       Opc = FoldedOpc;
776       // The extends the live range of NewVReg.
777       MRI.clearKillFlags(NewVReg);
778     }
779   }
780 
781   // Pull all virtual register into the appropriate class.
782   MRI.constrainRegClass(TrueReg, RC);
783   MRI.constrainRegClass(FalseReg, RC);
784 
785   // Insert the csel.
786   BuildMI(MBB, I, DL, get(Opc), DstReg)
787       .addReg(TrueReg)
788       .addReg(FalseReg)
789       .addImm(CC);
790 }
791 
792 /// Returns true if a MOVi32imm or MOVi64imm can be expanded to an  ORRxx.
793 static bool canBeExpandedToORR(const MachineInstr &MI, unsigned BitSize) {
794   uint64_t Imm = MI.getOperand(1).getImm();
795   uint64_t UImm = Imm << (64 - BitSize) >> (64 - BitSize);
796   uint64_t Encoding;
797   return AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding);
798 }
799 
800 // FIXME: this implementation should be micro-architecture dependent, so a
801 // micro-architecture target hook should be introduced here in future.
802 bool AArch64InstrInfo::isAsCheapAsAMove(const MachineInstr &MI) const {
803   if (!Subtarget.hasCustomCheapAsMoveHandling())
804     return MI.isAsCheapAsAMove();
805 
806   const unsigned Opcode = MI.getOpcode();
807 
808   // Firstly, check cases gated by features.
809 
810   if (Subtarget.hasZeroCycleZeroingFP()) {
811     if (Opcode == AArch64::FMOVH0 ||
812         Opcode == AArch64::FMOVS0 ||
813         Opcode == AArch64::FMOVD0)
814       return true;
815   }
816 
817   if (Subtarget.hasZeroCycleZeroingGP()) {
818     if (Opcode == TargetOpcode::COPY &&
819         (MI.getOperand(1).getReg() == AArch64::WZR ||
820          MI.getOperand(1).getReg() == AArch64::XZR))
821       return true;
822   }
823 
824   // Secondly, check cases specific to sub-targets.
825 
826   if (Subtarget.hasExynosCheapAsMoveHandling()) {
827     if (isExynosCheapAsMove(MI))
828       return true;
829 
830     return MI.isAsCheapAsAMove();
831   }
832 
833   // Finally, check generic cases.
834 
835   switch (Opcode) {
836   default:
837     return false;
838 
839   // add/sub on register without shift
840   case AArch64::ADDWri:
841   case AArch64::ADDXri:
842   case AArch64::SUBWri:
843   case AArch64::SUBXri:
844     return (MI.getOperand(3).getImm() == 0);
845 
846   // logical ops on immediate
847   case AArch64::ANDWri:
848   case AArch64::ANDXri:
849   case AArch64::EORWri:
850   case AArch64::EORXri:
851   case AArch64::ORRWri:
852   case AArch64::ORRXri:
853     return true;
854 
855   // logical ops on register without shift
856   case AArch64::ANDWrr:
857   case AArch64::ANDXrr:
858   case AArch64::BICWrr:
859   case AArch64::BICXrr:
860   case AArch64::EONWrr:
861   case AArch64::EONXrr:
862   case AArch64::EORWrr:
863   case AArch64::EORXrr:
864   case AArch64::ORNWrr:
865   case AArch64::ORNXrr:
866   case AArch64::ORRWrr:
867   case AArch64::ORRXrr:
868     return true;
869 
870   // If MOVi32imm or MOVi64imm can be expanded into ORRWri or
871   // ORRXri, it is as cheap as MOV
872   case AArch64::MOVi32imm:
873     return canBeExpandedToORR(MI, 32);
874   case AArch64::MOVi64imm:
875     return canBeExpandedToORR(MI, 64);
876   }
877 
878   llvm_unreachable("Unknown opcode to check as cheap as a move!");
879 }
880 
881 bool AArch64InstrInfo::isFalkorShiftExtFast(const MachineInstr &MI) {
882   switch (MI.getOpcode()) {
883   default:
884     return false;
885 
886   case AArch64::ADDWrs:
887   case AArch64::ADDXrs:
888   case AArch64::ADDSWrs:
889   case AArch64::ADDSXrs: {
890     unsigned Imm = MI.getOperand(3).getImm();
891     unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
892     if (ShiftVal == 0)
893       return true;
894     return AArch64_AM::getShiftType(Imm) == AArch64_AM::LSL && ShiftVal <= 5;
895   }
896 
897   case AArch64::ADDWrx:
898   case AArch64::ADDXrx:
899   case AArch64::ADDXrx64:
900   case AArch64::ADDSWrx:
901   case AArch64::ADDSXrx:
902   case AArch64::ADDSXrx64: {
903     unsigned Imm = MI.getOperand(3).getImm();
904     switch (AArch64_AM::getArithExtendType(Imm)) {
905     default:
906       return false;
907     case AArch64_AM::UXTB:
908     case AArch64_AM::UXTH:
909     case AArch64_AM::UXTW:
910     case AArch64_AM::UXTX:
911       return AArch64_AM::getArithShiftValue(Imm) <= 4;
912     }
913   }
914 
915   case AArch64::SUBWrs:
916   case AArch64::SUBSWrs: {
917     unsigned Imm = MI.getOperand(3).getImm();
918     unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
919     return ShiftVal == 0 ||
920            (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 31);
921   }
922 
923   case AArch64::SUBXrs:
924   case AArch64::SUBSXrs: {
925     unsigned Imm = MI.getOperand(3).getImm();
926     unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
927     return ShiftVal == 0 ||
928            (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 63);
929   }
930 
931   case AArch64::SUBWrx:
932   case AArch64::SUBXrx:
933   case AArch64::SUBXrx64:
934   case AArch64::SUBSWrx:
935   case AArch64::SUBSXrx:
936   case AArch64::SUBSXrx64: {
937     unsigned Imm = MI.getOperand(3).getImm();
938     switch (AArch64_AM::getArithExtendType(Imm)) {
939     default:
940       return false;
941     case AArch64_AM::UXTB:
942     case AArch64_AM::UXTH:
943     case AArch64_AM::UXTW:
944     case AArch64_AM::UXTX:
945       return AArch64_AM::getArithShiftValue(Imm) == 0;
946     }
947   }
948 
949   case AArch64::LDRBBroW:
950   case AArch64::LDRBBroX:
951   case AArch64::LDRBroW:
952   case AArch64::LDRBroX:
953   case AArch64::LDRDroW:
954   case AArch64::LDRDroX:
955   case AArch64::LDRHHroW:
956   case AArch64::LDRHHroX:
957   case AArch64::LDRHroW:
958   case AArch64::LDRHroX:
959   case AArch64::LDRQroW:
960   case AArch64::LDRQroX:
961   case AArch64::LDRSBWroW:
962   case AArch64::LDRSBWroX:
963   case AArch64::LDRSBXroW:
964   case AArch64::LDRSBXroX:
965   case AArch64::LDRSHWroW:
966   case AArch64::LDRSHWroX:
967   case AArch64::LDRSHXroW:
968   case AArch64::LDRSHXroX:
969   case AArch64::LDRSWroW:
970   case AArch64::LDRSWroX:
971   case AArch64::LDRSroW:
972   case AArch64::LDRSroX:
973   case AArch64::LDRWroW:
974   case AArch64::LDRWroX:
975   case AArch64::LDRXroW:
976   case AArch64::LDRXroX:
977   case AArch64::PRFMroW:
978   case AArch64::PRFMroX:
979   case AArch64::STRBBroW:
980   case AArch64::STRBBroX:
981   case AArch64::STRBroW:
982   case AArch64::STRBroX:
983   case AArch64::STRDroW:
984   case AArch64::STRDroX:
985   case AArch64::STRHHroW:
986   case AArch64::STRHHroX:
987   case AArch64::STRHroW:
988   case AArch64::STRHroX:
989   case AArch64::STRQroW:
990   case AArch64::STRQroX:
991   case AArch64::STRSroW:
992   case AArch64::STRSroX:
993   case AArch64::STRWroW:
994   case AArch64::STRWroX:
995   case AArch64::STRXroW:
996   case AArch64::STRXroX: {
997     unsigned IsSigned = MI.getOperand(3).getImm();
998     return !IsSigned;
999   }
1000   }
1001 }
1002 
1003 bool AArch64InstrInfo::isSEHInstruction(const MachineInstr &MI) {
1004   unsigned Opc = MI.getOpcode();
1005   switch (Opc) {
1006     default:
1007       return false;
1008     case AArch64::SEH_StackAlloc:
1009     case AArch64::SEH_SaveFPLR:
1010     case AArch64::SEH_SaveFPLR_X:
1011     case AArch64::SEH_SaveReg:
1012     case AArch64::SEH_SaveReg_X:
1013     case AArch64::SEH_SaveRegP:
1014     case AArch64::SEH_SaveRegP_X:
1015     case AArch64::SEH_SaveFReg:
1016     case AArch64::SEH_SaveFReg_X:
1017     case AArch64::SEH_SaveFRegP:
1018     case AArch64::SEH_SaveFRegP_X:
1019     case AArch64::SEH_SetFP:
1020     case AArch64::SEH_AddFP:
1021     case AArch64::SEH_Nop:
1022     case AArch64::SEH_PrologEnd:
1023     case AArch64::SEH_EpilogStart:
1024     case AArch64::SEH_EpilogEnd:
1025       return true;
1026   }
1027 }
1028 
1029 bool AArch64InstrInfo::isCoalescableExtInstr(const MachineInstr &MI,
1030                                              Register &SrcReg, Register &DstReg,
1031                                              unsigned &SubIdx) const {
1032   switch (MI.getOpcode()) {
1033   default:
1034     return false;
1035   case AArch64::SBFMXri: // aka sxtw
1036   case AArch64::UBFMXri: // aka uxtw
1037     // Check for the 32 -> 64 bit extension case, these instructions can do
1038     // much more.
1039     if (MI.getOperand(2).getImm() != 0 || MI.getOperand(3).getImm() != 31)
1040       return false;
1041     // This is a signed or unsigned 32 -> 64 bit extension.
1042     SrcReg = MI.getOperand(1).getReg();
1043     DstReg = MI.getOperand(0).getReg();
1044     SubIdx = AArch64::sub_32;
1045     return true;
1046   }
1047 }
1048 
1049 bool AArch64InstrInfo::areMemAccessesTriviallyDisjoint(
1050     const MachineInstr &MIa, const MachineInstr &MIb) const {
1051   const TargetRegisterInfo *TRI = &getRegisterInfo();
1052   const MachineOperand *BaseOpA = nullptr, *BaseOpB = nullptr;
1053   int64_t OffsetA = 0, OffsetB = 0;
1054   unsigned WidthA = 0, WidthB = 0;
1055   bool OffsetAIsScalable = false, OffsetBIsScalable = false;
1056 
1057   assert(MIa.mayLoadOrStore() && "MIa must be a load or store.");
1058   assert(MIb.mayLoadOrStore() && "MIb must be a load or store.");
1059 
1060   if (MIa.hasUnmodeledSideEffects() || MIb.hasUnmodeledSideEffects() ||
1061       MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef())
1062     return false;
1063 
1064   // Retrieve the base, offset from the base and width. Width
1065   // is the size of memory that is being loaded/stored (e.g. 1, 2, 4, 8).  If
1066   // base are identical, and the offset of a lower memory access +
1067   // the width doesn't overlap the offset of a higher memory access,
1068   // then the memory accesses are different.
1069   // If OffsetAIsScalable and OffsetBIsScalable are both true, they
1070   // are assumed to have the same scale (vscale).
1071   if (getMemOperandWithOffsetWidth(MIa, BaseOpA, OffsetA, OffsetAIsScalable,
1072                                    WidthA, TRI) &&
1073       getMemOperandWithOffsetWidth(MIb, BaseOpB, OffsetB, OffsetBIsScalable,
1074                                    WidthB, TRI)) {
1075     if (BaseOpA->isIdenticalTo(*BaseOpB) &&
1076         OffsetAIsScalable == OffsetBIsScalable) {
1077       int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB;
1078       int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA;
1079       int LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB;
1080       if (LowOffset + LowWidth <= HighOffset)
1081         return true;
1082     }
1083   }
1084   return false;
1085 }
1086 
1087 bool AArch64InstrInfo::isSchedulingBoundary(const MachineInstr &MI,
1088                                             const MachineBasicBlock *MBB,
1089                                             const MachineFunction &MF) const {
1090   if (TargetInstrInfo::isSchedulingBoundary(MI, MBB, MF))
1091     return true;
1092   switch (MI.getOpcode()) {
1093   case AArch64::HINT:
1094     // CSDB hints are scheduling barriers.
1095     if (MI.getOperand(0).getImm() == 0x14)
1096       return true;
1097     break;
1098   case AArch64::DSB:
1099   case AArch64::ISB:
1100     // DSB and ISB also are scheduling barriers.
1101     return true;
1102   default:;
1103   }
1104   return isSEHInstruction(MI);
1105 }
1106 
1107 /// analyzeCompare - For a comparison instruction, return the source registers
1108 /// in SrcReg and SrcReg2, and the value it compares against in CmpValue.
1109 /// Return true if the comparison instruction can be analyzed.
1110 bool AArch64InstrInfo::analyzeCompare(const MachineInstr &MI, Register &SrcReg,
1111                                       Register &SrcReg2, int &CmpMask,
1112                                       int &CmpValue) const {
1113   // The first operand can be a frame index where we'd normally expect a
1114   // register.
1115   assert(MI.getNumOperands() >= 2 && "All AArch64 cmps should have 2 operands");
1116   if (!MI.getOperand(1).isReg())
1117     return false;
1118 
1119   switch (MI.getOpcode()) {
1120   default:
1121     break;
1122   case AArch64::SUBSWrr:
1123   case AArch64::SUBSWrs:
1124   case AArch64::SUBSWrx:
1125   case AArch64::SUBSXrr:
1126   case AArch64::SUBSXrs:
1127   case AArch64::SUBSXrx:
1128   case AArch64::ADDSWrr:
1129   case AArch64::ADDSWrs:
1130   case AArch64::ADDSWrx:
1131   case AArch64::ADDSXrr:
1132   case AArch64::ADDSXrs:
1133   case AArch64::ADDSXrx:
1134     // Replace SUBSWrr with SUBWrr if NZCV is not used.
1135     SrcReg = MI.getOperand(1).getReg();
1136     SrcReg2 = MI.getOperand(2).getReg();
1137     CmpMask = ~0;
1138     CmpValue = 0;
1139     return true;
1140   case AArch64::SUBSWri:
1141   case AArch64::ADDSWri:
1142   case AArch64::SUBSXri:
1143   case AArch64::ADDSXri:
1144     SrcReg = MI.getOperand(1).getReg();
1145     SrcReg2 = 0;
1146     CmpMask = ~0;
1147     // FIXME: In order to convert CmpValue to 0 or 1
1148     CmpValue = MI.getOperand(2).getImm() != 0;
1149     return true;
1150   case AArch64::ANDSWri:
1151   case AArch64::ANDSXri:
1152     // ANDS does not use the same encoding scheme as the others xxxS
1153     // instructions.
1154     SrcReg = MI.getOperand(1).getReg();
1155     SrcReg2 = 0;
1156     CmpMask = ~0;
1157     // FIXME:The return val type of decodeLogicalImmediate is uint64_t,
1158     // while the type of CmpValue is int. When converting uint64_t to int,
1159     // the high 32 bits of uint64_t will be lost.
1160     // In fact it causes a bug in spec2006-483.xalancbmk
1161     // CmpValue is only used to compare with zero in OptimizeCompareInstr
1162     CmpValue = AArch64_AM::decodeLogicalImmediate(
1163                    MI.getOperand(2).getImm(),
1164                    MI.getOpcode() == AArch64::ANDSWri ? 32 : 64) != 0;
1165     return true;
1166   }
1167 
1168   return false;
1169 }
1170 
1171 static bool UpdateOperandRegClass(MachineInstr &Instr) {
1172   MachineBasicBlock *MBB = Instr.getParent();
1173   assert(MBB && "Can't get MachineBasicBlock here");
1174   MachineFunction *MF = MBB->getParent();
1175   assert(MF && "Can't get MachineFunction here");
1176   const TargetInstrInfo *TII = MF->getSubtarget().getInstrInfo();
1177   const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo();
1178   MachineRegisterInfo *MRI = &MF->getRegInfo();
1179 
1180   for (unsigned OpIdx = 0, EndIdx = Instr.getNumOperands(); OpIdx < EndIdx;
1181        ++OpIdx) {
1182     MachineOperand &MO = Instr.getOperand(OpIdx);
1183     const TargetRegisterClass *OpRegCstraints =
1184         Instr.getRegClassConstraint(OpIdx, TII, TRI);
1185 
1186     // If there's no constraint, there's nothing to do.
1187     if (!OpRegCstraints)
1188       continue;
1189     // If the operand is a frame index, there's nothing to do here.
1190     // A frame index operand will resolve correctly during PEI.
1191     if (MO.isFI())
1192       continue;
1193 
1194     assert(MO.isReg() &&
1195            "Operand has register constraints without being a register!");
1196 
1197     Register Reg = MO.getReg();
1198     if (Register::isPhysicalRegister(Reg)) {
1199       if (!OpRegCstraints->contains(Reg))
1200         return false;
1201     } else if (!OpRegCstraints->hasSubClassEq(MRI->getRegClass(Reg)) &&
1202                !MRI->constrainRegClass(Reg, OpRegCstraints))
1203       return false;
1204   }
1205 
1206   return true;
1207 }
1208 
1209 /// Return the opcode that does not set flags when possible - otherwise
1210 /// return the original opcode. The caller is responsible to do the actual
1211 /// substitution and legality checking.
1212 static unsigned convertToNonFlagSettingOpc(const MachineInstr &MI) {
1213   // Don't convert all compare instructions, because for some the zero register
1214   // encoding becomes the sp register.
1215   bool MIDefinesZeroReg = false;
1216   if (MI.definesRegister(AArch64::WZR) || MI.definesRegister(AArch64::XZR))
1217     MIDefinesZeroReg = true;
1218 
1219   switch (MI.getOpcode()) {
1220   default:
1221     return MI.getOpcode();
1222   case AArch64::ADDSWrr:
1223     return AArch64::ADDWrr;
1224   case AArch64::ADDSWri:
1225     return MIDefinesZeroReg ? AArch64::ADDSWri : AArch64::ADDWri;
1226   case AArch64::ADDSWrs:
1227     return MIDefinesZeroReg ? AArch64::ADDSWrs : AArch64::ADDWrs;
1228   case AArch64::ADDSWrx:
1229     return AArch64::ADDWrx;
1230   case AArch64::ADDSXrr:
1231     return AArch64::ADDXrr;
1232   case AArch64::ADDSXri:
1233     return MIDefinesZeroReg ? AArch64::ADDSXri : AArch64::ADDXri;
1234   case AArch64::ADDSXrs:
1235     return MIDefinesZeroReg ? AArch64::ADDSXrs : AArch64::ADDXrs;
1236   case AArch64::ADDSXrx:
1237     return AArch64::ADDXrx;
1238   case AArch64::SUBSWrr:
1239     return AArch64::SUBWrr;
1240   case AArch64::SUBSWri:
1241     return MIDefinesZeroReg ? AArch64::SUBSWri : AArch64::SUBWri;
1242   case AArch64::SUBSWrs:
1243     return MIDefinesZeroReg ? AArch64::SUBSWrs : AArch64::SUBWrs;
1244   case AArch64::SUBSWrx:
1245     return AArch64::SUBWrx;
1246   case AArch64::SUBSXrr:
1247     return AArch64::SUBXrr;
1248   case AArch64::SUBSXri:
1249     return MIDefinesZeroReg ? AArch64::SUBSXri : AArch64::SUBXri;
1250   case AArch64::SUBSXrs:
1251     return MIDefinesZeroReg ? AArch64::SUBSXrs : AArch64::SUBXrs;
1252   case AArch64::SUBSXrx:
1253     return AArch64::SUBXrx;
1254   }
1255 }
1256 
1257 enum AccessKind { AK_Write = 0x01, AK_Read = 0x10, AK_All = 0x11 };
1258 
1259 /// True when condition flags are accessed (either by writing or reading)
1260 /// on the instruction trace starting at From and ending at To.
1261 ///
1262 /// Note: If From and To are from different blocks it's assumed CC are accessed
1263 ///       on the path.
1264 static bool areCFlagsAccessedBetweenInstrs(
1265     MachineBasicBlock::iterator From, MachineBasicBlock::iterator To,
1266     const TargetRegisterInfo *TRI, const AccessKind AccessToCheck = AK_All) {
1267   // Early exit if To is at the beginning of the BB.
1268   if (To == To->getParent()->begin())
1269     return true;
1270 
1271   // Check whether the instructions are in the same basic block
1272   // If not, assume the condition flags might get modified somewhere.
1273   if (To->getParent() != From->getParent())
1274     return true;
1275 
1276   // From must be above To.
1277   assert(std::find_if(++To.getReverse(), To->getParent()->rend(),
1278                       [From](MachineInstr &MI) {
1279                         return MI.getIterator() == From;
1280                       }) != To->getParent()->rend());
1281 
1282   // We iterate backward starting at \p To until we hit \p From.
1283   for (const MachineInstr &Instr :
1284        instructionsWithoutDebug(++To.getReverse(), From.getReverse())) {
1285     if (((AccessToCheck & AK_Write) &&
1286          Instr.modifiesRegister(AArch64::NZCV, TRI)) ||
1287         ((AccessToCheck & AK_Read) && Instr.readsRegister(AArch64::NZCV, TRI)))
1288       return true;
1289   }
1290   return false;
1291 }
1292 
1293 /// Try to optimize a compare instruction. A compare instruction is an
1294 /// instruction which produces AArch64::NZCV. It can be truly compare
1295 /// instruction
1296 /// when there are no uses of its destination register.
1297 ///
1298 /// The following steps are tried in order:
1299 /// 1. Convert CmpInstr into an unconditional version.
1300 /// 2. Remove CmpInstr if above there is an instruction producing a needed
1301 ///    condition code or an instruction which can be converted into such an
1302 ///    instruction.
1303 ///    Only comparison with zero is supported.
1304 bool AArch64InstrInfo::optimizeCompareInstr(
1305     MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int CmpMask,
1306     int CmpValue, const MachineRegisterInfo *MRI) const {
1307   assert(CmpInstr.getParent());
1308   assert(MRI);
1309 
1310   // Replace SUBSWrr with SUBWrr if NZCV is not used.
1311   int DeadNZCVIdx = CmpInstr.findRegisterDefOperandIdx(AArch64::NZCV, true);
1312   if (DeadNZCVIdx != -1) {
1313     if (CmpInstr.definesRegister(AArch64::WZR) ||
1314         CmpInstr.definesRegister(AArch64::XZR)) {
1315       CmpInstr.eraseFromParent();
1316       return true;
1317     }
1318     unsigned Opc = CmpInstr.getOpcode();
1319     unsigned NewOpc = convertToNonFlagSettingOpc(CmpInstr);
1320     if (NewOpc == Opc)
1321       return false;
1322     const MCInstrDesc &MCID = get(NewOpc);
1323     CmpInstr.setDesc(MCID);
1324     CmpInstr.RemoveOperand(DeadNZCVIdx);
1325     bool succeeded = UpdateOperandRegClass(CmpInstr);
1326     (void)succeeded;
1327     assert(succeeded && "Some operands reg class are incompatible!");
1328     return true;
1329   }
1330 
1331   // Continue only if we have a "ri" where immediate is zero.
1332   // FIXME:CmpValue has already been converted to 0 or 1 in analyzeCompare
1333   // function.
1334   assert((CmpValue == 0 || CmpValue == 1) && "CmpValue must be 0 or 1!");
1335   if (CmpValue != 0 || SrcReg2 != 0)
1336     return false;
1337 
1338   // CmpInstr is a Compare instruction if destination register is not used.
1339   if (!MRI->use_nodbg_empty(CmpInstr.getOperand(0).getReg()))
1340     return false;
1341 
1342   return substituteCmpToZero(CmpInstr, SrcReg, MRI);
1343 }
1344 
1345 /// Get opcode of S version of Instr.
1346 /// If Instr is S version its opcode is returned.
1347 /// AArch64::INSTRUCTION_LIST_END is returned if Instr does not have S version
1348 /// or we are not interested in it.
1349 static unsigned sForm(MachineInstr &Instr) {
1350   switch (Instr.getOpcode()) {
1351   default:
1352     return AArch64::INSTRUCTION_LIST_END;
1353 
1354   case AArch64::ADDSWrr:
1355   case AArch64::ADDSWri:
1356   case AArch64::ADDSXrr:
1357   case AArch64::ADDSXri:
1358   case AArch64::SUBSWrr:
1359   case AArch64::SUBSWri:
1360   case AArch64::SUBSXrr:
1361   case AArch64::SUBSXri:
1362     return Instr.getOpcode();
1363 
1364   case AArch64::ADDWrr:
1365     return AArch64::ADDSWrr;
1366   case AArch64::ADDWri:
1367     return AArch64::ADDSWri;
1368   case AArch64::ADDXrr:
1369     return AArch64::ADDSXrr;
1370   case AArch64::ADDXri:
1371     return AArch64::ADDSXri;
1372   case AArch64::ADCWr:
1373     return AArch64::ADCSWr;
1374   case AArch64::ADCXr:
1375     return AArch64::ADCSXr;
1376   case AArch64::SUBWrr:
1377     return AArch64::SUBSWrr;
1378   case AArch64::SUBWri:
1379     return AArch64::SUBSWri;
1380   case AArch64::SUBXrr:
1381     return AArch64::SUBSXrr;
1382   case AArch64::SUBXri:
1383     return AArch64::SUBSXri;
1384   case AArch64::SBCWr:
1385     return AArch64::SBCSWr;
1386   case AArch64::SBCXr:
1387     return AArch64::SBCSXr;
1388   case AArch64::ANDWri:
1389     return AArch64::ANDSWri;
1390   case AArch64::ANDXri:
1391     return AArch64::ANDSXri;
1392   }
1393 }
1394 
1395 /// Check if AArch64::NZCV should be alive in successors of MBB.
1396 static bool areCFlagsAliveInSuccessors(MachineBasicBlock *MBB) {
1397   for (auto *BB : MBB->successors())
1398     if (BB->isLiveIn(AArch64::NZCV))
1399       return true;
1400   return false;
1401 }
1402 
1403 namespace {
1404 
1405 struct UsedNZCV {
1406   bool N = false;
1407   bool Z = false;
1408   bool C = false;
1409   bool V = false;
1410 
1411   UsedNZCV() = default;
1412 
1413   UsedNZCV &operator|=(const UsedNZCV &UsedFlags) {
1414     this->N |= UsedFlags.N;
1415     this->Z |= UsedFlags.Z;
1416     this->C |= UsedFlags.C;
1417     this->V |= UsedFlags.V;
1418     return *this;
1419   }
1420 };
1421 
1422 } // end anonymous namespace
1423 
1424 /// Find a condition code used by the instruction.
1425 /// Returns AArch64CC::Invalid if either the instruction does not use condition
1426 /// codes or we don't optimize CmpInstr in the presence of such instructions.
1427 static AArch64CC::CondCode findCondCodeUsedByInstr(const MachineInstr &Instr) {
1428   switch (Instr.getOpcode()) {
1429   default:
1430     return AArch64CC::Invalid;
1431 
1432   case AArch64::Bcc: {
1433     int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV);
1434     assert(Idx >= 2);
1435     return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 2).getImm());
1436   }
1437 
1438   case AArch64::CSINVWr:
1439   case AArch64::CSINVXr:
1440   case AArch64::CSINCWr:
1441   case AArch64::CSINCXr:
1442   case AArch64::CSELWr:
1443   case AArch64::CSELXr:
1444   case AArch64::CSNEGWr:
1445   case AArch64::CSNEGXr:
1446   case AArch64::FCSELSrrr:
1447   case AArch64::FCSELDrrr: {
1448     int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV);
1449     assert(Idx >= 1);
1450     return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 1).getImm());
1451   }
1452   }
1453 }
1454 
1455 static UsedNZCV getUsedNZCV(AArch64CC::CondCode CC) {
1456   assert(CC != AArch64CC::Invalid);
1457   UsedNZCV UsedFlags;
1458   switch (CC) {
1459   default:
1460     break;
1461 
1462   case AArch64CC::EQ: // Z set
1463   case AArch64CC::NE: // Z clear
1464     UsedFlags.Z = true;
1465     break;
1466 
1467   case AArch64CC::HI: // Z clear and C set
1468   case AArch64CC::LS: // Z set   or  C clear
1469     UsedFlags.Z = true;
1470     LLVM_FALLTHROUGH;
1471   case AArch64CC::HS: // C set
1472   case AArch64CC::LO: // C clear
1473     UsedFlags.C = true;
1474     break;
1475 
1476   case AArch64CC::MI: // N set
1477   case AArch64CC::PL: // N clear
1478     UsedFlags.N = true;
1479     break;
1480 
1481   case AArch64CC::VS: // V set
1482   case AArch64CC::VC: // V clear
1483     UsedFlags.V = true;
1484     break;
1485 
1486   case AArch64CC::GT: // Z clear, N and V the same
1487   case AArch64CC::LE: // Z set,   N and V differ
1488     UsedFlags.Z = true;
1489     LLVM_FALLTHROUGH;
1490   case AArch64CC::GE: // N and V the same
1491   case AArch64CC::LT: // N and V differ
1492     UsedFlags.N = true;
1493     UsedFlags.V = true;
1494     break;
1495   }
1496   return UsedFlags;
1497 }
1498 
1499 static bool isADDSRegImm(unsigned Opcode) {
1500   return Opcode == AArch64::ADDSWri || Opcode == AArch64::ADDSXri;
1501 }
1502 
1503 static bool isSUBSRegImm(unsigned Opcode) {
1504   return Opcode == AArch64::SUBSWri || Opcode == AArch64::SUBSXri;
1505 }
1506 
1507 /// Check if CmpInstr can be substituted by MI.
1508 ///
1509 /// CmpInstr can be substituted:
1510 /// - CmpInstr is either 'ADDS %vreg, 0' or 'SUBS %vreg, 0'
1511 /// - and, MI and CmpInstr are from the same MachineBB
1512 /// - and, condition flags are not alive in successors of the CmpInstr parent
1513 /// - and, if MI opcode is the S form there must be no defs of flags between
1514 ///        MI and CmpInstr
1515 ///        or if MI opcode is not the S form there must be neither defs of flags
1516 ///        nor uses of flags between MI and CmpInstr.
1517 /// - and  C/V flags are not used after CmpInstr
1518 static bool canInstrSubstituteCmpInstr(MachineInstr *MI, MachineInstr *CmpInstr,
1519                                        const TargetRegisterInfo *TRI) {
1520   assert(MI);
1521   assert(sForm(*MI) != AArch64::INSTRUCTION_LIST_END);
1522   assert(CmpInstr);
1523 
1524   const unsigned CmpOpcode = CmpInstr->getOpcode();
1525   if (!isADDSRegImm(CmpOpcode) && !isSUBSRegImm(CmpOpcode))
1526     return false;
1527 
1528   if (MI->getParent() != CmpInstr->getParent())
1529     return false;
1530 
1531   if (areCFlagsAliveInSuccessors(CmpInstr->getParent()))
1532     return false;
1533 
1534   AccessKind AccessToCheck = AK_Write;
1535   if (sForm(*MI) != MI->getOpcode())
1536     AccessToCheck = AK_All;
1537   if (areCFlagsAccessedBetweenInstrs(MI, CmpInstr, TRI, AccessToCheck))
1538     return false;
1539 
1540   UsedNZCV NZCVUsedAfterCmp;
1541   for (const MachineInstr &Instr :
1542        instructionsWithoutDebug(std::next(CmpInstr->getIterator()),
1543                                 CmpInstr->getParent()->instr_end())) {
1544     if (Instr.readsRegister(AArch64::NZCV, TRI)) {
1545       AArch64CC::CondCode CC = findCondCodeUsedByInstr(Instr);
1546       if (CC == AArch64CC::Invalid) // Unsupported conditional instruction
1547         return false;
1548       NZCVUsedAfterCmp |= getUsedNZCV(CC);
1549     }
1550 
1551     if (Instr.modifiesRegister(AArch64::NZCV, TRI))
1552       break;
1553   }
1554 
1555   return !NZCVUsedAfterCmp.C && !NZCVUsedAfterCmp.V;
1556 }
1557 
1558 /// Substitute an instruction comparing to zero with another instruction
1559 /// which produces needed condition flags.
1560 ///
1561 /// Return true on success.
1562 bool AArch64InstrInfo::substituteCmpToZero(
1563     MachineInstr &CmpInstr, unsigned SrcReg,
1564     const MachineRegisterInfo *MRI) const {
1565   assert(MRI);
1566   // Get the unique definition of SrcReg.
1567   MachineInstr *MI = MRI->getUniqueVRegDef(SrcReg);
1568   if (!MI)
1569     return false;
1570 
1571   const TargetRegisterInfo *TRI = &getRegisterInfo();
1572 
1573   unsigned NewOpc = sForm(*MI);
1574   if (NewOpc == AArch64::INSTRUCTION_LIST_END)
1575     return false;
1576 
1577   if (!canInstrSubstituteCmpInstr(MI, &CmpInstr, TRI))
1578     return false;
1579 
1580   // Update the instruction to set NZCV.
1581   MI->setDesc(get(NewOpc));
1582   CmpInstr.eraseFromParent();
1583   bool succeeded = UpdateOperandRegClass(*MI);
1584   (void)succeeded;
1585   assert(succeeded && "Some operands reg class are incompatible!");
1586   MI->addRegisterDefined(AArch64::NZCV, TRI);
1587   return true;
1588 }
1589 
1590 bool AArch64InstrInfo::expandPostRAPseudo(MachineInstr &MI) const {
1591   if (MI.getOpcode() != TargetOpcode::LOAD_STACK_GUARD &&
1592       MI.getOpcode() != AArch64::CATCHRET)
1593     return false;
1594 
1595   MachineBasicBlock &MBB = *MI.getParent();
1596   auto &Subtarget = MBB.getParent()->getSubtarget<AArch64Subtarget>();
1597   auto TRI = Subtarget.getRegisterInfo();
1598   DebugLoc DL = MI.getDebugLoc();
1599 
1600   if (MI.getOpcode() == AArch64::CATCHRET) {
1601     // Skip to the first instruction before the epilog.
1602     const TargetInstrInfo *TII =
1603       MBB.getParent()->getSubtarget().getInstrInfo();
1604     MachineBasicBlock *TargetMBB = MI.getOperand(0).getMBB();
1605     auto MBBI = MachineBasicBlock::iterator(MI);
1606     MachineBasicBlock::iterator FirstEpilogSEH = std::prev(MBBI);
1607     while (FirstEpilogSEH->getFlag(MachineInstr::FrameDestroy) &&
1608            FirstEpilogSEH != MBB.begin())
1609       FirstEpilogSEH = std::prev(FirstEpilogSEH);
1610     if (FirstEpilogSEH != MBB.begin())
1611       FirstEpilogSEH = std::next(FirstEpilogSEH);
1612     BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADRP))
1613         .addReg(AArch64::X0, RegState::Define)
1614         .addMBB(TargetMBB);
1615     BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADDXri))
1616         .addReg(AArch64::X0, RegState::Define)
1617         .addReg(AArch64::X0)
1618         .addMBB(TargetMBB)
1619         .addImm(0);
1620     return true;
1621   }
1622 
1623   Register Reg = MI.getOperand(0).getReg();
1624   const GlobalValue *GV =
1625       cast<GlobalValue>((*MI.memoperands_begin())->getValue());
1626   const TargetMachine &TM = MBB.getParent()->getTarget();
1627   unsigned OpFlags = Subtarget.ClassifyGlobalReference(GV, TM);
1628   const unsigned char MO_NC = AArch64II::MO_NC;
1629 
1630   if ((OpFlags & AArch64II::MO_GOT) != 0) {
1631     BuildMI(MBB, MI, DL, get(AArch64::LOADgot), Reg)
1632         .addGlobalAddress(GV, 0, OpFlags);
1633     if (Subtarget.isTargetILP32()) {
1634       unsigned Reg32 = TRI->getSubReg(Reg, AArch64::sub_32);
1635       BuildMI(MBB, MI, DL, get(AArch64::LDRWui))
1636           .addDef(Reg32, RegState::Dead)
1637           .addUse(Reg, RegState::Kill)
1638           .addImm(0)
1639           .addMemOperand(*MI.memoperands_begin())
1640           .addDef(Reg, RegState::Implicit);
1641     } else {
1642       BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
1643           .addReg(Reg, RegState::Kill)
1644           .addImm(0)
1645           .addMemOperand(*MI.memoperands_begin());
1646     }
1647   } else if (TM.getCodeModel() == CodeModel::Large) {
1648     assert(!Subtarget.isTargetILP32() && "how can large exist in ILP32?");
1649     BuildMI(MBB, MI, DL, get(AArch64::MOVZXi), Reg)
1650         .addGlobalAddress(GV, 0, AArch64II::MO_G0 | MO_NC)
1651         .addImm(0);
1652     BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
1653         .addReg(Reg, RegState::Kill)
1654         .addGlobalAddress(GV, 0, AArch64II::MO_G1 | MO_NC)
1655         .addImm(16);
1656     BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
1657         .addReg(Reg, RegState::Kill)
1658         .addGlobalAddress(GV, 0, AArch64II::MO_G2 | MO_NC)
1659         .addImm(32);
1660     BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
1661         .addReg(Reg, RegState::Kill)
1662         .addGlobalAddress(GV, 0, AArch64II::MO_G3)
1663         .addImm(48);
1664     BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
1665         .addReg(Reg, RegState::Kill)
1666         .addImm(0)
1667         .addMemOperand(*MI.memoperands_begin());
1668   } else if (TM.getCodeModel() == CodeModel::Tiny) {
1669     BuildMI(MBB, MI, DL, get(AArch64::ADR), Reg)
1670         .addGlobalAddress(GV, 0, OpFlags);
1671   } else {
1672     BuildMI(MBB, MI, DL, get(AArch64::ADRP), Reg)
1673         .addGlobalAddress(GV, 0, OpFlags | AArch64II::MO_PAGE);
1674     unsigned char LoFlags = OpFlags | AArch64II::MO_PAGEOFF | MO_NC;
1675     if (Subtarget.isTargetILP32()) {
1676       unsigned Reg32 = TRI->getSubReg(Reg, AArch64::sub_32);
1677       BuildMI(MBB, MI, DL, get(AArch64::LDRWui))
1678           .addDef(Reg32, RegState::Dead)
1679           .addUse(Reg, RegState::Kill)
1680           .addGlobalAddress(GV, 0, LoFlags)
1681           .addMemOperand(*MI.memoperands_begin())
1682           .addDef(Reg, RegState::Implicit);
1683     } else {
1684       BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
1685           .addReg(Reg, RegState::Kill)
1686           .addGlobalAddress(GV, 0, LoFlags)
1687           .addMemOperand(*MI.memoperands_begin());
1688     }
1689   }
1690 
1691   MBB.erase(MI);
1692 
1693   return true;
1694 }
1695 
1696 // Return true if this instruction simply sets its single destination register
1697 // to zero. This is equivalent to a register rename of the zero-register.
1698 bool AArch64InstrInfo::isGPRZero(const MachineInstr &MI) {
1699   switch (MI.getOpcode()) {
1700   default:
1701     break;
1702   case AArch64::MOVZWi:
1703   case AArch64::MOVZXi: // movz Rd, #0 (LSL #0)
1704     if (MI.getOperand(1).isImm() && MI.getOperand(1).getImm() == 0) {
1705       assert(MI.getDesc().getNumOperands() == 3 &&
1706              MI.getOperand(2).getImm() == 0 && "invalid MOVZi operands");
1707       return true;
1708     }
1709     break;
1710   case AArch64::ANDWri: // and Rd, Rzr, #imm
1711     return MI.getOperand(1).getReg() == AArch64::WZR;
1712   case AArch64::ANDXri:
1713     return MI.getOperand(1).getReg() == AArch64::XZR;
1714   case TargetOpcode::COPY:
1715     return MI.getOperand(1).getReg() == AArch64::WZR;
1716   }
1717   return false;
1718 }
1719 
1720 // Return true if this instruction simply renames a general register without
1721 // modifying bits.
1722 bool AArch64InstrInfo::isGPRCopy(const MachineInstr &MI) {
1723   switch (MI.getOpcode()) {
1724   default:
1725     break;
1726   case TargetOpcode::COPY: {
1727     // GPR32 copies will by lowered to ORRXrs
1728     Register DstReg = MI.getOperand(0).getReg();
1729     return (AArch64::GPR32RegClass.contains(DstReg) ||
1730             AArch64::GPR64RegClass.contains(DstReg));
1731   }
1732   case AArch64::ORRXrs: // orr Xd, Xzr, Xm (LSL #0)
1733     if (MI.getOperand(1).getReg() == AArch64::XZR) {
1734       assert(MI.getDesc().getNumOperands() == 4 &&
1735              MI.getOperand(3).getImm() == 0 && "invalid ORRrs operands");
1736       return true;
1737     }
1738     break;
1739   case AArch64::ADDXri: // add Xd, Xn, #0 (LSL #0)
1740     if (MI.getOperand(2).getImm() == 0) {
1741       assert(MI.getDesc().getNumOperands() == 4 &&
1742              MI.getOperand(3).getImm() == 0 && "invalid ADDXri operands");
1743       return true;
1744     }
1745     break;
1746   }
1747   return false;
1748 }
1749 
1750 // Return true if this instruction simply renames a general register without
1751 // modifying bits.
1752 bool AArch64InstrInfo::isFPRCopy(const MachineInstr &MI) {
1753   switch (MI.getOpcode()) {
1754   default:
1755     break;
1756   case TargetOpcode::COPY: {
1757     // FPR64 copies will by lowered to ORR.16b
1758     Register DstReg = MI.getOperand(0).getReg();
1759     return (AArch64::FPR64RegClass.contains(DstReg) ||
1760             AArch64::FPR128RegClass.contains(DstReg));
1761   }
1762   case AArch64::ORRv16i8:
1763     if (MI.getOperand(1).getReg() == MI.getOperand(2).getReg()) {
1764       assert(MI.getDesc().getNumOperands() == 3 && MI.getOperand(0).isReg() &&
1765              "invalid ORRv16i8 operands");
1766       return true;
1767     }
1768     break;
1769   }
1770   return false;
1771 }
1772 
1773 unsigned AArch64InstrInfo::isLoadFromStackSlot(const MachineInstr &MI,
1774                                                int &FrameIndex) const {
1775   switch (MI.getOpcode()) {
1776   default:
1777     break;
1778   case AArch64::LDRWui:
1779   case AArch64::LDRXui:
1780   case AArch64::LDRBui:
1781   case AArch64::LDRHui:
1782   case AArch64::LDRSui:
1783   case AArch64::LDRDui:
1784   case AArch64::LDRQui:
1785     if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() &&
1786         MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) {
1787       FrameIndex = MI.getOperand(1).getIndex();
1788       return MI.getOperand(0).getReg();
1789     }
1790     break;
1791   }
1792 
1793   return 0;
1794 }
1795 
1796 unsigned AArch64InstrInfo::isStoreToStackSlot(const MachineInstr &MI,
1797                                               int &FrameIndex) const {
1798   switch (MI.getOpcode()) {
1799   default:
1800     break;
1801   case AArch64::STRWui:
1802   case AArch64::STRXui:
1803   case AArch64::STRBui:
1804   case AArch64::STRHui:
1805   case AArch64::STRSui:
1806   case AArch64::STRDui:
1807   case AArch64::STRQui:
1808   case AArch64::LDR_PXI:
1809   case AArch64::STR_PXI:
1810     if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() &&
1811         MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) {
1812       FrameIndex = MI.getOperand(1).getIndex();
1813       return MI.getOperand(0).getReg();
1814     }
1815     break;
1816   }
1817   return 0;
1818 }
1819 
1820 /// Check all MachineMemOperands for a hint to suppress pairing.
1821 bool AArch64InstrInfo::isLdStPairSuppressed(const MachineInstr &MI) {
1822   return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) {
1823     return MMO->getFlags() & MOSuppressPair;
1824   });
1825 }
1826 
1827 /// Set a flag on the first MachineMemOperand to suppress pairing.
1828 void AArch64InstrInfo::suppressLdStPair(MachineInstr &MI) {
1829   if (MI.memoperands_empty())
1830     return;
1831   (*MI.memoperands_begin())->setFlags(MOSuppressPair);
1832 }
1833 
1834 /// Check all MachineMemOperands for a hint that the load/store is strided.
1835 bool AArch64InstrInfo::isStridedAccess(const MachineInstr &MI) {
1836   return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) {
1837     return MMO->getFlags() & MOStridedAccess;
1838   });
1839 }
1840 
1841 bool AArch64InstrInfo::isUnscaledLdSt(unsigned Opc) {
1842   switch (Opc) {
1843   default:
1844     return false;
1845   case AArch64::STURSi:
1846   case AArch64::STURDi:
1847   case AArch64::STURQi:
1848   case AArch64::STURBBi:
1849   case AArch64::STURHHi:
1850   case AArch64::STURWi:
1851   case AArch64::STURXi:
1852   case AArch64::LDURSi:
1853   case AArch64::LDURDi:
1854   case AArch64::LDURQi:
1855   case AArch64::LDURWi:
1856   case AArch64::LDURXi:
1857   case AArch64::LDURSWi:
1858   case AArch64::LDURHHi:
1859   case AArch64::LDURBBi:
1860   case AArch64::LDURSBWi:
1861   case AArch64::LDURSHWi:
1862     return true;
1863   }
1864 }
1865 
1866 Optional<unsigned> AArch64InstrInfo::getUnscaledLdSt(unsigned Opc) {
1867   switch (Opc) {
1868   default: return {};
1869   case AArch64::PRFMui: return AArch64::PRFUMi;
1870   case AArch64::LDRXui: return AArch64::LDURXi;
1871   case AArch64::LDRWui: return AArch64::LDURWi;
1872   case AArch64::LDRBui: return AArch64::LDURBi;
1873   case AArch64::LDRHui: return AArch64::LDURHi;
1874   case AArch64::LDRSui: return AArch64::LDURSi;
1875   case AArch64::LDRDui: return AArch64::LDURDi;
1876   case AArch64::LDRQui: return AArch64::LDURQi;
1877   case AArch64::LDRBBui: return AArch64::LDURBBi;
1878   case AArch64::LDRHHui: return AArch64::LDURHHi;
1879   case AArch64::LDRSBXui: return AArch64::LDURSBXi;
1880   case AArch64::LDRSBWui: return AArch64::LDURSBWi;
1881   case AArch64::LDRSHXui: return AArch64::LDURSHXi;
1882   case AArch64::LDRSHWui: return AArch64::LDURSHWi;
1883   case AArch64::LDRSWui: return AArch64::LDURSWi;
1884   case AArch64::STRXui: return AArch64::STURXi;
1885   case AArch64::STRWui: return AArch64::STURWi;
1886   case AArch64::STRBui: return AArch64::STURBi;
1887   case AArch64::STRHui: return AArch64::STURHi;
1888   case AArch64::STRSui: return AArch64::STURSi;
1889   case AArch64::STRDui: return AArch64::STURDi;
1890   case AArch64::STRQui: return AArch64::STURQi;
1891   case AArch64::STRBBui: return AArch64::STURBBi;
1892   case AArch64::STRHHui: return AArch64::STURHHi;
1893   }
1894 }
1895 
1896 unsigned AArch64InstrInfo::getLoadStoreImmIdx(unsigned Opc) {
1897   switch (Opc) {
1898   default:
1899     return 2;
1900   case AArch64::LDPXi:
1901   case AArch64::LDPDi:
1902   case AArch64::STPXi:
1903   case AArch64::STPDi:
1904   case AArch64::LDNPXi:
1905   case AArch64::LDNPDi:
1906   case AArch64::STNPXi:
1907   case AArch64::STNPDi:
1908   case AArch64::LDPQi:
1909   case AArch64::STPQi:
1910   case AArch64::LDNPQi:
1911   case AArch64::STNPQi:
1912   case AArch64::LDPWi:
1913   case AArch64::LDPSi:
1914   case AArch64::STPWi:
1915   case AArch64::STPSi:
1916   case AArch64::LDNPWi:
1917   case AArch64::LDNPSi:
1918   case AArch64::STNPWi:
1919   case AArch64::STNPSi:
1920   case AArch64::LDG:
1921   case AArch64::STGPi:
1922   case AArch64::LD1B_IMM:
1923   case AArch64::LD1H_IMM:
1924   case AArch64::LD1W_IMM:
1925   case AArch64::LD1D_IMM:
1926   case AArch64::ST1B_IMM:
1927   case AArch64::ST1H_IMM:
1928   case AArch64::ST1W_IMM:
1929   case AArch64::ST1D_IMM:
1930   case AArch64::LD1B_H_IMM:
1931   case AArch64::LD1SB_H_IMM:
1932   case AArch64::LD1H_S_IMM:
1933   case AArch64::LD1SH_S_IMM:
1934   case AArch64::LD1W_D_IMM:
1935   case AArch64::LD1SW_D_IMM:
1936   case AArch64::ST1B_H_IMM:
1937   case AArch64::ST1H_S_IMM:
1938   case AArch64::ST1W_D_IMM:
1939   case AArch64::LD1B_S_IMM:
1940   case AArch64::LD1SB_S_IMM:
1941   case AArch64::LD1H_D_IMM:
1942   case AArch64::LD1SH_D_IMM:
1943   case AArch64::ST1B_S_IMM:
1944   case AArch64::ST1H_D_IMM:
1945   case AArch64::LD1B_D_IMM:
1946   case AArch64::LD1SB_D_IMM:
1947   case AArch64::ST1B_D_IMM:
1948     return 3;
1949   case AArch64::ADDG:
1950   case AArch64::STGOffset:
1951   case AArch64::LDR_PXI:
1952   case AArch64::STR_PXI:
1953     return 2;
1954   }
1955 }
1956 
1957 bool AArch64InstrInfo::isPairableLdStInst(const MachineInstr &MI) {
1958   switch (MI.getOpcode()) {
1959   default:
1960     return false;
1961   // Scaled instructions.
1962   case AArch64::STRSui:
1963   case AArch64::STRDui:
1964   case AArch64::STRQui:
1965   case AArch64::STRXui:
1966   case AArch64::STRWui:
1967   case AArch64::LDRSui:
1968   case AArch64::LDRDui:
1969   case AArch64::LDRQui:
1970   case AArch64::LDRXui:
1971   case AArch64::LDRWui:
1972   case AArch64::LDRSWui:
1973   // Unscaled instructions.
1974   case AArch64::STURSi:
1975   case AArch64::STURDi:
1976   case AArch64::STURQi:
1977   case AArch64::STURWi:
1978   case AArch64::STURXi:
1979   case AArch64::LDURSi:
1980   case AArch64::LDURDi:
1981   case AArch64::LDURQi:
1982   case AArch64::LDURWi:
1983   case AArch64::LDURXi:
1984   case AArch64::LDURSWi:
1985     return true;
1986   }
1987 }
1988 
1989 unsigned AArch64InstrInfo::convertToFlagSettingOpc(unsigned Opc,
1990                                                    bool &Is64Bit) {
1991   switch (Opc) {
1992   default:
1993     llvm_unreachable("Opcode has no flag setting equivalent!");
1994   // 32-bit cases:
1995   case AArch64::ADDWri:
1996     Is64Bit = false;
1997     return AArch64::ADDSWri;
1998   case AArch64::ADDWrr:
1999     Is64Bit = false;
2000     return AArch64::ADDSWrr;
2001   case AArch64::ADDWrs:
2002     Is64Bit = false;
2003     return AArch64::ADDSWrs;
2004   case AArch64::ADDWrx:
2005     Is64Bit = false;
2006     return AArch64::ADDSWrx;
2007   case AArch64::ANDWri:
2008     Is64Bit = false;
2009     return AArch64::ANDSWri;
2010   case AArch64::ANDWrr:
2011     Is64Bit = false;
2012     return AArch64::ANDSWrr;
2013   case AArch64::ANDWrs:
2014     Is64Bit = false;
2015     return AArch64::ANDSWrs;
2016   case AArch64::BICWrr:
2017     Is64Bit = false;
2018     return AArch64::BICSWrr;
2019   case AArch64::BICWrs:
2020     Is64Bit = false;
2021     return AArch64::BICSWrs;
2022   case AArch64::SUBWri:
2023     Is64Bit = false;
2024     return AArch64::SUBSWri;
2025   case AArch64::SUBWrr:
2026     Is64Bit = false;
2027     return AArch64::SUBSWrr;
2028   case AArch64::SUBWrs:
2029     Is64Bit = false;
2030     return AArch64::SUBSWrs;
2031   case AArch64::SUBWrx:
2032     Is64Bit = false;
2033     return AArch64::SUBSWrx;
2034   // 64-bit cases:
2035   case AArch64::ADDXri:
2036     Is64Bit = true;
2037     return AArch64::ADDSXri;
2038   case AArch64::ADDXrr:
2039     Is64Bit = true;
2040     return AArch64::ADDSXrr;
2041   case AArch64::ADDXrs:
2042     Is64Bit = true;
2043     return AArch64::ADDSXrs;
2044   case AArch64::ADDXrx:
2045     Is64Bit = true;
2046     return AArch64::ADDSXrx;
2047   case AArch64::ANDXri:
2048     Is64Bit = true;
2049     return AArch64::ANDSXri;
2050   case AArch64::ANDXrr:
2051     Is64Bit = true;
2052     return AArch64::ANDSXrr;
2053   case AArch64::ANDXrs:
2054     Is64Bit = true;
2055     return AArch64::ANDSXrs;
2056   case AArch64::BICXrr:
2057     Is64Bit = true;
2058     return AArch64::BICSXrr;
2059   case AArch64::BICXrs:
2060     Is64Bit = true;
2061     return AArch64::BICSXrs;
2062   case AArch64::SUBXri:
2063     Is64Bit = true;
2064     return AArch64::SUBSXri;
2065   case AArch64::SUBXrr:
2066     Is64Bit = true;
2067     return AArch64::SUBSXrr;
2068   case AArch64::SUBXrs:
2069     Is64Bit = true;
2070     return AArch64::SUBSXrs;
2071   case AArch64::SUBXrx:
2072     Is64Bit = true;
2073     return AArch64::SUBSXrx;
2074   }
2075 }
2076 
2077 // Is this a candidate for ld/st merging or pairing?  For example, we don't
2078 // touch volatiles or load/stores that have a hint to avoid pair formation.
2079 bool AArch64InstrInfo::isCandidateToMergeOrPair(const MachineInstr &MI) const {
2080   // If this is a volatile load/store, don't mess with it.
2081   if (MI.hasOrderedMemoryRef())
2082     return false;
2083 
2084   // Make sure this is a reg/fi+imm (as opposed to an address reloc).
2085   assert((MI.getOperand(1).isReg() || MI.getOperand(1).isFI()) &&
2086          "Expected a reg or frame index operand.");
2087   if (!MI.getOperand(2).isImm())
2088     return false;
2089 
2090   // Can't merge/pair if the instruction modifies the base register.
2091   // e.g., ldr x0, [x0]
2092   // This case will never occur with an FI base.
2093   if (MI.getOperand(1).isReg()) {
2094     Register BaseReg = MI.getOperand(1).getReg();
2095     const TargetRegisterInfo *TRI = &getRegisterInfo();
2096     if (MI.modifiesRegister(BaseReg, TRI))
2097       return false;
2098   }
2099 
2100   // Check if this load/store has a hint to avoid pair formation.
2101   // MachineMemOperands hints are set by the AArch64StorePairSuppress pass.
2102   if (isLdStPairSuppressed(MI))
2103     return false;
2104 
2105   // Do not pair any callee-save store/reload instructions in the
2106   // prologue/epilogue if the CFI information encoded the operations as separate
2107   // instructions, as that will cause the size of the actual prologue to mismatch
2108   // with the prologue size recorded in the Windows CFI.
2109   const MCAsmInfo *MAI = MI.getMF()->getTarget().getMCAsmInfo();
2110   bool NeedsWinCFI = MAI->usesWindowsCFI() &&
2111                      MI.getMF()->getFunction().needsUnwindTableEntry();
2112   if (NeedsWinCFI && (MI.getFlag(MachineInstr::FrameSetup) ||
2113                       MI.getFlag(MachineInstr::FrameDestroy)))
2114     return false;
2115 
2116   // On some CPUs quad load/store pairs are slower than two single load/stores.
2117   if (Subtarget.isPaired128Slow()) {
2118     switch (MI.getOpcode()) {
2119     default:
2120       break;
2121     case AArch64::LDURQi:
2122     case AArch64::STURQi:
2123     case AArch64::LDRQui:
2124     case AArch64::STRQui:
2125       return false;
2126     }
2127   }
2128 
2129   return true;
2130 }
2131 
2132 bool AArch64InstrInfo::getMemOperandsWithOffsetWidth(
2133     const MachineInstr &LdSt, SmallVectorImpl<const MachineOperand *> &BaseOps,
2134     int64_t &Offset, bool &OffsetIsScalable, unsigned &Width,
2135     const TargetRegisterInfo *TRI) const {
2136   if (!LdSt.mayLoadOrStore())
2137     return false;
2138 
2139   const MachineOperand *BaseOp;
2140   if (!getMemOperandWithOffsetWidth(LdSt, BaseOp, Offset, OffsetIsScalable,
2141                                     Width, TRI))
2142     return false;
2143   BaseOps.push_back(BaseOp);
2144   return true;
2145 }
2146 
2147 bool AArch64InstrInfo::getMemOperandWithOffsetWidth(
2148     const MachineInstr &LdSt, const MachineOperand *&BaseOp, int64_t &Offset,
2149     bool &OffsetIsScalable, unsigned &Width,
2150     const TargetRegisterInfo *TRI) const {
2151   assert(LdSt.mayLoadOrStore() && "Expected a memory operation.");
2152   // Handle only loads/stores with base register followed by immediate offset.
2153   if (LdSt.getNumExplicitOperands() == 3) {
2154     // Non-paired instruction (e.g., ldr x1, [x0, #8]).
2155     if ((!LdSt.getOperand(1).isReg() && !LdSt.getOperand(1).isFI()) ||
2156         !LdSt.getOperand(2).isImm())
2157       return false;
2158   } else if (LdSt.getNumExplicitOperands() == 4) {
2159     // Paired instruction (e.g., ldp x1, x2, [x0, #8]).
2160     if (!LdSt.getOperand(1).isReg() ||
2161         (!LdSt.getOperand(2).isReg() && !LdSt.getOperand(2).isFI()) ||
2162         !LdSt.getOperand(3).isImm())
2163       return false;
2164   } else
2165     return false;
2166 
2167   // Get the scaling factor for the instruction and set the width for the
2168   // instruction.
2169   TypeSize Scale(0U, false);
2170   int64_t Dummy1, Dummy2;
2171 
2172   // If this returns false, then it's an instruction we don't want to handle.
2173   if (!getMemOpInfo(LdSt.getOpcode(), Scale, Width, Dummy1, Dummy2))
2174     return false;
2175 
2176   // Compute the offset. Offset is calculated as the immediate operand
2177   // multiplied by the scaling factor. Unscaled instructions have scaling factor
2178   // set to 1.
2179   if (LdSt.getNumExplicitOperands() == 3) {
2180     BaseOp = &LdSt.getOperand(1);
2181     Offset = LdSt.getOperand(2).getImm() * Scale.getKnownMinSize();
2182   } else {
2183     assert(LdSt.getNumExplicitOperands() == 4 && "invalid number of operands");
2184     BaseOp = &LdSt.getOperand(2);
2185     Offset = LdSt.getOperand(3).getImm() * Scale.getKnownMinSize();
2186   }
2187   OffsetIsScalable = Scale.isScalable();
2188 
2189   if (!BaseOp->isReg() && !BaseOp->isFI())
2190     return false;
2191 
2192   return true;
2193 }
2194 
2195 MachineOperand &
2196 AArch64InstrInfo::getMemOpBaseRegImmOfsOffsetOperand(MachineInstr &LdSt) const {
2197   assert(LdSt.mayLoadOrStore() && "Expected a memory operation.");
2198   MachineOperand &OfsOp = LdSt.getOperand(LdSt.getNumExplicitOperands() - 1);
2199   assert(OfsOp.isImm() && "Offset operand wasn't immediate.");
2200   return OfsOp;
2201 }
2202 
2203 bool AArch64InstrInfo::getMemOpInfo(unsigned Opcode, TypeSize &Scale,
2204                                     unsigned &Width, int64_t &MinOffset,
2205                                     int64_t &MaxOffset) {
2206   const unsigned SVEMaxBytesPerVector = AArch64::SVEMaxBitsPerVector / 8;
2207   switch (Opcode) {
2208   // Not a memory operation or something we want to handle.
2209   default:
2210     Scale = TypeSize::Fixed(0);
2211     Width = 0;
2212     MinOffset = MaxOffset = 0;
2213     return false;
2214   case AArch64::STRWpost:
2215   case AArch64::LDRWpost:
2216     Width = 32;
2217     Scale = TypeSize::Fixed(4);
2218     MinOffset = -256;
2219     MaxOffset = 255;
2220     break;
2221   case AArch64::LDURQi:
2222   case AArch64::STURQi:
2223     Width = 16;
2224     Scale = TypeSize::Fixed(1);
2225     MinOffset = -256;
2226     MaxOffset = 255;
2227     break;
2228   case AArch64::PRFUMi:
2229   case AArch64::LDURXi:
2230   case AArch64::LDURDi:
2231   case AArch64::STURXi:
2232   case AArch64::STURDi:
2233     Width = 8;
2234     Scale = TypeSize::Fixed(1);
2235     MinOffset = -256;
2236     MaxOffset = 255;
2237     break;
2238   case AArch64::LDURWi:
2239   case AArch64::LDURSi:
2240   case AArch64::LDURSWi:
2241   case AArch64::STURWi:
2242   case AArch64::STURSi:
2243     Width = 4;
2244     Scale = TypeSize::Fixed(1);
2245     MinOffset = -256;
2246     MaxOffset = 255;
2247     break;
2248   case AArch64::LDURHi:
2249   case AArch64::LDURHHi:
2250   case AArch64::LDURSHXi:
2251   case AArch64::LDURSHWi:
2252   case AArch64::STURHi:
2253   case AArch64::STURHHi:
2254     Width = 2;
2255     Scale = TypeSize::Fixed(1);
2256     MinOffset = -256;
2257     MaxOffset = 255;
2258     break;
2259   case AArch64::LDURBi:
2260   case AArch64::LDURBBi:
2261   case AArch64::LDURSBXi:
2262   case AArch64::LDURSBWi:
2263   case AArch64::STURBi:
2264   case AArch64::STURBBi:
2265     Width = 1;
2266     Scale = TypeSize::Fixed(1);
2267     MinOffset = -256;
2268     MaxOffset = 255;
2269     break;
2270   case AArch64::LDPQi:
2271   case AArch64::LDNPQi:
2272   case AArch64::STPQi:
2273   case AArch64::STNPQi:
2274     Scale = TypeSize::Fixed(16);
2275     Width = 32;
2276     MinOffset = -64;
2277     MaxOffset = 63;
2278     break;
2279   case AArch64::LDRQui:
2280   case AArch64::STRQui:
2281     Scale = TypeSize::Fixed(16);
2282     Width = 16;
2283     MinOffset = 0;
2284     MaxOffset = 4095;
2285     break;
2286   case AArch64::LDPXi:
2287   case AArch64::LDPDi:
2288   case AArch64::LDNPXi:
2289   case AArch64::LDNPDi:
2290   case AArch64::STPXi:
2291   case AArch64::STPDi:
2292   case AArch64::STNPXi:
2293   case AArch64::STNPDi:
2294     Scale = TypeSize::Fixed(8);
2295     Width = 16;
2296     MinOffset = -64;
2297     MaxOffset = 63;
2298     break;
2299   case AArch64::PRFMui:
2300   case AArch64::LDRXui:
2301   case AArch64::LDRDui:
2302   case AArch64::STRXui:
2303   case AArch64::STRDui:
2304     Scale = TypeSize::Fixed(8);
2305     Width = 8;
2306     MinOffset = 0;
2307     MaxOffset = 4095;
2308     break;
2309   case AArch64::LDPWi:
2310   case AArch64::LDPSi:
2311   case AArch64::LDNPWi:
2312   case AArch64::LDNPSi:
2313   case AArch64::STPWi:
2314   case AArch64::STPSi:
2315   case AArch64::STNPWi:
2316   case AArch64::STNPSi:
2317     Scale = TypeSize::Fixed(4);
2318     Width = 8;
2319     MinOffset = -64;
2320     MaxOffset = 63;
2321     break;
2322   case AArch64::LDRWui:
2323   case AArch64::LDRSui:
2324   case AArch64::LDRSWui:
2325   case AArch64::STRWui:
2326   case AArch64::STRSui:
2327     Scale = TypeSize::Fixed(4);
2328     Width = 4;
2329     MinOffset = 0;
2330     MaxOffset = 4095;
2331     break;
2332   case AArch64::LDRHui:
2333   case AArch64::LDRHHui:
2334   case AArch64::LDRSHWui:
2335   case AArch64::LDRSHXui:
2336   case AArch64::STRHui:
2337   case AArch64::STRHHui:
2338     Scale = TypeSize::Fixed(2);
2339     Width = 2;
2340     MinOffset = 0;
2341     MaxOffset = 4095;
2342     break;
2343   case AArch64::LDRBui:
2344   case AArch64::LDRBBui:
2345   case AArch64::LDRSBWui:
2346   case AArch64::LDRSBXui:
2347   case AArch64::STRBui:
2348   case AArch64::STRBBui:
2349     Scale = TypeSize::Fixed(1);
2350     Width = 1;
2351     MinOffset = 0;
2352     MaxOffset = 4095;
2353     break;
2354   case AArch64::ADDG:
2355     Scale = TypeSize::Fixed(16);
2356     Width = 0;
2357     MinOffset = 0;
2358     MaxOffset = 63;
2359     break;
2360   case AArch64::TAGPstack:
2361     Scale = TypeSize::Fixed(16);
2362     Width = 0;
2363     // TAGP with a negative offset turns into SUBP, which has a maximum offset
2364     // of 63 (not 64!).
2365     MinOffset = -63;
2366     MaxOffset = 63;
2367     break;
2368   case AArch64::LDG:
2369   case AArch64::STGOffset:
2370   case AArch64::STZGOffset:
2371     Scale = TypeSize::Fixed(16);
2372     Width = 16;
2373     MinOffset = -256;
2374     MaxOffset = 255;
2375     break;
2376   case AArch64::STR_ZZZZXI:
2377   case AArch64::LDR_ZZZZXI:
2378     Scale = TypeSize::Scalable(16);
2379     Width = SVEMaxBytesPerVector * 4;
2380     MinOffset = -256;
2381     MaxOffset = 252;
2382     break;
2383   case AArch64::STR_ZZZXI:
2384   case AArch64::LDR_ZZZXI:
2385     Scale = TypeSize::Scalable(16);
2386     Width = SVEMaxBytesPerVector * 3;
2387     MinOffset = -256;
2388     MaxOffset = 253;
2389     break;
2390   case AArch64::STR_ZZXI:
2391   case AArch64::LDR_ZZXI:
2392     Scale = TypeSize::Scalable(16);
2393     Width = SVEMaxBytesPerVector * 2;
2394     MinOffset = -256;
2395     MaxOffset = 254;
2396     break;
2397   case AArch64::LDR_PXI:
2398   case AArch64::STR_PXI:
2399     Scale = TypeSize::Scalable(2);
2400     Width = SVEMaxBytesPerVector / 8;
2401     MinOffset = -256;
2402     MaxOffset = 255;
2403     break;
2404   case AArch64::LDR_ZXI:
2405   case AArch64::STR_ZXI:
2406     Scale = TypeSize::Scalable(16);
2407     Width = SVEMaxBytesPerVector;
2408     MinOffset = -256;
2409     MaxOffset = 255;
2410     break;
2411   case AArch64::LD1B_IMM:
2412   case AArch64::LD1H_IMM:
2413   case AArch64::LD1W_IMM:
2414   case AArch64::LD1D_IMM:
2415   case AArch64::ST1B_IMM:
2416   case AArch64::ST1H_IMM:
2417   case AArch64::ST1W_IMM:
2418   case AArch64::ST1D_IMM:
2419     // A full vectors worth of data
2420     // Width = mbytes * elements
2421     Scale = TypeSize::Scalable(16);
2422     Width = SVEMaxBytesPerVector;
2423     MinOffset = -8;
2424     MaxOffset = 7;
2425     break;
2426   case AArch64::LD1B_H_IMM:
2427   case AArch64::LD1SB_H_IMM:
2428   case AArch64::LD1H_S_IMM:
2429   case AArch64::LD1SH_S_IMM:
2430   case AArch64::LD1W_D_IMM:
2431   case AArch64::LD1SW_D_IMM:
2432   case AArch64::ST1B_H_IMM:
2433   case AArch64::ST1H_S_IMM:
2434   case AArch64::ST1W_D_IMM:
2435     // A half vector worth of data
2436     // Width = mbytes * elements
2437     Scale = TypeSize::Scalable(8);
2438     Width = SVEMaxBytesPerVector / 2;
2439     MinOffset = -8;
2440     MaxOffset = 7;
2441     break;
2442   case AArch64::LD1B_S_IMM:
2443   case AArch64::LD1SB_S_IMM:
2444   case AArch64::LD1H_D_IMM:
2445   case AArch64::LD1SH_D_IMM:
2446   case AArch64::ST1B_S_IMM:
2447   case AArch64::ST1H_D_IMM:
2448     // A quarter vector worth of data
2449     // Width = mbytes * elements
2450     Scale = TypeSize::Scalable(4);
2451     Width = SVEMaxBytesPerVector / 4;
2452     MinOffset = -8;
2453     MaxOffset = 7;
2454     break;
2455   case AArch64::LD1B_D_IMM:
2456   case AArch64::LD1SB_D_IMM:
2457   case AArch64::ST1B_D_IMM:
2458     // A eighth vector worth of data
2459     // Width = mbytes * elements
2460     Scale = TypeSize::Scalable(2);
2461     Width = SVEMaxBytesPerVector / 8;
2462     MinOffset = -8;
2463     MaxOffset = 7;
2464     break;
2465   case AArch64::ST2GOffset:
2466   case AArch64::STZ2GOffset:
2467     Scale = TypeSize::Fixed(16);
2468     Width = 32;
2469     MinOffset = -256;
2470     MaxOffset = 255;
2471     break;
2472   case AArch64::STGPi:
2473     Scale = TypeSize::Fixed(16);
2474     Width = 16;
2475     MinOffset = -64;
2476     MaxOffset = 63;
2477     break;
2478   }
2479 
2480   return true;
2481 }
2482 
2483 // Scaling factor for unscaled load or store.
2484 int AArch64InstrInfo::getMemScale(unsigned Opc) {
2485   switch (Opc) {
2486   default:
2487     llvm_unreachable("Opcode has unknown scale!");
2488   case AArch64::LDRBBui:
2489   case AArch64::LDURBBi:
2490   case AArch64::LDRSBWui:
2491   case AArch64::LDURSBWi:
2492   case AArch64::STRBBui:
2493   case AArch64::STURBBi:
2494     return 1;
2495   case AArch64::LDRHHui:
2496   case AArch64::LDURHHi:
2497   case AArch64::LDRSHWui:
2498   case AArch64::LDURSHWi:
2499   case AArch64::STRHHui:
2500   case AArch64::STURHHi:
2501     return 2;
2502   case AArch64::LDRSui:
2503   case AArch64::LDURSi:
2504   case AArch64::LDRSWui:
2505   case AArch64::LDURSWi:
2506   case AArch64::LDRWui:
2507   case AArch64::LDURWi:
2508   case AArch64::STRSui:
2509   case AArch64::STURSi:
2510   case AArch64::STRWui:
2511   case AArch64::STURWi:
2512   case AArch64::LDPSi:
2513   case AArch64::LDPSWi:
2514   case AArch64::LDPWi:
2515   case AArch64::STPSi:
2516   case AArch64::STPWi:
2517     return 4;
2518   case AArch64::LDRDui:
2519   case AArch64::LDURDi:
2520   case AArch64::LDRXui:
2521   case AArch64::LDURXi:
2522   case AArch64::STRDui:
2523   case AArch64::STURDi:
2524   case AArch64::STRXui:
2525   case AArch64::STURXi:
2526   case AArch64::LDPDi:
2527   case AArch64::LDPXi:
2528   case AArch64::STPDi:
2529   case AArch64::STPXi:
2530     return 8;
2531   case AArch64::LDRQui:
2532   case AArch64::LDURQi:
2533   case AArch64::STRQui:
2534   case AArch64::STURQi:
2535   case AArch64::LDPQi:
2536   case AArch64::STPQi:
2537   case AArch64::STGOffset:
2538   case AArch64::STZGOffset:
2539   case AArch64::ST2GOffset:
2540   case AArch64::STZ2GOffset:
2541   case AArch64::STGPi:
2542     return 16;
2543   }
2544 }
2545 
2546 // Scale the unscaled offsets.  Returns false if the unscaled offset can't be
2547 // scaled.
2548 static bool scaleOffset(unsigned Opc, int64_t &Offset) {
2549   int Scale = AArch64InstrInfo::getMemScale(Opc);
2550 
2551   // If the byte-offset isn't a multiple of the stride, we can't scale this
2552   // offset.
2553   if (Offset % Scale != 0)
2554     return false;
2555 
2556   // Convert the byte-offset used by unscaled into an "element" offset used
2557   // by the scaled pair load/store instructions.
2558   Offset /= Scale;
2559   return true;
2560 }
2561 
2562 static bool canPairLdStOpc(unsigned FirstOpc, unsigned SecondOpc) {
2563   if (FirstOpc == SecondOpc)
2564     return true;
2565   // We can also pair sign-ext and zero-ext instructions.
2566   switch (FirstOpc) {
2567   default:
2568     return false;
2569   case AArch64::LDRWui:
2570   case AArch64::LDURWi:
2571     return SecondOpc == AArch64::LDRSWui || SecondOpc == AArch64::LDURSWi;
2572   case AArch64::LDRSWui:
2573   case AArch64::LDURSWi:
2574     return SecondOpc == AArch64::LDRWui || SecondOpc == AArch64::LDURWi;
2575   }
2576   // These instructions can't be paired based on their opcodes.
2577   return false;
2578 }
2579 
2580 static bool shouldClusterFI(const MachineFrameInfo &MFI, int FI1,
2581                             int64_t Offset1, unsigned Opcode1, int FI2,
2582                             int64_t Offset2, unsigned Opcode2) {
2583   // Accesses through fixed stack object frame indices may access a different
2584   // fixed stack slot. Check that the object offsets + offsets match.
2585   if (MFI.isFixedObjectIndex(FI1) && MFI.isFixedObjectIndex(FI2)) {
2586     int64_t ObjectOffset1 = MFI.getObjectOffset(FI1);
2587     int64_t ObjectOffset2 = MFI.getObjectOffset(FI2);
2588     assert(ObjectOffset1 <= ObjectOffset2 && "Object offsets are not ordered.");
2589     // Convert to scaled object offsets.
2590     int Scale1 = AArch64InstrInfo::getMemScale(Opcode1);
2591     if (ObjectOffset1 % Scale1 != 0)
2592       return false;
2593     ObjectOffset1 /= Scale1;
2594     int Scale2 = AArch64InstrInfo::getMemScale(Opcode2);
2595     if (ObjectOffset2 % Scale2 != 0)
2596       return false;
2597     ObjectOffset2 /= Scale2;
2598     ObjectOffset1 += Offset1;
2599     ObjectOffset2 += Offset2;
2600     return ObjectOffset1 + 1 == ObjectOffset2;
2601   }
2602 
2603   return FI1 == FI2;
2604 }
2605 
2606 /// Detect opportunities for ldp/stp formation.
2607 ///
2608 /// Only called for LdSt for which getMemOperandWithOffset returns true.
2609 bool AArch64InstrInfo::shouldClusterMemOps(
2610     ArrayRef<const MachineOperand *> BaseOps1,
2611     ArrayRef<const MachineOperand *> BaseOps2, unsigned NumLoads,
2612     unsigned NumBytes) const {
2613   assert(BaseOps1.size() == 1 && BaseOps2.size() == 1);
2614   const MachineOperand &BaseOp1 = *BaseOps1.front();
2615   const MachineOperand &BaseOp2 = *BaseOps2.front();
2616   const MachineInstr &FirstLdSt = *BaseOp1.getParent();
2617   const MachineInstr &SecondLdSt = *BaseOp2.getParent();
2618   if (BaseOp1.getType() != BaseOp2.getType())
2619     return false;
2620 
2621   assert((BaseOp1.isReg() || BaseOp1.isFI()) &&
2622          "Only base registers and frame indices are supported.");
2623 
2624   // Check for both base regs and base FI.
2625   if (BaseOp1.isReg() && BaseOp1.getReg() != BaseOp2.getReg())
2626     return false;
2627 
2628   // Only cluster up to a single pair.
2629   if (NumLoads > 2)
2630     return false;
2631 
2632   if (!isPairableLdStInst(FirstLdSt) || !isPairableLdStInst(SecondLdSt))
2633     return false;
2634 
2635   // Can we pair these instructions based on their opcodes?
2636   unsigned FirstOpc = FirstLdSt.getOpcode();
2637   unsigned SecondOpc = SecondLdSt.getOpcode();
2638   if (!canPairLdStOpc(FirstOpc, SecondOpc))
2639     return false;
2640 
2641   // Can't merge volatiles or load/stores that have a hint to avoid pair
2642   // formation, for example.
2643   if (!isCandidateToMergeOrPair(FirstLdSt) ||
2644       !isCandidateToMergeOrPair(SecondLdSt))
2645     return false;
2646 
2647   // isCandidateToMergeOrPair guarantees that operand 2 is an immediate.
2648   int64_t Offset1 = FirstLdSt.getOperand(2).getImm();
2649   if (isUnscaledLdSt(FirstOpc) && !scaleOffset(FirstOpc, Offset1))
2650     return false;
2651 
2652   int64_t Offset2 = SecondLdSt.getOperand(2).getImm();
2653   if (isUnscaledLdSt(SecondOpc) && !scaleOffset(SecondOpc, Offset2))
2654     return false;
2655 
2656   // Pairwise instructions have a 7-bit signed offset field.
2657   if (Offset1 > 63 || Offset1 < -64)
2658     return false;
2659 
2660   // The caller should already have ordered First/SecondLdSt by offset.
2661   // Note: except for non-equal frame index bases
2662   if (BaseOp1.isFI()) {
2663     assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 <= Offset2) &&
2664            "Caller should have ordered offsets.");
2665 
2666     const MachineFrameInfo &MFI =
2667         FirstLdSt.getParent()->getParent()->getFrameInfo();
2668     return shouldClusterFI(MFI, BaseOp1.getIndex(), Offset1, FirstOpc,
2669                            BaseOp2.getIndex(), Offset2, SecondOpc);
2670   }
2671 
2672   assert(Offset1 <= Offset2 && "Caller should have ordered offsets.");
2673 
2674   return Offset1 + 1 == Offset2;
2675 }
2676 
2677 static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB,
2678                                             unsigned Reg, unsigned SubIdx,
2679                                             unsigned State,
2680                                             const TargetRegisterInfo *TRI) {
2681   if (!SubIdx)
2682     return MIB.addReg(Reg, State);
2683 
2684   if (Register::isPhysicalRegister(Reg))
2685     return MIB.addReg(TRI->getSubReg(Reg, SubIdx), State);
2686   return MIB.addReg(Reg, State, SubIdx);
2687 }
2688 
2689 static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg,
2690                                         unsigned NumRegs) {
2691   // We really want the positive remainder mod 32 here, that happens to be
2692   // easily obtainable with a mask.
2693   return ((DestReg - SrcReg) & 0x1f) < NumRegs;
2694 }
2695 
2696 void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
2697                                         MachineBasicBlock::iterator I,
2698                                         const DebugLoc &DL, MCRegister DestReg,
2699                                         MCRegister SrcReg, bool KillSrc,
2700                                         unsigned Opcode,
2701                                         ArrayRef<unsigned> Indices) const {
2702   assert(Subtarget.hasNEON() && "Unexpected register copy without NEON");
2703   const TargetRegisterInfo *TRI = &getRegisterInfo();
2704   uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
2705   uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
2706   unsigned NumRegs = Indices.size();
2707 
2708   int SubReg = 0, End = NumRegs, Incr = 1;
2709   if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs)) {
2710     SubReg = NumRegs - 1;
2711     End = -1;
2712     Incr = -1;
2713   }
2714 
2715   for (; SubReg != End; SubReg += Incr) {
2716     const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
2717     AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
2718     AddSubReg(MIB, SrcReg, Indices[SubReg], 0, TRI);
2719     AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
2720   }
2721 }
2722 
2723 void AArch64InstrInfo::copyGPRRegTuple(MachineBasicBlock &MBB,
2724                                        MachineBasicBlock::iterator I,
2725                                        DebugLoc DL, unsigned DestReg,
2726                                        unsigned SrcReg, bool KillSrc,
2727                                        unsigned Opcode, unsigned ZeroReg,
2728                                        llvm::ArrayRef<unsigned> Indices) const {
2729   const TargetRegisterInfo *TRI = &getRegisterInfo();
2730   unsigned NumRegs = Indices.size();
2731 
2732 #ifndef NDEBUG
2733   uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
2734   uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
2735   assert(DestEncoding % NumRegs == 0 && SrcEncoding % NumRegs == 0 &&
2736          "GPR reg sequences should not be able to overlap");
2737 #endif
2738 
2739   for (unsigned SubReg = 0; SubReg != NumRegs; ++SubReg) {
2740     const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
2741     AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
2742     MIB.addReg(ZeroReg);
2743     AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
2744     MIB.addImm(0);
2745   }
2746 }
2747 
2748 void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
2749                                    MachineBasicBlock::iterator I,
2750                                    const DebugLoc &DL, MCRegister DestReg,
2751                                    MCRegister SrcReg, bool KillSrc) const {
2752   if (AArch64::GPR32spRegClass.contains(DestReg) &&
2753       (AArch64::GPR32spRegClass.contains(SrcReg) || SrcReg == AArch64::WZR)) {
2754     const TargetRegisterInfo *TRI = &getRegisterInfo();
2755 
2756     if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) {
2757       // If either operand is WSP, expand to ADD #0.
2758       if (Subtarget.hasZeroCycleRegMove()) {
2759         // Cyclone recognizes "ADD Xd, Xn, #0" as a zero-cycle register move.
2760         MCRegister DestRegX = TRI->getMatchingSuperReg(
2761             DestReg, AArch64::sub_32, &AArch64::GPR64spRegClass);
2762         MCRegister SrcRegX = TRI->getMatchingSuperReg(
2763             SrcReg, AArch64::sub_32, &AArch64::GPR64spRegClass);
2764         // This instruction is reading and writing X registers.  This may upset
2765         // the register scavenger and machine verifier, so we need to indicate
2766         // that we are reading an undefined value from SrcRegX, but a proper
2767         // value from SrcReg.
2768         BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestRegX)
2769             .addReg(SrcRegX, RegState::Undef)
2770             .addImm(0)
2771             .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0))
2772             .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
2773       } else {
2774         BuildMI(MBB, I, DL, get(AArch64::ADDWri), DestReg)
2775             .addReg(SrcReg, getKillRegState(KillSrc))
2776             .addImm(0)
2777             .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2778       }
2779     } else if (SrcReg == AArch64::WZR && Subtarget.hasZeroCycleZeroingGP()) {
2780       BuildMI(MBB, I, DL, get(AArch64::MOVZWi), DestReg)
2781           .addImm(0)
2782           .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2783     } else {
2784       if (Subtarget.hasZeroCycleRegMove()) {
2785         // Cyclone recognizes "ORR Xd, XZR, Xm" as a zero-cycle register move.
2786         MCRegister DestRegX = TRI->getMatchingSuperReg(
2787             DestReg, AArch64::sub_32, &AArch64::GPR64spRegClass);
2788         MCRegister SrcRegX = TRI->getMatchingSuperReg(
2789             SrcReg, AArch64::sub_32, &AArch64::GPR64spRegClass);
2790         // This instruction is reading and writing X registers.  This may upset
2791         // the register scavenger and machine verifier, so we need to indicate
2792         // that we are reading an undefined value from SrcRegX, but a proper
2793         // value from SrcReg.
2794         BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestRegX)
2795             .addReg(AArch64::XZR)
2796             .addReg(SrcRegX, RegState::Undef)
2797             .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
2798       } else {
2799         // Otherwise, expand to ORR WZR.
2800         BuildMI(MBB, I, DL, get(AArch64::ORRWrr), DestReg)
2801             .addReg(AArch64::WZR)
2802             .addReg(SrcReg, getKillRegState(KillSrc));
2803       }
2804     }
2805     return;
2806   }
2807 
2808   // Copy a Predicate register by ORRing with itself.
2809   if (AArch64::PPRRegClass.contains(DestReg) &&
2810       AArch64::PPRRegClass.contains(SrcReg)) {
2811     assert(Subtarget.hasSVE() && "Unexpected SVE register.");
2812     BuildMI(MBB, I, DL, get(AArch64::ORR_PPzPP), DestReg)
2813       .addReg(SrcReg) // Pg
2814       .addReg(SrcReg)
2815       .addReg(SrcReg, getKillRegState(KillSrc));
2816     return;
2817   }
2818 
2819   // Copy a Z register by ORRing with itself.
2820   if (AArch64::ZPRRegClass.contains(DestReg) &&
2821       AArch64::ZPRRegClass.contains(SrcReg)) {
2822     assert(Subtarget.hasSVE() && "Unexpected SVE register.");
2823     BuildMI(MBB, I, DL, get(AArch64::ORR_ZZZ), DestReg)
2824       .addReg(SrcReg)
2825       .addReg(SrcReg, getKillRegState(KillSrc));
2826     return;
2827   }
2828 
2829   // Copy a Z register pair by copying the individual sub-registers.
2830   if (AArch64::ZPR2RegClass.contains(DestReg) &&
2831       AArch64::ZPR2RegClass.contains(SrcReg)) {
2832     static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1};
2833     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
2834                      Indices);
2835     return;
2836   }
2837 
2838   // Copy a Z register triple by copying the individual sub-registers.
2839   if (AArch64::ZPR3RegClass.contains(DestReg) &&
2840       AArch64::ZPR3RegClass.contains(SrcReg)) {
2841     static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
2842                                        AArch64::zsub2};
2843     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
2844                      Indices);
2845     return;
2846   }
2847 
2848   // Copy a Z register quad by copying the individual sub-registers.
2849   if (AArch64::ZPR4RegClass.contains(DestReg) &&
2850       AArch64::ZPR4RegClass.contains(SrcReg)) {
2851     static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
2852                                        AArch64::zsub2, AArch64::zsub3};
2853     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
2854                      Indices);
2855     return;
2856   }
2857 
2858   if (AArch64::GPR64spRegClass.contains(DestReg) &&
2859       (AArch64::GPR64spRegClass.contains(SrcReg) || SrcReg == AArch64::XZR)) {
2860     if (DestReg == AArch64::SP || SrcReg == AArch64::SP) {
2861       // If either operand is SP, expand to ADD #0.
2862       BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestReg)
2863           .addReg(SrcReg, getKillRegState(KillSrc))
2864           .addImm(0)
2865           .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2866     } else if (SrcReg == AArch64::XZR && Subtarget.hasZeroCycleZeroingGP()) {
2867       BuildMI(MBB, I, DL, get(AArch64::MOVZXi), DestReg)
2868           .addImm(0)
2869           .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0));
2870     } else {
2871       // Otherwise, expand to ORR XZR.
2872       BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestReg)
2873           .addReg(AArch64::XZR)
2874           .addReg(SrcReg, getKillRegState(KillSrc));
2875     }
2876     return;
2877   }
2878 
2879   // Copy a DDDD register quad by copying the individual sub-registers.
2880   if (AArch64::DDDDRegClass.contains(DestReg) &&
2881       AArch64::DDDDRegClass.contains(SrcReg)) {
2882     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
2883                                        AArch64::dsub2, AArch64::dsub3};
2884     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
2885                      Indices);
2886     return;
2887   }
2888 
2889   // Copy a DDD register triple by copying the individual sub-registers.
2890   if (AArch64::DDDRegClass.contains(DestReg) &&
2891       AArch64::DDDRegClass.contains(SrcReg)) {
2892     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
2893                                        AArch64::dsub2};
2894     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
2895                      Indices);
2896     return;
2897   }
2898 
2899   // Copy a DD register pair by copying the individual sub-registers.
2900   if (AArch64::DDRegClass.contains(DestReg) &&
2901       AArch64::DDRegClass.contains(SrcReg)) {
2902     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1};
2903     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
2904                      Indices);
2905     return;
2906   }
2907 
2908   // Copy a QQQQ register quad by copying the individual sub-registers.
2909   if (AArch64::QQQQRegClass.contains(DestReg) &&
2910       AArch64::QQQQRegClass.contains(SrcReg)) {
2911     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
2912                                        AArch64::qsub2, AArch64::qsub3};
2913     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
2914                      Indices);
2915     return;
2916   }
2917 
2918   // Copy a QQQ register triple by copying the individual sub-registers.
2919   if (AArch64::QQQRegClass.contains(DestReg) &&
2920       AArch64::QQQRegClass.contains(SrcReg)) {
2921     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
2922                                        AArch64::qsub2};
2923     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
2924                      Indices);
2925     return;
2926   }
2927 
2928   // Copy a QQ register pair by copying the individual sub-registers.
2929   if (AArch64::QQRegClass.contains(DestReg) &&
2930       AArch64::QQRegClass.contains(SrcReg)) {
2931     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1};
2932     copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
2933                      Indices);
2934     return;
2935   }
2936 
2937   if (AArch64::XSeqPairsClassRegClass.contains(DestReg) &&
2938       AArch64::XSeqPairsClassRegClass.contains(SrcReg)) {
2939     static const unsigned Indices[] = {AArch64::sube64, AArch64::subo64};
2940     copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRXrs,
2941                     AArch64::XZR, Indices);
2942     return;
2943   }
2944 
2945   if (AArch64::WSeqPairsClassRegClass.contains(DestReg) &&
2946       AArch64::WSeqPairsClassRegClass.contains(SrcReg)) {
2947     static const unsigned Indices[] = {AArch64::sube32, AArch64::subo32};
2948     copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRWrs,
2949                     AArch64::WZR, Indices);
2950     return;
2951   }
2952 
2953   if (AArch64::FPR128RegClass.contains(DestReg) &&
2954       AArch64::FPR128RegClass.contains(SrcReg)) {
2955     if (Subtarget.hasNEON()) {
2956       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2957           .addReg(SrcReg)
2958           .addReg(SrcReg, getKillRegState(KillSrc));
2959     } else {
2960       BuildMI(MBB, I, DL, get(AArch64::STRQpre))
2961           .addReg(AArch64::SP, RegState::Define)
2962           .addReg(SrcReg, getKillRegState(KillSrc))
2963           .addReg(AArch64::SP)
2964           .addImm(-16);
2965       BuildMI(MBB, I, DL, get(AArch64::LDRQpre))
2966           .addReg(AArch64::SP, RegState::Define)
2967           .addReg(DestReg, RegState::Define)
2968           .addReg(AArch64::SP)
2969           .addImm(16);
2970     }
2971     return;
2972   }
2973 
2974   if (AArch64::FPR64RegClass.contains(DestReg) &&
2975       AArch64::FPR64RegClass.contains(SrcReg)) {
2976     if (Subtarget.hasNEON()) {
2977       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::dsub,
2978                                        &AArch64::FPR128RegClass);
2979       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::dsub,
2980                                       &AArch64::FPR128RegClass);
2981       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2982           .addReg(SrcReg)
2983           .addReg(SrcReg, getKillRegState(KillSrc));
2984     } else {
2985       BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestReg)
2986           .addReg(SrcReg, getKillRegState(KillSrc));
2987     }
2988     return;
2989   }
2990 
2991   if (AArch64::FPR32RegClass.contains(DestReg) &&
2992       AArch64::FPR32RegClass.contains(SrcReg)) {
2993     if (Subtarget.hasNEON()) {
2994       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::ssub,
2995                                        &AArch64::FPR128RegClass);
2996       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::ssub,
2997                                       &AArch64::FPR128RegClass);
2998       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
2999           .addReg(SrcReg)
3000           .addReg(SrcReg, getKillRegState(KillSrc));
3001     } else {
3002       BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
3003           .addReg(SrcReg, getKillRegState(KillSrc));
3004     }
3005     return;
3006   }
3007 
3008   if (AArch64::FPR16RegClass.contains(DestReg) &&
3009       AArch64::FPR16RegClass.contains(SrcReg)) {
3010     if (Subtarget.hasNEON()) {
3011       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub,
3012                                        &AArch64::FPR128RegClass);
3013       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub,
3014                                       &AArch64::FPR128RegClass);
3015       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
3016           .addReg(SrcReg)
3017           .addReg(SrcReg, getKillRegState(KillSrc));
3018     } else {
3019       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub,
3020                                        &AArch64::FPR32RegClass);
3021       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub,
3022                                       &AArch64::FPR32RegClass);
3023       BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
3024           .addReg(SrcReg, getKillRegState(KillSrc));
3025     }
3026     return;
3027   }
3028 
3029   if (AArch64::FPR8RegClass.contains(DestReg) &&
3030       AArch64::FPR8RegClass.contains(SrcReg)) {
3031     if (Subtarget.hasNEON()) {
3032       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub,
3033                                        &AArch64::FPR128RegClass);
3034       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub,
3035                                       &AArch64::FPR128RegClass);
3036       BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
3037           .addReg(SrcReg)
3038           .addReg(SrcReg, getKillRegState(KillSrc));
3039     } else {
3040       DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub,
3041                                        &AArch64::FPR32RegClass);
3042       SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub,
3043                                       &AArch64::FPR32RegClass);
3044       BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
3045           .addReg(SrcReg, getKillRegState(KillSrc));
3046     }
3047     return;
3048   }
3049 
3050   // Copies between GPR64 and FPR64.
3051   if (AArch64::FPR64RegClass.contains(DestReg) &&
3052       AArch64::GPR64RegClass.contains(SrcReg)) {
3053     BuildMI(MBB, I, DL, get(AArch64::FMOVXDr), DestReg)
3054         .addReg(SrcReg, getKillRegState(KillSrc));
3055     return;
3056   }
3057   if (AArch64::GPR64RegClass.contains(DestReg) &&
3058       AArch64::FPR64RegClass.contains(SrcReg)) {
3059     BuildMI(MBB, I, DL, get(AArch64::FMOVDXr), DestReg)
3060         .addReg(SrcReg, getKillRegState(KillSrc));
3061     return;
3062   }
3063   // Copies between GPR32 and FPR32.
3064   if (AArch64::FPR32RegClass.contains(DestReg) &&
3065       AArch64::GPR32RegClass.contains(SrcReg)) {
3066     BuildMI(MBB, I, DL, get(AArch64::FMOVWSr), DestReg)
3067         .addReg(SrcReg, getKillRegState(KillSrc));
3068     return;
3069   }
3070   if (AArch64::GPR32RegClass.contains(DestReg) &&
3071       AArch64::FPR32RegClass.contains(SrcReg)) {
3072     BuildMI(MBB, I, DL, get(AArch64::FMOVSWr), DestReg)
3073         .addReg(SrcReg, getKillRegState(KillSrc));
3074     return;
3075   }
3076 
3077   if (DestReg == AArch64::NZCV) {
3078     assert(AArch64::GPR64RegClass.contains(SrcReg) && "Invalid NZCV copy");
3079     BuildMI(MBB, I, DL, get(AArch64::MSR))
3080         .addImm(AArch64SysReg::NZCV)
3081         .addReg(SrcReg, getKillRegState(KillSrc))
3082         .addReg(AArch64::NZCV, RegState::Implicit | RegState::Define);
3083     return;
3084   }
3085 
3086   if (SrcReg == AArch64::NZCV) {
3087     assert(AArch64::GPR64RegClass.contains(DestReg) && "Invalid NZCV copy");
3088     BuildMI(MBB, I, DL, get(AArch64::MRS), DestReg)
3089         .addImm(AArch64SysReg::NZCV)
3090         .addReg(AArch64::NZCV, RegState::Implicit | getKillRegState(KillSrc));
3091     return;
3092   }
3093 
3094   llvm_unreachable("unimplemented reg-to-reg copy");
3095 }
3096 
3097 static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI,
3098                                     MachineBasicBlock &MBB,
3099                                     MachineBasicBlock::iterator InsertBefore,
3100                                     const MCInstrDesc &MCID,
3101                                     Register SrcReg, bool IsKill,
3102                                     unsigned SubIdx0, unsigned SubIdx1, int FI,
3103                                     MachineMemOperand *MMO) {
3104   Register SrcReg0 = SrcReg;
3105   Register SrcReg1 = SrcReg;
3106   if (Register::isPhysicalRegister(SrcReg)) {
3107     SrcReg0 = TRI.getSubReg(SrcReg, SubIdx0);
3108     SubIdx0 = 0;
3109     SrcReg1 = TRI.getSubReg(SrcReg, SubIdx1);
3110     SubIdx1 = 0;
3111   }
3112   BuildMI(MBB, InsertBefore, DebugLoc(), MCID)
3113       .addReg(SrcReg0, getKillRegState(IsKill), SubIdx0)
3114       .addReg(SrcReg1, getKillRegState(IsKill), SubIdx1)
3115       .addFrameIndex(FI)
3116       .addImm(0)
3117       .addMemOperand(MMO);
3118 }
3119 
3120 void AArch64InstrInfo::storeRegToStackSlot(
3121     MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register SrcReg,
3122     bool isKill, int FI, const TargetRegisterClass *RC,
3123     const TargetRegisterInfo *TRI) const {
3124   MachineFunction &MF = *MBB.getParent();
3125   MachineFrameInfo &MFI = MF.getFrameInfo();
3126 
3127   MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI);
3128   MachineMemOperand *MMO =
3129       MF.getMachineMemOperand(PtrInfo, MachineMemOperand::MOStore,
3130                               MFI.getObjectSize(FI), MFI.getObjectAlign(FI));
3131   unsigned Opc = 0;
3132   bool Offset = true;
3133   unsigned StackID = TargetStackID::Default;
3134   switch (TRI->getSpillSize(*RC)) {
3135   case 1:
3136     if (AArch64::FPR8RegClass.hasSubClassEq(RC))
3137       Opc = AArch64::STRBui;
3138     break;
3139   case 2:
3140     if (AArch64::FPR16RegClass.hasSubClassEq(RC))
3141       Opc = AArch64::STRHui;
3142     else if (AArch64::PPRRegClass.hasSubClassEq(RC)) {
3143       assert(Subtarget.hasSVE() && "Unexpected register store without SVE");
3144       Opc = AArch64::STR_PXI;
3145       StackID = TargetStackID::SVEVector;
3146     }
3147     break;
3148   case 4:
3149     if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
3150       Opc = AArch64::STRWui;
3151       if (Register::isVirtualRegister(SrcReg))
3152         MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR32RegClass);
3153       else
3154         assert(SrcReg != AArch64::WSP);
3155     } else if (AArch64::FPR32RegClass.hasSubClassEq(RC))
3156       Opc = AArch64::STRSui;
3157     break;
3158   case 8:
3159     if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) {
3160       Opc = AArch64::STRXui;
3161       if (Register::isVirtualRegister(SrcReg))
3162         MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass);
3163       else
3164         assert(SrcReg != AArch64::SP);
3165     } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) {
3166       Opc = AArch64::STRDui;
3167     } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) {
3168       storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI,
3169                               get(AArch64::STPWi), SrcReg, isKill,
3170                               AArch64::sube32, AArch64::subo32, FI, MMO);
3171       return;
3172     }
3173     break;
3174   case 16:
3175     if (AArch64::FPR128RegClass.hasSubClassEq(RC))
3176       Opc = AArch64::STRQui;
3177     else if (AArch64::DDRegClass.hasSubClassEq(RC)) {
3178       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
3179       Opc = AArch64::ST1Twov1d;
3180       Offset = false;
3181     } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) {
3182       storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI,
3183                               get(AArch64::STPXi), SrcReg, isKill,
3184                               AArch64::sube64, AArch64::subo64, FI, MMO);
3185       return;
3186     } else if (AArch64::ZPRRegClass.hasSubClassEq(RC)) {
3187       assert(Subtarget.hasSVE() && "Unexpected register store without SVE");
3188       Opc = AArch64::STR_ZXI;
3189       StackID = TargetStackID::SVEVector;
3190     }
3191     break;
3192   case 24:
3193     if (AArch64::DDDRegClass.hasSubClassEq(RC)) {
3194       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
3195       Opc = AArch64::ST1Threev1d;
3196       Offset = false;
3197     }
3198     break;
3199   case 32:
3200     if (AArch64::DDDDRegClass.hasSubClassEq(RC)) {
3201       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
3202       Opc = AArch64::ST1Fourv1d;
3203       Offset = false;
3204     } else if (AArch64::QQRegClass.hasSubClassEq(RC)) {
3205       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
3206       Opc = AArch64::ST1Twov2d;
3207       Offset = false;
3208     } else if (AArch64::ZPR2RegClass.hasSubClassEq(RC)) {
3209       assert(Subtarget.hasSVE() && "Unexpected register store without SVE");
3210       Opc = AArch64::STR_ZZXI;
3211       StackID = TargetStackID::SVEVector;
3212     }
3213     break;
3214   case 48:
3215     if (AArch64::QQQRegClass.hasSubClassEq(RC)) {
3216       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
3217       Opc = AArch64::ST1Threev2d;
3218       Offset = false;
3219     } else if (AArch64::ZPR3RegClass.hasSubClassEq(RC)) {
3220       assert(Subtarget.hasSVE() && "Unexpected register store without SVE");
3221       Opc = AArch64::STR_ZZZXI;
3222       StackID = TargetStackID::SVEVector;
3223     }
3224     break;
3225   case 64:
3226     if (AArch64::QQQQRegClass.hasSubClassEq(RC)) {
3227       assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
3228       Opc = AArch64::ST1Fourv2d;
3229       Offset = false;
3230     } else if (AArch64::ZPR4RegClass.hasSubClassEq(RC)) {
3231       assert(Subtarget.hasSVE() && "Unexpected register store without SVE");
3232       Opc = AArch64::STR_ZZZZXI;
3233       StackID = TargetStackID::SVEVector;
3234     }
3235     break;
3236   }
3237   assert(Opc && "Unknown register class");
3238   MFI.setStackID(FI, StackID);
3239 
3240   const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc))
3241                                      .addReg(SrcReg, getKillRegState(isKill))
3242                                      .addFrameIndex(FI);
3243 
3244   if (Offset)
3245     MI.addImm(0);
3246   MI.addMemOperand(MMO);
3247 }
3248 
3249 static void loadRegPairFromStackSlot(const TargetRegisterInfo &TRI,
3250                                      MachineBasicBlock &MBB,
3251                                      MachineBasicBlock::iterator InsertBefore,
3252                                      const MCInstrDesc &MCID,
3253                                      Register DestReg, unsigned SubIdx0,
3254                                      unsigned SubIdx1, int FI,
3255                                      MachineMemOperand *MMO) {
3256   Register DestReg0 = DestReg;
3257   Register DestReg1 = DestReg;
3258   bool IsUndef = true;
3259   if (Register::isPhysicalRegister(DestReg)) {
3260     DestReg0 = TRI.getSubReg(DestReg, SubIdx0);
3261     SubIdx0 = 0;
3262     DestReg1 = TRI.getSubReg(DestReg, SubIdx1);
3263     SubIdx1 = 0;
3264     IsUndef = false;
3265   }
3266   BuildMI(MBB, InsertBefore, DebugLoc(), MCID)
3267       .addReg(DestReg0, RegState::Define | getUndefRegState(IsUndef), SubIdx0)
3268       .addReg(DestReg1, RegState::Define | getUndefRegState(IsUndef), SubIdx1)
3269       .addFrameIndex(FI)
3270       .addImm(0)
3271       .addMemOperand(MMO);
3272 }
3273 
3274 void AArch64InstrInfo::loadRegFromStackSlot(
3275     MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register DestReg,
3276     int FI, const TargetRegisterClass *RC,
3277     const TargetRegisterInfo *TRI) const {
3278   MachineFunction &MF = *MBB.getParent();
3279   MachineFrameInfo &MFI = MF.getFrameInfo();
3280   MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI);
3281   MachineMemOperand *MMO =
3282       MF.getMachineMemOperand(PtrInfo, MachineMemOperand::MOLoad,
3283                               MFI.getObjectSize(FI), MFI.getObjectAlign(FI));
3284 
3285   unsigned Opc = 0;
3286   bool Offset = true;
3287   unsigned StackID = TargetStackID::Default;
3288   switch (TRI->getSpillSize(*RC)) {
3289   case 1:
3290     if (AArch64::FPR8RegClass.hasSubClassEq(RC))
3291       Opc = AArch64::LDRBui;
3292     break;
3293   case 2:
3294     if (AArch64::FPR16RegClass.hasSubClassEq(RC))
3295       Opc = AArch64::LDRHui;
3296     else if (AArch64::PPRRegClass.hasSubClassEq(RC)) {
3297       assert(Subtarget.hasSVE() && "Unexpected register load without SVE");
3298       Opc = AArch64::LDR_PXI;
3299       StackID = TargetStackID::SVEVector;
3300     }
3301     break;
3302   case 4:
3303     if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
3304       Opc = AArch64::LDRWui;
3305       if (Register::isVirtualRegister(DestReg))
3306         MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR32RegClass);
3307       else
3308         assert(DestReg != AArch64::WSP);
3309     } else if (AArch64::FPR32RegClass.hasSubClassEq(RC))
3310       Opc = AArch64::LDRSui;
3311     break;
3312   case 8:
3313     if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) {
3314       Opc = AArch64::LDRXui;
3315       if (Register::isVirtualRegister(DestReg))
3316         MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR64RegClass);
3317       else
3318         assert(DestReg != AArch64::SP);
3319     } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) {
3320       Opc = AArch64::LDRDui;
3321     } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) {
3322       loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI,
3323                                get(AArch64::LDPWi), DestReg, AArch64::sube32,
3324                                AArch64::subo32, FI, MMO);
3325       return;
3326     }
3327     break;
3328   case 16:
3329     if (AArch64::FPR128RegClass.hasSubClassEq(RC))
3330       Opc = AArch64::LDRQui;
3331     else if (AArch64::DDRegClass.hasSubClassEq(RC)) {
3332       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
3333       Opc = AArch64::LD1Twov1d;
3334       Offset = false;
3335     } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) {
3336       loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI,
3337                                get(AArch64::LDPXi), DestReg, AArch64::sube64,
3338                                AArch64::subo64, FI, MMO);
3339       return;
3340     } else if (AArch64::ZPRRegClass.hasSubClassEq(RC)) {
3341       assert(Subtarget.hasSVE() && "Unexpected register load without SVE");
3342       Opc = AArch64::LDR_ZXI;
3343       StackID = TargetStackID::SVEVector;
3344     }
3345     break;
3346   case 24:
3347     if (AArch64::DDDRegClass.hasSubClassEq(RC)) {
3348       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
3349       Opc = AArch64::LD1Threev1d;
3350       Offset = false;
3351     }
3352     break;
3353   case 32:
3354     if (AArch64::DDDDRegClass.hasSubClassEq(RC)) {
3355       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
3356       Opc = AArch64::LD1Fourv1d;
3357       Offset = false;
3358     } else if (AArch64::QQRegClass.hasSubClassEq(RC)) {
3359       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
3360       Opc = AArch64::LD1Twov2d;
3361       Offset = false;
3362     } else if (AArch64::ZPR2RegClass.hasSubClassEq(RC)) {
3363       assert(Subtarget.hasSVE() && "Unexpected register load without SVE");
3364       Opc = AArch64::LDR_ZZXI;
3365       StackID = TargetStackID::SVEVector;
3366     }
3367     break;
3368   case 48:
3369     if (AArch64::QQQRegClass.hasSubClassEq(RC)) {
3370       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
3371       Opc = AArch64::LD1Threev2d;
3372       Offset = false;
3373     } else if (AArch64::ZPR3RegClass.hasSubClassEq(RC)) {
3374       assert(Subtarget.hasSVE() && "Unexpected register load without SVE");
3375       Opc = AArch64::LDR_ZZZXI;
3376       StackID = TargetStackID::SVEVector;
3377     }
3378     break;
3379   case 64:
3380     if (AArch64::QQQQRegClass.hasSubClassEq(RC)) {
3381       assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
3382       Opc = AArch64::LD1Fourv2d;
3383       Offset = false;
3384     } else if (AArch64::ZPR4RegClass.hasSubClassEq(RC)) {
3385       assert(Subtarget.hasSVE() && "Unexpected register load without SVE");
3386       Opc = AArch64::LDR_ZZZZXI;
3387       StackID = TargetStackID::SVEVector;
3388     }
3389     break;
3390   }
3391 
3392   assert(Opc && "Unknown register class");
3393   MFI.setStackID(FI, StackID);
3394 
3395   const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc))
3396                                      .addReg(DestReg, getDefRegState(true))
3397                                      .addFrameIndex(FI);
3398   if (Offset)
3399     MI.addImm(0);
3400   MI.addMemOperand(MMO);
3401 }
3402 
3403 bool llvm::isNZCVTouchedInInstructionRange(const MachineInstr &DefMI,
3404                                            const MachineInstr &UseMI,
3405                                            const TargetRegisterInfo *TRI) {
3406   return any_of(instructionsWithoutDebug(std::next(DefMI.getIterator()),
3407                                          UseMI.getIterator()),
3408                 [TRI](const MachineInstr &I) {
3409                   return I.modifiesRegister(AArch64::NZCV, TRI) ||
3410                          I.readsRegister(AArch64::NZCV, TRI);
3411                 });
3412 }
3413 
3414 // Helper function to emit a frame offset adjustment from a given
3415 // pointer (SrcReg), stored into DestReg. This function is explicit
3416 // in that it requires the opcode.
3417 static void emitFrameOffsetAdj(MachineBasicBlock &MBB,
3418                                MachineBasicBlock::iterator MBBI,
3419                                const DebugLoc &DL, unsigned DestReg,
3420                                unsigned SrcReg, int64_t Offset, unsigned Opc,
3421                                const TargetInstrInfo *TII,
3422                                MachineInstr::MIFlag Flag, bool NeedsWinCFI,
3423                                bool *HasWinCFI) {
3424   int Sign = 1;
3425   unsigned MaxEncoding, ShiftSize;
3426   switch (Opc) {
3427   case AArch64::ADDXri:
3428   case AArch64::ADDSXri:
3429   case AArch64::SUBXri:
3430   case AArch64::SUBSXri:
3431     MaxEncoding = 0xfff;
3432     ShiftSize = 12;
3433     break;
3434   case AArch64::ADDVL_XXI:
3435   case AArch64::ADDPL_XXI:
3436     MaxEncoding = 31;
3437     ShiftSize = 0;
3438     if (Offset < 0) {
3439       MaxEncoding = 32;
3440       Sign = -1;
3441       Offset = -Offset;
3442     }
3443     break;
3444   default:
3445     llvm_unreachable("Unsupported opcode");
3446   }
3447 
3448   // FIXME: If the offset won't fit in 24-bits, compute the offset into a
3449   // scratch register.  If DestReg is a virtual register, use it as the
3450   // scratch register; otherwise, create a new virtual register (to be
3451   // replaced by the scavenger at the end of PEI).  That case can be optimized
3452   // slightly if DestReg is SP which is always 16-byte aligned, so the scratch
3453   // register can be loaded with offset%8 and the add/sub can use an extending
3454   // instruction with LSL#3.
3455   // Currently the function handles any offsets but generates a poor sequence
3456   // of code.
3457   //  assert(Offset < (1 << 24) && "unimplemented reg plus immediate");
3458 
3459   const unsigned MaxEncodableValue = MaxEncoding << ShiftSize;
3460   Register TmpReg = DestReg;
3461   if (TmpReg == AArch64::XZR)
3462     TmpReg = MBB.getParent()->getRegInfo().createVirtualRegister(
3463         &AArch64::GPR64RegClass);
3464   do {
3465     uint64_t ThisVal = std::min<uint64_t>(Offset, MaxEncodableValue);
3466     unsigned LocalShiftSize = 0;
3467     if (ThisVal > MaxEncoding) {
3468       ThisVal = ThisVal >> ShiftSize;
3469       LocalShiftSize = ShiftSize;
3470     }
3471     assert((ThisVal >> ShiftSize) <= MaxEncoding &&
3472            "Encoding cannot handle value that big");
3473 
3474     Offset -= ThisVal << LocalShiftSize;
3475     if (Offset == 0)
3476       TmpReg = DestReg;
3477     auto MBI = BuildMI(MBB, MBBI, DL, TII->get(Opc), TmpReg)
3478                    .addReg(SrcReg)
3479                    .addImm(Sign * (int)ThisVal);
3480     if (ShiftSize)
3481       MBI = MBI.addImm(
3482           AArch64_AM::getShifterImm(AArch64_AM::LSL, LocalShiftSize));
3483     MBI = MBI.setMIFlag(Flag);
3484 
3485     if (NeedsWinCFI) {
3486       assert(Sign == 1 && "SEH directives should always have a positive sign");
3487       int Imm = (int)(ThisVal << LocalShiftSize);
3488       if ((DestReg == AArch64::FP && SrcReg == AArch64::SP) ||
3489           (SrcReg == AArch64::FP && DestReg == AArch64::SP)) {
3490         if (HasWinCFI)
3491           *HasWinCFI = true;
3492         if (Imm == 0)
3493           BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_SetFP)).setMIFlag(Flag);
3494         else
3495           BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_AddFP))
3496               .addImm(Imm)
3497               .setMIFlag(Flag);
3498         assert(Offset == 0 && "Expected remaining offset to be zero to "
3499                               "emit a single SEH directive");
3500       } else if (DestReg == AArch64::SP) {
3501         if (HasWinCFI)
3502           *HasWinCFI = true;
3503         assert(SrcReg == AArch64::SP && "Unexpected SrcReg for SEH_StackAlloc");
3504         BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_StackAlloc))
3505             .addImm(Imm)
3506             .setMIFlag(Flag);
3507       }
3508       if (HasWinCFI)
3509         *HasWinCFI = true;
3510     }
3511 
3512     SrcReg = TmpReg;
3513   } while (Offset);
3514 }
3515 
3516 void llvm::emitFrameOffset(MachineBasicBlock &MBB,
3517                            MachineBasicBlock::iterator MBBI, const DebugLoc &DL,
3518                            unsigned DestReg, unsigned SrcReg,
3519                            StackOffset Offset, const TargetInstrInfo *TII,
3520                            MachineInstr::MIFlag Flag, bool SetNZCV,
3521                            bool NeedsWinCFI, bool *HasWinCFI) {
3522   int64_t Bytes, NumPredicateVectors, NumDataVectors;
3523   Offset.getForFrameOffset(Bytes, NumPredicateVectors, NumDataVectors);
3524 
3525   // First emit non-scalable frame offsets, or a simple 'mov'.
3526   if (Bytes || (!Offset && SrcReg != DestReg)) {
3527     assert((DestReg != AArch64::SP || Bytes % 8 == 0) &&
3528            "SP increment/decrement not 8-byte aligned");
3529     unsigned Opc = SetNZCV ? AArch64::ADDSXri : AArch64::ADDXri;
3530     if (Bytes < 0) {
3531       Bytes = -Bytes;
3532       Opc = SetNZCV ? AArch64::SUBSXri : AArch64::SUBXri;
3533     }
3534     emitFrameOffsetAdj(MBB, MBBI, DL, DestReg, SrcReg, Bytes, Opc, TII, Flag,
3535                        NeedsWinCFI, HasWinCFI);
3536     SrcReg = DestReg;
3537   }
3538 
3539   assert(!(SetNZCV && (NumPredicateVectors || NumDataVectors)) &&
3540          "SetNZCV not supported with SVE vectors");
3541   assert(!(NeedsWinCFI && (NumPredicateVectors || NumDataVectors)) &&
3542          "WinCFI not supported with SVE vectors");
3543 
3544   if (NumDataVectors) {
3545     emitFrameOffsetAdj(MBB, MBBI, DL, DestReg, SrcReg, NumDataVectors,
3546                        AArch64::ADDVL_XXI, TII, Flag, NeedsWinCFI, nullptr);
3547     SrcReg = DestReg;
3548   }
3549 
3550   if (NumPredicateVectors) {
3551     assert(DestReg != AArch64::SP && "Unaligned access to SP");
3552     emitFrameOffsetAdj(MBB, MBBI, DL, DestReg, SrcReg, NumPredicateVectors,
3553                        AArch64::ADDPL_XXI, TII, Flag, NeedsWinCFI, nullptr);
3554   }
3555 }
3556 
3557 MachineInstr *AArch64InstrInfo::foldMemoryOperandImpl(
3558     MachineFunction &MF, MachineInstr &MI, ArrayRef<unsigned> Ops,
3559     MachineBasicBlock::iterator InsertPt, int FrameIndex,
3560     LiveIntervals *LIS, VirtRegMap *VRM) const {
3561   // This is a bit of a hack. Consider this instruction:
3562   //
3563   //   %0 = COPY %sp; GPR64all:%0
3564   //
3565   // We explicitly chose GPR64all for the virtual register so such a copy might
3566   // be eliminated by RegisterCoalescer. However, that may not be possible, and
3567   // %0 may even spill. We can't spill %sp, and since it is in the GPR64all
3568   // register class, TargetInstrInfo::foldMemoryOperand() is going to try.
3569   //
3570   // To prevent that, we are going to constrain the %0 register class here.
3571   //
3572   // <rdar://problem/11522048>
3573   //
3574   if (MI.isFullCopy()) {
3575     Register DstReg = MI.getOperand(0).getReg();
3576     Register SrcReg = MI.getOperand(1).getReg();
3577     if (SrcReg == AArch64::SP && Register::isVirtualRegister(DstReg)) {
3578       MF.getRegInfo().constrainRegClass(DstReg, &AArch64::GPR64RegClass);
3579       return nullptr;
3580     }
3581     if (DstReg == AArch64::SP && Register::isVirtualRegister(SrcReg)) {
3582       MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass);
3583       return nullptr;
3584     }
3585   }
3586 
3587   // Handle the case where a copy is being spilled or filled but the source
3588   // and destination register class don't match.  For example:
3589   //
3590   //   %0 = COPY %xzr; GPR64common:%0
3591   //
3592   // In this case we can still safely fold away the COPY and generate the
3593   // following spill code:
3594   //
3595   //   STRXui %xzr, %stack.0
3596   //
3597   // This also eliminates spilled cross register class COPYs (e.g. between x and
3598   // d regs) of the same size.  For example:
3599   //
3600   //   %0 = COPY %1; GPR64:%0, FPR64:%1
3601   //
3602   // will be filled as
3603   //
3604   //   LDRDui %0, fi<#0>
3605   //
3606   // instead of
3607   //
3608   //   LDRXui %Temp, fi<#0>
3609   //   %0 = FMOV %Temp
3610   //
3611   if (MI.isCopy() && Ops.size() == 1 &&
3612       // Make sure we're only folding the explicit COPY defs/uses.
3613       (Ops[0] == 0 || Ops[0] == 1)) {
3614     bool IsSpill = Ops[0] == 0;
3615     bool IsFill = !IsSpill;
3616     const TargetRegisterInfo &TRI = *MF.getSubtarget().getRegisterInfo();
3617     const MachineRegisterInfo &MRI = MF.getRegInfo();
3618     MachineBasicBlock &MBB = *MI.getParent();
3619     const MachineOperand &DstMO = MI.getOperand(0);
3620     const MachineOperand &SrcMO = MI.getOperand(1);
3621     Register DstReg = DstMO.getReg();
3622     Register SrcReg = SrcMO.getReg();
3623     // This is slightly expensive to compute for physical regs since
3624     // getMinimalPhysRegClass is slow.
3625     auto getRegClass = [&](unsigned Reg) {
3626       return Register::isVirtualRegister(Reg) ? MRI.getRegClass(Reg)
3627                                               : TRI.getMinimalPhysRegClass(Reg);
3628     };
3629 
3630     if (DstMO.getSubReg() == 0 && SrcMO.getSubReg() == 0) {
3631       assert(TRI.getRegSizeInBits(*getRegClass(DstReg)) ==
3632                  TRI.getRegSizeInBits(*getRegClass(SrcReg)) &&
3633              "Mismatched register size in non subreg COPY");
3634       if (IsSpill)
3635         storeRegToStackSlot(MBB, InsertPt, SrcReg, SrcMO.isKill(), FrameIndex,
3636                             getRegClass(SrcReg), &TRI);
3637       else
3638         loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex,
3639                              getRegClass(DstReg), &TRI);
3640       return &*--InsertPt;
3641     }
3642 
3643     // Handle cases like spilling def of:
3644     //
3645     //   %0:sub_32<def,read-undef> = COPY %wzr; GPR64common:%0
3646     //
3647     // where the physical register source can be widened and stored to the full
3648     // virtual reg destination stack slot, in this case producing:
3649     //
3650     //   STRXui %xzr, %stack.0
3651     //
3652     if (IsSpill && DstMO.isUndef() && Register::isPhysicalRegister(SrcReg)) {
3653       assert(SrcMO.getSubReg() == 0 &&
3654              "Unexpected subreg on physical register");
3655       const TargetRegisterClass *SpillRC;
3656       unsigned SpillSubreg;
3657       switch (DstMO.getSubReg()) {
3658       default:
3659         SpillRC = nullptr;
3660         break;
3661       case AArch64::sub_32:
3662       case AArch64::ssub:
3663         if (AArch64::GPR32RegClass.contains(SrcReg)) {
3664           SpillRC = &AArch64::GPR64RegClass;
3665           SpillSubreg = AArch64::sub_32;
3666         } else if (AArch64::FPR32RegClass.contains(SrcReg)) {
3667           SpillRC = &AArch64::FPR64RegClass;
3668           SpillSubreg = AArch64::ssub;
3669         } else
3670           SpillRC = nullptr;
3671         break;
3672       case AArch64::dsub:
3673         if (AArch64::FPR64RegClass.contains(SrcReg)) {
3674           SpillRC = &AArch64::FPR128RegClass;
3675           SpillSubreg = AArch64::dsub;
3676         } else
3677           SpillRC = nullptr;
3678         break;
3679       }
3680 
3681       if (SpillRC)
3682         if (unsigned WidenedSrcReg =
3683                 TRI.getMatchingSuperReg(SrcReg, SpillSubreg, SpillRC)) {
3684           storeRegToStackSlot(MBB, InsertPt, WidenedSrcReg, SrcMO.isKill(),
3685                               FrameIndex, SpillRC, &TRI);
3686           return &*--InsertPt;
3687         }
3688     }
3689 
3690     // Handle cases like filling use of:
3691     //
3692     //   %0:sub_32<def,read-undef> = COPY %1; GPR64:%0, GPR32:%1
3693     //
3694     // where we can load the full virtual reg source stack slot, into the subreg
3695     // destination, in this case producing:
3696     //
3697     //   LDRWui %0:sub_32<def,read-undef>, %stack.0
3698     //
3699     if (IsFill && SrcMO.getSubReg() == 0 && DstMO.isUndef()) {
3700       const TargetRegisterClass *FillRC;
3701       switch (DstMO.getSubReg()) {
3702       default:
3703         FillRC = nullptr;
3704         break;
3705       case AArch64::sub_32:
3706         FillRC = &AArch64::GPR32RegClass;
3707         break;
3708       case AArch64::ssub:
3709         FillRC = &AArch64::FPR32RegClass;
3710         break;
3711       case AArch64::dsub:
3712         FillRC = &AArch64::FPR64RegClass;
3713         break;
3714       }
3715 
3716       if (FillRC) {
3717         assert(TRI.getRegSizeInBits(*getRegClass(SrcReg)) ==
3718                    TRI.getRegSizeInBits(*FillRC) &&
3719                "Mismatched regclass size on folded subreg COPY");
3720         loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex, FillRC, &TRI);
3721         MachineInstr &LoadMI = *--InsertPt;
3722         MachineOperand &LoadDst = LoadMI.getOperand(0);
3723         assert(LoadDst.getSubReg() == 0 && "unexpected subreg on fill load");
3724         LoadDst.setSubReg(DstMO.getSubReg());
3725         LoadDst.setIsUndef();
3726         return &LoadMI;
3727       }
3728     }
3729   }
3730 
3731   // Cannot fold.
3732   return nullptr;
3733 }
3734 
3735 int llvm::isAArch64FrameOffsetLegal(const MachineInstr &MI,
3736                                     StackOffset &SOffset,
3737                                     bool *OutUseUnscaledOp,
3738                                     unsigned *OutUnscaledOp,
3739                                     int64_t *EmittableOffset) {
3740   // Set output values in case of early exit.
3741   if (EmittableOffset)
3742     *EmittableOffset = 0;
3743   if (OutUseUnscaledOp)
3744     *OutUseUnscaledOp = false;
3745   if (OutUnscaledOp)
3746     *OutUnscaledOp = 0;
3747 
3748   // Exit early for structured vector spills/fills as they can't take an
3749   // immediate offset.
3750   switch (MI.getOpcode()) {
3751   default:
3752     break;
3753   case AArch64::LD1Twov2d:
3754   case AArch64::LD1Threev2d:
3755   case AArch64::LD1Fourv2d:
3756   case AArch64::LD1Twov1d:
3757   case AArch64::LD1Threev1d:
3758   case AArch64::LD1Fourv1d:
3759   case AArch64::ST1Twov2d:
3760   case AArch64::ST1Threev2d:
3761   case AArch64::ST1Fourv2d:
3762   case AArch64::ST1Twov1d:
3763   case AArch64::ST1Threev1d:
3764   case AArch64::ST1Fourv1d:
3765   case AArch64::IRG:
3766   case AArch64::IRGstack:
3767   case AArch64::STGloop:
3768   case AArch64::STZGloop:
3769     return AArch64FrameOffsetCannotUpdate;
3770   }
3771 
3772   // Get the min/max offset and the scale.
3773   TypeSize ScaleValue(0U, false);
3774   unsigned Width;
3775   int64_t MinOff, MaxOff;
3776   if (!AArch64InstrInfo::getMemOpInfo(MI.getOpcode(), ScaleValue, Width, MinOff,
3777                                       MaxOff))
3778     llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal");
3779 
3780   // Construct the complete offset.
3781   bool IsMulVL = ScaleValue.isScalable();
3782   unsigned Scale = ScaleValue.getKnownMinSize();
3783   int64_t Offset = IsMulVL ? SOffset.getScalableBytes() : SOffset.getBytes();
3784 
3785   const MachineOperand &ImmOpnd =
3786       MI.getOperand(AArch64InstrInfo::getLoadStoreImmIdx(MI.getOpcode()));
3787   Offset += ImmOpnd.getImm() * Scale;
3788 
3789   // If the offset doesn't match the scale, we rewrite the instruction to
3790   // use the unscaled instruction instead. Likewise, if we have a negative
3791   // offset and there is an unscaled op to use.
3792   Optional<unsigned> UnscaledOp =
3793       AArch64InstrInfo::getUnscaledLdSt(MI.getOpcode());
3794   bool useUnscaledOp = UnscaledOp && (Offset % Scale || Offset < 0);
3795   if (useUnscaledOp &&
3796       !AArch64InstrInfo::getMemOpInfo(*UnscaledOp, ScaleValue, Width, MinOff,
3797                                       MaxOff))
3798     llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal");
3799 
3800   Scale = ScaleValue.getKnownMinSize();
3801   assert(IsMulVL == ScaleValue.isScalable() &&
3802          "Unscaled opcode has different value for scalable");
3803 
3804   int64_t Remainder = Offset % Scale;
3805   assert(!(Remainder && useUnscaledOp) &&
3806          "Cannot have remainder when using unscaled op");
3807 
3808   assert(MinOff < MaxOff && "Unexpected Min/Max offsets");
3809   int64_t NewOffset = Offset / Scale;
3810   if (MinOff <= NewOffset && NewOffset <= MaxOff)
3811     Offset = Remainder;
3812   else {
3813     NewOffset = NewOffset < 0 ? MinOff : MaxOff;
3814     Offset = Offset - NewOffset * Scale + Remainder;
3815   }
3816 
3817   if (EmittableOffset)
3818     *EmittableOffset = NewOffset;
3819   if (OutUseUnscaledOp)
3820     *OutUseUnscaledOp = useUnscaledOp;
3821   if (OutUnscaledOp && UnscaledOp)
3822     *OutUnscaledOp = *UnscaledOp;
3823 
3824   if (IsMulVL)
3825     SOffset = StackOffset(Offset, MVT::nxv1i8) +
3826               StackOffset(SOffset.getBytes(), MVT::i8);
3827   else
3828     SOffset = StackOffset(Offset, MVT::i8) +
3829               StackOffset(SOffset.getScalableBytes(), MVT::nxv1i8);
3830   return AArch64FrameOffsetCanUpdate |
3831          (SOffset ? 0 : AArch64FrameOffsetIsLegal);
3832 }
3833 
3834 bool llvm::rewriteAArch64FrameIndex(MachineInstr &MI, unsigned FrameRegIdx,
3835                                     unsigned FrameReg, StackOffset &Offset,
3836                                     const AArch64InstrInfo *TII) {
3837   unsigned Opcode = MI.getOpcode();
3838   unsigned ImmIdx = FrameRegIdx + 1;
3839 
3840   if (Opcode == AArch64::ADDSXri || Opcode == AArch64::ADDXri) {
3841     Offset += StackOffset(MI.getOperand(ImmIdx).getImm(), MVT::i8);
3842     emitFrameOffset(*MI.getParent(), MI, MI.getDebugLoc(),
3843                     MI.getOperand(0).getReg(), FrameReg, Offset, TII,
3844                     MachineInstr::NoFlags, (Opcode == AArch64::ADDSXri));
3845     MI.eraseFromParent();
3846     Offset = StackOffset();
3847     return true;
3848   }
3849 
3850   int64_t NewOffset;
3851   unsigned UnscaledOp;
3852   bool UseUnscaledOp;
3853   int Status = isAArch64FrameOffsetLegal(MI, Offset, &UseUnscaledOp,
3854                                          &UnscaledOp, &NewOffset);
3855   if (Status & AArch64FrameOffsetCanUpdate) {
3856     if (Status & AArch64FrameOffsetIsLegal)
3857       // Replace the FrameIndex with FrameReg.
3858       MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false);
3859     if (UseUnscaledOp)
3860       MI.setDesc(TII->get(UnscaledOp));
3861 
3862     MI.getOperand(ImmIdx).ChangeToImmediate(NewOffset);
3863     return !Offset;
3864   }
3865 
3866   return false;
3867 }
3868 
3869 void AArch64InstrInfo::getNoop(MCInst &NopInst) const {
3870   NopInst.setOpcode(AArch64::HINT);
3871   NopInst.addOperand(MCOperand::createImm(0));
3872 }
3873 
3874 // AArch64 supports MachineCombiner.
3875 bool AArch64InstrInfo::useMachineCombiner() const { return true; }
3876 
3877 // True when Opc sets flag
3878 static bool isCombineInstrSettingFlag(unsigned Opc) {
3879   switch (Opc) {
3880   case AArch64::ADDSWrr:
3881   case AArch64::ADDSWri:
3882   case AArch64::ADDSXrr:
3883   case AArch64::ADDSXri:
3884   case AArch64::SUBSWrr:
3885   case AArch64::SUBSXrr:
3886   // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
3887   case AArch64::SUBSWri:
3888   case AArch64::SUBSXri:
3889     return true;
3890   default:
3891     break;
3892   }
3893   return false;
3894 }
3895 
3896 // 32b Opcodes that can be combined with a MUL
3897 static bool isCombineInstrCandidate32(unsigned Opc) {
3898   switch (Opc) {
3899   case AArch64::ADDWrr:
3900   case AArch64::ADDWri:
3901   case AArch64::SUBWrr:
3902   case AArch64::ADDSWrr:
3903   case AArch64::ADDSWri:
3904   case AArch64::SUBSWrr:
3905   // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
3906   case AArch64::SUBWri:
3907   case AArch64::SUBSWri:
3908     return true;
3909   default:
3910     break;
3911   }
3912   return false;
3913 }
3914 
3915 // 64b Opcodes that can be combined with a MUL
3916 static bool isCombineInstrCandidate64(unsigned Opc) {
3917   switch (Opc) {
3918   case AArch64::ADDXrr:
3919   case AArch64::ADDXri:
3920   case AArch64::SUBXrr:
3921   case AArch64::ADDSXrr:
3922   case AArch64::ADDSXri:
3923   case AArch64::SUBSXrr:
3924   // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
3925   case AArch64::SUBXri:
3926   case AArch64::SUBSXri:
3927   case AArch64::ADDv8i8:
3928   case AArch64::ADDv16i8:
3929   case AArch64::ADDv4i16:
3930   case AArch64::ADDv8i16:
3931   case AArch64::ADDv2i32:
3932   case AArch64::ADDv4i32:
3933   case AArch64::SUBv8i8:
3934   case AArch64::SUBv16i8:
3935   case AArch64::SUBv4i16:
3936   case AArch64::SUBv8i16:
3937   case AArch64::SUBv2i32:
3938   case AArch64::SUBv4i32:
3939     return true;
3940   default:
3941     break;
3942   }
3943   return false;
3944 }
3945 
3946 // FP Opcodes that can be combined with a FMUL.
3947 static bool isCombineInstrCandidateFP(const MachineInstr &Inst) {
3948   switch (Inst.getOpcode()) {
3949   default:
3950     break;
3951   case AArch64::FADDHrr:
3952   case AArch64::FADDSrr:
3953   case AArch64::FADDDrr:
3954   case AArch64::FADDv4f16:
3955   case AArch64::FADDv8f16:
3956   case AArch64::FADDv2f32:
3957   case AArch64::FADDv2f64:
3958   case AArch64::FADDv4f32:
3959   case AArch64::FSUBHrr:
3960   case AArch64::FSUBSrr:
3961   case AArch64::FSUBDrr:
3962   case AArch64::FSUBv4f16:
3963   case AArch64::FSUBv8f16:
3964   case AArch64::FSUBv2f32:
3965   case AArch64::FSUBv2f64:
3966   case AArch64::FSUBv4f32:
3967     TargetOptions Options = Inst.getParent()->getParent()->getTarget().Options;
3968     // We can fuse FADD/FSUB with FMUL, if fusion is either allowed globally by
3969     // the target options or if FADD/FSUB has the contract fast-math flag.
3970     return Options.UnsafeFPMath ||
3971            Options.AllowFPOpFusion == FPOpFusion::Fast ||
3972            Inst.getFlag(MachineInstr::FmContract);
3973     return true;
3974   }
3975   return false;
3976 }
3977 
3978 // Opcodes that can be combined with a MUL
3979 static bool isCombineInstrCandidate(unsigned Opc) {
3980   return (isCombineInstrCandidate32(Opc) || isCombineInstrCandidate64(Opc));
3981 }
3982 
3983 //
3984 // Utility routine that checks if \param MO is defined by an
3985 // \param CombineOpc instruction in the basic block \param MBB
3986 static bool canCombine(MachineBasicBlock &MBB, MachineOperand &MO,
3987                        unsigned CombineOpc, unsigned ZeroReg = 0,
3988                        bool CheckZeroReg = false) {
3989   MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3990   MachineInstr *MI = nullptr;
3991 
3992   if (MO.isReg() && Register::isVirtualRegister(MO.getReg()))
3993     MI = MRI.getUniqueVRegDef(MO.getReg());
3994   // And it needs to be in the trace (otherwise, it won't have a depth).
3995   if (!MI || MI->getParent() != &MBB || (unsigned)MI->getOpcode() != CombineOpc)
3996     return false;
3997   // Must only used by the user we combine with.
3998   if (!MRI.hasOneNonDBGUse(MI->getOperand(0).getReg()))
3999     return false;
4000 
4001   if (CheckZeroReg) {
4002     assert(MI->getNumOperands() >= 4 && MI->getOperand(0).isReg() &&
4003            MI->getOperand(1).isReg() && MI->getOperand(2).isReg() &&
4004            MI->getOperand(3).isReg() && "MAdd/MSub must have a least 4 regs");
4005     // The third input reg must be zero.
4006     if (MI->getOperand(3).getReg() != ZeroReg)
4007       return false;
4008   }
4009 
4010   return true;
4011 }
4012 
4013 //
4014 // Is \param MO defined by an integer multiply and can be combined?
4015 static bool canCombineWithMUL(MachineBasicBlock &MBB, MachineOperand &MO,
4016                               unsigned MulOpc, unsigned ZeroReg) {
4017   return canCombine(MBB, MO, MulOpc, ZeroReg, true);
4018 }
4019 
4020 //
4021 // Is \param MO defined by a floating-point multiply and can be combined?
4022 static bool canCombineWithFMUL(MachineBasicBlock &MBB, MachineOperand &MO,
4023                                unsigned MulOpc) {
4024   return canCombine(MBB, MO, MulOpc);
4025 }
4026 
4027 // TODO: There are many more machine instruction opcodes to match:
4028 //       1. Other data types (integer, vectors)
4029 //       2. Other math / logic operations (xor, or)
4030 //       3. Other forms of the same operation (intrinsics and other variants)
4031 bool AArch64InstrInfo::isAssociativeAndCommutative(
4032     const MachineInstr &Inst) const {
4033   switch (Inst.getOpcode()) {
4034   case AArch64::FADDDrr:
4035   case AArch64::FADDSrr:
4036   case AArch64::FADDv2f32:
4037   case AArch64::FADDv2f64:
4038   case AArch64::FADDv4f32:
4039   case AArch64::FMULDrr:
4040   case AArch64::FMULSrr:
4041   case AArch64::FMULX32:
4042   case AArch64::FMULX64:
4043   case AArch64::FMULXv2f32:
4044   case AArch64::FMULXv2f64:
4045   case AArch64::FMULXv4f32:
4046   case AArch64::FMULv2f32:
4047   case AArch64::FMULv2f64:
4048   case AArch64::FMULv4f32:
4049     return Inst.getParent()->getParent()->getTarget().Options.UnsafeFPMath;
4050   default:
4051     return false;
4052   }
4053 }
4054 
4055 /// Find instructions that can be turned into madd.
4056 static bool getMaddPatterns(MachineInstr &Root,
4057                             SmallVectorImpl<MachineCombinerPattern> &Patterns) {
4058   unsigned Opc = Root.getOpcode();
4059   MachineBasicBlock &MBB = *Root.getParent();
4060   bool Found = false;
4061 
4062   if (!isCombineInstrCandidate(Opc))
4063     return false;
4064   if (isCombineInstrSettingFlag(Opc)) {
4065     int Cmp_NZCV = Root.findRegisterDefOperandIdx(AArch64::NZCV, true);
4066     // When NZCV is live bail out.
4067     if (Cmp_NZCV == -1)
4068       return false;
4069     unsigned NewOpc = convertToNonFlagSettingOpc(Root);
4070     // When opcode can't change bail out.
4071     // CHECKME: do we miss any cases for opcode conversion?
4072     if (NewOpc == Opc)
4073       return false;
4074     Opc = NewOpc;
4075   }
4076 
4077   auto setFound = [&](int Opcode, int Operand, unsigned ZeroReg,
4078                       MachineCombinerPattern Pattern) {
4079     if (canCombineWithMUL(MBB, Root.getOperand(Operand), Opcode, ZeroReg)) {
4080       Patterns.push_back(Pattern);
4081       Found = true;
4082     }
4083   };
4084 
4085   auto setVFound = [&](int Opcode, int Operand, MachineCombinerPattern Pattern) {
4086     if (canCombine(MBB, Root.getOperand(Operand), Opcode)) {
4087       Patterns.push_back(Pattern);
4088       Found = true;
4089     }
4090   };
4091 
4092   typedef MachineCombinerPattern MCP;
4093 
4094   switch (Opc) {
4095   default:
4096     break;
4097   case AArch64::ADDWrr:
4098     assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
4099            "ADDWrr does not have register operands");
4100     setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULADDW_OP1);
4101     setFound(AArch64::MADDWrrr, 2, AArch64::WZR, MCP::MULADDW_OP2);
4102     break;
4103   case AArch64::ADDXrr:
4104     setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULADDX_OP1);
4105     setFound(AArch64::MADDXrrr, 2, AArch64::XZR, MCP::MULADDX_OP2);
4106     break;
4107   case AArch64::SUBWrr:
4108     setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULSUBW_OP1);
4109     setFound(AArch64::MADDWrrr, 2, AArch64::WZR, MCP::MULSUBW_OP2);
4110     break;
4111   case AArch64::SUBXrr:
4112     setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULSUBX_OP1);
4113     setFound(AArch64::MADDXrrr, 2, AArch64::XZR, MCP::MULSUBX_OP2);
4114     break;
4115   case AArch64::ADDWri:
4116     setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULADDWI_OP1);
4117     break;
4118   case AArch64::ADDXri:
4119     setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULADDXI_OP1);
4120     break;
4121   case AArch64::SUBWri:
4122     setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULSUBWI_OP1);
4123     break;
4124   case AArch64::SUBXri:
4125     setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULSUBXI_OP1);
4126     break;
4127   case AArch64::ADDv8i8:
4128     setVFound(AArch64::MULv8i8, 1, MCP::MULADDv8i8_OP1);
4129     setVFound(AArch64::MULv8i8, 2, MCP::MULADDv8i8_OP2);
4130     break;
4131   case AArch64::ADDv16i8:
4132     setVFound(AArch64::MULv16i8, 1, MCP::MULADDv16i8_OP1);
4133     setVFound(AArch64::MULv16i8, 2, MCP::MULADDv16i8_OP2);
4134     break;
4135   case AArch64::ADDv4i16:
4136     setVFound(AArch64::MULv4i16, 1, MCP::MULADDv4i16_OP1);
4137     setVFound(AArch64::MULv4i16, 2, MCP::MULADDv4i16_OP2);
4138     setVFound(AArch64::MULv4i16_indexed, 1, MCP::MULADDv4i16_indexed_OP1);
4139     setVFound(AArch64::MULv4i16_indexed, 2, MCP::MULADDv4i16_indexed_OP2);
4140     break;
4141   case AArch64::ADDv8i16:
4142     setVFound(AArch64::MULv8i16, 1, MCP::MULADDv8i16_OP1);
4143     setVFound(AArch64::MULv8i16, 2, MCP::MULADDv8i16_OP2);
4144     setVFound(AArch64::MULv8i16_indexed, 1, MCP::MULADDv8i16_indexed_OP1);
4145     setVFound(AArch64::MULv8i16_indexed, 2, MCP::MULADDv8i16_indexed_OP2);
4146     break;
4147   case AArch64::ADDv2i32:
4148     setVFound(AArch64::MULv2i32, 1, MCP::MULADDv2i32_OP1);
4149     setVFound(AArch64::MULv2i32, 2, MCP::MULADDv2i32_OP2);
4150     setVFound(AArch64::MULv2i32_indexed, 1, MCP::MULADDv2i32_indexed_OP1);
4151     setVFound(AArch64::MULv2i32_indexed, 2, MCP::MULADDv2i32_indexed_OP2);
4152     break;
4153   case AArch64::ADDv4i32:
4154     setVFound(AArch64::MULv4i32, 1, MCP::MULADDv4i32_OP1);
4155     setVFound(AArch64::MULv4i32, 2, MCP::MULADDv4i32_OP2);
4156     setVFound(AArch64::MULv4i32_indexed, 1, MCP::MULADDv4i32_indexed_OP1);
4157     setVFound(AArch64::MULv4i32_indexed, 2, MCP::MULADDv4i32_indexed_OP2);
4158     break;
4159   case AArch64::SUBv8i8:
4160     setVFound(AArch64::MULv8i8, 1, MCP::MULSUBv8i8_OP1);
4161     setVFound(AArch64::MULv8i8, 2, MCP::MULSUBv8i8_OP2);
4162     break;
4163   case AArch64::SUBv16i8:
4164     setVFound(AArch64::MULv16i8, 1, MCP::MULSUBv16i8_OP1);
4165     setVFound(AArch64::MULv16i8, 2, MCP::MULSUBv16i8_OP2);
4166     break;
4167   case AArch64::SUBv4i16:
4168     setVFound(AArch64::MULv4i16, 1, MCP::MULSUBv4i16_OP1);
4169     setVFound(AArch64::MULv4i16, 2, MCP::MULSUBv4i16_OP2);
4170     setVFound(AArch64::MULv4i16_indexed, 1, MCP::MULSUBv4i16_indexed_OP1);
4171     setVFound(AArch64::MULv4i16_indexed, 2, MCP::MULSUBv4i16_indexed_OP2);
4172     break;
4173   case AArch64::SUBv8i16:
4174     setVFound(AArch64::MULv8i16, 1, MCP::MULSUBv8i16_OP1);
4175     setVFound(AArch64::MULv8i16, 2, MCP::MULSUBv8i16_OP2);
4176     setVFound(AArch64::MULv8i16_indexed, 1, MCP::MULSUBv8i16_indexed_OP1);
4177     setVFound(AArch64::MULv8i16_indexed, 2, MCP::MULSUBv8i16_indexed_OP2);
4178     break;
4179   case AArch64::SUBv2i32:
4180     setVFound(AArch64::MULv2i32, 1, MCP::MULSUBv2i32_OP1);
4181     setVFound(AArch64::MULv2i32, 2, MCP::MULSUBv2i32_OP2);
4182     setVFound(AArch64::MULv2i32_indexed, 1, MCP::MULSUBv2i32_indexed_OP1);
4183     setVFound(AArch64::MULv2i32_indexed, 2, MCP::MULSUBv2i32_indexed_OP2);
4184     break;
4185   case AArch64::SUBv4i32:
4186     setVFound(AArch64::MULv4i32, 1, MCP::MULSUBv4i32_OP1);
4187     setVFound(AArch64::MULv4i32, 2, MCP::MULSUBv4i32_OP2);
4188     setVFound(AArch64::MULv4i32_indexed, 1, MCP::MULSUBv4i32_indexed_OP1);
4189     setVFound(AArch64::MULv4i32_indexed, 2, MCP::MULSUBv4i32_indexed_OP2);
4190     break;
4191   }
4192   return Found;
4193 }
4194 /// Floating-Point Support
4195 
4196 /// Find instructions that can be turned into madd.
4197 static bool getFMAPatterns(MachineInstr &Root,
4198                            SmallVectorImpl<MachineCombinerPattern> &Patterns) {
4199 
4200   if (!isCombineInstrCandidateFP(Root))
4201     return false;
4202 
4203   MachineBasicBlock &MBB = *Root.getParent();
4204   bool Found = false;
4205 
4206   auto Match = [&](int Opcode, int Operand,
4207                    MachineCombinerPattern Pattern) -> bool {
4208     if (canCombineWithFMUL(MBB, Root.getOperand(Operand), Opcode)) {
4209       Patterns.push_back(Pattern);
4210       return true;
4211     }
4212     return false;
4213   };
4214 
4215   typedef MachineCombinerPattern MCP;
4216 
4217   switch (Root.getOpcode()) {
4218   default:
4219     assert(false && "Unsupported FP instruction in combiner\n");
4220     break;
4221   case AArch64::FADDHrr:
4222     assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
4223            "FADDHrr does not have register operands");
4224 
4225     Found  = Match(AArch64::FMULHrr, 1, MCP::FMULADDH_OP1);
4226     Found |= Match(AArch64::FMULHrr, 2, MCP::FMULADDH_OP2);
4227     break;
4228   case AArch64::FADDSrr:
4229     assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
4230            "FADDSrr does not have register operands");
4231 
4232     Found |= Match(AArch64::FMULSrr, 1, MCP::FMULADDS_OP1) ||
4233              Match(AArch64::FMULv1i32_indexed, 1, MCP::FMLAv1i32_indexed_OP1);
4234 
4235     Found |= Match(AArch64::FMULSrr, 2, MCP::FMULADDS_OP2) ||
4236              Match(AArch64::FMULv1i32_indexed, 2, MCP::FMLAv1i32_indexed_OP2);
4237     break;
4238   case AArch64::FADDDrr:
4239     Found |= Match(AArch64::FMULDrr, 1, MCP::FMULADDD_OP1) ||
4240              Match(AArch64::FMULv1i64_indexed, 1, MCP::FMLAv1i64_indexed_OP1);
4241 
4242     Found |= Match(AArch64::FMULDrr, 2, MCP::FMULADDD_OP2) ||
4243              Match(AArch64::FMULv1i64_indexed, 2, MCP::FMLAv1i64_indexed_OP2);
4244     break;
4245   case AArch64::FADDv4f16:
4246     Found |= Match(AArch64::FMULv4i16_indexed, 1, MCP::FMLAv4i16_indexed_OP1) ||
4247              Match(AArch64::FMULv4f16, 1, MCP::FMLAv4f16_OP1);
4248 
4249     Found |= Match(AArch64::FMULv4i16_indexed, 2, MCP::FMLAv4i16_indexed_OP2) ||
4250              Match(AArch64::FMULv4f16, 2, MCP::FMLAv4f16_OP2);
4251     break;
4252   case AArch64::FADDv8f16:
4253     Found |= Match(AArch64::FMULv8i16_indexed, 1, MCP::FMLAv8i16_indexed_OP1) ||
4254              Match(AArch64::FMULv8f16, 1, MCP::FMLAv8f16_OP1);
4255 
4256     Found |= Match(AArch64::FMULv8i16_indexed, 2, MCP::FMLAv8i16_indexed_OP2) ||
4257              Match(AArch64::FMULv8f16, 2, MCP::FMLAv8f16_OP2);
4258     break;
4259   case AArch64::FADDv2f32:
4260     Found |= Match(AArch64::FMULv2i32_indexed, 1, MCP::FMLAv2i32_indexed_OP1) ||
4261              Match(AArch64::FMULv2f32, 1, MCP::FMLAv2f32_OP1);
4262 
4263     Found |= Match(AArch64::FMULv2i32_indexed, 2, MCP::FMLAv2i32_indexed_OP2) ||
4264              Match(AArch64::FMULv2f32, 2, MCP::FMLAv2f32_OP2);
4265     break;
4266   case AArch64::FADDv2f64:
4267     Found |= Match(AArch64::FMULv2i64_indexed, 1, MCP::FMLAv2i64_indexed_OP1) ||
4268              Match(AArch64::FMULv2f64, 1, MCP::FMLAv2f64_OP1);
4269 
4270     Found |= Match(AArch64::FMULv2i64_indexed, 2, MCP::FMLAv2i64_indexed_OP2) ||
4271              Match(AArch64::FMULv2f64, 2, MCP::FMLAv2f64_OP2);
4272     break;
4273   case AArch64::FADDv4f32:
4274     Found |= Match(AArch64::FMULv4i32_indexed, 1, MCP::FMLAv4i32_indexed_OP1) ||
4275              Match(AArch64::FMULv4f32, 1, MCP::FMLAv4f32_OP1);
4276 
4277     Found |= Match(AArch64::FMULv4i32_indexed, 2, MCP::FMLAv4i32_indexed_OP2) ||
4278              Match(AArch64::FMULv4f32, 2, MCP::FMLAv4f32_OP2);
4279     break;
4280   case AArch64::FSUBHrr:
4281     Found  = Match(AArch64::FMULHrr, 1, MCP::FMULSUBH_OP1);
4282     Found |= Match(AArch64::FMULHrr, 2, MCP::FMULSUBH_OP2);
4283     Found |= Match(AArch64::FNMULHrr, 1, MCP::FNMULSUBH_OP1);
4284     break;
4285   case AArch64::FSUBSrr:
4286     Found = Match(AArch64::FMULSrr, 1, MCP::FMULSUBS_OP1);
4287 
4288     Found |= Match(AArch64::FMULSrr, 2, MCP::FMULSUBS_OP2) ||
4289              Match(AArch64::FMULv1i32_indexed, 2, MCP::FMLSv1i32_indexed_OP2);
4290 
4291     Found |= Match(AArch64::FNMULSrr, 1, MCP::FNMULSUBS_OP1);
4292     break;
4293   case AArch64::FSUBDrr:
4294     Found = Match(AArch64::FMULDrr, 1, MCP::FMULSUBD_OP1);
4295 
4296     Found |= Match(AArch64::FMULDrr, 2, MCP::FMULSUBD_OP2) ||
4297              Match(AArch64::FMULv1i64_indexed, 2, MCP::FMLSv1i64_indexed_OP2);
4298 
4299     Found |= Match(AArch64::FNMULDrr, 1, MCP::FNMULSUBD_OP1);
4300     break;
4301   case AArch64::FSUBv4f16:
4302     Found |= Match(AArch64::FMULv4i16_indexed, 2, MCP::FMLSv4i16_indexed_OP2) ||
4303              Match(AArch64::FMULv4f16, 2, MCP::FMLSv4f16_OP2);
4304 
4305     Found |= Match(AArch64::FMULv4i16_indexed, 1, MCP::FMLSv4i16_indexed_OP1) ||
4306              Match(AArch64::FMULv4f16, 1, MCP::FMLSv4f16_OP1);
4307     break;
4308   case AArch64::FSUBv8f16:
4309     Found |= Match(AArch64::FMULv8i16_indexed, 2, MCP::FMLSv8i16_indexed_OP2) ||
4310              Match(AArch64::FMULv8f16, 2, MCP::FMLSv8f16_OP2);
4311 
4312     Found |= Match(AArch64::FMULv8i16_indexed, 1, MCP::FMLSv8i16_indexed_OP1) ||
4313              Match(AArch64::FMULv8f16, 1, MCP::FMLSv8f16_OP1);
4314     break;
4315   case AArch64::FSUBv2f32:
4316     Found |= Match(AArch64::FMULv2i32_indexed, 2, MCP::FMLSv2i32_indexed_OP2) ||
4317              Match(AArch64::FMULv2f32, 2, MCP::FMLSv2f32_OP2);
4318 
4319     Found |= Match(AArch64::FMULv2i32_indexed, 1, MCP::FMLSv2i32_indexed_OP1) ||
4320              Match(AArch64::FMULv2f32, 1, MCP::FMLSv2f32_OP1);
4321     break;
4322   case AArch64::FSUBv2f64:
4323     Found |= Match(AArch64::FMULv2i64_indexed, 2, MCP::FMLSv2i64_indexed_OP2) ||
4324              Match(AArch64::FMULv2f64, 2, MCP::FMLSv2f64_OP2);
4325 
4326     Found |= Match(AArch64::FMULv2i64_indexed, 1, MCP::FMLSv2i64_indexed_OP1) ||
4327              Match(AArch64::FMULv2f64, 1, MCP::FMLSv2f64_OP1);
4328     break;
4329   case AArch64::FSUBv4f32:
4330     Found |= Match(AArch64::FMULv4i32_indexed, 2, MCP::FMLSv4i32_indexed_OP2) ||
4331              Match(AArch64::FMULv4f32, 2, MCP::FMLSv4f32_OP2);
4332 
4333     Found |= Match(AArch64::FMULv4i32_indexed, 1, MCP::FMLSv4i32_indexed_OP1) ||
4334              Match(AArch64::FMULv4f32, 1, MCP::FMLSv4f32_OP1);
4335     break;
4336   }
4337   return Found;
4338 }
4339 
4340 /// Return true when a code sequence can improve throughput. It
4341 /// should be called only for instructions in loops.
4342 /// \param Pattern - combiner pattern
4343 bool AArch64InstrInfo::isThroughputPattern(
4344     MachineCombinerPattern Pattern) const {
4345   switch (Pattern) {
4346   default:
4347     break;
4348   case MachineCombinerPattern::FMULADDH_OP1:
4349   case MachineCombinerPattern::FMULADDH_OP2:
4350   case MachineCombinerPattern::FMULSUBH_OP1:
4351   case MachineCombinerPattern::FMULSUBH_OP2:
4352   case MachineCombinerPattern::FMULADDS_OP1:
4353   case MachineCombinerPattern::FMULADDS_OP2:
4354   case MachineCombinerPattern::FMULSUBS_OP1:
4355   case MachineCombinerPattern::FMULSUBS_OP2:
4356   case MachineCombinerPattern::FMULADDD_OP1:
4357   case MachineCombinerPattern::FMULADDD_OP2:
4358   case MachineCombinerPattern::FMULSUBD_OP1:
4359   case MachineCombinerPattern::FMULSUBD_OP2:
4360   case MachineCombinerPattern::FNMULSUBH_OP1:
4361   case MachineCombinerPattern::FNMULSUBS_OP1:
4362   case MachineCombinerPattern::FNMULSUBD_OP1:
4363   case MachineCombinerPattern::FMLAv4i16_indexed_OP1:
4364   case MachineCombinerPattern::FMLAv4i16_indexed_OP2:
4365   case MachineCombinerPattern::FMLAv8i16_indexed_OP1:
4366   case MachineCombinerPattern::FMLAv8i16_indexed_OP2:
4367   case MachineCombinerPattern::FMLAv1i32_indexed_OP1:
4368   case MachineCombinerPattern::FMLAv1i32_indexed_OP2:
4369   case MachineCombinerPattern::FMLAv1i64_indexed_OP1:
4370   case MachineCombinerPattern::FMLAv1i64_indexed_OP2:
4371   case MachineCombinerPattern::FMLAv4f16_OP2:
4372   case MachineCombinerPattern::FMLAv4f16_OP1:
4373   case MachineCombinerPattern::FMLAv8f16_OP1:
4374   case MachineCombinerPattern::FMLAv8f16_OP2:
4375   case MachineCombinerPattern::FMLAv2f32_OP2:
4376   case MachineCombinerPattern::FMLAv2f32_OP1:
4377   case MachineCombinerPattern::FMLAv2f64_OP1:
4378   case MachineCombinerPattern::FMLAv2f64_OP2:
4379   case MachineCombinerPattern::FMLAv2i32_indexed_OP1:
4380   case MachineCombinerPattern::FMLAv2i32_indexed_OP2:
4381   case MachineCombinerPattern::FMLAv2i64_indexed_OP1:
4382   case MachineCombinerPattern::FMLAv2i64_indexed_OP2:
4383   case MachineCombinerPattern::FMLAv4f32_OP1:
4384   case MachineCombinerPattern::FMLAv4f32_OP2:
4385   case MachineCombinerPattern::FMLAv4i32_indexed_OP1:
4386   case MachineCombinerPattern::FMLAv4i32_indexed_OP2:
4387   case MachineCombinerPattern::FMLSv4i16_indexed_OP1:
4388   case MachineCombinerPattern::FMLSv4i16_indexed_OP2:
4389   case MachineCombinerPattern::FMLSv8i16_indexed_OP1:
4390   case MachineCombinerPattern::FMLSv8i16_indexed_OP2:
4391   case MachineCombinerPattern::FMLSv1i32_indexed_OP2:
4392   case MachineCombinerPattern::FMLSv1i64_indexed_OP2:
4393   case MachineCombinerPattern::FMLSv2i32_indexed_OP2:
4394   case MachineCombinerPattern::FMLSv2i64_indexed_OP2:
4395   case MachineCombinerPattern::FMLSv4f16_OP1:
4396   case MachineCombinerPattern::FMLSv4f16_OP2:
4397   case MachineCombinerPattern::FMLSv8f16_OP1:
4398   case MachineCombinerPattern::FMLSv8f16_OP2:
4399   case MachineCombinerPattern::FMLSv2f32_OP2:
4400   case MachineCombinerPattern::FMLSv2f64_OP2:
4401   case MachineCombinerPattern::FMLSv4i32_indexed_OP2:
4402   case MachineCombinerPattern::FMLSv4f32_OP2:
4403   case MachineCombinerPattern::MULADDv8i8_OP1:
4404   case MachineCombinerPattern::MULADDv8i8_OP2:
4405   case MachineCombinerPattern::MULADDv16i8_OP1:
4406   case MachineCombinerPattern::MULADDv16i8_OP2:
4407   case MachineCombinerPattern::MULADDv4i16_OP1:
4408   case MachineCombinerPattern::MULADDv4i16_OP2:
4409   case MachineCombinerPattern::MULADDv8i16_OP1:
4410   case MachineCombinerPattern::MULADDv8i16_OP2:
4411   case MachineCombinerPattern::MULADDv2i32_OP1:
4412   case MachineCombinerPattern::MULADDv2i32_OP2:
4413   case MachineCombinerPattern::MULADDv4i32_OP1:
4414   case MachineCombinerPattern::MULADDv4i32_OP2:
4415   case MachineCombinerPattern::MULSUBv8i8_OP1:
4416   case MachineCombinerPattern::MULSUBv8i8_OP2:
4417   case MachineCombinerPattern::MULSUBv16i8_OP1:
4418   case MachineCombinerPattern::MULSUBv16i8_OP2:
4419   case MachineCombinerPattern::MULSUBv4i16_OP1:
4420   case MachineCombinerPattern::MULSUBv4i16_OP2:
4421   case MachineCombinerPattern::MULSUBv8i16_OP1:
4422   case MachineCombinerPattern::MULSUBv8i16_OP2:
4423   case MachineCombinerPattern::MULSUBv2i32_OP1:
4424   case MachineCombinerPattern::MULSUBv2i32_OP2:
4425   case MachineCombinerPattern::MULSUBv4i32_OP1:
4426   case MachineCombinerPattern::MULSUBv4i32_OP2:
4427   case MachineCombinerPattern::MULADDv4i16_indexed_OP1:
4428   case MachineCombinerPattern::MULADDv4i16_indexed_OP2:
4429   case MachineCombinerPattern::MULADDv8i16_indexed_OP1:
4430   case MachineCombinerPattern::MULADDv8i16_indexed_OP2:
4431   case MachineCombinerPattern::MULADDv2i32_indexed_OP1:
4432   case MachineCombinerPattern::MULADDv2i32_indexed_OP2:
4433   case MachineCombinerPattern::MULADDv4i32_indexed_OP1:
4434   case MachineCombinerPattern::MULADDv4i32_indexed_OP2:
4435   case MachineCombinerPattern::MULSUBv4i16_indexed_OP1:
4436   case MachineCombinerPattern::MULSUBv4i16_indexed_OP2:
4437   case MachineCombinerPattern::MULSUBv8i16_indexed_OP1:
4438   case MachineCombinerPattern::MULSUBv8i16_indexed_OP2:
4439   case MachineCombinerPattern::MULSUBv2i32_indexed_OP1:
4440   case MachineCombinerPattern::MULSUBv2i32_indexed_OP2:
4441   case MachineCombinerPattern::MULSUBv4i32_indexed_OP1:
4442   case MachineCombinerPattern::MULSUBv4i32_indexed_OP2:
4443     return true;
4444   } // end switch (Pattern)
4445   return false;
4446 }
4447 /// Return true when there is potentially a faster code sequence for an
4448 /// instruction chain ending in \p Root. All potential patterns are listed in
4449 /// the \p Pattern vector. Pattern should be sorted in priority order since the
4450 /// pattern evaluator stops checking as soon as it finds a faster sequence.
4451 
4452 bool AArch64InstrInfo::getMachineCombinerPatterns(
4453     MachineInstr &Root,
4454     SmallVectorImpl<MachineCombinerPattern> &Patterns) const {
4455   // Integer patterns
4456   if (getMaddPatterns(Root, Patterns))
4457     return true;
4458   // Floating point patterns
4459   if (getFMAPatterns(Root, Patterns))
4460     return true;
4461 
4462   return TargetInstrInfo::getMachineCombinerPatterns(Root, Patterns);
4463 }
4464 
4465 enum class FMAInstKind { Default, Indexed, Accumulator };
4466 /// genFusedMultiply - Generate fused multiply instructions.
4467 /// This function supports both integer and floating point instructions.
4468 /// A typical example:
4469 ///  F|MUL I=A,B,0
4470 ///  F|ADD R,I,C
4471 ///  ==> F|MADD R,A,B,C
4472 /// \param MF Containing MachineFunction
4473 /// \param MRI Register information
4474 /// \param TII Target information
4475 /// \param Root is the F|ADD instruction
4476 /// \param [out] InsInstrs is a vector of machine instructions and will
4477 /// contain the generated madd instruction
4478 /// \param IdxMulOpd is index of operand in Root that is the result of
4479 /// the F|MUL. In the example above IdxMulOpd is 1.
4480 /// \param MaddOpc the opcode fo the f|madd instruction
4481 /// \param RC Register class of operands
4482 /// \param kind of fma instruction (addressing mode) to be generated
4483 /// \param ReplacedAddend is the result register from the instruction
4484 /// replacing the non-combined operand, if any.
4485 static MachineInstr *
4486 genFusedMultiply(MachineFunction &MF, MachineRegisterInfo &MRI,
4487                  const TargetInstrInfo *TII, MachineInstr &Root,
4488                  SmallVectorImpl<MachineInstr *> &InsInstrs, unsigned IdxMulOpd,
4489                  unsigned MaddOpc, const TargetRegisterClass *RC,
4490                  FMAInstKind kind = FMAInstKind::Default,
4491                  const Register *ReplacedAddend = nullptr) {
4492   assert(IdxMulOpd == 1 || IdxMulOpd == 2);
4493 
4494   unsigned IdxOtherOpd = IdxMulOpd == 1 ? 2 : 1;
4495   MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg());
4496   Register ResultReg = Root.getOperand(0).getReg();
4497   Register SrcReg0 = MUL->getOperand(1).getReg();
4498   bool Src0IsKill = MUL->getOperand(1).isKill();
4499   Register SrcReg1 = MUL->getOperand(2).getReg();
4500   bool Src1IsKill = MUL->getOperand(2).isKill();
4501 
4502   unsigned SrcReg2;
4503   bool Src2IsKill;
4504   if (ReplacedAddend) {
4505     // If we just generated a new addend, we must be it's only use.
4506     SrcReg2 = *ReplacedAddend;
4507     Src2IsKill = true;
4508   } else {
4509     SrcReg2 = Root.getOperand(IdxOtherOpd).getReg();
4510     Src2IsKill = Root.getOperand(IdxOtherOpd).isKill();
4511   }
4512 
4513   if (Register::isVirtualRegister(ResultReg))
4514     MRI.constrainRegClass(ResultReg, RC);
4515   if (Register::isVirtualRegister(SrcReg0))
4516     MRI.constrainRegClass(SrcReg0, RC);
4517   if (Register::isVirtualRegister(SrcReg1))
4518     MRI.constrainRegClass(SrcReg1, RC);
4519   if (Register::isVirtualRegister(SrcReg2))
4520     MRI.constrainRegClass(SrcReg2, RC);
4521 
4522   MachineInstrBuilder MIB;
4523   if (kind == FMAInstKind::Default)
4524     MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
4525               .addReg(SrcReg0, getKillRegState(Src0IsKill))
4526               .addReg(SrcReg1, getKillRegState(Src1IsKill))
4527               .addReg(SrcReg2, getKillRegState(Src2IsKill));
4528   else if (kind == FMAInstKind::Indexed)
4529     MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
4530               .addReg(SrcReg2, getKillRegState(Src2IsKill))
4531               .addReg(SrcReg0, getKillRegState(Src0IsKill))
4532               .addReg(SrcReg1, getKillRegState(Src1IsKill))
4533               .addImm(MUL->getOperand(3).getImm());
4534   else if (kind == FMAInstKind::Accumulator)
4535     MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
4536               .addReg(SrcReg2, getKillRegState(Src2IsKill))
4537               .addReg(SrcReg0, getKillRegState(Src0IsKill))
4538               .addReg(SrcReg1, getKillRegState(Src1IsKill));
4539   else
4540     assert(false && "Invalid FMA instruction kind \n");
4541   // Insert the MADD (MADD, FMA, FMS, FMLA, FMSL)
4542   InsInstrs.push_back(MIB);
4543   return MUL;
4544 }
4545 
4546 /// genFusedMultiplyAcc - Helper to generate fused multiply accumulate
4547 /// instructions.
4548 ///
4549 /// \see genFusedMultiply
4550 static MachineInstr *genFusedMultiplyAcc(
4551     MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII,
4552     MachineInstr &Root, SmallVectorImpl<MachineInstr *> &InsInstrs,
4553     unsigned IdxMulOpd, unsigned MaddOpc, const TargetRegisterClass *RC) {
4554   return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
4555                           FMAInstKind::Accumulator);
4556 }
4557 
4558 /// genNeg - Helper to generate an intermediate negation of the second operand
4559 /// of Root
4560 static Register genNeg(MachineFunction &MF, MachineRegisterInfo &MRI,
4561                        const TargetInstrInfo *TII, MachineInstr &Root,
4562                        SmallVectorImpl<MachineInstr *> &InsInstrs,
4563                        DenseMap<unsigned, unsigned> &InstrIdxForVirtReg,
4564                        unsigned MnegOpc, const TargetRegisterClass *RC) {
4565   Register NewVR = MRI.createVirtualRegister(RC);
4566   MachineInstrBuilder MIB =
4567       BuildMI(MF, Root.getDebugLoc(), TII->get(MnegOpc), NewVR)
4568           .add(Root.getOperand(2));
4569   InsInstrs.push_back(MIB);
4570 
4571   assert(InstrIdxForVirtReg.empty());
4572   InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4573 
4574   return NewVR;
4575 }
4576 
4577 /// genFusedMultiplyAccNeg - Helper to generate fused multiply accumulate
4578 /// instructions with an additional negation of the accumulator
4579 static MachineInstr *genFusedMultiplyAccNeg(
4580     MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII,
4581     MachineInstr &Root, SmallVectorImpl<MachineInstr *> &InsInstrs,
4582     DenseMap<unsigned, unsigned> &InstrIdxForVirtReg, unsigned IdxMulOpd,
4583     unsigned MaddOpc, unsigned MnegOpc, const TargetRegisterClass *RC) {
4584   assert(IdxMulOpd == 1);
4585 
4586   Register NewVR =
4587       genNeg(MF, MRI, TII, Root, InsInstrs, InstrIdxForVirtReg, MnegOpc, RC);
4588   return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
4589                           FMAInstKind::Accumulator, &NewVR);
4590 }
4591 
4592 /// genFusedMultiplyIdx - Helper to generate fused multiply accumulate
4593 /// instructions.
4594 ///
4595 /// \see genFusedMultiply
4596 static MachineInstr *genFusedMultiplyIdx(
4597     MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII,
4598     MachineInstr &Root, SmallVectorImpl<MachineInstr *> &InsInstrs,
4599     unsigned IdxMulOpd, unsigned MaddOpc, const TargetRegisterClass *RC) {
4600   return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
4601                           FMAInstKind::Indexed);
4602 }
4603 
4604 /// genFusedMultiplyAccNeg - Helper to generate fused multiply accumulate
4605 /// instructions with an additional negation of the accumulator
4606 static MachineInstr *genFusedMultiplyIdxNeg(
4607     MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII,
4608     MachineInstr &Root, SmallVectorImpl<MachineInstr *> &InsInstrs,
4609     DenseMap<unsigned, unsigned> &InstrIdxForVirtReg, unsigned IdxMulOpd,
4610     unsigned MaddOpc, unsigned MnegOpc, const TargetRegisterClass *RC) {
4611   assert(IdxMulOpd == 1);
4612 
4613   Register NewVR =
4614       genNeg(MF, MRI, TII, Root, InsInstrs, InstrIdxForVirtReg, MnegOpc, RC);
4615 
4616   return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
4617                           FMAInstKind::Indexed, &NewVR);
4618 }
4619 
4620 /// genMaddR - Generate madd instruction and combine mul and add using
4621 /// an extra virtual register
4622 /// Example - an ADD intermediate needs to be stored in a register:
4623 ///   MUL I=A,B,0
4624 ///   ADD R,I,Imm
4625 ///   ==> ORR  V, ZR, Imm
4626 ///   ==> MADD R,A,B,V
4627 /// \param MF Containing MachineFunction
4628 /// \param MRI Register information
4629 /// \param TII Target information
4630 /// \param Root is the ADD instruction
4631 /// \param [out] InsInstrs is a vector of machine instructions and will
4632 /// contain the generated madd instruction
4633 /// \param IdxMulOpd is index of operand in Root that is the result of
4634 /// the MUL. In the example above IdxMulOpd is 1.
4635 /// \param MaddOpc the opcode fo the madd instruction
4636 /// \param VR is a virtual register that holds the value of an ADD operand
4637 /// (V in the example above).
4638 /// \param RC Register class of operands
4639 static MachineInstr *genMaddR(MachineFunction &MF, MachineRegisterInfo &MRI,
4640                               const TargetInstrInfo *TII, MachineInstr &Root,
4641                               SmallVectorImpl<MachineInstr *> &InsInstrs,
4642                               unsigned IdxMulOpd, unsigned MaddOpc, unsigned VR,
4643                               const TargetRegisterClass *RC) {
4644   assert(IdxMulOpd == 1 || IdxMulOpd == 2);
4645 
4646   MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg());
4647   Register ResultReg = Root.getOperand(0).getReg();
4648   Register SrcReg0 = MUL->getOperand(1).getReg();
4649   bool Src0IsKill = MUL->getOperand(1).isKill();
4650   Register SrcReg1 = MUL->getOperand(2).getReg();
4651   bool Src1IsKill = MUL->getOperand(2).isKill();
4652 
4653   if (Register::isVirtualRegister(ResultReg))
4654     MRI.constrainRegClass(ResultReg, RC);
4655   if (Register::isVirtualRegister(SrcReg0))
4656     MRI.constrainRegClass(SrcReg0, RC);
4657   if (Register::isVirtualRegister(SrcReg1))
4658     MRI.constrainRegClass(SrcReg1, RC);
4659   if (Register::isVirtualRegister(VR))
4660     MRI.constrainRegClass(VR, RC);
4661 
4662   MachineInstrBuilder MIB =
4663       BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg)
4664           .addReg(SrcReg0, getKillRegState(Src0IsKill))
4665           .addReg(SrcReg1, getKillRegState(Src1IsKill))
4666           .addReg(VR);
4667   // Insert the MADD
4668   InsInstrs.push_back(MIB);
4669   return MUL;
4670 }
4671 
4672 /// When getMachineCombinerPatterns() finds potential patterns,
4673 /// this function generates the instructions that could replace the
4674 /// original code sequence
4675 void AArch64InstrInfo::genAlternativeCodeSequence(
4676     MachineInstr &Root, MachineCombinerPattern Pattern,
4677     SmallVectorImpl<MachineInstr *> &InsInstrs,
4678     SmallVectorImpl<MachineInstr *> &DelInstrs,
4679     DenseMap<unsigned, unsigned> &InstrIdxForVirtReg) const {
4680   MachineBasicBlock &MBB = *Root.getParent();
4681   MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
4682   MachineFunction &MF = *MBB.getParent();
4683   const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo();
4684 
4685   MachineInstr *MUL;
4686   const TargetRegisterClass *RC;
4687   unsigned Opc;
4688   switch (Pattern) {
4689   default:
4690     // Reassociate instructions.
4691     TargetInstrInfo::genAlternativeCodeSequence(Root, Pattern, InsInstrs,
4692                                                 DelInstrs, InstrIdxForVirtReg);
4693     return;
4694   case MachineCombinerPattern::MULADDW_OP1:
4695   case MachineCombinerPattern::MULADDX_OP1:
4696     // MUL I=A,B,0
4697     // ADD R,I,C
4698     // ==> MADD R,A,B,C
4699     // --- Create(MADD);
4700     if (Pattern == MachineCombinerPattern::MULADDW_OP1) {
4701       Opc = AArch64::MADDWrrr;
4702       RC = &AArch64::GPR32RegClass;
4703     } else {
4704       Opc = AArch64::MADDXrrr;
4705       RC = &AArch64::GPR64RegClass;
4706     }
4707     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4708     break;
4709   case MachineCombinerPattern::MULADDW_OP2:
4710   case MachineCombinerPattern::MULADDX_OP2:
4711     // MUL I=A,B,0
4712     // ADD R,C,I
4713     // ==> MADD R,A,B,C
4714     // --- Create(MADD);
4715     if (Pattern == MachineCombinerPattern::MULADDW_OP2) {
4716       Opc = AArch64::MADDWrrr;
4717       RC = &AArch64::GPR32RegClass;
4718     } else {
4719       Opc = AArch64::MADDXrrr;
4720       RC = &AArch64::GPR64RegClass;
4721     }
4722     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4723     break;
4724   case MachineCombinerPattern::MULADDWI_OP1:
4725   case MachineCombinerPattern::MULADDXI_OP1: {
4726     // MUL I=A,B,0
4727     // ADD R,I,Imm
4728     // ==> ORR  V, ZR, Imm
4729     // ==> MADD R,A,B,V
4730     // --- Create(MADD);
4731     const TargetRegisterClass *OrrRC;
4732     unsigned BitSize, OrrOpc, ZeroReg;
4733     if (Pattern == MachineCombinerPattern::MULADDWI_OP1) {
4734       OrrOpc = AArch64::ORRWri;
4735       OrrRC = &AArch64::GPR32spRegClass;
4736       BitSize = 32;
4737       ZeroReg = AArch64::WZR;
4738       Opc = AArch64::MADDWrrr;
4739       RC = &AArch64::GPR32RegClass;
4740     } else {
4741       OrrOpc = AArch64::ORRXri;
4742       OrrRC = &AArch64::GPR64spRegClass;
4743       BitSize = 64;
4744       ZeroReg = AArch64::XZR;
4745       Opc = AArch64::MADDXrrr;
4746       RC = &AArch64::GPR64RegClass;
4747     }
4748     Register NewVR = MRI.createVirtualRegister(OrrRC);
4749     uint64_t Imm = Root.getOperand(2).getImm();
4750 
4751     if (Root.getOperand(3).isImm()) {
4752       unsigned Val = Root.getOperand(3).getImm();
4753       Imm = Imm << Val;
4754     }
4755     uint64_t UImm = SignExtend64(Imm, BitSize);
4756     uint64_t Encoding;
4757     if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) {
4758       MachineInstrBuilder MIB1 =
4759           BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR)
4760               .addReg(ZeroReg)
4761               .addImm(Encoding);
4762       InsInstrs.push_back(MIB1);
4763       InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4764       MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
4765     }
4766     break;
4767   }
4768   case MachineCombinerPattern::MULSUBW_OP1:
4769   case MachineCombinerPattern::MULSUBX_OP1: {
4770     // MUL I=A,B,0
4771     // SUB R,I, C
4772     // ==> SUB  V, 0, C
4773     // ==> MADD R,A,B,V // = -C + A*B
4774     // --- Create(MADD);
4775     const TargetRegisterClass *SubRC;
4776     unsigned SubOpc, ZeroReg;
4777     if (Pattern == MachineCombinerPattern::MULSUBW_OP1) {
4778       SubOpc = AArch64::SUBWrr;
4779       SubRC = &AArch64::GPR32spRegClass;
4780       ZeroReg = AArch64::WZR;
4781       Opc = AArch64::MADDWrrr;
4782       RC = &AArch64::GPR32RegClass;
4783     } else {
4784       SubOpc = AArch64::SUBXrr;
4785       SubRC = &AArch64::GPR64spRegClass;
4786       ZeroReg = AArch64::XZR;
4787       Opc = AArch64::MADDXrrr;
4788       RC = &AArch64::GPR64RegClass;
4789     }
4790     Register NewVR = MRI.createVirtualRegister(SubRC);
4791     // SUB NewVR, 0, C
4792     MachineInstrBuilder MIB1 =
4793         BuildMI(MF, Root.getDebugLoc(), TII->get(SubOpc), NewVR)
4794             .addReg(ZeroReg)
4795             .add(Root.getOperand(2));
4796     InsInstrs.push_back(MIB1);
4797     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4798     MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
4799     break;
4800   }
4801   case MachineCombinerPattern::MULSUBW_OP2:
4802   case MachineCombinerPattern::MULSUBX_OP2:
4803     // MUL I=A,B,0
4804     // SUB R,C,I
4805     // ==> MSUB R,A,B,C (computes C - A*B)
4806     // --- Create(MSUB);
4807     if (Pattern == MachineCombinerPattern::MULSUBW_OP2) {
4808       Opc = AArch64::MSUBWrrr;
4809       RC = &AArch64::GPR32RegClass;
4810     } else {
4811       Opc = AArch64::MSUBXrrr;
4812       RC = &AArch64::GPR64RegClass;
4813     }
4814     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4815     break;
4816   case MachineCombinerPattern::MULSUBWI_OP1:
4817   case MachineCombinerPattern::MULSUBXI_OP1: {
4818     // MUL I=A,B,0
4819     // SUB R,I, Imm
4820     // ==> ORR  V, ZR, -Imm
4821     // ==> MADD R,A,B,V // = -Imm + A*B
4822     // --- Create(MADD);
4823     const TargetRegisterClass *OrrRC;
4824     unsigned BitSize, OrrOpc, ZeroReg;
4825     if (Pattern == MachineCombinerPattern::MULSUBWI_OP1) {
4826       OrrOpc = AArch64::ORRWri;
4827       OrrRC = &AArch64::GPR32spRegClass;
4828       BitSize = 32;
4829       ZeroReg = AArch64::WZR;
4830       Opc = AArch64::MADDWrrr;
4831       RC = &AArch64::GPR32RegClass;
4832     } else {
4833       OrrOpc = AArch64::ORRXri;
4834       OrrRC = &AArch64::GPR64spRegClass;
4835       BitSize = 64;
4836       ZeroReg = AArch64::XZR;
4837       Opc = AArch64::MADDXrrr;
4838       RC = &AArch64::GPR64RegClass;
4839     }
4840     Register NewVR = MRI.createVirtualRegister(OrrRC);
4841     uint64_t Imm = Root.getOperand(2).getImm();
4842     if (Root.getOperand(3).isImm()) {
4843       unsigned Val = Root.getOperand(3).getImm();
4844       Imm = Imm << Val;
4845     }
4846     uint64_t UImm = SignExtend64(-Imm, BitSize);
4847     uint64_t Encoding;
4848     if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) {
4849       MachineInstrBuilder MIB1 =
4850           BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR)
4851               .addReg(ZeroReg)
4852               .addImm(Encoding);
4853       InsInstrs.push_back(MIB1);
4854       InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
4855       MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
4856     }
4857     break;
4858   }
4859 
4860   case MachineCombinerPattern::MULADDv8i8_OP1:
4861     Opc = AArch64::MLAv8i8;
4862     RC = &AArch64::FPR64RegClass;
4863     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4864     break;
4865   case MachineCombinerPattern::MULADDv8i8_OP2:
4866     Opc = AArch64::MLAv8i8;
4867     RC = &AArch64::FPR64RegClass;
4868     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4869     break;
4870   case MachineCombinerPattern::MULADDv16i8_OP1:
4871     Opc = AArch64::MLAv16i8;
4872     RC = &AArch64::FPR128RegClass;
4873     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4874     break;
4875   case MachineCombinerPattern::MULADDv16i8_OP2:
4876     Opc = AArch64::MLAv16i8;
4877     RC = &AArch64::FPR128RegClass;
4878     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4879     break;
4880   case MachineCombinerPattern::MULADDv4i16_OP1:
4881     Opc = AArch64::MLAv4i16;
4882     RC = &AArch64::FPR64RegClass;
4883     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4884     break;
4885   case MachineCombinerPattern::MULADDv4i16_OP2:
4886     Opc = AArch64::MLAv4i16;
4887     RC = &AArch64::FPR64RegClass;
4888     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4889     break;
4890   case MachineCombinerPattern::MULADDv8i16_OP1:
4891     Opc = AArch64::MLAv8i16;
4892     RC = &AArch64::FPR128RegClass;
4893     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4894     break;
4895   case MachineCombinerPattern::MULADDv8i16_OP2:
4896     Opc = AArch64::MLAv8i16;
4897     RC = &AArch64::FPR128RegClass;
4898     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4899     break;
4900   case MachineCombinerPattern::MULADDv2i32_OP1:
4901     Opc = AArch64::MLAv2i32;
4902     RC = &AArch64::FPR64RegClass;
4903     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4904     break;
4905   case MachineCombinerPattern::MULADDv2i32_OP2:
4906     Opc = AArch64::MLAv2i32;
4907     RC = &AArch64::FPR64RegClass;
4908     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4909     break;
4910   case MachineCombinerPattern::MULADDv4i32_OP1:
4911     Opc = AArch64::MLAv4i32;
4912     RC = &AArch64::FPR128RegClass;
4913     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4914     break;
4915   case MachineCombinerPattern::MULADDv4i32_OP2:
4916     Opc = AArch64::MLAv4i32;
4917     RC = &AArch64::FPR128RegClass;
4918     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4919     break;
4920 
4921   case MachineCombinerPattern::MULSUBv8i8_OP1:
4922     Opc = AArch64::MLAv8i8;
4923     RC = &AArch64::FPR64RegClass;
4924     MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
4925                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv8i8,
4926                                  RC);
4927     break;
4928   case MachineCombinerPattern::MULSUBv8i8_OP2:
4929     Opc = AArch64::MLSv8i8;
4930     RC = &AArch64::FPR64RegClass;
4931     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4932     break;
4933   case MachineCombinerPattern::MULSUBv16i8_OP1:
4934     Opc = AArch64::MLAv16i8;
4935     RC = &AArch64::FPR128RegClass;
4936     MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
4937                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv16i8,
4938                                  RC);
4939     break;
4940   case MachineCombinerPattern::MULSUBv16i8_OP2:
4941     Opc = AArch64::MLSv16i8;
4942     RC = &AArch64::FPR128RegClass;
4943     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4944     break;
4945   case MachineCombinerPattern::MULSUBv4i16_OP1:
4946     Opc = AArch64::MLAv4i16;
4947     RC = &AArch64::FPR64RegClass;
4948     MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
4949                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i16,
4950                                  RC);
4951     break;
4952   case MachineCombinerPattern::MULSUBv4i16_OP2:
4953     Opc = AArch64::MLSv4i16;
4954     RC = &AArch64::FPR64RegClass;
4955     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4956     break;
4957   case MachineCombinerPattern::MULSUBv8i16_OP1:
4958     Opc = AArch64::MLAv8i16;
4959     RC = &AArch64::FPR128RegClass;
4960     MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
4961                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv8i16,
4962                                  RC);
4963     break;
4964   case MachineCombinerPattern::MULSUBv8i16_OP2:
4965     Opc = AArch64::MLSv8i16;
4966     RC = &AArch64::FPR128RegClass;
4967     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4968     break;
4969   case MachineCombinerPattern::MULSUBv2i32_OP1:
4970     Opc = AArch64::MLAv2i32;
4971     RC = &AArch64::FPR64RegClass;
4972     MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
4973                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv2i32,
4974                                  RC);
4975     break;
4976   case MachineCombinerPattern::MULSUBv2i32_OP2:
4977     Opc = AArch64::MLSv2i32;
4978     RC = &AArch64::FPR64RegClass;
4979     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4980     break;
4981   case MachineCombinerPattern::MULSUBv4i32_OP1:
4982     Opc = AArch64::MLAv4i32;
4983     RC = &AArch64::FPR128RegClass;
4984     MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
4985                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i32,
4986                                  RC);
4987     break;
4988   case MachineCombinerPattern::MULSUBv4i32_OP2:
4989     Opc = AArch64::MLSv4i32;
4990     RC = &AArch64::FPR128RegClass;
4991     MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
4992     break;
4993 
4994   case MachineCombinerPattern::MULADDv4i16_indexed_OP1:
4995     Opc = AArch64::MLAv4i16_indexed;
4996     RC = &AArch64::FPR64RegClass;
4997     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
4998     break;
4999   case MachineCombinerPattern::MULADDv4i16_indexed_OP2:
5000     Opc = AArch64::MLAv4i16_indexed;
5001     RC = &AArch64::FPR64RegClass;
5002     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5003     break;
5004   case MachineCombinerPattern::MULADDv8i16_indexed_OP1:
5005     Opc = AArch64::MLAv8i16_indexed;
5006     RC = &AArch64::FPR128RegClass;
5007     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5008     break;
5009   case MachineCombinerPattern::MULADDv8i16_indexed_OP2:
5010     Opc = AArch64::MLAv8i16_indexed;
5011     RC = &AArch64::FPR128RegClass;
5012     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5013     break;
5014   case MachineCombinerPattern::MULADDv2i32_indexed_OP1:
5015     Opc = AArch64::MLAv2i32_indexed;
5016     RC = &AArch64::FPR64RegClass;
5017     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5018     break;
5019   case MachineCombinerPattern::MULADDv2i32_indexed_OP2:
5020     Opc = AArch64::MLAv2i32_indexed;
5021     RC = &AArch64::FPR64RegClass;
5022     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5023     break;
5024   case MachineCombinerPattern::MULADDv4i32_indexed_OP1:
5025     Opc = AArch64::MLAv4i32_indexed;
5026     RC = &AArch64::FPR128RegClass;
5027     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5028     break;
5029   case MachineCombinerPattern::MULADDv4i32_indexed_OP2:
5030     Opc = AArch64::MLAv4i32_indexed;
5031     RC = &AArch64::FPR128RegClass;
5032     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5033     break;
5034 
5035   case MachineCombinerPattern::MULSUBv4i16_indexed_OP1:
5036     Opc = AArch64::MLAv4i16_indexed;
5037     RC = &AArch64::FPR64RegClass;
5038     MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
5039                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i16,
5040                                  RC);
5041     break;
5042   case MachineCombinerPattern::MULSUBv4i16_indexed_OP2:
5043     Opc = AArch64::MLSv4i16_indexed;
5044     RC = &AArch64::FPR64RegClass;
5045     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5046     break;
5047   case MachineCombinerPattern::MULSUBv8i16_indexed_OP1:
5048     Opc = AArch64::MLAv8i16_indexed;
5049     RC = &AArch64::FPR128RegClass;
5050     MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
5051                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv8i16,
5052                                  RC);
5053     break;
5054   case MachineCombinerPattern::MULSUBv8i16_indexed_OP2:
5055     Opc = AArch64::MLSv8i16_indexed;
5056     RC = &AArch64::FPR128RegClass;
5057     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5058     break;
5059   case MachineCombinerPattern::MULSUBv2i32_indexed_OP1:
5060     Opc = AArch64::MLAv2i32_indexed;
5061     RC = &AArch64::FPR64RegClass;
5062     MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
5063                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv2i32,
5064                                  RC);
5065     break;
5066   case MachineCombinerPattern::MULSUBv2i32_indexed_OP2:
5067     Opc = AArch64::MLSv2i32_indexed;
5068     RC = &AArch64::FPR64RegClass;
5069     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5070     break;
5071   case MachineCombinerPattern::MULSUBv4i32_indexed_OP1:
5072     Opc = AArch64::MLAv4i32_indexed;
5073     RC = &AArch64::FPR128RegClass;
5074     MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
5075                                  InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i32,
5076                                  RC);
5077     break;
5078   case MachineCombinerPattern::MULSUBv4i32_indexed_OP2:
5079     Opc = AArch64::MLSv4i32_indexed;
5080     RC = &AArch64::FPR128RegClass;
5081     MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5082     break;
5083 
5084   // Floating Point Support
5085   case MachineCombinerPattern::FMULADDH_OP1:
5086     Opc = AArch64::FMADDHrrr;
5087     RC = &AArch64::FPR16RegClass;
5088     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5089     break;
5090   case MachineCombinerPattern::FMULADDS_OP1:
5091     Opc = AArch64::FMADDSrrr;
5092     RC = &AArch64::FPR32RegClass;
5093     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5094     break;
5095   case MachineCombinerPattern::FMULADDD_OP1:
5096     Opc = AArch64::FMADDDrrr;
5097     RC = &AArch64::FPR64RegClass;
5098     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5099     break;
5100 
5101   case MachineCombinerPattern::FMULADDH_OP2:
5102     Opc = AArch64::FMADDHrrr;
5103     RC = &AArch64::FPR16RegClass;
5104     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5105     break;
5106   case MachineCombinerPattern::FMULADDS_OP2:
5107     Opc = AArch64::FMADDSrrr;
5108     RC = &AArch64::FPR32RegClass;
5109     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5110     break;
5111   case MachineCombinerPattern::FMULADDD_OP2:
5112     Opc = AArch64::FMADDDrrr;
5113     RC = &AArch64::FPR64RegClass;
5114     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5115     break;
5116 
5117   case MachineCombinerPattern::FMLAv1i32_indexed_OP1:
5118     Opc = AArch64::FMLAv1i32_indexed;
5119     RC = &AArch64::FPR32RegClass;
5120     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5121                            FMAInstKind::Indexed);
5122     break;
5123   case MachineCombinerPattern::FMLAv1i32_indexed_OP2:
5124     Opc = AArch64::FMLAv1i32_indexed;
5125     RC = &AArch64::FPR32RegClass;
5126     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5127                            FMAInstKind::Indexed);
5128     break;
5129 
5130   case MachineCombinerPattern::FMLAv1i64_indexed_OP1:
5131     Opc = AArch64::FMLAv1i64_indexed;
5132     RC = &AArch64::FPR64RegClass;
5133     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5134                            FMAInstKind::Indexed);
5135     break;
5136   case MachineCombinerPattern::FMLAv1i64_indexed_OP2:
5137     Opc = AArch64::FMLAv1i64_indexed;
5138     RC = &AArch64::FPR64RegClass;
5139     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5140                            FMAInstKind::Indexed);
5141     break;
5142 
5143   case MachineCombinerPattern::FMLAv4i16_indexed_OP1:
5144     RC = &AArch64::FPR64RegClass;
5145     Opc = AArch64::FMLAv4i16_indexed;
5146     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5147                            FMAInstKind::Indexed);
5148     break;
5149   case MachineCombinerPattern::FMLAv4f16_OP1:
5150     RC = &AArch64::FPR64RegClass;
5151     Opc = AArch64::FMLAv4f16;
5152     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5153                            FMAInstKind::Accumulator);
5154     break;
5155   case MachineCombinerPattern::FMLAv4i16_indexed_OP2:
5156     RC = &AArch64::FPR64RegClass;
5157     Opc = AArch64::FMLAv4i16_indexed;
5158     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5159                            FMAInstKind::Indexed);
5160     break;
5161   case MachineCombinerPattern::FMLAv4f16_OP2:
5162     RC = &AArch64::FPR64RegClass;
5163     Opc = AArch64::FMLAv4f16;
5164     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5165                            FMAInstKind::Accumulator);
5166     break;
5167 
5168   case MachineCombinerPattern::FMLAv2i32_indexed_OP1:
5169   case MachineCombinerPattern::FMLAv2f32_OP1:
5170     RC = &AArch64::FPR64RegClass;
5171     if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP1) {
5172       Opc = AArch64::FMLAv2i32_indexed;
5173       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5174                              FMAInstKind::Indexed);
5175     } else {
5176       Opc = AArch64::FMLAv2f32;
5177       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5178                              FMAInstKind::Accumulator);
5179     }
5180     break;
5181   case MachineCombinerPattern::FMLAv2i32_indexed_OP2:
5182   case MachineCombinerPattern::FMLAv2f32_OP2:
5183     RC = &AArch64::FPR64RegClass;
5184     if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP2) {
5185       Opc = AArch64::FMLAv2i32_indexed;
5186       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5187                              FMAInstKind::Indexed);
5188     } else {
5189       Opc = AArch64::FMLAv2f32;
5190       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5191                              FMAInstKind::Accumulator);
5192     }
5193     break;
5194 
5195   case MachineCombinerPattern::FMLAv8i16_indexed_OP1:
5196     RC = &AArch64::FPR128RegClass;
5197     Opc = AArch64::FMLAv8i16_indexed;
5198     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5199                            FMAInstKind::Indexed);
5200     break;
5201   case MachineCombinerPattern::FMLAv8f16_OP1:
5202     RC = &AArch64::FPR128RegClass;
5203     Opc = AArch64::FMLAv8f16;
5204     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5205                            FMAInstKind::Accumulator);
5206     break;
5207   case MachineCombinerPattern::FMLAv8i16_indexed_OP2:
5208     RC = &AArch64::FPR128RegClass;
5209     Opc = AArch64::FMLAv8i16_indexed;
5210     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5211                            FMAInstKind::Indexed);
5212     break;
5213   case MachineCombinerPattern::FMLAv8f16_OP2:
5214     RC = &AArch64::FPR128RegClass;
5215     Opc = AArch64::FMLAv8f16;
5216     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5217                            FMAInstKind::Accumulator);
5218     break;
5219 
5220   case MachineCombinerPattern::FMLAv2i64_indexed_OP1:
5221   case MachineCombinerPattern::FMLAv2f64_OP1:
5222     RC = &AArch64::FPR128RegClass;
5223     if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP1) {
5224       Opc = AArch64::FMLAv2i64_indexed;
5225       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5226                              FMAInstKind::Indexed);
5227     } else {
5228       Opc = AArch64::FMLAv2f64;
5229       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5230                              FMAInstKind::Accumulator);
5231     }
5232     break;
5233   case MachineCombinerPattern::FMLAv2i64_indexed_OP2:
5234   case MachineCombinerPattern::FMLAv2f64_OP2:
5235     RC = &AArch64::FPR128RegClass;
5236     if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP2) {
5237       Opc = AArch64::FMLAv2i64_indexed;
5238       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5239                              FMAInstKind::Indexed);
5240     } else {
5241       Opc = AArch64::FMLAv2f64;
5242       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5243                              FMAInstKind::Accumulator);
5244     }
5245     break;
5246 
5247   case MachineCombinerPattern::FMLAv4i32_indexed_OP1:
5248   case MachineCombinerPattern::FMLAv4f32_OP1:
5249     RC = &AArch64::FPR128RegClass;
5250     if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP1) {
5251       Opc = AArch64::FMLAv4i32_indexed;
5252       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5253                              FMAInstKind::Indexed);
5254     } else {
5255       Opc = AArch64::FMLAv4f32;
5256       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5257                              FMAInstKind::Accumulator);
5258     }
5259     break;
5260 
5261   case MachineCombinerPattern::FMLAv4i32_indexed_OP2:
5262   case MachineCombinerPattern::FMLAv4f32_OP2:
5263     RC = &AArch64::FPR128RegClass;
5264     if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP2) {
5265       Opc = AArch64::FMLAv4i32_indexed;
5266       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5267                              FMAInstKind::Indexed);
5268     } else {
5269       Opc = AArch64::FMLAv4f32;
5270       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5271                              FMAInstKind::Accumulator);
5272     }
5273     break;
5274 
5275   case MachineCombinerPattern::FMULSUBH_OP1:
5276     Opc = AArch64::FNMSUBHrrr;
5277     RC = &AArch64::FPR16RegClass;
5278     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5279     break;
5280   case MachineCombinerPattern::FMULSUBS_OP1:
5281     Opc = AArch64::FNMSUBSrrr;
5282     RC = &AArch64::FPR32RegClass;
5283     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5284     break;
5285   case MachineCombinerPattern::FMULSUBD_OP1:
5286     Opc = AArch64::FNMSUBDrrr;
5287     RC = &AArch64::FPR64RegClass;
5288     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5289     break;
5290 
5291   case MachineCombinerPattern::FNMULSUBH_OP1:
5292     Opc = AArch64::FNMADDHrrr;
5293     RC = &AArch64::FPR16RegClass;
5294     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5295     break;
5296   case MachineCombinerPattern::FNMULSUBS_OP1:
5297     Opc = AArch64::FNMADDSrrr;
5298     RC = &AArch64::FPR32RegClass;
5299     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5300     break;
5301   case MachineCombinerPattern::FNMULSUBD_OP1:
5302     Opc = AArch64::FNMADDDrrr;
5303     RC = &AArch64::FPR64RegClass;
5304     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
5305     break;
5306 
5307   case MachineCombinerPattern::FMULSUBH_OP2:
5308     Opc = AArch64::FMSUBHrrr;
5309     RC = &AArch64::FPR16RegClass;
5310     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5311     break;
5312   case MachineCombinerPattern::FMULSUBS_OP2:
5313     Opc = AArch64::FMSUBSrrr;
5314     RC = &AArch64::FPR32RegClass;
5315     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5316     break;
5317   case MachineCombinerPattern::FMULSUBD_OP2:
5318     Opc = AArch64::FMSUBDrrr;
5319     RC = &AArch64::FPR64RegClass;
5320     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
5321     break;
5322 
5323   case MachineCombinerPattern::FMLSv1i32_indexed_OP2:
5324     Opc = AArch64::FMLSv1i32_indexed;
5325     RC = &AArch64::FPR32RegClass;
5326     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5327                            FMAInstKind::Indexed);
5328     break;
5329 
5330   case MachineCombinerPattern::FMLSv1i64_indexed_OP2:
5331     Opc = AArch64::FMLSv1i64_indexed;
5332     RC = &AArch64::FPR64RegClass;
5333     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5334                            FMAInstKind::Indexed);
5335     break;
5336 
5337   case MachineCombinerPattern::FMLSv4f16_OP1:
5338   case MachineCombinerPattern::FMLSv4i16_indexed_OP1: {
5339     RC = &AArch64::FPR64RegClass;
5340     Register NewVR = MRI.createVirtualRegister(RC);
5341     MachineInstrBuilder MIB1 =
5342         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv4f16), NewVR)
5343             .add(Root.getOperand(2));
5344     InsInstrs.push_back(MIB1);
5345     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
5346     if (Pattern == MachineCombinerPattern::FMLSv4f16_OP1) {
5347       Opc = AArch64::FMLAv4f16;
5348       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5349                              FMAInstKind::Accumulator, &NewVR);
5350     } else {
5351       Opc = AArch64::FMLAv4i16_indexed;
5352       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5353                              FMAInstKind::Indexed, &NewVR);
5354     }
5355     break;
5356   }
5357   case MachineCombinerPattern::FMLSv4f16_OP2:
5358     RC = &AArch64::FPR64RegClass;
5359     Opc = AArch64::FMLSv4f16;
5360     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5361                            FMAInstKind::Accumulator);
5362     break;
5363   case MachineCombinerPattern::FMLSv4i16_indexed_OP2:
5364     RC = &AArch64::FPR64RegClass;
5365     Opc = AArch64::FMLSv4i16_indexed;
5366     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5367                            FMAInstKind::Indexed);
5368     break;
5369 
5370   case MachineCombinerPattern::FMLSv2f32_OP2:
5371   case MachineCombinerPattern::FMLSv2i32_indexed_OP2:
5372     RC = &AArch64::FPR64RegClass;
5373     if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP2) {
5374       Opc = AArch64::FMLSv2i32_indexed;
5375       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5376                              FMAInstKind::Indexed);
5377     } else {
5378       Opc = AArch64::FMLSv2f32;
5379       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5380                              FMAInstKind::Accumulator);
5381     }
5382     break;
5383 
5384   case MachineCombinerPattern::FMLSv8f16_OP1:
5385   case MachineCombinerPattern::FMLSv8i16_indexed_OP1: {
5386     RC = &AArch64::FPR128RegClass;
5387     Register NewVR = MRI.createVirtualRegister(RC);
5388     MachineInstrBuilder MIB1 =
5389         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv8f16), NewVR)
5390             .add(Root.getOperand(2));
5391     InsInstrs.push_back(MIB1);
5392     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
5393     if (Pattern == MachineCombinerPattern::FMLSv8f16_OP1) {
5394       Opc = AArch64::FMLAv8f16;
5395       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5396                              FMAInstKind::Accumulator, &NewVR);
5397     } else {
5398       Opc = AArch64::FMLAv8i16_indexed;
5399       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5400                              FMAInstKind::Indexed, &NewVR);
5401     }
5402     break;
5403   }
5404   case MachineCombinerPattern::FMLSv8f16_OP2:
5405     RC = &AArch64::FPR128RegClass;
5406     Opc = AArch64::FMLSv8f16;
5407     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5408                            FMAInstKind::Accumulator);
5409     break;
5410   case MachineCombinerPattern::FMLSv8i16_indexed_OP2:
5411     RC = &AArch64::FPR128RegClass;
5412     Opc = AArch64::FMLSv8i16_indexed;
5413     MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5414                            FMAInstKind::Indexed);
5415     break;
5416 
5417   case MachineCombinerPattern::FMLSv2f64_OP2:
5418   case MachineCombinerPattern::FMLSv2i64_indexed_OP2:
5419     RC = &AArch64::FPR128RegClass;
5420     if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP2) {
5421       Opc = AArch64::FMLSv2i64_indexed;
5422       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5423                              FMAInstKind::Indexed);
5424     } else {
5425       Opc = AArch64::FMLSv2f64;
5426       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5427                              FMAInstKind::Accumulator);
5428     }
5429     break;
5430 
5431   case MachineCombinerPattern::FMLSv4f32_OP2:
5432   case MachineCombinerPattern::FMLSv4i32_indexed_OP2:
5433     RC = &AArch64::FPR128RegClass;
5434     if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP2) {
5435       Opc = AArch64::FMLSv4i32_indexed;
5436       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5437                              FMAInstKind::Indexed);
5438     } else {
5439       Opc = AArch64::FMLSv4f32;
5440       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
5441                              FMAInstKind::Accumulator);
5442     }
5443     break;
5444   case MachineCombinerPattern::FMLSv2f32_OP1:
5445   case MachineCombinerPattern::FMLSv2i32_indexed_OP1: {
5446     RC = &AArch64::FPR64RegClass;
5447     Register NewVR = MRI.createVirtualRegister(RC);
5448     MachineInstrBuilder MIB1 =
5449         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f32), NewVR)
5450             .add(Root.getOperand(2));
5451     InsInstrs.push_back(MIB1);
5452     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
5453     if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP1) {
5454       Opc = AArch64::FMLAv2i32_indexed;
5455       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5456                              FMAInstKind::Indexed, &NewVR);
5457     } else {
5458       Opc = AArch64::FMLAv2f32;
5459       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5460                              FMAInstKind::Accumulator, &NewVR);
5461     }
5462     break;
5463   }
5464   case MachineCombinerPattern::FMLSv4f32_OP1:
5465   case MachineCombinerPattern::FMLSv4i32_indexed_OP1: {
5466     RC = &AArch64::FPR128RegClass;
5467     Register NewVR = MRI.createVirtualRegister(RC);
5468     MachineInstrBuilder MIB1 =
5469         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv4f32), NewVR)
5470             .add(Root.getOperand(2));
5471     InsInstrs.push_back(MIB1);
5472     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
5473     if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP1) {
5474       Opc = AArch64::FMLAv4i32_indexed;
5475       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5476                              FMAInstKind::Indexed, &NewVR);
5477     } else {
5478       Opc = AArch64::FMLAv4f32;
5479       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5480                              FMAInstKind::Accumulator, &NewVR);
5481     }
5482     break;
5483   }
5484   case MachineCombinerPattern::FMLSv2f64_OP1:
5485   case MachineCombinerPattern::FMLSv2i64_indexed_OP1: {
5486     RC = &AArch64::FPR128RegClass;
5487     Register NewVR = MRI.createVirtualRegister(RC);
5488     MachineInstrBuilder MIB1 =
5489         BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f64), NewVR)
5490             .add(Root.getOperand(2));
5491     InsInstrs.push_back(MIB1);
5492     InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
5493     if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP1) {
5494       Opc = AArch64::FMLAv2i64_indexed;
5495       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5496                              FMAInstKind::Indexed, &NewVR);
5497     } else {
5498       Opc = AArch64::FMLAv2f64;
5499       MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
5500                              FMAInstKind::Accumulator, &NewVR);
5501     }
5502     break;
5503   }
5504   } // end switch (Pattern)
5505   // Record MUL and ADD/SUB for deletion
5506   DelInstrs.push_back(MUL);
5507   DelInstrs.push_back(&Root);
5508 }
5509 
5510 /// Replace csincr-branch sequence by simple conditional branch
5511 ///
5512 /// Examples:
5513 /// 1. \code
5514 ///   csinc  w9, wzr, wzr, <condition code>
5515 ///   tbnz   w9, #0, 0x44
5516 ///    \endcode
5517 /// to
5518 ///    \code
5519 ///   b.<inverted condition code>
5520 ///    \endcode
5521 ///
5522 /// 2. \code
5523 ///   csinc w9, wzr, wzr, <condition code>
5524 ///   tbz   w9, #0, 0x44
5525 ///    \endcode
5526 /// to
5527 ///    \code
5528 ///   b.<condition code>
5529 ///    \endcode
5530 ///
5531 /// Replace compare and branch sequence by TBZ/TBNZ instruction when the
5532 /// compare's constant operand is power of 2.
5533 ///
5534 /// Examples:
5535 ///    \code
5536 ///   and  w8, w8, #0x400
5537 ///   cbnz w8, L1
5538 ///    \endcode
5539 /// to
5540 ///    \code
5541 ///   tbnz w8, #10, L1
5542 ///    \endcode
5543 ///
5544 /// \param  MI Conditional Branch
5545 /// \return True when the simple conditional branch is generated
5546 ///
5547 bool AArch64InstrInfo::optimizeCondBranch(MachineInstr &MI) const {
5548   bool IsNegativeBranch = false;
5549   bool IsTestAndBranch = false;
5550   unsigned TargetBBInMI = 0;
5551   switch (MI.getOpcode()) {
5552   default:
5553     llvm_unreachable("Unknown branch instruction?");
5554   case AArch64::Bcc:
5555     return false;
5556   case AArch64::CBZW:
5557   case AArch64::CBZX:
5558     TargetBBInMI = 1;
5559     break;
5560   case AArch64::CBNZW:
5561   case AArch64::CBNZX:
5562     TargetBBInMI = 1;
5563     IsNegativeBranch = true;
5564     break;
5565   case AArch64::TBZW:
5566   case AArch64::TBZX:
5567     TargetBBInMI = 2;
5568     IsTestAndBranch = true;
5569     break;
5570   case AArch64::TBNZW:
5571   case AArch64::TBNZX:
5572     TargetBBInMI = 2;
5573     IsNegativeBranch = true;
5574     IsTestAndBranch = true;
5575     break;
5576   }
5577   // So we increment a zero register and test for bits other
5578   // than bit 0? Conservatively bail out in case the verifier
5579   // missed this case.
5580   if (IsTestAndBranch && MI.getOperand(1).getImm())
5581     return false;
5582 
5583   // Find Definition.
5584   assert(MI.getParent() && "Incomplete machine instruciton\n");
5585   MachineBasicBlock *MBB = MI.getParent();
5586   MachineFunction *MF = MBB->getParent();
5587   MachineRegisterInfo *MRI = &MF->getRegInfo();
5588   Register VReg = MI.getOperand(0).getReg();
5589   if (!Register::isVirtualRegister(VReg))
5590     return false;
5591 
5592   MachineInstr *DefMI = MRI->getVRegDef(VReg);
5593 
5594   // Look through COPY instructions to find definition.
5595   while (DefMI->isCopy()) {
5596     Register CopyVReg = DefMI->getOperand(1).getReg();
5597     if (!MRI->hasOneNonDBGUse(CopyVReg))
5598       return false;
5599     if (!MRI->hasOneDef(CopyVReg))
5600       return false;
5601     DefMI = MRI->getVRegDef(CopyVReg);
5602   }
5603 
5604   switch (DefMI->getOpcode()) {
5605   default:
5606     return false;
5607   // Fold AND into a TBZ/TBNZ if constant operand is power of 2.
5608   case AArch64::ANDWri:
5609   case AArch64::ANDXri: {
5610     if (IsTestAndBranch)
5611       return false;
5612     if (DefMI->getParent() != MBB)
5613       return false;
5614     if (!MRI->hasOneNonDBGUse(VReg))
5615       return false;
5616 
5617     bool Is32Bit = (DefMI->getOpcode() == AArch64::ANDWri);
5618     uint64_t Mask = AArch64_AM::decodeLogicalImmediate(
5619         DefMI->getOperand(2).getImm(), Is32Bit ? 32 : 64);
5620     if (!isPowerOf2_64(Mask))
5621       return false;
5622 
5623     MachineOperand &MO = DefMI->getOperand(1);
5624     Register NewReg = MO.getReg();
5625     if (!Register::isVirtualRegister(NewReg))
5626       return false;
5627 
5628     assert(!MRI->def_empty(NewReg) && "Register must be defined.");
5629 
5630     MachineBasicBlock &RefToMBB = *MBB;
5631     MachineBasicBlock *TBB = MI.getOperand(1).getMBB();
5632     DebugLoc DL = MI.getDebugLoc();
5633     unsigned Imm = Log2_64(Mask);
5634     unsigned Opc = (Imm < 32)
5635                        ? (IsNegativeBranch ? AArch64::TBNZW : AArch64::TBZW)
5636                        : (IsNegativeBranch ? AArch64::TBNZX : AArch64::TBZX);
5637     MachineInstr *NewMI = BuildMI(RefToMBB, MI, DL, get(Opc))
5638                               .addReg(NewReg)
5639                               .addImm(Imm)
5640                               .addMBB(TBB);
5641     // Register lives on to the CBZ now.
5642     MO.setIsKill(false);
5643 
5644     // For immediate smaller than 32, we need to use the 32-bit
5645     // variant (W) in all cases. Indeed the 64-bit variant does not
5646     // allow to encode them.
5647     // Therefore, if the input register is 64-bit, we need to take the
5648     // 32-bit sub-part.
5649     if (!Is32Bit && Imm < 32)
5650       NewMI->getOperand(0).setSubReg(AArch64::sub_32);
5651     MI.eraseFromParent();
5652     return true;
5653   }
5654   // Look for CSINC
5655   case AArch64::CSINCWr:
5656   case AArch64::CSINCXr: {
5657     if (!(DefMI->getOperand(1).getReg() == AArch64::WZR &&
5658           DefMI->getOperand(2).getReg() == AArch64::WZR) &&
5659         !(DefMI->getOperand(1).getReg() == AArch64::XZR &&
5660           DefMI->getOperand(2).getReg() == AArch64::XZR))
5661       return false;
5662 
5663     if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) != -1)
5664       return false;
5665 
5666     AArch64CC::CondCode CC = (AArch64CC::CondCode)DefMI->getOperand(3).getImm();
5667     // Convert only when the condition code is not modified between
5668     // the CSINC and the branch. The CC may be used by other
5669     // instructions in between.
5670     if (areCFlagsAccessedBetweenInstrs(DefMI, MI, &getRegisterInfo(), AK_Write))
5671       return false;
5672     MachineBasicBlock &RefToMBB = *MBB;
5673     MachineBasicBlock *TBB = MI.getOperand(TargetBBInMI).getMBB();
5674     DebugLoc DL = MI.getDebugLoc();
5675     if (IsNegativeBranch)
5676       CC = AArch64CC::getInvertedCondCode(CC);
5677     BuildMI(RefToMBB, MI, DL, get(AArch64::Bcc)).addImm(CC).addMBB(TBB);
5678     MI.eraseFromParent();
5679     return true;
5680   }
5681   }
5682 }
5683 
5684 std::pair<unsigned, unsigned>
5685 AArch64InstrInfo::decomposeMachineOperandsTargetFlags(unsigned TF) const {
5686   const unsigned Mask = AArch64II::MO_FRAGMENT;
5687   return std::make_pair(TF & Mask, TF & ~Mask);
5688 }
5689 
5690 ArrayRef<std::pair<unsigned, const char *>>
5691 AArch64InstrInfo::getSerializableDirectMachineOperandTargetFlags() const {
5692   using namespace AArch64II;
5693 
5694   static const std::pair<unsigned, const char *> TargetFlags[] = {
5695       {MO_PAGE, "aarch64-page"}, {MO_PAGEOFF, "aarch64-pageoff"},
5696       {MO_G3, "aarch64-g3"},     {MO_G2, "aarch64-g2"},
5697       {MO_G1, "aarch64-g1"},     {MO_G0, "aarch64-g0"},
5698       {MO_HI12, "aarch64-hi12"}};
5699   return makeArrayRef(TargetFlags);
5700 }
5701 
5702 ArrayRef<std::pair<unsigned, const char *>>
5703 AArch64InstrInfo::getSerializableBitmaskMachineOperandTargetFlags() const {
5704   using namespace AArch64II;
5705 
5706   static const std::pair<unsigned, const char *> TargetFlags[] = {
5707       {MO_COFFSTUB, "aarch64-coffstub"},
5708       {MO_GOT, "aarch64-got"},
5709       {MO_NC, "aarch64-nc"},
5710       {MO_S, "aarch64-s"},
5711       {MO_TLS, "aarch64-tls"},
5712       {MO_DLLIMPORT, "aarch64-dllimport"},
5713       {MO_PREL, "aarch64-prel"},
5714       {MO_TAGGED, "aarch64-tagged"}};
5715   return makeArrayRef(TargetFlags);
5716 }
5717 
5718 ArrayRef<std::pair<MachineMemOperand::Flags, const char *>>
5719 AArch64InstrInfo::getSerializableMachineMemOperandTargetFlags() const {
5720   static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] =
5721       {{MOSuppressPair, "aarch64-suppress-pair"},
5722        {MOStridedAccess, "aarch64-strided-access"}};
5723   return makeArrayRef(TargetFlags);
5724 }
5725 
5726 /// Constants defining how certain sequences should be outlined.
5727 /// This encompasses how an outlined function should be called, and what kind of
5728 /// frame should be emitted for that outlined function.
5729 ///
5730 /// \p MachineOutlinerDefault implies that the function should be called with
5731 /// a save and restore of LR to the stack.
5732 ///
5733 /// That is,
5734 ///
5735 /// I1     Save LR                    OUTLINED_FUNCTION:
5736 /// I2 --> BL OUTLINED_FUNCTION       I1
5737 /// I3     Restore LR                 I2
5738 ///                                   I3
5739 ///                                   RET
5740 ///
5741 /// * Call construction overhead: 3 (save + BL + restore)
5742 /// * Frame construction overhead: 1 (ret)
5743 /// * Requires stack fixups? Yes
5744 ///
5745 /// \p MachineOutlinerTailCall implies that the function is being created from
5746 /// a sequence of instructions ending in a return.
5747 ///
5748 /// That is,
5749 ///
5750 /// I1                             OUTLINED_FUNCTION:
5751 /// I2 --> B OUTLINED_FUNCTION     I1
5752 /// RET                            I2
5753 ///                                RET
5754 ///
5755 /// * Call construction overhead: 1 (B)
5756 /// * Frame construction overhead: 0 (Return included in sequence)
5757 /// * Requires stack fixups? No
5758 ///
5759 /// \p MachineOutlinerNoLRSave implies that the function should be called using
5760 /// a BL instruction, but doesn't require LR to be saved and restored. This
5761 /// happens when LR is known to be dead.
5762 ///
5763 /// That is,
5764 ///
5765 /// I1                                OUTLINED_FUNCTION:
5766 /// I2 --> BL OUTLINED_FUNCTION       I1
5767 /// I3                                I2
5768 ///                                   I3
5769 ///                                   RET
5770 ///
5771 /// * Call construction overhead: 1 (BL)
5772 /// * Frame construction overhead: 1 (RET)
5773 /// * Requires stack fixups? No
5774 ///
5775 /// \p MachineOutlinerThunk implies that the function is being created from
5776 /// a sequence of instructions ending in a call. The outlined function is
5777 /// called with a BL instruction, and the outlined function tail-calls the
5778 /// original call destination.
5779 ///
5780 /// That is,
5781 ///
5782 /// I1                                OUTLINED_FUNCTION:
5783 /// I2 --> BL OUTLINED_FUNCTION       I1
5784 /// BL f                              I2
5785 ///                                   B f
5786 /// * Call construction overhead: 1 (BL)
5787 /// * Frame construction overhead: 0
5788 /// * Requires stack fixups? No
5789 ///
5790 /// \p MachineOutlinerRegSave implies that the function should be called with a
5791 /// save and restore of LR to an available register. This allows us to avoid
5792 /// stack fixups. Note that this outlining variant is compatible with the
5793 /// NoLRSave case.
5794 ///
5795 /// That is,
5796 ///
5797 /// I1     Save LR                    OUTLINED_FUNCTION:
5798 /// I2 --> BL OUTLINED_FUNCTION       I1
5799 /// I3     Restore LR                 I2
5800 ///                                   I3
5801 ///                                   RET
5802 ///
5803 /// * Call construction overhead: 3 (save + BL + restore)
5804 /// * Frame construction overhead: 1 (ret)
5805 /// * Requires stack fixups? No
5806 enum MachineOutlinerClass {
5807   MachineOutlinerDefault,  /// Emit a save, restore, call, and return.
5808   MachineOutlinerTailCall, /// Only emit a branch.
5809   MachineOutlinerNoLRSave, /// Emit a call and return.
5810   MachineOutlinerThunk,    /// Emit a call and tail-call.
5811   MachineOutlinerRegSave   /// Same as default, but save to a register.
5812 };
5813 
5814 enum MachineOutlinerMBBFlags {
5815   LRUnavailableSomewhere = 0x2,
5816   HasCalls = 0x4,
5817   UnsafeRegsDead = 0x8
5818 };
5819 
5820 unsigned
5821 AArch64InstrInfo::findRegisterToSaveLRTo(const outliner::Candidate &C) const {
5822   assert(C.LRUWasSet && "LRU wasn't set?");
5823   MachineFunction *MF = C.getMF();
5824   const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>(
5825       MF->getSubtarget().getRegisterInfo());
5826 
5827   // Check if there is an available register across the sequence that we can
5828   // use.
5829   for (unsigned Reg : AArch64::GPR64RegClass) {
5830     if (!ARI->isReservedReg(*MF, Reg) &&
5831         Reg != AArch64::LR &&  // LR is not reserved, but don't use it.
5832         Reg != AArch64::X16 && // X16 is not guaranteed to be preserved.
5833         Reg != AArch64::X17 && // Ditto for X17.
5834         C.LRU.available(Reg) && C.UsedInSequence.available(Reg))
5835       return Reg;
5836   }
5837 
5838   // No suitable register. Return 0.
5839   return 0u;
5840 }
5841 
5842 static bool
5843 outliningCandidatesSigningScopeConsensus(const outliner::Candidate &a,
5844                                          const outliner::Candidate &b) {
5845   const auto &MFIa = a.getMF()->getInfo<AArch64FunctionInfo>();
5846   const auto &MFIb = b.getMF()->getInfo<AArch64FunctionInfo>();
5847 
5848   return MFIa->shouldSignReturnAddress(false) == MFIb->shouldSignReturnAddress(false) &&
5849          MFIa->shouldSignReturnAddress(true) == MFIb->shouldSignReturnAddress(true);
5850 }
5851 
5852 static bool
5853 outliningCandidatesSigningKeyConsensus(const outliner::Candidate &a,
5854                                        const outliner::Candidate &b) {
5855   const auto &MFIa = a.getMF()->getInfo<AArch64FunctionInfo>();
5856   const auto &MFIb = b.getMF()->getInfo<AArch64FunctionInfo>();
5857 
5858   return MFIa->shouldSignWithBKey() == MFIb->shouldSignWithBKey();
5859 }
5860 
5861 static bool outliningCandidatesV8_3OpsConsensus(const outliner::Candidate &a,
5862                                                 const outliner::Candidate &b) {
5863   const AArch64Subtarget &SubtargetA =
5864       a.getMF()->getSubtarget<AArch64Subtarget>();
5865   const AArch64Subtarget &SubtargetB =
5866       b.getMF()->getSubtarget<AArch64Subtarget>();
5867   return SubtargetA.hasV8_3aOps() == SubtargetB.hasV8_3aOps();
5868 }
5869 
5870 outliner::OutlinedFunction AArch64InstrInfo::getOutliningCandidateInfo(
5871     std::vector<outliner::Candidate> &RepeatedSequenceLocs) const {
5872   outliner::Candidate &FirstCand = RepeatedSequenceLocs[0];
5873   unsigned SequenceSize =
5874       std::accumulate(FirstCand.front(), std::next(FirstCand.back()), 0,
5875                       [this](unsigned Sum, const MachineInstr &MI) {
5876                         return Sum + getInstSizeInBytes(MI);
5877                       });
5878   unsigned NumBytesToCreateFrame = 0;
5879 
5880   // We only allow outlining for functions having exactly matching return
5881   // address signing attributes, i.e., all share the same value for the
5882   // attribute "sign-return-address" and all share the same type of key they
5883   // are signed with.
5884   // Additionally we require all functions to simultaniously either support
5885   // v8.3a features or not. Otherwise an outlined function could get signed
5886   // using dedicated v8.3 instructions and a call from a function that doesn't
5887   // support v8.3 instructions would therefore be invalid.
5888   if (std::adjacent_find(
5889           RepeatedSequenceLocs.begin(), RepeatedSequenceLocs.end(),
5890           [](const outliner::Candidate &a, const outliner::Candidate &b) {
5891             // Return true if a and b are non-equal w.r.t. return address
5892             // signing or support of v8.3a features
5893             if (outliningCandidatesSigningScopeConsensus(a, b) &&
5894                 outliningCandidatesSigningKeyConsensus(a, b) &&
5895                 outliningCandidatesV8_3OpsConsensus(a, b)) {
5896               return false;
5897             }
5898             return true;
5899           }) != RepeatedSequenceLocs.end()) {
5900     return outliner::OutlinedFunction();
5901   }
5902 
5903   // Since at this point all candidates agree on their return address signing
5904   // picking just one is fine. If the candidate functions potentially sign their
5905   // return addresses, the outlined function should do the same. Note that in
5906   // the case of "sign-return-address"="non-leaf" this is an assumption: It is
5907   // not certainly true that the outlined function will have to sign its return
5908   // address but this decision is made later, when the decision to outline
5909   // has already been made.
5910   // The same holds for the number of additional instructions we need: On
5911   // v8.3a RET can be replaced by RETAA/RETAB and no AUT instruction is
5912   // necessary. However, at this point we don't know if the outlined function
5913   // will have a RET instruction so we assume the worst.
5914   const TargetRegisterInfo &TRI = getRegisterInfo();
5915   if (FirstCand.getMF()
5916           ->getInfo<AArch64FunctionInfo>()
5917           ->shouldSignReturnAddress(true)) {
5918     // One PAC and one AUT instructions
5919     NumBytesToCreateFrame += 8;
5920 
5921     // We have to check if sp modifying instructions would get outlined.
5922     // If so we only allow outlining if sp is unchanged overall, so matching
5923     // sub and add instructions are okay to outline, all other sp modifications
5924     // are not
5925     auto hasIllegalSPModification = [&TRI](outliner::Candidate &C) {
5926       int SPValue = 0;
5927       MachineBasicBlock::iterator MBBI = C.front();
5928       for (;;) {
5929         if (MBBI->modifiesRegister(AArch64::SP, &TRI)) {
5930           switch (MBBI->getOpcode()) {
5931           case AArch64::ADDXri:
5932           case AArch64::ADDWri:
5933             assert(MBBI->getNumOperands() == 4 && "Wrong number of operands");
5934             assert(MBBI->getOperand(2).isImm() &&
5935                    "Expected operand to be immediate");
5936             assert(MBBI->getOperand(1).isReg() &&
5937                    "Expected operand to be a register");
5938             // Check if the add just increments sp. If so, we search for
5939             // matching sub instructions that decrement sp. If not, the
5940             // modification is illegal
5941             if (MBBI->getOperand(1).getReg() == AArch64::SP)
5942               SPValue += MBBI->getOperand(2).getImm();
5943             else
5944               return true;
5945             break;
5946           case AArch64::SUBXri:
5947           case AArch64::SUBWri:
5948             assert(MBBI->getNumOperands() == 4 && "Wrong number of operands");
5949             assert(MBBI->getOperand(2).isImm() &&
5950                    "Expected operand to be immediate");
5951             assert(MBBI->getOperand(1).isReg() &&
5952                    "Expected operand to be a register");
5953             // Check if the sub just decrements sp. If so, we search for
5954             // matching add instructions that increment sp. If not, the
5955             // modification is illegal
5956             if (MBBI->getOperand(1).getReg() == AArch64::SP)
5957               SPValue -= MBBI->getOperand(2).getImm();
5958             else
5959               return true;
5960             break;
5961           default:
5962             return true;
5963           }
5964         }
5965         if (MBBI == C.back())
5966           break;
5967         ++MBBI;
5968       }
5969       if (SPValue)
5970         return true;
5971       return false;
5972     };
5973     // Remove candidates with illegal stack modifying instructions
5974     RepeatedSequenceLocs.erase(std::remove_if(RepeatedSequenceLocs.begin(),
5975                                               RepeatedSequenceLocs.end(),
5976                                               hasIllegalSPModification),
5977                                RepeatedSequenceLocs.end());
5978 
5979     // If the sequence doesn't have enough candidates left, then we're done.
5980     if (RepeatedSequenceLocs.size() < 2)
5981       return outliner::OutlinedFunction();
5982   }
5983 
5984   // Properties about candidate MBBs that hold for all of them.
5985   unsigned FlagsSetInAll = 0xF;
5986 
5987   // Compute liveness information for each candidate, and set FlagsSetInAll.
5988   std::for_each(RepeatedSequenceLocs.begin(), RepeatedSequenceLocs.end(),
5989                 [&FlagsSetInAll](outliner::Candidate &C) {
5990                   FlagsSetInAll &= C.Flags;
5991                 });
5992 
5993   // According to the AArch64 Procedure Call Standard, the following are
5994   // undefined on entry/exit from a function call:
5995   //
5996   // * Registers x16, x17, (and thus w16, w17)
5997   // * Condition codes (and thus the NZCV register)
5998   //
5999   // Because if this, we can't outline any sequence of instructions where
6000   // one
6001   // of these registers is live into/across it. Thus, we need to delete
6002   // those
6003   // candidates.
6004   auto CantGuaranteeValueAcrossCall = [&TRI](outliner::Candidate &C) {
6005     // If the unsafe registers in this block are all dead, then we don't need
6006     // to compute liveness here.
6007     if (C.Flags & UnsafeRegsDead)
6008       return false;
6009     C.initLRU(TRI);
6010     LiveRegUnits LRU = C.LRU;
6011     return (!LRU.available(AArch64::W16) || !LRU.available(AArch64::W17) ||
6012             !LRU.available(AArch64::NZCV));
6013   };
6014 
6015   // Are there any candidates where those registers are live?
6016   if (!(FlagsSetInAll & UnsafeRegsDead)) {
6017     // Erase every candidate that violates the restrictions above. (It could be
6018     // true that we have viable candidates, so it's not worth bailing out in
6019     // the case that, say, 1 out of 20 candidates violate the restructions.)
6020     RepeatedSequenceLocs.erase(std::remove_if(RepeatedSequenceLocs.begin(),
6021                                               RepeatedSequenceLocs.end(),
6022                                               CantGuaranteeValueAcrossCall),
6023                                RepeatedSequenceLocs.end());
6024 
6025     // If the sequence doesn't have enough candidates left, then we're done.
6026     if (RepeatedSequenceLocs.size() < 2)
6027       return outliner::OutlinedFunction();
6028   }
6029 
6030   // At this point, we have only "safe" candidates to outline. Figure out
6031   // frame + call instruction information.
6032 
6033   unsigned LastInstrOpcode = RepeatedSequenceLocs[0].back()->getOpcode();
6034 
6035   // Helper lambda which sets call information for every candidate.
6036   auto SetCandidateCallInfo =
6037       [&RepeatedSequenceLocs](unsigned CallID, unsigned NumBytesForCall) {
6038         for (outliner::Candidate &C : RepeatedSequenceLocs)
6039           C.setCallInfo(CallID, NumBytesForCall);
6040       };
6041 
6042   unsigned FrameID = MachineOutlinerDefault;
6043   NumBytesToCreateFrame += 4;
6044 
6045   bool HasBTI = any_of(RepeatedSequenceLocs, [](outliner::Candidate &C) {
6046     return C.getMF()->getInfo<AArch64FunctionInfo>()->branchTargetEnforcement();
6047   });
6048 
6049   // We check to see if CFI Instructions are present, and if they are
6050   // we find the number of CFI Instructions in the candidates.
6051   unsigned CFICount = 0;
6052   MachineBasicBlock::iterator MBBI = RepeatedSequenceLocs[0].front();
6053   for (unsigned Loc = RepeatedSequenceLocs[0].getStartIdx();
6054        Loc < RepeatedSequenceLocs[0].getEndIdx() + 1; Loc++) {
6055     const std::vector<MCCFIInstruction> &CFIInstructions =
6056         RepeatedSequenceLocs[0].getMF()->getFrameInstructions();
6057     if (MBBI->isCFIInstruction()) {
6058       unsigned CFIIndex = MBBI->getOperand(0).getCFIIndex();
6059       MCCFIInstruction CFI = CFIInstructions[CFIIndex];
6060       CFICount++;
6061     }
6062     MBBI++;
6063   }
6064 
6065   // We compare the number of found CFI Instructions to  the number of CFI
6066   // instructions in the parent function for each candidate.  We must check this
6067   // since if we outline one of the CFI instructions in a function, we have to
6068   // outline them all for correctness. If we do not, the address offsets will be
6069   // incorrect between the two sections of the program.
6070   for (outliner::Candidate &C : RepeatedSequenceLocs) {
6071     std::vector<MCCFIInstruction> CFIInstructions =
6072         C.getMF()->getFrameInstructions();
6073 
6074     if (CFICount > 0 && CFICount != CFIInstructions.size())
6075       return outliner::OutlinedFunction();
6076   }
6077 
6078   // Returns true if an instructions is safe to fix up, false otherwise.
6079   auto IsSafeToFixup = [this, &TRI](MachineInstr &MI) {
6080     if (MI.isCall())
6081       return true;
6082 
6083     if (!MI.modifiesRegister(AArch64::SP, &TRI) &&
6084         !MI.readsRegister(AArch64::SP, &TRI))
6085       return true;
6086 
6087     // Any modification of SP will break our code to save/restore LR.
6088     // FIXME: We could handle some instructions which add a constant
6089     // offset to SP, with a bit more work.
6090     if (MI.modifiesRegister(AArch64::SP, &TRI))
6091       return false;
6092 
6093     // At this point, we have a stack instruction that we might need to
6094     // fix up. We'll handle it if it's a load or store.
6095     if (MI.mayLoadOrStore()) {
6096       const MachineOperand *Base; // Filled with the base operand of MI.
6097       int64_t Offset;             // Filled with the offset of MI.
6098       bool OffsetIsScalable;
6099 
6100       // Does it allow us to offset the base operand and is the base the
6101       // register SP?
6102       if (!getMemOperandWithOffset(MI, Base, Offset, OffsetIsScalable, &TRI) ||
6103           !Base->isReg() || Base->getReg() != AArch64::SP)
6104         return false;
6105 
6106       // Fixe-up code below assumes bytes.
6107       if (OffsetIsScalable)
6108         return false;
6109 
6110       // Find the minimum/maximum offset for this instruction and check
6111       // if fixing it up would be in range.
6112       int64_t MinOffset,
6113           MaxOffset;  // Unscaled offsets for the instruction.
6114       TypeSize Scale(0U, false); // The scale to multiply the offsets by.
6115       unsigned DummyWidth;
6116       getMemOpInfo(MI.getOpcode(), Scale, DummyWidth, MinOffset, MaxOffset);
6117 
6118       Offset += 16; // Update the offset to what it would be if we outlined.
6119       if (Offset < MinOffset * (int64_t)Scale.getFixedSize() ||
6120           Offset > MaxOffset * (int64_t)Scale.getFixedSize())
6121         return false;
6122 
6123       // It's in range, so we can outline it.
6124       return true;
6125     }
6126 
6127     // FIXME: Add handling for instructions like "add x0, sp, #8".
6128 
6129     // We can't fix it up, so don't outline it.
6130     return false;
6131   };
6132 
6133   // True if it's possible to fix up each stack instruction in this sequence.
6134   // Important for frames/call variants that modify the stack.
6135   bool AllStackInstrsSafe = std::all_of(
6136       FirstCand.front(), std::next(FirstCand.back()), IsSafeToFixup);
6137 
6138   // If the last instruction in any candidate is a terminator, then we should
6139   // tail call all of the candidates.
6140   if (RepeatedSequenceLocs[0].back()->isTerminator()) {
6141     FrameID = MachineOutlinerTailCall;
6142     NumBytesToCreateFrame = 0;
6143     SetCandidateCallInfo(MachineOutlinerTailCall, 4);
6144   }
6145 
6146   else if (LastInstrOpcode == AArch64::BL ||
6147            ((LastInstrOpcode == AArch64::BLR ||
6148              LastInstrOpcode == AArch64::BLRNoIP) &&
6149             !HasBTI)) {
6150     // FIXME: Do we need to check if the code after this uses the value of LR?
6151     FrameID = MachineOutlinerThunk;
6152     NumBytesToCreateFrame = 0;
6153     SetCandidateCallInfo(MachineOutlinerThunk, 4);
6154   }
6155 
6156   else {
6157     // We need to decide how to emit calls + frames. We can always emit the same
6158     // frame if we don't need to save to the stack. If we have to save to the
6159     // stack, then we need a different frame.
6160     unsigned NumBytesNoStackCalls = 0;
6161     std::vector<outliner::Candidate> CandidatesWithoutStackFixups;
6162 
6163     // Check if we have to save LR.
6164     for (outliner::Candidate &C : RepeatedSequenceLocs) {
6165       C.initLRU(TRI);
6166 
6167       // If we have a noreturn caller, then we're going to be conservative and
6168       // say that we have to save LR. If we don't have a ret at the end of the
6169       // block, then we can't reason about liveness accurately.
6170       //
6171       // FIXME: We can probably do better than always disabling this in
6172       // noreturn functions by fixing up the liveness info.
6173       bool IsNoReturn =
6174           C.getMF()->getFunction().hasFnAttribute(Attribute::NoReturn);
6175 
6176       // Is LR available? If so, we don't need a save.
6177       if (C.LRU.available(AArch64::LR) && !IsNoReturn) {
6178         NumBytesNoStackCalls += 4;
6179         C.setCallInfo(MachineOutlinerNoLRSave, 4);
6180         CandidatesWithoutStackFixups.push_back(C);
6181       }
6182 
6183       // Is an unused register available? If so, we won't modify the stack, so
6184       // we can outline with the same frame type as those that don't save LR.
6185       else if (findRegisterToSaveLRTo(C)) {
6186         NumBytesNoStackCalls += 12;
6187         C.setCallInfo(MachineOutlinerRegSave, 12);
6188         CandidatesWithoutStackFixups.push_back(C);
6189       }
6190 
6191       // Is SP used in the sequence at all? If not, we don't have to modify
6192       // the stack, so we are guaranteed to get the same frame.
6193       else if (C.UsedInSequence.available(AArch64::SP)) {
6194         NumBytesNoStackCalls += 12;
6195         C.setCallInfo(MachineOutlinerDefault, 12);
6196         CandidatesWithoutStackFixups.push_back(C);
6197       }
6198 
6199       // If we outline this, we need to modify the stack. Pretend we don't
6200       // outline this by saving all of its bytes.
6201       else {
6202         NumBytesNoStackCalls += SequenceSize;
6203       }
6204     }
6205 
6206     // If there are no places where we have to save LR, then note that we
6207     // don't have to update the stack. Otherwise, give every candidate the
6208     // default call type, as long as it's safe to do so.
6209     if (!AllStackInstrsSafe ||
6210         NumBytesNoStackCalls <= RepeatedSequenceLocs.size() * 12) {
6211       RepeatedSequenceLocs = CandidatesWithoutStackFixups;
6212       FrameID = MachineOutlinerNoLRSave;
6213     } else {
6214       SetCandidateCallInfo(MachineOutlinerDefault, 12);
6215 
6216       // Bugzilla ID: 46767
6217       // TODO: Check if fixing up the stack more than once is safe so we can
6218       // outline these.
6219       //
6220       // An outline resulting in a caller that requires stack fixups at the
6221       // callsite to a callee that also requires stack fixups can happen when
6222       // there are no available registers at the candidate callsite for a
6223       // candidate that itself also has calls.
6224       //
6225       // In other words if function_containing_sequence in the following pseudo
6226       // assembly requires that we save LR at the point of the call, but there
6227       // are no available registers: in this case we save using SP and as a
6228       // result the SP offsets requires stack fixups by multiples of 16.
6229       //
6230       // function_containing_sequence:
6231       //   ...
6232       //   save LR to SP <- Requires stack instr fixups in OUTLINED_FUNCTION_N
6233       //   call OUTLINED_FUNCTION_N
6234       //   restore LR from SP
6235       //   ...
6236       //
6237       // OUTLINED_FUNCTION_N:
6238       //   save LR to SP <- Requires stack instr fixups in OUTLINED_FUNCTION_N
6239       //   ...
6240       //   bl foo
6241       //   restore LR from SP
6242       //   ret
6243       //
6244       // Because the code to handle more than one stack fixup does not
6245       // currently have the proper checks for legality, these cases will assert
6246       // in the AArch64 MachineOutliner. This is because the code to do this
6247       // needs more hardening, testing, better checks that generated code is
6248       // legal, etc and because it is only verified to handle a single pass of
6249       // stack fixup.
6250       //
6251       // The assert happens in AArch64InstrInfo::buildOutlinedFrame to catch
6252       // these cases until they are known to be handled. Bugzilla 46767 is
6253       // referenced in comments at the assert site.
6254       //
6255       // To avoid asserting (or generating non-legal code on noassert builds)
6256       // we remove all candidates which would need more than one stack fixup by
6257       // pruning the cases where the candidate has calls while also having no
6258       // available LR and having no available general purpose registers to copy
6259       // LR to (ie one extra stack save/restore).
6260       //
6261       if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) {
6262         erase_if(RepeatedSequenceLocs, [this](outliner::Candidate &C) {
6263           return (std::any_of(
6264                      C.front(), std::next(C.back()),
6265                      [](const MachineInstr &MI) { return MI.isCall(); })) &&
6266                  (!C.LRU.available(AArch64::LR) || !findRegisterToSaveLRTo(C));
6267         });
6268       }
6269     }
6270 
6271     // If we dropped all of the candidates, bail out here.
6272     if (RepeatedSequenceLocs.size() < 2) {
6273       RepeatedSequenceLocs.clear();
6274       return outliner::OutlinedFunction();
6275     }
6276   }
6277 
6278   // Does every candidate's MBB contain a call? If so, then we might have a call
6279   // in the range.
6280   if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) {
6281     // Check if the range contains a call. These require a save + restore of the
6282     // link register.
6283     bool ModStackToSaveLR = false;
6284     if (std::any_of(FirstCand.front(), FirstCand.back(),
6285                     [](const MachineInstr &MI) { return MI.isCall(); }))
6286       ModStackToSaveLR = true;
6287 
6288     // Handle the last instruction separately. If this is a tail call, then the
6289     // last instruction is a call. We don't want to save + restore in this case.
6290     // However, it could be possible that the last instruction is a call without
6291     // it being valid to tail call this sequence. We should consider this as
6292     // well.
6293     else if (FrameID != MachineOutlinerThunk &&
6294              FrameID != MachineOutlinerTailCall && FirstCand.back()->isCall())
6295       ModStackToSaveLR = true;
6296 
6297     if (ModStackToSaveLR) {
6298       // We can't fix up the stack. Bail out.
6299       if (!AllStackInstrsSafe) {
6300         RepeatedSequenceLocs.clear();
6301         return outliner::OutlinedFunction();
6302       }
6303 
6304       // Save + restore LR.
6305       NumBytesToCreateFrame += 8;
6306     }
6307   }
6308 
6309   // If we have CFI instructions, we can only outline if the outlined section
6310   // can be a tail call
6311   if (FrameID != MachineOutlinerTailCall && CFICount > 0)
6312     return outliner::OutlinedFunction();
6313 
6314   return outliner::OutlinedFunction(RepeatedSequenceLocs, SequenceSize,
6315                                     NumBytesToCreateFrame, FrameID);
6316 }
6317 
6318 bool AArch64InstrInfo::isFunctionSafeToOutlineFrom(
6319     MachineFunction &MF, bool OutlineFromLinkOnceODRs) const {
6320   const Function &F = MF.getFunction();
6321 
6322   // Can F be deduplicated by the linker? If it can, don't outline from it.
6323   if (!OutlineFromLinkOnceODRs && F.hasLinkOnceODRLinkage())
6324     return false;
6325 
6326   // Don't outline from functions with section markings; the program could
6327   // expect that all the code is in the named section.
6328   // FIXME: Allow outlining from multiple functions with the same section
6329   // marking.
6330   if (F.hasSection())
6331     return false;
6332 
6333   // Outlining from functions with redzones is unsafe since the outliner may
6334   // modify the stack. Check if hasRedZone is true or unknown; if yes, don't
6335   // outline from it.
6336   AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>();
6337   if (!AFI || AFI->hasRedZone().getValueOr(true))
6338     return false;
6339 
6340   // FIXME: Teach the outliner to generate/handle Windows unwind info.
6341   if (MF.getTarget().getMCAsmInfo()->usesWindowsCFI())
6342     return false;
6343 
6344   // It's safe to outline from MF.
6345   return true;
6346 }
6347 
6348 bool AArch64InstrInfo::isMBBSafeToOutlineFrom(MachineBasicBlock &MBB,
6349                                               unsigned &Flags) const {
6350   // Check if LR is available through all of the MBB. If it's not, then set
6351   // a flag.
6352   assert(MBB.getParent()->getRegInfo().tracksLiveness() &&
6353          "Suitable Machine Function for outlining must track liveness");
6354   LiveRegUnits LRU(getRegisterInfo());
6355 
6356   std::for_each(MBB.rbegin(), MBB.rend(),
6357                 [&LRU](MachineInstr &MI) { LRU.accumulate(MI); });
6358 
6359   // Check if each of the unsafe registers are available...
6360   bool W16AvailableInBlock = LRU.available(AArch64::W16);
6361   bool W17AvailableInBlock = LRU.available(AArch64::W17);
6362   bool NZCVAvailableInBlock = LRU.available(AArch64::NZCV);
6363 
6364   // If all of these are dead (and not live out), we know we don't have to check
6365   // them later.
6366   if (W16AvailableInBlock && W17AvailableInBlock && NZCVAvailableInBlock)
6367     Flags |= MachineOutlinerMBBFlags::UnsafeRegsDead;
6368 
6369   // Now, add the live outs to the set.
6370   LRU.addLiveOuts(MBB);
6371 
6372   // If any of these registers is available in the MBB, but also a live out of
6373   // the block, then we know outlining is unsafe.
6374   if (W16AvailableInBlock && !LRU.available(AArch64::W16))
6375     return false;
6376   if (W17AvailableInBlock && !LRU.available(AArch64::W17))
6377     return false;
6378   if (NZCVAvailableInBlock && !LRU.available(AArch64::NZCV))
6379     return false;
6380 
6381   // Check if there's a call inside this MachineBasicBlock. If there is, then
6382   // set a flag.
6383   if (any_of(MBB, [](MachineInstr &MI) { return MI.isCall(); }))
6384     Flags |= MachineOutlinerMBBFlags::HasCalls;
6385 
6386   MachineFunction *MF = MBB.getParent();
6387 
6388   // In the event that we outline, we may have to save LR. If there is an
6389   // available register in the MBB, then we'll always save LR there. Check if
6390   // this is true.
6391   bool CanSaveLR = false;
6392   const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>(
6393       MF->getSubtarget().getRegisterInfo());
6394 
6395   // Check if there is an available register across the sequence that we can
6396   // use.
6397   for (unsigned Reg : AArch64::GPR64RegClass) {
6398     if (!ARI->isReservedReg(*MF, Reg) && Reg != AArch64::LR &&
6399         Reg != AArch64::X16 && Reg != AArch64::X17 && LRU.available(Reg)) {
6400       CanSaveLR = true;
6401       break;
6402     }
6403   }
6404 
6405   // Check if we have a register we can save LR to, and if LR was used
6406   // somewhere. If both of those things are true, then we need to evaluate the
6407   // safety of outlining stack instructions later.
6408   if (!CanSaveLR && !LRU.available(AArch64::LR))
6409     Flags |= MachineOutlinerMBBFlags::LRUnavailableSomewhere;
6410 
6411   return true;
6412 }
6413 
6414 outliner::InstrType
6415 AArch64InstrInfo::getOutliningType(MachineBasicBlock::iterator &MIT,
6416                                    unsigned Flags) const {
6417   MachineInstr &MI = *MIT;
6418   MachineBasicBlock *MBB = MI.getParent();
6419   MachineFunction *MF = MBB->getParent();
6420   AArch64FunctionInfo *FuncInfo = MF->getInfo<AArch64FunctionInfo>();
6421 
6422   // Don't outline anything used for return address signing. The outlined
6423   // function will get signed later if needed
6424   switch (MI.getOpcode()) {
6425   case AArch64::PACIASP:
6426   case AArch64::PACIBSP:
6427   case AArch64::AUTIASP:
6428   case AArch64::AUTIBSP:
6429   case AArch64::RETAA:
6430   case AArch64::RETAB:
6431   case AArch64::EMITBKEY:
6432     return outliner::InstrType::Illegal;
6433   }
6434 
6435   // Don't outline LOHs.
6436   if (FuncInfo->getLOHRelated().count(&MI))
6437     return outliner::InstrType::Illegal;
6438 
6439   // We can only outline these if we will tail call the outlined function, or
6440   // fix up the CFI offsets. Currently, CFI instructions are outlined only if
6441   // in a tail call.
6442   //
6443   // FIXME: If the proper fixups for the offset are implemented, this should be
6444   // possible.
6445   if (MI.isCFIInstruction())
6446     return outliner::InstrType::Legal;
6447 
6448   // Don't allow debug values to impact outlining type.
6449   if (MI.isDebugInstr() || MI.isIndirectDebugValue())
6450     return outliner::InstrType::Invisible;
6451 
6452   // At this point, KILL instructions don't really tell us much so we can go
6453   // ahead and skip over them.
6454   if (MI.isKill())
6455     return outliner::InstrType::Invisible;
6456 
6457   // Is this a terminator for a basic block?
6458   if (MI.isTerminator()) {
6459 
6460     // Is this the end of a function?
6461     if (MI.getParent()->succ_empty())
6462       return outliner::InstrType::Legal;
6463 
6464     // It's not, so don't outline it.
6465     return outliner::InstrType::Illegal;
6466   }
6467 
6468   // Make sure none of the operands are un-outlinable.
6469   for (const MachineOperand &MOP : MI.operands()) {
6470     if (MOP.isCPI() || MOP.isJTI() || MOP.isCFIIndex() || MOP.isFI() ||
6471         MOP.isTargetIndex())
6472       return outliner::InstrType::Illegal;
6473 
6474     // If it uses LR or W30 explicitly, then don't touch it.
6475     if (MOP.isReg() && !MOP.isImplicit() &&
6476         (MOP.getReg() == AArch64::LR || MOP.getReg() == AArch64::W30))
6477       return outliner::InstrType::Illegal;
6478   }
6479 
6480   // Special cases for instructions that can always be outlined, but will fail
6481   // the later tests. e.g, ADRPs, which are PC-relative use LR, but can always
6482   // be outlined because they don't require a *specific* value to be in LR.
6483   if (MI.getOpcode() == AArch64::ADRP)
6484     return outliner::InstrType::Legal;
6485 
6486   // If MI is a call we might be able to outline it. We don't want to outline
6487   // any calls that rely on the position of items on the stack. When we outline
6488   // something containing a call, we have to emit a save and restore of LR in
6489   // the outlined function. Currently, this always happens by saving LR to the
6490   // stack. Thus, if we outline, say, half the parameters for a function call
6491   // plus the call, then we'll break the callee's expectations for the layout
6492   // of the stack.
6493   //
6494   // FIXME: Allow calls to functions which construct a stack frame, as long
6495   // as they don't access arguments on the stack.
6496   // FIXME: Figure out some way to analyze functions defined in other modules.
6497   // We should be able to compute the memory usage based on the IR calling
6498   // convention, even if we can't see the definition.
6499   if (MI.isCall()) {
6500     // Get the function associated with the call. Look at each operand and find
6501     // the one that represents the callee and get its name.
6502     const Function *Callee = nullptr;
6503     for (const MachineOperand &MOP : MI.operands()) {
6504       if (MOP.isGlobal()) {
6505         Callee = dyn_cast<Function>(MOP.getGlobal());
6506         break;
6507       }
6508     }
6509 
6510     // Never outline calls to mcount.  There isn't any rule that would require
6511     // this, but the Linux kernel's "ftrace" feature depends on it.
6512     if (Callee && Callee->getName() == "\01_mcount")
6513       return outliner::InstrType::Illegal;
6514 
6515     // If we don't know anything about the callee, assume it depends on the
6516     // stack layout of the caller. In that case, it's only legal to outline
6517     // as a tail-call. Explicitly list the call instructions we know about so we
6518     // don't get unexpected results with call pseudo-instructions.
6519     auto UnknownCallOutlineType = outliner::InstrType::Illegal;
6520     if (MI.getOpcode() == AArch64::BLR ||
6521         MI.getOpcode() == AArch64::BLRNoIP || MI.getOpcode() == AArch64::BL)
6522       UnknownCallOutlineType = outliner::InstrType::LegalTerminator;
6523 
6524     if (!Callee)
6525       return UnknownCallOutlineType;
6526 
6527     // We have a function we have information about. Check it if it's something
6528     // can safely outline.
6529     MachineFunction *CalleeMF = MF->getMMI().getMachineFunction(*Callee);
6530 
6531     // We don't know what's going on with the callee at all. Don't touch it.
6532     if (!CalleeMF)
6533       return UnknownCallOutlineType;
6534 
6535     // Check if we know anything about the callee saves on the function. If we
6536     // don't, then don't touch it, since that implies that we haven't
6537     // computed anything about its stack frame yet.
6538     MachineFrameInfo &MFI = CalleeMF->getFrameInfo();
6539     if (!MFI.isCalleeSavedInfoValid() || MFI.getStackSize() > 0 ||
6540         MFI.getNumObjects() > 0)
6541       return UnknownCallOutlineType;
6542 
6543     // At this point, we can say that CalleeMF ought to not pass anything on the
6544     // stack. Therefore, we can outline it.
6545     return outliner::InstrType::Legal;
6546   }
6547 
6548   // Don't outline positions.
6549   if (MI.isPosition())
6550     return outliner::InstrType::Illegal;
6551 
6552   // Don't touch the link register or W30.
6553   if (MI.readsRegister(AArch64::W30, &getRegisterInfo()) ||
6554       MI.modifiesRegister(AArch64::W30, &getRegisterInfo()))
6555     return outliner::InstrType::Illegal;
6556 
6557   // Don't outline BTI instructions, because that will prevent the outlining
6558   // site from being indirectly callable.
6559   if (MI.getOpcode() == AArch64::HINT) {
6560     int64_t Imm = MI.getOperand(0).getImm();
6561     if (Imm == 32 || Imm == 34 || Imm == 36 || Imm == 38)
6562       return outliner::InstrType::Illegal;
6563   }
6564 
6565   return outliner::InstrType::Legal;
6566 }
6567 
6568 void AArch64InstrInfo::fixupPostOutline(MachineBasicBlock &MBB) const {
6569   for (MachineInstr &MI : MBB) {
6570     const MachineOperand *Base;
6571     unsigned Width;
6572     int64_t Offset;
6573     bool OffsetIsScalable;
6574 
6575     // Is this a load or store with an immediate offset with SP as the base?
6576     if (!MI.mayLoadOrStore() ||
6577         !getMemOperandWithOffsetWidth(MI, Base, Offset, OffsetIsScalable, Width,
6578                                       &RI) ||
6579         (Base->isReg() && Base->getReg() != AArch64::SP))
6580       continue;
6581 
6582     // It is, so we have to fix it up.
6583     TypeSize Scale(0U, false);
6584     int64_t Dummy1, Dummy2;
6585 
6586     MachineOperand &StackOffsetOperand = getMemOpBaseRegImmOfsOffsetOperand(MI);
6587     assert(StackOffsetOperand.isImm() && "Stack offset wasn't immediate!");
6588     getMemOpInfo(MI.getOpcode(), Scale, Width, Dummy1, Dummy2);
6589     assert(Scale != 0 && "Unexpected opcode!");
6590     assert(!OffsetIsScalable && "Expected offset to be a byte offset");
6591 
6592     // We've pushed the return address to the stack, so add 16 to the offset.
6593     // This is safe, since we already checked if it would overflow when we
6594     // checked if this instruction was legal to outline.
6595     int64_t NewImm = (Offset + 16) / (int64_t)Scale.getFixedSize();
6596     StackOffsetOperand.setImm(NewImm);
6597   }
6598 }
6599 
6600 static void signOutlinedFunction(MachineFunction &MF, MachineBasicBlock &MBB,
6601                                  bool ShouldSignReturnAddr,
6602                                  bool ShouldSignReturnAddrWithAKey) {
6603   if (ShouldSignReturnAddr) {
6604     MachineBasicBlock::iterator MBBPAC = MBB.begin();
6605     MachineBasicBlock::iterator MBBAUT = MBB.getFirstTerminator();
6606     const AArch64Subtarget &Subtarget = MF.getSubtarget<AArch64Subtarget>();
6607     const TargetInstrInfo *TII = Subtarget.getInstrInfo();
6608     DebugLoc DL;
6609 
6610     if (MBBAUT != MBB.end())
6611       DL = MBBAUT->getDebugLoc();
6612 
6613     // At the very beginning of the basic block we insert the following
6614     // depending on the key type
6615     //
6616     // a_key:                   b_key:
6617     //    PACIASP                   EMITBKEY
6618     //    CFI_INSTRUCTION           PACIBSP
6619     //                              CFI_INSTRUCTION
6620     if (ShouldSignReturnAddrWithAKey) {
6621       BuildMI(MBB, MBBPAC, DebugLoc(), TII->get(AArch64::PACIASP))
6622           .setMIFlag(MachineInstr::FrameSetup);
6623     } else {
6624       BuildMI(MBB, MBBPAC, DebugLoc(), TII->get(AArch64::EMITBKEY))
6625           .setMIFlag(MachineInstr::FrameSetup);
6626       BuildMI(MBB, MBBPAC, DebugLoc(), TII->get(AArch64::PACIBSP))
6627           .setMIFlag(MachineInstr::FrameSetup);
6628     }
6629     unsigned CFIIndex =
6630         MF.addFrameInst(MCCFIInstruction::createNegateRAState(nullptr));
6631     BuildMI(MBB, MBBPAC, DebugLoc(), TII->get(AArch64::CFI_INSTRUCTION))
6632         .addCFIIndex(CFIIndex)
6633         .setMIFlags(MachineInstr::FrameSetup);
6634 
6635     // If v8.3a features are available we can replace a RET instruction by
6636     // RETAA or RETAB and omit the AUT instructions
6637     if (Subtarget.hasV8_3aOps() && MBBAUT != MBB.end() &&
6638         MBBAUT->getOpcode() == AArch64::RET) {
6639       BuildMI(MBB, MBBAUT, DL,
6640               TII->get(ShouldSignReturnAddrWithAKey ? AArch64::RETAA
6641                                                     : AArch64::RETAB))
6642           .copyImplicitOps(*MBBAUT);
6643       MBB.erase(MBBAUT);
6644     } else {
6645       BuildMI(MBB, MBBAUT, DL,
6646               TII->get(ShouldSignReturnAddrWithAKey ? AArch64::AUTIASP
6647                                                     : AArch64::AUTIBSP))
6648           .setMIFlag(MachineInstr::FrameDestroy);
6649     }
6650   }
6651 }
6652 
6653 void AArch64InstrInfo::buildOutlinedFrame(
6654     MachineBasicBlock &MBB, MachineFunction &MF,
6655     const outliner::OutlinedFunction &OF) const {
6656 
6657   AArch64FunctionInfo *FI = MF.getInfo<AArch64FunctionInfo>();
6658 
6659   if (OF.FrameConstructionID == MachineOutlinerTailCall)
6660     FI->setOutliningStyle("Tail Call");
6661   else if (OF.FrameConstructionID == MachineOutlinerThunk) {
6662     // For thunk outlining, rewrite the last instruction from a call to a
6663     // tail-call.
6664     MachineInstr *Call = &*--MBB.instr_end();
6665     unsigned TailOpcode;
6666     if (Call->getOpcode() == AArch64::BL) {
6667       TailOpcode = AArch64::TCRETURNdi;
6668     } else {
6669       assert(Call->getOpcode() == AArch64::BLR ||
6670              Call->getOpcode() == AArch64::BLRNoIP);
6671       TailOpcode = AArch64::TCRETURNriALL;
6672     }
6673     MachineInstr *TC = BuildMI(MF, DebugLoc(), get(TailOpcode))
6674                            .add(Call->getOperand(0))
6675                            .addImm(0);
6676     MBB.insert(MBB.end(), TC);
6677     Call->eraseFromParent();
6678 
6679     FI->setOutliningStyle("Thunk");
6680   }
6681 
6682   bool IsLeafFunction = true;
6683 
6684   // Is there a call in the outlined range?
6685   auto IsNonTailCall = [](const MachineInstr &MI) {
6686     return MI.isCall() && !MI.isReturn();
6687   };
6688 
6689   if (std::any_of(MBB.instr_begin(), MBB.instr_end(), IsNonTailCall)) {
6690     // Fix up the instructions in the range, since we're going to modify the
6691     // stack.
6692 
6693     // Bugzilla ID: 46767
6694     // TODO: Check if fixing up twice is safe so we can outline these.
6695     assert(OF.FrameConstructionID != MachineOutlinerDefault &&
6696            "Can only fix up stack references once");
6697     fixupPostOutline(MBB);
6698 
6699     IsLeafFunction = false;
6700 
6701     // LR has to be a live in so that we can save it.
6702     if (!MBB.isLiveIn(AArch64::LR))
6703       MBB.addLiveIn(AArch64::LR);
6704 
6705     MachineBasicBlock::iterator It = MBB.begin();
6706     MachineBasicBlock::iterator Et = MBB.end();
6707 
6708     if (OF.FrameConstructionID == MachineOutlinerTailCall ||
6709         OF.FrameConstructionID == MachineOutlinerThunk)
6710       Et = std::prev(MBB.end());
6711 
6712     // Insert a save before the outlined region
6713     MachineInstr *STRXpre = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre))
6714                                 .addReg(AArch64::SP, RegState::Define)
6715                                 .addReg(AArch64::LR)
6716                                 .addReg(AArch64::SP)
6717                                 .addImm(-16);
6718     It = MBB.insert(It, STRXpre);
6719 
6720     const TargetSubtargetInfo &STI = MF.getSubtarget();
6721     const MCRegisterInfo *MRI = STI.getRegisterInfo();
6722     unsigned DwarfReg = MRI->getDwarfRegNum(AArch64::LR, true);
6723 
6724     // Add a CFI saying the stack was moved 16 B down.
6725     int64_t StackPosEntry =
6726         MF.addFrameInst(MCCFIInstruction::cfiDefCfaOffset(nullptr, 16));
6727     BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION))
6728         .addCFIIndex(StackPosEntry)
6729         .setMIFlags(MachineInstr::FrameSetup);
6730 
6731     // Add a CFI saying that the LR that we want to find is now 16 B higher than
6732     // before.
6733     int64_t LRPosEntry =
6734         MF.addFrameInst(MCCFIInstruction::createOffset(nullptr, DwarfReg, -16));
6735     BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION))
6736         .addCFIIndex(LRPosEntry)
6737         .setMIFlags(MachineInstr::FrameSetup);
6738 
6739     // Insert a restore before the terminator for the function.
6740     MachineInstr *LDRXpost = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost))
6741                                  .addReg(AArch64::SP, RegState::Define)
6742                                  .addReg(AArch64::LR, RegState::Define)
6743                                  .addReg(AArch64::SP)
6744                                  .addImm(16);
6745     Et = MBB.insert(Et, LDRXpost);
6746   }
6747 
6748   // If a bunch of candidates reach this point they must agree on their return
6749   // address signing. It is therefore enough to just consider the signing
6750   // behaviour of one of them
6751   const auto &MFI = *OF.Candidates.front().getMF()->getInfo<AArch64FunctionInfo>();
6752   bool ShouldSignReturnAddr = MFI.shouldSignReturnAddress(!IsLeafFunction);
6753 
6754   // a_key is the default
6755   bool ShouldSignReturnAddrWithAKey = !MFI.shouldSignWithBKey();
6756 
6757   // If this is a tail call outlined function, then there's already a return.
6758   if (OF.FrameConstructionID == MachineOutlinerTailCall ||
6759       OF.FrameConstructionID == MachineOutlinerThunk) {
6760     signOutlinedFunction(MF, MBB, ShouldSignReturnAddr,
6761                          ShouldSignReturnAddrWithAKey);
6762     return;
6763   }
6764 
6765   // It's not a tail call, so we have to insert the return ourselves.
6766 
6767   // LR has to be a live in so that we can return to it.
6768   if (!MBB.isLiveIn(AArch64::LR))
6769     MBB.addLiveIn(AArch64::LR);
6770 
6771   MachineInstr *ret = BuildMI(MF, DebugLoc(), get(AArch64::RET))
6772                           .addReg(AArch64::LR);
6773   MBB.insert(MBB.end(), ret);
6774 
6775   signOutlinedFunction(MF, MBB, ShouldSignReturnAddr,
6776                        ShouldSignReturnAddrWithAKey);
6777 
6778   FI->setOutliningStyle("Function");
6779 
6780   // Did we have to modify the stack by saving the link register?
6781   if (OF.FrameConstructionID != MachineOutlinerDefault)
6782     return;
6783 
6784   // We modified the stack.
6785   // Walk over the basic block and fix up all the stack accesses.
6786   fixupPostOutline(MBB);
6787 }
6788 
6789 MachineBasicBlock::iterator AArch64InstrInfo::insertOutlinedCall(
6790     Module &M, MachineBasicBlock &MBB, MachineBasicBlock::iterator &It,
6791     MachineFunction &MF, const outliner::Candidate &C) const {
6792 
6793   // Are we tail calling?
6794   if (C.CallConstructionID == MachineOutlinerTailCall) {
6795     // If yes, then we can just branch to the label.
6796     It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::TCRETURNdi))
6797                             .addGlobalAddress(M.getNamedValue(MF.getName()))
6798                             .addImm(0));
6799     return It;
6800   }
6801 
6802   // Are we saving the link register?
6803   if (C.CallConstructionID == MachineOutlinerNoLRSave ||
6804       C.CallConstructionID == MachineOutlinerThunk) {
6805     // No, so just insert the call.
6806     It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL))
6807                             .addGlobalAddress(M.getNamedValue(MF.getName())));
6808     return It;
6809   }
6810 
6811   // We want to return the spot where we inserted the call.
6812   MachineBasicBlock::iterator CallPt;
6813 
6814   // Instructions for saving and restoring LR around the call instruction we're
6815   // going to insert.
6816   MachineInstr *Save;
6817   MachineInstr *Restore;
6818   // Can we save to a register?
6819   if (C.CallConstructionID == MachineOutlinerRegSave) {
6820     // FIXME: This logic should be sunk into a target-specific interface so that
6821     // we don't have to recompute the register.
6822     unsigned Reg = findRegisterToSaveLRTo(C);
6823     assert(Reg != 0 && "No callee-saved register available?");
6824 
6825     // Save and restore LR from that register.
6826     Save = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), Reg)
6827                .addReg(AArch64::XZR)
6828                .addReg(AArch64::LR)
6829                .addImm(0);
6830     Restore = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), AArch64::LR)
6831                 .addReg(AArch64::XZR)
6832                 .addReg(Reg)
6833                 .addImm(0);
6834   } else {
6835     // We have the default case. Save and restore from SP.
6836     Save = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre))
6837                .addReg(AArch64::SP, RegState::Define)
6838                .addReg(AArch64::LR)
6839                .addReg(AArch64::SP)
6840                .addImm(-16);
6841     Restore = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost))
6842                   .addReg(AArch64::SP, RegState::Define)
6843                   .addReg(AArch64::LR, RegState::Define)
6844                   .addReg(AArch64::SP)
6845                   .addImm(16);
6846   }
6847 
6848   It = MBB.insert(It, Save);
6849   It++;
6850 
6851   // Insert the call.
6852   It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL))
6853                           .addGlobalAddress(M.getNamedValue(MF.getName())));
6854   CallPt = It;
6855   It++;
6856 
6857   It = MBB.insert(It, Restore);
6858   return CallPt;
6859 }
6860 
6861 bool AArch64InstrInfo::shouldOutlineFromFunctionByDefault(
6862   MachineFunction &MF) const {
6863   return MF.getFunction().hasMinSize();
6864 }
6865 
6866 Optional<DestSourcePair>
6867 AArch64InstrInfo::isCopyInstrImpl(const MachineInstr &MI) const {
6868 
6869   // AArch64::ORRWrs and AArch64::ORRXrs with WZR/XZR reg
6870   // and zero immediate operands used as an alias for mov instruction.
6871   if (MI.getOpcode() == AArch64::ORRWrs &&
6872       MI.getOperand(1).getReg() == AArch64::WZR &&
6873       MI.getOperand(3).getImm() == 0x0) {
6874     return DestSourcePair{MI.getOperand(0), MI.getOperand(2)};
6875   }
6876 
6877   if (MI.getOpcode() == AArch64::ORRXrs &&
6878       MI.getOperand(1).getReg() == AArch64::XZR &&
6879       MI.getOperand(3).getImm() == 0x0) {
6880     return DestSourcePair{MI.getOperand(0), MI.getOperand(2)};
6881   }
6882 
6883   return None;
6884 }
6885 
6886 Optional<RegImmPair> AArch64InstrInfo::isAddImmediate(const MachineInstr &MI,
6887                                                       Register Reg) const {
6888   int Sign = 1;
6889   int64_t Offset = 0;
6890 
6891   // TODO: Handle cases where Reg is a super- or sub-register of the
6892   // destination register.
6893   const MachineOperand &Op0 = MI.getOperand(0);
6894   if (!Op0.isReg() || Reg != Op0.getReg())
6895     return None;
6896 
6897   switch (MI.getOpcode()) {
6898   default:
6899     return None;
6900   case AArch64::SUBWri:
6901   case AArch64::SUBXri:
6902   case AArch64::SUBSWri:
6903   case AArch64::SUBSXri:
6904     Sign *= -1;
6905     LLVM_FALLTHROUGH;
6906   case AArch64::ADDSWri:
6907   case AArch64::ADDSXri:
6908   case AArch64::ADDWri:
6909   case AArch64::ADDXri: {
6910     // TODO: Third operand can be global address (usually some string).
6911     if (!MI.getOperand(0).isReg() || !MI.getOperand(1).isReg() ||
6912         !MI.getOperand(2).isImm())
6913       return None;
6914     int Shift = MI.getOperand(3).getImm();
6915     assert((Shift == 0 || Shift == 12) && "Shift can be either 0 or 12");
6916     Offset = Sign * (MI.getOperand(2).getImm() << Shift);
6917   }
6918   }
6919   return RegImmPair{MI.getOperand(1).getReg(), Offset};
6920 }
6921 
6922 /// If the given ORR instruction is a copy, and \p DescribedReg overlaps with
6923 /// the destination register then, if possible, describe the value in terms of
6924 /// the source register.
6925 static Optional<ParamLoadedValue>
6926 describeORRLoadedValue(const MachineInstr &MI, Register DescribedReg,
6927                        const TargetInstrInfo *TII,
6928                        const TargetRegisterInfo *TRI) {
6929   auto DestSrc = TII->isCopyInstr(MI);
6930   if (!DestSrc)
6931     return None;
6932 
6933   Register DestReg = DestSrc->Destination->getReg();
6934   Register SrcReg = DestSrc->Source->getReg();
6935 
6936   auto Expr = DIExpression::get(MI.getMF()->getFunction().getContext(), {});
6937 
6938   // If the described register is the destination, just return the source.
6939   if (DestReg == DescribedReg)
6940     return ParamLoadedValue(MachineOperand::CreateReg(SrcReg, false), Expr);
6941 
6942   // ORRWrs zero-extends to 64-bits, so we need to consider such cases.
6943   if (MI.getOpcode() == AArch64::ORRWrs &&
6944       TRI->isSuperRegister(DestReg, DescribedReg))
6945     return ParamLoadedValue(MachineOperand::CreateReg(SrcReg, false), Expr);
6946 
6947   // We may need to describe the lower part of a ORRXrs move.
6948   if (MI.getOpcode() == AArch64::ORRXrs &&
6949       TRI->isSubRegister(DestReg, DescribedReg)) {
6950     Register SrcSubReg = TRI->getSubReg(SrcReg, AArch64::sub_32);
6951     return ParamLoadedValue(MachineOperand::CreateReg(SrcSubReg, false), Expr);
6952   }
6953 
6954   assert(!TRI->isSuperOrSubRegisterEq(DestReg, DescribedReg) &&
6955          "Unhandled ORR[XW]rs copy case");
6956 
6957   return None;
6958 }
6959 
6960 Optional<ParamLoadedValue>
6961 AArch64InstrInfo::describeLoadedValue(const MachineInstr &MI,
6962                                       Register Reg) const {
6963   const MachineFunction *MF = MI.getMF();
6964   const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo();
6965   switch (MI.getOpcode()) {
6966   case AArch64::MOVZWi:
6967   case AArch64::MOVZXi: {
6968     // MOVZWi may be used for producing zero-extended 32-bit immediates in
6969     // 64-bit parameters, so we need to consider super-registers.
6970     if (!TRI->isSuperRegisterEq(MI.getOperand(0).getReg(), Reg))
6971       return None;
6972 
6973     if (!MI.getOperand(1).isImm())
6974       return None;
6975     int64_t Immediate = MI.getOperand(1).getImm();
6976     int Shift = MI.getOperand(2).getImm();
6977     return ParamLoadedValue(MachineOperand::CreateImm(Immediate << Shift),
6978                             nullptr);
6979   }
6980   case AArch64::ORRWrs:
6981   case AArch64::ORRXrs:
6982     return describeORRLoadedValue(MI, Reg, this, TRI);
6983   }
6984 
6985   return TargetInstrInfo::describeLoadedValue(MI, Reg);
6986 }
6987 
6988 uint64_t AArch64InstrInfo::getElementSizeForOpcode(unsigned Opc) const {
6989   return get(Opc).TSFlags & AArch64::ElementSizeMask;
6990 }
6991 
6992 unsigned llvm::getBLRCallOpcode(const MachineFunction &MF) {
6993   if (MF.getSubtarget<AArch64Subtarget>().hardenSlsBlr())
6994     return AArch64::BLRNoIP;
6995   else
6996     return AArch64::BLR;
6997 }
6998 
6999 #define GET_INSTRINFO_HELPERS
7000 #define GET_INSTRMAP_INFO
7001 #include "AArch64GenInstrInfo.inc"
7002