1 //===-- ARMISelDAGToDAG.cpp - A dag to dag inst selector for ARM ----------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 // This file defines an instruction selector for the ARM target.
11 //
12 //===----------------------------------------------------------------------===//
13 
14 #include "ARM.h"
15 #include "ARMBaseInstrInfo.h"
16 #include "ARMTargetMachine.h"
17 #include "MCTargetDesc/ARMAddressingModes.h"
18 #include "llvm/ADT/StringSwitch.h"
19 #include "llvm/CodeGen/MachineFrameInfo.h"
20 #include "llvm/CodeGen/MachineFunction.h"
21 #include "llvm/CodeGen/MachineInstrBuilder.h"
22 #include "llvm/CodeGen/MachineRegisterInfo.h"
23 #include "llvm/CodeGen/SelectionDAG.h"
24 #include "llvm/CodeGen/SelectionDAGISel.h"
25 #include "llvm/IR/CallingConv.h"
26 #include "llvm/IR/Constants.h"
27 #include "llvm/IR/DerivedTypes.h"
28 #include "llvm/IR/Function.h"
29 #include "llvm/IR/Intrinsics.h"
30 #include "llvm/IR/LLVMContext.h"
31 #include "llvm/Support/CommandLine.h"
32 #include "llvm/Support/Debug.h"
33 #include "llvm/Support/ErrorHandling.h"
34 #include "llvm/Target/TargetLowering.h"
35 #include "llvm/Target/TargetOptions.h"
36 
37 using namespace llvm;
38 
39 #define DEBUG_TYPE "arm-isel"
40 
41 static cl::opt<bool>
42 DisableShifterOp("disable-shifter-op", cl::Hidden,
43   cl::desc("Disable isel of shifter-op"),
44   cl::init(false));
45 
46 static cl::opt<bool>
47 CheckVMLxHazard("check-vmlx-hazard", cl::Hidden,
48   cl::desc("Check fp vmla / vmls hazard at isel time"),
49   cl::init(true));
50 
51 //===--------------------------------------------------------------------===//
52 /// ARMDAGToDAGISel - ARM specific code to select ARM machine
53 /// instructions for SelectionDAG operations.
54 ///
55 namespace {
56 
57 enum AddrMode2Type {
58   AM2_BASE, // Simple AM2 (+-imm12)
59   AM2_SHOP  // Shifter-op AM2
60 };
61 
62 class ARMDAGToDAGISel : public SelectionDAGISel {
63   /// Subtarget - Keep a pointer to the ARMSubtarget around so that we can
64   /// make the right decision when generating code for different targets.
65   const ARMSubtarget *Subtarget;
66 
67 public:
68   explicit ARMDAGToDAGISel(ARMBaseTargetMachine &tm, CodeGenOpt::Level OptLevel)
69       : SelectionDAGISel(tm, OptLevel) {}
70 
71   bool runOnMachineFunction(MachineFunction &MF) override {
72     // Reset the subtarget each time through.
73     Subtarget = &MF.getSubtarget<ARMSubtarget>();
74     SelectionDAGISel::runOnMachineFunction(MF);
75     return true;
76   }
77 
78   const char *getPassName() const override {
79     return "ARM Instruction Selection";
80   }
81 
82   void PreprocessISelDAG() override;
83 
84   /// getI32Imm - Return a target constant of type i32 with the specified
85   /// value.
86   inline SDValue getI32Imm(unsigned Imm, SDLoc dl) {
87     return CurDAG->getTargetConstant(Imm, dl, MVT::i32);
88   }
89 
90   SDNode *Select(SDNode *N) override;
91 
92 
93   bool hasNoVMLxHazardUse(SDNode *N) const;
94   bool isShifterOpProfitable(const SDValue &Shift,
95                              ARM_AM::ShiftOpc ShOpcVal, unsigned ShAmt);
96   bool SelectRegShifterOperand(SDValue N, SDValue &A,
97                                SDValue &B, SDValue &C,
98                                bool CheckProfitability = true);
99   bool SelectImmShifterOperand(SDValue N, SDValue &A,
100                                SDValue &B, bool CheckProfitability = true);
101   bool SelectShiftRegShifterOperand(SDValue N, SDValue &A,
102                                     SDValue &B, SDValue &C) {
103     // Don't apply the profitability check
104     return SelectRegShifterOperand(N, A, B, C, false);
105   }
106   bool SelectShiftImmShifterOperand(SDValue N, SDValue &A,
107                                     SDValue &B) {
108     // Don't apply the profitability check
109     return SelectImmShifterOperand(N, A, B, false);
110   }
111 
112   bool SelectAddrModeImm12(SDValue N, SDValue &Base, SDValue &OffImm);
113   bool SelectLdStSOReg(SDValue N, SDValue &Base, SDValue &Offset, SDValue &Opc);
114 
115   AddrMode2Type SelectAddrMode2Worker(SDValue N, SDValue &Base,
116                                       SDValue &Offset, SDValue &Opc);
117   bool SelectAddrMode2Base(SDValue N, SDValue &Base, SDValue &Offset,
118                            SDValue &Opc) {
119     return SelectAddrMode2Worker(N, Base, Offset, Opc) == AM2_BASE;
120   }
121 
122   bool SelectAddrMode2ShOp(SDValue N, SDValue &Base, SDValue &Offset,
123                            SDValue &Opc) {
124     return SelectAddrMode2Worker(N, Base, Offset, Opc) == AM2_SHOP;
125   }
126 
127   bool SelectAddrMode2(SDValue N, SDValue &Base, SDValue &Offset,
128                        SDValue &Opc) {
129     SelectAddrMode2Worker(N, Base, Offset, Opc);
130 //    return SelectAddrMode2ShOp(N, Base, Offset, Opc);
131     // This always matches one way or another.
132     return true;
133   }
134 
135   bool SelectCMOVPred(SDValue N, SDValue &Pred, SDValue &Reg) {
136     const ConstantSDNode *CN = cast<ConstantSDNode>(N);
137     Pred = CurDAG->getTargetConstant(CN->getZExtValue(), SDLoc(N), MVT::i32);
138     Reg = CurDAG->getRegister(ARM::CPSR, MVT::i32);
139     return true;
140   }
141 
142   bool SelectAddrMode2OffsetReg(SDNode *Op, SDValue N,
143                              SDValue &Offset, SDValue &Opc);
144   bool SelectAddrMode2OffsetImm(SDNode *Op, SDValue N,
145                              SDValue &Offset, SDValue &Opc);
146   bool SelectAddrMode2OffsetImmPre(SDNode *Op, SDValue N,
147                              SDValue &Offset, SDValue &Opc);
148   bool SelectAddrOffsetNone(SDValue N, SDValue &Base);
149   bool SelectAddrMode3(SDValue N, SDValue &Base,
150                        SDValue &Offset, SDValue &Opc);
151   bool SelectAddrMode3Offset(SDNode *Op, SDValue N,
152                              SDValue &Offset, SDValue &Opc);
153   bool SelectAddrMode5(SDValue N, SDValue &Base,
154                        SDValue &Offset);
155   bool SelectAddrMode6(SDNode *Parent, SDValue N, SDValue &Addr,SDValue &Align);
156   bool SelectAddrMode6Offset(SDNode *Op, SDValue N, SDValue &Offset);
157 
158   bool SelectAddrModePC(SDValue N, SDValue &Offset, SDValue &Label);
159 
160   // Thumb Addressing Modes:
161   bool SelectThumbAddrModeRR(SDValue N, SDValue &Base, SDValue &Offset);
162   bool SelectThumbAddrModeImm5S(SDValue N, unsigned Scale, SDValue &Base,
163                                 SDValue &OffImm);
164   bool SelectThumbAddrModeImm5S1(SDValue N, SDValue &Base,
165                                  SDValue &OffImm);
166   bool SelectThumbAddrModeImm5S2(SDValue N, SDValue &Base,
167                                  SDValue &OffImm);
168   bool SelectThumbAddrModeImm5S4(SDValue N, SDValue &Base,
169                                  SDValue &OffImm);
170   bool SelectThumbAddrModeSP(SDValue N, SDValue &Base, SDValue &OffImm);
171 
172   // Thumb 2 Addressing Modes:
173   bool SelectT2AddrModeImm12(SDValue N, SDValue &Base, SDValue &OffImm);
174   bool SelectT2AddrModeImm8(SDValue N, SDValue &Base,
175                             SDValue &OffImm);
176   bool SelectT2AddrModeImm8Offset(SDNode *Op, SDValue N,
177                                  SDValue &OffImm);
178   bool SelectT2AddrModeSoReg(SDValue N, SDValue &Base,
179                              SDValue &OffReg, SDValue &ShImm);
180   bool SelectT2AddrModeExclusive(SDValue N, SDValue &Base, SDValue &OffImm);
181 
182   inline bool is_so_imm(unsigned Imm) const {
183     return ARM_AM::getSOImmVal(Imm) != -1;
184   }
185 
186   inline bool is_so_imm_not(unsigned Imm) const {
187     return ARM_AM::getSOImmVal(~Imm) != -1;
188   }
189 
190   inline bool is_t2_so_imm(unsigned Imm) const {
191     return ARM_AM::getT2SOImmVal(Imm) != -1;
192   }
193 
194   inline bool is_t2_so_imm_not(unsigned Imm) const {
195     return ARM_AM::getT2SOImmVal(~Imm) != -1;
196   }
197 
198   // Include the pieces autogenerated from the target description.
199 #include "ARMGenDAGISel.inc"
200 
201 private:
202   /// SelectARMIndexedLoad - Indexed (pre/post inc/dec) load matching code for
203   /// ARM.
204   SDNode *SelectARMIndexedLoad(SDNode *N);
205   SDNode *SelectT2IndexedLoad(SDNode *N);
206 
207   /// SelectVLD - Select NEON load intrinsics.  NumVecs should be
208   /// 1, 2, 3 or 4.  The opcode arrays specify the instructions used for
209   /// loads of D registers and even subregs and odd subregs of Q registers.
210   /// For NumVecs <= 2, QOpcodes1 is not used.
211   SDNode *SelectVLD(SDNode *N, bool isUpdating, unsigned NumVecs,
212                     const uint16_t *DOpcodes,
213                     const uint16_t *QOpcodes0, const uint16_t *QOpcodes1);
214 
215   /// SelectVST - Select NEON store intrinsics.  NumVecs should
216   /// be 1, 2, 3 or 4.  The opcode arrays specify the instructions used for
217   /// stores of D registers and even subregs and odd subregs of Q registers.
218   /// For NumVecs <= 2, QOpcodes1 is not used.
219   SDNode *SelectVST(SDNode *N, bool isUpdating, unsigned NumVecs,
220                     const uint16_t *DOpcodes,
221                     const uint16_t *QOpcodes0, const uint16_t *QOpcodes1);
222 
223   /// SelectVLDSTLane - Select NEON load/store lane intrinsics.  NumVecs should
224   /// be 2, 3 or 4.  The opcode arrays specify the instructions used for
225   /// load/store of D registers and Q registers.
226   SDNode *SelectVLDSTLane(SDNode *N, bool IsLoad,
227                           bool isUpdating, unsigned NumVecs,
228                           const uint16_t *DOpcodes, const uint16_t *QOpcodes);
229 
230   /// SelectVLDDup - Select NEON load-duplicate intrinsics.  NumVecs
231   /// should be 2, 3 or 4.  The opcode array specifies the instructions used
232   /// for loading D registers.  (Q registers are not supported.)
233   SDNode *SelectVLDDup(SDNode *N, bool isUpdating, unsigned NumVecs,
234                        const uint16_t *Opcodes);
235 
236   /// SelectVTBL - Select NEON VTBL and VTBX intrinsics.  NumVecs should be 2,
237   /// 3 or 4.  These are custom-selected so that a REG_SEQUENCE can be
238   /// generated to force the table registers to be consecutive.
239   SDNode *SelectVTBL(SDNode *N, bool IsExt, unsigned NumVecs, unsigned Opc);
240 
241   /// SelectV6T2BitfieldExtractOp - Select SBFX/UBFX instructions for ARM.
242   SDNode *SelectV6T2BitfieldExtractOp(SDNode *N, bool isSigned);
243 
244   // Select special operations if node forms integer ABS pattern
245   SDNode *SelectABSOp(SDNode *N);
246 
247   SDNode *SelectReadRegister(SDNode *N);
248   SDNode *SelectWriteRegister(SDNode *N);
249 
250   SDNode *SelectInlineAsm(SDNode *N);
251 
252   SDNode *SelectConcatVector(SDNode *N);
253 
254   SDNode *SelectSMLAWSMULW(SDNode *N);
255 
256   SDNode *SelectCMP_SWAP(SDNode *N);
257 
258   /// SelectInlineAsmMemoryOperand - Implement addressing mode selection for
259   /// inline asm expressions.
260   bool SelectInlineAsmMemoryOperand(const SDValue &Op, unsigned ConstraintID,
261                                     std::vector<SDValue> &OutOps) override;
262 
263   // Form pairs of consecutive R, S, D, or Q registers.
264   SDNode *createGPRPairNode(EVT VT, SDValue V0, SDValue V1);
265   SDNode *createSRegPairNode(EVT VT, SDValue V0, SDValue V1);
266   SDNode *createDRegPairNode(EVT VT, SDValue V0, SDValue V1);
267   SDNode *createQRegPairNode(EVT VT, SDValue V0, SDValue V1);
268 
269   // Form sequences of 4 consecutive S, D, or Q registers.
270   SDNode *createQuadSRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3);
271   SDNode *createQuadDRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3);
272   SDNode *createQuadQRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3);
273 
274   // Get the alignment operand for a NEON VLD or VST instruction.
275   SDValue GetVLDSTAlign(SDValue Align, SDLoc dl, unsigned NumVecs,
276                         bool is64BitVector);
277 
278   /// Returns the number of instructions required to materialize the given
279   /// constant in a register, or 3 if a literal pool load is needed.
280   unsigned ConstantMaterializationCost(unsigned Val) const;
281 
282   /// Checks if N is a multiplication by a constant where we can extract out a
283   /// power of two from the constant so that it can be used in a shift, but only
284   /// if it simplifies the materialization of the constant. Returns true if it
285   /// is, and assigns to PowerOfTwo the power of two that should be extracted
286   /// out and to NewMulConst the new constant to be multiplied by.
287   bool canExtractShiftFromMul(const SDValue &N, unsigned MaxShift,
288                               unsigned &PowerOfTwo, SDValue &NewMulConst) const;
289 
290   /// Replace N with M in CurDAG, in a way that also ensures that M gets
291   /// selected when N would have been selected.
292   void replaceDAGValue(const SDValue &N, SDValue M);
293 };
294 }
295 
296 /// isInt32Immediate - This method tests to see if the node is a 32-bit constant
297 /// operand. If so Imm will receive the 32-bit value.
298 static bool isInt32Immediate(SDNode *N, unsigned &Imm) {
299   if (N->getOpcode() == ISD::Constant && N->getValueType(0) == MVT::i32) {
300     Imm = cast<ConstantSDNode>(N)->getZExtValue();
301     return true;
302   }
303   return false;
304 }
305 
306 // isInt32Immediate - This method tests to see if a constant operand.
307 // If so Imm will receive the 32 bit value.
308 static bool isInt32Immediate(SDValue N, unsigned &Imm) {
309   return isInt32Immediate(N.getNode(), Imm);
310 }
311 
312 // isOpcWithIntImmediate - This method tests to see if the node is a specific
313 // opcode and that it has a immediate integer right operand.
314 // If so Imm will receive the 32 bit value.
315 static bool isOpcWithIntImmediate(SDNode *N, unsigned Opc, unsigned& Imm) {
316   return N->getOpcode() == Opc &&
317          isInt32Immediate(N->getOperand(1).getNode(), Imm);
318 }
319 
320 /// \brief Check whether a particular node is a constant value representable as
321 /// (N * Scale) where (N in [\p RangeMin, \p RangeMax).
322 ///
323 /// \param ScaledConstant [out] - On success, the pre-scaled constant value.
324 static bool isScaledConstantInRange(SDValue Node, int Scale,
325                                     int RangeMin, int RangeMax,
326                                     int &ScaledConstant) {
327   assert(Scale > 0 && "Invalid scale!");
328 
329   // Check that this is a constant.
330   const ConstantSDNode *C = dyn_cast<ConstantSDNode>(Node);
331   if (!C)
332     return false;
333 
334   ScaledConstant = (int) C->getZExtValue();
335   if ((ScaledConstant % Scale) != 0)
336     return false;
337 
338   ScaledConstant /= Scale;
339   return ScaledConstant >= RangeMin && ScaledConstant < RangeMax;
340 }
341 
342 void ARMDAGToDAGISel::PreprocessISelDAG() {
343   if (!Subtarget->hasV6T2Ops())
344     return;
345 
346   bool isThumb2 = Subtarget->isThumb();
347   for (SelectionDAG::allnodes_iterator I = CurDAG->allnodes_begin(),
348        E = CurDAG->allnodes_end(); I != E; ) {
349     SDNode *N = &*I++; // Preincrement iterator to avoid invalidation issues.
350 
351     if (N->getOpcode() != ISD::ADD)
352       continue;
353 
354     // Look for (add X1, (and (srl X2, c1), c2)) where c2 is constant with
355     // leading zeros, followed by consecutive set bits, followed by 1 or 2
356     // trailing zeros, e.g. 1020.
357     // Transform the expression to
358     // (add X1, (shl (and (srl X2, c1), (c2>>tz)), tz)) where tz is the number
359     // of trailing zeros of c2. The left shift would be folded as an shifter
360     // operand of 'add' and the 'and' and 'srl' would become a bits extraction
361     // node (UBFX).
362 
363     SDValue N0 = N->getOperand(0);
364     SDValue N1 = N->getOperand(1);
365     unsigned And_imm = 0;
366     if (!isOpcWithIntImmediate(N1.getNode(), ISD::AND, And_imm)) {
367       if (isOpcWithIntImmediate(N0.getNode(), ISD::AND, And_imm))
368         std::swap(N0, N1);
369     }
370     if (!And_imm)
371       continue;
372 
373     // Check if the AND mask is an immediate of the form: 000.....1111111100
374     unsigned TZ = countTrailingZeros(And_imm);
375     if (TZ != 1 && TZ != 2)
376       // Be conservative here. Shifter operands aren't always free. e.g. On
377       // Swift, left shifter operand of 1 / 2 for free but others are not.
378       // e.g.
379       //  ubfx   r3, r1, #16, #8
380       //  ldr.w  r3, [r0, r3, lsl #2]
381       // vs.
382       //  mov.w  r9, #1020
383       //  and.w  r2, r9, r1, lsr #14
384       //  ldr    r2, [r0, r2]
385       continue;
386     And_imm >>= TZ;
387     if (And_imm & (And_imm + 1))
388       continue;
389 
390     // Look for (and (srl X, c1), c2).
391     SDValue Srl = N1.getOperand(0);
392     unsigned Srl_imm = 0;
393     if (!isOpcWithIntImmediate(Srl.getNode(), ISD::SRL, Srl_imm) ||
394         (Srl_imm <= 2))
395       continue;
396 
397     // Make sure first operand is not a shifter operand which would prevent
398     // folding of the left shift.
399     SDValue CPTmp0;
400     SDValue CPTmp1;
401     SDValue CPTmp2;
402     if (isThumb2) {
403       if (SelectImmShifterOperand(N0, CPTmp0, CPTmp1))
404         continue;
405     } else {
406       if (SelectImmShifterOperand(N0, CPTmp0, CPTmp1) ||
407           SelectRegShifterOperand(N0, CPTmp0, CPTmp1, CPTmp2))
408         continue;
409     }
410 
411     // Now make the transformation.
412     Srl = CurDAG->getNode(ISD::SRL, SDLoc(Srl), MVT::i32,
413                           Srl.getOperand(0),
414                           CurDAG->getConstant(Srl_imm + TZ, SDLoc(Srl),
415                                               MVT::i32));
416     N1 = CurDAG->getNode(ISD::AND, SDLoc(N1), MVT::i32,
417                          Srl,
418                          CurDAG->getConstant(And_imm, SDLoc(Srl), MVT::i32));
419     N1 = CurDAG->getNode(ISD::SHL, SDLoc(N1), MVT::i32,
420                          N1, CurDAG->getConstant(TZ, SDLoc(Srl), MVT::i32));
421     CurDAG->UpdateNodeOperands(N, N0, N1);
422   }
423 }
424 
425 /// hasNoVMLxHazardUse - Return true if it's desirable to select a FP MLA / MLS
426 /// node. VFP / NEON fp VMLA / VMLS instructions have special RAW hazards (at
427 /// least on current ARM implementations) which should be avoidded.
428 bool ARMDAGToDAGISel::hasNoVMLxHazardUse(SDNode *N) const {
429   if (OptLevel == CodeGenOpt::None)
430     return true;
431 
432   if (!CheckVMLxHazard)
433     return true;
434 
435   if (!Subtarget->isCortexA7() && !Subtarget->isCortexA8() &&
436       !Subtarget->isCortexA9() && !Subtarget->isSwift())
437     return true;
438 
439   if (!N->hasOneUse())
440     return false;
441 
442   SDNode *Use = *N->use_begin();
443   if (Use->getOpcode() == ISD::CopyToReg)
444     return true;
445   if (Use->isMachineOpcode()) {
446     const ARMBaseInstrInfo *TII = static_cast<const ARMBaseInstrInfo *>(
447         CurDAG->getSubtarget().getInstrInfo());
448 
449     const MCInstrDesc &MCID = TII->get(Use->getMachineOpcode());
450     if (MCID.mayStore())
451       return true;
452     unsigned Opcode = MCID.getOpcode();
453     if (Opcode == ARM::VMOVRS || Opcode == ARM::VMOVRRD)
454       return true;
455     // vmlx feeding into another vmlx. We actually want to unfold
456     // the use later in the MLxExpansion pass. e.g.
457     // vmla
458     // vmla (stall 8 cycles)
459     //
460     // vmul (5 cycles)
461     // vadd (5 cycles)
462     // vmla
463     // This adds up to about 18 - 19 cycles.
464     //
465     // vmla
466     // vmul (stall 4 cycles)
467     // vadd adds up to about 14 cycles.
468     return TII->isFpMLxInstruction(Opcode);
469   }
470 
471   return false;
472 }
473 
474 bool ARMDAGToDAGISel::isShifterOpProfitable(const SDValue &Shift,
475                                             ARM_AM::ShiftOpc ShOpcVal,
476                                             unsigned ShAmt) {
477   if (!Subtarget->isLikeA9() && !Subtarget->isSwift())
478     return true;
479   if (Shift.hasOneUse())
480     return true;
481   // R << 2 is free.
482   return ShOpcVal == ARM_AM::lsl &&
483          (ShAmt == 2 || (Subtarget->isSwift() && ShAmt == 1));
484 }
485 
486 unsigned ARMDAGToDAGISel::ConstantMaterializationCost(unsigned Val) const {
487   if (Subtarget->isThumb()) {
488     if (Val <= 255) return 1;                               // MOV
489     if (Subtarget->hasV6T2Ops() && Val <= 0xffff) return 1; // MOVW
490     if (~Val <= 255) return 2;                              // MOV + MVN
491     if (ARM_AM::isThumbImmShiftedVal(Val)) return 2;        // MOV + LSL
492   } else {
493     if (ARM_AM::getSOImmVal(Val) != -1) return 1;           // MOV
494     if (ARM_AM::getSOImmVal(~Val) != -1) return 1;          // MVN
495     if (Subtarget->hasV6T2Ops() && Val <= 0xffff) return 1; // MOVW
496     if (ARM_AM::isSOImmTwoPartVal(Val)) return 2;           // two instrs
497   }
498   if (Subtarget->useMovt(*MF)) return 2; // MOVW + MOVT
499   return 3; // Literal pool load
500 }
501 
502 bool ARMDAGToDAGISel::canExtractShiftFromMul(const SDValue &N,
503                                              unsigned MaxShift,
504                                              unsigned &PowerOfTwo,
505                                              SDValue &NewMulConst) const {
506   assert(N.getOpcode() == ISD::MUL);
507   assert(MaxShift > 0);
508 
509   // If the multiply is used in more than one place then changing the constant
510   // will make other uses incorrect, so don't.
511   if (!N.hasOneUse()) return false;
512   // Check if the multiply is by a constant
513   ConstantSDNode *MulConst = dyn_cast<ConstantSDNode>(N.getOperand(1));
514   if (!MulConst) return false;
515   // If the constant is used in more than one place then modifying it will mean
516   // we need to materialize two constants instead of one, which is a bad idea.
517   if (!MulConst->hasOneUse()) return false;
518   unsigned MulConstVal = MulConst->getZExtValue();
519   if (MulConstVal == 0) return false;
520 
521   // Find the largest power of 2 that MulConstVal is a multiple of
522   PowerOfTwo = MaxShift;
523   while ((MulConstVal % (1 << PowerOfTwo)) != 0) {
524     --PowerOfTwo;
525     if (PowerOfTwo == 0) return false;
526   }
527 
528   // Only optimise if the new cost is better
529   unsigned NewMulConstVal = MulConstVal / (1 << PowerOfTwo);
530   NewMulConst = CurDAG->getConstant(NewMulConstVal, SDLoc(N), MVT::i32);
531   unsigned OldCost = ConstantMaterializationCost(MulConstVal);
532   unsigned NewCost = ConstantMaterializationCost(NewMulConstVal);
533   return NewCost < OldCost;
534 }
535 
536 void ARMDAGToDAGISel::replaceDAGValue(const SDValue &N, SDValue M) {
537   CurDAG->RepositionNode(N.getNode()->getIterator(), M.getNode());
538   CurDAG->ReplaceAllUsesWith(N, M);
539 }
540 
541 bool ARMDAGToDAGISel::SelectImmShifterOperand(SDValue N,
542                                               SDValue &BaseReg,
543                                               SDValue &Opc,
544                                               bool CheckProfitability) {
545   if (DisableShifterOp)
546     return false;
547 
548   // If N is a multiply-by-constant and it's profitable to extract a shift and
549   // use it in a shifted operand do so.
550   if (N.getOpcode() == ISD::MUL) {
551     unsigned PowerOfTwo = 0;
552     SDValue NewMulConst;
553     if (canExtractShiftFromMul(N, 31, PowerOfTwo, NewMulConst)) {
554       BaseReg = SDValue(Select(CurDAG->getNode(ISD::MUL, SDLoc(N), MVT::i32,
555                                                N.getOperand(0), NewMulConst)
556                                    .getNode()),
557                         0);
558       replaceDAGValue(N.getOperand(1), NewMulConst);
559       Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ARM_AM::lsl,
560                                                           PowerOfTwo),
561                                       SDLoc(N), MVT::i32);
562       return true;
563     }
564   }
565 
566   ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode());
567 
568   // Don't match base register only case. That is matched to a separate
569   // lower complexity pattern with explicit register operand.
570   if (ShOpcVal == ARM_AM::no_shift) return false;
571 
572   BaseReg = N.getOperand(0);
573   unsigned ShImmVal = 0;
574   ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1));
575   if (!RHS) return false;
576   ShImmVal = RHS->getZExtValue() & 31;
577   Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, ShImmVal),
578                                   SDLoc(N), MVT::i32);
579   return true;
580 }
581 
582 bool ARMDAGToDAGISel::SelectRegShifterOperand(SDValue N,
583                                               SDValue &BaseReg,
584                                               SDValue &ShReg,
585                                               SDValue &Opc,
586                                               bool CheckProfitability) {
587   if (DisableShifterOp)
588     return false;
589 
590   ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode());
591 
592   // Don't match base register only case. That is matched to a separate
593   // lower complexity pattern with explicit register operand.
594   if (ShOpcVal == ARM_AM::no_shift) return false;
595 
596   BaseReg = N.getOperand(0);
597   unsigned ShImmVal = 0;
598   ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1));
599   if (RHS) return false;
600 
601   ShReg = N.getOperand(1);
602   if (CheckProfitability && !isShifterOpProfitable(N, ShOpcVal, ShImmVal))
603     return false;
604   Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, ShImmVal),
605                                   SDLoc(N), MVT::i32);
606   return true;
607 }
608 
609 
610 bool ARMDAGToDAGISel::SelectAddrModeImm12(SDValue N,
611                                           SDValue &Base,
612                                           SDValue &OffImm) {
613   // Match simple R + imm12 operands.
614 
615   // Base only.
616   if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB &&
617       !CurDAG->isBaseWithConstantOffset(N)) {
618     if (N.getOpcode() == ISD::FrameIndex) {
619       // Match frame index.
620       int FI = cast<FrameIndexSDNode>(N)->getIndex();
621       Base = CurDAG->getTargetFrameIndex(
622           FI, TLI->getPointerTy(CurDAG->getDataLayout()));
623       OffImm  = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
624       return true;
625     }
626 
627     if (N.getOpcode() == ARMISD::Wrapper &&
628         N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress &&
629         N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol &&
630         N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) {
631       Base = N.getOperand(0);
632     } else
633       Base = N;
634     OffImm  = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
635     return true;
636   }
637 
638   if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
639     int RHSC = (int)RHS->getSExtValue();
640     if (N.getOpcode() == ISD::SUB)
641       RHSC = -RHSC;
642 
643     if (RHSC > -0x1000 && RHSC < 0x1000) { // 12 bits
644       Base   = N.getOperand(0);
645       if (Base.getOpcode() == ISD::FrameIndex) {
646         int FI = cast<FrameIndexSDNode>(Base)->getIndex();
647         Base = CurDAG->getTargetFrameIndex(
648             FI, TLI->getPointerTy(CurDAG->getDataLayout()));
649       }
650       OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32);
651       return true;
652     }
653   }
654 
655   // Base only.
656   Base = N;
657   OffImm  = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
658   return true;
659 }
660 
661 
662 
663 bool ARMDAGToDAGISel::SelectLdStSOReg(SDValue N, SDValue &Base, SDValue &Offset,
664                                       SDValue &Opc) {
665   if (N.getOpcode() == ISD::MUL &&
666       ((!Subtarget->isLikeA9() && !Subtarget->isSwift()) || N.hasOneUse())) {
667     if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
668       // X * [3,5,9] -> X + X * [2,4,8] etc.
669       int RHSC = (int)RHS->getZExtValue();
670       if (RHSC & 1) {
671         RHSC = RHSC & ~1;
672         ARM_AM::AddrOpc AddSub = ARM_AM::add;
673         if (RHSC < 0) {
674           AddSub = ARM_AM::sub;
675           RHSC = - RHSC;
676         }
677         if (isPowerOf2_32(RHSC)) {
678           unsigned ShAmt = Log2_32(RHSC);
679           Base = Offset = N.getOperand(0);
680           Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt,
681                                                             ARM_AM::lsl),
682                                           SDLoc(N), MVT::i32);
683           return true;
684         }
685       }
686     }
687   }
688 
689   if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB &&
690       // ISD::OR that is equivalent to an ISD::ADD.
691       !CurDAG->isBaseWithConstantOffset(N))
692     return false;
693 
694   // Leave simple R +/- imm12 operands for LDRi12
695   if (N.getOpcode() == ISD::ADD || N.getOpcode() == ISD::OR) {
696     int RHSC;
697     if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1,
698                                 -0x1000+1, 0x1000, RHSC)) // 12 bits.
699       return false;
700   }
701 
702   // Otherwise this is R +/- [possibly shifted] R.
703   ARM_AM::AddrOpc AddSub = N.getOpcode() == ISD::SUB ? ARM_AM::sub:ARM_AM::add;
704   ARM_AM::ShiftOpc ShOpcVal =
705     ARM_AM::getShiftOpcForNode(N.getOperand(1).getOpcode());
706   unsigned ShAmt = 0;
707 
708   Base   = N.getOperand(0);
709   Offset = N.getOperand(1);
710 
711   if (ShOpcVal != ARM_AM::no_shift) {
712     // Check to see if the RHS of the shift is a constant, if not, we can't fold
713     // it.
714     if (ConstantSDNode *Sh =
715            dyn_cast<ConstantSDNode>(N.getOperand(1).getOperand(1))) {
716       ShAmt = Sh->getZExtValue();
717       if (isShifterOpProfitable(Offset, ShOpcVal, ShAmt))
718         Offset = N.getOperand(1).getOperand(0);
719       else {
720         ShAmt = 0;
721         ShOpcVal = ARM_AM::no_shift;
722       }
723     } else {
724       ShOpcVal = ARM_AM::no_shift;
725     }
726   }
727 
728   // Try matching (R shl C) + (R).
729   if (N.getOpcode() != ISD::SUB && ShOpcVal == ARM_AM::no_shift &&
730       !(Subtarget->isLikeA9() || Subtarget->isSwift() ||
731         N.getOperand(0).hasOneUse())) {
732     ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOperand(0).getOpcode());
733     if (ShOpcVal != ARM_AM::no_shift) {
734       // Check to see if the RHS of the shift is a constant, if not, we can't
735       // fold it.
736       if (ConstantSDNode *Sh =
737           dyn_cast<ConstantSDNode>(N.getOperand(0).getOperand(1))) {
738         ShAmt = Sh->getZExtValue();
739         if (isShifterOpProfitable(N.getOperand(0), ShOpcVal, ShAmt)) {
740           Offset = N.getOperand(0).getOperand(0);
741           Base = N.getOperand(1);
742         } else {
743           ShAmt = 0;
744           ShOpcVal = ARM_AM::no_shift;
745         }
746       } else {
747         ShOpcVal = ARM_AM::no_shift;
748       }
749     }
750   }
751 
752   // If Offset is a multiply-by-constant and it's profitable to extract a shift
753   // and use it in a shifted operand do so.
754   if (Offset.getOpcode() == ISD::MUL && N.hasOneUse()) {
755     unsigned PowerOfTwo = 0;
756     SDValue NewMulConst;
757     if (canExtractShiftFromMul(Offset, 31, PowerOfTwo, NewMulConst)) {
758       replaceDAGValue(Offset.getOperand(1), NewMulConst);
759       ShAmt = PowerOfTwo;
760       ShOpcVal = ARM_AM::lsl;
761     }
762   }
763 
764   Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal),
765                                   SDLoc(N), MVT::i32);
766   return true;
767 }
768 
769 
770 //-----
771 
772 AddrMode2Type ARMDAGToDAGISel::SelectAddrMode2Worker(SDValue N,
773                                                      SDValue &Base,
774                                                      SDValue &Offset,
775                                                      SDValue &Opc) {
776   if (N.getOpcode() == ISD::MUL &&
777       (!(Subtarget->isLikeA9() || Subtarget->isSwift()) || N.hasOneUse())) {
778     if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
779       // X * [3,5,9] -> X + X * [2,4,8] etc.
780       int RHSC = (int)RHS->getZExtValue();
781       if (RHSC & 1) {
782         RHSC = RHSC & ~1;
783         ARM_AM::AddrOpc AddSub = ARM_AM::add;
784         if (RHSC < 0) {
785           AddSub = ARM_AM::sub;
786           RHSC = - RHSC;
787         }
788         if (isPowerOf2_32(RHSC)) {
789           unsigned ShAmt = Log2_32(RHSC);
790           Base = Offset = N.getOperand(0);
791           Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt,
792                                                             ARM_AM::lsl),
793                                           SDLoc(N), MVT::i32);
794           return AM2_SHOP;
795         }
796       }
797     }
798   }
799 
800   if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB &&
801       // ISD::OR that is equivalent to an ADD.
802       !CurDAG->isBaseWithConstantOffset(N)) {
803     Base = N;
804     if (N.getOpcode() == ISD::FrameIndex) {
805       int FI = cast<FrameIndexSDNode>(N)->getIndex();
806       Base = CurDAG->getTargetFrameIndex(
807           FI, TLI->getPointerTy(CurDAG->getDataLayout()));
808     } else if (N.getOpcode() == ARMISD::Wrapper &&
809                N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress &&
810                N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol &&
811                N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) {
812       Base = N.getOperand(0);
813     }
814     Offset = CurDAG->getRegister(0, MVT::i32);
815     Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(ARM_AM::add, 0,
816                                                       ARM_AM::no_shift),
817                                     SDLoc(N), MVT::i32);
818     return AM2_BASE;
819   }
820 
821   // Match simple R +/- imm12 operands.
822   if (N.getOpcode() != ISD::SUB) {
823     int RHSC;
824     if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1,
825                                 -0x1000+1, 0x1000, RHSC)) { // 12 bits.
826       Base = N.getOperand(0);
827       if (Base.getOpcode() == ISD::FrameIndex) {
828         int FI = cast<FrameIndexSDNode>(Base)->getIndex();
829         Base = CurDAG->getTargetFrameIndex(
830             FI, TLI->getPointerTy(CurDAG->getDataLayout()));
831       }
832       Offset = CurDAG->getRegister(0, MVT::i32);
833 
834       ARM_AM::AddrOpc AddSub = ARM_AM::add;
835       if (RHSC < 0) {
836         AddSub = ARM_AM::sub;
837         RHSC = - RHSC;
838       }
839       Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, RHSC,
840                                                         ARM_AM::no_shift),
841                                       SDLoc(N), MVT::i32);
842       return AM2_BASE;
843     }
844   }
845 
846   if ((Subtarget->isLikeA9() || Subtarget->isSwift()) && !N.hasOneUse()) {
847     // Compute R +/- (R << N) and reuse it.
848     Base = N;
849     Offset = CurDAG->getRegister(0, MVT::i32);
850     Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(ARM_AM::add, 0,
851                                                       ARM_AM::no_shift),
852                                     SDLoc(N), MVT::i32);
853     return AM2_BASE;
854   }
855 
856   // Otherwise this is R +/- [possibly shifted] R.
857   ARM_AM::AddrOpc AddSub = N.getOpcode() != ISD::SUB ? ARM_AM::add:ARM_AM::sub;
858   ARM_AM::ShiftOpc ShOpcVal =
859     ARM_AM::getShiftOpcForNode(N.getOperand(1).getOpcode());
860   unsigned ShAmt = 0;
861 
862   Base   = N.getOperand(0);
863   Offset = N.getOperand(1);
864 
865   if (ShOpcVal != ARM_AM::no_shift) {
866     // Check to see if the RHS of the shift is a constant, if not, we can't fold
867     // it.
868     if (ConstantSDNode *Sh =
869            dyn_cast<ConstantSDNode>(N.getOperand(1).getOperand(1))) {
870       ShAmt = Sh->getZExtValue();
871       if (isShifterOpProfitable(Offset, ShOpcVal, ShAmt))
872         Offset = N.getOperand(1).getOperand(0);
873       else {
874         ShAmt = 0;
875         ShOpcVal = ARM_AM::no_shift;
876       }
877     } else {
878       ShOpcVal = ARM_AM::no_shift;
879     }
880   }
881 
882   // Try matching (R shl C) + (R).
883   if (N.getOpcode() != ISD::SUB && ShOpcVal == ARM_AM::no_shift &&
884       !(Subtarget->isLikeA9() || Subtarget->isSwift() ||
885         N.getOperand(0).hasOneUse())) {
886     ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOperand(0).getOpcode());
887     if (ShOpcVal != ARM_AM::no_shift) {
888       // Check to see if the RHS of the shift is a constant, if not, we can't
889       // fold it.
890       if (ConstantSDNode *Sh =
891           dyn_cast<ConstantSDNode>(N.getOperand(0).getOperand(1))) {
892         ShAmt = Sh->getZExtValue();
893         if (isShifterOpProfitable(N.getOperand(0), ShOpcVal, ShAmt)) {
894           Offset = N.getOperand(0).getOperand(0);
895           Base = N.getOperand(1);
896         } else {
897           ShAmt = 0;
898           ShOpcVal = ARM_AM::no_shift;
899         }
900       } else {
901         ShOpcVal = ARM_AM::no_shift;
902       }
903     }
904   }
905 
906   Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal),
907                                   SDLoc(N), MVT::i32);
908   return AM2_SHOP;
909 }
910 
911 bool ARMDAGToDAGISel::SelectAddrMode2OffsetReg(SDNode *Op, SDValue N,
912                                             SDValue &Offset, SDValue &Opc) {
913   unsigned Opcode = Op->getOpcode();
914   ISD::MemIndexedMode AM = (Opcode == ISD::LOAD)
915     ? cast<LoadSDNode>(Op)->getAddressingMode()
916     : cast<StoreSDNode>(Op)->getAddressingMode();
917   ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC)
918     ? ARM_AM::add : ARM_AM::sub;
919   int Val;
920   if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val))
921     return false;
922 
923   Offset = N;
924   ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode());
925   unsigned ShAmt = 0;
926   if (ShOpcVal != ARM_AM::no_shift) {
927     // Check to see if the RHS of the shift is a constant, if not, we can't fold
928     // it.
929     if (ConstantSDNode *Sh = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
930       ShAmt = Sh->getZExtValue();
931       if (isShifterOpProfitable(N, ShOpcVal, ShAmt))
932         Offset = N.getOperand(0);
933       else {
934         ShAmt = 0;
935         ShOpcVal = ARM_AM::no_shift;
936       }
937     } else {
938       ShOpcVal = ARM_AM::no_shift;
939     }
940   }
941 
942   Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal),
943                                   SDLoc(N), MVT::i32);
944   return true;
945 }
946 
947 bool ARMDAGToDAGISel::SelectAddrMode2OffsetImmPre(SDNode *Op, SDValue N,
948                                             SDValue &Offset, SDValue &Opc) {
949   unsigned Opcode = Op->getOpcode();
950   ISD::MemIndexedMode AM = (Opcode == ISD::LOAD)
951     ? cast<LoadSDNode>(Op)->getAddressingMode()
952     : cast<StoreSDNode>(Op)->getAddressingMode();
953   ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC)
954     ? ARM_AM::add : ARM_AM::sub;
955   int Val;
956   if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) { // 12 bits.
957     if (AddSub == ARM_AM::sub) Val *= -1;
958     Offset = CurDAG->getRegister(0, MVT::i32);
959     Opc = CurDAG->getTargetConstant(Val, SDLoc(Op), MVT::i32);
960     return true;
961   }
962 
963   return false;
964 }
965 
966 
967 bool ARMDAGToDAGISel::SelectAddrMode2OffsetImm(SDNode *Op, SDValue N,
968                                             SDValue &Offset, SDValue &Opc) {
969   unsigned Opcode = Op->getOpcode();
970   ISD::MemIndexedMode AM = (Opcode == ISD::LOAD)
971     ? cast<LoadSDNode>(Op)->getAddressingMode()
972     : cast<StoreSDNode>(Op)->getAddressingMode();
973   ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC)
974     ? ARM_AM::add : ARM_AM::sub;
975   int Val;
976   if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) { // 12 bits.
977     Offset = CurDAG->getRegister(0, MVT::i32);
978     Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, Val,
979                                                       ARM_AM::no_shift),
980                                     SDLoc(Op), MVT::i32);
981     return true;
982   }
983 
984   return false;
985 }
986 
987 bool ARMDAGToDAGISel::SelectAddrOffsetNone(SDValue N, SDValue &Base) {
988   Base = N;
989   return true;
990 }
991 
992 bool ARMDAGToDAGISel::SelectAddrMode3(SDValue N,
993                                       SDValue &Base, SDValue &Offset,
994                                       SDValue &Opc) {
995   if (N.getOpcode() == ISD::SUB) {
996     // X - C  is canonicalize to X + -C, no need to handle it here.
997     Base = N.getOperand(0);
998     Offset = N.getOperand(1);
999     Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::sub, 0), SDLoc(N),
1000                                     MVT::i32);
1001     return true;
1002   }
1003 
1004   if (!CurDAG->isBaseWithConstantOffset(N)) {
1005     Base = N;
1006     if (N.getOpcode() == ISD::FrameIndex) {
1007       int FI = cast<FrameIndexSDNode>(N)->getIndex();
1008       Base = CurDAG->getTargetFrameIndex(
1009           FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1010     }
1011     Offset = CurDAG->getRegister(0, MVT::i32);
1012     Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::add, 0), SDLoc(N),
1013                                     MVT::i32);
1014     return true;
1015   }
1016 
1017   // If the RHS is +/- imm8, fold into addr mode.
1018   int RHSC;
1019   if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1,
1020                               -256 + 1, 256, RHSC)) { // 8 bits.
1021     Base = N.getOperand(0);
1022     if (Base.getOpcode() == ISD::FrameIndex) {
1023       int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1024       Base = CurDAG->getTargetFrameIndex(
1025           FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1026     }
1027     Offset = CurDAG->getRegister(0, MVT::i32);
1028 
1029     ARM_AM::AddrOpc AddSub = ARM_AM::add;
1030     if (RHSC < 0) {
1031       AddSub = ARM_AM::sub;
1032       RHSC = -RHSC;
1033     }
1034     Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, RHSC), SDLoc(N),
1035                                     MVT::i32);
1036     return true;
1037   }
1038 
1039   Base = N.getOperand(0);
1040   Offset = N.getOperand(1);
1041   Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::add, 0), SDLoc(N),
1042                                   MVT::i32);
1043   return true;
1044 }
1045 
1046 bool ARMDAGToDAGISel::SelectAddrMode3Offset(SDNode *Op, SDValue N,
1047                                             SDValue &Offset, SDValue &Opc) {
1048   unsigned Opcode = Op->getOpcode();
1049   ISD::MemIndexedMode AM = (Opcode == ISD::LOAD)
1050     ? cast<LoadSDNode>(Op)->getAddressingMode()
1051     : cast<StoreSDNode>(Op)->getAddressingMode();
1052   ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC)
1053     ? ARM_AM::add : ARM_AM::sub;
1054   int Val;
1055   if (isScaledConstantInRange(N, /*Scale=*/1, 0, 256, Val)) { // 12 bits.
1056     Offset = CurDAG->getRegister(0, MVT::i32);
1057     Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, Val), SDLoc(Op),
1058                                     MVT::i32);
1059     return true;
1060   }
1061 
1062   Offset = N;
1063   Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, 0), SDLoc(Op),
1064                                   MVT::i32);
1065   return true;
1066 }
1067 
1068 bool ARMDAGToDAGISel::SelectAddrMode5(SDValue N,
1069                                       SDValue &Base, SDValue &Offset) {
1070   if (!CurDAG->isBaseWithConstantOffset(N)) {
1071     Base = N;
1072     if (N.getOpcode() == ISD::FrameIndex) {
1073       int FI = cast<FrameIndexSDNode>(N)->getIndex();
1074       Base = CurDAG->getTargetFrameIndex(
1075           FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1076     } else if (N.getOpcode() == ARMISD::Wrapper &&
1077                N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress &&
1078                N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol &&
1079                N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) {
1080       Base = N.getOperand(0);
1081     }
1082     Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(ARM_AM::add, 0),
1083                                        SDLoc(N), MVT::i32);
1084     return true;
1085   }
1086 
1087   // If the RHS is +/- imm8, fold into addr mode.
1088   int RHSC;
1089   if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/4,
1090                               -256 + 1, 256, RHSC)) {
1091     Base = N.getOperand(0);
1092     if (Base.getOpcode() == ISD::FrameIndex) {
1093       int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1094       Base = CurDAG->getTargetFrameIndex(
1095           FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1096     }
1097 
1098     ARM_AM::AddrOpc AddSub = ARM_AM::add;
1099     if (RHSC < 0) {
1100       AddSub = ARM_AM::sub;
1101       RHSC = -RHSC;
1102     }
1103     Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(AddSub, RHSC),
1104                                        SDLoc(N), MVT::i32);
1105     return true;
1106   }
1107 
1108   Base = N;
1109   Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(ARM_AM::add, 0),
1110                                      SDLoc(N), MVT::i32);
1111   return true;
1112 }
1113 
1114 bool ARMDAGToDAGISel::SelectAddrMode6(SDNode *Parent, SDValue N, SDValue &Addr,
1115                                       SDValue &Align) {
1116   Addr = N;
1117 
1118   unsigned Alignment = 0;
1119 
1120   MemSDNode *MemN = cast<MemSDNode>(Parent);
1121 
1122   if (isa<LSBaseSDNode>(MemN) ||
1123       ((MemN->getOpcode() == ARMISD::VST1_UPD ||
1124         MemN->getOpcode() == ARMISD::VLD1_UPD) &&
1125        MemN->getConstantOperandVal(MemN->getNumOperands() - 1) == 1)) {
1126     // This case occurs only for VLD1-lane/dup and VST1-lane instructions.
1127     // The maximum alignment is equal to the memory size being referenced.
1128     unsigned MMOAlign = MemN->getAlignment();
1129     unsigned MemSize = MemN->getMemoryVT().getSizeInBits() / 8;
1130     if (MMOAlign >= MemSize && MemSize > 1)
1131       Alignment = MemSize;
1132   } else {
1133     // All other uses of addrmode6 are for intrinsics.  For now just record
1134     // the raw alignment value; it will be refined later based on the legal
1135     // alignment operands for the intrinsic.
1136     Alignment = MemN->getAlignment();
1137   }
1138 
1139   Align = CurDAG->getTargetConstant(Alignment, SDLoc(N), MVT::i32);
1140   return true;
1141 }
1142 
1143 bool ARMDAGToDAGISel::SelectAddrMode6Offset(SDNode *Op, SDValue N,
1144                                             SDValue &Offset) {
1145   LSBaseSDNode *LdSt = cast<LSBaseSDNode>(Op);
1146   ISD::MemIndexedMode AM = LdSt->getAddressingMode();
1147   if (AM != ISD::POST_INC)
1148     return false;
1149   Offset = N;
1150   if (ConstantSDNode *NC = dyn_cast<ConstantSDNode>(N)) {
1151     if (NC->getZExtValue() * 8 == LdSt->getMemoryVT().getSizeInBits())
1152       Offset = CurDAG->getRegister(0, MVT::i32);
1153   }
1154   return true;
1155 }
1156 
1157 bool ARMDAGToDAGISel::SelectAddrModePC(SDValue N,
1158                                        SDValue &Offset, SDValue &Label) {
1159   if (N.getOpcode() == ARMISD::PIC_ADD && N.hasOneUse()) {
1160     Offset = N.getOperand(0);
1161     SDValue N1 = N.getOperand(1);
1162     Label = CurDAG->getTargetConstant(cast<ConstantSDNode>(N1)->getZExtValue(),
1163                                       SDLoc(N), MVT::i32);
1164     return true;
1165   }
1166 
1167   return false;
1168 }
1169 
1170 
1171 //===----------------------------------------------------------------------===//
1172 //                         Thumb Addressing Modes
1173 //===----------------------------------------------------------------------===//
1174 
1175 bool ARMDAGToDAGISel::SelectThumbAddrModeRR(SDValue N,
1176                                             SDValue &Base, SDValue &Offset){
1177   if (N.getOpcode() != ISD::ADD && !CurDAG->isBaseWithConstantOffset(N)) {
1178     ConstantSDNode *NC = dyn_cast<ConstantSDNode>(N);
1179     if (!NC || !NC->isNullValue())
1180       return false;
1181 
1182     Base = Offset = N;
1183     return true;
1184   }
1185 
1186   Base = N.getOperand(0);
1187   Offset = N.getOperand(1);
1188   return true;
1189 }
1190 
1191 bool
1192 ARMDAGToDAGISel::SelectThumbAddrModeImm5S(SDValue N, unsigned Scale,
1193                                           SDValue &Base, SDValue &OffImm) {
1194   if (!CurDAG->isBaseWithConstantOffset(N)) {
1195     if (N.getOpcode() == ISD::ADD) {
1196       return false; // We want to select register offset instead
1197     } else if (N.getOpcode() == ARMISD::Wrapper &&
1198         N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress &&
1199         N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol &&
1200         N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) {
1201       Base = N.getOperand(0);
1202     } else {
1203       Base = N;
1204     }
1205 
1206     OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
1207     return true;
1208   }
1209 
1210   // If the RHS is + imm5 * scale, fold into addr mode.
1211   int RHSC;
1212   if (isScaledConstantInRange(N.getOperand(1), Scale, 0, 32, RHSC)) {
1213     Base = N.getOperand(0);
1214     OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32);
1215     return true;
1216   }
1217 
1218   // Offset is too large, so use register offset instead.
1219   return false;
1220 }
1221 
1222 bool
1223 ARMDAGToDAGISel::SelectThumbAddrModeImm5S4(SDValue N, SDValue &Base,
1224                                            SDValue &OffImm) {
1225   return SelectThumbAddrModeImm5S(N, 4, Base, OffImm);
1226 }
1227 
1228 bool
1229 ARMDAGToDAGISel::SelectThumbAddrModeImm5S2(SDValue N, SDValue &Base,
1230                                            SDValue &OffImm) {
1231   return SelectThumbAddrModeImm5S(N, 2, Base, OffImm);
1232 }
1233 
1234 bool
1235 ARMDAGToDAGISel::SelectThumbAddrModeImm5S1(SDValue N, SDValue &Base,
1236                                            SDValue &OffImm) {
1237   return SelectThumbAddrModeImm5S(N, 1, Base, OffImm);
1238 }
1239 
1240 bool ARMDAGToDAGISel::SelectThumbAddrModeSP(SDValue N,
1241                                             SDValue &Base, SDValue &OffImm) {
1242   if (N.getOpcode() == ISD::FrameIndex) {
1243     int FI = cast<FrameIndexSDNode>(N)->getIndex();
1244     // Only multiples of 4 are allowed for the offset, so the frame object
1245     // alignment must be at least 4.
1246     MachineFrameInfo *MFI = MF->getFrameInfo();
1247     if (MFI->getObjectAlignment(FI) < 4)
1248       MFI->setObjectAlignment(FI, 4);
1249     Base = CurDAG->getTargetFrameIndex(
1250         FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1251     OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
1252     return true;
1253   }
1254 
1255   if (!CurDAG->isBaseWithConstantOffset(N))
1256     return false;
1257 
1258   RegisterSDNode *LHSR = dyn_cast<RegisterSDNode>(N.getOperand(0));
1259   if (N.getOperand(0).getOpcode() == ISD::FrameIndex ||
1260       (LHSR && LHSR->getReg() == ARM::SP)) {
1261     // If the RHS is + imm8 * scale, fold into addr mode.
1262     int RHSC;
1263     if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/4, 0, 256, RHSC)) {
1264       Base = N.getOperand(0);
1265       if (Base.getOpcode() == ISD::FrameIndex) {
1266         int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1267         // For LHS+RHS to result in an offset that's a multiple of 4 the object
1268         // indexed by the LHS must be 4-byte aligned.
1269         MachineFrameInfo *MFI = MF->getFrameInfo();
1270         if (MFI->getObjectAlignment(FI) < 4)
1271           MFI->setObjectAlignment(FI, 4);
1272         Base = CurDAG->getTargetFrameIndex(
1273             FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1274       }
1275       OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32);
1276       return true;
1277     }
1278   }
1279 
1280   return false;
1281 }
1282 
1283 
1284 //===----------------------------------------------------------------------===//
1285 //                        Thumb 2 Addressing Modes
1286 //===----------------------------------------------------------------------===//
1287 
1288 
1289 bool ARMDAGToDAGISel::SelectT2AddrModeImm12(SDValue N,
1290                                             SDValue &Base, SDValue &OffImm) {
1291   // Match simple R + imm12 operands.
1292 
1293   // Base only.
1294   if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB &&
1295       !CurDAG->isBaseWithConstantOffset(N)) {
1296     if (N.getOpcode() == ISD::FrameIndex) {
1297       // Match frame index.
1298       int FI = cast<FrameIndexSDNode>(N)->getIndex();
1299       Base = CurDAG->getTargetFrameIndex(
1300           FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1301       OffImm  = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
1302       return true;
1303     }
1304 
1305     if (N.getOpcode() == ARMISD::Wrapper &&
1306         N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress &&
1307         N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol &&
1308         N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) {
1309       Base = N.getOperand(0);
1310       if (Base.getOpcode() == ISD::TargetConstantPool)
1311         return false;  // We want to select t2LDRpci instead.
1312     } else
1313       Base = N;
1314     OffImm  = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
1315     return true;
1316   }
1317 
1318   if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1319     if (SelectT2AddrModeImm8(N, Base, OffImm))
1320       // Let t2LDRi8 handle (R - imm8).
1321       return false;
1322 
1323     int RHSC = (int)RHS->getZExtValue();
1324     if (N.getOpcode() == ISD::SUB)
1325       RHSC = -RHSC;
1326 
1327     if (RHSC >= 0 && RHSC < 0x1000) { // 12 bits (unsigned)
1328       Base   = N.getOperand(0);
1329       if (Base.getOpcode() == ISD::FrameIndex) {
1330         int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1331         Base = CurDAG->getTargetFrameIndex(
1332             FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1333       }
1334       OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32);
1335       return true;
1336     }
1337   }
1338 
1339   // Base only.
1340   Base = N;
1341   OffImm  = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
1342   return true;
1343 }
1344 
1345 bool ARMDAGToDAGISel::SelectT2AddrModeImm8(SDValue N,
1346                                            SDValue &Base, SDValue &OffImm) {
1347   // Match simple R - imm8 operands.
1348   if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB &&
1349       !CurDAG->isBaseWithConstantOffset(N))
1350     return false;
1351 
1352   if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1353     int RHSC = (int)RHS->getSExtValue();
1354     if (N.getOpcode() == ISD::SUB)
1355       RHSC = -RHSC;
1356 
1357     if ((RHSC >= -255) && (RHSC < 0)) { // 8 bits (always negative)
1358       Base = N.getOperand(0);
1359       if (Base.getOpcode() == ISD::FrameIndex) {
1360         int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1361         Base = CurDAG->getTargetFrameIndex(
1362             FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1363       }
1364       OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32);
1365       return true;
1366     }
1367   }
1368 
1369   return false;
1370 }
1371 
1372 bool ARMDAGToDAGISel::SelectT2AddrModeImm8Offset(SDNode *Op, SDValue N,
1373                                                  SDValue &OffImm){
1374   unsigned Opcode = Op->getOpcode();
1375   ISD::MemIndexedMode AM = (Opcode == ISD::LOAD)
1376     ? cast<LoadSDNode>(Op)->getAddressingMode()
1377     : cast<StoreSDNode>(Op)->getAddressingMode();
1378   int RHSC;
1379   if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x100, RHSC)) { // 8 bits.
1380     OffImm = ((AM == ISD::PRE_INC) || (AM == ISD::POST_INC))
1381       ? CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32)
1382       : CurDAG->getTargetConstant(-RHSC, SDLoc(N), MVT::i32);
1383     return true;
1384   }
1385 
1386   return false;
1387 }
1388 
1389 bool ARMDAGToDAGISel::SelectT2AddrModeSoReg(SDValue N,
1390                                             SDValue &Base,
1391                                             SDValue &OffReg, SDValue &ShImm) {
1392   // (R - imm8) should be handled by t2LDRi8. The rest are handled by t2LDRi12.
1393   if (N.getOpcode() != ISD::ADD && !CurDAG->isBaseWithConstantOffset(N))
1394     return false;
1395 
1396   // Leave (R + imm12) for t2LDRi12, (R - imm8) for t2LDRi8.
1397   if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1398     int RHSC = (int)RHS->getZExtValue();
1399     if (RHSC >= 0 && RHSC < 0x1000) // 12 bits (unsigned)
1400       return false;
1401     else if (RHSC < 0 && RHSC >= -255) // 8 bits
1402       return false;
1403   }
1404 
1405   // Look for (R + R) or (R + (R << [1,2,3])).
1406   unsigned ShAmt = 0;
1407   Base   = N.getOperand(0);
1408   OffReg = N.getOperand(1);
1409 
1410   // Swap if it is ((R << c) + R).
1411   ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(OffReg.getOpcode());
1412   if (ShOpcVal != ARM_AM::lsl) {
1413     ShOpcVal = ARM_AM::getShiftOpcForNode(Base.getOpcode());
1414     if (ShOpcVal == ARM_AM::lsl)
1415       std::swap(Base, OffReg);
1416   }
1417 
1418   if (ShOpcVal == ARM_AM::lsl) {
1419     // Check to see if the RHS of the shift is a constant, if not, we can't fold
1420     // it.
1421     if (ConstantSDNode *Sh = dyn_cast<ConstantSDNode>(OffReg.getOperand(1))) {
1422       ShAmt = Sh->getZExtValue();
1423       if (ShAmt < 4 && isShifterOpProfitable(OffReg, ShOpcVal, ShAmt))
1424         OffReg = OffReg.getOperand(0);
1425       else {
1426         ShAmt = 0;
1427       }
1428     }
1429   }
1430 
1431   // If OffReg is a multiply-by-constant and it's profitable to extract a shift
1432   // and use it in a shifted operand do so.
1433   if (OffReg.getOpcode() == ISD::MUL && N.hasOneUse()) {
1434     unsigned PowerOfTwo = 0;
1435     SDValue NewMulConst;
1436     if (canExtractShiftFromMul(OffReg, 3, PowerOfTwo, NewMulConst)) {
1437       replaceDAGValue(OffReg.getOperand(1), NewMulConst);
1438       ShAmt = PowerOfTwo;
1439     }
1440   }
1441 
1442   ShImm = CurDAG->getTargetConstant(ShAmt, SDLoc(N), MVT::i32);
1443 
1444   return true;
1445 }
1446 
1447 bool ARMDAGToDAGISel::SelectT2AddrModeExclusive(SDValue N, SDValue &Base,
1448                                                 SDValue &OffImm) {
1449   // This *must* succeed since it's used for the irreplaceable ldrex and strex
1450   // instructions.
1451   Base = N;
1452   OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32);
1453 
1454   if (N.getOpcode() != ISD::ADD || !CurDAG->isBaseWithConstantOffset(N))
1455     return true;
1456 
1457   ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1));
1458   if (!RHS)
1459     return true;
1460 
1461   uint32_t RHSC = (int)RHS->getZExtValue();
1462   if (RHSC > 1020 || RHSC % 4 != 0)
1463     return true;
1464 
1465   Base = N.getOperand(0);
1466   if (Base.getOpcode() == ISD::FrameIndex) {
1467     int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1468     Base = CurDAG->getTargetFrameIndex(
1469         FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1470   }
1471 
1472   OffImm = CurDAG->getTargetConstant(RHSC/4, SDLoc(N), MVT::i32);
1473   return true;
1474 }
1475 
1476 //===--------------------------------------------------------------------===//
1477 
1478 /// getAL - Returns a ARMCC::AL immediate node.
1479 static inline SDValue getAL(SelectionDAG *CurDAG, SDLoc dl) {
1480   return CurDAG->getTargetConstant((uint64_t)ARMCC::AL, dl, MVT::i32);
1481 }
1482 
1483 SDNode *ARMDAGToDAGISel::SelectARMIndexedLoad(SDNode *N) {
1484   LoadSDNode *LD = cast<LoadSDNode>(N);
1485   ISD::MemIndexedMode AM = LD->getAddressingMode();
1486   if (AM == ISD::UNINDEXED)
1487     return nullptr;
1488 
1489   EVT LoadedVT = LD->getMemoryVT();
1490   SDValue Offset, AMOpc;
1491   bool isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC);
1492   unsigned Opcode = 0;
1493   bool Match = false;
1494   if (LoadedVT == MVT::i32 && isPre &&
1495       SelectAddrMode2OffsetImmPre(N, LD->getOffset(), Offset, AMOpc)) {
1496     Opcode = ARM::LDR_PRE_IMM;
1497     Match = true;
1498   } else if (LoadedVT == MVT::i32 && !isPre &&
1499       SelectAddrMode2OffsetImm(N, LD->getOffset(), Offset, AMOpc)) {
1500     Opcode = ARM::LDR_POST_IMM;
1501     Match = true;
1502   } else if (LoadedVT == MVT::i32 &&
1503       SelectAddrMode2OffsetReg(N, LD->getOffset(), Offset, AMOpc)) {
1504     Opcode = isPre ? ARM::LDR_PRE_REG : ARM::LDR_POST_REG;
1505     Match = true;
1506 
1507   } else if (LoadedVT == MVT::i16 &&
1508              SelectAddrMode3Offset(N, LD->getOffset(), Offset, AMOpc)) {
1509     Match = true;
1510     Opcode = (LD->getExtensionType() == ISD::SEXTLOAD)
1511       ? (isPre ? ARM::LDRSH_PRE : ARM::LDRSH_POST)
1512       : (isPre ? ARM::LDRH_PRE : ARM::LDRH_POST);
1513   } else if (LoadedVT == MVT::i8 || LoadedVT == MVT::i1) {
1514     if (LD->getExtensionType() == ISD::SEXTLOAD) {
1515       if (SelectAddrMode3Offset(N, LD->getOffset(), Offset, AMOpc)) {
1516         Match = true;
1517         Opcode = isPre ? ARM::LDRSB_PRE : ARM::LDRSB_POST;
1518       }
1519     } else {
1520       if (isPre &&
1521           SelectAddrMode2OffsetImmPre(N, LD->getOffset(), Offset, AMOpc)) {
1522         Match = true;
1523         Opcode = ARM::LDRB_PRE_IMM;
1524       } else if (!isPre &&
1525                   SelectAddrMode2OffsetImm(N, LD->getOffset(), Offset, AMOpc)) {
1526         Match = true;
1527         Opcode = ARM::LDRB_POST_IMM;
1528       } else if (SelectAddrMode2OffsetReg(N, LD->getOffset(), Offset, AMOpc)) {
1529         Match = true;
1530         Opcode = isPre ? ARM::LDRB_PRE_REG : ARM::LDRB_POST_REG;
1531       }
1532     }
1533   }
1534 
1535   if (Match) {
1536     if (Opcode == ARM::LDR_PRE_IMM || Opcode == ARM::LDRB_PRE_IMM) {
1537       SDValue Chain = LD->getChain();
1538       SDValue Base = LD->getBasePtr();
1539       SDValue Ops[]= { Base, AMOpc, getAL(CurDAG, SDLoc(N)),
1540                        CurDAG->getRegister(0, MVT::i32), Chain };
1541       return CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32,
1542                                     MVT::i32, MVT::Other, Ops);
1543     } else {
1544       SDValue Chain = LD->getChain();
1545       SDValue Base = LD->getBasePtr();
1546       SDValue Ops[]= { Base, Offset, AMOpc, getAL(CurDAG, SDLoc(N)),
1547                        CurDAG->getRegister(0, MVT::i32), Chain };
1548       return CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32,
1549                                     MVT::i32, MVT::Other, Ops);
1550     }
1551   }
1552 
1553   return nullptr;
1554 }
1555 
1556 SDNode *ARMDAGToDAGISel::SelectT2IndexedLoad(SDNode *N) {
1557   LoadSDNode *LD = cast<LoadSDNode>(N);
1558   ISD::MemIndexedMode AM = LD->getAddressingMode();
1559   if (AM == ISD::UNINDEXED)
1560     return nullptr;
1561 
1562   EVT LoadedVT = LD->getMemoryVT();
1563   bool isSExtLd = LD->getExtensionType() == ISD::SEXTLOAD;
1564   SDValue Offset;
1565   bool isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC);
1566   unsigned Opcode = 0;
1567   bool Match = false;
1568   if (SelectT2AddrModeImm8Offset(N, LD->getOffset(), Offset)) {
1569     switch (LoadedVT.getSimpleVT().SimpleTy) {
1570     case MVT::i32:
1571       Opcode = isPre ? ARM::t2LDR_PRE : ARM::t2LDR_POST;
1572       break;
1573     case MVT::i16:
1574       if (isSExtLd)
1575         Opcode = isPre ? ARM::t2LDRSH_PRE : ARM::t2LDRSH_POST;
1576       else
1577         Opcode = isPre ? ARM::t2LDRH_PRE : ARM::t2LDRH_POST;
1578       break;
1579     case MVT::i8:
1580     case MVT::i1:
1581       if (isSExtLd)
1582         Opcode = isPre ? ARM::t2LDRSB_PRE : ARM::t2LDRSB_POST;
1583       else
1584         Opcode = isPre ? ARM::t2LDRB_PRE : ARM::t2LDRB_POST;
1585       break;
1586     default:
1587       return nullptr;
1588     }
1589     Match = true;
1590   }
1591 
1592   if (Match) {
1593     SDValue Chain = LD->getChain();
1594     SDValue Base = LD->getBasePtr();
1595     SDValue Ops[]= { Base, Offset, getAL(CurDAG, SDLoc(N)),
1596                      CurDAG->getRegister(0, MVT::i32), Chain };
1597     return CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, MVT::i32,
1598                                   MVT::Other, Ops);
1599   }
1600 
1601   return nullptr;
1602 }
1603 
1604 /// \brief Form a GPRPair pseudo register from a pair of GPR regs.
1605 SDNode *ARMDAGToDAGISel::createGPRPairNode(EVT VT, SDValue V0, SDValue V1) {
1606   SDLoc dl(V0.getNode());
1607   SDValue RegClass =
1608     CurDAG->getTargetConstant(ARM::GPRPairRegClassID, dl, MVT::i32);
1609   SDValue SubReg0 = CurDAG->getTargetConstant(ARM::gsub_0, dl, MVT::i32);
1610   SDValue SubReg1 = CurDAG->getTargetConstant(ARM::gsub_1, dl, MVT::i32);
1611   const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 };
1612   return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops);
1613 }
1614 
1615 /// \brief Form a D register from a pair of S registers.
1616 SDNode *ARMDAGToDAGISel::createSRegPairNode(EVT VT, SDValue V0, SDValue V1) {
1617   SDLoc dl(V0.getNode());
1618   SDValue RegClass =
1619     CurDAG->getTargetConstant(ARM::DPR_VFP2RegClassID, dl, MVT::i32);
1620   SDValue SubReg0 = CurDAG->getTargetConstant(ARM::ssub_0, dl, MVT::i32);
1621   SDValue SubReg1 = CurDAG->getTargetConstant(ARM::ssub_1, dl, MVT::i32);
1622   const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 };
1623   return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops);
1624 }
1625 
1626 /// \brief Form a quad register from a pair of D registers.
1627 SDNode *ARMDAGToDAGISel::createDRegPairNode(EVT VT, SDValue V0, SDValue V1) {
1628   SDLoc dl(V0.getNode());
1629   SDValue RegClass = CurDAG->getTargetConstant(ARM::QPRRegClassID, dl,
1630                                                MVT::i32);
1631   SDValue SubReg0 = CurDAG->getTargetConstant(ARM::dsub_0, dl, MVT::i32);
1632   SDValue SubReg1 = CurDAG->getTargetConstant(ARM::dsub_1, dl, MVT::i32);
1633   const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 };
1634   return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops);
1635 }
1636 
1637 /// \brief Form 4 consecutive D registers from a pair of Q registers.
1638 SDNode *ARMDAGToDAGISel::createQRegPairNode(EVT VT, SDValue V0, SDValue V1) {
1639   SDLoc dl(V0.getNode());
1640   SDValue RegClass = CurDAG->getTargetConstant(ARM::QQPRRegClassID, dl,
1641                                                MVT::i32);
1642   SDValue SubReg0 = CurDAG->getTargetConstant(ARM::qsub_0, dl, MVT::i32);
1643   SDValue SubReg1 = CurDAG->getTargetConstant(ARM::qsub_1, dl, MVT::i32);
1644   const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 };
1645   return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops);
1646 }
1647 
1648 /// \brief Form 4 consecutive S registers.
1649 SDNode *ARMDAGToDAGISel::createQuadSRegsNode(EVT VT, SDValue V0, SDValue V1,
1650                                    SDValue V2, SDValue V3) {
1651   SDLoc dl(V0.getNode());
1652   SDValue RegClass =
1653     CurDAG->getTargetConstant(ARM::QPR_VFP2RegClassID, dl, MVT::i32);
1654   SDValue SubReg0 = CurDAG->getTargetConstant(ARM::ssub_0, dl, MVT::i32);
1655   SDValue SubReg1 = CurDAG->getTargetConstant(ARM::ssub_1, dl, MVT::i32);
1656   SDValue SubReg2 = CurDAG->getTargetConstant(ARM::ssub_2, dl, MVT::i32);
1657   SDValue SubReg3 = CurDAG->getTargetConstant(ARM::ssub_3, dl, MVT::i32);
1658   const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1,
1659                                     V2, SubReg2, V3, SubReg3 };
1660   return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops);
1661 }
1662 
1663 /// \brief Form 4 consecutive D registers.
1664 SDNode *ARMDAGToDAGISel::createQuadDRegsNode(EVT VT, SDValue V0, SDValue V1,
1665                                    SDValue V2, SDValue V3) {
1666   SDLoc dl(V0.getNode());
1667   SDValue RegClass = CurDAG->getTargetConstant(ARM::QQPRRegClassID, dl,
1668                                                MVT::i32);
1669   SDValue SubReg0 = CurDAG->getTargetConstant(ARM::dsub_0, dl, MVT::i32);
1670   SDValue SubReg1 = CurDAG->getTargetConstant(ARM::dsub_1, dl, MVT::i32);
1671   SDValue SubReg2 = CurDAG->getTargetConstant(ARM::dsub_2, dl, MVT::i32);
1672   SDValue SubReg3 = CurDAG->getTargetConstant(ARM::dsub_3, dl, MVT::i32);
1673   const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1,
1674                                     V2, SubReg2, V3, SubReg3 };
1675   return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops);
1676 }
1677 
1678 /// \brief Form 4 consecutive Q registers.
1679 SDNode *ARMDAGToDAGISel::createQuadQRegsNode(EVT VT, SDValue V0, SDValue V1,
1680                                    SDValue V2, SDValue V3) {
1681   SDLoc dl(V0.getNode());
1682   SDValue RegClass = CurDAG->getTargetConstant(ARM::QQQQPRRegClassID, dl,
1683                                                MVT::i32);
1684   SDValue SubReg0 = CurDAG->getTargetConstant(ARM::qsub_0, dl, MVT::i32);
1685   SDValue SubReg1 = CurDAG->getTargetConstant(ARM::qsub_1, dl, MVT::i32);
1686   SDValue SubReg2 = CurDAG->getTargetConstant(ARM::qsub_2, dl, MVT::i32);
1687   SDValue SubReg3 = CurDAG->getTargetConstant(ARM::qsub_3, dl, MVT::i32);
1688   const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1,
1689                                     V2, SubReg2, V3, SubReg3 };
1690   return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops);
1691 }
1692 
1693 /// GetVLDSTAlign - Get the alignment (in bytes) for the alignment operand
1694 /// of a NEON VLD or VST instruction.  The supported values depend on the
1695 /// number of registers being loaded.
1696 SDValue ARMDAGToDAGISel::GetVLDSTAlign(SDValue Align, SDLoc dl,
1697                                        unsigned NumVecs, bool is64BitVector) {
1698   unsigned NumRegs = NumVecs;
1699   if (!is64BitVector && NumVecs < 3)
1700     NumRegs *= 2;
1701 
1702   unsigned Alignment = cast<ConstantSDNode>(Align)->getZExtValue();
1703   if (Alignment >= 32 && NumRegs == 4)
1704     Alignment = 32;
1705   else if (Alignment >= 16 && (NumRegs == 2 || NumRegs == 4))
1706     Alignment = 16;
1707   else if (Alignment >= 8)
1708     Alignment = 8;
1709   else
1710     Alignment = 0;
1711 
1712   return CurDAG->getTargetConstant(Alignment, dl, MVT::i32);
1713 }
1714 
1715 static bool isVLDfixed(unsigned Opc)
1716 {
1717   switch (Opc) {
1718   default: return false;
1719   case ARM::VLD1d8wb_fixed : return true;
1720   case ARM::VLD1d16wb_fixed : return true;
1721   case ARM::VLD1d64Qwb_fixed : return true;
1722   case ARM::VLD1d32wb_fixed : return true;
1723   case ARM::VLD1d64wb_fixed : return true;
1724   case ARM::VLD1d64TPseudoWB_fixed : return true;
1725   case ARM::VLD1d64QPseudoWB_fixed : return true;
1726   case ARM::VLD1q8wb_fixed : return true;
1727   case ARM::VLD1q16wb_fixed : return true;
1728   case ARM::VLD1q32wb_fixed : return true;
1729   case ARM::VLD1q64wb_fixed : return true;
1730   case ARM::VLD2d8wb_fixed : return true;
1731   case ARM::VLD2d16wb_fixed : return true;
1732   case ARM::VLD2d32wb_fixed : return true;
1733   case ARM::VLD2q8PseudoWB_fixed : return true;
1734   case ARM::VLD2q16PseudoWB_fixed : return true;
1735   case ARM::VLD2q32PseudoWB_fixed : return true;
1736   case ARM::VLD2DUPd8wb_fixed : return true;
1737   case ARM::VLD2DUPd16wb_fixed : return true;
1738   case ARM::VLD2DUPd32wb_fixed : return true;
1739   }
1740 }
1741 
1742 static bool isVSTfixed(unsigned Opc)
1743 {
1744   switch (Opc) {
1745   default: return false;
1746   case ARM::VST1d8wb_fixed : return true;
1747   case ARM::VST1d16wb_fixed : return true;
1748   case ARM::VST1d32wb_fixed : return true;
1749   case ARM::VST1d64wb_fixed : return true;
1750   case ARM::VST1q8wb_fixed : return true;
1751   case ARM::VST1q16wb_fixed : return true;
1752   case ARM::VST1q32wb_fixed : return true;
1753   case ARM::VST1q64wb_fixed : return true;
1754   case ARM::VST1d64TPseudoWB_fixed : return true;
1755   case ARM::VST1d64QPseudoWB_fixed : return true;
1756   case ARM::VST2d8wb_fixed : return true;
1757   case ARM::VST2d16wb_fixed : return true;
1758   case ARM::VST2d32wb_fixed : return true;
1759   case ARM::VST2q8PseudoWB_fixed : return true;
1760   case ARM::VST2q16PseudoWB_fixed : return true;
1761   case ARM::VST2q32PseudoWB_fixed : return true;
1762   }
1763 }
1764 
1765 // Get the register stride update opcode of a VLD/VST instruction that
1766 // is otherwise equivalent to the given fixed stride updating instruction.
1767 static unsigned getVLDSTRegisterUpdateOpcode(unsigned Opc) {
1768   assert((isVLDfixed(Opc) || isVSTfixed(Opc))
1769     && "Incorrect fixed stride updating instruction.");
1770   switch (Opc) {
1771   default: break;
1772   case ARM::VLD1d8wb_fixed: return ARM::VLD1d8wb_register;
1773   case ARM::VLD1d16wb_fixed: return ARM::VLD1d16wb_register;
1774   case ARM::VLD1d32wb_fixed: return ARM::VLD1d32wb_register;
1775   case ARM::VLD1d64wb_fixed: return ARM::VLD1d64wb_register;
1776   case ARM::VLD1q8wb_fixed: return ARM::VLD1q8wb_register;
1777   case ARM::VLD1q16wb_fixed: return ARM::VLD1q16wb_register;
1778   case ARM::VLD1q32wb_fixed: return ARM::VLD1q32wb_register;
1779   case ARM::VLD1q64wb_fixed: return ARM::VLD1q64wb_register;
1780   case ARM::VLD1d64Twb_fixed: return ARM::VLD1d64Twb_register;
1781   case ARM::VLD1d64Qwb_fixed: return ARM::VLD1d64Qwb_register;
1782   case ARM::VLD1d64TPseudoWB_fixed: return ARM::VLD1d64TPseudoWB_register;
1783   case ARM::VLD1d64QPseudoWB_fixed: return ARM::VLD1d64QPseudoWB_register;
1784 
1785   case ARM::VST1d8wb_fixed: return ARM::VST1d8wb_register;
1786   case ARM::VST1d16wb_fixed: return ARM::VST1d16wb_register;
1787   case ARM::VST1d32wb_fixed: return ARM::VST1d32wb_register;
1788   case ARM::VST1d64wb_fixed: return ARM::VST1d64wb_register;
1789   case ARM::VST1q8wb_fixed: return ARM::VST1q8wb_register;
1790   case ARM::VST1q16wb_fixed: return ARM::VST1q16wb_register;
1791   case ARM::VST1q32wb_fixed: return ARM::VST1q32wb_register;
1792   case ARM::VST1q64wb_fixed: return ARM::VST1q64wb_register;
1793   case ARM::VST1d64TPseudoWB_fixed: return ARM::VST1d64TPseudoWB_register;
1794   case ARM::VST1d64QPseudoWB_fixed: return ARM::VST1d64QPseudoWB_register;
1795 
1796   case ARM::VLD2d8wb_fixed: return ARM::VLD2d8wb_register;
1797   case ARM::VLD2d16wb_fixed: return ARM::VLD2d16wb_register;
1798   case ARM::VLD2d32wb_fixed: return ARM::VLD2d32wb_register;
1799   case ARM::VLD2q8PseudoWB_fixed: return ARM::VLD2q8PseudoWB_register;
1800   case ARM::VLD2q16PseudoWB_fixed: return ARM::VLD2q16PseudoWB_register;
1801   case ARM::VLD2q32PseudoWB_fixed: return ARM::VLD2q32PseudoWB_register;
1802 
1803   case ARM::VST2d8wb_fixed: return ARM::VST2d8wb_register;
1804   case ARM::VST2d16wb_fixed: return ARM::VST2d16wb_register;
1805   case ARM::VST2d32wb_fixed: return ARM::VST2d32wb_register;
1806   case ARM::VST2q8PseudoWB_fixed: return ARM::VST2q8PseudoWB_register;
1807   case ARM::VST2q16PseudoWB_fixed: return ARM::VST2q16PseudoWB_register;
1808   case ARM::VST2q32PseudoWB_fixed: return ARM::VST2q32PseudoWB_register;
1809 
1810   case ARM::VLD2DUPd8wb_fixed: return ARM::VLD2DUPd8wb_register;
1811   case ARM::VLD2DUPd16wb_fixed: return ARM::VLD2DUPd16wb_register;
1812   case ARM::VLD2DUPd32wb_fixed: return ARM::VLD2DUPd32wb_register;
1813   }
1814   return Opc; // If not one we handle, return it unchanged.
1815 }
1816 
1817 SDNode *ARMDAGToDAGISel::SelectVLD(SDNode *N, bool isUpdating, unsigned NumVecs,
1818                                    const uint16_t *DOpcodes,
1819                                    const uint16_t *QOpcodes0,
1820                                    const uint16_t *QOpcodes1) {
1821   assert(NumVecs >= 1 && NumVecs <= 4 && "VLD NumVecs out-of-range");
1822   SDLoc dl(N);
1823 
1824   SDValue MemAddr, Align;
1825   unsigned AddrOpIdx = isUpdating ? 1 : 2;
1826   if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align))
1827     return nullptr;
1828 
1829   SDValue Chain = N->getOperand(0);
1830   EVT VT = N->getValueType(0);
1831   bool is64BitVector = VT.is64BitVector();
1832   Align = GetVLDSTAlign(Align, dl, NumVecs, is64BitVector);
1833 
1834   unsigned OpcodeIndex;
1835   switch (VT.getSimpleVT().SimpleTy) {
1836   default: llvm_unreachable("unhandled vld type");
1837     // Double-register operations:
1838   case MVT::v8i8:  OpcodeIndex = 0; break;
1839   case MVT::v4i16: OpcodeIndex = 1; break;
1840   case MVT::v2f32:
1841   case MVT::v2i32: OpcodeIndex = 2; break;
1842   case MVT::v1i64: OpcodeIndex = 3; break;
1843     // Quad-register operations:
1844   case MVT::v16i8: OpcodeIndex = 0; break;
1845   case MVT::v8i16: OpcodeIndex = 1; break;
1846   case MVT::v4f32:
1847   case MVT::v4i32: OpcodeIndex = 2; break;
1848   case MVT::v2f64:
1849   case MVT::v2i64: OpcodeIndex = 3;
1850     assert(NumVecs == 1 && "v2i64 type only supported for VLD1");
1851     break;
1852   }
1853 
1854   EVT ResTy;
1855   if (NumVecs == 1)
1856     ResTy = VT;
1857   else {
1858     unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs;
1859     if (!is64BitVector)
1860       ResTyElts *= 2;
1861     ResTy = EVT::getVectorVT(*CurDAG->getContext(), MVT::i64, ResTyElts);
1862   }
1863   std::vector<EVT> ResTys;
1864   ResTys.push_back(ResTy);
1865   if (isUpdating)
1866     ResTys.push_back(MVT::i32);
1867   ResTys.push_back(MVT::Other);
1868 
1869   SDValue Pred = getAL(CurDAG, dl);
1870   SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
1871   SDNode *VLd;
1872   SmallVector<SDValue, 7> Ops;
1873 
1874   // Double registers and VLD1/VLD2 quad registers are directly supported.
1875   if (is64BitVector || NumVecs <= 2) {
1876     unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] :
1877                     QOpcodes0[OpcodeIndex]);
1878     Ops.push_back(MemAddr);
1879     Ops.push_back(Align);
1880     if (isUpdating) {
1881       SDValue Inc = N->getOperand(AddrOpIdx + 1);
1882       // FIXME: VLD1/VLD2 fixed increment doesn't need Reg0. Remove the reg0
1883       // case entirely when the rest are updated to that form, too.
1884       if ((NumVecs <= 2) && !isa<ConstantSDNode>(Inc.getNode()))
1885         Opc = getVLDSTRegisterUpdateOpcode(Opc);
1886       // FIXME: We use a VLD1 for v1i64 even if the pseudo says vld2/3/4, so
1887       // check for that explicitly too. Horribly hacky, but temporary.
1888       if ((NumVecs > 2 && !isVLDfixed(Opc)) ||
1889           !isa<ConstantSDNode>(Inc.getNode()))
1890         Ops.push_back(isa<ConstantSDNode>(Inc.getNode()) ? Reg0 : Inc);
1891     }
1892     Ops.push_back(Pred);
1893     Ops.push_back(Reg0);
1894     Ops.push_back(Chain);
1895     VLd = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
1896 
1897   } else {
1898     // Otherwise, quad registers are loaded with two separate instructions,
1899     // where one loads the even registers and the other loads the odd registers.
1900     EVT AddrTy = MemAddr.getValueType();
1901 
1902     // Load the even subregs.  This is always an updating load, so that it
1903     // provides the address to the second load for the odd subregs.
1904     SDValue ImplDef =
1905       SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, ResTy), 0);
1906     const SDValue OpsA[] = { MemAddr, Align, Reg0, ImplDef, Pred, Reg0, Chain };
1907     SDNode *VLdA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], dl,
1908                                           ResTy, AddrTy, MVT::Other, OpsA);
1909     Chain = SDValue(VLdA, 2);
1910 
1911     // Load the odd subregs.
1912     Ops.push_back(SDValue(VLdA, 1));
1913     Ops.push_back(Align);
1914     if (isUpdating) {
1915       SDValue Inc = N->getOperand(AddrOpIdx + 1);
1916       assert(isa<ConstantSDNode>(Inc.getNode()) &&
1917              "only constant post-increment update allowed for VLD3/4");
1918       (void)Inc;
1919       Ops.push_back(Reg0);
1920     }
1921     Ops.push_back(SDValue(VLdA, 0));
1922     Ops.push_back(Pred);
1923     Ops.push_back(Reg0);
1924     Ops.push_back(Chain);
1925     VLd = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys, Ops);
1926   }
1927 
1928   // Transfer memoperands.
1929   MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1);
1930   MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand();
1931   cast<MachineSDNode>(VLd)->setMemRefs(MemOp, MemOp + 1);
1932 
1933   if (NumVecs == 1)
1934     return VLd;
1935 
1936   // Extract out the subregisters.
1937   SDValue SuperReg = SDValue(VLd, 0);
1938   assert(ARM::dsub_7 == ARM::dsub_0+7 &&
1939          ARM::qsub_3 == ARM::qsub_0+3 && "Unexpected subreg numbering");
1940   unsigned Sub0 = (is64BitVector ? ARM::dsub_0 : ARM::qsub_0);
1941   for (unsigned Vec = 0; Vec < NumVecs; ++Vec)
1942     ReplaceUses(SDValue(N, Vec),
1943                 CurDAG->getTargetExtractSubreg(Sub0 + Vec, dl, VT, SuperReg));
1944   ReplaceUses(SDValue(N, NumVecs), SDValue(VLd, 1));
1945   if (isUpdating)
1946     ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLd, 2));
1947   return nullptr;
1948 }
1949 
1950 SDNode *ARMDAGToDAGISel::SelectVST(SDNode *N, bool isUpdating, unsigned NumVecs,
1951                                    const uint16_t *DOpcodes,
1952                                    const uint16_t *QOpcodes0,
1953                                    const uint16_t *QOpcodes1) {
1954   assert(NumVecs >= 1 && NumVecs <= 4 && "VST NumVecs out-of-range");
1955   SDLoc dl(N);
1956 
1957   SDValue MemAddr, Align;
1958   unsigned AddrOpIdx = isUpdating ? 1 : 2;
1959   unsigned Vec0Idx = 3; // AddrOpIdx + (isUpdating ? 2 : 1)
1960   if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align))
1961     return nullptr;
1962 
1963   MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1);
1964   MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand();
1965 
1966   SDValue Chain = N->getOperand(0);
1967   EVT VT = N->getOperand(Vec0Idx).getValueType();
1968   bool is64BitVector = VT.is64BitVector();
1969   Align = GetVLDSTAlign(Align, dl, NumVecs, is64BitVector);
1970 
1971   unsigned OpcodeIndex;
1972   switch (VT.getSimpleVT().SimpleTy) {
1973   default: llvm_unreachable("unhandled vst type");
1974     // Double-register operations:
1975   case MVT::v8i8:  OpcodeIndex = 0; break;
1976   case MVT::v4i16: OpcodeIndex = 1; break;
1977   case MVT::v2f32:
1978   case MVT::v2i32: OpcodeIndex = 2; break;
1979   case MVT::v1i64: OpcodeIndex = 3; break;
1980     // Quad-register operations:
1981   case MVT::v16i8: OpcodeIndex = 0; break;
1982   case MVT::v8i16: OpcodeIndex = 1; break;
1983   case MVT::v4f32:
1984   case MVT::v4i32: OpcodeIndex = 2; break;
1985   case MVT::v2f64:
1986   case MVT::v2i64: OpcodeIndex = 3;
1987     assert(NumVecs == 1 && "v2i64 type only supported for VST1");
1988     break;
1989   }
1990 
1991   std::vector<EVT> ResTys;
1992   if (isUpdating)
1993     ResTys.push_back(MVT::i32);
1994   ResTys.push_back(MVT::Other);
1995 
1996   SDValue Pred = getAL(CurDAG, dl);
1997   SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
1998   SmallVector<SDValue, 7> Ops;
1999 
2000   // Double registers and VST1/VST2 quad registers are directly supported.
2001   if (is64BitVector || NumVecs <= 2) {
2002     SDValue SrcReg;
2003     if (NumVecs == 1) {
2004       SrcReg = N->getOperand(Vec0Idx);
2005     } else if (is64BitVector) {
2006       // Form a REG_SEQUENCE to force register allocation.
2007       SDValue V0 = N->getOperand(Vec0Idx + 0);
2008       SDValue V1 = N->getOperand(Vec0Idx + 1);
2009       if (NumVecs == 2)
2010         SrcReg = SDValue(createDRegPairNode(MVT::v2i64, V0, V1), 0);
2011       else {
2012         SDValue V2 = N->getOperand(Vec0Idx + 2);
2013         // If it's a vst3, form a quad D-register and leave the last part as
2014         // an undef.
2015         SDValue V3 = (NumVecs == 3)
2016           ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,dl,VT), 0)
2017           : N->getOperand(Vec0Idx + 3);
2018         SrcReg = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0);
2019       }
2020     } else {
2021       // Form a QQ register.
2022       SDValue Q0 = N->getOperand(Vec0Idx);
2023       SDValue Q1 = N->getOperand(Vec0Idx + 1);
2024       SrcReg = SDValue(createQRegPairNode(MVT::v4i64, Q0, Q1), 0);
2025     }
2026 
2027     unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] :
2028                     QOpcodes0[OpcodeIndex]);
2029     Ops.push_back(MemAddr);
2030     Ops.push_back(Align);
2031     if (isUpdating) {
2032       SDValue Inc = N->getOperand(AddrOpIdx + 1);
2033       // FIXME: VST1/VST2 fixed increment doesn't need Reg0. Remove the reg0
2034       // case entirely when the rest are updated to that form, too.
2035       if (NumVecs <= 2 && !isa<ConstantSDNode>(Inc.getNode()))
2036         Opc = getVLDSTRegisterUpdateOpcode(Opc);
2037       // FIXME: We use a VST1 for v1i64 even if the pseudo says vld2/3/4, so
2038       // check for that explicitly too. Horribly hacky, but temporary.
2039       if  (!isa<ConstantSDNode>(Inc.getNode()))
2040         Ops.push_back(Inc);
2041       else if (NumVecs > 2 && !isVSTfixed(Opc))
2042         Ops.push_back(Reg0);
2043     }
2044     Ops.push_back(SrcReg);
2045     Ops.push_back(Pred);
2046     Ops.push_back(Reg0);
2047     Ops.push_back(Chain);
2048     SDNode *VSt = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2049 
2050     // Transfer memoperands.
2051     cast<MachineSDNode>(VSt)->setMemRefs(MemOp, MemOp + 1);
2052 
2053     return VSt;
2054   }
2055 
2056   // Otherwise, quad registers are stored with two separate instructions,
2057   // where one stores the even registers and the other stores the odd registers.
2058 
2059   // Form the QQQQ REG_SEQUENCE.
2060   SDValue V0 = N->getOperand(Vec0Idx + 0);
2061   SDValue V1 = N->getOperand(Vec0Idx + 1);
2062   SDValue V2 = N->getOperand(Vec0Idx + 2);
2063   SDValue V3 = (NumVecs == 3)
2064     ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0)
2065     : N->getOperand(Vec0Idx + 3);
2066   SDValue RegSeq = SDValue(createQuadQRegsNode(MVT::v8i64, V0, V1, V2, V3), 0);
2067 
2068   // Store the even D registers.  This is always an updating store, so that it
2069   // provides the address to the second store for the odd subregs.
2070   const SDValue OpsA[] = { MemAddr, Align, Reg0, RegSeq, Pred, Reg0, Chain };
2071   SDNode *VStA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], dl,
2072                                         MemAddr.getValueType(),
2073                                         MVT::Other, OpsA);
2074   cast<MachineSDNode>(VStA)->setMemRefs(MemOp, MemOp + 1);
2075   Chain = SDValue(VStA, 1);
2076 
2077   // Store the odd D registers.
2078   Ops.push_back(SDValue(VStA, 0));
2079   Ops.push_back(Align);
2080   if (isUpdating) {
2081     SDValue Inc = N->getOperand(AddrOpIdx + 1);
2082     assert(isa<ConstantSDNode>(Inc.getNode()) &&
2083            "only constant post-increment update allowed for VST3/4");
2084     (void)Inc;
2085     Ops.push_back(Reg0);
2086   }
2087   Ops.push_back(RegSeq);
2088   Ops.push_back(Pred);
2089   Ops.push_back(Reg0);
2090   Ops.push_back(Chain);
2091   SDNode *VStB = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys,
2092                                         Ops);
2093   cast<MachineSDNode>(VStB)->setMemRefs(MemOp, MemOp + 1);
2094   return VStB;
2095 }
2096 
2097 SDNode *ARMDAGToDAGISel::SelectVLDSTLane(SDNode *N, bool IsLoad,
2098                                          bool isUpdating, unsigned NumVecs,
2099                                          const uint16_t *DOpcodes,
2100                                          const uint16_t *QOpcodes) {
2101   assert(NumVecs >=2 && NumVecs <= 4 && "VLDSTLane NumVecs out-of-range");
2102   SDLoc dl(N);
2103 
2104   SDValue MemAddr, Align;
2105   unsigned AddrOpIdx = isUpdating ? 1 : 2;
2106   unsigned Vec0Idx = 3; // AddrOpIdx + (isUpdating ? 2 : 1)
2107   if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align))
2108     return nullptr;
2109 
2110   MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1);
2111   MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2112 
2113   SDValue Chain = N->getOperand(0);
2114   unsigned Lane =
2115     cast<ConstantSDNode>(N->getOperand(Vec0Idx + NumVecs))->getZExtValue();
2116   EVT VT = N->getOperand(Vec0Idx).getValueType();
2117   bool is64BitVector = VT.is64BitVector();
2118 
2119   unsigned Alignment = 0;
2120   if (NumVecs != 3) {
2121     Alignment = cast<ConstantSDNode>(Align)->getZExtValue();
2122     unsigned NumBytes = NumVecs * VT.getVectorElementType().getSizeInBits()/8;
2123     if (Alignment > NumBytes)
2124       Alignment = NumBytes;
2125     if (Alignment < 8 && Alignment < NumBytes)
2126       Alignment = 0;
2127     // Alignment must be a power of two; make sure of that.
2128     Alignment = (Alignment & -Alignment);
2129     if (Alignment == 1)
2130       Alignment = 0;
2131   }
2132   Align = CurDAG->getTargetConstant(Alignment, dl, MVT::i32);
2133 
2134   unsigned OpcodeIndex;
2135   switch (VT.getSimpleVT().SimpleTy) {
2136   default: llvm_unreachable("unhandled vld/vst lane type");
2137     // Double-register operations:
2138   case MVT::v8i8:  OpcodeIndex = 0; break;
2139   case MVT::v4i16: OpcodeIndex = 1; break;
2140   case MVT::v2f32:
2141   case MVT::v2i32: OpcodeIndex = 2; break;
2142     // Quad-register operations:
2143   case MVT::v8i16: OpcodeIndex = 0; break;
2144   case MVT::v4f32:
2145   case MVT::v4i32: OpcodeIndex = 1; break;
2146   }
2147 
2148   std::vector<EVT> ResTys;
2149   if (IsLoad) {
2150     unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs;
2151     if (!is64BitVector)
2152       ResTyElts *= 2;
2153     ResTys.push_back(EVT::getVectorVT(*CurDAG->getContext(),
2154                                       MVT::i64, ResTyElts));
2155   }
2156   if (isUpdating)
2157     ResTys.push_back(MVT::i32);
2158   ResTys.push_back(MVT::Other);
2159 
2160   SDValue Pred = getAL(CurDAG, dl);
2161   SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
2162 
2163   SmallVector<SDValue, 8> Ops;
2164   Ops.push_back(MemAddr);
2165   Ops.push_back(Align);
2166   if (isUpdating) {
2167     SDValue Inc = N->getOperand(AddrOpIdx + 1);
2168     Ops.push_back(isa<ConstantSDNode>(Inc.getNode()) ? Reg0 : Inc);
2169   }
2170 
2171   SDValue SuperReg;
2172   SDValue V0 = N->getOperand(Vec0Idx + 0);
2173   SDValue V1 = N->getOperand(Vec0Idx + 1);
2174   if (NumVecs == 2) {
2175     if (is64BitVector)
2176       SuperReg = SDValue(createDRegPairNode(MVT::v2i64, V0, V1), 0);
2177     else
2178       SuperReg = SDValue(createQRegPairNode(MVT::v4i64, V0, V1), 0);
2179   } else {
2180     SDValue V2 = N->getOperand(Vec0Idx + 2);
2181     SDValue V3 = (NumVecs == 3)
2182       ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0)
2183       : N->getOperand(Vec0Idx + 3);
2184     if (is64BitVector)
2185       SuperReg = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0);
2186     else
2187       SuperReg = SDValue(createQuadQRegsNode(MVT::v8i64, V0, V1, V2, V3), 0);
2188   }
2189   Ops.push_back(SuperReg);
2190   Ops.push_back(getI32Imm(Lane, dl));
2191   Ops.push_back(Pred);
2192   Ops.push_back(Reg0);
2193   Ops.push_back(Chain);
2194 
2195   unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] :
2196                                   QOpcodes[OpcodeIndex]);
2197   SDNode *VLdLn = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2198   cast<MachineSDNode>(VLdLn)->setMemRefs(MemOp, MemOp + 1);
2199   if (!IsLoad)
2200     return VLdLn;
2201 
2202   // Extract the subregisters.
2203   SuperReg = SDValue(VLdLn, 0);
2204   assert(ARM::dsub_7 == ARM::dsub_0+7 &&
2205          ARM::qsub_3 == ARM::qsub_0+3 && "Unexpected subreg numbering");
2206   unsigned Sub0 = is64BitVector ? ARM::dsub_0 : ARM::qsub_0;
2207   for (unsigned Vec = 0; Vec < NumVecs; ++Vec)
2208     ReplaceUses(SDValue(N, Vec),
2209                 CurDAG->getTargetExtractSubreg(Sub0 + Vec, dl, VT, SuperReg));
2210   ReplaceUses(SDValue(N, NumVecs), SDValue(VLdLn, 1));
2211   if (isUpdating)
2212     ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLdLn, 2));
2213   return nullptr;
2214 }
2215 
2216 SDNode *ARMDAGToDAGISel::SelectVLDDup(SDNode *N, bool isUpdating,
2217                                       unsigned NumVecs,
2218                                       const uint16_t *Opcodes) {
2219   assert(NumVecs >=2 && NumVecs <= 4 && "VLDDup NumVecs out-of-range");
2220   SDLoc dl(N);
2221 
2222   SDValue MemAddr, Align;
2223   if (!SelectAddrMode6(N, N->getOperand(1), MemAddr, Align))
2224     return nullptr;
2225 
2226   MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1);
2227   MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2228 
2229   SDValue Chain = N->getOperand(0);
2230   EVT VT = N->getValueType(0);
2231 
2232   unsigned Alignment = 0;
2233   if (NumVecs != 3) {
2234     Alignment = cast<ConstantSDNode>(Align)->getZExtValue();
2235     unsigned NumBytes = NumVecs * VT.getVectorElementType().getSizeInBits()/8;
2236     if (Alignment > NumBytes)
2237       Alignment = NumBytes;
2238     if (Alignment < 8 && Alignment < NumBytes)
2239       Alignment = 0;
2240     // Alignment must be a power of two; make sure of that.
2241     Alignment = (Alignment & -Alignment);
2242     if (Alignment == 1)
2243       Alignment = 0;
2244   }
2245   Align = CurDAG->getTargetConstant(Alignment, dl, MVT::i32);
2246 
2247   unsigned OpcodeIndex;
2248   switch (VT.getSimpleVT().SimpleTy) {
2249   default: llvm_unreachable("unhandled vld-dup type");
2250   case MVT::v8i8:  OpcodeIndex = 0; break;
2251   case MVT::v4i16: OpcodeIndex = 1; break;
2252   case MVT::v2f32:
2253   case MVT::v2i32: OpcodeIndex = 2; break;
2254   }
2255 
2256   SDValue Pred = getAL(CurDAG, dl);
2257   SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
2258   SDValue SuperReg;
2259   unsigned Opc = Opcodes[OpcodeIndex];
2260   SmallVector<SDValue, 6> Ops;
2261   Ops.push_back(MemAddr);
2262   Ops.push_back(Align);
2263   if (isUpdating) {
2264     // fixed-stride update instructions don't have an explicit writeback
2265     // operand. It's implicit in the opcode itself.
2266     SDValue Inc = N->getOperand(2);
2267     if (!isa<ConstantSDNode>(Inc.getNode()))
2268       Ops.push_back(Inc);
2269     // FIXME: VLD3 and VLD4 haven't been updated to that form yet.
2270     else if (NumVecs > 2)
2271       Ops.push_back(Reg0);
2272   }
2273   Ops.push_back(Pred);
2274   Ops.push_back(Reg0);
2275   Ops.push_back(Chain);
2276 
2277   unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs;
2278   std::vector<EVT> ResTys;
2279   ResTys.push_back(EVT::getVectorVT(*CurDAG->getContext(), MVT::i64,ResTyElts));
2280   if (isUpdating)
2281     ResTys.push_back(MVT::i32);
2282   ResTys.push_back(MVT::Other);
2283   SDNode *VLdDup = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2284   cast<MachineSDNode>(VLdDup)->setMemRefs(MemOp, MemOp + 1);
2285   SuperReg = SDValue(VLdDup, 0);
2286 
2287   // Extract the subregisters.
2288   assert(ARM::dsub_7 == ARM::dsub_0+7 && "Unexpected subreg numbering");
2289   unsigned SubIdx = ARM::dsub_0;
2290   for (unsigned Vec = 0; Vec < NumVecs; ++Vec)
2291     ReplaceUses(SDValue(N, Vec),
2292                 CurDAG->getTargetExtractSubreg(SubIdx+Vec, dl, VT, SuperReg));
2293   ReplaceUses(SDValue(N, NumVecs), SDValue(VLdDup, 1));
2294   if (isUpdating)
2295     ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLdDup, 2));
2296   return nullptr;
2297 }
2298 
2299 SDNode *ARMDAGToDAGISel::SelectVTBL(SDNode *N, bool IsExt, unsigned NumVecs,
2300                                     unsigned Opc) {
2301   assert(NumVecs >= 2 && NumVecs <= 4 && "VTBL NumVecs out-of-range");
2302   SDLoc dl(N);
2303   EVT VT = N->getValueType(0);
2304   unsigned FirstTblReg = IsExt ? 2 : 1;
2305 
2306   // Form a REG_SEQUENCE to force register allocation.
2307   SDValue RegSeq;
2308   SDValue V0 = N->getOperand(FirstTblReg + 0);
2309   SDValue V1 = N->getOperand(FirstTblReg + 1);
2310   if (NumVecs == 2)
2311     RegSeq = SDValue(createDRegPairNode(MVT::v16i8, V0, V1), 0);
2312   else {
2313     SDValue V2 = N->getOperand(FirstTblReg + 2);
2314     // If it's a vtbl3, form a quad D-register and leave the last part as
2315     // an undef.
2316     SDValue V3 = (NumVecs == 3)
2317       ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0)
2318       : N->getOperand(FirstTblReg + 3);
2319     RegSeq = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0);
2320   }
2321 
2322   SmallVector<SDValue, 6> Ops;
2323   if (IsExt)
2324     Ops.push_back(N->getOperand(1));
2325   Ops.push_back(RegSeq);
2326   Ops.push_back(N->getOperand(FirstTblReg + NumVecs));
2327   Ops.push_back(getAL(CurDAG, dl)); // predicate
2328   Ops.push_back(CurDAG->getRegister(0, MVT::i32)); // predicate register
2329   return CurDAG->getMachineNode(Opc, dl, VT, Ops);
2330 }
2331 
2332 SDNode *ARMDAGToDAGISel::SelectV6T2BitfieldExtractOp(SDNode *N,
2333                                                      bool isSigned) {
2334   if (!Subtarget->hasV6T2Ops())
2335     return nullptr;
2336 
2337   unsigned Opc = isSigned
2338     ? (Subtarget->isThumb() ? ARM::t2SBFX : ARM::SBFX)
2339     : (Subtarget->isThumb() ? ARM::t2UBFX : ARM::UBFX);
2340   SDLoc dl(N);
2341 
2342   // For unsigned extracts, check for a shift right and mask
2343   unsigned And_imm = 0;
2344   if (N->getOpcode() == ISD::AND) {
2345     if (isOpcWithIntImmediate(N, ISD::AND, And_imm)) {
2346 
2347       // The immediate is a mask of the low bits iff imm & (imm+1) == 0
2348       if (And_imm & (And_imm + 1))
2349         return nullptr;
2350 
2351       unsigned Srl_imm = 0;
2352       if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRL,
2353                                 Srl_imm)) {
2354         assert(Srl_imm > 0 && Srl_imm < 32 && "bad amount in shift node!");
2355 
2356         // Note: The width operand is encoded as width-1.
2357         unsigned Width = countTrailingOnes(And_imm) - 1;
2358         unsigned LSB = Srl_imm;
2359 
2360         SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
2361 
2362         if ((LSB + Width + 1) == N->getValueType(0).getSizeInBits()) {
2363           // It's cheaper to use a right shift to extract the top bits.
2364           if (Subtarget->isThumb()) {
2365             Opc = isSigned ? ARM::t2ASRri : ARM::t2LSRri;
2366             SDValue Ops[] = { N->getOperand(0).getOperand(0),
2367                               CurDAG->getTargetConstant(LSB, dl, MVT::i32),
2368                               getAL(CurDAG, dl), Reg0, Reg0 };
2369             return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops);
2370           }
2371 
2372           // ARM models shift instructions as MOVsi with shifter operand.
2373           ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(ISD::SRL);
2374           SDValue ShOpc =
2375             CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, LSB), dl,
2376                                       MVT::i32);
2377           SDValue Ops[] = { N->getOperand(0).getOperand(0), ShOpc,
2378                             getAL(CurDAG, dl), Reg0, Reg0 };
2379           return CurDAG->SelectNodeTo(N, ARM::MOVsi, MVT::i32, Ops);
2380         }
2381 
2382         SDValue Ops[] = { N->getOperand(0).getOperand(0),
2383                           CurDAG->getTargetConstant(LSB, dl, MVT::i32),
2384                           CurDAG->getTargetConstant(Width, dl, MVT::i32),
2385                           getAL(CurDAG, dl), Reg0 };
2386         return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops);
2387       }
2388     }
2389     return nullptr;
2390   }
2391 
2392   // Otherwise, we're looking for a shift of a shift
2393   unsigned Shl_imm = 0;
2394   if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SHL, Shl_imm)) {
2395     assert(Shl_imm > 0 && Shl_imm < 32 && "bad amount in shift node!");
2396     unsigned Srl_imm = 0;
2397     if (isInt32Immediate(N->getOperand(1), Srl_imm)) {
2398       assert(Srl_imm > 0 && Srl_imm < 32 && "bad amount in shift node!");
2399       // Note: The width operand is encoded as width-1.
2400       unsigned Width = 32 - Srl_imm - 1;
2401       int LSB = Srl_imm - Shl_imm;
2402       if (LSB < 0)
2403         return nullptr;
2404       SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
2405       SDValue Ops[] = { N->getOperand(0).getOperand(0),
2406                         CurDAG->getTargetConstant(LSB, dl, MVT::i32),
2407                         CurDAG->getTargetConstant(Width, dl, MVT::i32),
2408                         getAL(CurDAG, dl), Reg0 };
2409       return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops);
2410     }
2411   }
2412 
2413   if (N->getOpcode() == ISD::SIGN_EXTEND_INREG) {
2414     unsigned Width = cast<VTSDNode>(N->getOperand(1))->getVT().getSizeInBits();
2415     unsigned LSB = 0;
2416     if (!isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRL, LSB) &&
2417         !isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRA, LSB))
2418       return nullptr;
2419 
2420     if (LSB + Width > 32)
2421       return nullptr;
2422 
2423     SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
2424     SDValue Ops[] = { N->getOperand(0).getOperand(0),
2425                       CurDAG->getTargetConstant(LSB, dl, MVT::i32),
2426                       CurDAG->getTargetConstant(Width - 1, dl, MVT::i32),
2427                       getAL(CurDAG, dl), Reg0 };
2428     return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops);
2429   }
2430 
2431   return nullptr;
2432 }
2433 
2434 /// Target-specific DAG combining for ISD::XOR.
2435 /// Target-independent combining lowers SELECT_CC nodes of the form
2436 /// select_cc setg[ge] X,  0,  X, -X
2437 /// select_cc setgt    X, -1,  X, -X
2438 /// select_cc setl[te] X,  0, -X,  X
2439 /// select_cc setlt    X,  1, -X,  X
2440 /// which represent Integer ABS into:
2441 /// Y = sra (X, size(X)-1); xor (add (X, Y), Y)
2442 /// ARM instruction selection detects the latter and matches it to
2443 /// ARM::ABS or ARM::t2ABS machine node.
2444 SDNode *ARMDAGToDAGISel::SelectABSOp(SDNode *N){
2445   SDValue XORSrc0 = N->getOperand(0);
2446   SDValue XORSrc1 = N->getOperand(1);
2447   EVT VT = N->getValueType(0);
2448 
2449   if (Subtarget->isThumb1Only())
2450     return nullptr;
2451 
2452   if (XORSrc0.getOpcode() != ISD::ADD || XORSrc1.getOpcode() != ISD::SRA)
2453     return nullptr;
2454 
2455   SDValue ADDSrc0 = XORSrc0.getOperand(0);
2456   SDValue ADDSrc1 = XORSrc0.getOperand(1);
2457   SDValue SRASrc0 = XORSrc1.getOperand(0);
2458   SDValue SRASrc1 = XORSrc1.getOperand(1);
2459   ConstantSDNode *SRAConstant =  dyn_cast<ConstantSDNode>(SRASrc1);
2460   EVT XType = SRASrc0.getValueType();
2461   unsigned Size = XType.getSizeInBits() - 1;
2462 
2463   if (ADDSrc1 == XORSrc1 && ADDSrc0 == SRASrc0 &&
2464       XType.isInteger() && SRAConstant != nullptr &&
2465       Size == SRAConstant->getZExtValue()) {
2466     unsigned Opcode = Subtarget->isThumb2() ? ARM::t2ABS : ARM::ABS;
2467     return CurDAG->SelectNodeTo(N, Opcode, VT, ADDSrc0);
2468   }
2469 
2470   return nullptr;
2471 }
2472 
2473 static bool SearchSignedMulShort(SDValue SignExt, unsigned *Opc, SDValue &Src1,
2474                                  bool Accumulate) {
2475   // For SM*WB, we need to some form of sext.
2476   // For SM*WT, we need to search for (sra X, 16)
2477   // Src1 then gets set to X.
2478   if ((SignExt.getOpcode() == ISD::SIGN_EXTEND ||
2479        SignExt.getOpcode() == ISD::SIGN_EXTEND_INREG ||
2480        SignExt.getOpcode() == ISD::AssertSext) &&
2481        SignExt.getValueType() == MVT::i32) {
2482 
2483     *Opc = Accumulate ? ARM::SMLAWB : ARM::SMULWB;
2484     Src1 = SignExt.getOperand(0);
2485     return true;
2486   }
2487 
2488   if (SignExt.getOpcode() != ISD::SRA)
2489     return false;
2490 
2491   ConstantSDNode *SRASrc1 = dyn_cast<ConstantSDNode>(SignExt.getOperand(1));
2492   if (!SRASrc1 || SRASrc1->getZExtValue() != 16)
2493     return false;
2494 
2495   SDValue Op0 = SignExt.getOperand(0);
2496 
2497   // The sign extend operand for SM*WB could be generated by a shl and ashr.
2498   if (Op0.getOpcode() == ISD::SHL) {
2499     SDValue SHL = Op0;
2500     ConstantSDNode *SHLSrc1 = dyn_cast<ConstantSDNode>(SHL.getOperand(1));
2501     if (!SHLSrc1 || SHLSrc1->getZExtValue() != 16)
2502       return false;
2503 
2504     *Opc = Accumulate ? ARM::SMLAWB : ARM::SMULWB;
2505     Src1 = Op0.getOperand(0);
2506     return true;
2507   }
2508   *Opc = Accumulate ? ARM::SMLAWT : ARM::SMULWT;
2509   Src1 = SignExt.getOperand(0);
2510   return true;
2511 }
2512 
2513 static bool SearchSignedMulLong(SDValue OR, unsigned *Opc, SDValue &Src0,
2514                                 SDValue &Src1, bool Accumulate) {
2515   // First we look for:
2516   // (add (or (srl ?, 16), (shl ?, 16)))
2517   if (OR.getOpcode() != ISD::OR)
2518     return false;
2519 
2520   SDValue SRL = OR.getOperand(0);
2521   SDValue SHL = OR.getOperand(1);
2522 
2523   if (SRL.getOpcode() != ISD::SRL || SHL.getOpcode() != ISD::SHL) {
2524     SRL = OR.getOperand(1);
2525     SHL = OR.getOperand(0);
2526     if (SRL.getOpcode() != ISD::SRL || SHL.getOpcode() != ISD::SHL)
2527       return false;
2528   }
2529 
2530   ConstantSDNode *SRLSrc1 = dyn_cast<ConstantSDNode>(SRL.getOperand(1));
2531   ConstantSDNode *SHLSrc1 = dyn_cast<ConstantSDNode>(SHL.getOperand(1));
2532   if (!SRLSrc1 || !SHLSrc1 || SRLSrc1->getZExtValue() != 16 ||
2533       SHLSrc1->getZExtValue() != 16)
2534     return false;
2535 
2536   // The first operands to the shifts need to be the two results from the
2537   // same smul_lohi node.
2538   if ((SRL.getOperand(0).getNode() != SHL.getOperand(0).getNode()) ||
2539        SRL.getOperand(0).getOpcode() != ISD::SMUL_LOHI)
2540     return false;
2541 
2542   SDNode *SMULLOHI = SRL.getOperand(0).getNode();
2543   if (SRL.getOperand(0) != SDValue(SMULLOHI, 0) ||
2544       SHL.getOperand(0) != SDValue(SMULLOHI, 1))
2545     return false;
2546 
2547   // Now we have:
2548   // (add (or (srl (smul_lohi ?, ?), 16), (shl (smul_lohi ?, ?), 16)))
2549   // For SMLAW[B|T] smul_lohi will take a 32-bit and a 16-bit arguments.
2550   // For SMLAWB the 16-bit value will signed extended somehow.
2551   // For SMLAWT only the SRA is required.
2552 
2553   // Check both sides of SMUL_LOHI
2554   if (SearchSignedMulShort(SMULLOHI->getOperand(0), Opc, Src1, Accumulate)) {
2555     Src0 = SMULLOHI->getOperand(1);
2556   } else if (SearchSignedMulShort(SMULLOHI->getOperand(1), Opc, Src1,
2557                                   Accumulate)) {
2558     Src0 = SMULLOHI->getOperand(0);
2559   } else {
2560     return false;
2561   }
2562   return true;
2563 }
2564 
2565 SDNode *ARMDAGToDAGISel::SelectSMLAWSMULW(SDNode *N) {
2566   SDLoc dl(N);
2567   SDValue Src0 = N->getOperand(0);
2568   SDValue Src1 = N->getOperand(1);
2569   SDValue A, B;
2570   unsigned Opc = 0;
2571 
2572   if (N->getOpcode() == ISD::ADD) {
2573     if (Src0.getOpcode() != ISD::OR && Src1.getOpcode() != ISD::OR)
2574       return nullptr;
2575 
2576     SDValue Acc;
2577     if (SearchSignedMulLong(Src0, &Opc, A, B, true)) {
2578       Acc = Src1;
2579     } else if (SearchSignedMulLong(Src1, &Opc, A, B, true)) {
2580       Acc = Src0;
2581     } else {
2582       return nullptr;
2583     }
2584     if (Opc == 0)
2585       return nullptr;
2586 
2587     SDValue Ops[] = { A, B, Acc, getAL(CurDAG, dl),
2588                       CurDAG->getRegister(0, MVT::i32) };
2589     return CurDAG->SelectNodeTo(N, Opc, MVT::i32, MVT::Other, Ops);
2590   } else if (N->getOpcode() == ISD::OR &&
2591              SearchSignedMulLong(SDValue(N, 0), &Opc, A, B, false)) {
2592     if (Opc == 0)
2593       return nullptr;
2594 
2595     SDValue Ops[] = { A, B, getAL(CurDAG, dl),
2596                       CurDAG->getRegister(0, MVT::i32)};
2597     return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops);
2598   }
2599   return nullptr;
2600 }
2601 
2602 /// We've got special pseudo-instructions for these
2603 SDNode *ARMDAGToDAGISel::SelectCMP_SWAP(SDNode *N) {
2604   unsigned Opcode;
2605   EVT MemTy = cast<MemSDNode>(N)->getMemoryVT();
2606   if (MemTy == MVT::i8)
2607     Opcode = ARM::CMP_SWAP_8;
2608   else if (MemTy == MVT::i16)
2609     Opcode = ARM::CMP_SWAP_16;
2610   else if (MemTy == MVT::i32)
2611     Opcode = ARM::CMP_SWAP_32;
2612   else
2613     llvm_unreachable("Unknown AtomicCmpSwap type");
2614 
2615   SDValue Ops[] = {N->getOperand(1), N->getOperand(2), N->getOperand(3),
2616                    N->getOperand(0)};
2617   SDNode *CmpSwap = CurDAG->getMachineNode(
2618       Opcode, SDLoc(N),
2619       CurDAG->getVTList(MVT::i32, MVT::i32, MVT::Other), Ops);
2620 
2621   MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1);
2622   MemOp[0] = cast<MemSDNode>(N)->getMemOperand();
2623   cast<MachineSDNode>(CmpSwap)->setMemRefs(MemOp, MemOp + 1);
2624 
2625   ReplaceUses(SDValue(N, 0), SDValue(CmpSwap, 0));
2626   ReplaceUses(SDValue(N, 1), SDValue(CmpSwap, 2));
2627   return nullptr;
2628 }
2629 
2630 SDNode *ARMDAGToDAGISel::SelectConcatVector(SDNode *N) {
2631   // The only time a CONCAT_VECTORS operation can have legal types is when
2632   // two 64-bit vectors are concatenated to a 128-bit vector.
2633   EVT VT = N->getValueType(0);
2634   if (!VT.is128BitVector() || N->getNumOperands() != 2)
2635     llvm_unreachable("unexpected CONCAT_VECTORS");
2636   return createDRegPairNode(VT, N->getOperand(0), N->getOperand(1));
2637 }
2638 
2639 SDNode *ARMDAGToDAGISel::Select(SDNode *N) {
2640   SDLoc dl(N);
2641 
2642   if (N->isMachineOpcode()) {
2643     N->setNodeId(-1);
2644     return nullptr;   // Already selected.
2645   }
2646 
2647   switch (N->getOpcode()) {
2648   default: break;
2649   case ISD::ADD:
2650   case ISD::OR: {
2651     SDNode *ResNode = SelectSMLAWSMULW(N);
2652     if (ResNode)
2653       return ResNode;
2654     break;
2655   }
2656   case ISD::WRITE_REGISTER: {
2657     SDNode *ResNode = SelectWriteRegister(N);
2658     if (ResNode)
2659       return ResNode;
2660     break;
2661   }
2662   case ISD::READ_REGISTER: {
2663     SDNode *ResNode = SelectReadRegister(N);
2664     if (ResNode)
2665       return ResNode;
2666     break;
2667   }
2668   case ISD::INLINEASM: {
2669     SDNode *ResNode = SelectInlineAsm(N);
2670     if (ResNode)
2671       return ResNode;
2672     break;
2673   }
2674   case ISD::XOR: {
2675     // Select special operations if XOR node forms integer ABS pattern
2676     SDNode *ResNode = SelectABSOp(N);
2677     if (ResNode)
2678       return ResNode;
2679     // Other cases are autogenerated.
2680     break;
2681   }
2682   case ISD::Constant: {
2683     unsigned Val = cast<ConstantSDNode>(N)->getZExtValue();
2684     // If we can't materialize the constant we need to use a literal pool
2685     if (ConstantMaterializationCost(Val) > 2) {
2686       SDValue CPIdx = CurDAG->getTargetConstantPool(
2687           ConstantInt::get(Type::getInt32Ty(*CurDAG->getContext()), Val),
2688           TLI->getPointerTy(CurDAG->getDataLayout()));
2689 
2690       SDNode *ResNode;
2691       if (Subtarget->isThumb()) {
2692         SDValue Pred = getAL(CurDAG, dl);
2693         SDValue PredReg = CurDAG->getRegister(0, MVT::i32);
2694         SDValue Ops[] = { CPIdx, Pred, PredReg, CurDAG->getEntryNode() };
2695         ResNode = CurDAG->getMachineNode(ARM::tLDRpci, dl, MVT::i32, MVT::Other,
2696                                          Ops);
2697       } else {
2698         SDValue Ops[] = {
2699           CPIdx,
2700           CurDAG->getTargetConstant(0, dl, MVT::i32),
2701           getAL(CurDAG, dl),
2702           CurDAG->getRegister(0, MVT::i32),
2703           CurDAG->getEntryNode()
2704         };
2705         ResNode=CurDAG->getMachineNode(ARM::LDRcp, dl, MVT::i32, MVT::Other,
2706                                        Ops);
2707       }
2708       ReplaceUses(SDValue(N, 0), SDValue(ResNode, 0));
2709       return nullptr;
2710     }
2711 
2712     // Other cases are autogenerated.
2713     break;
2714   }
2715   case ISD::FrameIndex: {
2716     // Selects to ADDri FI, 0 which in turn will become ADDri SP, imm.
2717     int FI = cast<FrameIndexSDNode>(N)->getIndex();
2718     SDValue TFI = CurDAG->getTargetFrameIndex(
2719         FI, TLI->getPointerTy(CurDAG->getDataLayout()));
2720     if (Subtarget->isThumb1Only()) {
2721       // Set the alignment of the frame object to 4, to avoid having to generate
2722       // more than one ADD
2723       MachineFrameInfo *MFI = MF->getFrameInfo();
2724       if (MFI->getObjectAlignment(FI) < 4)
2725         MFI->setObjectAlignment(FI, 4);
2726       return CurDAG->SelectNodeTo(N, ARM::tADDframe, MVT::i32, TFI,
2727                                   CurDAG->getTargetConstant(0, dl, MVT::i32));
2728     } else {
2729       unsigned Opc = ((Subtarget->isThumb() && Subtarget->hasThumb2()) ?
2730                       ARM::t2ADDri : ARM::ADDri);
2731       SDValue Ops[] = { TFI, CurDAG->getTargetConstant(0, dl, MVT::i32),
2732                         getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32),
2733                         CurDAG->getRegister(0, MVT::i32) };
2734       return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops);
2735     }
2736   }
2737   case ISD::SRL:
2738     if (SDNode *I = SelectV6T2BitfieldExtractOp(N, false))
2739       return I;
2740     break;
2741   case ISD::SIGN_EXTEND_INREG:
2742   case ISD::SRA:
2743     if (SDNode *I = SelectV6T2BitfieldExtractOp(N, true))
2744       return I;
2745     break;
2746   case ISD::MUL:
2747     if (Subtarget->isThumb1Only())
2748       break;
2749     if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(1))) {
2750       unsigned RHSV = C->getZExtValue();
2751       if (!RHSV) break;
2752       if (isPowerOf2_32(RHSV-1)) {  // 2^n+1?
2753         unsigned ShImm = Log2_32(RHSV-1);
2754         if (ShImm >= 32)
2755           break;
2756         SDValue V = N->getOperand(0);
2757         ShImm = ARM_AM::getSORegOpc(ARM_AM::lsl, ShImm);
2758         SDValue ShImmOp = CurDAG->getTargetConstant(ShImm, dl, MVT::i32);
2759         SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
2760         if (Subtarget->isThumb()) {
2761           SDValue Ops[] = { V, V, ShImmOp, getAL(CurDAG, dl), Reg0, Reg0 };
2762           return CurDAG->SelectNodeTo(N, ARM::t2ADDrs, MVT::i32, Ops);
2763         } else {
2764           SDValue Ops[] = { V, V, Reg0, ShImmOp, getAL(CurDAG, dl), Reg0,
2765                             Reg0 };
2766           return CurDAG->SelectNodeTo(N, ARM::ADDrsi, MVT::i32, Ops);
2767         }
2768       }
2769       if (isPowerOf2_32(RHSV+1)) {  // 2^n-1?
2770         unsigned ShImm = Log2_32(RHSV+1);
2771         if (ShImm >= 32)
2772           break;
2773         SDValue V = N->getOperand(0);
2774         ShImm = ARM_AM::getSORegOpc(ARM_AM::lsl, ShImm);
2775         SDValue ShImmOp = CurDAG->getTargetConstant(ShImm, dl, MVT::i32);
2776         SDValue Reg0 = CurDAG->getRegister(0, MVT::i32);
2777         if (Subtarget->isThumb()) {
2778           SDValue Ops[] = { V, V, ShImmOp, getAL(CurDAG, dl), Reg0, Reg0 };
2779           return CurDAG->SelectNodeTo(N, ARM::t2RSBrs, MVT::i32, Ops);
2780         } else {
2781           SDValue Ops[] = { V, V, Reg0, ShImmOp, getAL(CurDAG, dl), Reg0,
2782                             Reg0 };
2783           return CurDAG->SelectNodeTo(N, ARM::RSBrsi, MVT::i32, Ops);
2784         }
2785       }
2786     }
2787     break;
2788   case ISD::AND: {
2789     // Check for unsigned bitfield extract
2790     if (SDNode *I = SelectV6T2BitfieldExtractOp(N, false))
2791       return I;
2792 
2793     // (and (or x, c2), c1) and top 16-bits of c1 and c2 match, lower 16-bits
2794     // of c1 are 0xffff, and lower 16-bit of c2 are 0. That is, the top 16-bits
2795     // are entirely contributed by c2 and lower 16-bits are entirely contributed
2796     // by x. That's equal to (or (and x, 0xffff), (and c1, 0xffff0000)).
2797     // Select it to: "movt x, ((c1 & 0xffff) >> 16)
2798     EVT VT = N->getValueType(0);
2799     if (VT != MVT::i32)
2800       break;
2801     unsigned Opc = (Subtarget->isThumb() && Subtarget->hasThumb2())
2802       ? ARM::t2MOVTi16
2803       : (Subtarget->hasV6T2Ops() ? ARM::MOVTi16 : 0);
2804     if (!Opc)
2805       break;
2806     SDValue N0 = N->getOperand(0), N1 = N->getOperand(1);
2807     ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
2808     if (!N1C)
2809       break;
2810     if (N0.getOpcode() == ISD::OR && N0.getNode()->hasOneUse()) {
2811       SDValue N2 = N0.getOperand(1);
2812       ConstantSDNode *N2C = dyn_cast<ConstantSDNode>(N2);
2813       if (!N2C)
2814         break;
2815       unsigned N1CVal = N1C->getZExtValue();
2816       unsigned N2CVal = N2C->getZExtValue();
2817       if ((N1CVal & 0xffff0000U) == (N2CVal & 0xffff0000U) &&
2818           (N1CVal & 0xffffU) == 0xffffU &&
2819           (N2CVal & 0xffffU) == 0x0U) {
2820         SDValue Imm16 = CurDAG->getTargetConstant((N2CVal & 0xFFFF0000U) >> 16,
2821                                                   dl, MVT::i32);
2822         SDValue Ops[] = { N0.getOperand(0), Imm16,
2823                           getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32) };
2824         return CurDAG->getMachineNode(Opc, dl, VT, Ops);
2825       }
2826     }
2827     break;
2828   }
2829   case ARMISD::VMOVRRD:
2830     return CurDAG->getMachineNode(ARM::VMOVRRD, dl, MVT::i32, MVT::i32,
2831                                   N->getOperand(0), getAL(CurDAG, dl),
2832                                   CurDAG->getRegister(0, MVT::i32));
2833   case ISD::UMUL_LOHI: {
2834     if (Subtarget->isThumb1Only())
2835       break;
2836     if (Subtarget->isThumb()) {
2837       SDValue Ops[] = { N->getOperand(0), N->getOperand(1),
2838                         getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32) };
2839       return CurDAG->getMachineNode(ARM::t2UMULL, dl, MVT::i32, MVT::i32, Ops);
2840     } else {
2841       SDValue Ops[] = { N->getOperand(0), N->getOperand(1),
2842                         getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32),
2843                         CurDAG->getRegister(0, MVT::i32) };
2844       return CurDAG->getMachineNode(Subtarget->hasV6Ops() ?
2845                                     ARM::UMULL : ARM::UMULLv5,
2846                                     dl, MVT::i32, MVT::i32, Ops);
2847     }
2848   }
2849   case ISD::SMUL_LOHI: {
2850     if (Subtarget->isThumb1Only())
2851       break;
2852     if (Subtarget->isThumb()) {
2853       SDValue Ops[] = { N->getOperand(0), N->getOperand(1),
2854                         getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32) };
2855       return CurDAG->getMachineNode(ARM::t2SMULL, dl, MVT::i32, MVT::i32, Ops);
2856     } else {
2857       SDValue Ops[] = { N->getOperand(0), N->getOperand(1),
2858                         getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32),
2859                         CurDAG->getRegister(0, MVT::i32) };
2860       return CurDAG->getMachineNode(Subtarget->hasV6Ops() ?
2861                                     ARM::SMULL : ARM::SMULLv5,
2862                                     dl, MVT::i32, MVT::i32, Ops);
2863     }
2864   }
2865   case ARMISD::UMLAL:{
2866     if (Subtarget->isThumb()) {
2867       SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2),
2868                         N->getOperand(3), getAL(CurDAG, dl),
2869                         CurDAG->getRegister(0, MVT::i32)};
2870       return CurDAG->getMachineNode(ARM::t2UMLAL, dl, MVT::i32, MVT::i32, Ops);
2871     }else{
2872       SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2),
2873                         N->getOperand(3), getAL(CurDAG, dl),
2874                         CurDAG->getRegister(0, MVT::i32),
2875                         CurDAG->getRegister(0, MVT::i32) };
2876       return CurDAG->getMachineNode(Subtarget->hasV6Ops() ?
2877                                       ARM::UMLAL : ARM::UMLALv5,
2878                                       dl, MVT::i32, MVT::i32, Ops);
2879     }
2880   }
2881   case ARMISD::SMLAL:{
2882     if (Subtarget->isThumb()) {
2883       SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2),
2884                         N->getOperand(3), getAL(CurDAG, dl),
2885                         CurDAG->getRegister(0, MVT::i32)};
2886       return CurDAG->getMachineNode(ARM::t2SMLAL, dl, MVT::i32, MVT::i32, Ops);
2887     }else{
2888       SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2),
2889                         N->getOperand(3), getAL(CurDAG, dl),
2890                         CurDAG->getRegister(0, MVT::i32),
2891                         CurDAG->getRegister(0, MVT::i32) };
2892       return CurDAG->getMachineNode(Subtarget->hasV6Ops() ?
2893                                       ARM::SMLAL : ARM::SMLALv5,
2894                                       dl, MVT::i32, MVT::i32, Ops);
2895     }
2896   }
2897   case ISD::LOAD: {
2898     SDNode *ResNode = nullptr;
2899     if (Subtarget->isThumb() && Subtarget->hasThumb2())
2900       ResNode = SelectT2IndexedLoad(N);
2901     else
2902       ResNode = SelectARMIndexedLoad(N);
2903     if (ResNode)
2904       return ResNode;
2905     // Other cases are autogenerated.
2906     break;
2907   }
2908   case ARMISD::BRCOND: {
2909     // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc)
2910     // Emits: (Bcc:void (bb:Other):$dst, (imm:i32):$cc)
2911     // Pattern complexity = 6  cost = 1  size = 0
2912 
2913     // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc)
2914     // Emits: (tBcc:void (bb:Other):$dst, (imm:i32):$cc)
2915     // Pattern complexity = 6  cost = 1  size = 0
2916 
2917     // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc)
2918     // Emits: (t2Bcc:void (bb:Other):$dst, (imm:i32):$cc)
2919     // Pattern complexity = 6  cost = 1  size = 0
2920 
2921     unsigned Opc = Subtarget->isThumb() ?
2922       ((Subtarget->hasThumb2()) ? ARM::t2Bcc : ARM::tBcc) : ARM::Bcc;
2923     SDValue Chain = N->getOperand(0);
2924     SDValue N1 = N->getOperand(1);
2925     SDValue N2 = N->getOperand(2);
2926     SDValue N3 = N->getOperand(3);
2927     SDValue InFlag = N->getOperand(4);
2928     assert(N1.getOpcode() == ISD::BasicBlock);
2929     assert(N2.getOpcode() == ISD::Constant);
2930     assert(N3.getOpcode() == ISD::Register);
2931 
2932     SDValue Tmp2 = CurDAG->getTargetConstant(((unsigned)
2933                                cast<ConstantSDNode>(N2)->getZExtValue()), dl,
2934                                MVT::i32);
2935     SDValue Ops[] = { N1, Tmp2, N3, Chain, InFlag };
2936     SDNode *ResNode = CurDAG->getMachineNode(Opc, dl, MVT::Other,
2937                                              MVT::Glue, Ops);
2938     Chain = SDValue(ResNode, 0);
2939     if (N->getNumValues() == 2) {
2940       InFlag = SDValue(ResNode, 1);
2941       ReplaceUses(SDValue(N, 1), InFlag);
2942     }
2943     ReplaceUses(SDValue(N, 0),
2944                 SDValue(Chain.getNode(), Chain.getResNo()));
2945     return nullptr;
2946   }
2947   case ARMISD::VZIP: {
2948     unsigned Opc = 0;
2949     EVT VT = N->getValueType(0);
2950     switch (VT.getSimpleVT().SimpleTy) {
2951     default: return nullptr;
2952     case MVT::v8i8:  Opc = ARM::VZIPd8; break;
2953     case MVT::v4i16: Opc = ARM::VZIPd16; break;
2954     case MVT::v2f32:
2955     // vzip.32 Dd, Dm is a pseudo-instruction expanded to vtrn.32 Dd, Dm.
2956     case MVT::v2i32: Opc = ARM::VTRNd32; break;
2957     case MVT::v16i8: Opc = ARM::VZIPq8; break;
2958     case MVT::v8i16: Opc = ARM::VZIPq16; break;
2959     case MVT::v4f32:
2960     case MVT::v4i32: Opc = ARM::VZIPq32; break;
2961     }
2962     SDValue Pred = getAL(CurDAG, dl);
2963     SDValue PredReg = CurDAG->getRegister(0, MVT::i32);
2964     SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg };
2965     return CurDAG->getMachineNode(Opc, dl, VT, VT, Ops);
2966   }
2967   case ARMISD::VUZP: {
2968     unsigned Opc = 0;
2969     EVT VT = N->getValueType(0);
2970     switch (VT.getSimpleVT().SimpleTy) {
2971     default: return nullptr;
2972     case MVT::v8i8:  Opc = ARM::VUZPd8; break;
2973     case MVT::v4i16: Opc = ARM::VUZPd16; break;
2974     case MVT::v2f32:
2975     // vuzp.32 Dd, Dm is a pseudo-instruction expanded to vtrn.32 Dd, Dm.
2976     case MVT::v2i32: Opc = ARM::VTRNd32; break;
2977     case MVT::v16i8: Opc = ARM::VUZPq8; break;
2978     case MVT::v8i16: Opc = ARM::VUZPq16; break;
2979     case MVT::v4f32:
2980     case MVT::v4i32: Opc = ARM::VUZPq32; break;
2981     }
2982     SDValue Pred = getAL(CurDAG, dl);
2983     SDValue PredReg = CurDAG->getRegister(0, MVT::i32);
2984     SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg };
2985     return CurDAG->getMachineNode(Opc, dl, VT, VT, Ops);
2986   }
2987   case ARMISD::VTRN: {
2988     unsigned Opc = 0;
2989     EVT VT = N->getValueType(0);
2990     switch (VT.getSimpleVT().SimpleTy) {
2991     default: return nullptr;
2992     case MVT::v8i8:  Opc = ARM::VTRNd8; break;
2993     case MVT::v4i16: Opc = ARM::VTRNd16; break;
2994     case MVT::v2f32:
2995     case MVT::v2i32: Opc = ARM::VTRNd32; break;
2996     case MVT::v16i8: Opc = ARM::VTRNq8; break;
2997     case MVT::v8i16: Opc = ARM::VTRNq16; break;
2998     case MVT::v4f32:
2999     case MVT::v4i32: Opc = ARM::VTRNq32; break;
3000     }
3001     SDValue Pred = getAL(CurDAG, dl);
3002     SDValue PredReg = CurDAG->getRegister(0, MVT::i32);
3003     SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg };
3004     return CurDAG->getMachineNode(Opc, dl, VT, VT, Ops);
3005   }
3006   case ARMISD::BUILD_VECTOR: {
3007     EVT VecVT = N->getValueType(0);
3008     EVT EltVT = VecVT.getVectorElementType();
3009     unsigned NumElts = VecVT.getVectorNumElements();
3010     if (EltVT == MVT::f64) {
3011       assert(NumElts == 2 && "unexpected type for BUILD_VECTOR");
3012       return createDRegPairNode(VecVT, N->getOperand(0), N->getOperand(1));
3013     }
3014     assert(EltVT == MVT::f32 && "unexpected type for BUILD_VECTOR");
3015     if (NumElts == 2)
3016       return createSRegPairNode(VecVT, N->getOperand(0), N->getOperand(1));
3017     assert(NumElts == 4 && "unexpected type for BUILD_VECTOR");
3018     return createQuadSRegsNode(VecVT, N->getOperand(0), N->getOperand(1),
3019                      N->getOperand(2), N->getOperand(3));
3020   }
3021 
3022   case ARMISD::VLD2DUP: {
3023     static const uint16_t Opcodes[] = { ARM::VLD2DUPd8, ARM::VLD2DUPd16,
3024                                         ARM::VLD2DUPd32 };
3025     return SelectVLDDup(N, false, 2, Opcodes);
3026   }
3027 
3028   case ARMISD::VLD3DUP: {
3029     static const uint16_t Opcodes[] = { ARM::VLD3DUPd8Pseudo,
3030                                         ARM::VLD3DUPd16Pseudo,
3031                                         ARM::VLD3DUPd32Pseudo };
3032     return SelectVLDDup(N, false, 3, Opcodes);
3033   }
3034 
3035   case ARMISD::VLD4DUP: {
3036     static const uint16_t Opcodes[] = { ARM::VLD4DUPd8Pseudo,
3037                                         ARM::VLD4DUPd16Pseudo,
3038                                         ARM::VLD4DUPd32Pseudo };
3039     return SelectVLDDup(N, false, 4, Opcodes);
3040   }
3041 
3042   case ARMISD::VLD2DUP_UPD: {
3043     static const uint16_t Opcodes[] = { ARM::VLD2DUPd8wb_fixed,
3044                                         ARM::VLD2DUPd16wb_fixed,
3045                                         ARM::VLD2DUPd32wb_fixed };
3046     return SelectVLDDup(N, true, 2, Opcodes);
3047   }
3048 
3049   case ARMISD::VLD3DUP_UPD: {
3050     static const uint16_t Opcodes[] = { ARM::VLD3DUPd8Pseudo_UPD,
3051                                         ARM::VLD3DUPd16Pseudo_UPD,
3052                                         ARM::VLD3DUPd32Pseudo_UPD };
3053     return SelectVLDDup(N, true, 3, Opcodes);
3054   }
3055 
3056   case ARMISD::VLD4DUP_UPD: {
3057     static const uint16_t Opcodes[] = { ARM::VLD4DUPd8Pseudo_UPD,
3058                                         ARM::VLD4DUPd16Pseudo_UPD,
3059                                         ARM::VLD4DUPd32Pseudo_UPD };
3060     return SelectVLDDup(N, true, 4, Opcodes);
3061   }
3062 
3063   case ARMISD::VLD1_UPD: {
3064     static const uint16_t DOpcodes[] = { ARM::VLD1d8wb_fixed,
3065                                          ARM::VLD1d16wb_fixed,
3066                                          ARM::VLD1d32wb_fixed,
3067                                          ARM::VLD1d64wb_fixed };
3068     static const uint16_t QOpcodes[] = { ARM::VLD1q8wb_fixed,
3069                                          ARM::VLD1q16wb_fixed,
3070                                          ARM::VLD1q32wb_fixed,
3071                                          ARM::VLD1q64wb_fixed };
3072     return SelectVLD(N, true, 1, DOpcodes, QOpcodes, nullptr);
3073   }
3074 
3075   case ARMISD::VLD2_UPD: {
3076     static const uint16_t DOpcodes[] = { ARM::VLD2d8wb_fixed,
3077                                          ARM::VLD2d16wb_fixed,
3078                                          ARM::VLD2d32wb_fixed,
3079                                          ARM::VLD1q64wb_fixed};
3080     static const uint16_t QOpcodes[] = { ARM::VLD2q8PseudoWB_fixed,
3081                                          ARM::VLD2q16PseudoWB_fixed,
3082                                          ARM::VLD2q32PseudoWB_fixed };
3083     return SelectVLD(N, true, 2, DOpcodes, QOpcodes, nullptr);
3084   }
3085 
3086   case ARMISD::VLD3_UPD: {
3087     static const uint16_t DOpcodes[] = { ARM::VLD3d8Pseudo_UPD,
3088                                          ARM::VLD3d16Pseudo_UPD,
3089                                          ARM::VLD3d32Pseudo_UPD,
3090                                          ARM::VLD1d64TPseudoWB_fixed};
3091     static const uint16_t QOpcodes0[] = { ARM::VLD3q8Pseudo_UPD,
3092                                           ARM::VLD3q16Pseudo_UPD,
3093                                           ARM::VLD3q32Pseudo_UPD };
3094     static const uint16_t QOpcodes1[] = { ARM::VLD3q8oddPseudo_UPD,
3095                                           ARM::VLD3q16oddPseudo_UPD,
3096                                           ARM::VLD3q32oddPseudo_UPD };
3097     return SelectVLD(N, true, 3, DOpcodes, QOpcodes0, QOpcodes1);
3098   }
3099 
3100   case ARMISD::VLD4_UPD: {
3101     static const uint16_t DOpcodes[] = { ARM::VLD4d8Pseudo_UPD,
3102                                          ARM::VLD4d16Pseudo_UPD,
3103                                          ARM::VLD4d32Pseudo_UPD,
3104                                          ARM::VLD1d64QPseudoWB_fixed};
3105     static const uint16_t QOpcodes0[] = { ARM::VLD4q8Pseudo_UPD,
3106                                           ARM::VLD4q16Pseudo_UPD,
3107                                           ARM::VLD4q32Pseudo_UPD };
3108     static const uint16_t QOpcodes1[] = { ARM::VLD4q8oddPseudo_UPD,
3109                                           ARM::VLD4q16oddPseudo_UPD,
3110                                           ARM::VLD4q32oddPseudo_UPD };
3111     return SelectVLD(N, true, 4, DOpcodes, QOpcodes0, QOpcodes1);
3112   }
3113 
3114   case ARMISD::VLD2LN_UPD: {
3115     static const uint16_t DOpcodes[] = { ARM::VLD2LNd8Pseudo_UPD,
3116                                          ARM::VLD2LNd16Pseudo_UPD,
3117                                          ARM::VLD2LNd32Pseudo_UPD };
3118     static const uint16_t QOpcodes[] = { ARM::VLD2LNq16Pseudo_UPD,
3119                                          ARM::VLD2LNq32Pseudo_UPD };
3120     return SelectVLDSTLane(N, true, true, 2, DOpcodes, QOpcodes);
3121   }
3122 
3123   case ARMISD::VLD3LN_UPD: {
3124     static const uint16_t DOpcodes[] = { ARM::VLD3LNd8Pseudo_UPD,
3125                                          ARM::VLD3LNd16Pseudo_UPD,
3126                                          ARM::VLD3LNd32Pseudo_UPD };
3127     static const uint16_t QOpcodes[] = { ARM::VLD3LNq16Pseudo_UPD,
3128                                          ARM::VLD3LNq32Pseudo_UPD };
3129     return SelectVLDSTLane(N, true, true, 3, DOpcodes, QOpcodes);
3130   }
3131 
3132   case ARMISD::VLD4LN_UPD: {
3133     static const uint16_t DOpcodes[] = { ARM::VLD4LNd8Pseudo_UPD,
3134                                          ARM::VLD4LNd16Pseudo_UPD,
3135                                          ARM::VLD4LNd32Pseudo_UPD };
3136     static const uint16_t QOpcodes[] = { ARM::VLD4LNq16Pseudo_UPD,
3137                                          ARM::VLD4LNq32Pseudo_UPD };
3138     return SelectVLDSTLane(N, true, true, 4, DOpcodes, QOpcodes);
3139   }
3140 
3141   case ARMISD::VST1_UPD: {
3142     static const uint16_t DOpcodes[] = { ARM::VST1d8wb_fixed,
3143                                          ARM::VST1d16wb_fixed,
3144                                          ARM::VST1d32wb_fixed,
3145                                          ARM::VST1d64wb_fixed };
3146     static const uint16_t QOpcodes[] = { ARM::VST1q8wb_fixed,
3147                                          ARM::VST1q16wb_fixed,
3148                                          ARM::VST1q32wb_fixed,
3149                                          ARM::VST1q64wb_fixed };
3150     return SelectVST(N, true, 1, DOpcodes, QOpcodes, nullptr);
3151   }
3152 
3153   case ARMISD::VST2_UPD: {
3154     static const uint16_t DOpcodes[] = { ARM::VST2d8wb_fixed,
3155                                          ARM::VST2d16wb_fixed,
3156                                          ARM::VST2d32wb_fixed,
3157                                          ARM::VST1q64wb_fixed};
3158     static const uint16_t QOpcodes[] = { ARM::VST2q8PseudoWB_fixed,
3159                                          ARM::VST2q16PseudoWB_fixed,
3160                                          ARM::VST2q32PseudoWB_fixed };
3161     return SelectVST(N, true, 2, DOpcodes, QOpcodes, nullptr);
3162   }
3163 
3164   case ARMISD::VST3_UPD: {
3165     static const uint16_t DOpcodes[] = { ARM::VST3d8Pseudo_UPD,
3166                                          ARM::VST3d16Pseudo_UPD,
3167                                          ARM::VST3d32Pseudo_UPD,
3168                                          ARM::VST1d64TPseudoWB_fixed};
3169     static const uint16_t QOpcodes0[] = { ARM::VST3q8Pseudo_UPD,
3170                                           ARM::VST3q16Pseudo_UPD,
3171                                           ARM::VST3q32Pseudo_UPD };
3172     static const uint16_t QOpcodes1[] = { ARM::VST3q8oddPseudo_UPD,
3173                                           ARM::VST3q16oddPseudo_UPD,
3174                                           ARM::VST3q32oddPseudo_UPD };
3175     return SelectVST(N, true, 3, DOpcodes, QOpcodes0, QOpcodes1);
3176   }
3177 
3178   case ARMISD::VST4_UPD: {
3179     static const uint16_t DOpcodes[] = { ARM::VST4d8Pseudo_UPD,
3180                                          ARM::VST4d16Pseudo_UPD,
3181                                          ARM::VST4d32Pseudo_UPD,
3182                                          ARM::VST1d64QPseudoWB_fixed};
3183     static const uint16_t QOpcodes0[] = { ARM::VST4q8Pseudo_UPD,
3184                                           ARM::VST4q16Pseudo_UPD,
3185                                           ARM::VST4q32Pseudo_UPD };
3186     static const uint16_t QOpcodes1[] = { ARM::VST4q8oddPseudo_UPD,
3187                                           ARM::VST4q16oddPseudo_UPD,
3188                                           ARM::VST4q32oddPseudo_UPD };
3189     return SelectVST(N, true, 4, DOpcodes, QOpcodes0, QOpcodes1);
3190   }
3191 
3192   case ARMISD::VST2LN_UPD: {
3193     static const uint16_t DOpcodes[] = { ARM::VST2LNd8Pseudo_UPD,
3194                                          ARM::VST2LNd16Pseudo_UPD,
3195                                          ARM::VST2LNd32Pseudo_UPD };
3196     static const uint16_t QOpcodes[] = { ARM::VST2LNq16Pseudo_UPD,
3197                                          ARM::VST2LNq32Pseudo_UPD };
3198     return SelectVLDSTLane(N, false, true, 2, DOpcodes, QOpcodes);
3199   }
3200 
3201   case ARMISD::VST3LN_UPD: {
3202     static const uint16_t DOpcodes[] = { ARM::VST3LNd8Pseudo_UPD,
3203                                          ARM::VST3LNd16Pseudo_UPD,
3204                                          ARM::VST3LNd32Pseudo_UPD };
3205     static const uint16_t QOpcodes[] = { ARM::VST3LNq16Pseudo_UPD,
3206                                          ARM::VST3LNq32Pseudo_UPD };
3207     return SelectVLDSTLane(N, false, true, 3, DOpcodes, QOpcodes);
3208   }
3209 
3210   case ARMISD::VST4LN_UPD: {
3211     static const uint16_t DOpcodes[] = { ARM::VST4LNd8Pseudo_UPD,
3212                                          ARM::VST4LNd16Pseudo_UPD,
3213                                          ARM::VST4LNd32Pseudo_UPD };
3214     static const uint16_t QOpcodes[] = { ARM::VST4LNq16Pseudo_UPD,
3215                                          ARM::VST4LNq32Pseudo_UPD };
3216     return SelectVLDSTLane(N, false, true, 4, DOpcodes, QOpcodes);
3217   }
3218 
3219   case ISD::INTRINSIC_VOID:
3220   case ISD::INTRINSIC_W_CHAIN: {
3221     unsigned IntNo = cast<ConstantSDNode>(N->getOperand(1))->getZExtValue();
3222     switch (IntNo) {
3223     default:
3224       break;
3225 
3226     case Intrinsic::arm_ldaexd:
3227     case Intrinsic::arm_ldrexd: {
3228       SDLoc dl(N);
3229       SDValue Chain = N->getOperand(0);
3230       SDValue MemAddr = N->getOperand(2);
3231       bool isThumb = Subtarget->isThumb() && Subtarget->hasV8MBaselineOps();
3232 
3233       bool IsAcquire = IntNo == Intrinsic::arm_ldaexd;
3234       unsigned NewOpc = isThumb ? (IsAcquire ? ARM::t2LDAEXD : ARM::t2LDREXD)
3235                                 : (IsAcquire ? ARM::LDAEXD : ARM::LDREXD);
3236 
3237       // arm_ldrexd returns a i64 value in {i32, i32}
3238       std::vector<EVT> ResTys;
3239       if (isThumb) {
3240         ResTys.push_back(MVT::i32);
3241         ResTys.push_back(MVT::i32);
3242       } else
3243         ResTys.push_back(MVT::Untyped);
3244       ResTys.push_back(MVT::Other);
3245 
3246       // Place arguments in the right order.
3247       SmallVector<SDValue, 7> Ops;
3248       Ops.push_back(MemAddr);
3249       Ops.push_back(getAL(CurDAG, dl));
3250       Ops.push_back(CurDAG->getRegister(0, MVT::i32));
3251       Ops.push_back(Chain);
3252       SDNode *Ld = CurDAG->getMachineNode(NewOpc, dl, ResTys, Ops);
3253       // Transfer memoperands.
3254       MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1);
3255       MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand();
3256       cast<MachineSDNode>(Ld)->setMemRefs(MemOp, MemOp + 1);
3257 
3258       // Remap uses.
3259       SDValue OutChain = isThumb ? SDValue(Ld, 2) : SDValue(Ld, 1);
3260       if (!SDValue(N, 0).use_empty()) {
3261         SDValue Result;
3262         if (isThumb)
3263           Result = SDValue(Ld, 0);
3264         else {
3265           SDValue SubRegIdx =
3266             CurDAG->getTargetConstant(ARM::gsub_0, dl, MVT::i32);
3267           SDNode *ResNode = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
3268               dl, MVT::i32, SDValue(Ld, 0), SubRegIdx);
3269           Result = SDValue(ResNode,0);
3270         }
3271         ReplaceUses(SDValue(N, 0), Result);
3272       }
3273       if (!SDValue(N, 1).use_empty()) {
3274         SDValue Result;
3275         if (isThumb)
3276           Result = SDValue(Ld, 1);
3277         else {
3278           SDValue SubRegIdx =
3279             CurDAG->getTargetConstant(ARM::gsub_1, dl, MVT::i32);
3280           SDNode *ResNode = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
3281               dl, MVT::i32, SDValue(Ld, 0), SubRegIdx);
3282           Result = SDValue(ResNode,0);
3283         }
3284         ReplaceUses(SDValue(N, 1), Result);
3285       }
3286       ReplaceUses(SDValue(N, 2), OutChain);
3287       return nullptr;
3288     }
3289     case Intrinsic::arm_stlexd:
3290     case Intrinsic::arm_strexd: {
3291       SDLoc dl(N);
3292       SDValue Chain = N->getOperand(0);
3293       SDValue Val0 = N->getOperand(2);
3294       SDValue Val1 = N->getOperand(3);
3295       SDValue MemAddr = N->getOperand(4);
3296 
3297       // Store exclusive double return a i32 value which is the return status
3298       // of the issued store.
3299       const EVT ResTys[] = {MVT::i32, MVT::Other};
3300 
3301       bool isThumb = Subtarget->isThumb() && Subtarget->hasThumb2();
3302       // Place arguments in the right order.
3303       SmallVector<SDValue, 7> Ops;
3304       if (isThumb) {
3305         Ops.push_back(Val0);
3306         Ops.push_back(Val1);
3307       } else
3308         // arm_strexd uses GPRPair.
3309         Ops.push_back(SDValue(createGPRPairNode(MVT::Untyped, Val0, Val1), 0));
3310       Ops.push_back(MemAddr);
3311       Ops.push_back(getAL(CurDAG, dl));
3312       Ops.push_back(CurDAG->getRegister(0, MVT::i32));
3313       Ops.push_back(Chain);
3314 
3315       bool IsRelease = IntNo == Intrinsic::arm_stlexd;
3316       unsigned NewOpc = isThumb ? (IsRelease ? ARM::t2STLEXD : ARM::t2STREXD)
3317                                 : (IsRelease ? ARM::STLEXD : ARM::STREXD);
3318 
3319       SDNode *St = CurDAG->getMachineNode(NewOpc, dl, ResTys, Ops);
3320       // Transfer memoperands.
3321       MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1);
3322       MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand();
3323       cast<MachineSDNode>(St)->setMemRefs(MemOp, MemOp + 1);
3324 
3325       return St;
3326     }
3327 
3328     case Intrinsic::arm_neon_vld1: {
3329       static const uint16_t DOpcodes[] = { ARM::VLD1d8, ARM::VLD1d16,
3330                                            ARM::VLD1d32, ARM::VLD1d64 };
3331       static const uint16_t QOpcodes[] = { ARM::VLD1q8, ARM::VLD1q16,
3332                                            ARM::VLD1q32, ARM::VLD1q64};
3333       return SelectVLD(N, false, 1, DOpcodes, QOpcodes, nullptr);
3334     }
3335 
3336     case Intrinsic::arm_neon_vld2: {
3337       static const uint16_t DOpcodes[] = { ARM::VLD2d8, ARM::VLD2d16,
3338                                            ARM::VLD2d32, ARM::VLD1q64 };
3339       static const uint16_t QOpcodes[] = { ARM::VLD2q8Pseudo, ARM::VLD2q16Pseudo,
3340                                            ARM::VLD2q32Pseudo };
3341       return SelectVLD(N, false, 2, DOpcodes, QOpcodes, nullptr);
3342     }
3343 
3344     case Intrinsic::arm_neon_vld3: {
3345       static const uint16_t DOpcodes[] = { ARM::VLD3d8Pseudo,
3346                                            ARM::VLD3d16Pseudo,
3347                                            ARM::VLD3d32Pseudo,
3348                                            ARM::VLD1d64TPseudo };
3349       static const uint16_t QOpcodes0[] = { ARM::VLD3q8Pseudo_UPD,
3350                                             ARM::VLD3q16Pseudo_UPD,
3351                                             ARM::VLD3q32Pseudo_UPD };
3352       static const uint16_t QOpcodes1[] = { ARM::VLD3q8oddPseudo,
3353                                             ARM::VLD3q16oddPseudo,
3354                                             ARM::VLD3q32oddPseudo };
3355       return SelectVLD(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1);
3356     }
3357 
3358     case Intrinsic::arm_neon_vld4: {
3359       static const uint16_t DOpcodes[] = { ARM::VLD4d8Pseudo,
3360                                            ARM::VLD4d16Pseudo,
3361                                            ARM::VLD4d32Pseudo,
3362                                            ARM::VLD1d64QPseudo };
3363       static const uint16_t QOpcodes0[] = { ARM::VLD4q8Pseudo_UPD,
3364                                             ARM::VLD4q16Pseudo_UPD,
3365                                             ARM::VLD4q32Pseudo_UPD };
3366       static const uint16_t QOpcodes1[] = { ARM::VLD4q8oddPseudo,
3367                                             ARM::VLD4q16oddPseudo,
3368                                             ARM::VLD4q32oddPseudo };
3369       return SelectVLD(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1);
3370     }
3371 
3372     case Intrinsic::arm_neon_vld2lane: {
3373       static const uint16_t DOpcodes[] = { ARM::VLD2LNd8Pseudo,
3374                                            ARM::VLD2LNd16Pseudo,
3375                                            ARM::VLD2LNd32Pseudo };
3376       static const uint16_t QOpcodes[] = { ARM::VLD2LNq16Pseudo,
3377                                            ARM::VLD2LNq32Pseudo };
3378       return SelectVLDSTLane(N, true, false, 2, DOpcodes, QOpcodes);
3379     }
3380 
3381     case Intrinsic::arm_neon_vld3lane: {
3382       static const uint16_t DOpcodes[] = { ARM::VLD3LNd8Pseudo,
3383                                            ARM::VLD3LNd16Pseudo,
3384                                            ARM::VLD3LNd32Pseudo };
3385       static const uint16_t QOpcodes[] = { ARM::VLD3LNq16Pseudo,
3386                                            ARM::VLD3LNq32Pseudo };
3387       return SelectVLDSTLane(N, true, false, 3, DOpcodes, QOpcodes);
3388     }
3389 
3390     case Intrinsic::arm_neon_vld4lane: {
3391       static const uint16_t DOpcodes[] = { ARM::VLD4LNd8Pseudo,
3392                                            ARM::VLD4LNd16Pseudo,
3393                                            ARM::VLD4LNd32Pseudo };
3394       static const uint16_t QOpcodes[] = { ARM::VLD4LNq16Pseudo,
3395                                            ARM::VLD4LNq32Pseudo };
3396       return SelectVLDSTLane(N, true, false, 4, DOpcodes, QOpcodes);
3397     }
3398 
3399     case Intrinsic::arm_neon_vst1: {
3400       static const uint16_t DOpcodes[] = { ARM::VST1d8, ARM::VST1d16,
3401                                            ARM::VST1d32, ARM::VST1d64 };
3402       static const uint16_t QOpcodes[] = { ARM::VST1q8, ARM::VST1q16,
3403                                            ARM::VST1q32, ARM::VST1q64 };
3404       return SelectVST(N, false, 1, DOpcodes, QOpcodes, nullptr);
3405     }
3406 
3407     case Intrinsic::arm_neon_vst2: {
3408       static const uint16_t DOpcodes[] = { ARM::VST2d8, ARM::VST2d16,
3409                                            ARM::VST2d32, ARM::VST1q64 };
3410       static uint16_t QOpcodes[] = { ARM::VST2q8Pseudo, ARM::VST2q16Pseudo,
3411                                      ARM::VST2q32Pseudo };
3412       return SelectVST(N, false, 2, DOpcodes, QOpcodes, nullptr);
3413     }
3414 
3415     case Intrinsic::arm_neon_vst3: {
3416       static const uint16_t DOpcodes[] = { ARM::VST3d8Pseudo,
3417                                            ARM::VST3d16Pseudo,
3418                                            ARM::VST3d32Pseudo,
3419                                            ARM::VST1d64TPseudo };
3420       static const uint16_t QOpcodes0[] = { ARM::VST3q8Pseudo_UPD,
3421                                             ARM::VST3q16Pseudo_UPD,
3422                                             ARM::VST3q32Pseudo_UPD };
3423       static const uint16_t QOpcodes1[] = { ARM::VST3q8oddPseudo,
3424                                             ARM::VST3q16oddPseudo,
3425                                             ARM::VST3q32oddPseudo };
3426       return SelectVST(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1);
3427     }
3428 
3429     case Intrinsic::arm_neon_vst4: {
3430       static const uint16_t DOpcodes[] = { ARM::VST4d8Pseudo,
3431                                            ARM::VST4d16Pseudo,
3432                                            ARM::VST4d32Pseudo,
3433                                            ARM::VST1d64QPseudo };
3434       static const uint16_t QOpcodes0[] = { ARM::VST4q8Pseudo_UPD,
3435                                             ARM::VST4q16Pseudo_UPD,
3436                                             ARM::VST4q32Pseudo_UPD };
3437       static const uint16_t QOpcodes1[] = { ARM::VST4q8oddPseudo,
3438                                             ARM::VST4q16oddPseudo,
3439                                             ARM::VST4q32oddPseudo };
3440       return SelectVST(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1);
3441     }
3442 
3443     case Intrinsic::arm_neon_vst2lane: {
3444       static const uint16_t DOpcodes[] = { ARM::VST2LNd8Pseudo,
3445                                            ARM::VST2LNd16Pseudo,
3446                                            ARM::VST2LNd32Pseudo };
3447       static const uint16_t QOpcodes[] = { ARM::VST2LNq16Pseudo,
3448                                            ARM::VST2LNq32Pseudo };
3449       return SelectVLDSTLane(N, false, false, 2, DOpcodes, QOpcodes);
3450     }
3451 
3452     case Intrinsic::arm_neon_vst3lane: {
3453       static const uint16_t DOpcodes[] = { ARM::VST3LNd8Pseudo,
3454                                            ARM::VST3LNd16Pseudo,
3455                                            ARM::VST3LNd32Pseudo };
3456       static const uint16_t QOpcodes[] = { ARM::VST3LNq16Pseudo,
3457                                            ARM::VST3LNq32Pseudo };
3458       return SelectVLDSTLane(N, false, false, 3, DOpcodes, QOpcodes);
3459     }
3460 
3461     case Intrinsic::arm_neon_vst4lane: {
3462       static const uint16_t DOpcodes[] = { ARM::VST4LNd8Pseudo,
3463                                            ARM::VST4LNd16Pseudo,
3464                                            ARM::VST4LNd32Pseudo };
3465       static const uint16_t QOpcodes[] = { ARM::VST4LNq16Pseudo,
3466                                            ARM::VST4LNq32Pseudo };
3467       return SelectVLDSTLane(N, false, false, 4, DOpcodes, QOpcodes);
3468     }
3469     }
3470     break;
3471   }
3472 
3473   case ISD::INTRINSIC_WO_CHAIN: {
3474     unsigned IntNo = cast<ConstantSDNode>(N->getOperand(0))->getZExtValue();
3475     switch (IntNo) {
3476     default:
3477       break;
3478 
3479     case Intrinsic::arm_neon_vtbl2:
3480       return SelectVTBL(N, false, 2, ARM::VTBL2);
3481     case Intrinsic::arm_neon_vtbl3:
3482       return SelectVTBL(N, false, 3, ARM::VTBL3Pseudo);
3483     case Intrinsic::arm_neon_vtbl4:
3484       return SelectVTBL(N, false, 4, ARM::VTBL4Pseudo);
3485 
3486     case Intrinsic::arm_neon_vtbx2:
3487       return SelectVTBL(N, true, 2, ARM::VTBX2);
3488     case Intrinsic::arm_neon_vtbx3:
3489       return SelectVTBL(N, true, 3, ARM::VTBX3Pseudo);
3490     case Intrinsic::arm_neon_vtbx4:
3491       return SelectVTBL(N, true, 4, ARM::VTBX4Pseudo);
3492     }
3493     break;
3494   }
3495 
3496   case ARMISD::VTBL1: {
3497     SDLoc dl(N);
3498     EVT VT = N->getValueType(0);
3499     SmallVector<SDValue, 6> Ops;
3500 
3501     Ops.push_back(N->getOperand(0));
3502     Ops.push_back(N->getOperand(1));
3503     Ops.push_back(getAL(CurDAG, dl));                // Predicate
3504     Ops.push_back(CurDAG->getRegister(0, MVT::i32)); // Predicate Register
3505     return CurDAG->getMachineNode(ARM::VTBL1, dl, VT, Ops);
3506   }
3507   case ARMISD::VTBL2: {
3508     SDLoc dl(N);
3509     EVT VT = N->getValueType(0);
3510 
3511     // Form a REG_SEQUENCE to force register allocation.
3512     SDValue V0 = N->getOperand(0);
3513     SDValue V1 = N->getOperand(1);
3514     SDValue RegSeq = SDValue(createDRegPairNode(MVT::v16i8, V0, V1), 0);
3515 
3516     SmallVector<SDValue, 6> Ops;
3517     Ops.push_back(RegSeq);
3518     Ops.push_back(N->getOperand(2));
3519     Ops.push_back(getAL(CurDAG, dl));                // Predicate
3520     Ops.push_back(CurDAG->getRegister(0, MVT::i32)); // Predicate Register
3521     return CurDAG->getMachineNode(ARM::VTBL2, dl, VT, Ops);
3522   }
3523 
3524   case ISD::CONCAT_VECTORS:
3525     return SelectConcatVector(N);
3526 
3527   case ISD::ATOMIC_CMP_SWAP:
3528       return SelectCMP_SWAP(N);
3529   }
3530 
3531   return SelectCode(N);
3532 }
3533 
3534 // Inspect a register string of the form
3535 // cp<coprocessor>:<opc1>:c<CRn>:c<CRm>:<opc2> (32bit) or
3536 // cp<coprocessor>:<opc1>:c<CRm> (64bit) inspect the fields of the string
3537 // and obtain the integer operands from them, adding these operands to the
3538 // provided vector.
3539 static void getIntOperandsFromRegisterString(StringRef RegString,
3540                                              SelectionDAG *CurDAG, SDLoc DL,
3541                                              std::vector<SDValue>& Ops) {
3542   SmallVector<StringRef, 5> Fields;
3543   RegString.split(Fields, ':');
3544 
3545   if (Fields.size() > 1) {
3546     bool AllIntFields = true;
3547 
3548     for (StringRef Field : Fields) {
3549       // Need to trim out leading 'cp' characters and get the integer field.
3550       unsigned IntField;
3551       AllIntFields &= !Field.trim("CPcp").getAsInteger(10, IntField);
3552       Ops.push_back(CurDAG->getTargetConstant(IntField, DL, MVT::i32));
3553     }
3554 
3555     assert(AllIntFields &&
3556             "Unexpected non-integer value in special register string.");
3557   }
3558 }
3559 
3560 // Maps a Banked Register string to its mask value. The mask value returned is
3561 // for use in the MRSbanked / MSRbanked instruction nodes as the Banked Register
3562 // mask operand, which expresses which register is to be used, e.g. r8, and in
3563 // which mode it is to be used, e.g. usr. Returns -1 to signify that the string
3564 // was invalid.
3565 static inline int getBankedRegisterMask(StringRef RegString) {
3566   return StringSwitch<int>(RegString.lower())
3567           .Case("r8_usr", 0x00)
3568           .Case("r9_usr", 0x01)
3569           .Case("r10_usr", 0x02)
3570           .Case("r11_usr", 0x03)
3571           .Case("r12_usr", 0x04)
3572           .Case("sp_usr", 0x05)
3573           .Case("lr_usr", 0x06)
3574           .Case("r8_fiq", 0x08)
3575           .Case("r9_fiq", 0x09)
3576           .Case("r10_fiq", 0x0a)
3577           .Case("r11_fiq", 0x0b)
3578           .Case("r12_fiq", 0x0c)
3579           .Case("sp_fiq", 0x0d)
3580           .Case("lr_fiq", 0x0e)
3581           .Case("lr_irq", 0x10)
3582           .Case("sp_irq", 0x11)
3583           .Case("lr_svc", 0x12)
3584           .Case("sp_svc", 0x13)
3585           .Case("lr_abt", 0x14)
3586           .Case("sp_abt", 0x15)
3587           .Case("lr_und", 0x16)
3588           .Case("sp_und", 0x17)
3589           .Case("lr_mon", 0x1c)
3590           .Case("sp_mon", 0x1d)
3591           .Case("elr_hyp", 0x1e)
3592           .Case("sp_hyp", 0x1f)
3593           .Case("spsr_fiq", 0x2e)
3594           .Case("spsr_irq", 0x30)
3595           .Case("spsr_svc", 0x32)
3596           .Case("spsr_abt", 0x34)
3597           .Case("spsr_und", 0x36)
3598           .Case("spsr_mon", 0x3c)
3599           .Case("spsr_hyp", 0x3e)
3600           .Default(-1);
3601 }
3602 
3603 // Maps a MClass special register string to its value for use in the
3604 // t2MRS_M / t2MSR_M instruction nodes as the SYSm value operand.
3605 // Returns -1 to signify that the string was invalid.
3606 static inline int getMClassRegisterSYSmValueMask(StringRef RegString) {
3607   return StringSwitch<int>(RegString.lower())
3608           .Case("apsr", 0x0)
3609           .Case("iapsr", 0x1)
3610           .Case("eapsr", 0x2)
3611           .Case("xpsr", 0x3)
3612           .Case("ipsr", 0x5)
3613           .Case("epsr", 0x6)
3614           .Case("iepsr", 0x7)
3615           .Case("msp", 0x8)
3616           .Case("psp", 0x9)
3617           .Case("primask", 0x10)
3618           .Case("basepri", 0x11)
3619           .Case("basepri_max", 0x12)
3620           .Case("faultmask", 0x13)
3621           .Case("control", 0x14)
3622           .Case("msplim", 0x0a)
3623           .Case("psplim", 0x0b)
3624           .Case("sp", 0x18)
3625           .Default(-1);
3626 }
3627 
3628 // The flags here are common to those allowed for apsr in the A class cores and
3629 // those allowed for the special registers in the M class cores. Returns a
3630 // value representing which flags were present, -1 if invalid.
3631 static inline int getMClassFlagsMask(StringRef Flags, bool hasDSP) {
3632   if (Flags.empty())
3633     return 0x2 | (int)hasDSP;
3634 
3635   return StringSwitch<int>(Flags)
3636           .Case("g", 0x1)
3637           .Case("nzcvq", 0x2)
3638           .Case("nzcvqg", 0x3)
3639           .Default(-1);
3640 }
3641 
3642 static int getMClassRegisterMask(StringRef Reg, StringRef Flags, bool IsRead,
3643                                  const ARMSubtarget *Subtarget) {
3644   // Ensure that the register (without flags) was a valid M Class special
3645   // register.
3646   int SYSmvalue = getMClassRegisterSYSmValueMask(Reg);
3647   if (SYSmvalue == -1)
3648     return -1;
3649 
3650   // basepri, basepri_max and faultmask are only valid for V7m.
3651   if (!Subtarget->hasV7Ops() && SYSmvalue >= 0x11 && SYSmvalue <= 0x13)
3652     return -1;
3653 
3654   if (Subtarget->has8MSecExt() && Flags.lower() == "ns") {
3655     Flags = "";
3656     SYSmvalue |= 0x80;
3657   }
3658 
3659   if (!Subtarget->has8MSecExt() &&
3660       (SYSmvalue == 0xa || SYSmvalue == 0xb || SYSmvalue > 0x14))
3661     return -1;
3662 
3663   if (!Subtarget->hasV8MMainlineOps() &&
3664       (SYSmvalue == 0x8a || SYSmvalue == 0x8b || SYSmvalue == 0x91 ||
3665        SYSmvalue == 0x93))
3666     return -1;
3667 
3668   // If it was a read then we won't be expecting flags and so at this point
3669   // we can return the mask.
3670   if (IsRead) {
3671     if (Flags.empty())
3672       return SYSmvalue;
3673     else
3674       return -1;
3675   }
3676 
3677   // We know we are now handling a write so need to get the mask for the flags.
3678   int Mask = getMClassFlagsMask(Flags, Subtarget->hasDSP());
3679 
3680   // Only apsr, iapsr, eapsr, xpsr can have flags. The other register values
3681   // shouldn't have flags present.
3682   if ((SYSmvalue < 0x4 && Mask == -1) || (SYSmvalue > 0x4 && !Flags.empty()))
3683     return -1;
3684 
3685   // The _g and _nzcvqg versions are only valid if the DSP extension is
3686   // available.
3687   if (!Subtarget->hasDSP() && (Mask & 0x1))
3688     return -1;
3689 
3690   // The register was valid so need to put the mask in the correct place
3691   // (the flags need to be in bits 11-10) and combine with the SYSmvalue to
3692   // construct the operand for the instruction node.
3693   if (SYSmvalue < 0x4)
3694     return SYSmvalue | Mask << 10;
3695 
3696   return SYSmvalue;
3697 }
3698 
3699 static int getARClassRegisterMask(StringRef Reg, StringRef Flags) {
3700   // The mask operand contains the special register (R Bit) in bit 4, whether
3701   // the register is spsr (R bit is 1) or one of cpsr/apsr (R bit is 0), and
3702   // bits 3-0 contains the fields to be accessed in the special register, set by
3703   // the flags provided with the register.
3704   int Mask = 0;
3705   if (Reg == "apsr") {
3706     // The flags permitted for apsr are the same flags that are allowed in
3707     // M class registers. We get the flag value and then shift the flags into
3708     // the correct place to combine with the mask.
3709     Mask = getMClassFlagsMask(Flags, true);
3710     if (Mask == -1)
3711       return -1;
3712     return Mask << 2;
3713   }
3714 
3715   if (Reg != "cpsr" && Reg != "spsr") {
3716     return -1;
3717   }
3718 
3719   // This is the same as if the flags were "fc"
3720   if (Flags.empty() || Flags == "all")
3721     return Mask | 0x9;
3722 
3723   // Inspect the supplied flags string and set the bits in the mask for
3724   // the relevant and valid flags allowed for cpsr and spsr.
3725   for (char Flag : Flags) {
3726     int FlagVal;
3727     switch (Flag) {
3728       case 'c':
3729         FlagVal = 0x1;
3730         break;
3731       case 'x':
3732         FlagVal = 0x2;
3733         break;
3734       case 's':
3735         FlagVal = 0x4;
3736         break;
3737       case 'f':
3738         FlagVal = 0x8;
3739         break;
3740       default:
3741         FlagVal = 0;
3742     }
3743 
3744     // This avoids allowing strings where the same flag bit appears twice.
3745     if (!FlagVal || (Mask & FlagVal))
3746       return -1;
3747     Mask |= FlagVal;
3748   }
3749 
3750   // If the register is spsr then we need to set the R bit.
3751   if (Reg == "spsr")
3752     Mask |= 0x10;
3753 
3754   return Mask;
3755 }
3756 
3757 // Lower the read_register intrinsic to ARM specific DAG nodes
3758 // using the supplied metadata string to select the instruction node to use
3759 // and the registers/masks to construct as operands for the node.
3760 SDNode *ARMDAGToDAGISel::SelectReadRegister(SDNode *N){
3761   const MDNodeSDNode *MD = dyn_cast<MDNodeSDNode>(N->getOperand(1));
3762   const MDString *RegString = dyn_cast<MDString>(MD->getMD()->getOperand(0));
3763   bool IsThumb2 = Subtarget->isThumb2();
3764   SDLoc DL(N);
3765 
3766   std::vector<SDValue> Ops;
3767   getIntOperandsFromRegisterString(RegString->getString(), CurDAG, DL, Ops);
3768 
3769   if (!Ops.empty()) {
3770     // If the special register string was constructed of fields (as defined
3771     // in the ACLE) then need to lower to MRC node (32 bit) or
3772     // MRRC node(64 bit), we can make the distinction based on the number of
3773     // operands we have.
3774     unsigned Opcode;
3775     SmallVector<EVT, 3> ResTypes;
3776     if (Ops.size() == 5){
3777       Opcode = IsThumb2 ? ARM::t2MRC : ARM::MRC;
3778       ResTypes.append({ MVT::i32, MVT::Other });
3779     } else {
3780       assert(Ops.size() == 3 &&
3781               "Invalid number of fields in special register string.");
3782       Opcode = IsThumb2 ? ARM::t2MRRC : ARM::MRRC;
3783       ResTypes.append({ MVT::i32, MVT::i32, MVT::Other });
3784     }
3785 
3786     Ops.push_back(getAL(CurDAG, DL));
3787     Ops.push_back(CurDAG->getRegister(0, MVT::i32));
3788     Ops.push_back(N->getOperand(0));
3789     return CurDAG->getMachineNode(Opcode, DL, ResTypes, Ops);
3790   }
3791 
3792   std::string SpecialReg = RegString->getString().lower();
3793 
3794   int BankedReg = getBankedRegisterMask(SpecialReg);
3795   if (BankedReg != -1) {
3796     Ops = { CurDAG->getTargetConstant(BankedReg, DL, MVT::i32),
3797             getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32),
3798             N->getOperand(0) };
3799     return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRSbanked : ARM::MRSbanked,
3800                                   DL, MVT::i32, MVT::Other, Ops);
3801   }
3802 
3803   // The VFP registers are read by creating SelectionDAG nodes with opcodes
3804   // corresponding to the register that is being read from. So we switch on the
3805   // string to find which opcode we need to use.
3806   unsigned Opcode = StringSwitch<unsigned>(SpecialReg)
3807                     .Case("fpscr", ARM::VMRS)
3808                     .Case("fpexc", ARM::VMRS_FPEXC)
3809                     .Case("fpsid", ARM::VMRS_FPSID)
3810                     .Case("mvfr0", ARM::VMRS_MVFR0)
3811                     .Case("mvfr1", ARM::VMRS_MVFR1)
3812                     .Case("mvfr2", ARM::VMRS_MVFR2)
3813                     .Case("fpinst", ARM::VMRS_FPINST)
3814                     .Case("fpinst2", ARM::VMRS_FPINST2)
3815                     .Default(0);
3816 
3817   // If an opcode was found then we can lower the read to a VFP instruction.
3818   if (Opcode) {
3819     if (!Subtarget->hasVFP2())
3820       return nullptr;
3821     if (Opcode == ARM::VMRS_MVFR2 && !Subtarget->hasFPARMv8())
3822       return nullptr;
3823 
3824     Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32),
3825             N->getOperand(0) };
3826     return CurDAG->getMachineNode(Opcode, DL, MVT::i32, MVT::Other, Ops);
3827   }
3828 
3829   // If the target is M Class then need to validate that the register string
3830   // is an acceptable value, so check that a mask can be constructed from the
3831   // string.
3832   if (Subtarget->isMClass()) {
3833     StringRef Flags = "", Reg = SpecialReg;
3834     if (Reg.endswith("_ns")) {
3835       Flags = "ns";
3836       Reg = Reg.drop_back(3);
3837     }
3838 
3839     int SYSmValue = getMClassRegisterMask(Reg, Flags, true, Subtarget);
3840     if (SYSmValue == -1)
3841       return nullptr;
3842 
3843     SDValue Ops[] = { CurDAG->getTargetConstant(SYSmValue, DL, MVT::i32),
3844                       getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32),
3845                       N->getOperand(0) };
3846     return CurDAG->getMachineNode(ARM::t2MRS_M, DL, MVT::i32, MVT::Other, Ops);
3847   }
3848 
3849   // Here we know the target is not M Class so we need to check if it is one
3850   // of the remaining possible values which are apsr, cpsr or spsr.
3851   if (SpecialReg == "apsr" || SpecialReg == "cpsr") {
3852     Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32),
3853             N->getOperand(0) };
3854     return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRS_AR : ARM::MRS, DL,
3855                                   MVT::i32, MVT::Other, Ops);
3856   }
3857 
3858   if (SpecialReg == "spsr") {
3859     Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32),
3860             N->getOperand(0) };
3861     return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRSsys_AR : ARM::MRSsys,
3862                                   DL, MVT::i32, MVT::Other, Ops);
3863   }
3864 
3865   return nullptr;
3866 }
3867 
3868 // Lower the write_register intrinsic to ARM specific DAG nodes
3869 // using the supplied metadata string to select the instruction node to use
3870 // and the registers/masks to use in the nodes
3871 SDNode *ARMDAGToDAGISel::SelectWriteRegister(SDNode *N){
3872   const MDNodeSDNode *MD = dyn_cast<MDNodeSDNode>(N->getOperand(1));
3873   const MDString *RegString = dyn_cast<MDString>(MD->getMD()->getOperand(0));
3874   bool IsThumb2 = Subtarget->isThumb2();
3875   SDLoc DL(N);
3876 
3877   std::vector<SDValue> Ops;
3878   getIntOperandsFromRegisterString(RegString->getString(), CurDAG, DL, Ops);
3879 
3880   if (!Ops.empty()) {
3881     // If the special register string was constructed of fields (as defined
3882     // in the ACLE) then need to lower to MCR node (32 bit) or
3883     // MCRR node(64 bit), we can make the distinction based on the number of
3884     // operands we have.
3885     unsigned Opcode;
3886     if (Ops.size() == 5) {
3887       Opcode = IsThumb2 ? ARM::t2MCR : ARM::MCR;
3888       Ops.insert(Ops.begin()+2, N->getOperand(2));
3889     } else {
3890       assert(Ops.size() == 3 &&
3891               "Invalid number of fields in special register string.");
3892       Opcode = IsThumb2 ? ARM::t2MCRR : ARM::MCRR;
3893       SDValue WriteValue[] = { N->getOperand(2), N->getOperand(3) };
3894       Ops.insert(Ops.begin()+2, WriteValue, WriteValue+2);
3895     }
3896 
3897     Ops.push_back(getAL(CurDAG, DL));
3898     Ops.push_back(CurDAG->getRegister(0, MVT::i32));
3899     Ops.push_back(N->getOperand(0));
3900 
3901     return CurDAG->getMachineNode(Opcode, DL, MVT::Other, Ops);
3902   }
3903 
3904   std::string SpecialReg = RegString->getString().lower();
3905   int BankedReg = getBankedRegisterMask(SpecialReg);
3906   if (BankedReg != -1) {
3907     Ops = { CurDAG->getTargetConstant(BankedReg, DL, MVT::i32), N->getOperand(2),
3908             getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32),
3909             N->getOperand(0) };
3910     return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MSRbanked : ARM::MSRbanked,
3911                                   DL, MVT::Other, Ops);
3912   }
3913 
3914   // The VFP registers are written to by creating SelectionDAG nodes with
3915   // opcodes corresponding to the register that is being written. So we switch
3916   // on the string to find which opcode we need to use.
3917   unsigned Opcode = StringSwitch<unsigned>(SpecialReg)
3918                     .Case("fpscr", ARM::VMSR)
3919                     .Case("fpexc", ARM::VMSR_FPEXC)
3920                     .Case("fpsid", ARM::VMSR_FPSID)
3921                     .Case("fpinst", ARM::VMSR_FPINST)
3922                     .Case("fpinst2", ARM::VMSR_FPINST2)
3923                     .Default(0);
3924 
3925   if (Opcode) {
3926     if (!Subtarget->hasVFP2())
3927       return nullptr;
3928     Ops = { N->getOperand(2), getAL(CurDAG, DL),
3929             CurDAG->getRegister(0, MVT::i32), N->getOperand(0) };
3930     return CurDAG->getMachineNode(Opcode, DL, MVT::Other, Ops);
3931   }
3932 
3933   std::pair<StringRef, StringRef> Fields;
3934   Fields = StringRef(SpecialReg).rsplit('_');
3935   std::string Reg = Fields.first.str();
3936   StringRef Flags = Fields.second;
3937 
3938   // If the target was M Class then need to validate the special register value
3939   // and retrieve the mask for use in the instruction node.
3940   if (Subtarget->isMClass()) {
3941     // basepri_max gets split so need to correct Reg and Flags.
3942     if (SpecialReg == "basepri_max") {
3943       Reg = SpecialReg;
3944       Flags = "";
3945     }
3946     int SYSmValue = getMClassRegisterMask(Reg, Flags, false, Subtarget);
3947     if (SYSmValue == -1)
3948       return nullptr;
3949 
3950     SDValue Ops[] = { CurDAG->getTargetConstant(SYSmValue, DL, MVT::i32),
3951                       N->getOperand(2), getAL(CurDAG, DL),
3952                       CurDAG->getRegister(0, MVT::i32), N->getOperand(0) };
3953     return CurDAG->getMachineNode(ARM::t2MSR_M, DL, MVT::Other, Ops);
3954   }
3955 
3956   // We then check to see if a valid mask can be constructed for one of the
3957   // register string values permitted for the A and R class cores. These values
3958   // are apsr, spsr and cpsr; these are also valid on older cores.
3959   int Mask = getARClassRegisterMask(Reg, Flags);
3960   if (Mask != -1) {
3961     Ops = { CurDAG->getTargetConstant(Mask, DL, MVT::i32), N->getOperand(2),
3962             getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32),
3963             N->getOperand(0) };
3964     return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MSR_AR : ARM::MSR,
3965                                   DL, MVT::Other, Ops);
3966   }
3967 
3968   return nullptr;
3969 }
3970 
3971 SDNode *ARMDAGToDAGISel::SelectInlineAsm(SDNode *N){
3972   std::vector<SDValue> AsmNodeOperands;
3973   unsigned Flag, Kind;
3974   bool Changed = false;
3975   unsigned NumOps = N->getNumOperands();
3976 
3977   // Normally, i64 data is bounded to two arbitrary GRPs for "%r" constraint.
3978   // However, some instrstions (e.g. ldrexd/strexd in ARM mode) require
3979   // (even/even+1) GPRs and use %n and %Hn to refer to the individual regs
3980   // respectively. Since there is no constraint to explicitly specify a
3981   // reg pair, we use GPRPair reg class for "%r" for 64-bit data. For Thumb,
3982   // the 64-bit data may be referred by H, Q, R modifiers, so we still pack
3983   // them into a GPRPair.
3984 
3985   SDLoc dl(N);
3986   SDValue Glue = N->getGluedNode() ? N->getOperand(NumOps-1)
3987                                    : SDValue(nullptr,0);
3988 
3989   SmallVector<bool, 8> OpChanged;
3990   // Glue node will be appended late.
3991   for(unsigned i = 0, e = N->getGluedNode() ? NumOps - 1 : NumOps; i < e; ++i) {
3992     SDValue op = N->getOperand(i);
3993     AsmNodeOperands.push_back(op);
3994 
3995     if (i < InlineAsm::Op_FirstOperand)
3996       continue;
3997 
3998     if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(i))) {
3999       Flag = C->getZExtValue();
4000       Kind = InlineAsm::getKind(Flag);
4001     }
4002     else
4003       continue;
4004 
4005     // Immediate operands to inline asm in the SelectionDAG are modeled with
4006     // two operands. The first is a constant of value InlineAsm::Kind_Imm, and
4007     // the second is a constant with the value of the immediate. If we get here
4008     // and we have a Kind_Imm, skip the next operand, and continue.
4009     if (Kind == InlineAsm::Kind_Imm) {
4010       SDValue op = N->getOperand(++i);
4011       AsmNodeOperands.push_back(op);
4012       continue;
4013     }
4014 
4015     unsigned NumRegs = InlineAsm::getNumOperandRegisters(Flag);
4016     if (NumRegs)
4017       OpChanged.push_back(false);
4018 
4019     unsigned DefIdx = 0;
4020     bool IsTiedToChangedOp = false;
4021     // If it's a use that is tied with a previous def, it has no
4022     // reg class constraint.
4023     if (Changed && InlineAsm::isUseOperandTiedToDef(Flag, DefIdx))
4024       IsTiedToChangedOp = OpChanged[DefIdx];
4025 
4026     if (Kind != InlineAsm::Kind_RegUse && Kind != InlineAsm::Kind_RegDef
4027         && Kind != InlineAsm::Kind_RegDefEarlyClobber)
4028       continue;
4029 
4030     unsigned RC;
4031     bool HasRC = InlineAsm::hasRegClassConstraint(Flag, RC);
4032     if ((!IsTiedToChangedOp && (!HasRC || RC != ARM::GPRRegClassID))
4033         || NumRegs != 2)
4034       continue;
4035 
4036     assert((i+2 < NumOps) && "Invalid number of operands in inline asm");
4037     SDValue V0 = N->getOperand(i+1);
4038     SDValue V1 = N->getOperand(i+2);
4039     unsigned Reg0 = cast<RegisterSDNode>(V0)->getReg();
4040     unsigned Reg1 = cast<RegisterSDNode>(V1)->getReg();
4041     SDValue PairedReg;
4042     MachineRegisterInfo &MRI = MF->getRegInfo();
4043 
4044     if (Kind == InlineAsm::Kind_RegDef ||
4045         Kind == InlineAsm::Kind_RegDefEarlyClobber) {
4046       // Replace the two GPRs with 1 GPRPair and copy values from GPRPair to
4047       // the original GPRs.
4048 
4049       unsigned GPVR = MRI.createVirtualRegister(&ARM::GPRPairRegClass);
4050       PairedReg = CurDAG->getRegister(GPVR, MVT::Untyped);
4051       SDValue Chain = SDValue(N,0);
4052 
4053       SDNode *GU = N->getGluedUser();
4054       SDValue RegCopy = CurDAG->getCopyFromReg(Chain, dl, GPVR, MVT::Untyped,
4055                                                Chain.getValue(1));
4056 
4057       // Extract values from a GPRPair reg and copy to the original GPR reg.
4058       SDValue Sub0 = CurDAG->getTargetExtractSubreg(ARM::gsub_0, dl, MVT::i32,
4059                                                     RegCopy);
4060       SDValue Sub1 = CurDAG->getTargetExtractSubreg(ARM::gsub_1, dl, MVT::i32,
4061                                                     RegCopy);
4062       SDValue T0 = CurDAG->getCopyToReg(Sub0, dl, Reg0, Sub0,
4063                                         RegCopy.getValue(1));
4064       SDValue T1 = CurDAG->getCopyToReg(Sub1, dl, Reg1, Sub1, T0.getValue(1));
4065 
4066       // Update the original glue user.
4067       std::vector<SDValue> Ops(GU->op_begin(), GU->op_end()-1);
4068       Ops.push_back(T1.getValue(1));
4069       CurDAG->UpdateNodeOperands(GU, Ops);
4070     }
4071     else {
4072       // For Kind  == InlineAsm::Kind_RegUse, we first copy two GPRs into a
4073       // GPRPair and then pass the GPRPair to the inline asm.
4074       SDValue Chain = AsmNodeOperands[InlineAsm::Op_InputChain];
4075 
4076       // As REG_SEQ doesn't take RegisterSDNode, we copy them first.
4077       SDValue T0 = CurDAG->getCopyFromReg(Chain, dl, Reg0, MVT::i32,
4078                                           Chain.getValue(1));
4079       SDValue T1 = CurDAG->getCopyFromReg(Chain, dl, Reg1, MVT::i32,
4080                                           T0.getValue(1));
4081       SDValue Pair = SDValue(createGPRPairNode(MVT::Untyped, T0, T1), 0);
4082 
4083       // Copy REG_SEQ into a GPRPair-typed VR and replace the original two
4084       // i32 VRs of inline asm with it.
4085       unsigned GPVR = MRI.createVirtualRegister(&ARM::GPRPairRegClass);
4086       PairedReg = CurDAG->getRegister(GPVR, MVT::Untyped);
4087       Chain = CurDAG->getCopyToReg(T1, dl, GPVR, Pair, T1.getValue(1));
4088 
4089       AsmNodeOperands[InlineAsm::Op_InputChain] = Chain;
4090       Glue = Chain.getValue(1);
4091     }
4092 
4093     Changed = true;
4094 
4095     if(PairedReg.getNode()) {
4096       OpChanged[OpChanged.size() -1 ] = true;
4097       Flag = InlineAsm::getFlagWord(Kind, 1 /* RegNum*/);
4098       if (IsTiedToChangedOp)
4099         Flag = InlineAsm::getFlagWordForMatchingOp(Flag, DefIdx);
4100       else
4101         Flag = InlineAsm::getFlagWordForRegClass(Flag, ARM::GPRPairRegClassID);
4102       // Replace the current flag.
4103       AsmNodeOperands[AsmNodeOperands.size() -1] = CurDAG->getTargetConstant(
4104           Flag, dl, MVT::i32);
4105       // Add the new register node and skip the original two GPRs.
4106       AsmNodeOperands.push_back(PairedReg);
4107       // Skip the next two GPRs.
4108       i += 2;
4109     }
4110   }
4111 
4112   if (Glue.getNode())
4113     AsmNodeOperands.push_back(Glue);
4114   if (!Changed)
4115     return nullptr;
4116 
4117   SDValue New = CurDAG->getNode(ISD::INLINEASM, SDLoc(N),
4118       CurDAG->getVTList(MVT::Other, MVT::Glue), AsmNodeOperands);
4119   New->setNodeId(-1);
4120   return New.getNode();
4121 }
4122 
4123 
4124 bool ARMDAGToDAGISel::
4125 SelectInlineAsmMemoryOperand(const SDValue &Op, unsigned ConstraintID,
4126                              std::vector<SDValue> &OutOps) {
4127   switch(ConstraintID) {
4128   default:
4129     llvm_unreachable("Unexpected asm memory constraint");
4130   case InlineAsm::Constraint_i:
4131     // FIXME: It seems strange that 'i' is needed here since it's supposed to
4132     //        be an immediate and not a memory constraint.
4133     // Fallthrough.
4134   case InlineAsm::Constraint_m:
4135   case InlineAsm::Constraint_o:
4136   case InlineAsm::Constraint_Q:
4137   case InlineAsm::Constraint_Um:
4138   case InlineAsm::Constraint_Un:
4139   case InlineAsm::Constraint_Uq:
4140   case InlineAsm::Constraint_Us:
4141   case InlineAsm::Constraint_Ut:
4142   case InlineAsm::Constraint_Uv:
4143   case InlineAsm::Constraint_Uy:
4144     // Require the address to be in a register.  That is safe for all ARM
4145     // variants and it is hard to do anything much smarter without knowing
4146     // how the operand is used.
4147     OutOps.push_back(Op);
4148     return false;
4149   }
4150   return true;
4151 }
4152 
4153 /// createARMISelDag - This pass converts a legalized DAG into a
4154 /// ARM-specific DAG, ready for instruction scheduling.
4155 ///
4156 FunctionPass *llvm::createARMISelDag(ARMBaseTargetMachine &TM,
4157                                      CodeGenOpt::Level OptLevel) {
4158   return new ARMDAGToDAGISel(TM, OptLevel);
4159 }
4160