1 //===-- ARMISelDAGToDAG.cpp - A dag to dag inst selector for ARM ----------===// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //===----------------------------------------------------------------------===// 9 // 10 // This file defines an instruction selector for the ARM target. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "ARM.h" 15 #include "ARMBaseInstrInfo.h" 16 #include "ARMTargetMachine.h" 17 #include "MCTargetDesc/ARMAddressingModes.h" 18 #include "llvm/ADT/StringSwitch.h" 19 #include "llvm/CodeGen/MachineFrameInfo.h" 20 #include "llvm/CodeGen/MachineFunction.h" 21 #include "llvm/CodeGen/MachineInstrBuilder.h" 22 #include "llvm/CodeGen/MachineRegisterInfo.h" 23 #include "llvm/CodeGen/SelectionDAG.h" 24 #include "llvm/CodeGen/SelectionDAGISel.h" 25 #include "llvm/IR/CallingConv.h" 26 #include "llvm/IR/Constants.h" 27 #include "llvm/IR/DerivedTypes.h" 28 #include "llvm/IR/Function.h" 29 #include "llvm/IR/Intrinsics.h" 30 #include "llvm/IR/LLVMContext.h" 31 #include "llvm/Support/CommandLine.h" 32 #include "llvm/Support/Debug.h" 33 #include "llvm/Support/ErrorHandling.h" 34 #include "llvm/Target/TargetLowering.h" 35 #include "llvm/Target/TargetOptions.h" 36 37 using namespace llvm; 38 39 #define DEBUG_TYPE "arm-isel" 40 41 static cl::opt<bool> 42 DisableShifterOp("disable-shifter-op", cl::Hidden, 43 cl::desc("Disable isel of shifter-op"), 44 cl::init(false)); 45 46 static cl::opt<bool> 47 CheckVMLxHazard("check-vmlx-hazard", cl::Hidden, 48 cl::desc("Check fp vmla / vmls hazard at isel time"), 49 cl::init(true)); 50 51 //===--------------------------------------------------------------------===// 52 /// ARMDAGToDAGISel - ARM specific code to select ARM machine 53 /// instructions for SelectionDAG operations. 54 /// 55 namespace { 56 57 enum AddrMode2Type { 58 AM2_BASE, // Simple AM2 (+-imm12) 59 AM2_SHOP // Shifter-op AM2 60 }; 61 62 class ARMDAGToDAGISel : public SelectionDAGISel { 63 /// Subtarget - Keep a pointer to the ARMSubtarget around so that we can 64 /// make the right decision when generating code for different targets. 65 const ARMSubtarget *Subtarget; 66 67 public: 68 explicit ARMDAGToDAGISel(ARMBaseTargetMachine &tm, CodeGenOpt::Level OptLevel) 69 : SelectionDAGISel(tm, OptLevel) {} 70 71 bool runOnMachineFunction(MachineFunction &MF) override { 72 // Reset the subtarget each time through. 73 Subtarget = &MF.getSubtarget<ARMSubtarget>(); 74 SelectionDAGISel::runOnMachineFunction(MF); 75 return true; 76 } 77 78 const char *getPassName() const override { 79 return "ARM Instruction Selection"; 80 } 81 82 void PreprocessISelDAG() override; 83 84 /// getI32Imm - Return a target constant of type i32 with the specified 85 /// value. 86 inline SDValue getI32Imm(unsigned Imm, SDLoc dl) { 87 return CurDAG->getTargetConstant(Imm, dl, MVT::i32); 88 } 89 90 SDNode *Select(SDNode *N) override; 91 92 93 bool hasNoVMLxHazardUse(SDNode *N) const; 94 bool isShifterOpProfitable(const SDValue &Shift, 95 ARM_AM::ShiftOpc ShOpcVal, unsigned ShAmt); 96 bool SelectRegShifterOperand(SDValue N, SDValue &A, 97 SDValue &B, SDValue &C, 98 bool CheckProfitability = true); 99 bool SelectImmShifterOperand(SDValue N, SDValue &A, 100 SDValue &B, bool CheckProfitability = true); 101 bool SelectShiftRegShifterOperand(SDValue N, SDValue &A, 102 SDValue &B, SDValue &C) { 103 // Don't apply the profitability check 104 return SelectRegShifterOperand(N, A, B, C, false); 105 } 106 bool SelectShiftImmShifterOperand(SDValue N, SDValue &A, 107 SDValue &B) { 108 // Don't apply the profitability check 109 return SelectImmShifterOperand(N, A, B, false); 110 } 111 112 bool SelectAddrModeImm12(SDValue N, SDValue &Base, SDValue &OffImm); 113 bool SelectLdStSOReg(SDValue N, SDValue &Base, SDValue &Offset, SDValue &Opc); 114 115 AddrMode2Type SelectAddrMode2Worker(SDValue N, SDValue &Base, 116 SDValue &Offset, SDValue &Opc); 117 bool SelectAddrMode2Base(SDValue N, SDValue &Base, SDValue &Offset, 118 SDValue &Opc) { 119 return SelectAddrMode2Worker(N, Base, Offset, Opc) == AM2_BASE; 120 } 121 122 bool SelectAddrMode2ShOp(SDValue N, SDValue &Base, SDValue &Offset, 123 SDValue &Opc) { 124 return SelectAddrMode2Worker(N, Base, Offset, Opc) == AM2_SHOP; 125 } 126 127 bool SelectAddrMode2(SDValue N, SDValue &Base, SDValue &Offset, 128 SDValue &Opc) { 129 SelectAddrMode2Worker(N, Base, Offset, Opc); 130 // return SelectAddrMode2ShOp(N, Base, Offset, Opc); 131 // This always matches one way or another. 132 return true; 133 } 134 135 bool SelectCMOVPred(SDValue N, SDValue &Pred, SDValue &Reg) { 136 const ConstantSDNode *CN = cast<ConstantSDNode>(N); 137 Pred = CurDAG->getTargetConstant(CN->getZExtValue(), SDLoc(N), MVT::i32); 138 Reg = CurDAG->getRegister(ARM::CPSR, MVT::i32); 139 return true; 140 } 141 142 bool SelectAddrMode2OffsetReg(SDNode *Op, SDValue N, 143 SDValue &Offset, SDValue &Opc); 144 bool SelectAddrMode2OffsetImm(SDNode *Op, SDValue N, 145 SDValue &Offset, SDValue &Opc); 146 bool SelectAddrMode2OffsetImmPre(SDNode *Op, SDValue N, 147 SDValue &Offset, SDValue &Opc); 148 bool SelectAddrOffsetNone(SDValue N, SDValue &Base); 149 bool SelectAddrMode3(SDValue N, SDValue &Base, 150 SDValue &Offset, SDValue &Opc); 151 bool SelectAddrMode3Offset(SDNode *Op, SDValue N, 152 SDValue &Offset, SDValue &Opc); 153 bool SelectAddrMode5(SDValue N, SDValue &Base, 154 SDValue &Offset); 155 bool SelectAddrMode6(SDNode *Parent, SDValue N, SDValue &Addr,SDValue &Align); 156 bool SelectAddrMode6Offset(SDNode *Op, SDValue N, SDValue &Offset); 157 158 bool SelectAddrModePC(SDValue N, SDValue &Offset, SDValue &Label); 159 160 // Thumb Addressing Modes: 161 bool SelectThumbAddrModeRR(SDValue N, SDValue &Base, SDValue &Offset); 162 bool SelectThumbAddrModeImm5S(SDValue N, unsigned Scale, SDValue &Base, 163 SDValue &OffImm); 164 bool SelectThumbAddrModeImm5S1(SDValue N, SDValue &Base, 165 SDValue &OffImm); 166 bool SelectThumbAddrModeImm5S2(SDValue N, SDValue &Base, 167 SDValue &OffImm); 168 bool SelectThumbAddrModeImm5S4(SDValue N, SDValue &Base, 169 SDValue &OffImm); 170 bool SelectThumbAddrModeSP(SDValue N, SDValue &Base, SDValue &OffImm); 171 172 // Thumb 2 Addressing Modes: 173 bool SelectT2AddrModeImm12(SDValue N, SDValue &Base, SDValue &OffImm); 174 bool SelectT2AddrModeImm8(SDValue N, SDValue &Base, 175 SDValue &OffImm); 176 bool SelectT2AddrModeImm8Offset(SDNode *Op, SDValue N, 177 SDValue &OffImm); 178 bool SelectT2AddrModeSoReg(SDValue N, SDValue &Base, 179 SDValue &OffReg, SDValue &ShImm); 180 bool SelectT2AddrModeExclusive(SDValue N, SDValue &Base, SDValue &OffImm); 181 182 inline bool is_so_imm(unsigned Imm) const { 183 return ARM_AM::getSOImmVal(Imm) != -1; 184 } 185 186 inline bool is_so_imm_not(unsigned Imm) const { 187 return ARM_AM::getSOImmVal(~Imm) != -1; 188 } 189 190 inline bool is_t2_so_imm(unsigned Imm) const { 191 return ARM_AM::getT2SOImmVal(Imm) != -1; 192 } 193 194 inline bool is_t2_so_imm_not(unsigned Imm) const { 195 return ARM_AM::getT2SOImmVal(~Imm) != -1; 196 } 197 198 // Include the pieces autogenerated from the target description. 199 #include "ARMGenDAGISel.inc" 200 201 private: 202 /// SelectARMIndexedLoad - Indexed (pre/post inc/dec) load matching code for 203 /// ARM. 204 SDNode *SelectARMIndexedLoad(SDNode *N); 205 SDNode *SelectT2IndexedLoad(SDNode *N); 206 207 /// SelectVLD - Select NEON load intrinsics. NumVecs should be 208 /// 1, 2, 3 or 4. The opcode arrays specify the instructions used for 209 /// loads of D registers and even subregs and odd subregs of Q registers. 210 /// For NumVecs <= 2, QOpcodes1 is not used. 211 SDNode *SelectVLD(SDNode *N, bool isUpdating, unsigned NumVecs, 212 const uint16_t *DOpcodes, 213 const uint16_t *QOpcodes0, const uint16_t *QOpcodes1); 214 215 /// SelectVST - Select NEON store intrinsics. NumVecs should 216 /// be 1, 2, 3 or 4. The opcode arrays specify the instructions used for 217 /// stores of D registers and even subregs and odd subregs of Q registers. 218 /// For NumVecs <= 2, QOpcodes1 is not used. 219 SDNode *SelectVST(SDNode *N, bool isUpdating, unsigned NumVecs, 220 const uint16_t *DOpcodes, 221 const uint16_t *QOpcodes0, const uint16_t *QOpcodes1); 222 223 /// SelectVLDSTLane - Select NEON load/store lane intrinsics. NumVecs should 224 /// be 2, 3 or 4. The opcode arrays specify the instructions used for 225 /// load/store of D registers and Q registers. 226 SDNode *SelectVLDSTLane(SDNode *N, bool IsLoad, 227 bool isUpdating, unsigned NumVecs, 228 const uint16_t *DOpcodes, const uint16_t *QOpcodes); 229 230 /// SelectVLDDup - Select NEON load-duplicate intrinsics. NumVecs 231 /// should be 2, 3 or 4. The opcode array specifies the instructions used 232 /// for loading D registers. (Q registers are not supported.) 233 SDNode *SelectVLDDup(SDNode *N, bool isUpdating, unsigned NumVecs, 234 const uint16_t *Opcodes); 235 236 /// SelectVTBL - Select NEON VTBL and VTBX intrinsics. NumVecs should be 2, 237 /// 3 or 4. These are custom-selected so that a REG_SEQUENCE can be 238 /// generated to force the table registers to be consecutive. 239 SDNode *SelectVTBL(SDNode *N, bool IsExt, unsigned NumVecs, unsigned Opc); 240 241 /// SelectV6T2BitfieldExtractOp - Select SBFX/UBFX instructions for ARM. 242 SDNode *SelectV6T2BitfieldExtractOp(SDNode *N, bool isSigned); 243 244 // Select special operations if node forms integer ABS pattern 245 SDNode *SelectABSOp(SDNode *N); 246 247 SDNode *SelectReadRegister(SDNode *N); 248 SDNode *SelectWriteRegister(SDNode *N); 249 250 SDNode *SelectInlineAsm(SDNode *N); 251 252 SDNode *SelectConcatVector(SDNode *N); 253 254 SDNode *SelectSMLAWSMULW(SDNode *N); 255 256 SDNode *SelectCMP_SWAP(SDNode *N); 257 258 /// SelectInlineAsmMemoryOperand - Implement addressing mode selection for 259 /// inline asm expressions. 260 bool SelectInlineAsmMemoryOperand(const SDValue &Op, unsigned ConstraintID, 261 std::vector<SDValue> &OutOps) override; 262 263 // Form pairs of consecutive R, S, D, or Q registers. 264 SDNode *createGPRPairNode(EVT VT, SDValue V0, SDValue V1); 265 SDNode *createSRegPairNode(EVT VT, SDValue V0, SDValue V1); 266 SDNode *createDRegPairNode(EVT VT, SDValue V0, SDValue V1); 267 SDNode *createQRegPairNode(EVT VT, SDValue V0, SDValue V1); 268 269 // Form sequences of 4 consecutive S, D, or Q registers. 270 SDNode *createQuadSRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3); 271 SDNode *createQuadDRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3); 272 SDNode *createQuadQRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3); 273 274 // Get the alignment operand for a NEON VLD or VST instruction. 275 SDValue GetVLDSTAlign(SDValue Align, SDLoc dl, unsigned NumVecs, 276 bool is64BitVector); 277 278 /// Returns the number of instructions required to materialize the given 279 /// constant in a register, or 3 if a literal pool load is needed. 280 unsigned ConstantMaterializationCost(unsigned Val) const; 281 282 /// Checks if N is a multiplication by a constant where we can extract out a 283 /// power of two from the constant so that it can be used in a shift, but only 284 /// if it simplifies the materialization of the constant. Returns true if it 285 /// is, and assigns to PowerOfTwo the power of two that should be extracted 286 /// out and to NewMulConst the new constant to be multiplied by. 287 bool canExtractShiftFromMul(const SDValue &N, unsigned MaxShift, 288 unsigned &PowerOfTwo, SDValue &NewMulConst) const; 289 290 /// Replace N with M in CurDAG, in a way that also ensures that M gets 291 /// selected when N would have been selected. 292 void replaceDAGValue(const SDValue &N, SDValue M); 293 }; 294 } 295 296 /// isInt32Immediate - This method tests to see if the node is a 32-bit constant 297 /// operand. If so Imm will receive the 32-bit value. 298 static bool isInt32Immediate(SDNode *N, unsigned &Imm) { 299 if (N->getOpcode() == ISD::Constant && N->getValueType(0) == MVT::i32) { 300 Imm = cast<ConstantSDNode>(N)->getZExtValue(); 301 return true; 302 } 303 return false; 304 } 305 306 // isInt32Immediate - This method tests to see if a constant operand. 307 // If so Imm will receive the 32 bit value. 308 static bool isInt32Immediate(SDValue N, unsigned &Imm) { 309 return isInt32Immediate(N.getNode(), Imm); 310 } 311 312 // isOpcWithIntImmediate - This method tests to see if the node is a specific 313 // opcode and that it has a immediate integer right operand. 314 // If so Imm will receive the 32 bit value. 315 static bool isOpcWithIntImmediate(SDNode *N, unsigned Opc, unsigned& Imm) { 316 return N->getOpcode() == Opc && 317 isInt32Immediate(N->getOperand(1).getNode(), Imm); 318 } 319 320 /// \brief Check whether a particular node is a constant value representable as 321 /// (N * Scale) where (N in [\p RangeMin, \p RangeMax). 322 /// 323 /// \param ScaledConstant [out] - On success, the pre-scaled constant value. 324 static bool isScaledConstantInRange(SDValue Node, int Scale, 325 int RangeMin, int RangeMax, 326 int &ScaledConstant) { 327 assert(Scale > 0 && "Invalid scale!"); 328 329 // Check that this is a constant. 330 const ConstantSDNode *C = dyn_cast<ConstantSDNode>(Node); 331 if (!C) 332 return false; 333 334 ScaledConstant = (int) C->getZExtValue(); 335 if ((ScaledConstant % Scale) != 0) 336 return false; 337 338 ScaledConstant /= Scale; 339 return ScaledConstant >= RangeMin && ScaledConstant < RangeMax; 340 } 341 342 void ARMDAGToDAGISel::PreprocessISelDAG() { 343 if (!Subtarget->hasV6T2Ops()) 344 return; 345 346 bool isThumb2 = Subtarget->isThumb(); 347 for (SelectionDAG::allnodes_iterator I = CurDAG->allnodes_begin(), 348 E = CurDAG->allnodes_end(); I != E; ) { 349 SDNode *N = &*I++; // Preincrement iterator to avoid invalidation issues. 350 351 if (N->getOpcode() != ISD::ADD) 352 continue; 353 354 // Look for (add X1, (and (srl X2, c1), c2)) where c2 is constant with 355 // leading zeros, followed by consecutive set bits, followed by 1 or 2 356 // trailing zeros, e.g. 1020. 357 // Transform the expression to 358 // (add X1, (shl (and (srl X2, c1), (c2>>tz)), tz)) where tz is the number 359 // of trailing zeros of c2. The left shift would be folded as an shifter 360 // operand of 'add' and the 'and' and 'srl' would become a bits extraction 361 // node (UBFX). 362 363 SDValue N0 = N->getOperand(0); 364 SDValue N1 = N->getOperand(1); 365 unsigned And_imm = 0; 366 if (!isOpcWithIntImmediate(N1.getNode(), ISD::AND, And_imm)) { 367 if (isOpcWithIntImmediate(N0.getNode(), ISD::AND, And_imm)) 368 std::swap(N0, N1); 369 } 370 if (!And_imm) 371 continue; 372 373 // Check if the AND mask is an immediate of the form: 000.....1111111100 374 unsigned TZ = countTrailingZeros(And_imm); 375 if (TZ != 1 && TZ != 2) 376 // Be conservative here. Shifter operands aren't always free. e.g. On 377 // Swift, left shifter operand of 1 / 2 for free but others are not. 378 // e.g. 379 // ubfx r3, r1, #16, #8 380 // ldr.w r3, [r0, r3, lsl #2] 381 // vs. 382 // mov.w r9, #1020 383 // and.w r2, r9, r1, lsr #14 384 // ldr r2, [r0, r2] 385 continue; 386 And_imm >>= TZ; 387 if (And_imm & (And_imm + 1)) 388 continue; 389 390 // Look for (and (srl X, c1), c2). 391 SDValue Srl = N1.getOperand(0); 392 unsigned Srl_imm = 0; 393 if (!isOpcWithIntImmediate(Srl.getNode(), ISD::SRL, Srl_imm) || 394 (Srl_imm <= 2)) 395 continue; 396 397 // Make sure first operand is not a shifter operand which would prevent 398 // folding of the left shift. 399 SDValue CPTmp0; 400 SDValue CPTmp1; 401 SDValue CPTmp2; 402 if (isThumb2) { 403 if (SelectImmShifterOperand(N0, CPTmp0, CPTmp1)) 404 continue; 405 } else { 406 if (SelectImmShifterOperand(N0, CPTmp0, CPTmp1) || 407 SelectRegShifterOperand(N0, CPTmp0, CPTmp1, CPTmp2)) 408 continue; 409 } 410 411 // Now make the transformation. 412 Srl = CurDAG->getNode(ISD::SRL, SDLoc(Srl), MVT::i32, 413 Srl.getOperand(0), 414 CurDAG->getConstant(Srl_imm + TZ, SDLoc(Srl), 415 MVT::i32)); 416 N1 = CurDAG->getNode(ISD::AND, SDLoc(N1), MVT::i32, 417 Srl, 418 CurDAG->getConstant(And_imm, SDLoc(Srl), MVT::i32)); 419 N1 = CurDAG->getNode(ISD::SHL, SDLoc(N1), MVT::i32, 420 N1, CurDAG->getConstant(TZ, SDLoc(Srl), MVT::i32)); 421 CurDAG->UpdateNodeOperands(N, N0, N1); 422 } 423 } 424 425 /// hasNoVMLxHazardUse - Return true if it's desirable to select a FP MLA / MLS 426 /// node. VFP / NEON fp VMLA / VMLS instructions have special RAW hazards (at 427 /// least on current ARM implementations) which should be avoidded. 428 bool ARMDAGToDAGISel::hasNoVMLxHazardUse(SDNode *N) const { 429 if (OptLevel == CodeGenOpt::None) 430 return true; 431 432 if (!CheckVMLxHazard) 433 return true; 434 435 if (!Subtarget->isCortexA7() && !Subtarget->isCortexA8() && 436 !Subtarget->isCortexA9() && !Subtarget->isSwift()) 437 return true; 438 439 if (!N->hasOneUse()) 440 return false; 441 442 SDNode *Use = *N->use_begin(); 443 if (Use->getOpcode() == ISD::CopyToReg) 444 return true; 445 if (Use->isMachineOpcode()) { 446 const ARMBaseInstrInfo *TII = static_cast<const ARMBaseInstrInfo *>( 447 CurDAG->getSubtarget().getInstrInfo()); 448 449 const MCInstrDesc &MCID = TII->get(Use->getMachineOpcode()); 450 if (MCID.mayStore()) 451 return true; 452 unsigned Opcode = MCID.getOpcode(); 453 if (Opcode == ARM::VMOVRS || Opcode == ARM::VMOVRRD) 454 return true; 455 // vmlx feeding into another vmlx. We actually want to unfold 456 // the use later in the MLxExpansion pass. e.g. 457 // vmla 458 // vmla (stall 8 cycles) 459 // 460 // vmul (5 cycles) 461 // vadd (5 cycles) 462 // vmla 463 // This adds up to about 18 - 19 cycles. 464 // 465 // vmla 466 // vmul (stall 4 cycles) 467 // vadd adds up to about 14 cycles. 468 return TII->isFpMLxInstruction(Opcode); 469 } 470 471 return false; 472 } 473 474 bool ARMDAGToDAGISel::isShifterOpProfitable(const SDValue &Shift, 475 ARM_AM::ShiftOpc ShOpcVal, 476 unsigned ShAmt) { 477 if (!Subtarget->isLikeA9() && !Subtarget->isSwift()) 478 return true; 479 if (Shift.hasOneUse()) 480 return true; 481 // R << 2 is free. 482 return ShOpcVal == ARM_AM::lsl && 483 (ShAmt == 2 || (Subtarget->isSwift() && ShAmt == 1)); 484 } 485 486 unsigned ARMDAGToDAGISel::ConstantMaterializationCost(unsigned Val) const { 487 if (Subtarget->isThumb()) { 488 if (Val <= 255) return 1; // MOV 489 if (Subtarget->hasV6T2Ops() && Val <= 0xffff) return 1; // MOVW 490 if (~Val <= 255) return 2; // MOV + MVN 491 if (ARM_AM::isThumbImmShiftedVal(Val)) return 2; // MOV + LSL 492 } else { 493 if (ARM_AM::getSOImmVal(Val) != -1) return 1; // MOV 494 if (ARM_AM::getSOImmVal(~Val) != -1) return 1; // MVN 495 if (Subtarget->hasV6T2Ops() && Val <= 0xffff) return 1; // MOVW 496 if (ARM_AM::isSOImmTwoPartVal(Val)) return 2; // two instrs 497 } 498 if (Subtarget->useMovt(*MF)) return 2; // MOVW + MOVT 499 return 3; // Literal pool load 500 } 501 502 bool ARMDAGToDAGISel::canExtractShiftFromMul(const SDValue &N, 503 unsigned MaxShift, 504 unsigned &PowerOfTwo, 505 SDValue &NewMulConst) const { 506 assert(N.getOpcode() == ISD::MUL); 507 assert(MaxShift > 0); 508 509 // If the multiply is used in more than one place then changing the constant 510 // will make other uses incorrect, so don't. 511 if (!N.hasOneUse()) return false; 512 // Check if the multiply is by a constant 513 ConstantSDNode *MulConst = dyn_cast<ConstantSDNode>(N.getOperand(1)); 514 if (!MulConst) return false; 515 // If the constant is used in more than one place then modifying it will mean 516 // we need to materialize two constants instead of one, which is a bad idea. 517 if (!MulConst->hasOneUse()) return false; 518 unsigned MulConstVal = MulConst->getZExtValue(); 519 if (MulConstVal == 0) return false; 520 521 // Find the largest power of 2 that MulConstVal is a multiple of 522 PowerOfTwo = MaxShift; 523 while ((MulConstVal % (1 << PowerOfTwo)) != 0) { 524 --PowerOfTwo; 525 if (PowerOfTwo == 0) return false; 526 } 527 528 // Only optimise if the new cost is better 529 unsigned NewMulConstVal = MulConstVal / (1 << PowerOfTwo); 530 NewMulConst = CurDAG->getConstant(NewMulConstVal, SDLoc(N), MVT::i32); 531 unsigned OldCost = ConstantMaterializationCost(MulConstVal); 532 unsigned NewCost = ConstantMaterializationCost(NewMulConstVal); 533 return NewCost < OldCost; 534 } 535 536 void ARMDAGToDAGISel::replaceDAGValue(const SDValue &N, SDValue M) { 537 CurDAG->RepositionNode(N.getNode()->getIterator(), M.getNode()); 538 CurDAG->ReplaceAllUsesWith(N, M); 539 } 540 541 bool ARMDAGToDAGISel::SelectImmShifterOperand(SDValue N, 542 SDValue &BaseReg, 543 SDValue &Opc, 544 bool CheckProfitability) { 545 if (DisableShifterOp) 546 return false; 547 548 // If N is a multiply-by-constant and it's profitable to extract a shift and 549 // use it in a shifted operand do so. 550 if (N.getOpcode() == ISD::MUL) { 551 unsigned PowerOfTwo = 0; 552 SDValue NewMulConst; 553 if (canExtractShiftFromMul(N, 31, PowerOfTwo, NewMulConst)) { 554 BaseReg = SDValue(Select(CurDAG->getNode(ISD::MUL, SDLoc(N), MVT::i32, 555 N.getOperand(0), NewMulConst) 556 .getNode()), 557 0); 558 replaceDAGValue(N.getOperand(1), NewMulConst); 559 Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ARM_AM::lsl, 560 PowerOfTwo), 561 SDLoc(N), MVT::i32); 562 return true; 563 } 564 } 565 566 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode()); 567 568 // Don't match base register only case. That is matched to a separate 569 // lower complexity pattern with explicit register operand. 570 if (ShOpcVal == ARM_AM::no_shift) return false; 571 572 BaseReg = N.getOperand(0); 573 unsigned ShImmVal = 0; 574 ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1)); 575 if (!RHS) return false; 576 ShImmVal = RHS->getZExtValue() & 31; 577 Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, ShImmVal), 578 SDLoc(N), MVT::i32); 579 return true; 580 } 581 582 bool ARMDAGToDAGISel::SelectRegShifterOperand(SDValue N, 583 SDValue &BaseReg, 584 SDValue &ShReg, 585 SDValue &Opc, 586 bool CheckProfitability) { 587 if (DisableShifterOp) 588 return false; 589 590 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode()); 591 592 // Don't match base register only case. That is matched to a separate 593 // lower complexity pattern with explicit register operand. 594 if (ShOpcVal == ARM_AM::no_shift) return false; 595 596 BaseReg = N.getOperand(0); 597 unsigned ShImmVal = 0; 598 ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1)); 599 if (RHS) return false; 600 601 ShReg = N.getOperand(1); 602 if (CheckProfitability && !isShifterOpProfitable(N, ShOpcVal, ShImmVal)) 603 return false; 604 Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, ShImmVal), 605 SDLoc(N), MVT::i32); 606 return true; 607 } 608 609 610 bool ARMDAGToDAGISel::SelectAddrModeImm12(SDValue N, 611 SDValue &Base, 612 SDValue &OffImm) { 613 // Match simple R + imm12 operands. 614 615 // Base only. 616 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 617 !CurDAG->isBaseWithConstantOffset(N)) { 618 if (N.getOpcode() == ISD::FrameIndex) { 619 // Match frame index. 620 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 621 Base = CurDAG->getTargetFrameIndex( 622 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 623 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 624 return true; 625 } 626 627 if (N.getOpcode() == ARMISD::Wrapper && 628 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 629 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 630 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 631 Base = N.getOperand(0); 632 } else 633 Base = N; 634 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 635 return true; 636 } 637 638 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 639 int RHSC = (int)RHS->getSExtValue(); 640 if (N.getOpcode() == ISD::SUB) 641 RHSC = -RHSC; 642 643 if (RHSC > -0x1000 && RHSC < 0x1000) { // 12 bits 644 Base = N.getOperand(0); 645 if (Base.getOpcode() == ISD::FrameIndex) { 646 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 647 Base = CurDAG->getTargetFrameIndex( 648 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 649 } 650 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 651 return true; 652 } 653 } 654 655 // Base only. 656 Base = N; 657 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 658 return true; 659 } 660 661 662 663 bool ARMDAGToDAGISel::SelectLdStSOReg(SDValue N, SDValue &Base, SDValue &Offset, 664 SDValue &Opc) { 665 if (N.getOpcode() == ISD::MUL && 666 ((!Subtarget->isLikeA9() && !Subtarget->isSwift()) || N.hasOneUse())) { 667 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 668 // X * [3,5,9] -> X + X * [2,4,8] etc. 669 int RHSC = (int)RHS->getZExtValue(); 670 if (RHSC & 1) { 671 RHSC = RHSC & ~1; 672 ARM_AM::AddrOpc AddSub = ARM_AM::add; 673 if (RHSC < 0) { 674 AddSub = ARM_AM::sub; 675 RHSC = - RHSC; 676 } 677 if (isPowerOf2_32(RHSC)) { 678 unsigned ShAmt = Log2_32(RHSC); 679 Base = Offset = N.getOperand(0); 680 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, 681 ARM_AM::lsl), 682 SDLoc(N), MVT::i32); 683 return true; 684 } 685 } 686 } 687 } 688 689 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 690 // ISD::OR that is equivalent to an ISD::ADD. 691 !CurDAG->isBaseWithConstantOffset(N)) 692 return false; 693 694 // Leave simple R +/- imm12 operands for LDRi12 695 if (N.getOpcode() == ISD::ADD || N.getOpcode() == ISD::OR) { 696 int RHSC; 697 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1, 698 -0x1000+1, 0x1000, RHSC)) // 12 bits. 699 return false; 700 } 701 702 // Otherwise this is R +/- [possibly shifted] R. 703 ARM_AM::AddrOpc AddSub = N.getOpcode() == ISD::SUB ? ARM_AM::sub:ARM_AM::add; 704 ARM_AM::ShiftOpc ShOpcVal = 705 ARM_AM::getShiftOpcForNode(N.getOperand(1).getOpcode()); 706 unsigned ShAmt = 0; 707 708 Base = N.getOperand(0); 709 Offset = N.getOperand(1); 710 711 if (ShOpcVal != ARM_AM::no_shift) { 712 // Check to see if the RHS of the shift is a constant, if not, we can't fold 713 // it. 714 if (ConstantSDNode *Sh = 715 dyn_cast<ConstantSDNode>(N.getOperand(1).getOperand(1))) { 716 ShAmt = Sh->getZExtValue(); 717 if (isShifterOpProfitable(Offset, ShOpcVal, ShAmt)) 718 Offset = N.getOperand(1).getOperand(0); 719 else { 720 ShAmt = 0; 721 ShOpcVal = ARM_AM::no_shift; 722 } 723 } else { 724 ShOpcVal = ARM_AM::no_shift; 725 } 726 } 727 728 // Try matching (R shl C) + (R). 729 if (N.getOpcode() != ISD::SUB && ShOpcVal == ARM_AM::no_shift && 730 !(Subtarget->isLikeA9() || Subtarget->isSwift() || 731 N.getOperand(0).hasOneUse())) { 732 ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOperand(0).getOpcode()); 733 if (ShOpcVal != ARM_AM::no_shift) { 734 // Check to see if the RHS of the shift is a constant, if not, we can't 735 // fold it. 736 if (ConstantSDNode *Sh = 737 dyn_cast<ConstantSDNode>(N.getOperand(0).getOperand(1))) { 738 ShAmt = Sh->getZExtValue(); 739 if (isShifterOpProfitable(N.getOperand(0), ShOpcVal, ShAmt)) { 740 Offset = N.getOperand(0).getOperand(0); 741 Base = N.getOperand(1); 742 } else { 743 ShAmt = 0; 744 ShOpcVal = ARM_AM::no_shift; 745 } 746 } else { 747 ShOpcVal = ARM_AM::no_shift; 748 } 749 } 750 } 751 752 // If Offset is a multiply-by-constant and it's profitable to extract a shift 753 // and use it in a shifted operand do so. 754 if (Offset.getOpcode() == ISD::MUL && N.hasOneUse()) { 755 unsigned PowerOfTwo = 0; 756 SDValue NewMulConst; 757 if (canExtractShiftFromMul(Offset, 31, PowerOfTwo, NewMulConst)) { 758 replaceDAGValue(Offset.getOperand(1), NewMulConst); 759 ShAmt = PowerOfTwo; 760 ShOpcVal = ARM_AM::lsl; 761 } 762 } 763 764 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal), 765 SDLoc(N), MVT::i32); 766 return true; 767 } 768 769 770 //----- 771 772 AddrMode2Type ARMDAGToDAGISel::SelectAddrMode2Worker(SDValue N, 773 SDValue &Base, 774 SDValue &Offset, 775 SDValue &Opc) { 776 if (N.getOpcode() == ISD::MUL && 777 (!(Subtarget->isLikeA9() || Subtarget->isSwift()) || N.hasOneUse())) { 778 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 779 // X * [3,5,9] -> X + X * [2,4,8] etc. 780 int RHSC = (int)RHS->getZExtValue(); 781 if (RHSC & 1) { 782 RHSC = RHSC & ~1; 783 ARM_AM::AddrOpc AddSub = ARM_AM::add; 784 if (RHSC < 0) { 785 AddSub = ARM_AM::sub; 786 RHSC = - RHSC; 787 } 788 if (isPowerOf2_32(RHSC)) { 789 unsigned ShAmt = Log2_32(RHSC); 790 Base = Offset = N.getOperand(0); 791 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, 792 ARM_AM::lsl), 793 SDLoc(N), MVT::i32); 794 return AM2_SHOP; 795 } 796 } 797 } 798 } 799 800 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 801 // ISD::OR that is equivalent to an ADD. 802 !CurDAG->isBaseWithConstantOffset(N)) { 803 Base = N; 804 if (N.getOpcode() == ISD::FrameIndex) { 805 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 806 Base = CurDAG->getTargetFrameIndex( 807 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 808 } else if (N.getOpcode() == ARMISD::Wrapper && 809 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 810 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 811 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 812 Base = N.getOperand(0); 813 } 814 Offset = CurDAG->getRegister(0, MVT::i32); 815 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(ARM_AM::add, 0, 816 ARM_AM::no_shift), 817 SDLoc(N), MVT::i32); 818 return AM2_BASE; 819 } 820 821 // Match simple R +/- imm12 operands. 822 if (N.getOpcode() != ISD::SUB) { 823 int RHSC; 824 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1, 825 -0x1000+1, 0x1000, RHSC)) { // 12 bits. 826 Base = N.getOperand(0); 827 if (Base.getOpcode() == ISD::FrameIndex) { 828 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 829 Base = CurDAG->getTargetFrameIndex( 830 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 831 } 832 Offset = CurDAG->getRegister(0, MVT::i32); 833 834 ARM_AM::AddrOpc AddSub = ARM_AM::add; 835 if (RHSC < 0) { 836 AddSub = ARM_AM::sub; 837 RHSC = - RHSC; 838 } 839 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, RHSC, 840 ARM_AM::no_shift), 841 SDLoc(N), MVT::i32); 842 return AM2_BASE; 843 } 844 } 845 846 if ((Subtarget->isLikeA9() || Subtarget->isSwift()) && !N.hasOneUse()) { 847 // Compute R +/- (R << N) and reuse it. 848 Base = N; 849 Offset = CurDAG->getRegister(0, MVT::i32); 850 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(ARM_AM::add, 0, 851 ARM_AM::no_shift), 852 SDLoc(N), MVT::i32); 853 return AM2_BASE; 854 } 855 856 // Otherwise this is R +/- [possibly shifted] R. 857 ARM_AM::AddrOpc AddSub = N.getOpcode() != ISD::SUB ? ARM_AM::add:ARM_AM::sub; 858 ARM_AM::ShiftOpc ShOpcVal = 859 ARM_AM::getShiftOpcForNode(N.getOperand(1).getOpcode()); 860 unsigned ShAmt = 0; 861 862 Base = N.getOperand(0); 863 Offset = N.getOperand(1); 864 865 if (ShOpcVal != ARM_AM::no_shift) { 866 // Check to see if the RHS of the shift is a constant, if not, we can't fold 867 // it. 868 if (ConstantSDNode *Sh = 869 dyn_cast<ConstantSDNode>(N.getOperand(1).getOperand(1))) { 870 ShAmt = Sh->getZExtValue(); 871 if (isShifterOpProfitable(Offset, ShOpcVal, ShAmt)) 872 Offset = N.getOperand(1).getOperand(0); 873 else { 874 ShAmt = 0; 875 ShOpcVal = ARM_AM::no_shift; 876 } 877 } else { 878 ShOpcVal = ARM_AM::no_shift; 879 } 880 } 881 882 // Try matching (R shl C) + (R). 883 if (N.getOpcode() != ISD::SUB && ShOpcVal == ARM_AM::no_shift && 884 !(Subtarget->isLikeA9() || Subtarget->isSwift() || 885 N.getOperand(0).hasOneUse())) { 886 ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOperand(0).getOpcode()); 887 if (ShOpcVal != ARM_AM::no_shift) { 888 // Check to see if the RHS of the shift is a constant, if not, we can't 889 // fold it. 890 if (ConstantSDNode *Sh = 891 dyn_cast<ConstantSDNode>(N.getOperand(0).getOperand(1))) { 892 ShAmt = Sh->getZExtValue(); 893 if (isShifterOpProfitable(N.getOperand(0), ShOpcVal, ShAmt)) { 894 Offset = N.getOperand(0).getOperand(0); 895 Base = N.getOperand(1); 896 } else { 897 ShAmt = 0; 898 ShOpcVal = ARM_AM::no_shift; 899 } 900 } else { 901 ShOpcVal = ARM_AM::no_shift; 902 } 903 } 904 } 905 906 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal), 907 SDLoc(N), MVT::i32); 908 return AM2_SHOP; 909 } 910 911 bool ARMDAGToDAGISel::SelectAddrMode2OffsetReg(SDNode *Op, SDValue N, 912 SDValue &Offset, SDValue &Opc) { 913 unsigned Opcode = Op->getOpcode(); 914 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 915 ? cast<LoadSDNode>(Op)->getAddressingMode() 916 : cast<StoreSDNode>(Op)->getAddressingMode(); 917 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 918 ? ARM_AM::add : ARM_AM::sub; 919 int Val; 920 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) 921 return false; 922 923 Offset = N; 924 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode()); 925 unsigned ShAmt = 0; 926 if (ShOpcVal != ARM_AM::no_shift) { 927 // Check to see if the RHS of the shift is a constant, if not, we can't fold 928 // it. 929 if (ConstantSDNode *Sh = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 930 ShAmt = Sh->getZExtValue(); 931 if (isShifterOpProfitable(N, ShOpcVal, ShAmt)) 932 Offset = N.getOperand(0); 933 else { 934 ShAmt = 0; 935 ShOpcVal = ARM_AM::no_shift; 936 } 937 } else { 938 ShOpcVal = ARM_AM::no_shift; 939 } 940 } 941 942 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal), 943 SDLoc(N), MVT::i32); 944 return true; 945 } 946 947 bool ARMDAGToDAGISel::SelectAddrMode2OffsetImmPre(SDNode *Op, SDValue N, 948 SDValue &Offset, SDValue &Opc) { 949 unsigned Opcode = Op->getOpcode(); 950 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 951 ? cast<LoadSDNode>(Op)->getAddressingMode() 952 : cast<StoreSDNode>(Op)->getAddressingMode(); 953 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 954 ? ARM_AM::add : ARM_AM::sub; 955 int Val; 956 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) { // 12 bits. 957 if (AddSub == ARM_AM::sub) Val *= -1; 958 Offset = CurDAG->getRegister(0, MVT::i32); 959 Opc = CurDAG->getTargetConstant(Val, SDLoc(Op), MVT::i32); 960 return true; 961 } 962 963 return false; 964 } 965 966 967 bool ARMDAGToDAGISel::SelectAddrMode2OffsetImm(SDNode *Op, SDValue N, 968 SDValue &Offset, SDValue &Opc) { 969 unsigned Opcode = Op->getOpcode(); 970 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 971 ? cast<LoadSDNode>(Op)->getAddressingMode() 972 : cast<StoreSDNode>(Op)->getAddressingMode(); 973 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 974 ? ARM_AM::add : ARM_AM::sub; 975 int Val; 976 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) { // 12 bits. 977 Offset = CurDAG->getRegister(0, MVT::i32); 978 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, Val, 979 ARM_AM::no_shift), 980 SDLoc(Op), MVT::i32); 981 return true; 982 } 983 984 return false; 985 } 986 987 bool ARMDAGToDAGISel::SelectAddrOffsetNone(SDValue N, SDValue &Base) { 988 Base = N; 989 return true; 990 } 991 992 bool ARMDAGToDAGISel::SelectAddrMode3(SDValue N, 993 SDValue &Base, SDValue &Offset, 994 SDValue &Opc) { 995 if (N.getOpcode() == ISD::SUB) { 996 // X - C is canonicalize to X + -C, no need to handle it here. 997 Base = N.getOperand(0); 998 Offset = N.getOperand(1); 999 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::sub, 0), SDLoc(N), 1000 MVT::i32); 1001 return true; 1002 } 1003 1004 if (!CurDAG->isBaseWithConstantOffset(N)) { 1005 Base = N; 1006 if (N.getOpcode() == ISD::FrameIndex) { 1007 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 1008 Base = CurDAG->getTargetFrameIndex( 1009 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1010 } 1011 Offset = CurDAG->getRegister(0, MVT::i32); 1012 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::add, 0), SDLoc(N), 1013 MVT::i32); 1014 return true; 1015 } 1016 1017 // If the RHS is +/- imm8, fold into addr mode. 1018 int RHSC; 1019 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1, 1020 -256 + 1, 256, RHSC)) { // 8 bits. 1021 Base = N.getOperand(0); 1022 if (Base.getOpcode() == ISD::FrameIndex) { 1023 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1024 Base = CurDAG->getTargetFrameIndex( 1025 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1026 } 1027 Offset = CurDAG->getRegister(0, MVT::i32); 1028 1029 ARM_AM::AddrOpc AddSub = ARM_AM::add; 1030 if (RHSC < 0) { 1031 AddSub = ARM_AM::sub; 1032 RHSC = -RHSC; 1033 } 1034 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, RHSC), SDLoc(N), 1035 MVT::i32); 1036 return true; 1037 } 1038 1039 Base = N.getOperand(0); 1040 Offset = N.getOperand(1); 1041 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::add, 0), SDLoc(N), 1042 MVT::i32); 1043 return true; 1044 } 1045 1046 bool ARMDAGToDAGISel::SelectAddrMode3Offset(SDNode *Op, SDValue N, 1047 SDValue &Offset, SDValue &Opc) { 1048 unsigned Opcode = Op->getOpcode(); 1049 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 1050 ? cast<LoadSDNode>(Op)->getAddressingMode() 1051 : cast<StoreSDNode>(Op)->getAddressingMode(); 1052 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 1053 ? ARM_AM::add : ARM_AM::sub; 1054 int Val; 1055 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 256, Val)) { // 12 bits. 1056 Offset = CurDAG->getRegister(0, MVT::i32); 1057 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, Val), SDLoc(Op), 1058 MVT::i32); 1059 return true; 1060 } 1061 1062 Offset = N; 1063 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, 0), SDLoc(Op), 1064 MVT::i32); 1065 return true; 1066 } 1067 1068 bool ARMDAGToDAGISel::SelectAddrMode5(SDValue N, 1069 SDValue &Base, SDValue &Offset) { 1070 if (!CurDAG->isBaseWithConstantOffset(N)) { 1071 Base = N; 1072 if (N.getOpcode() == ISD::FrameIndex) { 1073 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 1074 Base = CurDAG->getTargetFrameIndex( 1075 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1076 } else if (N.getOpcode() == ARMISD::Wrapper && 1077 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 1078 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 1079 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 1080 Base = N.getOperand(0); 1081 } 1082 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(ARM_AM::add, 0), 1083 SDLoc(N), MVT::i32); 1084 return true; 1085 } 1086 1087 // If the RHS is +/- imm8, fold into addr mode. 1088 int RHSC; 1089 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/4, 1090 -256 + 1, 256, RHSC)) { 1091 Base = N.getOperand(0); 1092 if (Base.getOpcode() == ISD::FrameIndex) { 1093 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1094 Base = CurDAG->getTargetFrameIndex( 1095 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1096 } 1097 1098 ARM_AM::AddrOpc AddSub = ARM_AM::add; 1099 if (RHSC < 0) { 1100 AddSub = ARM_AM::sub; 1101 RHSC = -RHSC; 1102 } 1103 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(AddSub, RHSC), 1104 SDLoc(N), MVT::i32); 1105 return true; 1106 } 1107 1108 Base = N; 1109 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(ARM_AM::add, 0), 1110 SDLoc(N), MVT::i32); 1111 return true; 1112 } 1113 1114 bool ARMDAGToDAGISel::SelectAddrMode6(SDNode *Parent, SDValue N, SDValue &Addr, 1115 SDValue &Align) { 1116 Addr = N; 1117 1118 unsigned Alignment = 0; 1119 1120 MemSDNode *MemN = cast<MemSDNode>(Parent); 1121 1122 if (isa<LSBaseSDNode>(MemN) || 1123 ((MemN->getOpcode() == ARMISD::VST1_UPD || 1124 MemN->getOpcode() == ARMISD::VLD1_UPD) && 1125 MemN->getConstantOperandVal(MemN->getNumOperands() - 1) == 1)) { 1126 // This case occurs only for VLD1-lane/dup and VST1-lane instructions. 1127 // The maximum alignment is equal to the memory size being referenced. 1128 unsigned MMOAlign = MemN->getAlignment(); 1129 unsigned MemSize = MemN->getMemoryVT().getSizeInBits() / 8; 1130 if (MMOAlign >= MemSize && MemSize > 1) 1131 Alignment = MemSize; 1132 } else { 1133 // All other uses of addrmode6 are for intrinsics. For now just record 1134 // the raw alignment value; it will be refined later based on the legal 1135 // alignment operands for the intrinsic. 1136 Alignment = MemN->getAlignment(); 1137 } 1138 1139 Align = CurDAG->getTargetConstant(Alignment, SDLoc(N), MVT::i32); 1140 return true; 1141 } 1142 1143 bool ARMDAGToDAGISel::SelectAddrMode6Offset(SDNode *Op, SDValue N, 1144 SDValue &Offset) { 1145 LSBaseSDNode *LdSt = cast<LSBaseSDNode>(Op); 1146 ISD::MemIndexedMode AM = LdSt->getAddressingMode(); 1147 if (AM != ISD::POST_INC) 1148 return false; 1149 Offset = N; 1150 if (ConstantSDNode *NC = dyn_cast<ConstantSDNode>(N)) { 1151 if (NC->getZExtValue() * 8 == LdSt->getMemoryVT().getSizeInBits()) 1152 Offset = CurDAG->getRegister(0, MVT::i32); 1153 } 1154 return true; 1155 } 1156 1157 bool ARMDAGToDAGISel::SelectAddrModePC(SDValue N, 1158 SDValue &Offset, SDValue &Label) { 1159 if (N.getOpcode() == ARMISD::PIC_ADD && N.hasOneUse()) { 1160 Offset = N.getOperand(0); 1161 SDValue N1 = N.getOperand(1); 1162 Label = CurDAG->getTargetConstant(cast<ConstantSDNode>(N1)->getZExtValue(), 1163 SDLoc(N), MVT::i32); 1164 return true; 1165 } 1166 1167 return false; 1168 } 1169 1170 1171 //===----------------------------------------------------------------------===// 1172 // Thumb Addressing Modes 1173 //===----------------------------------------------------------------------===// 1174 1175 bool ARMDAGToDAGISel::SelectThumbAddrModeRR(SDValue N, 1176 SDValue &Base, SDValue &Offset){ 1177 if (N.getOpcode() != ISD::ADD && !CurDAG->isBaseWithConstantOffset(N)) { 1178 ConstantSDNode *NC = dyn_cast<ConstantSDNode>(N); 1179 if (!NC || !NC->isNullValue()) 1180 return false; 1181 1182 Base = Offset = N; 1183 return true; 1184 } 1185 1186 Base = N.getOperand(0); 1187 Offset = N.getOperand(1); 1188 return true; 1189 } 1190 1191 bool 1192 ARMDAGToDAGISel::SelectThumbAddrModeImm5S(SDValue N, unsigned Scale, 1193 SDValue &Base, SDValue &OffImm) { 1194 if (!CurDAG->isBaseWithConstantOffset(N)) { 1195 if (N.getOpcode() == ISD::ADD) { 1196 return false; // We want to select register offset instead 1197 } else if (N.getOpcode() == ARMISD::Wrapper && 1198 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 1199 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 1200 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 1201 Base = N.getOperand(0); 1202 } else { 1203 Base = N; 1204 } 1205 1206 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1207 return true; 1208 } 1209 1210 // If the RHS is + imm5 * scale, fold into addr mode. 1211 int RHSC; 1212 if (isScaledConstantInRange(N.getOperand(1), Scale, 0, 32, RHSC)) { 1213 Base = N.getOperand(0); 1214 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1215 return true; 1216 } 1217 1218 // Offset is too large, so use register offset instead. 1219 return false; 1220 } 1221 1222 bool 1223 ARMDAGToDAGISel::SelectThumbAddrModeImm5S4(SDValue N, SDValue &Base, 1224 SDValue &OffImm) { 1225 return SelectThumbAddrModeImm5S(N, 4, Base, OffImm); 1226 } 1227 1228 bool 1229 ARMDAGToDAGISel::SelectThumbAddrModeImm5S2(SDValue N, SDValue &Base, 1230 SDValue &OffImm) { 1231 return SelectThumbAddrModeImm5S(N, 2, Base, OffImm); 1232 } 1233 1234 bool 1235 ARMDAGToDAGISel::SelectThumbAddrModeImm5S1(SDValue N, SDValue &Base, 1236 SDValue &OffImm) { 1237 return SelectThumbAddrModeImm5S(N, 1, Base, OffImm); 1238 } 1239 1240 bool ARMDAGToDAGISel::SelectThumbAddrModeSP(SDValue N, 1241 SDValue &Base, SDValue &OffImm) { 1242 if (N.getOpcode() == ISD::FrameIndex) { 1243 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 1244 // Only multiples of 4 are allowed for the offset, so the frame object 1245 // alignment must be at least 4. 1246 MachineFrameInfo *MFI = MF->getFrameInfo(); 1247 if (MFI->getObjectAlignment(FI) < 4) 1248 MFI->setObjectAlignment(FI, 4); 1249 Base = CurDAG->getTargetFrameIndex( 1250 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1251 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1252 return true; 1253 } 1254 1255 if (!CurDAG->isBaseWithConstantOffset(N)) 1256 return false; 1257 1258 RegisterSDNode *LHSR = dyn_cast<RegisterSDNode>(N.getOperand(0)); 1259 if (N.getOperand(0).getOpcode() == ISD::FrameIndex || 1260 (LHSR && LHSR->getReg() == ARM::SP)) { 1261 // If the RHS is + imm8 * scale, fold into addr mode. 1262 int RHSC; 1263 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/4, 0, 256, RHSC)) { 1264 Base = N.getOperand(0); 1265 if (Base.getOpcode() == ISD::FrameIndex) { 1266 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1267 // For LHS+RHS to result in an offset that's a multiple of 4 the object 1268 // indexed by the LHS must be 4-byte aligned. 1269 MachineFrameInfo *MFI = MF->getFrameInfo(); 1270 if (MFI->getObjectAlignment(FI) < 4) 1271 MFI->setObjectAlignment(FI, 4); 1272 Base = CurDAG->getTargetFrameIndex( 1273 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1274 } 1275 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1276 return true; 1277 } 1278 } 1279 1280 return false; 1281 } 1282 1283 1284 //===----------------------------------------------------------------------===// 1285 // Thumb 2 Addressing Modes 1286 //===----------------------------------------------------------------------===// 1287 1288 1289 bool ARMDAGToDAGISel::SelectT2AddrModeImm12(SDValue N, 1290 SDValue &Base, SDValue &OffImm) { 1291 // Match simple R + imm12 operands. 1292 1293 // Base only. 1294 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 1295 !CurDAG->isBaseWithConstantOffset(N)) { 1296 if (N.getOpcode() == ISD::FrameIndex) { 1297 // Match frame index. 1298 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 1299 Base = CurDAG->getTargetFrameIndex( 1300 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1301 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1302 return true; 1303 } 1304 1305 if (N.getOpcode() == ARMISD::Wrapper && 1306 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 1307 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 1308 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 1309 Base = N.getOperand(0); 1310 if (Base.getOpcode() == ISD::TargetConstantPool) 1311 return false; // We want to select t2LDRpci instead. 1312 } else 1313 Base = N; 1314 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1315 return true; 1316 } 1317 1318 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 1319 if (SelectT2AddrModeImm8(N, Base, OffImm)) 1320 // Let t2LDRi8 handle (R - imm8). 1321 return false; 1322 1323 int RHSC = (int)RHS->getZExtValue(); 1324 if (N.getOpcode() == ISD::SUB) 1325 RHSC = -RHSC; 1326 1327 if (RHSC >= 0 && RHSC < 0x1000) { // 12 bits (unsigned) 1328 Base = N.getOperand(0); 1329 if (Base.getOpcode() == ISD::FrameIndex) { 1330 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1331 Base = CurDAG->getTargetFrameIndex( 1332 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1333 } 1334 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1335 return true; 1336 } 1337 } 1338 1339 // Base only. 1340 Base = N; 1341 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1342 return true; 1343 } 1344 1345 bool ARMDAGToDAGISel::SelectT2AddrModeImm8(SDValue N, 1346 SDValue &Base, SDValue &OffImm) { 1347 // Match simple R - imm8 operands. 1348 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 1349 !CurDAG->isBaseWithConstantOffset(N)) 1350 return false; 1351 1352 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 1353 int RHSC = (int)RHS->getSExtValue(); 1354 if (N.getOpcode() == ISD::SUB) 1355 RHSC = -RHSC; 1356 1357 if ((RHSC >= -255) && (RHSC < 0)) { // 8 bits (always negative) 1358 Base = N.getOperand(0); 1359 if (Base.getOpcode() == ISD::FrameIndex) { 1360 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1361 Base = CurDAG->getTargetFrameIndex( 1362 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1363 } 1364 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1365 return true; 1366 } 1367 } 1368 1369 return false; 1370 } 1371 1372 bool ARMDAGToDAGISel::SelectT2AddrModeImm8Offset(SDNode *Op, SDValue N, 1373 SDValue &OffImm){ 1374 unsigned Opcode = Op->getOpcode(); 1375 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 1376 ? cast<LoadSDNode>(Op)->getAddressingMode() 1377 : cast<StoreSDNode>(Op)->getAddressingMode(); 1378 int RHSC; 1379 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x100, RHSC)) { // 8 bits. 1380 OffImm = ((AM == ISD::PRE_INC) || (AM == ISD::POST_INC)) 1381 ? CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32) 1382 : CurDAG->getTargetConstant(-RHSC, SDLoc(N), MVT::i32); 1383 return true; 1384 } 1385 1386 return false; 1387 } 1388 1389 bool ARMDAGToDAGISel::SelectT2AddrModeSoReg(SDValue N, 1390 SDValue &Base, 1391 SDValue &OffReg, SDValue &ShImm) { 1392 // (R - imm8) should be handled by t2LDRi8. The rest are handled by t2LDRi12. 1393 if (N.getOpcode() != ISD::ADD && !CurDAG->isBaseWithConstantOffset(N)) 1394 return false; 1395 1396 // Leave (R + imm12) for t2LDRi12, (R - imm8) for t2LDRi8. 1397 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 1398 int RHSC = (int)RHS->getZExtValue(); 1399 if (RHSC >= 0 && RHSC < 0x1000) // 12 bits (unsigned) 1400 return false; 1401 else if (RHSC < 0 && RHSC >= -255) // 8 bits 1402 return false; 1403 } 1404 1405 // Look for (R + R) or (R + (R << [1,2,3])). 1406 unsigned ShAmt = 0; 1407 Base = N.getOperand(0); 1408 OffReg = N.getOperand(1); 1409 1410 // Swap if it is ((R << c) + R). 1411 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(OffReg.getOpcode()); 1412 if (ShOpcVal != ARM_AM::lsl) { 1413 ShOpcVal = ARM_AM::getShiftOpcForNode(Base.getOpcode()); 1414 if (ShOpcVal == ARM_AM::lsl) 1415 std::swap(Base, OffReg); 1416 } 1417 1418 if (ShOpcVal == ARM_AM::lsl) { 1419 // Check to see if the RHS of the shift is a constant, if not, we can't fold 1420 // it. 1421 if (ConstantSDNode *Sh = dyn_cast<ConstantSDNode>(OffReg.getOperand(1))) { 1422 ShAmt = Sh->getZExtValue(); 1423 if (ShAmt < 4 && isShifterOpProfitable(OffReg, ShOpcVal, ShAmt)) 1424 OffReg = OffReg.getOperand(0); 1425 else { 1426 ShAmt = 0; 1427 } 1428 } 1429 } 1430 1431 // If OffReg is a multiply-by-constant and it's profitable to extract a shift 1432 // and use it in a shifted operand do so. 1433 if (OffReg.getOpcode() == ISD::MUL && N.hasOneUse()) { 1434 unsigned PowerOfTwo = 0; 1435 SDValue NewMulConst; 1436 if (canExtractShiftFromMul(OffReg, 3, PowerOfTwo, NewMulConst)) { 1437 replaceDAGValue(OffReg.getOperand(1), NewMulConst); 1438 ShAmt = PowerOfTwo; 1439 } 1440 } 1441 1442 ShImm = CurDAG->getTargetConstant(ShAmt, SDLoc(N), MVT::i32); 1443 1444 return true; 1445 } 1446 1447 bool ARMDAGToDAGISel::SelectT2AddrModeExclusive(SDValue N, SDValue &Base, 1448 SDValue &OffImm) { 1449 // This *must* succeed since it's used for the irreplaceable ldrex and strex 1450 // instructions. 1451 Base = N; 1452 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1453 1454 if (N.getOpcode() != ISD::ADD || !CurDAG->isBaseWithConstantOffset(N)) 1455 return true; 1456 1457 ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1)); 1458 if (!RHS) 1459 return true; 1460 1461 uint32_t RHSC = (int)RHS->getZExtValue(); 1462 if (RHSC > 1020 || RHSC % 4 != 0) 1463 return true; 1464 1465 Base = N.getOperand(0); 1466 if (Base.getOpcode() == ISD::FrameIndex) { 1467 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1468 Base = CurDAG->getTargetFrameIndex( 1469 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1470 } 1471 1472 OffImm = CurDAG->getTargetConstant(RHSC/4, SDLoc(N), MVT::i32); 1473 return true; 1474 } 1475 1476 //===--------------------------------------------------------------------===// 1477 1478 /// getAL - Returns a ARMCC::AL immediate node. 1479 static inline SDValue getAL(SelectionDAG *CurDAG, SDLoc dl) { 1480 return CurDAG->getTargetConstant((uint64_t)ARMCC::AL, dl, MVT::i32); 1481 } 1482 1483 SDNode *ARMDAGToDAGISel::SelectARMIndexedLoad(SDNode *N) { 1484 LoadSDNode *LD = cast<LoadSDNode>(N); 1485 ISD::MemIndexedMode AM = LD->getAddressingMode(); 1486 if (AM == ISD::UNINDEXED) 1487 return nullptr; 1488 1489 EVT LoadedVT = LD->getMemoryVT(); 1490 SDValue Offset, AMOpc; 1491 bool isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC); 1492 unsigned Opcode = 0; 1493 bool Match = false; 1494 if (LoadedVT == MVT::i32 && isPre && 1495 SelectAddrMode2OffsetImmPre(N, LD->getOffset(), Offset, AMOpc)) { 1496 Opcode = ARM::LDR_PRE_IMM; 1497 Match = true; 1498 } else if (LoadedVT == MVT::i32 && !isPre && 1499 SelectAddrMode2OffsetImm(N, LD->getOffset(), Offset, AMOpc)) { 1500 Opcode = ARM::LDR_POST_IMM; 1501 Match = true; 1502 } else if (LoadedVT == MVT::i32 && 1503 SelectAddrMode2OffsetReg(N, LD->getOffset(), Offset, AMOpc)) { 1504 Opcode = isPre ? ARM::LDR_PRE_REG : ARM::LDR_POST_REG; 1505 Match = true; 1506 1507 } else if (LoadedVT == MVT::i16 && 1508 SelectAddrMode3Offset(N, LD->getOffset(), Offset, AMOpc)) { 1509 Match = true; 1510 Opcode = (LD->getExtensionType() == ISD::SEXTLOAD) 1511 ? (isPre ? ARM::LDRSH_PRE : ARM::LDRSH_POST) 1512 : (isPre ? ARM::LDRH_PRE : ARM::LDRH_POST); 1513 } else if (LoadedVT == MVT::i8 || LoadedVT == MVT::i1) { 1514 if (LD->getExtensionType() == ISD::SEXTLOAD) { 1515 if (SelectAddrMode3Offset(N, LD->getOffset(), Offset, AMOpc)) { 1516 Match = true; 1517 Opcode = isPre ? ARM::LDRSB_PRE : ARM::LDRSB_POST; 1518 } 1519 } else { 1520 if (isPre && 1521 SelectAddrMode2OffsetImmPre(N, LD->getOffset(), Offset, AMOpc)) { 1522 Match = true; 1523 Opcode = ARM::LDRB_PRE_IMM; 1524 } else if (!isPre && 1525 SelectAddrMode2OffsetImm(N, LD->getOffset(), Offset, AMOpc)) { 1526 Match = true; 1527 Opcode = ARM::LDRB_POST_IMM; 1528 } else if (SelectAddrMode2OffsetReg(N, LD->getOffset(), Offset, AMOpc)) { 1529 Match = true; 1530 Opcode = isPre ? ARM::LDRB_PRE_REG : ARM::LDRB_POST_REG; 1531 } 1532 } 1533 } 1534 1535 if (Match) { 1536 if (Opcode == ARM::LDR_PRE_IMM || Opcode == ARM::LDRB_PRE_IMM) { 1537 SDValue Chain = LD->getChain(); 1538 SDValue Base = LD->getBasePtr(); 1539 SDValue Ops[]= { Base, AMOpc, getAL(CurDAG, SDLoc(N)), 1540 CurDAG->getRegister(0, MVT::i32), Chain }; 1541 return CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, 1542 MVT::i32, MVT::Other, Ops); 1543 } else { 1544 SDValue Chain = LD->getChain(); 1545 SDValue Base = LD->getBasePtr(); 1546 SDValue Ops[]= { Base, Offset, AMOpc, getAL(CurDAG, SDLoc(N)), 1547 CurDAG->getRegister(0, MVT::i32), Chain }; 1548 return CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, 1549 MVT::i32, MVT::Other, Ops); 1550 } 1551 } 1552 1553 return nullptr; 1554 } 1555 1556 SDNode *ARMDAGToDAGISel::SelectT2IndexedLoad(SDNode *N) { 1557 LoadSDNode *LD = cast<LoadSDNode>(N); 1558 ISD::MemIndexedMode AM = LD->getAddressingMode(); 1559 if (AM == ISD::UNINDEXED) 1560 return nullptr; 1561 1562 EVT LoadedVT = LD->getMemoryVT(); 1563 bool isSExtLd = LD->getExtensionType() == ISD::SEXTLOAD; 1564 SDValue Offset; 1565 bool isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC); 1566 unsigned Opcode = 0; 1567 bool Match = false; 1568 if (SelectT2AddrModeImm8Offset(N, LD->getOffset(), Offset)) { 1569 switch (LoadedVT.getSimpleVT().SimpleTy) { 1570 case MVT::i32: 1571 Opcode = isPre ? ARM::t2LDR_PRE : ARM::t2LDR_POST; 1572 break; 1573 case MVT::i16: 1574 if (isSExtLd) 1575 Opcode = isPre ? ARM::t2LDRSH_PRE : ARM::t2LDRSH_POST; 1576 else 1577 Opcode = isPre ? ARM::t2LDRH_PRE : ARM::t2LDRH_POST; 1578 break; 1579 case MVT::i8: 1580 case MVT::i1: 1581 if (isSExtLd) 1582 Opcode = isPre ? ARM::t2LDRSB_PRE : ARM::t2LDRSB_POST; 1583 else 1584 Opcode = isPre ? ARM::t2LDRB_PRE : ARM::t2LDRB_POST; 1585 break; 1586 default: 1587 return nullptr; 1588 } 1589 Match = true; 1590 } 1591 1592 if (Match) { 1593 SDValue Chain = LD->getChain(); 1594 SDValue Base = LD->getBasePtr(); 1595 SDValue Ops[]= { Base, Offset, getAL(CurDAG, SDLoc(N)), 1596 CurDAG->getRegister(0, MVT::i32), Chain }; 1597 return CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, MVT::i32, 1598 MVT::Other, Ops); 1599 } 1600 1601 return nullptr; 1602 } 1603 1604 /// \brief Form a GPRPair pseudo register from a pair of GPR regs. 1605 SDNode *ARMDAGToDAGISel::createGPRPairNode(EVT VT, SDValue V0, SDValue V1) { 1606 SDLoc dl(V0.getNode()); 1607 SDValue RegClass = 1608 CurDAG->getTargetConstant(ARM::GPRPairRegClassID, dl, MVT::i32); 1609 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::gsub_0, dl, MVT::i32); 1610 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::gsub_1, dl, MVT::i32); 1611 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1612 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1613 } 1614 1615 /// \brief Form a D register from a pair of S registers. 1616 SDNode *ARMDAGToDAGISel::createSRegPairNode(EVT VT, SDValue V0, SDValue V1) { 1617 SDLoc dl(V0.getNode()); 1618 SDValue RegClass = 1619 CurDAG->getTargetConstant(ARM::DPR_VFP2RegClassID, dl, MVT::i32); 1620 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::ssub_0, dl, MVT::i32); 1621 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::ssub_1, dl, MVT::i32); 1622 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1623 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1624 } 1625 1626 /// \brief Form a quad register from a pair of D registers. 1627 SDNode *ARMDAGToDAGISel::createDRegPairNode(EVT VT, SDValue V0, SDValue V1) { 1628 SDLoc dl(V0.getNode()); 1629 SDValue RegClass = CurDAG->getTargetConstant(ARM::QPRRegClassID, dl, 1630 MVT::i32); 1631 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::dsub_0, dl, MVT::i32); 1632 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::dsub_1, dl, MVT::i32); 1633 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1634 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1635 } 1636 1637 /// \brief Form 4 consecutive D registers from a pair of Q registers. 1638 SDNode *ARMDAGToDAGISel::createQRegPairNode(EVT VT, SDValue V0, SDValue V1) { 1639 SDLoc dl(V0.getNode()); 1640 SDValue RegClass = CurDAG->getTargetConstant(ARM::QQPRRegClassID, dl, 1641 MVT::i32); 1642 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::qsub_0, dl, MVT::i32); 1643 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::qsub_1, dl, MVT::i32); 1644 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1645 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1646 } 1647 1648 /// \brief Form 4 consecutive S registers. 1649 SDNode *ARMDAGToDAGISel::createQuadSRegsNode(EVT VT, SDValue V0, SDValue V1, 1650 SDValue V2, SDValue V3) { 1651 SDLoc dl(V0.getNode()); 1652 SDValue RegClass = 1653 CurDAG->getTargetConstant(ARM::QPR_VFP2RegClassID, dl, MVT::i32); 1654 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::ssub_0, dl, MVT::i32); 1655 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::ssub_1, dl, MVT::i32); 1656 SDValue SubReg2 = CurDAG->getTargetConstant(ARM::ssub_2, dl, MVT::i32); 1657 SDValue SubReg3 = CurDAG->getTargetConstant(ARM::ssub_3, dl, MVT::i32); 1658 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1, 1659 V2, SubReg2, V3, SubReg3 }; 1660 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1661 } 1662 1663 /// \brief Form 4 consecutive D registers. 1664 SDNode *ARMDAGToDAGISel::createQuadDRegsNode(EVT VT, SDValue V0, SDValue V1, 1665 SDValue V2, SDValue V3) { 1666 SDLoc dl(V0.getNode()); 1667 SDValue RegClass = CurDAG->getTargetConstant(ARM::QQPRRegClassID, dl, 1668 MVT::i32); 1669 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::dsub_0, dl, MVT::i32); 1670 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::dsub_1, dl, MVT::i32); 1671 SDValue SubReg2 = CurDAG->getTargetConstant(ARM::dsub_2, dl, MVT::i32); 1672 SDValue SubReg3 = CurDAG->getTargetConstant(ARM::dsub_3, dl, MVT::i32); 1673 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1, 1674 V2, SubReg2, V3, SubReg3 }; 1675 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1676 } 1677 1678 /// \brief Form 4 consecutive Q registers. 1679 SDNode *ARMDAGToDAGISel::createQuadQRegsNode(EVT VT, SDValue V0, SDValue V1, 1680 SDValue V2, SDValue V3) { 1681 SDLoc dl(V0.getNode()); 1682 SDValue RegClass = CurDAG->getTargetConstant(ARM::QQQQPRRegClassID, dl, 1683 MVT::i32); 1684 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::qsub_0, dl, MVT::i32); 1685 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::qsub_1, dl, MVT::i32); 1686 SDValue SubReg2 = CurDAG->getTargetConstant(ARM::qsub_2, dl, MVT::i32); 1687 SDValue SubReg3 = CurDAG->getTargetConstant(ARM::qsub_3, dl, MVT::i32); 1688 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1, 1689 V2, SubReg2, V3, SubReg3 }; 1690 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1691 } 1692 1693 /// GetVLDSTAlign - Get the alignment (in bytes) for the alignment operand 1694 /// of a NEON VLD or VST instruction. The supported values depend on the 1695 /// number of registers being loaded. 1696 SDValue ARMDAGToDAGISel::GetVLDSTAlign(SDValue Align, SDLoc dl, 1697 unsigned NumVecs, bool is64BitVector) { 1698 unsigned NumRegs = NumVecs; 1699 if (!is64BitVector && NumVecs < 3) 1700 NumRegs *= 2; 1701 1702 unsigned Alignment = cast<ConstantSDNode>(Align)->getZExtValue(); 1703 if (Alignment >= 32 && NumRegs == 4) 1704 Alignment = 32; 1705 else if (Alignment >= 16 && (NumRegs == 2 || NumRegs == 4)) 1706 Alignment = 16; 1707 else if (Alignment >= 8) 1708 Alignment = 8; 1709 else 1710 Alignment = 0; 1711 1712 return CurDAG->getTargetConstant(Alignment, dl, MVT::i32); 1713 } 1714 1715 static bool isVLDfixed(unsigned Opc) 1716 { 1717 switch (Opc) { 1718 default: return false; 1719 case ARM::VLD1d8wb_fixed : return true; 1720 case ARM::VLD1d16wb_fixed : return true; 1721 case ARM::VLD1d64Qwb_fixed : return true; 1722 case ARM::VLD1d32wb_fixed : return true; 1723 case ARM::VLD1d64wb_fixed : return true; 1724 case ARM::VLD1d64TPseudoWB_fixed : return true; 1725 case ARM::VLD1d64QPseudoWB_fixed : return true; 1726 case ARM::VLD1q8wb_fixed : return true; 1727 case ARM::VLD1q16wb_fixed : return true; 1728 case ARM::VLD1q32wb_fixed : return true; 1729 case ARM::VLD1q64wb_fixed : return true; 1730 case ARM::VLD2d8wb_fixed : return true; 1731 case ARM::VLD2d16wb_fixed : return true; 1732 case ARM::VLD2d32wb_fixed : return true; 1733 case ARM::VLD2q8PseudoWB_fixed : return true; 1734 case ARM::VLD2q16PseudoWB_fixed : return true; 1735 case ARM::VLD2q32PseudoWB_fixed : return true; 1736 case ARM::VLD2DUPd8wb_fixed : return true; 1737 case ARM::VLD2DUPd16wb_fixed : return true; 1738 case ARM::VLD2DUPd32wb_fixed : return true; 1739 } 1740 } 1741 1742 static bool isVSTfixed(unsigned Opc) 1743 { 1744 switch (Opc) { 1745 default: return false; 1746 case ARM::VST1d8wb_fixed : return true; 1747 case ARM::VST1d16wb_fixed : return true; 1748 case ARM::VST1d32wb_fixed : return true; 1749 case ARM::VST1d64wb_fixed : return true; 1750 case ARM::VST1q8wb_fixed : return true; 1751 case ARM::VST1q16wb_fixed : return true; 1752 case ARM::VST1q32wb_fixed : return true; 1753 case ARM::VST1q64wb_fixed : return true; 1754 case ARM::VST1d64TPseudoWB_fixed : return true; 1755 case ARM::VST1d64QPseudoWB_fixed : return true; 1756 case ARM::VST2d8wb_fixed : return true; 1757 case ARM::VST2d16wb_fixed : return true; 1758 case ARM::VST2d32wb_fixed : return true; 1759 case ARM::VST2q8PseudoWB_fixed : return true; 1760 case ARM::VST2q16PseudoWB_fixed : return true; 1761 case ARM::VST2q32PseudoWB_fixed : return true; 1762 } 1763 } 1764 1765 // Get the register stride update opcode of a VLD/VST instruction that 1766 // is otherwise equivalent to the given fixed stride updating instruction. 1767 static unsigned getVLDSTRegisterUpdateOpcode(unsigned Opc) { 1768 assert((isVLDfixed(Opc) || isVSTfixed(Opc)) 1769 && "Incorrect fixed stride updating instruction."); 1770 switch (Opc) { 1771 default: break; 1772 case ARM::VLD1d8wb_fixed: return ARM::VLD1d8wb_register; 1773 case ARM::VLD1d16wb_fixed: return ARM::VLD1d16wb_register; 1774 case ARM::VLD1d32wb_fixed: return ARM::VLD1d32wb_register; 1775 case ARM::VLD1d64wb_fixed: return ARM::VLD1d64wb_register; 1776 case ARM::VLD1q8wb_fixed: return ARM::VLD1q8wb_register; 1777 case ARM::VLD1q16wb_fixed: return ARM::VLD1q16wb_register; 1778 case ARM::VLD1q32wb_fixed: return ARM::VLD1q32wb_register; 1779 case ARM::VLD1q64wb_fixed: return ARM::VLD1q64wb_register; 1780 case ARM::VLD1d64Twb_fixed: return ARM::VLD1d64Twb_register; 1781 case ARM::VLD1d64Qwb_fixed: return ARM::VLD1d64Qwb_register; 1782 case ARM::VLD1d64TPseudoWB_fixed: return ARM::VLD1d64TPseudoWB_register; 1783 case ARM::VLD1d64QPseudoWB_fixed: return ARM::VLD1d64QPseudoWB_register; 1784 1785 case ARM::VST1d8wb_fixed: return ARM::VST1d8wb_register; 1786 case ARM::VST1d16wb_fixed: return ARM::VST1d16wb_register; 1787 case ARM::VST1d32wb_fixed: return ARM::VST1d32wb_register; 1788 case ARM::VST1d64wb_fixed: return ARM::VST1d64wb_register; 1789 case ARM::VST1q8wb_fixed: return ARM::VST1q8wb_register; 1790 case ARM::VST1q16wb_fixed: return ARM::VST1q16wb_register; 1791 case ARM::VST1q32wb_fixed: return ARM::VST1q32wb_register; 1792 case ARM::VST1q64wb_fixed: return ARM::VST1q64wb_register; 1793 case ARM::VST1d64TPseudoWB_fixed: return ARM::VST1d64TPseudoWB_register; 1794 case ARM::VST1d64QPseudoWB_fixed: return ARM::VST1d64QPseudoWB_register; 1795 1796 case ARM::VLD2d8wb_fixed: return ARM::VLD2d8wb_register; 1797 case ARM::VLD2d16wb_fixed: return ARM::VLD2d16wb_register; 1798 case ARM::VLD2d32wb_fixed: return ARM::VLD2d32wb_register; 1799 case ARM::VLD2q8PseudoWB_fixed: return ARM::VLD2q8PseudoWB_register; 1800 case ARM::VLD2q16PseudoWB_fixed: return ARM::VLD2q16PseudoWB_register; 1801 case ARM::VLD2q32PseudoWB_fixed: return ARM::VLD2q32PseudoWB_register; 1802 1803 case ARM::VST2d8wb_fixed: return ARM::VST2d8wb_register; 1804 case ARM::VST2d16wb_fixed: return ARM::VST2d16wb_register; 1805 case ARM::VST2d32wb_fixed: return ARM::VST2d32wb_register; 1806 case ARM::VST2q8PseudoWB_fixed: return ARM::VST2q8PseudoWB_register; 1807 case ARM::VST2q16PseudoWB_fixed: return ARM::VST2q16PseudoWB_register; 1808 case ARM::VST2q32PseudoWB_fixed: return ARM::VST2q32PseudoWB_register; 1809 1810 case ARM::VLD2DUPd8wb_fixed: return ARM::VLD2DUPd8wb_register; 1811 case ARM::VLD2DUPd16wb_fixed: return ARM::VLD2DUPd16wb_register; 1812 case ARM::VLD2DUPd32wb_fixed: return ARM::VLD2DUPd32wb_register; 1813 } 1814 return Opc; // If not one we handle, return it unchanged. 1815 } 1816 1817 SDNode *ARMDAGToDAGISel::SelectVLD(SDNode *N, bool isUpdating, unsigned NumVecs, 1818 const uint16_t *DOpcodes, 1819 const uint16_t *QOpcodes0, 1820 const uint16_t *QOpcodes1) { 1821 assert(NumVecs >= 1 && NumVecs <= 4 && "VLD NumVecs out-of-range"); 1822 SDLoc dl(N); 1823 1824 SDValue MemAddr, Align; 1825 unsigned AddrOpIdx = isUpdating ? 1 : 2; 1826 if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align)) 1827 return nullptr; 1828 1829 SDValue Chain = N->getOperand(0); 1830 EVT VT = N->getValueType(0); 1831 bool is64BitVector = VT.is64BitVector(); 1832 Align = GetVLDSTAlign(Align, dl, NumVecs, is64BitVector); 1833 1834 unsigned OpcodeIndex; 1835 switch (VT.getSimpleVT().SimpleTy) { 1836 default: llvm_unreachable("unhandled vld type"); 1837 // Double-register operations: 1838 case MVT::v8i8: OpcodeIndex = 0; break; 1839 case MVT::v4i16: OpcodeIndex = 1; break; 1840 case MVT::v2f32: 1841 case MVT::v2i32: OpcodeIndex = 2; break; 1842 case MVT::v1i64: OpcodeIndex = 3; break; 1843 // Quad-register operations: 1844 case MVT::v16i8: OpcodeIndex = 0; break; 1845 case MVT::v8i16: OpcodeIndex = 1; break; 1846 case MVT::v4f32: 1847 case MVT::v4i32: OpcodeIndex = 2; break; 1848 case MVT::v2f64: 1849 case MVT::v2i64: OpcodeIndex = 3; 1850 assert(NumVecs == 1 && "v2i64 type only supported for VLD1"); 1851 break; 1852 } 1853 1854 EVT ResTy; 1855 if (NumVecs == 1) 1856 ResTy = VT; 1857 else { 1858 unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs; 1859 if (!is64BitVector) 1860 ResTyElts *= 2; 1861 ResTy = EVT::getVectorVT(*CurDAG->getContext(), MVT::i64, ResTyElts); 1862 } 1863 std::vector<EVT> ResTys; 1864 ResTys.push_back(ResTy); 1865 if (isUpdating) 1866 ResTys.push_back(MVT::i32); 1867 ResTys.push_back(MVT::Other); 1868 1869 SDValue Pred = getAL(CurDAG, dl); 1870 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 1871 SDNode *VLd; 1872 SmallVector<SDValue, 7> Ops; 1873 1874 // Double registers and VLD1/VLD2 quad registers are directly supported. 1875 if (is64BitVector || NumVecs <= 2) { 1876 unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] : 1877 QOpcodes0[OpcodeIndex]); 1878 Ops.push_back(MemAddr); 1879 Ops.push_back(Align); 1880 if (isUpdating) { 1881 SDValue Inc = N->getOperand(AddrOpIdx + 1); 1882 // FIXME: VLD1/VLD2 fixed increment doesn't need Reg0. Remove the reg0 1883 // case entirely when the rest are updated to that form, too. 1884 if ((NumVecs <= 2) && !isa<ConstantSDNode>(Inc.getNode())) 1885 Opc = getVLDSTRegisterUpdateOpcode(Opc); 1886 // FIXME: We use a VLD1 for v1i64 even if the pseudo says vld2/3/4, so 1887 // check for that explicitly too. Horribly hacky, but temporary. 1888 if ((NumVecs > 2 && !isVLDfixed(Opc)) || 1889 !isa<ConstantSDNode>(Inc.getNode())) 1890 Ops.push_back(isa<ConstantSDNode>(Inc.getNode()) ? Reg0 : Inc); 1891 } 1892 Ops.push_back(Pred); 1893 Ops.push_back(Reg0); 1894 Ops.push_back(Chain); 1895 VLd = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 1896 1897 } else { 1898 // Otherwise, quad registers are loaded with two separate instructions, 1899 // where one loads the even registers and the other loads the odd registers. 1900 EVT AddrTy = MemAddr.getValueType(); 1901 1902 // Load the even subregs. This is always an updating load, so that it 1903 // provides the address to the second load for the odd subregs. 1904 SDValue ImplDef = 1905 SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, ResTy), 0); 1906 const SDValue OpsA[] = { MemAddr, Align, Reg0, ImplDef, Pred, Reg0, Chain }; 1907 SDNode *VLdA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], dl, 1908 ResTy, AddrTy, MVT::Other, OpsA); 1909 Chain = SDValue(VLdA, 2); 1910 1911 // Load the odd subregs. 1912 Ops.push_back(SDValue(VLdA, 1)); 1913 Ops.push_back(Align); 1914 if (isUpdating) { 1915 SDValue Inc = N->getOperand(AddrOpIdx + 1); 1916 assert(isa<ConstantSDNode>(Inc.getNode()) && 1917 "only constant post-increment update allowed for VLD3/4"); 1918 (void)Inc; 1919 Ops.push_back(Reg0); 1920 } 1921 Ops.push_back(SDValue(VLdA, 0)); 1922 Ops.push_back(Pred); 1923 Ops.push_back(Reg0); 1924 Ops.push_back(Chain); 1925 VLd = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys, Ops); 1926 } 1927 1928 // Transfer memoperands. 1929 MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1); 1930 MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 1931 cast<MachineSDNode>(VLd)->setMemRefs(MemOp, MemOp + 1); 1932 1933 if (NumVecs == 1) 1934 return VLd; 1935 1936 // Extract out the subregisters. 1937 SDValue SuperReg = SDValue(VLd, 0); 1938 assert(ARM::dsub_7 == ARM::dsub_0+7 && 1939 ARM::qsub_3 == ARM::qsub_0+3 && "Unexpected subreg numbering"); 1940 unsigned Sub0 = (is64BitVector ? ARM::dsub_0 : ARM::qsub_0); 1941 for (unsigned Vec = 0; Vec < NumVecs; ++Vec) 1942 ReplaceUses(SDValue(N, Vec), 1943 CurDAG->getTargetExtractSubreg(Sub0 + Vec, dl, VT, SuperReg)); 1944 ReplaceUses(SDValue(N, NumVecs), SDValue(VLd, 1)); 1945 if (isUpdating) 1946 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLd, 2)); 1947 return nullptr; 1948 } 1949 1950 SDNode *ARMDAGToDAGISel::SelectVST(SDNode *N, bool isUpdating, unsigned NumVecs, 1951 const uint16_t *DOpcodes, 1952 const uint16_t *QOpcodes0, 1953 const uint16_t *QOpcodes1) { 1954 assert(NumVecs >= 1 && NumVecs <= 4 && "VST NumVecs out-of-range"); 1955 SDLoc dl(N); 1956 1957 SDValue MemAddr, Align; 1958 unsigned AddrOpIdx = isUpdating ? 1 : 2; 1959 unsigned Vec0Idx = 3; // AddrOpIdx + (isUpdating ? 2 : 1) 1960 if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align)) 1961 return nullptr; 1962 1963 MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1); 1964 MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 1965 1966 SDValue Chain = N->getOperand(0); 1967 EVT VT = N->getOperand(Vec0Idx).getValueType(); 1968 bool is64BitVector = VT.is64BitVector(); 1969 Align = GetVLDSTAlign(Align, dl, NumVecs, is64BitVector); 1970 1971 unsigned OpcodeIndex; 1972 switch (VT.getSimpleVT().SimpleTy) { 1973 default: llvm_unreachable("unhandled vst type"); 1974 // Double-register operations: 1975 case MVT::v8i8: OpcodeIndex = 0; break; 1976 case MVT::v4i16: OpcodeIndex = 1; break; 1977 case MVT::v2f32: 1978 case MVT::v2i32: OpcodeIndex = 2; break; 1979 case MVT::v1i64: OpcodeIndex = 3; break; 1980 // Quad-register operations: 1981 case MVT::v16i8: OpcodeIndex = 0; break; 1982 case MVT::v8i16: OpcodeIndex = 1; break; 1983 case MVT::v4f32: 1984 case MVT::v4i32: OpcodeIndex = 2; break; 1985 case MVT::v2f64: 1986 case MVT::v2i64: OpcodeIndex = 3; 1987 assert(NumVecs == 1 && "v2i64 type only supported for VST1"); 1988 break; 1989 } 1990 1991 std::vector<EVT> ResTys; 1992 if (isUpdating) 1993 ResTys.push_back(MVT::i32); 1994 ResTys.push_back(MVT::Other); 1995 1996 SDValue Pred = getAL(CurDAG, dl); 1997 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 1998 SmallVector<SDValue, 7> Ops; 1999 2000 // Double registers and VST1/VST2 quad registers are directly supported. 2001 if (is64BitVector || NumVecs <= 2) { 2002 SDValue SrcReg; 2003 if (NumVecs == 1) { 2004 SrcReg = N->getOperand(Vec0Idx); 2005 } else if (is64BitVector) { 2006 // Form a REG_SEQUENCE to force register allocation. 2007 SDValue V0 = N->getOperand(Vec0Idx + 0); 2008 SDValue V1 = N->getOperand(Vec0Idx + 1); 2009 if (NumVecs == 2) 2010 SrcReg = SDValue(createDRegPairNode(MVT::v2i64, V0, V1), 0); 2011 else { 2012 SDValue V2 = N->getOperand(Vec0Idx + 2); 2013 // If it's a vst3, form a quad D-register and leave the last part as 2014 // an undef. 2015 SDValue V3 = (NumVecs == 3) 2016 ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,dl,VT), 0) 2017 : N->getOperand(Vec0Idx + 3); 2018 SrcReg = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0); 2019 } 2020 } else { 2021 // Form a QQ register. 2022 SDValue Q0 = N->getOperand(Vec0Idx); 2023 SDValue Q1 = N->getOperand(Vec0Idx + 1); 2024 SrcReg = SDValue(createQRegPairNode(MVT::v4i64, Q0, Q1), 0); 2025 } 2026 2027 unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] : 2028 QOpcodes0[OpcodeIndex]); 2029 Ops.push_back(MemAddr); 2030 Ops.push_back(Align); 2031 if (isUpdating) { 2032 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2033 // FIXME: VST1/VST2 fixed increment doesn't need Reg0. Remove the reg0 2034 // case entirely when the rest are updated to that form, too. 2035 if (NumVecs <= 2 && !isa<ConstantSDNode>(Inc.getNode())) 2036 Opc = getVLDSTRegisterUpdateOpcode(Opc); 2037 // FIXME: We use a VST1 for v1i64 even if the pseudo says vld2/3/4, so 2038 // check for that explicitly too. Horribly hacky, but temporary. 2039 if (!isa<ConstantSDNode>(Inc.getNode())) 2040 Ops.push_back(Inc); 2041 else if (NumVecs > 2 && !isVSTfixed(Opc)) 2042 Ops.push_back(Reg0); 2043 } 2044 Ops.push_back(SrcReg); 2045 Ops.push_back(Pred); 2046 Ops.push_back(Reg0); 2047 Ops.push_back(Chain); 2048 SDNode *VSt = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 2049 2050 // Transfer memoperands. 2051 cast<MachineSDNode>(VSt)->setMemRefs(MemOp, MemOp + 1); 2052 2053 return VSt; 2054 } 2055 2056 // Otherwise, quad registers are stored with two separate instructions, 2057 // where one stores the even registers and the other stores the odd registers. 2058 2059 // Form the QQQQ REG_SEQUENCE. 2060 SDValue V0 = N->getOperand(Vec0Idx + 0); 2061 SDValue V1 = N->getOperand(Vec0Idx + 1); 2062 SDValue V2 = N->getOperand(Vec0Idx + 2); 2063 SDValue V3 = (NumVecs == 3) 2064 ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0) 2065 : N->getOperand(Vec0Idx + 3); 2066 SDValue RegSeq = SDValue(createQuadQRegsNode(MVT::v8i64, V0, V1, V2, V3), 0); 2067 2068 // Store the even D registers. This is always an updating store, so that it 2069 // provides the address to the second store for the odd subregs. 2070 const SDValue OpsA[] = { MemAddr, Align, Reg0, RegSeq, Pred, Reg0, Chain }; 2071 SDNode *VStA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], dl, 2072 MemAddr.getValueType(), 2073 MVT::Other, OpsA); 2074 cast<MachineSDNode>(VStA)->setMemRefs(MemOp, MemOp + 1); 2075 Chain = SDValue(VStA, 1); 2076 2077 // Store the odd D registers. 2078 Ops.push_back(SDValue(VStA, 0)); 2079 Ops.push_back(Align); 2080 if (isUpdating) { 2081 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2082 assert(isa<ConstantSDNode>(Inc.getNode()) && 2083 "only constant post-increment update allowed for VST3/4"); 2084 (void)Inc; 2085 Ops.push_back(Reg0); 2086 } 2087 Ops.push_back(RegSeq); 2088 Ops.push_back(Pred); 2089 Ops.push_back(Reg0); 2090 Ops.push_back(Chain); 2091 SDNode *VStB = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys, 2092 Ops); 2093 cast<MachineSDNode>(VStB)->setMemRefs(MemOp, MemOp + 1); 2094 return VStB; 2095 } 2096 2097 SDNode *ARMDAGToDAGISel::SelectVLDSTLane(SDNode *N, bool IsLoad, 2098 bool isUpdating, unsigned NumVecs, 2099 const uint16_t *DOpcodes, 2100 const uint16_t *QOpcodes) { 2101 assert(NumVecs >=2 && NumVecs <= 4 && "VLDSTLane NumVecs out-of-range"); 2102 SDLoc dl(N); 2103 2104 SDValue MemAddr, Align; 2105 unsigned AddrOpIdx = isUpdating ? 1 : 2; 2106 unsigned Vec0Idx = 3; // AddrOpIdx + (isUpdating ? 2 : 1) 2107 if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align)) 2108 return nullptr; 2109 2110 MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1); 2111 MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 2112 2113 SDValue Chain = N->getOperand(0); 2114 unsigned Lane = 2115 cast<ConstantSDNode>(N->getOperand(Vec0Idx + NumVecs))->getZExtValue(); 2116 EVT VT = N->getOperand(Vec0Idx).getValueType(); 2117 bool is64BitVector = VT.is64BitVector(); 2118 2119 unsigned Alignment = 0; 2120 if (NumVecs != 3) { 2121 Alignment = cast<ConstantSDNode>(Align)->getZExtValue(); 2122 unsigned NumBytes = NumVecs * VT.getVectorElementType().getSizeInBits()/8; 2123 if (Alignment > NumBytes) 2124 Alignment = NumBytes; 2125 if (Alignment < 8 && Alignment < NumBytes) 2126 Alignment = 0; 2127 // Alignment must be a power of two; make sure of that. 2128 Alignment = (Alignment & -Alignment); 2129 if (Alignment == 1) 2130 Alignment = 0; 2131 } 2132 Align = CurDAG->getTargetConstant(Alignment, dl, MVT::i32); 2133 2134 unsigned OpcodeIndex; 2135 switch (VT.getSimpleVT().SimpleTy) { 2136 default: llvm_unreachable("unhandled vld/vst lane type"); 2137 // Double-register operations: 2138 case MVT::v8i8: OpcodeIndex = 0; break; 2139 case MVT::v4i16: OpcodeIndex = 1; break; 2140 case MVT::v2f32: 2141 case MVT::v2i32: OpcodeIndex = 2; break; 2142 // Quad-register operations: 2143 case MVT::v8i16: OpcodeIndex = 0; break; 2144 case MVT::v4f32: 2145 case MVT::v4i32: OpcodeIndex = 1; break; 2146 } 2147 2148 std::vector<EVT> ResTys; 2149 if (IsLoad) { 2150 unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs; 2151 if (!is64BitVector) 2152 ResTyElts *= 2; 2153 ResTys.push_back(EVT::getVectorVT(*CurDAG->getContext(), 2154 MVT::i64, ResTyElts)); 2155 } 2156 if (isUpdating) 2157 ResTys.push_back(MVT::i32); 2158 ResTys.push_back(MVT::Other); 2159 2160 SDValue Pred = getAL(CurDAG, dl); 2161 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2162 2163 SmallVector<SDValue, 8> Ops; 2164 Ops.push_back(MemAddr); 2165 Ops.push_back(Align); 2166 if (isUpdating) { 2167 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2168 Ops.push_back(isa<ConstantSDNode>(Inc.getNode()) ? Reg0 : Inc); 2169 } 2170 2171 SDValue SuperReg; 2172 SDValue V0 = N->getOperand(Vec0Idx + 0); 2173 SDValue V1 = N->getOperand(Vec0Idx + 1); 2174 if (NumVecs == 2) { 2175 if (is64BitVector) 2176 SuperReg = SDValue(createDRegPairNode(MVT::v2i64, V0, V1), 0); 2177 else 2178 SuperReg = SDValue(createQRegPairNode(MVT::v4i64, V0, V1), 0); 2179 } else { 2180 SDValue V2 = N->getOperand(Vec0Idx + 2); 2181 SDValue V3 = (NumVecs == 3) 2182 ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0) 2183 : N->getOperand(Vec0Idx + 3); 2184 if (is64BitVector) 2185 SuperReg = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0); 2186 else 2187 SuperReg = SDValue(createQuadQRegsNode(MVT::v8i64, V0, V1, V2, V3), 0); 2188 } 2189 Ops.push_back(SuperReg); 2190 Ops.push_back(getI32Imm(Lane, dl)); 2191 Ops.push_back(Pred); 2192 Ops.push_back(Reg0); 2193 Ops.push_back(Chain); 2194 2195 unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] : 2196 QOpcodes[OpcodeIndex]); 2197 SDNode *VLdLn = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 2198 cast<MachineSDNode>(VLdLn)->setMemRefs(MemOp, MemOp + 1); 2199 if (!IsLoad) 2200 return VLdLn; 2201 2202 // Extract the subregisters. 2203 SuperReg = SDValue(VLdLn, 0); 2204 assert(ARM::dsub_7 == ARM::dsub_0+7 && 2205 ARM::qsub_3 == ARM::qsub_0+3 && "Unexpected subreg numbering"); 2206 unsigned Sub0 = is64BitVector ? ARM::dsub_0 : ARM::qsub_0; 2207 for (unsigned Vec = 0; Vec < NumVecs; ++Vec) 2208 ReplaceUses(SDValue(N, Vec), 2209 CurDAG->getTargetExtractSubreg(Sub0 + Vec, dl, VT, SuperReg)); 2210 ReplaceUses(SDValue(N, NumVecs), SDValue(VLdLn, 1)); 2211 if (isUpdating) 2212 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLdLn, 2)); 2213 return nullptr; 2214 } 2215 2216 SDNode *ARMDAGToDAGISel::SelectVLDDup(SDNode *N, bool isUpdating, 2217 unsigned NumVecs, 2218 const uint16_t *Opcodes) { 2219 assert(NumVecs >=2 && NumVecs <= 4 && "VLDDup NumVecs out-of-range"); 2220 SDLoc dl(N); 2221 2222 SDValue MemAddr, Align; 2223 if (!SelectAddrMode6(N, N->getOperand(1), MemAddr, Align)) 2224 return nullptr; 2225 2226 MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1); 2227 MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 2228 2229 SDValue Chain = N->getOperand(0); 2230 EVT VT = N->getValueType(0); 2231 2232 unsigned Alignment = 0; 2233 if (NumVecs != 3) { 2234 Alignment = cast<ConstantSDNode>(Align)->getZExtValue(); 2235 unsigned NumBytes = NumVecs * VT.getVectorElementType().getSizeInBits()/8; 2236 if (Alignment > NumBytes) 2237 Alignment = NumBytes; 2238 if (Alignment < 8 && Alignment < NumBytes) 2239 Alignment = 0; 2240 // Alignment must be a power of two; make sure of that. 2241 Alignment = (Alignment & -Alignment); 2242 if (Alignment == 1) 2243 Alignment = 0; 2244 } 2245 Align = CurDAG->getTargetConstant(Alignment, dl, MVT::i32); 2246 2247 unsigned OpcodeIndex; 2248 switch (VT.getSimpleVT().SimpleTy) { 2249 default: llvm_unreachable("unhandled vld-dup type"); 2250 case MVT::v8i8: OpcodeIndex = 0; break; 2251 case MVT::v4i16: OpcodeIndex = 1; break; 2252 case MVT::v2f32: 2253 case MVT::v2i32: OpcodeIndex = 2; break; 2254 } 2255 2256 SDValue Pred = getAL(CurDAG, dl); 2257 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2258 SDValue SuperReg; 2259 unsigned Opc = Opcodes[OpcodeIndex]; 2260 SmallVector<SDValue, 6> Ops; 2261 Ops.push_back(MemAddr); 2262 Ops.push_back(Align); 2263 if (isUpdating) { 2264 // fixed-stride update instructions don't have an explicit writeback 2265 // operand. It's implicit in the opcode itself. 2266 SDValue Inc = N->getOperand(2); 2267 if (!isa<ConstantSDNode>(Inc.getNode())) 2268 Ops.push_back(Inc); 2269 // FIXME: VLD3 and VLD4 haven't been updated to that form yet. 2270 else if (NumVecs > 2) 2271 Ops.push_back(Reg0); 2272 } 2273 Ops.push_back(Pred); 2274 Ops.push_back(Reg0); 2275 Ops.push_back(Chain); 2276 2277 unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs; 2278 std::vector<EVT> ResTys; 2279 ResTys.push_back(EVT::getVectorVT(*CurDAG->getContext(), MVT::i64,ResTyElts)); 2280 if (isUpdating) 2281 ResTys.push_back(MVT::i32); 2282 ResTys.push_back(MVT::Other); 2283 SDNode *VLdDup = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 2284 cast<MachineSDNode>(VLdDup)->setMemRefs(MemOp, MemOp + 1); 2285 SuperReg = SDValue(VLdDup, 0); 2286 2287 // Extract the subregisters. 2288 assert(ARM::dsub_7 == ARM::dsub_0+7 && "Unexpected subreg numbering"); 2289 unsigned SubIdx = ARM::dsub_0; 2290 for (unsigned Vec = 0; Vec < NumVecs; ++Vec) 2291 ReplaceUses(SDValue(N, Vec), 2292 CurDAG->getTargetExtractSubreg(SubIdx+Vec, dl, VT, SuperReg)); 2293 ReplaceUses(SDValue(N, NumVecs), SDValue(VLdDup, 1)); 2294 if (isUpdating) 2295 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLdDup, 2)); 2296 return nullptr; 2297 } 2298 2299 SDNode *ARMDAGToDAGISel::SelectVTBL(SDNode *N, bool IsExt, unsigned NumVecs, 2300 unsigned Opc) { 2301 assert(NumVecs >= 2 && NumVecs <= 4 && "VTBL NumVecs out-of-range"); 2302 SDLoc dl(N); 2303 EVT VT = N->getValueType(0); 2304 unsigned FirstTblReg = IsExt ? 2 : 1; 2305 2306 // Form a REG_SEQUENCE to force register allocation. 2307 SDValue RegSeq; 2308 SDValue V0 = N->getOperand(FirstTblReg + 0); 2309 SDValue V1 = N->getOperand(FirstTblReg + 1); 2310 if (NumVecs == 2) 2311 RegSeq = SDValue(createDRegPairNode(MVT::v16i8, V0, V1), 0); 2312 else { 2313 SDValue V2 = N->getOperand(FirstTblReg + 2); 2314 // If it's a vtbl3, form a quad D-register and leave the last part as 2315 // an undef. 2316 SDValue V3 = (NumVecs == 3) 2317 ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0) 2318 : N->getOperand(FirstTblReg + 3); 2319 RegSeq = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0); 2320 } 2321 2322 SmallVector<SDValue, 6> Ops; 2323 if (IsExt) 2324 Ops.push_back(N->getOperand(1)); 2325 Ops.push_back(RegSeq); 2326 Ops.push_back(N->getOperand(FirstTblReg + NumVecs)); 2327 Ops.push_back(getAL(CurDAG, dl)); // predicate 2328 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); // predicate register 2329 return CurDAG->getMachineNode(Opc, dl, VT, Ops); 2330 } 2331 2332 SDNode *ARMDAGToDAGISel::SelectV6T2BitfieldExtractOp(SDNode *N, 2333 bool isSigned) { 2334 if (!Subtarget->hasV6T2Ops()) 2335 return nullptr; 2336 2337 unsigned Opc = isSigned 2338 ? (Subtarget->isThumb() ? ARM::t2SBFX : ARM::SBFX) 2339 : (Subtarget->isThumb() ? ARM::t2UBFX : ARM::UBFX); 2340 SDLoc dl(N); 2341 2342 // For unsigned extracts, check for a shift right and mask 2343 unsigned And_imm = 0; 2344 if (N->getOpcode() == ISD::AND) { 2345 if (isOpcWithIntImmediate(N, ISD::AND, And_imm)) { 2346 2347 // The immediate is a mask of the low bits iff imm & (imm+1) == 0 2348 if (And_imm & (And_imm + 1)) 2349 return nullptr; 2350 2351 unsigned Srl_imm = 0; 2352 if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRL, 2353 Srl_imm)) { 2354 assert(Srl_imm > 0 && Srl_imm < 32 && "bad amount in shift node!"); 2355 2356 // Note: The width operand is encoded as width-1. 2357 unsigned Width = countTrailingOnes(And_imm) - 1; 2358 unsigned LSB = Srl_imm; 2359 2360 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2361 2362 if ((LSB + Width + 1) == N->getValueType(0).getSizeInBits()) { 2363 // It's cheaper to use a right shift to extract the top bits. 2364 if (Subtarget->isThumb()) { 2365 Opc = isSigned ? ARM::t2ASRri : ARM::t2LSRri; 2366 SDValue Ops[] = { N->getOperand(0).getOperand(0), 2367 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 2368 getAL(CurDAG, dl), Reg0, Reg0 }; 2369 return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 2370 } 2371 2372 // ARM models shift instructions as MOVsi with shifter operand. 2373 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(ISD::SRL); 2374 SDValue ShOpc = 2375 CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, LSB), dl, 2376 MVT::i32); 2377 SDValue Ops[] = { N->getOperand(0).getOperand(0), ShOpc, 2378 getAL(CurDAG, dl), Reg0, Reg0 }; 2379 return CurDAG->SelectNodeTo(N, ARM::MOVsi, MVT::i32, Ops); 2380 } 2381 2382 SDValue Ops[] = { N->getOperand(0).getOperand(0), 2383 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 2384 CurDAG->getTargetConstant(Width, dl, MVT::i32), 2385 getAL(CurDAG, dl), Reg0 }; 2386 return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 2387 } 2388 } 2389 return nullptr; 2390 } 2391 2392 // Otherwise, we're looking for a shift of a shift 2393 unsigned Shl_imm = 0; 2394 if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SHL, Shl_imm)) { 2395 assert(Shl_imm > 0 && Shl_imm < 32 && "bad amount in shift node!"); 2396 unsigned Srl_imm = 0; 2397 if (isInt32Immediate(N->getOperand(1), Srl_imm)) { 2398 assert(Srl_imm > 0 && Srl_imm < 32 && "bad amount in shift node!"); 2399 // Note: The width operand is encoded as width-1. 2400 unsigned Width = 32 - Srl_imm - 1; 2401 int LSB = Srl_imm - Shl_imm; 2402 if (LSB < 0) 2403 return nullptr; 2404 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2405 SDValue Ops[] = { N->getOperand(0).getOperand(0), 2406 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 2407 CurDAG->getTargetConstant(Width, dl, MVT::i32), 2408 getAL(CurDAG, dl), Reg0 }; 2409 return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 2410 } 2411 } 2412 2413 if (N->getOpcode() == ISD::SIGN_EXTEND_INREG) { 2414 unsigned Width = cast<VTSDNode>(N->getOperand(1))->getVT().getSizeInBits(); 2415 unsigned LSB = 0; 2416 if (!isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRL, LSB) && 2417 !isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRA, LSB)) 2418 return nullptr; 2419 2420 if (LSB + Width > 32) 2421 return nullptr; 2422 2423 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2424 SDValue Ops[] = { N->getOperand(0).getOperand(0), 2425 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 2426 CurDAG->getTargetConstant(Width - 1, dl, MVT::i32), 2427 getAL(CurDAG, dl), Reg0 }; 2428 return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 2429 } 2430 2431 return nullptr; 2432 } 2433 2434 /// Target-specific DAG combining for ISD::XOR. 2435 /// Target-independent combining lowers SELECT_CC nodes of the form 2436 /// select_cc setg[ge] X, 0, X, -X 2437 /// select_cc setgt X, -1, X, -X 2438 /// select_cc setl[te] X, 0, -X, X 2439 /// select_cc setlt X, 1, -X, X 2440 /// which represent Integer ABS into: 2441 /// Y = sra (X, size(X)-1); xor (add (X, Y), Y) 2442 /// ARM instruction selection detects the latter and matches it to 2443 /// ARM::ABS or ARM::t2ABS machine node. 2444 SDNode *ARMDAGToDAGISel::SelectABSOp(SDNode *N){ 2445 SDValue XORSrc0 = N->getOperand(0); 2446 SDValue XORSrc1 = N->getOperand(1); 2447 EVT VT = N->getValueType(0); 2448 2449 if (Subtarget->isThumb1Only()) 2450 return nullptr; 2451 2452 if (XORSrc0.getOpcode() != ISD::ADD || XORSrc1.getOpcode() != ISD::SRA) 2453 return nullptr; 2454 2455 SDValue ADDSrc0 = XORSrc0.getOperand(0); 2456 SDValue ADDSrc1 = XORSrc0.getOperand(1); 2457 SDValue SRASrc0 = XORSrc1.getOperand(0); 2458 SDValue SRASrc1 = XORSrc1.getOperand(1); 2459 ConstantSDNode *SRAConstant = dyn_cast<ConstantSDNode>(SRASrc1); 2460 EVT XType = SRASrc0.getValueType(); 2461 unsigned Size = XType.getSizeInBits() - 1; 2462 2463 if (ADDSrc1 == XORSrc1 && ADDSrc0 == SRASrc0 && 2464 XType.isInteger() && SRAConstant != nullptr && 2465 Size == SRAConstant->getZExtValue()) { 2466 unsigned Opcode = Subtarget->isThumb2() ? ARM::t2ABS : ARM::ABS; 2467 return CurDAG->SelectNodeTo(N, Opcode, VT, ADDSrc0); 2468 } 2469 2470 return nullptr; 2471 } 2472 2473 static bool SearchSignedMulShort(SDValue SignExt, unsigned *Opc, SDValue &Src1, 2474 bool Accumulate) { 2475 // For SM*WB, we need to some form of sext. 2476 // For SM*WT, we need to search for (sra X, 16) 2477 // Src1 then gets set to X. 2478 if ((SignExt.getOpcode() == ISD::SIGN_EXTEND || 2479 SignExt.getOpcode() == ISD::SIGN_EXTEND_INREG || 2480 SignExt.getOpcode() == ISD::AssertSext) && 2481 SignExt.getValueType() == MVT::i32) { 2482 2483 *Opc = Accumulate ? ARM::SMLAWB : ARM::SMULWB; 2484 Src1 = SignExt.getOperand(0); 2485 return true; 2486 } 2487 2488 if (SignExt.getOpcode() != ISD::SRA) 2489 return false; 2490 2491 ConstantSDNode *SRASrc1 = dyn_cast<ConstantSDNode>(SignExt.getOperand(1)); 2492 if (!SRASrc1 || SRASrc1->getZExtValue() != 16) 2493 return false; 2494 2495 SDValue Op0 = SignExt.getOperand(0); 2496 2497 // The sign extend operand for SM*WB could be generated by a shl and ashr. 2498 if (Op0.getOpcode() == ISD::SHL) { 2499 SDValue SHL = Op0; 2500 ConstantSDNode *SHLSrc1 = dyn_cast<ConstantSDNode>(SHL.getOperand(1)); 2501 if (!SHLSrc1 || SHLSrc1->getZExtValue() != 16) 2502 return false; 2503 2504 *Opc = Accumulate ? ARM::SMLAWB : ARM::SMULWB; 2505 Src1 = Op0.getOperand(0); 2506 return true; 2507 } 2508 *Opc = Accumulate ? ARM::SMLAWT : ARM::SMULWT; 2509 Src1 = SignExt.getOperand(0); 2510 return true; 2511 } 2512 2513 static bool SearchSignedMulLong(SDValue OR, unsigned *Opc, SDValue &Src0, 2514 SDValue &Src1, bool Accumulate) { 2515 // First we look for: 2516 // (add (or (srl ?, 16), (shl ?, 16))) 2517 if (OR.getOpcode() != ISD::OR) 2518 return false; 2519 2520 SDValue SRL = OR.getOperand(0); 2521 SDValue SHL = OR.getOperand(1); 2522 2523 if (SRL.getOpcode() != ISD::SRL || SHL.getOpcode() != ISD::SHL) { 2524 SRL = OR.getOperand(1); 2525 SHL = OR.getOperand(0); 2526 if (SRL.getOpcode() != ISD::SRL || SHL.getOpcode() != ISD::SHL) 2527 return false; 2528 } 2529 2530 ConstantSDNode *SRLSrc1 = dyn_cast<ConstantSDNode>(SRL.getOperand(1)); 2531 ConstantSDNode *SHLSrc1 = dyn_cast<ConstantSDNode>(SHL.getOperand(1)); 2532 if (!SRLSrc1 || !SHLSrc1 || SRLSrc1->getZExtValue() != 16 || 2533 SHLSrc1->getZExtValue() != 16) 2534 return false; 2535 2536 // The first operands to the shifts need to be the two results from the 2537 // same smul_lohi node. 2538 if ((SRL.getOperand(0).getNode() != SHL.getOperand(0).getNode()) || 2539 SRL.getOperand(0).getOpcode() != ISD::SMUL_LOHI) 2540 return false; 2541 2542 SDNode *SMULLOHI = SRL.getOperand(0).getNode(); 2543 if (SRL.getOperand(0) != SDValue(SMULLOHI, 0) || 2544 SHL.getOperand(0) != SDValue(SMULLOHI, 1)) 2545 return false; 2546 2547 // Now we have: 2548 // (add (or (srl (smul_lohi ?, ?), 16), (shl (smul_lohi ?, ?), 16))) 2549 // For SMLAW[B|T] smul_lohi will take a 32-bit and a 16-bit arguments. 2550 // For SMLAWB the 16-bit value will signed extended somehow. 2551 // For SMLAWT only the SRA is required. 2552 2553 // Check both sides of SMUL_LOHI 2554 if (SearchSignedMulShort(SMULLOHI->getOperand(0), Opc, Src1, Accumulate)) { 2555 Src0 = SMULLOHI->getOperand(1); 2556 } else if (SearchSignedMulShort(SMULLOHI->getOperand(1), Opc, Src1, 2557 Accumulate)) { 2558 Src0 = SMULLOHI->getOperand(0); 2559 } else { 2560 return false; 2561 } 2562 return true; 2563 } 2564 2565 SDNode *ARMDAGToDAGISel::SelectSMLAWSMULW(SDNode *N) { 2566 SDLoc dl(N); 2567 SDValue Src0 = N->getOperand(0); 2568 SDValue Src1 = N->getOperand(1); 2569 SDValue A, B; 2570 unsigned Opc = 0; 2571 2572 if (N->getOpcode() == ISD::ADD) { 2573 if (Src0.getOpcode() != ISD::OR && Src1.getOpcode() != ISD::OR) 2574 return nullptr; 2575 2576 SDValue Acc; 2577 if (SearchSignedMulLong(Src0, &Opc, A, B, true)) { 2578 Acc = Src1; 2579 } else if (SearchSignedMulLong(Src1, &Opc, A, B, true)) { 2580 Acc = Src0; 2581 } else { 2582 return nullptr; 2583 } 2584 if (Opc == 0) 2585 return nullptr; 2586 2587 SDValue Ops[] = { A, B, Acc, getAL(CurDAG, dl), 2588 CurDAG->getRegister(0, MVT::i32) }; 2589 return CurDAG->SelectNodeTo(N, Opc, MVT::i32, MVT::Other, Ops); 2590 } else if (N->getOpcode() == ISD::OR && 2591 SearchSignedMulLong(SDValue(N, 0), &Opc, A, B, false)) { 2592 if (Opc == 0) 2593 return nullptr; 2594 2595 SDValue Ops[] = { A, B, getAL(CurDAG, dl), 2596 CurDAG->getRegister(0, MVT::i32)}; 2597 return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 2598 } 2599 return nullptr; 2600 } 2601 2602 /// We've got special pseudo-instructions for these 2603 SDNode *ARMDAGToDAGISel::SelectCMP_SWAP(SDNode *N) { 2604 unsigned Opcode; 2605 EVT MemTy = cast<MemSDNode>(N)->getMemoryVT(); 2606 if (MemTy == MVT::i8) 2607 Opcode = ARM::CMP_SWAP_8; 2608 else if (MemTy == MVT::i16) 2609 Opcode = ARM::CMP_SWAP_16; 2610 else if (MemTy == MVT::i32) 2611 Opcode = ARM::CMP_SWAP_32; 2612 else 2613 llvm_unreachable("Unknown AtomicCmpSwap type"); 2614 2615 SDValue Ops[] = {N->getOperand(1), N->getOperand(2), N->getOperand(3), 2616 N->getOperand(0)}; 2617 SDNode *CmpSwap = CurDAG->getMachineNode( 2618 Opcode, SDLoc(N), 2619 CurDAG->getVTList(MVT::i32, MVT::i32, MVT::Other), Ops); 2620 2621 MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1); 2622 MemOp[0] = cast<MemSDNode>(N)->getMemOperand(); 2623 cast<MachineSDNode>(CmpSwap)->setMemRefs(MemOp, MemOp + 1); 2624 2625 ReplaceUses(SDValue(N, 0), SDValue(CmpSwap, 0)); 2626 ReplaceUses(SDValue(N, 1), SDValue(CmpSwap, 2)); 2627 return nullptr; 2628 } 2629 2630 SDNode *ARMDAGToDAGISel::SelectConcatVector(SDNode *N) { 2631 // The only time a CONCAT_VECTORS operation can have legal types is when 2632 // two 64-bit vectors are concatenated to a 128-bit vector. 2633 EVT VT = N->getValueType(0); 2634 if (!VT.is128BitVector() || N->getNumOperands() != 2) 2635 llvm_unreachable("unexpected CONCAT_VECTORS"); 2636 return createDRegPairNode(VT, N->getOperand(0), N->getOperand(1)); 2637 } 2638 2639 SDNode *ARMDAGToDAGISel::Select(SDNode *N) { 2640 SDLoc dl(N); 2641 2642 if (N->isMachineOpcode()) { 2643 N->setNodeId(-1); 2644 return nullptr; // Already selected. 2645 } 2646 2647 switch (N->getOpcode()) { 2648 default: break; 2649 case ISD::ADD: 2650 case ISD::OR: { 2651 SDNode *ResNode = SelectSMLAWSMULW(N); 2652 if (ResNode) 2653 return ResNode; 2654 break; 2655 } 2656 case ISD::WRITE_REGISTER: { 2657 SDNode *ResNode = SelectWriteRegister(N); 2658 if (ResNode) 2659 return ResNode; 2660 break; 2661 } 2662 case ISD::READ_REGISTER: { 2663 SDNode *ResNode = SelectReadRegister(N); 2664 if (ResNode) 2665 return ResNode; 2666 break; 2667 } 2668 case ISD::INLINEASM: { 2669 SDNode *ResNode = SelectInlineAsm(N); 2670 if (ResNode) 2671 return ResNode; 2672 break; 2673 } 2674 case ISD::XOR: { 2675 // Select special operations if XOR node forms integer ABS pattern 2676 SDNode *ResNode = SelectABSOp(N); 2677 if (ResNode) 2678 return ResNode; 2679 // Other cases are autogenerated. 2680 break; 2681 } 2682 case ISD::Constant: { 2683 unsigned Val = cast<ConstantSDNode>(N)->getZExtValue(); 2684 // If we can't materialize the constant we need to use a literal pool 2685 if (ConstantMaterializationCost(Val) > 2) { 2686 SDValue CPIdx = CurDAG->getTargetConstantPool( 2687 ConstantInt::get(Type::getInt32Ty(*CurDAG->getContext()), Val), 2688 TLI->getPointerTy(CurDAG->getDataLayout())); 2689 2690 SDNode *ResNode; 2691 if (Subtarget->isThumb()) { 2692 SDValue Pred = getAL(CurDAG, dl); 2693 SDValue PredReg = CurDAG->getRegister(0, MVT::i32); 2694 SDValue Ops[] = { CPIdx, Pred, PredReg, CurDAG->getEntryNode() }; 2695 ResNode = CurDAG->getMachineNode(ARM::tLDRpci, dl, MVT::i32, MVT::Other, 2696 Ops); 2697 } else { 2698 SDValue Ops[] = { 2699 CPIdx, 2700 CurDAG->getTargetConstant(0, dl, MVT::i32), 2701 getAL(CurDAG, dl), 2702 CurDAG->getRegister(0, MVT::i32), 2703 CurDAG->getEntryNode() 2704 }; 2705 ResNode=CurDAG->getMachineNode(ARM::LDRcp, dl, MVT::i32, MVT::Other, 2706 Ops); 2707 } 2708 ReplaceUses(SDValue(N, 0), SDValue(ResNode, 0)); 2709 return nullptr; 2710 } 2711 2712 // Other cases are autogenerated. 2713 break; 2714 } 2715 case ISD::FrameIndex: { 2716 // Selects to ADDri FI, 0 which in turn will become ADDri SP, imm. 2717 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 2718 SDValue TFI = CurDAG->getTargetFrameIndex( 2719 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 2720 if (Subtarget->isThumb1Only()) { 2721 // Set the alignment of the frame object to 4, to avoid having to generate 2722 // more than one ADD 2723 MachineFrameInfo *MFI = MF->getFrameInfo(); 2724 if (MFI->getObjectAlignment(FI) < 4) 2725 MFI->setObjectAlignment(FI, 4); 2726 return CurDAG->SelectNodeTo(N, ARM::tADDframe, MVT::i32, TFI, 2727 CurDAG->getTargetConstant(0, dl, MVT::i32)); 2728 } else { 2729 unsigned Opc = ((Subtarget->isThumb() && Subtarget->hasThumb2()) ? 2730 ARM::t2ADDri : ARM::ADDri); 2731 SDValue Ops[] = { TFI, CurDAG->getTargetConstant(0, dl, MVT::i32), 2732 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32), 2733 CurDAG->getRegister(0, MVT::i32) }; 2734 return CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 2735 } 2736 } 2737 case ISD::SRL: 2738 if (SDNode *I = SelectV6T2BitfieldExtractOp(N, false)) 2739 return I; 2740 break; 2741 case ISD::SIGN_EXTEND_INREG: 2742 case ISD::SRA: 2743 if (SDNode *I = SelectV6T2BitfieldExtractOp(N, true)) 2744 return I; 2745 break; 2746 case ISD::MUL: 2747 if (Subtarget->isThumb1Only()) 2748 break; 2749 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(1))) { 2750 unsigned RHSV = C->getZExtValue(); 2751 if (!RHSV) break; 2752 if (isPowerOf2_32(RHSV-1)) { // 2^n+1? 2753 unsigned ShImm = Log2_32(RHSV-1); 2754 if (ShImm >= 32) 2755 break; 2756 SDValue V = N->getOperand(0); 2757 ShImm = ARM_AM::getSORegOpc(ARM_AM::lsl, ShImm); 2758 SDValue ShImmOp = CurDAG->getTargetConstant(ShImm, dl, MVT::i32); 2759 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2760 if (Subtarget->isThumb()) { 2761 SDValue Ops[] = { V, V, ShImmOp, getAL(CurDAG, dl), Reg0, Reg0 }; 2762 return CurDAG->SelectNodeTo(N, ARM::t2ADDrs, MVT::i32, Ops); 2763 } else { 2764 SDValue Ops[] = { V, V, Reg0, ShImmOp, getAL(CurDAG, dl), Reg0, 2765 Reg0 }; 2766 return CurDAG->SelectNodeTo(N, ARM::ADDrsi, MVT::i32, Ops); 2767 } 2768 } 2769 if (isPowerOf2_32(RHSV+1)) { // 2^n-1? 2770 unsigned ShImm = Log2_32(RHSV+1); 2771 if (ShImm >= 32) 2772 break; 2773 SDValue V = N->getOperand(0); 2774 ShImm = ARM_AM::getSORegOpc(ARM_AM::lsl, ShImm); 2775 SDValue ShImmOp = CurDAG->getTargetConstant(ShImm, dl, MVT::i32); 2776 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2777 if (Subtarget->isThumb()) { 2778 SDValue Ops[] = { V, V, ShImmOp, getAL(CurDAG, dl), Reg0, Reg0 }; 2779 return CurDAG->SelectNodeTo(N, ARM::t2RSBrs, MVT::i32, Ops); 2780 } else { 2781 SDValue Ops[] = { V, V, Reg0, ShImmOp, getAL(CurDAG, dl), Reg0, 2782 Reg0 }; 2783 return CurDAG->SelectNodeTo(N, ARM::RSBrsi, MVT::i32, Ops); 2784 } 2785 } 2786 } 2787 break; 2788 case ISD::AND: { 2789 // Check for unsigned bitfield extract 2790 if (SDNode *I = SelectV6T2BitfieldExtractOp(N, false)) 2791 return I; 2792 2793 // (and (or x, c2), c1) and top 16-bits of c1 and c2 match, lower 16-bits 2794 // of c1 are 0xffff, and lower 16-bit of c2 are 0. That is, the top 16-bits 2795 // are entirely contributed by c2 and lower 16-bits are entirely contributed 2796 // by x. That's equal to (or (and x, 0xffff), (and c1, 0xffff0000)). 2797 // Select it to: "movt x, ((c1 & 0xffff) >> 16) 2798 EVT VT = N->getValueType(0); 2799 if (VT != MVT::i32) 2800 break; 2801 unsigned Opc = (Subtarget->isThumb() && Subtarget->hasThumb2()) 2802 ? ARM::t2MOVTi16 2803 : (Subtarget->hasV6T2Ops() ? ARM::MOVTi16 : 0); 2804 if (!Opc) 2805 break; 2806 SDValue N0 = N->getOperand(0), N1 = N->getOperand(1); 2807 ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1); 2808 if (!N1C) 2809 break; 2810 if (N0.getOpcode() == ISD::OR && N0.getNode()->hasOneUse()) { 2811 SDValue N2 = N0.getOperand(1); 2812 ConstantSDNode *N2C = dyn_cast<ConstantSDNode>(N2); 2813 if (!N2C) 2814 break; 2815 unsigned N1CVal = N1C->getZExtValue(); 2816 unsigned N2CVal = N2C->getZExtValue(); 2817 if ((N1CVal & 0xffff0000U) == (N2CVal & 0xffff0000U) && 2818 (N1CVal & 0xffffU) == 0xffffU && 2819 (N2CVal & 0xffffU) == 0x0U) { 2820 SDValue Imm16 = CurDAG->getTargetConstant((N2CVal & 0xFFFF0000U) >> 16, 2821 dl, MVT::i32); 2822 SDValue Ops[] = { N0.getOperand(0), Imm16, 2823 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32) }; 2824 return CurDAG->getMachineNode(Opc, dl, VT, Ops); 2825 } 2826 } 2827 break; 2828 } 2829 case ARMISD::VMOVRRD: 2830 return CurDAG->getMachineNode(ARM::VMOVRRD, dl, MVT::i32, MVT::i32, 2831 N->getOperand(0), getAL(CurDAG, dl), 2832 CurDAG->getRegister(0, MVT::i32)); 2833 case ISD::UMUL_LOHI: { 2834 if (Subtarget->isThumb1Only()) 2835 break; 2836 if (Subtarget->isThumb()) { 2837 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), 2838 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32) }; 2839 return CurDAG->getMachineNode(ARM::t2UMULL, dl, MVT::i32, MVT::i32, Ops); 2840 } else { 2841 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), 2842 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32), 2843 CurDAG->getRegister(0, MVT::i32) }; 2844 return CurDAG->getMachineNode(Subtarget->hasV6Ops() ? 2845 ARM::UMULL : ARM::UMULLv5, 2846 dl, MVT::i32, MVT::i32, Ops); 2847 } 2848 } 2849 case ISD::SMUL_LOHI: { 2850 if (Subtarget->isThumb1Only()) 2851 break; 2852 if (Subtarget->isThumb()) { 2853 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), 2854 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32) }; 2855 return CurDAG->getMachineNode(ARM::t2SMULL, dl, MVT::i32, MVT::i32, Ops); 2856 } else { 2857 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), 2858 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32), 2859 CurDAG->getRegister(0, MVT::i32) }; 2860 return CurDAG->getMachineNode(Subtarget->hasV6Ops() ? 2861 ARM::SMULL : ARM::SMULLv5, 2862 dl, MVT::i32, MVT::i32, Ops); 2863 } 2864 } 2865 case ARMISD::UMLAL:{ 2866 if (Subtarget->isThumb()) { 2867 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 2868 N->getOperand(3), getAL(CurDAG, dl), 2869 CurDAG->getRegister(0, MVT::i32)}; 2870 return CurDAG->getMachineNode(ARM::t2UMLAL, dl, MVT::i32, MVT::i32, Ops); 2871 }else{ 2872 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 2873 N->getOperand(3), getAL(CurDAG, dl), 2874 CurDAG->getRegister(0, MVT::i32), 2875 CurDAG->getRegister(0, MVT::i32) }; 2876 return CurDAG->getMachineNode(Subtarget->hasV6Ops() ? 2877 ARM::UMLAL : ARM::UMLALv5, 2878 dl, MVT::i32, MVT::i32, Ops); 2879 } 2880 } 2881 case ARMISD::SMLAL:{ 2882 if (Subtarget->isThumb()) { 2883 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 2884 N->getOperand(3), getAL(CurDAG, dl), 2885 CurDAG->getRegister(0, MVT::i32)}; 2886 return CurDAG->getMachineNode(ARM::t2SMLAL, dl, MVT::i32, MVT::i32, Ops); 2887 }else{ 2888 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 2889 N->getOperand(3), getAL(CurDAG, dl), 2890 CurDAG->getRegister(0, MVT::i32), 2891 CurDAG->getRegister(0, MVT::i32) }; 2892 return CurDAG->getMachineNode(Subtarget->hasV6Ops() ? 2893 ARM::SMLAL : ARM::SMLALv5, 2894 dl, MVT::i32, MVT::i32, Ops); 2895 } 2896 } 2897 case ISD::LOAD: { 2898 SDNode *ResNode = nullptr; 2899 if (Subtarget->isThumb() && Subtarget->hasThumb2()) 2900 ResNode = SelectT2IndexedLoad(N); 2901 else 2902 ResNode = SelectARMIndexedLoad(N); 2903 if (ResNode) 2904 return ResNode; 2905 // Other cases are autogenerated. 2906 break; 2907 } 2908 case ARMISD::BRCOND: { 2909 // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc) 2910 // Emits: (Bcc:void (bb:Other):$dst, (imm:i32):$cc) 2911 // Pattern complexity = 6 cost = 1 size = 0 2912 2913 // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc) 2914 // Emits: (tBcc:void (bb:Other):$dst, (imm:i32):$cc) 2915 // Pattern complexity = 6 cost = 1 size = 0 2916 2917 // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc) 2918 // Emits: (t2Bcc:void (bb:Other):$dst, (imm:i32):$cc) 2919 // Pattern complexity = 6 cost = 1 size = 0 2920 2921 unsigned Opc = Subtarget->isThumb() ? 2922 ((Subtarget->hasThumb2()) ? ARM::t2Bcc : ARM::tBcc) : ARM::Bcc; 2923 SDValue Chain = N->getOperand(0); 2924 SDValue N1 = N->getOperand(1); 2925 SDValue N2 = N->getOperand(2); 2926 SDValue N3 = N->getOperand(3); 2927 SDValue InFlag = N->getOperand(4); 2928 assert(N1.getOpcode() == ISD::BasicBlock); 2929 assert(N2.getOpcode() == ISD::Constant); 2930 assert(N3.getOpcode() == ISD::Register); 2931 2932 SDValue Tmp2 = CurDAG->getTargetConstant(((unsigned) 2933 cast<ConstantSDNode>(N2)->getZExtValue()), dl, 2934 MVT::i32); 2935 SDValue Ops[] = { N1, Tmp2, N3, Chain, InFlag }; 2936 SDNode *ResNode = CurDAG->getMachineNode(Opc, dl, MVT::Other, 2937 MVT::Glue, Ops); 2938 Chain = SDValue(ResNode, 0); 2939 if (N->getNumValues() == 2) { 2940 InFlag = SDValue(ResNode, 1); 2941 ReplaceUses(SDValue(N, 1), InFlag); 2942 } 2943 ReplaceUses(SDValue(N, 0), 2944 SDValue(Chain.getNode(), Chain.getResNo())); 2945 return nullptr; 2946 } 2947 case ARMISD::VZIP: { 2948 unsigned Opc = 0; 2949 EVT VT = N->getValueType(0); 2950 switch (VT.getSimpleVT().SimpleTy) { 2951 default: return nullptr; 2952 case MVT::v8i8: Opc = ARM::VZIPd8; break; 2953 case MVT::v4i16: Opc = ARM::VZIPd16; break; 2954 case MVT::v2f32: 2955 // vzip.32 Dd, Dm is a pseudo-instruction expanded to vtrn.32 Dd, Dm. 2956 case MVT::v2i32: Opc = ARM::VTRNd32; break; 2957 case MVT::v16i8: Opc = ARM::VZIPq8; break; 2958 case MVT::v8i16: Opc = ARM::VZIPq16; break; 2959 case MVT::v4f32: 2960 case MVT::v4i32: Opc = ARM::VZIPq32; break; 2961 } 2962 SDValue Pred = getAL(CurDAG, dl); 2963 SDValue PredReg = CurDAG->getRegister(0, MVT::i32); 2964 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg }; 2965 return CurDAG->getMachineNode(Opc, dl, VT, VT, Ops); 2966 } 2967 case ARMISD::VUZP: { 2968 unsigned Opc = 0; 2969 EVT VT = N->getValueType(0); 2970 switch (VT.getSimpleVT().SimpleTy) { 2971 default: return nullptr; 2972 case MVT::v8i8: Opc = ARM::VUZPd8; break; 2973 case MVT::v4i16: Opc = ARM::VUZPd16; break; 2974 case MVT::v2f32: 2975 // vuzp.32 Dd, Dm is a pseudo-instruction expanded to vtrn.32 Dd, Dm. 2976 case MVT::v2i32: Opc = ARM::VTRNd32; break; 2977 case MVT::v16i8: Opc = ARM::VUZPq8; break; 2978 case MVT::v8i16: Opc = ARM::VUZPq16; break; 2979 case MVT::v4f32: 2980 case MVT::v4i32: Opc = ARM::VUZPq32; break; 2981 } 2982 SDValue Pred = getAL(CurDAG, dl); 2983 SDValue PredReg = CurDAG->getRegister(0, MVT::i32); 2984 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg }; 2985 return CurDAG->getMachineNode(Opc, dl, VT, VT, Ops); 2986 } 2987 case ARMISD::VTRN: { 2988 unsigned Opc = 0; 2989 EVT VT = N->getValueType(0); 2990 switch (VT.getSimpleVT().SimpleTy) { 2991 default: return nullptr; 2992 case MVT::v8i8: Opc = ARM::VTRNd8; break; 2993 case MVT::v4i16: Opc = ARM::VTRNd16; break; 2994 case MVT::v2f32: 2995 case MVT::v2i32: Opc = ARM::VTRNd32; break; 2996 case MVT::v16i8: Opc = ARM::VTRNq8; break; 2997 case MVT::v8i16: Opc = ARM::VTRNq16; break; 2998 case MVT::v4f32: 2999 case MVT::v4i32: Opc = ARM::VTRNq32; break; 3000 } 3001 SDValue Pred = getAL(CurDAG, dl); 3002 SDValue PredReg = CurDAG->getRegister(0, MVT::i32); 3003 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg }; 3004 return CurDAG->getMachineNode(Opc, dl, VT, VT, Ops); 3005 } 3006 case ARMISD::BUILD_VECTOR: { 3007 EVT VecVT = N->getValueType(0); 3008 EVT EltVT = VecVT.getVectorElementType(); 3009 unsigned NumElts = VecVT.getVectorNumElements(); 3010 if (EltVT == MVT::f64) { 3011 assert(NumElts == 2 && "unexpected type for BUILD_VECTOR"); 3012 return createDRegPairNode(VecVT, N->getOperand(0), N->getOperand(1)); 3013 } 3014 assert(EltVT == MVT::f32 && "unexpected type for BUILD_VECTOR"); 3015 if (NumElts == 2) 3016 return createSRegPairNode(VecVT, N->getOperand(0), N->getOperand(1)); 3017 assert(NumElts == 4 && "unexpected type for BUILD_VECTOR"); 3018 return createQuadSRegsNode(VecVT, N->getOperand(0), N->getOperand(1), 3019 N->getOperand(2), N->getOperand(3)); 3020 } 3021 3022 case ARMISD::VLD2DUP: { 3023 static const uint16_t Opcodes[] = { ARM::VLD2DUPd8, ARM::VLD2DUPd16, 3024 ARM::VLD2DUPd32 }; 3025 return SelectVLDDup(N, false, 2, Opcodes); 3026 } 3027 3028 case ARMISD::VLD3DUP: { 3029 static const uint16_t Opcodes[] = { ARM::VLD3DUPd8Pseudo, 3030 ARM::VLD3DUPd16Pseudo, 3031 ARM::VLD3DUPd32Pseudo }; 3032 return SelectVLDDup(N, false, 3, Opcodes); 3033 } 3034 3035 case ARMISD::VLD4DUP: { 3036 static const uint16_t Opcodes[] = { ARM::VLD4DUPd8Pseudo, 3037 ARM::VLD4DUPd16Pseudo, 3038 ARM::VLD4DUPd32Pseudo }; 3039 return SelectVLDDup(N, false, 4, Opcodes); 3040 } 3041 3042 case ARMISD::VLD2DUP_UPD: { 3043 static const uint16_t Opcodes[] = { ARM::VLD2DUPd8wb_fixed, 3044 ARM::VLD2DUPd16wb_fixed, 3045 ARM::VLD2DUPd32wb_fixed }; 3046 return SelectVLDDup(N, true, 2, Opcodes); 3047 } 3048 3049 case ARMISD::VLD3DUP_UPD: { 3050 static const uint16_t Opcodes[] = { ARM::VLD3DUPd8Pseudo_UPD, 3051 ARM::VLD3DUPd16Pseudo_UPD, 3052 ARM::VLD3DUPd32Pseudo_UPD }; 3053 return SelectVLDDup(N, true, 3, Opcodes); 3054 } 3055 3056 case ARMISD::VLD4DUP_UPD: { 3057 static const uint16_t Opcodes[] = { ARM::VLD4DUPd8Pseudo_UPD, 3058 ARM::VLD4DUPd16Pseudo_UPD, 3059 ARM::VLD4DUPd32Pseudo_UPD }; 3060 return SelectVLDDup(N, true, 4, Opcodes); 3061 } 3062 3063 case ARMISD::VLD1_UPD: { 3064 static const uint16_t DOpcodes[] = { ARM::VLD1d8wb_fixed, 3065 ARM::VLD1d16wb_fixed, 3066 ARM::VLD1d32wb_fixed, 3067 ARM::VLD1d64wb_fixed }; 3068 static const uint16_t QOpcodes[] = { ARM::VLD1q8wb_fixed, 3069 ARM::VLD1q16wb_fixed, 3070 ARM::VLD1q32wb_fixed, 3071 ARM::VLD1q64wb_fixed }; 3072 return SelectVLD(N, true, 1, DOpcodes, QOpcodes, nullptr); 3073 } 3074 3075 case ARMISD::VLD2_UPD: { 3076 static const uint16_t DOpcodes[] = { ARM::VLD2d8wb_fixed, 3077 ARM::VLD2d16wb_fixed, 3078 ARM::VLD2d32wb_fixed, 3079 ARM::VLD1q64wb_fixed}; 3080 static const uint16_t QOpcodes[] = { ARM::VLD2q8PseudoWB_fixed, 3081 ARM::VLD2q16PseudoWB_fixed, 3082 ARM::VLD2q32PseudoWB_fixed }; 3083 return SelectVLD(N, true, 2, DOpcodes, QOpcodes, nullptr); 3084 } 3085 3086 case ARMISD::VLD3_UPD: { 3087 static const uint16_t DOpcodes[] = { ARM::VLD3d8Pseudo_UPD, 3088 ARM::VLD3d16Pseudo_UPD, 3089 ARM::VLD3d32Pseudo_UPD, 3090 ARM::VLD1d64TPseudoWB_fixed}; 3091 static const uint16_t QOpcodes0[] = { ARM::VLD3q8Pseudo_UPD, 3092 ARM::VLD3q16Pseudo_UPD, 3093 ARM::VLD3q32Pseudo_UPD }; 3094 static const uint16_t QOpcodes1[] = { ARM::VLD3q8oddPseudo_UPD, 3095 ARM::VLD3q16oddPseudo_UPD, 3096 ARM::VLD3q32oddPseudo_UPD }; 3097 return SelectVLD(N, true, 3, DOpcodes, QOpcodes0, QOpcodes1); 3098 } 3099 3100 case ARMISD::VLD4_UPD: { 3101 static const uint16_t DOpcodes[] = { ARM::VLD4d8Pseudo_UPD, 3102 ARM::VLD4d16Pseudo_UPD, 3103 ARM::VLD4d32Pseudo_UPD, 3104 ARM::VLD1d64QPseudoWB_fixed}; 3105 static const uint16_t QOpcodes0[] = { ARM::VLD4q8Pseudo_UPD, 3106 ARM::VLD4q16Pseudo_UPD, 3107 ARM::VLD4q32Pseudo_UPD }; 3108 static const uint16_t QOpcodes1[] = { ARM::VLD4q8oddPseudo_UPD, 3109 ARM::VLD4q16oddPseudo_UPD, 3110 ARM::VLD4q32oddPseudo_UPD }; 3111 return SelectVLD(N, true, 4, DOpcodes, QOpcodes0, QOpcodes1); 3112 } 3113 3114 case ARMISD::VLD2LN_UPD: { 3115 static const uint16_t DOpcodes[] = { ARM::VLD2LNd8Pseudo_UPD, 3116 ARM::VLD2LNd16Pseudo_UPD, 3117 ARM::VLD2LNd32Pseudo_UPD }; 3118 static const uint16_t QOpcodes[] = { ARM::VLD2LNq16Pseudo_UPD, 3119 ARM::VLD2LNq32Pseudo_UPD }; 3120 return SelectVLDSTLane(N, true, true, 2, DOpcodes, QOpcodes); 3121 } 3122 3123 case ARMISD::VLD3LN_UPD: { 3124 static const uint16_t DOpcodes[] = { ARM::VLD3LNd8Pseudo_UPD, 3125 ARM::VLD3LNd16Pseudo_UPD, 3126 ARM::VLD3LNd32Pseudo_UPD }; 3127 static const uint16_t QOpcodes[] = { ARM::VLD3LNq16Pseudo_UPD, 3128 ARM::VLD3LNq32Pseudo_UPD }; 3129 return SelectVLDSTLane(N, true, true, 3, DOpcodes, QOpcodes); 3130 } 3131 3132 case ARMISD::VLD4LN_UPD: { 3133 static const uint16_t DOpcodes[] = { ARM::VLD4LNd8Pseudo_UPD, 3134 ARM::VLD4LNd16Pseudo_UPD, 3135 ARM::VLD4LNd32Pseudo_UPD }; 3136 static const uint16_t QOpcodes[] = { ARM::VLD4LNq16Pseudo_UPD, 3137 ARM::VLD4LNq32Pseudo_UPD }; 3138 return SelectVLDSTLane(N, true, true, 4, DOpcodes, QOpcodes); 3139 } 3140 3141 case ARMISD::VST1_UPD: { 3142 static const uint16_t DOpcodes[] = { ARM::VST1d8wb_fixed, 3143 ARM::VST1d16wb_fixed, 3144 ARM::VST1d32wb_fixed, 3145 ARM::VST1d64wb_fixed }; 3146 static const uint16_t QOpcodes[] = { ARM::VST1q8wb_fixed, 3147 ARM::VST1q16wb_fixed, 3148 ARM::VST1q32wb_fixed, 3149 ARM::VST1q64wb_fixed }; 3150 return SelectVST(N, true, 1, DOpcodes, QOpcodes, nullptr); 3151 } 3152 3153 case ARMISD::VST2_UPD: { 3154 static const uint16_t DOpcodes[] = { ARM::VST2d8wb_fixed, 3155 ARM::VST2d16wb_fixed, 3156 ARM::VST2d32wb_fixed, 3157 ARM::VST1q64wb_fixed}; 3158 static const uint16_t QOpcodes[] = { ARM::VST2q8PseudoWB_fixed, 3159 ARM::VST2q16PseudoWB_fixed, 3160 ARM::VST2q32PseudoWB_fixed }; 3161 return SelectVST(N, true, 2, DOpcodes, QOpcodes, nullptr); 3162 } 3163 3164 case ARMISD::VST3_UPD: { 3165 static const uint16_t DOpcodes[] = { ARM::VST3d8Pseudo_UPD, 3166 ARM::VST3d16Pseudo_UPD, 3167 ARM::VST3d32Pseudo_UPD, 3168 ARM::VST1d64TPseudoWB_fixed}; 3169 static const uint16_t QOpcodes0[] = { ARM::VST3q8Pseudo_UPD, 3170 ARM::VST3q16Pseudo_UPD, 3171 ARM::VST3q32Pseudo_UPD }; 3172 static const uint16_t QOpcodes1[] = { ARM::VST3q8oddPseudo_UPD, 3173 ARM::VST3q16oddPseudo_UPD, 3174 ARM::VST3q32oddPseudo_UPD }; 3175 return SelectVST(N, true, 3, DOpcodes, QOpcodes0, QOpcodes1); 3176 } 3177 3178 case ARMISD::VST4_UPD: { 3179 static const uint16_t DOpcodes[] = { ARM::VST4d8Pseudo_UPD, 3180 ARM::VST4d16Pseudo_UPD, 3181 ARM::VST4d32Pseudo_UPD, 3182 ARM::VST1d64QPseudoWB_fixed}; 3183 static const uint16_t QOpcodes0[] = { ARM::VST4q8Pseudo_UPD, 3184 ARM::VST4q16Pseudo_UPD, 3185 ARM::VST4q32Pseudo_UPD }; 3186 static const uint16_t QOpcodes1[] = { ARM::VST4q8oddPseudo_UPD, 3187 ARM::VST4q16oddPseudo_UPD, 3188 ARM::VST4q32oddPseudo_UPD }; 3189 return SelectVST(N, true, 4, DOpcodes, QOpcodes0, QOpcodes1); 3190 } 3191 3192 case ARMISD::VST2LN_UPD: { 3193 static const uint16_t DOpcodes[] = { ARM::VST2LNd8Pseudo_UPD, 3194 ARM::VST2LNd16Pseudo_UPD, 3195 ARM::VST2LNd32Pseudo_UPD }; 3196 static const uint16_t QOpcodes[] = { ARM::VST2LNq16Pseudo_UPD, 3197 ARM::VST2LNq32Pseudo_UPD }; 3198 return SelectVLDSTLane(N, false, true, 2, DOpcodes, QOpcodes); 3199 } 3200 3201 case ARMISD::VST3LN_UPD: { 3202 static const uint16_t DOpcodes[] = { ARM::VST3LNd8Pseudo_UPD, 3203 ARM::VST3LNd16Pseudo_UPD, 3204 ARM::VST3LNd32Pseudo_UPD }; 3205 static const uint16_t QOpcodes[] = { ARM::VST3LNq16Pseudo_UPD, 3206 ARM::VST3LNq32Pseudo_UPD }; 3207 return SelectVLDSTLane(N, false, true, 3, DOpcodes, QOpcodes); 3208 } 3209 3210 case ARMISD::VST4LN_UPD: { 3211 static const uint16_t DOpcodes[] = { ARM::VST4LNd8Pseudo_UPD, 3212 ARM::VST4LNd16Pseudo_UPD, 3213 ARM::VST4LNd32Pseudo_UPD }; 3214 static const uint16_t QOpcodes[] = { ARM::VST4LNq16Pseudo_UPD, 3215 ARM::VST4LNq32Pseudo_UPD }; 3216 return SelectVLDSTLane(N, false, true, 4, DOpcodes, QOpcodes); 3217 } 3218 3219 case ISD::INTRINSIC_VOID: 3220 case ISD::INTRINSIC_W_CHAIN: { 3221 unsigned IntNo = cast<ConstantSDNode>(N->getOperand(1))->getZExtValue(); 3222 switch (IntNo) { 3223 default: 3224 break; 3225 3226 case Intrinsic::arm_ldaexd: 3227 case Intrinsic::arm_ldrexd: { 3228 SDLoc dl(N); 3229 SDValue Chain = N->getOperand(0); 3230 SDValue MemAddr = N->getOperand(2); 3231 bool isThumb = Subtarget->isThumb() && Subtarget->hasV8MBaselineOps(); 3232 3233 bool IsAcquire = IntNo == Intrinsic::arm_ldaexd; 3234 unsigned NewOpc = isThumb ? (IsAcquire ? ARM::t2LDAEXD : ARM::t2LDREXD) 3235 : (IsAcquire ? ARM::LDAEXD : ARM::LDREXD); 3236 3237 // arm_ldrexd returns a i64 value in {i32, i32} 3238 std::vector<EVT> ResTys; 3239 if (isThumb) { 3240 ResTys.push_back(MVT::i32); 3241 ResTys.push_back(MVT::i32); 3242 } else 3243 ResTys.push_back(MVT::Untyped); 3244 ResTys.push_back(MVT::Other); 3245 3246 // Place arguments in the right order. 3247 SmallVector<SDValue, 7> Ops; 3248 Ops.push_back(MemAddr); 3249 Ops.push_back(getAL(CurDAG, dl)); 3250 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 3251 Ops.push_back(Chain); 3252 SDNode *Ld = CurDAG->getMachineNode(NewOpc, dl, ResTys, Ops); 3253 // Transfer memoperands. 3254 MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1); 3255 MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 3256 cast<MachineSDNode>(Ld)->setMemRefs(MemOp, MemOp + 1); 3257 3258 // Remap uses. 3259 SDValue OutChain = isThumb ? SDValue(Ld, 2) : SDValue(Ld, 1); 3260 if (!SDValue(N, 0).use_empty()) { 3261 SDValue Result; 3262 if (isThumb) 3263 Result = SDValue(Ld, 0); 3264 else { 3265 SDValue SubRegIdx = 3266 CurDAG->getTargetConstant(ARM::gsub_0, dl, MVT::i32); 3267 SDNode *ResNode = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, 3268 dl, MVT::i32, SDValue(Ld, 0), SubRegIdx); 3269 Result = SDValue(ResNode,0); 3270 } 3271 ReplaceUses(SDValue(N, 0), Result); 3272 } 3273 if (!SDValue(N, 1).use_empty()) { 3274 SDValue Result; 3275 if (isThumb) 3276 Result = SDValue(Ld, 1); 3277 else { 3278 SDValue SubRegIdx = 3279 CurDAG->getTargetConstant(ARM::gsub_1, dl, MVT::i32); 3280 SDNode *ResNode = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, 3281 dl, MVT::i32, SDValue(Ld, 0), SubRegIdx); 3282 Result = SDValue(ResNode,0); 3283 } 3284 ReplaceUses(SDValue(N, 1), Result); 3285 } 3286 ReplaceUses(SDValue(N, 2), OutChain); 3287 return nullptr; 3288 } 3289 case Intrinsic::arm_stlexd: 3290 case Intrinsic::arm_strexd: { 3291 SDLoc dl(N); 3292 SDValue Chain = N->getOperand(0); 3293 SDValue Val0 = N->getOperand(2); 3294 SDValue Val1 = N->getOperand(3); 3295 SDValue MemAddr = N->getOperand(4); 3296 3297 // Store exclusive double return a i32 value which is the return status 3298 // of the issued store. 3299 const EVT ResTys[] = {MVT::i32, MVT::Other}; 3300 3301 bool isThumb = Subtarget->isThumb() && Subtarget->hasThumb2(); 3302 // Place arguments in the right order. 3303 SmallVector<SDValue, 7> Ops; 3304 if (isThumb) { 3305 Ops.push_back(Val0); 3306 Ops.push_back(Val1); 3307 } else 3308 // arm_strexd uses GPRPair. 3309 Ops.push_back(SDValue(createGPRPairNode(MVT::Untyped, Val0, Val1), 0)); 3310 Ops.push_back(MemAddr); 3311 Ops.push_back(getAL(CurDAG, dl)); 3312 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 3313 Ops.push_back(Chain); 3314 3315 bool IsRelease = IntNo == Intrinsic::arm_stlexd; 3316 unsigned NewOpc = isThumb ? (IsRelease ? ARM::t2STLEXD : ARM::t2STREXD) 3317 : (IsRelease ? ARM::STLEXD : ARM::STREXD); 3318 3319 SDNode *St = CurDAG->getMachineNode(NewOpc, dl, ResTys, Ops); 3320 // Transfer memoperands. 3321 MachineSDNode::mmo_iterator MemOp = MF->allocateMemRefsArray(1); 3322 MemOp[0] = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 3323 cast<MachineSDNode>(St)->setMemRefs(MemOp, MemOp + 1); 3324 3325 return St; 3326 } 3327 3328 case Intrinsic::arm_neon_vld1: { 3329 static const uint16_t DOpcodes[] = { ARM::VLD1d8, ARM::VLD1d16, 3330 ARM::VLD1d32, ARM::VLD1d64 }; 3331 static const uint16_t QOpcodes[] = { ARM::VLD1q8, ARM::VLD1q16, 3332 ARM::VLD1q32, ARM::VLD1q64}; 3333 return SelectVLD(N, false, 1, DOpcodes, QOpcodes, nullptr); 3334 } 3335 3336 case Intrinsic::arm_neon_vld2: { 3337 static const uint16_t DOpcodes[] = { ARM::VLD2d8, ARM::VLD2d16, 3338 ARM::VLD2d32, ARM::VLD1q64 }; 3339 static const uint16_t QOpcodes[] = { ARM::VLD2q8Pseudo, ARM::VLD2q16Pseudo, 3340 ARM::VLD2q32Pseudo }; 3341 return SelectVLD(N, false, 2, DOpcodes, QOpcodes, nullptr); 3342 } 3343 3344 case Intrinsic::arm_neon_vld3: { 3345 static const uint16_t DOpcodes[] = { ARM::VLD3d8Pseudo, 3346 ARM::VLD3d16Pseudo, 3347 ARM::VLD3d32Pseudo, 3348 ARM::VLD1d64TPseudo }; 3349 static const uint16_t QOpcodes0[] = { ARM::VLD3q8Pseudo_UPD, 3350 ARM::VLD3q16Pseudo_UPD, 3351 ARM::VLD3q32Pseudo_UPD }; 3352 static const uint16_t QOpcodes1[] = { ARM::VLD3q8oddPseudo, 3353 ARM::VLD3q16oddPseudo, 3354 ARM::VLD3q32oddPseudo }; 3355 return SelectVLD(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1); 3356 } 3357 3358 case Intrinsic::arm_neon_vld4: { 3359 static const uint16_t DOpcodes[] = { ARM::VLD4d8Pseudo, 3360 ARM::VLD4d16Pseudo, 3361 ARM::VLD4d32Pseudo, 3362 ARM::VLD1d64QPseudo }; 3363 static const uint16_t QOpcodes0[] = { ARM::VLD4q8Pseudo_UPD, 3364 ARM::VLD4q16Pseudo_UPD, 3365 ARM::VLD4q32Pseudo_UPD }; 3366 static const uint16_t QOpcodes1[] = { ARM::VLD4q8oddPseudo, 3367 ARM::VLD4q16oddPseudo, 3368 ARM::VLD4q32oddPseudo }; 3369 return SelectVLD(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1); 3370 } 3371 3372 case Intrinsic::arm_neon_vld2lane: { 3373 static const uint16_t DOpcodes[] = { ARM::VLD2LNd8Pseudo, 3374 ARM::VLD2LNd16Pseudo, 3375 ARM::VLD2LNd32Pseudo }; 3376 static const uint16_t QOpcodes[] = { ARM::VLD2LNq16Pseudo, 3377 ARM::VLD2LNq32Pseudo }; 3378 return SelectVLDSTLane(N, true, false, 2, DOpcodes, QOpcodes); 3379 } 3380 3381 case Intrinsic::arm_neon_vld3lane: { 3382 static const uint16_t DOpcodes[] = { ARM::VLD3LNd8Pseudo, 3383 ARM::VLD3LNd16Pseudo, 3384 ARM::VLD3LNd32Pseudo }; 3385 static const uint16_t QOpcodes[] = { ARM::VLD3LNq16Pseudo, 3386 ARM::VLD3LNq32Pseudo }; 3387 return SelectVLDSTLane(N, true, false, 3, DOpcodes, QOpcodes); 3388 } 3389 3390 case Intrinsic::arm_neon_vld4lane: { 3391 static const uint16_t DOpcodes[] = { ARM::VLD4LNd8Pseudo, 3392 ARM::VLD4LNd16Pseudo, 3393 ARM::VLD4LNd32Pseudo }; 3394 static const uint16_t QOpcodes[] = { ARM::VLD4LNq16Pseudo, 3395 ARM::VLD4LNq32Pseudo }; 3396 return SelectVLDSTLane(N, true, false, 4, DOpcodes, QOpcodes); 3397 } 3398 3399 case Intrinsic::arm_neon_vst1: { 3400 static const uint16_t DOpcodes[] = { ARM::VST1d8, ARM::VST1d16, 3401 ARM::VST1d32, ARM::VST1d64 }; 3402 static const uint16_t QOpcodes[] = { ARM::VST1q8, ARM::VST1q16, 3403 ARM::VST1q32, ARM::VST1q64 }; 3404 return SelectVST(N, false, 1, DOpcodes, QOpcodes, nullptr); 3405 } 3406 3407 case Intrinsic::arm_neon_vst2: { 3408 static const uint16_t DOpcodes[] = { ARM::VST2d8, ARM::VST2d16, 3409 ARM::VST2d32, ARM::VST1q64 }; 3410 static uint16_t QOpcodes[] = { ARM::VST2q8Pseudo, ARM::VST2q16Pseudo, 3411 ARM::VST2q32Pseudo }; 3412 return SelectVST(N, false, 2, DOpcodes, QOpcodes, nullptr); 3413 } 3414 3415 case Intrinsic::arm_neon_vst3: { 3416 static const uint16_t DOpcodes[] = { ARM::VST3d8Pseudo, 3417 ARM::VST3d16Pseudo, 3418 ARM::VST3d32Pseudo, 3419 ARM::VST1d64TPseudo }; 3420 static const uint16_t QOpcodes0[] = { ARM::VST3q8Pseudo_UPD, 3421 ARM::VST3q16Pseudo_UPD, 3422 ARM::VST3q32Pseudo_UPD }; 3423 static const uint16_t QOpcodes1[] = { ARM::VST3q8oddPseudo, 3424 ARM::VST3q16oddPseudo, 3425 ARM::VST3q32oddPseudo }; 3426 return SelectVST(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1); 3427 } 3428 3429 case Intrinsic::arm_neon_vst4: { 3430 static const uint16_t DOpcodes[] = { ARM::VST4d8Pseudo, 3431 ARM::VST4d16Pseudo, 3432 ARM::VST4d32Pseudo, 3433 ARM::VST1d64QPseudo }; 3434 static const uint16_t QOpcodes0[] = { ARM::VST4q8Pseudo_UPD, 3435 ARM::VST4q16Pseudo_UPD, 3436 ARM::VST4q32Pseudo_UPD }; 3437 static const uint16_t QOpcodes1[] = { ARM::VST4q8oddPseudo, 3438 ARM::VST4q16oddPseudo, 3439 ARM::VST4q32oddPseudo }; 3440 return SelectVST(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1); 3441 } 3442 3443 case Intrinsic::arm_neon_vst2lane: { 3444 static const uint16_t DOpcodes[] = { ARM::VST2LNd8Pseudo, 3445 ARM::VST2LNd16Pseudo, 3446 ARM::VST2LNd32Pseudo }; 3447 static const uint16_t QOpcodes[] = { ARM::VST2LNq16Pseudo, 3448 ARM::VST2LNq32Pseudo }; 3449 return SelectVLDSTLane(N, false, false, 2, DOpcodes, QOpcodes); 3450 } 3451 3452 case Intrinsic::arm_neon_vst3lane: { 3453 static const uint16_t DOpcodes[] = { ARM::VST3LNd8Pseudo, 3454 ARM::VST3LNd16Pseudo, 3455 ARM::VST3LNd32Pseudo }; 3456 static const uint16_t QOpcodes[] = { ARM::VST3LNq16Pseudo, 3457 ARM::VST3LNq32Pseudo }; 3458 return SelectVLDSTLane(N, false, false, 3, DOpcodes, QOpcodes); 3459 } 3460 3461 case Intrinsic::arm_neon_vst4lane: { 3462 static const uint16_t DOpcodes[] = { ARM::VST4LNd8Pseudo, 3463 ARM::VST4LNd16Pseudo, 3464 ARM::VST4LNd32Pseudo }; 3465 static const uint16_t QOpcodes[] = { ARM::VST4LNq16Pseudo, 3466 ARM::VST4LNq32Pseudo }; 3467 return SelectVLDSTLane(N, false, false, 4, DOpcodes, QOpcodes); 3468 } 3469 } 3470 break; 3471 } 3472 3473 case ISD::INTRINSIC_WO_CHAIN: { 3474 unsigned IntNo = cast<ConstantSDNode>(N->getOperand(0))->getZExtValue(); 3475 switch (IntNo) { 3476 default: 3477 break; 3478 3479 case Intrinsic::arm_neon_vtbl2: 3480 return SelectVTBL(N, false, 2, ARM::VTBL2); 3481 case Intrinsic::arm_neon_vtbl3: 3482 return SelectVTBL(N, false, 3, ARM::VTBL3Pseudo); 3483 case Intrinsic::arm_neon_vtbl4: 3484 return SelectVTBL(N, false, 4, ARM::VTBL4Pseudo); 3485 3486 case Intrinsic::arm_neon_vtbx2: 3487 return SelectVTBL(N, true, 2, ARM::VTBX2); 3488 case Intrinsic::arm_neon_vtbx3: 3489 return SelectVTBL(N, true, 3, ARM::VTBX3Pseudo); 3490 case Intrinsic::arm_neon_vtbx4: 3491 return SelectVTBL(N, true, 4, ARM::VTBX4Pseudo); 3492 } 3493 break; 3494 } 3495 3496 case ARMISD::VTBL1: { 3497 SDLoc dl(N); 3498 EVT VT = N->getValueType(0); 3499 SmallVector<SDValue, 6> Ops; 3500 3501 Ops.push_back(N->getOperand(0)); 3502 Ops.push_back(N->getOperand(1)); 3503 Ops.push_back(getAL(CurDAG, dl)); // Predicate 3504 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); // Predicate Register 3505 return CurDAG->getMachineNode(ARM::VTBL1, dl, VT, Ops); 3506 } 3507 case ARMISD::VTBL2: { 3508 SDLoc dl(N); 3509 EVT VT = N->getValueType(0); 3510 3511 // Form a REG_SEQUENCE to force register allocation. 3512 SDValue V0 = N->getOperand(0); 3513 SDValue V1 = N->getOperand(1); 3514 SDValue RegSeq = SDValue(createDRegPairNode(MVT::v16i8, V0, V1), 0); 3515 3516 SmallVector<SDValue, 6> Ops; 3517 Ops.push_back(RegSeq); 3518 Ops.push_back(N->getOperand(2)); 3519 Ops.push_back(getAL(CurDAG, dl)); // Predicate 3520 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); // Predicate Register 3521 return CurDAG->getMachineNode(ARM::VTBL2, dl, VT, Ops); 3522 } 3523 3524 case ISD::CONCAT_VECTORS: 3525 return SelectConcatVector(N); 3526 3527 case ISD::ATOMIC_CMP_SWAP: 3528 return SelectCMP_SWAP(N); 3529 } 3530 3531 return SelectCode(N); 3532 } 3533 3534 // Inspect a register string of the form 3535 // cp<coprocessor>:<opc1>:c<CRn>:c<CRm>:<opc2> (32bit) or 3536 // cp<coprocessor>:<opc1>:c<CRm> (64bit) inspect the fields of the string 3537 // and obtain the integer operands from them, adding these operands to the 3538 // provided vector. 3539 static void getIntOperandsFromRegisterString(StringRef RegString, 3540 SelectionDAG *CurDAG, SDLoc DL, 3541 std::vector<SDValue>& Ops) { 3542 SmallVector<StringRef, 5> Fields; 3543 RegString.split(Fields, ':'); 3544 3545 if (Fields.size() > 1) { 3546 bool AllIntFields = true; 3547 3548 for (StringRef Field : Fields) { 3549 // Need to trim out leading 'cp' characters and get the integer field. 3550 unsigned IntField; 3551 AllIntFields &= !Field.trim("CPcp").getAsInteger(10, IntField); 3552 Ops.push_back(CurDAG->getTargetConstant(IntField, DL, MVT::i32)); 3553 } 3554 3555 assert(AllIntFields && 3556 "Unexpected non-integer value in special register string."); 3557 } 3558 } 3559 3560 // Maps a Banked Register string to its mask value. The mask value returned is 3561 // for use in the MRSbanked / MSRbanked instruction nodes as the Banked Register 3562 // mask operand, which expresses which register is to be used, e.g. r8, and in 3563 // which mode it is to be used, e.g. usr. Returns -1 to signify that the string 3564 // was invalid. 3565 static inline int getBankedRegisterMask(StringRef RegString) { 3566 return StringSwitch<int>(RegString.lower()) 3567 .Case("r8_usr", 0x00) 3568 .Case("r9_usr", 0x01) 3569 .Case("r10_usr", 0x02) 3570 .Case("r11_usr", 0x03) 3571 .Case("r12_usr", 0x04) 3572 .Case("sp_usr", 0x05) 3573 .Case("lr_usr", 0x06) 3574 .Case("r8_fiq", 0x08) 3575 .Case("r9_fiq", 0x09) 3576 .Case("r10_fiq", 0x0a) 3577 .Case("r11_fiq", 0x0b) 3578 .Case("r12_fiq", 0x0c) 3579 .Case("sp_fiq", 0x0d) 3580 .Case("lr_fiq", 0x0e) 3581 .Case("lr_irq", 0x10) 3582 .Case("sp_irq", 0x11) 3583 .Case("lr_svc", 0x12) 3584 .Case("sp_svc", 0x13) 3585 .Case("lr_abt", 0x14) 3586 .Case("sp_abt", 0x15) 3587 .Case("lr_und", 0x16) 3588 .Case("sp_und", 0x17) 3589 .Case("lr_mon", 0x1c) 3590 .Case("sp_mon", 0x1d) 3591 .Case("elr_hyp", 0x1e) 3592 .Case("sp_hyp", 0x1f) 3593 .Case("spsr_fiq", 0x2e) 3594 .Case("spsr_irq", 0x30) 3595 .Case("spsr_svc", 0x32) 3596 .Case("spsr_abt", 0x34) 3597 .Case("spsr_und", 0x36) 3598 .Case("spsr_mon", 0x3c) 3599 .Case("spsr_hyp", 0x3e) 3600 .Default(-1); 3601 } 3602 3603 // Maps a MClass special register string to its value for use in the 3604 // t2MRS_M / t2MSR_M instruction nodes as the SYSm value operand. 3605 // Returns -1 to signify that the string was invalid. 3606 static inline int getMClassRegisterSYSmValueMask(StringRef RegString) { 3607 return StringSwitch<int>(RegString.lower()) 3608 .Case("apsr", 0x0) 3609 .Case("iapsr", 0x1) 3610 .Case("eapsr", 0x2) 3611 .Case("xpsr", 0x3) 3612 .Case("ipsr", 0x5) 3613 .Case("epsr", 0x6) 3614 .Case("iepsr", 0x7) 3615 .Case("msp", 0x8) 3616 .Case("psp", 0x9) 3617 .Case("primask", 0x10) 3618 .Case("basepri", 0x11) 3619 .Case("basepri_max", 0x12) 3620 .Case("faultmask", 0x13) 3621 .Case("control", 0x14) 3622 .Case("msplim", 0x0a) 3623 .Case("psplim", 0x0b) 3624 .Case("sp", 0x18) 3625 .Default(-1); 3626 } 3627 3628 // The flags here are common to those allowed for apsr in the A class cores and 3629 // those allowed for the special registers in the M class cores. Returns a 3630 // value representing which flags were present, -1 if invalid. 3631 static inline int getMClassFlagsMask(StringRef Flags, bool hasDSP) { 3632 if (Flags.empty()) 3633 return 0x2 | (int)hasDSP; 3634 3635 return StringSwitch<int>(Flags) 3636 .Case("g", 0x1) 3637 .Case("nzcvq", 0x2) 3638 .Case("nzcvqg", 0x3) 3639 .Default(-1); 3640 } 3641 3642 static int getMClassRegisterMask(StringRef Reg, StringRef Flags, bool IsRead, 3643 const ARMSubtarget *Subtarget) { 3644 // Ensure that the register (without flags) was a valid M Class special 3645 // register. 3646 int SYSmvalue = getMClassRegisterSYSmValueMask(Reg); 3647 if (SYSmvalue == -1) 3648 return -1; 3649 3650 // basepri, basepri_max and faultmask are only valid for V7m. 3651 if (!Subtarget->hasV7Ops() && SYSmvalue >= 0x11 && SYSmvalue <= 0x13) 3652 return -1; 3653 3654 if (Subtarget->has8MSecExt() && Flags.lower() == "ns") { 3655 Flags = ""; 3656 SYSmvalue |= 0x80; 3657 } 3658 3659 if (!Subtarget->has8MSecExt() && 3660 (SYSmvalue == 0xa || SYSmvalue == 0xb || SYSmvalue > 0x14)) 3661 return -1; 3662 3663 if (!Subtarget->hasV8MMainlineOps() && 3664 (SYSmvalue == 0x8a || SYSmvalue == 0x8b || SYSmvalue == 0x91 || 3665 SYSmvalue == 0x93)) 3666 return -1; 3667 3668 // If it was a read then we won't be expecting flags and so at this point 3669 // we can return the mask. 3670 if (IsRead) { 3671 if (Flags.empty()) 3672 return SYSmvalue; 3673 else 3674 return -1; 3675 } 3676 3677 // We know we are now handling a write so need to get the mask for the flags. 3678 int Mask = getMClassFlagsMask(Flags, Subtarget->hasDSP()); 3679 3680 // Only apsr, iapsr, eapsr, xpsr can have flags. The other register values 3681 // shouldn't have flags present. 3682 if ((SYSmvalue < 0x4 && Mask == -1) || (SYSmvalue > 0x4 && !Flags.empty())) 3683 return -1; 3684 3685 // The _g and _nzcvqg versions are only valid if the DSP extension is 3686 // available. 3687 if (!Subtarget->hasDSP() && (Mask & 0x1)) 3688 return -1; 3689 3690 // The register was valid so need to put the mask in the correct place 3691 // (the flags need to be in bits 11-10) and combine with the SYSmvalue to 3692 // construct the operand for the instruction node. 3693 if (SYSmvalue < 0x4) 3694 return SYSmvalue | Mask << 10; 3695 3696 return SYSmvalue; 3697 } 3698 3699 static int getARClassRegisterMask(StringRef Reg, StringRef Flags) { 3700 // The mask operand contains the special register (R Bit) in bit 4, whether 3701 // the register is spsr (R bit is 1) or one of cpsr/apsr (R bit is 0), and 3702 // bits 3-0 contains the fields to be accessed in the special register, set by 3703 // the flags provided with the register. 3704 int Mask = 0; 3705 if (Reg == "apsr") { 3706 // The flags permitted for apsr are the same flags that are allowed in 3707 // M class registers. We get the flag value and then shift the flags into 3708 // the correct place to combine with the mask. 3709 Mask = getMClassFlagsMask(Flags, true); 3710 if (Mask == -1) 3711 return -1; 3712 return Mask << 2; 3713 } 3714 3715 if (Reg != "cpsr" && Reg != "spsr") { 3716 return -1; 3717 } 3718 3719 // This is the same as if the flags were "fc" 3720 if (Flags.empty() || Flags == "all") 3721 return Mask | 0x9; 3722 3723 // Inspect the supplied flags string and set the bits in the mask for 3724 // the relevant and valid flags allowed for cpsr and spsr. 3725 for (char Flag : Flags) { 3726 int FlagVal; 3727 switch (Flag) { 3728 case 'c': 3729 FlagVal = 0x1; 3730 break; 3731 case 'x': 3732 FlagVal = 0x2; 3733 break; 3734 case 's': 3735 FlagVal = 0x4; 3736 break; 3737 case 'f': 3738 FlagVal = 0x8; 3739 break; 3740 default: 3741 FlagVal = 0; 3742 } 3743 3744 // This avoids allowing strings where the same flag bit appears twice. 3745 if (!FlagVal || (Mask & FlagVal)) 3746 return -1; 3747 Mask |= FlagVal; 3748 } 3749 3750 // If the register is spsr then we need to set the R bit. 3751 if (Reg == "spsr") 3752 Mask |= 0x10; 3753 3754 return Mask; 3755 } 3756 3757 // Lower the read_register intrinsic to ARM specific DAG nodes 3758 // using the supplied metadata string to select the instruction node to use 3759 // and the registers/masks to construct as operands for the node. 3760 SDNode *ARMDAGToDAGISel::SelectReadRegister(SDNode *N){ 3761 const MDNodeSDNode *MD = dyn_cast<MDNodeSDNode>(N->getOperand(1)); 3762 const MDString *RegString = dyn_cast<MDString>(MD->getMD()->getOperand(0)); 3763 bool IsThumb2 = Subtarget->isThumb2(); 3764 SDLoc DL(N); 3765 3766 std::vector<SDValue> Ops; 3767 getIntOperandsFromRegisterString(RegString->getString(), CurDAG, DL, Ops); 3768 3769 if (!Ops.empty()) { 3770 // If the special register string was constructed of fields (as defined 3771 // in the ACLE) then need to lower to MRC node (32 bit) or 3772 // MRRC node(64 bit), we can make the distinction based on the number of 3773 // operands we have. 3774 unsigned Opcode; 3775 SmallVector<EVT, 3> ResTypes; 3776 if (Ops.size() == 5){ 3777 Opcode = IsThumb2 ? ARM::t2MRC : ARM::MRC; 3778 ResTypes.append({ MVT::i32, MVT::Other }); 3779 } else { 3780 assert(Ops.size() == 3 && 3781 "Invalid number of fields in special register string."); 3782 Opcode = IsThumb2 ? ARM::t2MRRC : ARM::MRRC; 3783 ResTypes.append({ MVT::i32, MVT::i32, MVT::Other }); 3784 } 3785 3786 Ops.push_back(getAL(CurDAG, DL)); 3787 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 3788 Ops.push_back(N->getOperand(0)); 3789 return CurDAG->getMachineNode(Opcode, DL, ResTypes, Ops); 3790 } 3791 3792 std::string SpecialReg = RegString->getString().lower(); 3793 3794 int BankedReg = getBankedRegisterMask(SpecialReg); 3795 if (BankedReg != -1) { 3796 Ops = { CurDAG->getTargetConstant(BankedReg, DL, MVT::i32), 3797 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 3798 N->getOperand(0) }; 3799 return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRSbanked : ARM::MRSbanked, 3800 DL, MVT::i32, MVT::Other, Ops); 3801 } 3802 3803 // The VFP registers are read by creating SelectionDAG nodes with opcodes 3804 // corresponding to the register that is being read from. So we switch on the 3805 // string to find which opcode we need to use. 3806 unsigned Opcode = StringSwitch<unsigned>(SpecialReg) 3807 .Case("fpscr", ARM::VMRS) 3808 .Case("fpexc", ARM::VMRS_FPEXC) 3809 .Case("fpsid", ARM::VMRS_FPSID) 3810 .Case("mvfr0", ARM::VMRS_MVFR0) 3811 .Case("mvfr1", ARM::VMRS_MVFR1) 3812 .Case("mvfr2", ARM::VMRS_MVFR2) 3813 .Case("fpinst", ARM::VMRS_FPINST) 3814 .Case("fpinst2", ARM::VMRS_FPINST2) 3815 .Default(0); 3816 3817 // If an opcode was found then we can lower the read to a VFP instruction. 3818 if (Opcode) { 3819 if (!Subtarget->hasVFP2()) 3820 return nullptr; 3821 if (Opcode == ARM::VMRS_MVFR2 && !Subtarget->hasFPARMv8()) 3822 return nullptr; 3823 3824 Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 3825 N->getOperand(0) }; 3826 return CurDAG->getMachineNode(Opcode, DL, MVT::i32, MVT::Other, Ops); 3827 } 3828 3829 // If the target is M Class then need to validate that the register string 3830 // is an acceptable value, so check that a mask can be constructed from the 3831 // string. 3832 if (Subtarget->isMClass()) { 3833 StringRef Flags = "", Reg = SpecialReg; 3834 if (Reg.endswith("_ns")) { 3835 Flags = "ns"; 3836 Reg = Reg.drop_back(3); 3837 } 3838 3839 int SYSmValue = getMClassRegisterMask(Reg, Flags, true, Subtarget); 3840 if (SYSmValue == -1) 3841 return nullptr; 3842 3843 SDValue Ops[] = { CurDAG->getTargetConstant(SYSmValue, DL, MVT::i32), 3844 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 3845 N->getOperand(0) }; 3846 return CurDAG->getMachineNode(ARM::t2MRS_M, DL, MVT::i32, MVT::Other, Ops); 3847 } 3848 3849 // Here we know the target is not M Class so we need to check if it is one 3850 // of the remaining possible values which are apsr, cpsr or spsr. 3851 if (SpecialReg == "apsr" || SpecialReg == "cpsr") { 3852 Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 3853 N->getOperand(0) }; 3854 return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRS_AR : ARM::MRS, DL, 3855 MVT::i32, MVT::Other, Ops); 3856 } 3857 3858 if (SpecialReg == "spsr") { 3859 Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 3860 N->getOperand(0) }; 3861 return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRSsys_AR : ARM::MRSsys, 3862 DL, MVT::i32, MVT::Other, Ops); 3863 } 3864 3865 return nullptr; 3866 } 3867 3868 // Lower the write_register intrinsic to ARM specific DAG nodes 3869 // using the supplied metadata string to select the instruction node to use 3870 // and the registers/masks to use in the nodes 3871 SDNode *ARMDAGToDAGISel::SelectWriteRegister(SDNode *N){ 3872 const MDNodeSDNode *MD = dyn_cast<MDNodeSDNode>(N->getOperand(1)); 3873 const MDString *RegString = dyn_cast<MDString>(MD->getMD()->getOperand(0)); 3874 bool IsThumb2 = Subtarget->isThumb2(); 3875 SDLoc DL(N); 3876 3877 std::vector<SDValue> Ops; 3878 getIntOperandsFromRegisterString(RegString->getString(), CurDAG, DL, Ops); 3879 3880 if (!Ops.empty()) { 3881 // If the special register string was constructed of fields (as defined 3882 // in the ACLE) then need to lower to MCR node (32 bit) or 3883 // MCRR node(64 bit), we can make the distinction based on the number of 3884 // operands we have. 3885 unsigned Opcode; 3886 if (Ops.size() == 5) { 3887 Opcode = IsThumb2 ? ARM::t2MCR : ARM::MCR; 3888 Ops.insert(Ops.begin()+2, N->getOperand(2)); 3889 } else { 3890 assert(Ops.size() == 3 && 3891 "Invalid number of fields in special register string."); 3892 Opcode = IsThumb2 ? ARM::t2MCRR : ARM::MCRR; 3893 SDValue WriteValue[] = { N->getOperand(2), N->getOperand(3) }; 3894 Ops.insert(Ops.begin()+2, WriteValue, WriteValue+2); 3895 } 3896 3897 Ops.push_back(getAL(CurDAG, DL)); 3898 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 3899 Ops.push_back(N->getOperand(0)); 3900 3901 return CurDAG->getMachineNode(Opcode, DL, MVT::Other, Ops); 3902 } 3903 3904 std::string SpecialReg = RegString->getString().lower(); 3905 int BankedReg = getBankedRegisterMask(SpecialReg); 3906 if (BankedReg != -1) { 3907 Ops = { CurDAG->getTargetConstant(BankedReg, DL, MVT::i32), N->getOperand(2), 3908 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 3909 N->getOperand(0) }; 3910 return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MSRbanked : ARM::MSRbanked, 3911 DL, MVT::Other, Ops); 3912 } 3913 3914 // The VFP registers are written to by creating SelectionDAG nodes with 3915 // opcodes corresponding to the register that is being written. So we switch 3916 // on the string to find which opcode we need to use. 3917 unsigned Opcode = StringSwitch<unsigned>(SpecialReg) 3918 .Case("fpscr", ARM::VMSR) 3919 .Case("fpexc", ARM::VMSR_FPEXC) 3920 .Case("fpsid", ARM::VMSR_FPSID) 3921 .Case("fpinst", ARM::VMSR_FPINST) 3922 .Case("fpinst2", ARM::VMSR_FPINST2) 3923 .Default(0); 3924 3925 if (Opcode) { 3926 if (!Subtarget->hasVFP2()) 3927 return nullptr; 3928 Ops = { N->getOperand(2), getAL(CurDAG, DL), 3929 CurDAG->getRegister(0, MVT::i32), N->getOperand(0) }; 3930 return CurDAG->getMachineNode(Opcode, DL, MVT::Other, Ops); 3931 } 3932 3933 std::pair<StringRef, StringRef> Fields; 3934 Fields = StringRef(SpecialReg).rsplit('_'); 3935 std::string Reg = Fields.first.str(); 3936 StringRef Flags = Fields.second; 3937 3938 // If the target was M Class then need to validate the special register value 3939 // and retrieve the mask for use in the instruction node. 3940 if (Subtarget->isMClass()) { 3941 // basepri_max gets split so need to correct Reg and Flags. 3942 if (SpecialReg == "basepri_max") { 3943 Reg = SpecialReg; 3944 Flags = ""; 3945 } 3946 int SYSmValue = getMClassRegisterMask(Reg, Flags, false, Subtarget); 3947 if (SYSmValue == -1) 3948 return nullptr; 3949 3950 SDValue Ops[] = { CurDAG->getTargetConstant(SYSmValue, DL, MVT::i32), 3951 N->getOperand(2), getAL(CurDAG, DL), 3952 CurDAG->getRegister(0, MVT::i32), N->getOperand(0) }; 3953 return CurDAG->getMachineNode(ARM::t2MSR_M, DL, MVT::Other, Ops); 3954 } 3955 3956 // We then check to see if a valid mask can be constructed for one of the 3957 // register string values permitted for the A and R class cores. These values 3958 // are apsr, spsr and cpsr; these are also valid on older cores. 3959 int Mask = getARClassRegisterMask(Reg, Flags); 3960 if (Mask != -1) { 3961 Ops = { CurDAG->getTargetConstant(Mask, DL, MVT::i32), N->getOperand(2), 3962 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 3963 N->getOperand(0) }; 3964 return CurDAG->getMachineNode(IsThumb2 ? ARM::t2MSR_AR : ARM::MSR, 3965 DL, MVT::Other, Ops); 3966 } 3967 3968 return nullptr; 3969 } 3970 3971 SDNode *ARMDAGToDAGISel::SelectInlineAsm(SDNode *N){ 3972 std::vector<SDValue> AsmNodeOperands; 3973 unsigned Flag, Kind; 3974 bool Changed = false; 3975 unsigned NumOps = N->getNumOperands(); 3976 3977 // Normally, i64 data is bounded to two arbitrary GRPs for "%r" constraint. 3978 // However, some instrstions (e.g. ldrexd/strexd in ARM mode) require 3979 // (even/even+1) GPRs and use %n and %Hn to refer to the individual regs 3980 // respectively. Since there is no constraint to explicitly specify a 3981 // reg pair, we use GPRPair reg class for "%r" for 64-bit data. For Thumb, 3982 // the 64-bit data may be referred by H, Q, R modifiers, so we still pack 3983 // them into a GPRPair. 3984 3985 SDLoc dl(N); 3986 SDValue Glue = N->getGluedNode() ? N->getOperand(NumOps-1) 3987 : SDValue(nullptr,0); 3988 3989 SmallVector<bool, 8> OpChanged; 3990 // Glue node will be appended late. 3991 for(unsigned i = 0, e = N->getGluedNode() ? NumOps - 1 : NumOps; i < e; ++i) { 3992 SDValue op = N->getOperand(i); 3993 AsmNodeOperands.push_back(op); 3994 3995 if (i < InlineAsm::Op_FirstOperand) 3996 continue; 3997 3998 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(i))) { 3999 Flag = C->getZExtValue(); 4000 Kind = InlineAsm::getKind(Flag); 4001 } 4002 else 4003 continue; 4004 4005 // Immediate operands to inline asm in the SelectionDAG are modeled with 4006 // two operands. The first is a constant of value InlineAsm::Kind_Imm, and 4007 // the second is a constant with the value of the immediate. If we get here 4008 // and we have a Kind_Imm, skip the next operand, and continue. 4009 if (Kind == InlineAsm::Kind_Imm) { 4010 SDValue op = N->getOperand(++i); 4011 AsmNodeOperands.push_back(op); 4012 continue; 4013 } 4014 4015 unsigned NumRegs = InlineAsm::getNumOperandRegisters(Flag); 4016 if (NumRegs) 4017 OpChanged.push_back(false); 4018 4019 unsigned DefIdx = 0; 4020 bool IsTiedToChangedOp = false; 4021 // If it's a use that is tied with a previous def, it has no 4022 // reg class constraint. 4023 if (Changed && InlineAsm::isUseOperandTiedToDef(Flag, DefIdx)) 4024 IsTiedToChangedOp = OpChanged[DefIdx]; 4025 4026 if (Kind != InlineAsm::Kind_RegUse && Kind != InlineAsm::Kind_RegDef 4027 && Kind != InlineAsm::Kind_RegDefEarlyClobber) 4028 continue; 4029 4030 unsigned RC; 4031 bool HasRC = InlineAsm::hasRegClassConstraint(Flag, RC); 4032 if ((!IsTiedToChangedOp && (!HasRC || RC != ARM::GPRRegClassID)) 4033 || NumRegs != 2) 4034 continue; 4035 4036 assert((i+2 < NumOps) && "Invalid number of operands in inline asm"); 4037 SDValue V0 = N->getOperand(i+1); 4038 SDValue V1 = N->getOperand(i+2); 4039 unsigned Reg0 = cast<RegisterSDNode>(V0)->getReg(); 4040 unsigned Reg1 = cast<RegisterSDNode>(V1)->getReg(); 4041 SDValue PairedReg; 4042 MachineRegisterInfo &MRI = MF->getRegInfo(); 4043 4044 if (Kind == InlineAsm::Kind_RegDef || 4045 Kind == InlineAsm::Kind_RegDefEarlyClobber) { 4046 // Replace the two GPRs with 1 GPRPair and copy values from GPRPair to 4047 // the original GPRs. 4048 4049 unsigned GPVR = MRI.createVirtualRegister(&ARM::GPRPairRegClass); 4050 PairedReg = CurDAG->getRegister(GPVR, MVT::Untyped); 4051 SDValue Chain = SDValue(N,0); 4052 4053 SDNode *GU = N->getGluedUser(); 4054 SDValue RegCopy = CurDAG->getCopyFromReg(Chain, dl, GPVR, MVT::Untyped, 4055 Chain.getValue(1)); 4056 4057 // Extract values from a GPRPair reg and copy to the original GPR reg. 4058 SDValue Sub0 = CurDAG->getTargetExtractSubreg(ARM::gsub_0, dl, MVT::i32, 4059 RegCopy); 4060 SDValue Sub1 = CurDAG->getTargetExtractSubreg(ARM::gsub_1, dl, MVT::i32, 4061 RegCopy); 4062 SDValue T0 = CurDAG->getCopyToReg(Sub0, dl, Reg0, Sub0, 4063 RegCopy.getValue(1)); 4064 SDValue T1 = CurDAG->getCopyToReg(Sub1, dl, Reg1, Sub1, T0.getValue(1)); 4065 4066 // Update the original glue user. 4067 std::vector<SDValue> Ops(GU->op_begin(), GU->op_end()-1); 4068 Ops.push_back(T1.getValue(1)); 4069 CurDAG->UpdateNodeOperands(GU, Ops); 4070 } 4071 else { 4072 // For Kind == InlineAsm::Kind_RegUse, we first copy two GPRs into a 4073 // GPRPair and then pass the GPRPair to the inline asm. 4074 SDValue Chain = AsmNodeOperands[InlineAsm::Op_InputChain]; 4075 4076 // As REG_SEQ doesn't take RegisterSDNode, we copy them first. 4077 SDValue T0 = CurDAG->getCopyFromReg(Chain, dl, Reg0, MVT::i32, 4078 Chain.getValue(1)); 4079 SDValue T1 = CurDAG->getCopyFromReg(Chain, dl, Reg1, MVT::i32, 4080 T0.getValue(1)); 4081 SDValue Pair = SDValue(createGPRPairNode(MVT::Untyped, T0, T1), 0); 4082 4083 // Copy REG_SEQ into a GPRPair-typed VR and replace the original two 4084 // i32 VRs of inline asm with it. 4085 unsigned GPVR = MRI.createVirtualRegister(&ARM::GPRPairRegClass); 4086 PairedReg = CurDAG->getRegister(GPVR, MVT::Untyped); 4087 Chain = CurDAG->getCopyToReg(T1, dl, GPVR, Pair, T1.getValue(1)); 4088 4089 AsmNodeOperands[InlineAsm::Op_InputChain] = Chain; 4090 Glue = Chain.getValue(1); 4091 } 4092 4093 Changed = true; 4094 4095 if(PairedReg.getNode()) { 4096 OpChanged[OpChanged.size() -1 ] = true; 4097 Flag = InlineAsm::getFlagWord(Kind, 1 /* RegNum*/); 4098 if (IsTiedToChangedOp) 4099 Flag = InlineAsm::getFlagWordForMatchingOp(Flag, DefIdx); 4100 else 4101 Flag = InlineAsm::getFlagWordForRegClass(Flag, ARM::GPRPairRegClassID); 4102 // Replace the current flag. 4103 AsmNodeOperands[AsmNodeOperands.size() -1] = CurDAG->getTargetConstant( 4104 Flag, dl, MVT::i32); 4105 // Add the new register node and skip the original two GPRs. 4106 AsmNodeOperands.push_back(PairedReg); 4107 // Skip the next two GPRs. 4108 i += 2; 4109 } 4110 } 4111 4112 if (Glue.getNode()) 4113 AsmNodeOperands.push_back(Glue); 4114 if (!Changed) 4115 return nullptr; 4116 4117 SDValue New = CurDAG->getNode(ISD::INLINEASM, SDLoc(N), 4118 CurDAG->getVTList(MVT::Other, MVT::Glue), AsmNodeOperands); 4119 New->setNodeId(-1); 4120 return New.getNode(); 4121 } 4122 4123 4124 bool ARMDAGToDAGISel:: 4125 SelectInlineAsmMemoryOperand(const SDValue &Op, unsigned ConstraintID, 4126 std::vector<SDValue> &OutOps) { 4127 switch(ConstraintID) { 4128 default: 4129 llvm_unreachable("Unexpected asm memory constraint"); 4130 case InlineAsm::Constraint_i: 4131 // FIXME: It seems strange that 'i' is needed here since it's supposed to 4132 // be an immediate and not a memory constraint. 4133 // Fallthrough. 4134 case InlineAsm::Constraint_m: 4135 case InlineAsm::Constraint_o: 4136 case InlineAsm::Constraint_Q: 4137 case InlineAsm::Constraint_Um: 4138 case InlineAsm::Constraint_Un: 4139 case InlineAsm::Constraint_Uq: 4140 case InlineAsm::Constraint_Us: 4141 case InlineAsm::Constraint_Ut: 4142 case InlineAsm::Constraint_Uv: 4143 case InlineAsm::Constraint_Uy: 4144 // Require the address to be in a register. That is safe for all ARM 4145 // variants and it is hard to do anything much smarter without knowing 4146 // how the operand is used. 4147 OutOps.push_back(Op); 4148 return false; 4149 } 4150 return true; 4151 } 4152 4153 /// createARMISelDag - This pass converts a legalized DAG into a 4154 /// ARM-specific DAG, ready for instruction scheduling. 4155 /// 4156 FunctionPass *llvm::createARMISelDag(ARMBaseTargetMachine &TM, 4157 CodeGenOpt::Level OptLevel) { 4158 return new ARMDAGToDAGISel(TM, OptLevel); 4159 } 4160