1 //===-- ARMISelDAGToDAG.cpp - A dag to dag inst selector for ARM ----------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This file defines an instruction selector for the ARM target. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "ARM.h" 14 #include "ARMBaseInstrInfo.h" 15 #include "ARMTargetMachine.h" 16 #include "MCTargetDesc/ARMAddressingModes.h" 17 #include "Utils/ARMBaseInfo.h" 18 #include "llvm/ADT/StringSwitch.h" 19 #include "llvm/CodeGen/MachineFrameInfo.h" 20 #include "llvm/CodeGen/MachineFunction.h" 21 #include "llvm/CodeGen/MachineInstrBuilder.h" 22 #include "llvm/CodeGen/MachineRegisterInfo.h" 23 #include "llvm/CodeGen/SelectionDAG.h" 24 #include "llvm/CodeGen/SelectionDAGISel.h" 25 #include "llvm/CodeGen/TargetLowering.h" 26 #include "llvm/IR/CallingConv.h" 27 #include "llvm/IR/Constants.h" 28 #include "llvm/IR/DerivedTypes.h" 29 #include "llvm/IR/Function.h" 30 #include "llvm/IR/Intrinsics.h" 31 #include "llvm/IR/IntrinsicsARM.h" 32 #include "llvm/IR/LLVMContext.h" 33 #include "llvm/Support/CommandLine.h" 34 #include "llvm/Support/Debug.h" 35 #include "llvm/Support/ErrorHandling.h" 36 #include "llvm/Target/TargetOptions.h" 37 38 using namespace llvm; 39 40 #define DEBUG_TYPE "arm-isel" 41 42 static cl::opt<bool> 43 DisableShifterOp("disable-shifter-op", cl::Hidden, 44 cl::desc("Disable isel of shifter-op"), 45 cl::init(false)); 46 47 //===--------------------------------------------------------------------===// 48 /// ARMDAGToDAGISel - ARM specific code to select ARM machine 49 /// instructions for SelectionDAG operations. 50 /// 51 namespace { 52 53 class ARMDAGToDAGISel : public SelectionDAGISel { 54 /// Subtarget - Keep a pointer to the ARMSubtarget around so that we can 55 /// make the right decision when generating code for different targets. 56 const ARMSubtarget *Subtarget; 57 58 public: 59 explicit ARMDAGToDAGISel(ARMBaseTargetMachine &tm, CodeGenOpt::Level OptLevel) 60 : SelectionDAGISel(tm, OptLevel) {} 61 62 bool runOnMachineFunction(MachineFunction &MF) override { 63 // Reset the subtarget each time through. 64 Subtarget = &MF.getSubtarget<ARMSubtarget>(); 65 SelectionDAGISel::runOnMachineFunction(MF); 66 return true; 67 } 68 69 StringRef getPassName() const override { return "ARM Instruction Selection"; } 70 71 void PreprocessISelDAG() override; 72 73 /// getI32Imm - Return a target constant of type i32 with the specified 74 /// value. 75 inline SDValue getI32Imm(unsigned Imm, const SDLoc &dl) { 76 return CurDAG->getTargetConstant(Imm, dl, MVT::i32); 77 } 78 79 void Select(SDNode *N) override; 80 81 bool hasNoVMLxHazardUse(SDNode *N) const; 82 bool isShifterOpProfitable(const SDValue &Shift, 83 ARM_AM::ShiftOpc ShOpcVal, unsigned ShAmt); 84 bool SelectRegShifterOperand(SDValue N, SDValue &A, 85 SDValue &B, SDValue &C, 86 bool CheckProfitability = true); 87 bool SelectImmShifterOperand(SDValue N, SDValue &A, 88 SDValue &B, bool CheckProfitability = true); 89 bool SelectShiftRegShifterOperand(SDValue N, SDValue &A, 90 SDValue &B, SDValue &C) { 91 // Don't apply the profitability check 92 return SelectRegShifterOperand(N, A, B, C, false); 93 } 94 bool SelectShiftImmShifterOperand(SDValue N, SDValue &A, 95 SDValue &B) { 96 // Don't apply the profitability check 97 return SelectImmShifterOperand(N, A, B, false); 98 } 99 100 bool SelectAddLikeOr(SDNode *Parent, SDValue N, SDValue &Out); 101 102 bool SelectAddrModeImm12(SDValue N, SDValue &Base, SDValue &OffImm); 103 bool SelectLdStSOReg(SDValue N, SDValue &Base, SDValue &Offset, SDValue &Opc); 104 105 bool SelectCMOVPred(SDValue N, SDValue &Pred, SDValue &Reg) { 106 const ConstantSDNode *CN = cast<ConstantSDNode>(N); 107 Pred = CurDAG->getTargetConstant(CN->getZExtValue(), SDLoc(N), MVT::i32); 108 Reg = CurDAG->getRegister(ARM::CPSR, MVT::i32); 109 return true; 110 } 111 112 bool SelectAddrMode2OffsetReg(SDNode *Op, SDValue N, 113 SDValue &Offset, SDValue &Opc); 114 bool SelectAddrMode2OffsetImm(SDNode *Op, SDValue N, 115 SDValue &Offset, SDValue &Opc); 116 bool SelectAddrMode2OffsetImmPre(SDNode *Op, SDValue N, 117 SDValue &Offset, SDValue &Opc); 118 bool SelectAddrOffsetNone(SDValue N, SDValue &Base); 119 bool SelectAddrMode3(SDValue N, SDValue &Base, 120 SDValue &Offset, SDValue &Opc); 121 bool SelectAddrMode3Offset(SDNode *Op, SDValue N, 122 SDValue &Offset, SDValue &Opc); 123 bool IsAddressingMode5(SDValue N, SDValue &Base, SDValue &Offset, bool FP16); 124 bool SelectAddrMode5(SDValue N, SDValue &Base, SDValue &Offset); 125 bool SelectAddrMode5FP16(SDValue N, SDValue &Base, SDValue &Offset); 126 bool SelectAddrMode6(SDNode *Parent, SDValue N, SDValue &Addr,SDValue &Align); 127 bool SelectAddrMode6Offset(SDNode *Op, SDValue N, SDValue &Offset); 128 129 bool SelectAddrModePC(SDValue N, SDValue &Offset, SDValue &Label); 130 131 // Thumb Addressing Modes: 132 bool SelectThumbAddrModeRR(SDValue N, SDValue &Base, SDValue &Offset); 133 bool SelectThumbAddrModeRRSext(SDValue N, SDValue &Base, SDValue &Offset); 134 bool SelectThumbAddrModeImm5S(SDValue N, unsigned Scale, SDValue &Base, 135 SDValue &OffImm); 136 bool SelectThumbAddrModeImm5S1(SDValue N, SDValue &Base, 137 SDValue &OffImm); 138 bool SelectThumbAddrModeImm5S2(SDValue N, SDValue &Base, 139 SDValue &OffImm); 140 bool SelectThumbAddrModeImm5S4(SDValue N, SDValue &Base, 141 SDValue &OffImm); 142 bool SelectThumbAddrModeSP(SDValue N, SDValue &Base, SDValue &OffImm); 143 template <unsigned Shift> 144 bool SelectTAddrModeImm7(SDValue N, SDValue &Base, SDValue &OffImm); 145 146 // Thumb 2 Addressing Modes: 147 bool SelectT2AddrModeImm12(SDValue N, SDValue &Base, SDValue &OffImm); 148 template <unsigned Shift> 149 bool SelectT2AddrModeImm8(SDValue N, SDValue &Base, SDValue &OffImm); 150 bool SelectT2AddrModeImm8(SDValue N, SDValue &Base, 151 SDValue &OffImm); 152 bool SelectT2AddrModeImm8Offset(SDNode *Op, SDValue N, 153 SDValue &OffImm); 154 template <unsigned Shift> 155 bool SelectT2AddrModeImm7Offset(SDNode *Op, SDValue N, SDValue &OffImm); 156 bool SelectT2AddrModeImm7Offset(SDNode *Op, SDValue N, SDValue &OffImm, 157 unsigned Shift); 158 template <unsigned Shift> 159 bool SelectT2AddrModeImm7(SDValue N, SDValue &Base, SDValue &OffImm); 160 bool SelectT2AddrModeSoReg(SDValue N, SDValue &Base, 161 SDValue &OffReg, SDValue &ShImm); 162 bool SelectT2AddrModeExclusive(SDValue N, SDValue &Base, SDValue &OffImm); 163 164 template<int Min, int Max> 165 bool SelectImmediateInRange(SDValue N, SDValue &OffImm); 166 167 inline bool is_so_imm(unsigned Imm) const { 168 return ARM_AM::getSOImmVal(Imm) != -1; 169 } 170 171 inline bool is_so_imm_not(unsigned Imm) const { 172 return ARM_AM::getSOImmVal(~Imm) != -1; 173 } 174 175 inline bool is_t2_so_imm(unsigned Imm) const { 176 return ARM_AM::getT2SOImmVal(Imm) != -1; 177 } 178 179 inline bool is_t2_so_imm_not(unsigned Imm) const { 180 return ARM_AM::getT2SOImmVal(~Imm) != -1; 181 } 182 183 // Include the pieces autogenerated from the target description. 184 #include "ARMGenDAGISel.inc" 185 186 private: 187 void transferMemOperands(SDNode *Src, SDNode *Dst); 188 189 /// Indexed (pre/post inc/dec) load matching code for ARM. 190 bool tryARMIndexedLoad(SDNode *N); 191 bool tryT1IndexedLoad(SDNode *N); 192 bool tryT2IndexedLoad(SDNode *N); 193 bool tryMVEIndexedLoad(SDNode *N); 194 195 /// SelectVLD - Select NEON load intrinsics. NumVecs should be 196 /// 1, 2, 3 or 4. The opcode arrays specify the instructions used for 197 /// loads of D registers and even subregs and odd subregs of Q registers. 198 /// For NumVecs <= 2, QOpcodes1 is not used. 199 void SelectVLD(SDNode *N, bool isUpdating, unsigned NumVecs, 200 const uint16_t *DOpcodes, const uint16_t *QOpcodes0, 201 const uint16_t *QOpcodes1); 202 203 /// SelectVST - Select NEON store intrinsics. NumVecs should 204 /// be 1, 2, 3 or 4. The opcode arrays specify the instructions used for 205 /// stores of D registers and even subregs and odd subregs of Q registers. 206 /// For NumVecs <= 2, QOpcodes1 is not used. 207 void SelectVST(SDNode *N, bool isUpdating, unsigned NumVecs, 208 const uint16_t *DOpcodes, const uint16_t *QOpcodes0, 209 const uint16_t *QOpcodes1); 210 211 /// SelectVLDSTLane - Select NEON load/store lane intrinsics. NumVecs should 212 /// be 2, 3 or 4. The opcode arrays specify the instructions used for 213 /// load/store of D registers and Q registers. 214 void SelectVLDSTLane(SDNode *N, bool IsLoad, bool isUpdating, 215 unsigned NumVecs, const uint16_t *DOpcodes, 216 const uint16_t *QOpcodes); 217 218 /// Helper functions for setting up clusters of MVE predication operands. 219 template <typename SDValueVector> 220 void AddMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc, 221 SDValue PredicateMask); 222 template <typename SDValueVector> 223 void AddMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc, 224 SDValue PredicateMask, SDValue Inactive); 225 226 template <typename SDValueVector> 227 void AddEmptyMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc); 228 template <typename SDValueVector> 229 void AddEmptyMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc, EVT InactiveTy); 230 231 /// SelectMVE_WB - Select MVE writeback load/store intrinsics. 232 void SelectMVE_WB(SDNode *N, const uint16_t *Opcodes, bool Predicated); 233 234 /// SelectMVE_LongShift - Select MVE 64-bit scalar shift intrinsics. 235 void SelectMVE_LongShift(SDNode *N, uint16_t Opcode, bool Immediate, 236 bool HasSaturationOperand); 237 238 /// SelectMVE_VADCSBC - Select MVE vector add/sub-with-carry intrinsics. 239 void SelectMVE_VADCSBC(SDNode *N, uint16_t OpcodeWithCarry, 240 uint16_t OpcodeWithNoCarry, bool Add, bool Predicated); 241 242 /// SelectMVE_VSHLC - Select MVE intrinsics for a shift that carries between 243 /// vector lanes. 244 void SelectMVE_VSHLC(SDNode *N, bool Predicated); 245 246 /// Select long MVE vector reductions with two vector operands 247 /// Stride is the number of vector element widths the instruction can operate 248 /// on: 249 /// 2 for long non-rounding variants, vml{a,s}ldav[a][x]: [i16, i32] 250 /// 1 for long rounding variants: vrml{a,s}ldavh[a][x]: [i32] 251 /// Stride is used when addressing the OpcodesS array which contains multiple 252 /// opcodes for each element width. 253 /// TySize is the index into the list of element types listed above 254 void SelectBaseMVE_VMLLDAV(SDNode *N, bool Predicated, 255 const uint16_t *OpcodesS, const uint16_t *OpcodesU, 256 size_t Stride, size_t TySize); 257 258 /// Select a 64-bit MVE vector reduction with two vector operands 259 /// arm_mve_vmlldava_[predicated] 260 void SelectMVE_VMLLDAV(SDNode *N, bool Predicated, const uint16_t *OpcodesS, 261 const uint16_t *OpcodesU); 262 /// Select a 72-bit MVE vector rounding reduction with two vector operands 263 /// int_arm_mve_vrmlldavha[_predicated] 264 void SelectMVE_VRMLLDAVH(SDNode *N, bool Predicated, const uint16_t *OpcodesS, 265 const uint16_t *OpcodesU); 266 267 /// SelectMVE_VLD - Select MVE interleaving load intrinsics. NumVecs 268 /// should be 2 or 4. The opcode array specifies the instructions 269 /// used for 8, 16 and 32-bit lane sizes respectively, and each 270 /// pointer points to a set of NumVecs sub-opcodes used for the 271 /// different stages (e.g. VLD20 versus VLD21) of each load family. 272 void SelectMVE_VLD(SDNode *N, unsigned NumVecs, 273 const uint16_t *const *Opcodes, bool HasWriteback); 274 275 /// SelectMVE_VxDUP - Select MVE incrementing-dup instructions. Opcodes is an 276 /// array of 3 elements for the 8, 16 and 32-bit lane sizes. 277 void SelectMVE_VxDUP(SDNode *N, const uint16_t *Opcodes, 278 bool Wrapping, bool Predicated); 279 280 /// SelectVLDDup - Select NEON load-duplicate intrinsics. NumVecs 281 /// should be 1, 2, 3 or 4. The opcode array specifies the instructions used 282 /// for loading D registers. 283 void SelectVLDDup(SDNode *N, bool IsIntrinsic, bool isUpdating, 284 unsigned NumVecs, const uint16_t *DOpcodes, 285 const uint16_t *QOpcodes0 = nullptr, 286 const uint16_t *QOpcodes1 = nullptr); 287 288 /// Try to select SBFX/UBFX instructions for ARM. 289 bool tryV6T2BitfieldExtractOp(SDNode *N, bool isSigned); 290 291 // Select special operations if node forms integer ABS pattern 292 bool tryABSOp(SDNode *N); 293 294 bool tryReadRegister(SDNode *N); 295 bool tryWriteRegister(SDNode *N); 296 297 bool tryInlineAsm(SDNode *N); 298 299 void SelectCMPZ(SDNode *N, bool &SwitchEQNEToPLMI); 300 301 void SelectCMP_SWAP(SDNode *N); 302 303 /// SelectInlineAsmMemoryOperand - Implement addressing mode selection for 304 /// inline asm expressions. 305 bool SelectInlineAsmMemoryOperand(const SDValue &Op, unsigned ConstraintID, 306 std::vector<SDValue> &OutOps) override; 307 308 // Form pairs of consecutive R, S, D, or Q registers. 309 SDNode *createGPRPairNode(EVT VT, SDValue V0, SDValue V1); 310 SDNode *createSRegPairNode(EVT VT, SDValue V0, SDValue V1); 311 SDNode *createDRegPairNode(EVT VT, SDValue V0, SDValue V1); 312 SDNode *createQRegPairNode(EVT VT, SDValue V0, SDValue V1); 313 314 // Form sequences of 4 consecutive S, D, or Q registers. 315 SDNode *createQuadSRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3); 316 SDNode *createQuadDRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3); 317 SDNode *createQuadQRegsNode(EVT VT, SDValue V0, SDValue V1, SDValue V2, SDValue V3); 318 319 // Get the alignment operand for a NEON VLD or VST instruction. 320 SDValue GetVLDSTAlign(SDValue Align, const SDLoc &dl, unsigned NumVecs, 321 bool is64BitVector); 322 323 /// Checks if N is a multiplication by a constant where we can extract out a 324 /// power of two from the constant so that it can be used in a shift, but only 325 /// if it simplifies the materialization of the constant. Returns true if it 326 /// is, and assigns to PowerOfTwo the power of two that should be extracted 327 /// out and to NewMulConst the new constant to be multiplied by. 328 bool canExtractShiftFromMul(const SDValue &N, unsigned MaxShift, 329 unsigned &PowerOfTwo, SDValue &NewMulConst) const; 330 331 /// Replace N with M in CurDAG, in a way that also ensures that M gets 332 /// selected when N would have been selected. 333 void replaceDAGValue(const SDValue &N, SDValue M); 334 }; 335 } 336 337 /// isInt32Immediate - This method tests to see if the node is a 32-bit constant 338 /// operand. If so Imm will receive the 32-bit value. 339 static bool isInt32Immediate(SDNode *N, unsigned &Imm) { 340 if (N->getOpcode() == ISD::Constant && N->getValueType(0) == MVT::i32) { 341 Imm = cast<ConstantSDNode>(N)->getZExtValue(); 342 return true; 343 } 344 return false; 345 } 346 347 // isInt32Immediate - This method tests to see if a constant operand. 348 // If so Imm will receive the 32 bit value. 349 static bool isInt32Immediate(SDValue N, unsigned &Imm) { 350 return isInt32Immediate(N.getNode(), Imm); 351 } 352 353 // isOpcWithIntImmediate - This method tests to see if the node is a specific 354 // opcode and that it has a immediate integer right operand. 355 // If so Imm will receive the 32 bit value. 356 static bool isOpcWithIntImmediate(SDNode *N, unsigned Opc, unsigned& Imm) { 357 return N->getOpcode() == Opc && 358 isInt32Immediate(N->getOperand(1).getNode(), Imm); 359 } 360 361 /// Check whether a particular node is a constant value representable as 362 /// (N * Scale) where (N in [\p RangeMin, \p RangeMax). 363 /// 364 /// \param ScaledConstant [out] - On success, the pre-scaled constant value. 365 static bool isScaledConstantInRange(SDValue Node, int Scale, 366 int RangeMin, int RangeMax, 367 int &ScaledConstant) { 368 assert(Scale > 0 && "Invalid scale!"); 369 370 // Check that this is a constant. 371 const ConstantSDNode *C = dyn_cast<ConstantSDNode>(Node); 372 if (!C) 373 return false; 374 375 ScaledConstant = (int) C->getZExtValue(); 376 if ((ScaledConstant % Scale) != 0) 377 return false; 378 379 ScaledConstant /= Scale; 380 return ScaledConstant >= RangeMin && ScaledConstant < RangeMax; 381 } 382 383 void ARMDAGToDAGISel::PreprocessISelDAG() { 384 if (!Subtarget->hasV6T2Ops()) 385 return; 386 387 bool isThumb2 = Subtarget->isThumb(); 388 for (SelectionDAG::allnodes_iterator I = CurDAG->allnodes_begin(), 389 E = CurDAG->allnodes_end(); I != E; ) { 390 SDNode *N = &*I++; // Preincrement iterator to avoid invalidation issues. 391 392 if (N->getOpcode() != ISD::ADD) 393 continue; 394 395 // Look for (add X1, (and (srl X2, c1), c2)) where c2 is constant with 396 // leading zeros, followed by consecutive set bits, followed by 1 or 2 397 // trailing zeros, e.g. 1020. 398 // Transform the expression to 399 // (add X1, (shl (and (srl X2, c1), (c2>>tz)), tz)) where tz is the number 400 // of trailing zeros of c2. The left shift would be folded as an shifter 401 // operand of 'add' and the 'and' and 'srl' would become a bits extraction 402 // node (UBFX). 403 404 SDValue N0 = N->getOperand(0); 405 SDValue N1 = N->getOperand(1); 406 unsigned And_imm = 0; 407 if (!isOpcWithIntImmediate(N1.getNode(), ISD::AND, And_imm)) { 408 if (isOpcWithIntImmediate(N0.getNode(), ISD::AND, And_imm)) 409 std::swap(N0, N1); 410 } 411 if (!And_imm) 412 continue; 413 414 // Check if the AND mask is an immediate of the form: 000.....1111111100 415 unsigned TZ = countTrailingZeros(And_imm); 416 if (TZ != 1 && TZ != 2) 417 // Be conservative here. Shifter operands aren't always free. e.g. On 418 // Swift, left shifter operand of 1 / 2 for free but others are not. 419 // e.g. 420 // ubfx r3, r1, #16, #8 421 // ldr.w r3, [r0, r3, lsl #2] 422 // vs. 423 // mov.w r9, #1020 424 // and.w r2, r9, r1, lsr #14 425 // ldr r2, [r0, r2] 426 continue; 427 And_imm >>= TZ; 428 if (And_imm & (And_imm + 1)) 429 continue; 430 431 // Look for (and (srl X, c1), c2). 432 SDValue Srl = N1.getOperand(0); 433 unsigned Srl_imm = 0; 434 if (!isOpcWithIntImmediate(Srl.getNode(), ISD::SRL, Srl_imm) || 435 (Srl_imm <= 2)) 436 continue; 437 438 // Make sure first operand is not a shifter operand which would prevent 439 // folding of the left shift. 440 SDValue CPTmp0; 441 SDValue CPTmp1; 442 SDValue CPTmp2; 443 if (isThumb2) { 444 if (SelectImmShifterOperand(N0, CPTmp0, CPTmp1)) 445 continue; 446 } else { 447 if (SelectImmShifterOperand(N0, CPTmp0, CPTmp1) || 448 SelectRegShifterOperand(N0, CPTmp0, CPTmp1, CPTmp2)) 449 continue; 450 } 451 452 // Now make the transformation. 453 Srl = CurDAG->getNode(ISD::SRL, SDLoc(Srl), MVT::i32, 454 Srl.getOperand(0), 455 CurDAG->getConstant(Srl_imm + TZ, SDLoc(Srl), 456 MVT::i32)); 457 N1 = CurDAG->getNode(ISD::AND, SDLoc(N1), MVT::i32, 458 Srl, 459 CurDAG->getConstant(And_imm, SDLoc(Srl), MVT::i32)); 460 N1 = CurDAG->getNode(ISD::SHL, SDLoc(N1), MVT::i32, 461 N1, CurDAG->getConstant(TZ, SDLoc(Srl), MVT::i32)); 462 CurDAG->UpdateNodeOperands(N, N0, N1); 463 } 464 } 465 466 /// hasNoVMLxHazardUse - Return true if it's desirable to select a FP MLA / MLS 467 /// node. VFP / NEON fp VMLA / VMLS instructions have special RAW hazards (at 468 /// least on current ARM implementations) which should be avoidded. 469 bool ARMDAGToDAGISel::hasNoVMLxHazardUse(SDNode *N) const { 470 if (OptLevel == CodeGenOpt::None) 471 return true; 472 473 if (!Subtarget->hasVMLxHazards()) 474 return true; 475 476 if (!N->hasOneUse()) 477 return false; 478 479 SDNode *Use = *N->use_begin(); 480 if (Use->getOpcode() == ISD::CopyToReg) 481 return true; 482 if (Use->isMachineOpcode()) { 483 const ARMBaseInstrInfo *TII = static_cast<const ARMBaseInstrInfo *>( 484 CurDAG->getSubtarget().getInstrInfo()); 485 486 const MCInstrDesc &MCID = TII->get(Use->getMachineOpcode()); 487 if (MCID.mayStore()) 488 return true; 489 unsigned Opcode = MCID.getOpcode(); 490 if (Opcode == ARM::VMOVRS || Opcode == ARM::VMOVRRD) 491 return true; 492 // vmlx feeding into another vmlx. We actually want to unfold 493 // the use later in the MLxExpansion pass. e.g. 494 // vmla 495 // vmla (stall 8 cycles) 496 // 497 // vmul (5 cycles) 498 // vadd (5 cycles) 499 // vmla 500 // This adds up to about 18 - 19 cycles. 501 // 502 // vmla 503 // vmul (stall 4 cycles) 504 // vadd adds up to about 14 cycles. 505 return TII->isFpMLxInstruction(Opcode); 506 } 507 508 return false; 509 } 510 511 bool ARMDAGToDAGISel::isShifterOpProfitable(const SDValue &Shift, 512 ARM_AM::ShiftOpc ShOpcVal, 513 unsigned ShAmt) { 514 if (!Subtarget->isLikeA9() && !Subtarget->isSwift()) 515 return true; 516 if (Shift.hasOneUse()) 517 return true; 518 // R << 2 is free. 519 return ShOpcVal == ARM_AM::lsl && 520 (ShAmt == 2 || (Subtarget->isSwift() && ShAmt == 1)); 521 } 522 523 bool ARMDAGToDAGISel::canExtractShiftFromMul(const SDValue &N, 524 unsigned MaxShift, 525 unsigned &PowerOfTwo, 526 SDValue &NewMulConst) const { 527 assert(N.getOpcode() == ISD::MUL); 528 assert(MaxShift > 0); 529 530 // If the multiply is used in more than one place then changing the constant 531 // will make other uses incorrect, so don't. 532 if (!N.hasOneUse()) return false; 533 // Check if the multiply is by a constant 534 ConstantSDNode *MulConst = dyn_cast<ConstantSDNode>(N.getOperand(1)); 535 if (!MulConst) return false; 536 // If the constant is used in more than one place then modifying it will mean 537 // we need to materialize two constants instead of one, which is a bad idea. 538 if (!MulConst->hasOneUse()) return false; 539 unsigned MulConstVal = MulConst->getZExtValue(); 540 if (MulConstVal == 0) return false; 541 542 // Find the largest power of 2 that MulConstVal is a multiple of 543 PowerOfTwo = MaxShift; 544 while ((MulConstVal % (1 << PowerOfTwo)) != 0) { 545 --PowerOfTwo; 546 if (PowerOfTwo == 0) return false; 547 } 548 549 // Only optimise if the new cost is better 550 unsigned NewMulConstVal = MulConstVal / (1 << PowerOfTwo); 551 NewMulConst = CurDAG->getConstant(NewMulConstVal, SDLoc(N), MVT::i32); 552 unsigned OldCost = ConstantMaterializationCost(MulConstVal, Subtarget); 553 unsigned NewCost = ConstantMaterializationCost(NewMulConstVal, Subtarget); 554 return NewCost < OldCost; 555 } 556 557 void ARMDAGToDAGISel::replaceDAGValue(const SDValue &N, SDValue M) { 558 CurDAG->RepositionNode(N.getNode()->getIterator(), M.getNode()); 559 ReplaceUses(N, M); 560 } 561 562 bool ARMDAGToDAGISel::SelectImmShifterOperand(SDValue N, 563 SDValue &BaseReg, 564 SDValue &Opc, 565 bool CheckProfitability) { 566 if (DisableShifterOp) 567 return false; 568 569 // If N is a multiply-by-constant and it's profitable to extract a shift and 570 // use it in a shifted operand do so. 571 if (N.getOpcode() == ISD::MUL) { 572 unsigned PowerOfTwo = 0; 573 SDValue NewMulConst; 574 if (canExtractShiftFromMul(N, 31, PowerOfTwo, NewMulConst)) { 575 HandleSDNode Handle(N); 576 SDLoc Loc(N); 577 replaceDAGValue(N.getOperand(1), NewMulConst); 578 BaseReg = Handle.getValue(); 579 Opc = CurDAG->getTargetConstant( 580 ARM_AM::getSORegOpc(ARM_AM::lsl, PowerOfTwo), Loc, MVT::i32); 581 return true; 582 } 583 } 584 585 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode()); 586 587 // Don't match base register only case. That is matched to a separate 588 // lower complexity pattern with explicit register operand. 589 if (ShOpcVal == ARM_AM::no_shift) return false; 590 591 BaseReg = N.getOperand(0); 592 unsigned ShImmVal = 0; 593 ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1)); 594 if (!RHS) return false; 595 ShImmVal = RHS->getZExtValue() & 31; 596 Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, ShImmVal), 597 SDLoc(N), MVT::i32); 598 return true; 599 } 600 601 bool ARMDAGToDAGISel::SelectRegShifterOperand(SDValue N, 602 SDValue &BaseReg, 603 SDValue &ShReg, 604 SDValue &Opc, 605 bool CheckProfitability) { 606 if (DisableShifterOp) 607 return false; 608 609 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode()); 610 611 // Don't match base register only case. That is matched to a separate 612 // lower complexity pattern with explicit register operand. 613 if (ShOpcVal == ARM_AM::no_shift) return false; 614 615 BaseReg = N.getOperand(0); 616 unsigned ShImmVal = 0; 617 ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1)); 618 if (RHS) return false; 619 620 ShReg = N.getOperand(1); 621 if (CheckProfitability && !isShifterOpProfitable(N, ShOpcVal, ShImmVal)) 622 return false; 623 Opc = CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, ShImmVal), 624 SDLoc(N), MVT::i32); 625 return true; 626 } 627 628 // Determine whether an ISD::OR's operands are suitable to turn the operation 629 // into an addition, which often has more compact encodings. 630 bool ARMDAGToDAGISel::SelectAddLikeOr(SDNode *Parent, SDValue N, SDValue &Out) { 631 assert(Parent->getOpcode() == ISD::OR && "unexpected parent"); 632 Out = N; 633 return CurDAG->haveNoCommonBitsSet(N, Parent->getOperand(1)); 634 } 635 636 637 bool ARMDAGToDAGISel::SelectAddrModeImm12(SDValue N, 638 SDValue &Base, 639 SDValue &OffImm) { 640 // Match simple R + imm12 operands. 641 642 // Base only. 643 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 644 !CurDAG->isBaseWithConstantOffset(N)) { 645 if (N.getOpcode() == ISD::FrameIndex) { 646 // Match frame index. 647 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 648 Base = CurDAG->getTargetFrameIndex( 649 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 650 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 651 return true; 652 } 653 654 if (N.getOpcode() == ARMISD::Wrapper && 655 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 656 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 657 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 658 Base = N.getOperand(0); 659 } else 660 Base = N; 661 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 662 return true; 663 } 664 665 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 666 int RHSC = (int)RHS->getSExtValue(); 667 if (N.getOpcode() == ISD::SUB) 668 RHSC = -RHSC; 669 670 if (RHSC > -0x1000 && RHSC < 0x1000) { // 12 bits 671 Base = N.getOperand(0); 672 if (Base.getOpcode() == ISD::FrameIndex) { 673 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 674 Base = CurDAG->getTargetFrameIndex( 675 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 676 } 677 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 678 return true; 679 } 680 } 681 682 // Base only. 683 Base = N; 684 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 685 return true; 686 } 687 688 689 690 bool ARMDAGToDAGISel::SelectLdStSOReg(SDValue N, SDValue &Base, SDValue &Offset, 691 SDValue &Opc) { 692 if (N.getOpcode() == ISD::MUL && 693 ((!Subtarget->isLikeA9() && !Subtarget->isSwift()) || N.hasOneUse())) { 694 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 695 // X * [3,5,9] -> X + X * [2,4,8] etc. 696 int RHSC = (int)RHS->getZExtValue(); 697 if (RHSC & 1) { 698 RHSC = RHSC & ~1; 699 ARM_AM::AddrOpc AddSub = ARM_AM::add; 700 if (RHSC < 0) { 701 AddSub = ARM_AM::sub; 702 RHSC = - RHSC; 703 } 704 if (isPowerOf2_32(RHSC)) { 705 unsigned ShAmt = Log2_32(RHSC); 706 Base = Offset = N.getOperand(0); 707 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, 708 ARM_AM::lsl), 709 SDLoc(N), MVT::i32); 710 return true; 711 } 712 } 713 } 714 } 715 716 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 717 // ISD::OR that is equivalent to an ISD::ADD. 718 !CurDAG->isBaseWithConstantOffset(N)) 719 return false; 720 721 // Leave simple R +/- imm12 operands for LDRi12 722 if (N.getOpcode() == ISD::ADD || N.getOpcode() == ISD::OR) { 723 int RHSC; 724 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1, 725 -0x1000+1, 0x1000, RHSC)) // 12 bits. 726 return false; 727 } 728 729 // Otherwise this is R +/- [possibly shifted] R. 730 ARM_AM::AddrOpc AddSub = N.getOpcode() == ISD::SUB ? ARM_AM::sub:ARM_AM::add; 731 ARM_AM::ShiftOpc ShOpcVal = 732 ARM_AM::getShiftOpcForNode(N.getOperand(1).getOpcode()); 733 unsigned ShAmt = 0; 734 735 Base = N.getOperand(0); 736 Offset = N.getOperand(1); 737 738 if (ShOpcVal != ARM_AM::no_shift) { 739 // Check to see if the RHS of the shift is a constant, if not, we can't fold 740 // it. 741 if (ConstantSDNode *Sh = 742 dyn_cast<ConstantSDNode>(N.getOperand(1).getOperand(1))) { 743 ShAmt = Sh->getZExtValue(); 744 if (isShifterOpProfitable(Offset, ShOpcVal, ShAmt)) 745 Offset = N.getOperand(1).getOperand(0); 746 else { 747 ShAmt = 0; 748 ShOpcVal = ARM_AM::no_shift; 749 } 750 } else { 751 ShOpcVal = ARM_AM::no_shift; 752 } 753 } 754 755 // Try matching (R shl C) + (R). 756 if (N.getOpcode() != ISD::SUB && ShOpcVal == ARM_AM::no_shift && 757 !(Subtarget->isLikeA9() || Subtarget->isSwift() || 758 N.getOperand(0).hasOneUse())) { 759 ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOperand(0).getOpcode()); 760 if (ShOpcVal != ARM_AM::no_shift) { 761 // Check to see if the RHS of the shift is a constant, if not, we can't 762 // fold it. 763 if (ConstantSDNode *Sh = 764 dyn_cast<ConstantSDNode>(N.getOperand(0).getOperand(1))) { 765 ShAmt = Sh->getZExtValue(); 766 if (isShifterOpProfitable(N.getOperand(0), ShOpcVal, ShAmt)) { 767 Offset = N.getOperand(0).getOperand(0); 768 Base = N.getOperand(1); 769 } else { 770 ShAmt = 0; 771 ShOpcVal = ARM_AM::no_shift; 772 } 773 } else { 774 ShOpcVal = ARM_AM::no_shift; 775 } 776 } 777 } 778 779 // If Offset is a multiply-by-constant and it's profitable to extract a shift 780 // and use it in a shifted operand do so. 781 if (Offset.getOpcode() == ISD::MUL && N.hasOneUse()) { 782 unsigned PowerOfTwo = 0; 783 SDValue NewMulConst; 784 if (canExtractShiftFromMul(Offset, 31, PowerOfTwo, NewMulConst)) { 785 HandleSDNode Handle(Offset); 786 replaceDAGValue(Offset.getOperand(1), NewMulConst); 787 Offset = Handle.getValue(); 788 ShAmt = PowerOfTwo; 789 ShOpcVal = ARM_AM::lsl; 790 } 791 } 792 793 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal), 794 SDLoc(N), MVT::i32); 795 return true; 796 } 797 798 bool ARMDAGToDAGISel::SelectAddrMode2OffsetReg(SDNode *Op, SDValue N, 799 SDValue &Offset, SDValue &Opc) { 800 unsigned Opcode = Op->getOpcode(); 801 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 802 ? cast<LoadSDNode>(Op)->getAddressingMode() 803 : cast<StoreSDNode>(Op)->getAddressingMode(); 804 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 805 ? ARM_AM::add : ARM_AM::sub; 806 int Val; 807 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) 808 return false; 809 810 Offset = N; 811 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(N.getOpcode()); 812 unsigned ShAmt = 0; 813 if (ShOpcVal != ARM_AM::no_shift) { 814 // Check to see if the RHS of the shift is a constant, if not, we can't fold 815 // it. 816 if (ConstantSDNode *Sh = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 817 ShAmt = Sh->getZExtValue(); 818 if (isShifterOpProfitable(N, ShOpcVal, ShAmt)) 819 Offset = N.getOperand(0); 820 else { 821 ShAmt = 0; 822 ShOpcVal = ARM_AM::no_shift; 823 } 824 } else { 825 ShOpcVal = ARM_AM::no_shift; 826 } 827 } 828 829 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, ShAmt, ShOpcVal), 830 SDLoc(N), MVT::i32); 831 return true; 832 } 833 834 bool ARMDAGToDAGISel::SelectAddrMode2OffsetImmPre(SDNode *Op, SDValue N, 835 SDValue &Offset, SDValue &Opc) { 836 unsigned Opcode = Op->getOpcode(); 837 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 838 ? cast<LoadSDNode>(Op)->getAddressingMode() 839 : cast<StoreSDNode>(Op)->getAddressingMode(); 840 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 841 ? ARM_AM::add : ARM_AM::sub; 842 int Val; 843 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) { // 12 bits. 844 if (AddSub == ARM_AM::sub) Val *= -1; 845 Offset = CurDAG->getRegister(0, MVT::i32); 846 Opc = CurDAG->getTargetConstant(Val, SDLoc(Op), MVT::i32); 847 return true; 848 } 849 850 return false; 851 } 852 853 854 bool ARMDAGToDAGISel::SelectAddrMode2OffsetImm(SDNode *Op, SDValue N, 855 SDValue &Offset, SDValue &Opc) { 856 unsigned Opcode = Op->getOpcode(); 857 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 858 ? cast<LoadSDNode>(Op)->getAddressingMode() 859 : cast<StoreSDNode>(Op)->getAddressingMode(); 860 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 861 ? ARM_AM::add : ARM_AM::sub; 862 int Val; 863 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x1000, Val)) { // 12 bits. 864 Offset = CurDAG->getRegister(0, MVT::i32); 865 Opc = CurDAG->getTargetConstant(ARM_AM::getAM2Opc(AddSub, Val, 866 ARM_AM::no_shift), 867 SDLoc(Op), MVT::i32); 868 return true; 869 } 870 871 return false; 872 } 873 874 bool ARMDAGToDAGISel::SelectAddrOffsetNone(SDValue N, SDValue &Base) { 875 Base = N; 876 return true; 877 } 878 879 bool ARMDAGToDAGISel::SelectAddrMode3(SDValue N, 880 SDValue &Base, SDValue &Offset, 881 SDValue &Opc) { 882 if (N.getOpcode() == ISD::SUB) { 883 // X - C is canonicalize to X + -C, no need to handle it here. 884 Base = N.getOperand(0); 885 Offset = N.getOperand(1); 886 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::sub, 0), SDLoc(N), 887 MVT::i32); 888 return true; 889 } 890 891 if (!CurDAG->isBaseWithConstantOffset(N)) { 892 Base = N; 893 if (N.getOpcode() == ISD::FrameIndex) { 894 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 895 Base = CurDAG->getTargetFrameIndex( 896 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 897 } 898 Offset = CurDAG->getRegister(0, MVT::i32); 899 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::add, 0), SDLoc(N), 900 MVT::i32); 901 return true; 902 } 903 904 // If the RHS is +/- imm8, fold into addr mode. 905 int RHSC; 906 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/1, 907 -256 + 1, 256, RHSC)) { // 8 bits. 908 Base = N.getOperand(0); 909 if (Base.getOpcode() == ISD::FrameIndex) { 910 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 911 Base = CurDAG->getTargetFrameIndex( 912 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 913 } 914 Offset = CurDAG->getRegister(0, MVT::i32); 915 916 ARM_AM::AddrOpc AddSub = ARM_AM::add; 917 if (RHSC < 0) { 918 AddSub = ARM_AM::sub; 919 RHSC = -RHSC; 920 } 921 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, RHSC), SDLoc(N), 922 MVT::i32); 923 return true; 924 } 925 926 Base = N.getOperand(0); 927 Offset = N.getOperand(1); 928 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(ARM_AM::add, 0), SDLoc(N), 929 MVT::i32); 930 return true; 931 } 932 933 bool ARMDAGToDAGISel::SelectAddrMode3Offset(SDNode *Op, SDValue N, 934 SDValue &Offset, SDValue &Opc) { 935 unsigned Opcode = Op->getOpcode(); 936 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 937 ? cast<LoadSDNode>(Op)->getAddressingMode() 938 : cast<StoreSDNode>(Op)->getAddressingMode(); 939 ARM_AM::AddrOpc AddSub = (AM == ISD::PRE_INC || AM == ISD::POST_INC) 940 ? ARM_AM::add : ARM_AM::sub; 941 int Val; 942 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 256, Val)) { // 12 bits. 943 Offset = CurDAG->getRegister(0, MVT::i32); 944 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, Val), SDLoc(Op), 945 MVT::i32); 946 return true; 947 } 948 949 Offset = N; 950 Opc = CurDAG->getTargetConstant(ARM_AM::getAM3Opc(AddSub, 0), SDLoc(Op), 951 MVT::i32); 952 return true; 953 } 954 955 bool ARMDAGToDAGISel::IsAddressingMode5(SDValue N, SDValue &Base, SDValue &Offset, 956 bool FP16) { 957 if (!CurDAG->isBaseWithConstantOffset(N)) { 958 Base = N; 959 if (N.getOpcode() == ISD::FrameIndex) { 960 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 961 Base = CurDAG->getTargetFrameIndex( 962 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 963 } else if (N.getOpcode() == ARMISD::Wrapper && 964 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 965 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 966 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 967 Base = N.getOperand(0); 968 } 969 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(ARM_AM::add, 0), 970 SDLoc(N), MVT::i32); 971 return true; 972 } 973 974 // If the RHS is +/- imm8, fold into addr mode. 975 int RHSC; 976 const int Scale = FP16 ? 2 : 4; 977 978 if (isScaledConstantInRange(N.getOperand(1), Scale, -255, 256, RHSC)) { 979 Base = N.getOperand(0); 980 if (Base.getOpcode() == ISD::FrameIndex) { 981 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 982 Base = CurDAG->getTargetFrameIndex( 983 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 984 } 985 986 ARM_AM::AddrOpc AddSub = ARM_AM::add; 987 if (RHSC < 0) { 988 AddSub = ARM_AM::sub; 989 RHSC = -RHSC; 990 } 991 992 if (FP16) 993 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5FP16Opc(AddSub, RHSC), 994 SDLoc(N), MVT::i32); 995 else 996 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(AddSub, RHSC), 997 SDLoc(N), MVT::i32); 998 999 return true; 1000 } 1001 1002 Base = N; 1003 1004 if (FP16) 1005 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5FP16Opc(ARM_AM::add, 0), 1006 SDLoc(N), MVT::i32); 1007 else 1008 Offset = CurDAG->getTargetConstant(ARM_AM::getAM5Opc(ARM_AM::add, 0), 1009 SDLoc(N), MVT::i32); 1010 1011 return true; 1012 } 1013 1014 bool ARMDAGToDAGISel::SelectAddrMode5(SDValue N, 1015 SDValue &Base, SDValue &Offset) { 1016 return IsAddressingMode5(N, Base, Offset, /*FP16=*/ false); 1017 } 1018 1019 bool ARMDAGToDAGISel::SelectAddrMode5FP16(SDValue N, 1020 SDValue &Base, SDValue &Offset) { 1021 return IsAddressingMode5(N, Base, Offset, /*FP16=*/ true); 1022 } 1023 1024 bool ARMDAGToDAGISel::SelectAddrMode6(SDNode *Parent, SDValue N, SDValue &Addr, 1025 SDValue &Align) { 1026 Addr = N; 1027 1028 unsigned Alignment = 0; 1029 1030 MemSDNode *MemN = cast<MemSDNode>(Parent); 1031 1032 if (isa<LSBaseSDNode>(MemN) || 1033 ((MemN->getOpcode() == ARMISD::VST1_UPD || 1034 MemN->getOpcode() == ARMISD::VLD1_UPD) && 1035 MemN->getConstantOperandVal(MemN->getNumOperands() - 1) == 1)) { 1036 // This case occurs only for VLD1-lane/dup and VST1-lane instructions. 1037 // The maximum alignment is equal to the memory size being referenced. 1038 unsigned MMOAlign = MemN->getAlignment(); 1039 unsigned MemSize = MemN->getMemoryVT().getSizeInBits() / 8; 1040 if (MMOAlign >= MemSize && MemSize > 1) 1041 Alignment = MemSize; 1042 } else { 1043 // All other uses of addrmode6 are for intrinsics. For now just record 1044 // the raw alignment value; it will be refined later based on the legal 1045 // alignment operands for the intrinsic. 1046 Alignment = MemN->getAlignment(); 1047 } 1048 1049 Align = CurDAG->getTargetConstant(Alignment, SDLoc(N), MVT::i32); 1050 return true; 1051 } 1052 1053 bool ARMDAGToDAGISel::SelectAddrMode6Offset(SDNode *Op, SDValue N, 1054 SDValue &Offset) { 1055 LSBaseSDNode *LdSt = cast<LSBaseSDNode>(Op); 1056 ISD::MemIndexedMode AM = LdSt->getAddressingMode(); 1057 if (AM != ISD::POST_INC) 1058 return false; 1059 Offset = N; 1060 if (ConstantSDNode *NC = dyn_cast<ConstantSDNode>(N)) { 1061 if (NC->getZExtValue() * 8 == LdSt->getMemoryVT().getSizeInBits()) 1062 Offset = CurDAG->getRegister(0, MVT::i32); 1063 } 1064 return true; 1065 } 1066 1067 bool ARMDAGToDAGISel::SelectAddrModePC(SDValue N, 1068 SDValue &Offset, SDValue &Label) { 1069 if (N.getOpcode() == ARMISD::PIC_ADD && N.hasOneUse()) { 1070 Offset = N.getOperand(0); 1071 SDValue N1 = N.getOperand(1); 1072 Label = CurDAG->getTargetConstant(cast<ConstantSDNode>(N1)->getZExtValue(), 1073 SDLoc(N), MVT::i32); 1074 return true; 1075 } 1076 1077 return false; 1078 } 1079 1080 1081 //===----------------------------------------------------------------------===// 1082 // Thumb Addressing Modes 1083 //===----------------------------------------------------------------------===// 1084 1085 static bool shouldUseZeroOffsetLdSt(SDValue N) { 1086 // Negative numbers are difficult to materialise in thumb1. If we are 1087 // selecting the add of a negative, instead try to select ri with a zero 1088 // offset, so create the add node directly which will become a sub. 1089 if (N.getOpcode() != ISD::ADD) 1090 return false; 1091 1092 // Look for an imm which is not legal for ld/st, but is legal for sub. 1093 if (auto C = dyn_cast<ConstantSDNode>(N.getOperand(1))) 1094 return C->getSExtValue() < 0 && C->getSExtValue() >= -255; 1095 1096 return false; 1097 } 1098 1099 bool ARMDAGToDAGISel::SelectThumbAddrModeRRSext(SDValue N, SDValue &Base, 1100 SDValue &Offset) { 1101 if (N.getOpcode() != ISD::ADD && !CurDAG->isBaseWithConstantOffset(N)) { 1102 ConstantSDNode *NC = dyn_cast<ConstantSDNode>(N); 1103 if (!NC || !NC->isNullValue()) 1104 return false; 1105 1106 Base = Offset = N; 1107 return true; 1108 } 1109 1110 Base = N.getOperand(0); 1111 Offset = N.getOperand(1); 1112 return true; 1113 } 1114 1115 bool ARMDAGToDAGISel::SelectThumbAddrModeRR(SDValue N, SDValue &Base, 1116 SDValue &Offset) { 1117 if (shouldUseZeroOffsetLdSt(N)) 1118 return false; // Select ri instead 1119 return SelectThumbAddrModeRRSext(N, Base, Offset); 1120 } 1121 1122 bool 1123 ARMDAGToDAGISel::SelectThumbAddrModeImm5S(SDValue N, unsigned Scale, 1124 SDValue &Base, SDValue &OffImm) { 1125 if (shouldUseZeroOffsetLdSt(N)) { 1126 Base = N; 1127 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1128 return true; 1129 } 1130 1131 if (!CurDAG->isBaseWithConstantOffset(N)) { 1132 if (N.getOpcode() == ISD::ADD) { 1133 return false; // We want to select register offset instead 1134 } else if (N.getOpcode() == ARMISD::Wrapper && 1135 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 1136 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 1137 N.getOperand(0).getOpcode() != ISD::TargetConstantPool && 1138 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 1139 Base = N.getOperand(0); 1140 } else { 1141 Base = N; 1142 } 1143 1144 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1145 return true; 1146 } 1147 1148 // If the RHS is + imm5 * scale, fold into addr mode. 1149 int RHSC; 1150 if (isScaledConstantInRange(N.getOperand(1), Scale, 0, 32, RHSC)) { 1151 Base = N.getOperand(0); 1152 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1153 return true; 1154 } 1155 1156 // Offset is too large, so use register offset instead. 1157 return false; 1158 } 1159 1160 bool 1161 ARMDAGToDAGISel::SelectThumbAddrModeImm5S4(SDValue N, SDValue &Base, 1162 SDValue &OffImm) { 1163 return SelectThumbAddrModeImm5S(N, 4, Base, OffImm); 1164 } 1165 1166 bool 1167 ARMDAGToDAGISel::SelectThumbAddrModeImm5S2(SDValue N, SDValue &Base, 1168 SDValue &OffImm) { 1169 return SelectThumbAddrModeImm5S(N, 2, Base, OffImm); 1170 } 1171 1172 bool 1173 ARMDAGToDAGISel::SelectThumbAddrModeImm5S1(SDValue N, SDValue &Base, 1174 SDValue &OffImm) { 1175 return SelectThumbAddrModeImm5S(N, 1, Base, OffImm); 1176 } 1177 1178 bool ARMDAGToDAGISel::SelectThumbAddrModeSP(SDValue N, 1179 SDValue &Base, SDValue &OffImm) { 1180 if (N.getOpcode() == ISD::FrameIndex) { 1181 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 1182 // Only multiples of 4 are allowed for the offset, so the frame object 1183 // alignment must be at least 4. 1184 MachineFrameInfo &MFI = MF->getFrameInfo(); 1185 if (MFI.getObjectAlignment(FI) < 4) 1186 MFI.setObjectAlignment(FI, 4); 1187 Base = CurDAG->getTargetFrameIndex( 1188 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1189 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1190 return true; 1191 } 1192 1193 if (!CurDAG->isBaseWithConstantOffset(N)) 1194 return false; 1195 1196 if (N.getOperand(0).getOpcode() == ISD::FrameIndex) { 1197 // If the RHS is + imm8 * scale, fold into addr mode. 1198 int RHSC; 1199 if (isScaledConstantInRange(N.getOperand(1), /*Scale=*/4, 0, 256, RHSC)) { 1200 Base = N.getOperand(0); 1201 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1202 // Make sure the offset is inside the object, or we might fail to 1203 // allocate an emergency spill slot. (An out-of-range access is UB, but 1204 // it could show up anyway.) 1205 MachineFrameInfo &MFI = MF->getFrameInfo(); 1206 if (RHSC * 4 < MFI.getObjectSize(FI)) { 1207 // For LHS+RHS to result in an offset that's a multiple of 4 the object 1208 // indexed by the LHS must be 4-byte aligned. 1209 if (!MFI.isFixedObjectIndex(FI) && MFI.getObjectAlignment(FI) < 4) 1210 MFI.setObjectAlignment(FI, 4); 1211 if (MFI.getObjectAlignment(FI) >= 4) { 1212 Base = CurDAG->getTargetFrameIndex( 1213 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1214 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1215 return true; 1216 } 1217 } 1218 } 1219 } 1220 1221 return false; 1222 } 1223 1224 template <unsigned Shift> 1225 bool ARMDAGToDAGISel::SelectTAddrModeImm7(SDValue N, SDValue &Base, 1226 SDValue &OffImm) { 1227 if (N.getOpcode() == ISD::SUB || CurDAG->isBaseWithConstantOffset(N)) { 1228 int RHSC; 1229 if (isScaledConstantInRange(N.getOperand(1), 1 << Shift, -0x7f, 0x80, 1230 RHSC)) { 1231 Base = N.getOperand(0); 1232 if (N.getOpcode() == ISD::SUB) 1233 RHSC = -RHSC; 1234 OffImm = 1235 CurDAG->getTargetConstant(RHSC * (1 << Shift), SDLoc(N), MVT::i32); 1236 return true; 1237 } 1238 } 1239 1240 // Base only. 1241 Base = N; 1242 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1243 return true; 1244 } 1245 1246 1247 //===----------------------------------------------------------------------===// 1248 // Thumb 2 Addressing Modes 1249 //===----------------------------------------------------------------------===// 1250 1251 1252 bool ARMDAGToDAGISel::SelectT2AddrModeImm12(SDValue N, 1253 SDValue &Base, SDValue &OffImm) { 1254 // Match simple R + imm12 operands. 1255 1256 // Base only. 1257 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 1258 !CurDAG->isBaseWithConstantOffset(N)) { 1259 if (N.getOpcode() == ISD::FrameIndex) { 1260 // Match frame index. 1261 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 1262 Base = CurDAG->getTargetFrameIndex( 1263 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1264 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1265 return true; 1266 } 1267 1268 if (N.getOpcode() == ARMISD::Wrapper && 1269 N.getOperand(0).getOpcode() != ISD::TargetGlobalAddress && 1270 N.getOperand(0).getOpcode() != ISD::TargetExternalSymbol && 1271 N.getOperand(0).getOpcode() != ISD::TargetGlobalTLSAddress) { 1272 Base = N.getOperand(0); 1273 if (Base.getOpcode() == ISD::TargetConstantPool) 1274 return false; // We want to select t2LDRpci instead. 1275 } else 1276 Base = N; 1277 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1278 return true; 1279 } 1280 1281 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 1282 if (SelectT2AddrModeImm8(N, Base, OffImm)) 1283 // Let t2LDRi8 handle (R - imm8). 1284 return false; 1285 1286 int RHSC = (int)RHS->getZExtValue(); 1287 if (N.getOpcode() == ISD::SUB) 1288 RHSC = -RHSC; 1289 1290 if (RHSC >= 0 && RHSC < 0x1000) { // 12 bits (unsigned) 1291 Base = N.getOperand(0); 1292 if (Base.getOpcode() == ISD::FrameIndex) { 1293 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1294 Base = CurDAG->getTargetFrameIndex( 1295 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1296 } 1297 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1298 return true; 1299 } 1300 } 1301 1302 // Base only. 1303 Base = N; 1304 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1305 return true; 1306 } 1307 1308 template <unsigned Shift> 1309 bool ARMDAGToDAGISel::SelectT2AddrModeImm8(SDValue N, SDValue &Base, 1310 SDValue &OffImm) { 1311 if (N.getOpcode() == ISD::SUB || CurDAG->isBaseWithConstantOffset(N)) { 1312 int RHSC; 1313 if (isScaledConstantInRange(N.getOperand(1), 1 << Shift, -255, 256, RHSC)) { 1314 Base = N.getOperand(0); 1315 if (Base.getOpcode() == ISD::FrameIndex) { 1316 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1317 Base = CurDAG->getTargetFrameIndex( 1318 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1319 } 1320 1321 if (N.getOpcode() == ISD::SUB) 1322 RHSC = -RHSC; 1323 OffImm = 1324 CurDAG->getTargetConstant(RHSC * (1 << Shift), SDLoc(N), MVT::i32); 1325 return true; 1326 } 1327 } 1328 1329 // Base only. 1330 Base = N; 1331 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1332 return true; 1333 } 1334 1335 bool ARMDAGToDAGISel::SelectT2AddrModeImm8(SDValue N, 1336 SDValue &Base, SDValue &OffImm) { 1337 // Match simple R - imm8 operands. 1338 if (N.getOpcode() != ISD::ADD && N.getOpcode() != ISD::SUB && 1339 !CurDAG->isBaseWithConstantOffset(N)) 1340 return false; 1341 1342 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 1343 int RHSC = (int)RHS->getSExtValue(); 1344 if (N.getOpcode() == ISD::SUB) 1345 RHSC = -RHSC; 1346 1347 if ((RHSC >= -255) && (RHSC < 0)) { // 8 bits (always negative) 1348 Base = N.getOperand(0); 1349 if (Base.getOpcode() == ISD::FrameIndex) { 1350 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1351 Base = CurDAG->getTargetFrameIndex( 1352 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1353 } 1354 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32); 1355 return true; 1356 } 1357 } 1358 1359 return false; 1360 } 1361 1362 bool ARMDAGToDAGISel::SelectT2AddrModeImm8Offset(SDNode *Op, SDValue N, 1363 SDValue &OffImm){ 1364 unsigned Opcode = Op->getOpcode(); 1365 ISD::MemIndexedMode AM = (Opcode == ISD::LOAD) 1366 ? cast<LoadSDNode>(Op)->getAddressingMode() 1367 : cast<StoreSDNode>(Op)->getAddressingMode(); 1368 int RHSC; 1369 if (isScaledConstantInRange(N, /*Scale=*/1, 0, 0x100, RHSC)) { // 8 bits. 1370 OffImm = ((AM == ISD::PRE_INC) || (AM == ISD::POST_INC)) 1371 ? CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i32) 1372 : CurDAG->getTargetConstant(-RHSC, SDLoc(N), MVT::i32); 1373 return true; 1374 } 1375 1376 return false; 1377 } 1378 1379 template <unsigned Shift> 1380 bool ARMDAGToDAGISel::SelectT2AddrModeImm7(SDValue N, SDValue &Base, 1381 SDValue &OffImm) { 1382 if (N.getOpcode() == ISD::SUB || CurDAG->isBaseWithConstantOffset(N)) { 1383 int RHSC; 1384 if (isScaledConstantInRange(N.getOperand(1), 1 << Shift, -0x7f, 0x80, 1385 RHSC)) { 1386 Base = N.getOperand(0); 1387 if (Base.getOpcode() == ISD::FrameIndex) { 1388 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1389 Base = CurDAG->getTargetFrameIndex( 1390 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1391 } 1392 1393 if (N.getOpcode() == ISD::SUB) 1394 RHSC = -RHSC; 1395 OffImm = 1396 CurDAG->getTargetConstant(RHSC * (1 << Shift), SDLoc(N), MVT::i32); 1397 return true; 1398 } 1399 } 1400 1401 // Base only. 1402 Base = N; 1403 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1404 return true; 1405 } 1406 1407 template <unsigned Shift> 1408 bool ARMDAGToDAGISel::SelectT2AddrModeImm7Offset(SDNode *Op, SDValue N, 1409 SDValue &OffImm) { 1410 return SelectT2AddrModeImm7Offset(Op, N, OffImm, Shift); 1411 } 1412 1413 bool ARMDAGToDAGISel::SelectT2AddrModeImm7Offset(SDNode *Op, SDValue N, 1414 SDValue &OffImm, 1415 unsigned Shift) { 1416 unsigned Opcode = Op->getOpcode(); 1417 ISD::MemIndexedMode AM; 1418 switch (Opcode) { 1419 case ISD::LOAD: 1420 AM = cast<LoadSDNode>(Op)->getAddressingMode(); 1421 break; 1422 case ISD::STORE: 1423 AM = cast<StoreSDNode>(Op)->getAddressingMode(); 1424 break; 1425 case ISD::MLOAD: 1426 AM = cast<MaskedLoadSDNode>(Op)->getAddressingMode(); 1427 break; 1428 case ISD::MSTORE: 1429 AM = cast<MaskedStoreSDNode>(Op)->getAddressingMode(); 1430 break; 1431 default: 1432 llvm_unreachable("Unexpected Opcode for Imm7Offset"); 1433 } 1434 1435 int RHSC; 1436 // 7 bit constant, shifted by Shift. 1437 if (isScaledConstantInRange(N, 1 << Shift, 0, 0x80, RHSC)) { 1438 OffImm = 1439 ((AM == ISD::PRE_INC) || (AM == ISD::POST_INC)) 1440 ? CurDAG->getTargetConstant(RHSC * (1 << Shift), SDLoc(N), MVT::i32) 1441 : CurDAG->getTargetConstant(-RHSC * (1 << Shift), SDLoc(N), 1442 MVT::i32); 1443 return true; 1444 } 1445 return false; 1446 } 1447 1448 template <int Min, int Max> 1449 bool ARMDAGToDAGISel::SelectImmediateInRange(SDValue N, SDValue &OffImm) { 1450 int Val; 1451 if (isScaledConstantInRange(N, 1, Min, Max, Val)) { 1452 OffImm = CurDAG->getTargetConstant(Val, SDLoc(N), MVT::i32); 1453 return true; 1454 } 1455 return false; 1456 } 1457 1458 bool ARMDAGToDAGISel::SelectT2AddrModeSoReg(SDValue N, 1459 SDValue &Base, 1460 SDValue &OffReg, SDValue &ShImm) { 1461 // (R - imm8) should be handled by t2LDRi8. The rest are handled by t2LDRi12. 1462 if (N.getOpcode() != ISD::ADD && !CurDAG->isBaseWithConstantOffset(N)) 1463 return false; 1464 1465 // Leave (R + imm12) for t2LDRi12, (R - imm8) for t2LDRi8. 1466 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) { 1467 int RHSC = (int)RHS->getZExtValue(); 1468 if (RHSC >= 0 && RHSC < 0x1000) // 12 bits (unsigned) 1469 return false; 1470 else if (RHSC < 0 && RHSC >= -255) // 8 bits 1471 return false; 1472 } 1473 1474 // Look for (R + R) or (R + (R << [1,2,3])). 1475 unsigned ShAmt = 0; 1476 Base = N.getOperand(0); 1477 OffReg = N.getOperand(1); 1478 1479 // Swap if it is ((R << c) + R). 1480 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(OffReg.getOpcode()); 1481 if (ShOpcVal != ARM_AM::lsl) { 1482 ShOpcVal = ARM_AM::getShiftOpcForNode(Base.getOpcode()); 1483 if (ShOpcVal == ARM_AM::lsl) 1484 std::swap(Base, OffReg); 1485 } 1486 1487 if (ShOpcVal == ARM_AM::lsl) { 1488 // Check to see if the RHS of the shift is a constant, if not, we can't fold 1489 // it. 1490 if (ConstantSDNode *Sh = dyn_cast<ConstantSDNode>(OffReg.getOperand(1))) { 1491 ShAmt = Sh->getZExtValue(); 1492 if (ShAmt < 4 && isShifterOpProfitable(OffReg, ShOpcVal, ShAmt)) 1493 OffReg = OffReg.getOperand(0); 1494 else { 1495 ShAmt = 0; 1496 } 1497 } 1498 } 1499 1500 // If OffReg is a multiply-by-constant and it's profitable to extract a shift 1501 // and use it in a shifted operand do so. 1502 if (OffReg.getOpcode() == ISD::MUL && N.hasOneUse()) { 1503 unsigned PowerOfTwo = 0; 1504 SDValue NewMulConst; 1505 if (canExtractShiftFromMul(OffReg, 3, PowerOfTwo, NewMulConst)) { 1506 HandleSDNode Handle(OffReg); 1507 replaceDAGValue(OffReg.getOperand(1), NewMulConst); 1508 OffReg = Handle.getValue(); 1509 ShAmt = PowerOfTwo; 1510 } 1511 } 1512 1513 ShImm = CurDAG->getTargetConstant(ShAmt, SDLoc(N), MVT::i32); 1514 1515 return true; 1516 } 1517 1518 bool ARMDAGToDAGISel::SelectT2AddrModeExclusive(SDValue N, SDValue &Base, 1519 SDValue &OffImm) { 1520 // This *must* succeed since it's used for the irreplaceable ldrex and strex 1521 // instructions. 1522 Base = N; 1523 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i32); 1524 1525 if (N.getOpcode() != ISD::ADD || !CurDAG->isBaseWithConstantOffset(N)) 1526 return true; 1527 1528 ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1)); 1529 if (!RHS) 1530 return true; 1531 1532 uint32_t RHSC = (int)RHS->getZExtValue(); 1533 if (RHSC > 1020 || RHSC % 4 != 0) 1534 return true; 1535 1536 Base = N.getOperand(0); 1537 if (Base.getOpcode() == ISD::FrameIndex) { 1538 int FI = cast<FrameIndexSDNode>(Base)->getIndex(); 1539 Base = CurDAG->getTargetFrameIndex( 1540 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 1541 } 1542 1543 OffImm = CurDAG->getTargetConstant(RHSC/4, SDLoc(N), MVT::i32); 1544 return true; 1545 } 1546 1547 //===--------------------------------------------------------------------===// 1548 1549 /// getAL - Returns a ARMCC::AL immediate node. 1550 static inline SDValue getAL(SelectionDAG *CurDAG, const SDLoc &dl) { 1551 return CurDAG->getTargetConstant((uint64_t)ARMCC::AL, dl, MVT::i32); 1552 } 1553 1554 void ARMDAGToDAGISel::transferMemOperands(SDNode *N, SDNode *Result) { 1555 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand(); 1556 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Result), {MemOp}); 1557 } 1558 1559 bool ARMDAGToDAGISel::tryARMIndexedLoad(SDNode *N) { 1560 LoadSDNode *LD = cast<LoadSDNode>(N); 1561 ISD::MemIndexedMode AM = LD->getAddressingMode(); 1562 if (AM == ISD::UNINDEXED) 1563 return false; 1564 1565 EVT LoadedVT = LD->getMemoryVT(); 1566 SDValue Offset, AMOpc; 1567 bool isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC); 1568 unsigned Opcode = 0; 1569 bool Match = false; 1570 if (LoadedVT == MVT::i32 && isPre && 1571 SelectAddrMode2OffsetImmPre(N, LD->getOffset(), Offset, AMOpc)) { 1572 Opcode = ARM::LDR_PRE_IMM; 1573 Match = true; 1574 } else if (LoadedVT == MVT::i32 && !isPre && 1575 SelectAddrMode2OffsetImm(N, LD->getOffset(), Offset, AMOpc)) { 1576 Opcode = ARM::LDR_POST_IMM; 1577 Match = true; 1578 } else if (LoadedVT == MVT::i32 && 1579 SelectAddrMode2OffsetReg(N, LD->getOffset(), Offset, AMOpc)) { 1580 Opcode = isPre ? ARM::LDR_PRE_REG : ARM::LDR_POST_REG; 1581 Match = true; 1582 1583 } else if (LoadedVT == MVT::i16 && 1584 SelectAddrMode3Offset(N, LD->getOffset(), Offset, AMOpc)) { 1585 Match = true; 1586 Opcode = (LD->getExtensionType() == ISD::SEXTLOAD) 1587 ? (isPre ? ARM::LDRSH_PRE : ARM::LDRSH_POST) 1588 : (isPre ? ARM::LDRH_PRE : ARM::LDRH_POST); 1589 } else if (LoadedVT == MVT::i8 || LoadedVT == MVT::i1) { 1590 if (LD->getExtensionType() == ISD::SEXTLOAD) { 1591 if (SelectAddrMode3Offset(N, LD->getOffset(), Offset, AMOpc)) { 1592 Match = true; 1593 Opcode = isPre ? ARM::LDRSB_PRE : ARM::LDRSB_POST; 1594 } 1595 } else { 1596 if (isPre && 1597 SelectAddrMode2OffsetImmPre(N, LD->getOffset(), Offset, AMOpc)) { 1598 Match = true; 1599 Opcode = ARM::LDRB_PRE_IMM; 1600 } else if (!isPre && 1601 SelectAddrMode2OffsetImm(N, LD->getOffset(), Offset, AMOpc)) { 1602 Match = true; 1603 Opcode = ARM::LDRB_POST_IMM; 1604 } else if (SelectAddrMode2OffsetReg(N, LD->getOffset(), Offset, AMOpc)) { 1605 Match = true; 1606 Opcode = isPre ? ARM::LDRB_PRE_REG : ARM::LDRB_POST_REG; 1607 } 1608 } 1609 } 1610 1611 if (Match) { 1612 if (Opcode == ARM::LDR_PRE_IMM || Opcode == ARM::LDRB_PRE_IMM) { 1613 SDValue Chain = LD->getChain(); 1614 SDValue Base = LD->getBasePtr(); 1615 SDValue Ops[]= { Base, AMOpc, getAL(CurDAG, SDLoc(N)), 1616 CurDAG->getRegister(0, MVT::i32), Chain }; 1617 SDNode *New = CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, MVT::i32, 1618 MVT::Other, Ops); 1619 transferMemOperands(N, New); 1620 ReplaceNode(N, New); 1621 return true; 1622 } else { 1623 SDValue Chain = LD->getChain(); 1624 SDValue Base = LD->getBasePtr(); 1625 SDValue Ops[]= { Base, Offset, AMOpc, getAL(CurDAG, SDLoc(N)), 1626 CurDAG->getRegister(0, MVT::i32), Chain }; 1627 SDNode *New = CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, MVT::i32, 1628 MVT::Other, Ops); 1629 transferMemOperands(N, New); 1630 ReplaceNode(N, New); 1631 return true; 1632 } 1633 } 1634 1635 return false; 1636 } 1637 1638 bool ARMDAGToDAGISel::tryT1IndexedLoad(SDNode *N) { 1639 LoadSDNode *LD = cast<LoadSDNode>(N); 1640 EVT LoadedVT = LD->getMemoryVT(); 1641 ISD::MemIndexedMode AM = LD->getAddressingMode(); 1642 if (AM != ISD::POST_INC || LD->getExtensionType() != ISD::NON_EXTLOAD || 1643 LoadedVT.getSimpleVT().SimpleTy != MVT::i32) 1644 return false; 1645 1646 auto *COffs = dyn_cast<ConstantSDNode>(LD->getOffset()); 1647 if (!COffs || COffs->getZExtValue() != 4) 1648 return false; 1649 1650 // A T1 post-indexed load is just a single register LDM: LDM r0!, {r1}. 1651 // The encoding of LDM is not how the rest of ISel expects a post-inc load to 1652 // look however, so we use a pseudo here and switch it for a tLDMIA_UPD after 1653 // ISel. 1654 SDValue Chain = LD->getChain(); 1655 SDValue Base = LD->getBasePtr(); 1656 SDValue Ops[]= { Base, getAL(CurDAG, SDLoc(N)), 1657 CurDAG->getRegister(0, MVT::i32), Chain }; 1658 SDNode *New = CurDAG->getMachineNode(ARM::tLDR_postidx, SDLoc(N), MVT::i32, 1659 MVT::i32, MVT::Other, Ops); 1660 transferMemOperands(N, New); 1661 ReplaceNode(N, New); 1662 return true; 1663 } 1664 1665 bool ARMDAGToDAGISel::tryT2IndexedLoad(SDNode *N) { 1666 LoadSDNode *LD = cast<LoadSDNode>(N); 1667 ISD::MemIndexedMode AM = LD->getAddressingMode(); 1668 if (AM == ISD::UNINDEXED) 1669 return false; 1670 1671 EVT LoadedVT = LD->getMemoryVT(); 1672 bool isSExtLd = LD->getExtensionType() == ISD::SEXTLOAD; 1673 SDValue Offset; 1674 bool isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC); 1675 unsigned Opcode = 0; 1676 bool Match = false; 1677 if (SelectT2AddrModeImm8Offset(N, LD->getOffset(), Offset)) { 1678 switch (LoadedVT.getSimpleVT().SimpleTy) { 1679 case MVT::i32: 1680 Opcode = isPre ? ARM::t2LDR_PRE : ARM::t2LDR_POST; 1681 break; 1682 case MVT::i16: 1683 if (isSExtLd) 1684 Opcode = isPre ? ARM::t2LDRSH_PRE : ARM::t2LDRSH_POST; 1685 else 1686 Opcode = isPre ? ARM::t2LDRH_PRE : ARM::t2LDRH_POST; 1687 break; 1688 case MVT::i8: 1689 case MVT::i1: 1690 if (isSExtLd) 1691 Opcode = isPre ? ARM::t2LDRSB_PRE : ARM::t2LDRSB_POST; 1692 else 1693 Opcode = isPre ? ARM::t2LDRB_PRE : ARM::t2LDRB_POST; 1694 break; 1695 default: 1696 return false; 1697 } 1698 Match = true; 1699 } 1700 1701 if (Match) { 1702 SDValue Chain = LD->getChain(); 1703 SDValue Base = LD->getBasePtr(); 1704 SDValue Ops[]= { Base, Offset, getAL(CurDAG, SDLoc(N)), 1705 CurDAG->getRegister(0, MVT::i32), Chain }; 1706 SDNode *New = CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, MVT::i32, 1707 MVT::Other, Ops); 1708 transferMemOperands(N, New); 1709 ReplaceNode(N, New); 1710 return true; 1711 } 1712 1713 return false; 1714 } 1715 1716 bool ARMDAGToDAGISel::tryMVEIndexedLoad(SDNode *N) { 1717 EVT LoadedVT; 1718 unsigned Opcode = 0; 1719 bool isSExtLd, isPre; 1720 unsigned Align; 1721 ARMVCC::VPTCodes Pred; 1722 SDValue PredReg; 1723 SDValue Chain, Base, Offset; 1724 1725 if (LoadSDNode *LD = dyn_cast<LoadSDNode>(N)) { 1726 ISD::MemIndexedMode AM = LD->getAddressingMode(); 1727 if (AM == ISD::UNINDEXED) 1728 return false; 1729 LoadedVT = LD->getMemoryVT(); 1730 if (!LoadedVT.isVector()) 1731 return false; 1732 1733 Chain = LD->getChain(); 1734 Base = LD->getBasePtr(); 1735 Offset = LD->getOffset(); 1736 Align = LD->getAlignment(); 1737 isSExtLd = LD->getExtensionType() == ISD::SEXTLOAD; 1738 isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC); 1739 Pred = ARMVCC::None; 1740 PredReg = CurDAG->getRegister(0, MVT::i32); 1741 } else if (MaskedLoadSDNode *LD = dyn_cast<MaskedLoadSDNode>(N)) { 1742 ISD::MemIndexedMode AM = LD->getAddressingMode(); 1743 if (AM == ISD::UNINDEXED) 1744 return false; 1745 LoadedVT = LD->getMemoryVT(); 1746 if (!LoadedVT.isVector()) 1747 return false; 1748 1749 Chain = LD->getChain(); 1750 Base = LD->getBasePtr(); 1751 Offset = LD->getOffset(); 1752 Align = LD->getAlignment(); 1753 isSExtLd = LD->getExtensionType() == ISD::SEXTLOAD; 1754 isPre = (AM == ISD::PRE_INC) || (AM == ISD::PRE_DEC); 1755 Pred = ARMVCC::Then; 1756 PredReg = LD->getMask(); 1757 } else 1758 llvm_unreachable("Expected a Load or a Masked Load!"); 1759 1760 // We allow LE non-masked loads to change the type (for example use a vldrb.8 1761 // as opposed to a vldrw.32). This can allow extra addressing modes or 1762 // alignments for what is otherwise an equivalent instruction. 1763 bool CanChangeType = Subtarget->isLittle() && !isa<MaskedLoadSDNode>(N); 1764 1765 SDValue NewOffset; 1766 if (Align >= 2 && LoadedVT == MVT::v4i16 && 1767 SelectT2AddrModeImm7Offset(N, Offset, NewOffset, 1)) { 1768 if (isSExtLd) 1769 Opcode = isPre ? ARM::MVE_VLDRHS32_pre : ARM::MVE_VLDRHS32_post; 1770 else 1771 Opcode = isPre ? ARM::MVE_VLDRHU32_pre : ARM::MVE_VLDRHU32_post; 1772 } else if (LoadedVT == MVT::v8i8 && 1773 SelectT2AddrModeImm7Offset(N, Offset, NewOffset, 0)) { 1774 if (isSExtLd) 1775 Opcode = isPre ? ARM::MVE_VLDRBS16_pre : ARM::MVE_VLDRBS16_post; 1776 else 1777 Opcode = isPre ? ARM::MVE_VLDRBU16_pre : ARM::MVE_VLDRBU16_post; 1778 } else if (LoadedVT == MVT::v4i8 && 1779 SelectT2AddrModeImm7Offset(N, Offset, NewOffset, 0)) { 1780 if (isSExtLd) 1781 Opcode = isPre ? ARM::MVE_VLDRBS32_pre : ARM::MVE_VLDRBS32_post; 1782 else 1783 Opcode = isPre ? ARM::MVE_VLDRBU32_pre : ARM::MVE_VLDRBU32_post; 1784 } else if (Align >= 4 && 1785 (CanChangeType || LoadedVT == MVT::v4i32 || 1786 LoadedVT == MVT::v4f32) && 1787 SelectT2AddrModeImm7Offset(N, Offset, NewOffset, 2)) 1788 Opcode = isPre ? ARM::MVE_VLDRWU32_pre : ARM::MVE_VLDRWU32_post; 1789 else if (Align >= 2 && 1790 (CanChangeType || LoadedVT == MVT::v8i16 || 1791 LoadedVT == MVT::v8f16) && 1792 SelectT2AddrModeImm7Offset(N, Offset, NewOffset, 1)) 1793 Opcode = isPre ? ARM::MVE_VLDRHU16_pre : ARM::MVE_VLDRHU16_post; 1794 else if ((CanChangeType || LoadedVT == MVT::v16i8) && 1795 SelectT2AddrModeImm7Offset(N, Offset, NewOffset, 0)) 1796 Opcode = isPre ? ARM::MVE_VLDRBU8_pre : ARM::MVE_VLDRBU8_post; 1797 else 1798 return false; 1799 1800 SDValue Ops[] = {Base, NewOffset, 1801 CurDAG->getTargetConstant(Pred, SDLoc(N), MVT::i32), PredReg, 1802 Chain}; 1803 SDNode *New = CurDAG->getMachineNode(Opcode, SDLoc(N), MVT::i32, 1804 N->getValueType(0), MVT::Other, Ops); 1805 transferMemOperands(N, New); 1806 ReplaceUses(SDValue(N, 0), SDValue(New, 1)); 1807 ReplaceUses(SDValue(N, 1), SDValue(New, 0)); 1808 ReplaceUses(SDValue(N, 2), SDValue(New, 2)); 1809 CurDAG->RemoveDeadNode(N); 1810 return true; 1811 } 1812 1813 /// Form a GPRPair pseudo register from a pair of GPR regs. 1814 SDNode *ARMDAGToDAGISel::createGPRPairNode(EVT VT, SDValue V0, SDValue V1) { 1815 SDLoc dl(V0.getNode()); 1816 SDValue RegClass = 1817 CurDAG->getTargetConstant(ARM::GPRPairRegClassID, dl, MVT::i32); 1818 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::gsub_0, dl, MVT::i32); 1819 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::gsub_1, dl, MVT::i32); 1820 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1821 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1822 } 1823 1824 /// Form a D register from a pair of S registers. 1825 SDNode *ARMDAGToDAGISel::createSRegPairNode(EVT VT, SDValue V0, SDValue V1) { 1826 SDLoc dl(V0.getNode()); 1827 SDValue RegClass = 1828 CurDAG->getTargetConstant(ARM::DPR_VFP2RegClassID, dl, MVT::i32); 1829 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::ssub_0, dl, MVT::i32); 1830 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::ssub_1, dl, MVT::i32); 1831 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1832 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1833 } 1834 1835 /// Form a quad register from a pair of D registers. 1836 SDNode *ARMDAGToDAGISel::createDRegPairNode(EVT VT, SDValue V0, SDValue V1) { 1837 SDLoc dl(V0.getNode()); 1838 SDValue RegClass = CurDAG->getTargetConstant(ARM::QPRRegClassID, dl, 1839 MVT::i32); 1840 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::dsub_0, dl, MVT::i32); 1841 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::dsub_1, dl, MVT::i32); 1842 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1843 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1844 } 1845 1846 /// Form 4 consecutive D registers from a pair of Q registers. 1847 SDNode *ARMDAGToDAGISel::createQRegPairNode(EVT VT, SDValue V0, SDValue V1) { 1848 SDLoc dl(V0.getNode()); 1849 SDValue RegClass = CurDAG->getTargetConstant(ARM::QQPRRegClassID, dl, 1850 MVT::i32); 1851 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::qsub_0, dl, MVT::i32); 1852 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::qsub_1, dl, MVT::i32); 1853 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1 }; 1854 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1855 } 1856 1857 /// Form 4 consecutive S registers. 1858 SDNode *ARMDAGToDAGISel::createQuadSRegsNode(EVT VT, SDValue V0, SDValue V1, 1859 SDValue V2, SDValue V3) { 1860 SDLoc dl(V0.getNode()); 1861 SDValue RegClass = 1862 CurDAG->getTargetConstant(ARM::QPR_VFP2RegClassID, dl, MVT::i32); 1863 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::ssub_0, dl, MVT::i32); 1864 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::ssub_1, dl, MVT::i32); 1865 SDValue SubReg2 = CurDAG->getTargetConstant(ARM::ssub_2, dl, MVT::i32); 1866 SDValue SubReg3 = CurDAG->getTargetConstant(ARM::ssub_3, dl, MVT::i32); 1867 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1, 1868 V2, SubReg2, V3, SubReg3 }; 1869 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1870 } 1871 1872 /// Form 4 consecutive D registers. 1873 SDNode *ARMDAGToDAGISel::createQuadDRegsNode(EVT VT, SDValue V0, SDValue V1, 1874 SDValue V2, SDValue V3) { 1875 SDLoc dl(V0.getNode()); 1876 SDValue RegClass = CurDAG->getTargetConstant(ARM::QQPRRegClassID, dl, 1877 MVT::i32); 1878 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::dsub_0, dl, MVT::i32); 1879 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::dsub_1, dl, MVT::i32); 1880 SDValue SubReg2 = CurDAG->getTargetConstant(ARM::dsub_2, dl, MVT::i32); 1881 SDValue SubReg3 = CurDAG->getTargetConstant(ARM::dsub_3, dl, MVT::i32); 1882 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1, 1883 V2, SubReg2, V3, SubReg3 }; 1884 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1885 } 1886 1887 /// Form 4 consecutive Q registers. 1888 SDNode *ARMDAGToDAGISel::createQuadQRegsNode(EVT VT, SDValue V0, SDValue V1, 1889 SDValue V2, SDValue V3) { 1890 SDLoc dl(V0.getNode()); 1891 SDValue RegClass = CurDAG->getTargetConstant(ARM::QQQQPRRegClassID, dl, 1892 MVT::i32); 1893 SDValue SubReg0 = CurDAG->getTargetConstant(ARM::qsub_0, dl, MVT::i32); 1894 SDValue SubReg1 = CurDAG->getTargetConstant(ARM::qsub_1, dl, MVT::i32); 1895 SDValue SubReg2 = CurDAG->getTargetConstant(ARM::qsub_2, dl, MVT::i32); 1896 SDValue SubReg3 = CurDAG->getTargetConstant(ARM::qsub_3, dl, MVT::i32); 1897 const SDValue Ops[] = { RegClass, V0, SubReg0, V1, SubReg1, 1898 V2, SubReg2, V3, SubReg3 }; 1899 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, dl, VT, Ops); 1900 } 1901 1902 /// GetVLDSTAlign - Get the alignment (in bytes) for the alignment operand 1903 /// of a NEON VLD or VST instruction. The supported values depend on the 1904 /// number of registers being loaded. 1905 SDValue ARMDAGToDAGISel::GetVLDSTAlign(SDValue Align, const SDLoc &dl, 1906 unsigned NumVecs, bool is64BitVector) { 1907 unsigned NumRegs = NumVecs; 1908 if (!is64BitVector && NumVecs < 3) 1909 NumRegs *= 2; 1910 1911 unsigned Alignment = cast<ConstantSDNode>(Align)->getZExtValue(); 1912 if (Alignment >= 32 && NumRegs == 4) 1913 Alignment = 32; 1914 else if (Alignment >= 16 && (NumRegs == 2 || NumRegs == 4)) 1915 Alignment = 16; 1916 else if (Alignment >= 8) 1917 Alignment = 8; 1918 else 1919 Alignment = 0; 1920 1921 return CurDAG->getTargetConstant(Alignment, dl, MVT::i32); 1922 } 1923 1924 static bool isVLDfixed(unsigned Opc) 1925 { 1926 switch (Opc) { 1927 default: return false; 1928 case ARM::VLD1d8wb_fixed : return true; 1929 case ARM::VLD1d16wb_fixed : return true; 1930 case ARM::VLD1d64Qwb_fixed : return true; 1931 case ARM::VLD1d32wb_fixed : return true; 1932 case ARM::VLD1d64wb_fixed : return true; 1933 case ARM::VLD1d64TPseudoWB_fixed : return true; 1934 case ARM::VLD1d64QPseudoWB_fixed : return true; 1935 case ARM::VLD1q8wb_fixed : return true; 1936 case ARM::VLD1q16wb_fixed : return true; 1937 case ARM::VLD1q32wb_fixed : return true; 1938 case ARM::VLD1q64wb_fixed : return true; 1939 case ARM::VLD1DUPd8wb_fixed : return true; 1940 case ARM::VLD1DUPd16wb_fixed : return true; 1941 case ARM::VLD1DUPd32wb_fixed : return true; 1942 case ARM::VLD1DUPq8wb_fixed : return true; 1943 case ARM::VLD1DUPq16wb_fixed : return true; 1944 case ARM::VLD1DUPq32wb_fixed : return true; 1945 case ARM::VLD2d8wb_fixed : return true; 1946 case ARM::VLD2d16wb_fixed : return true; 1947 case ARM::VLD2d32wb_fixed : return true; 1948 case ARM::VLD2q8PseudoWB_fixed : return true; 1949 case ARM::VLD2q16PseudoWB_fixed : return true; 1950 case ARM::VLD2q32PseudoWB_fixed : return true; 1951 case ARM::VLD2DUPd8wb_fixed : return true; 1952 case ARM::VLD2DUPd16wb_fixed : return true; 1953 case ARM::VLD2DUPd32wb_fixed : return true; 1954 } 1955 } 1956 1957 static bool isVSTfixed(unsigned Opc) 1958 { 1959 switch (Opc) { 1960 default: return false; 1961 case ARM::VST1d8wb_fixed : return true; 1962 case ARM::VST1d16wb_fixed : return true; 1963 case ARM::VST1d32wb_fixed : return true; 1964 case ARM::VST1d64wb_fixed : return true; 1965 case ARM::VST1q8wb_fixed : return true; 1966 case ARM::VST1q16wb_fixed : return true; 1967 case ARM::VST1q32wb_fixed : return true; 1968 case ARM::VST1q64wb_fixed : return true; 1969 case ARM::VST1d64TPseudoWB_fixed : return true; 1970 case ARM::VST1d64QPseudoWB_fixed : return true; 1971 case ARM::VST2d8wb_fixed : return true; 1972 case ARM::VST2d16wb_fixed : return true; 1973 case ARM::VST2d32wb_fixed : return true; 1974 case ARM::VST2q8PseudoWB_fixed : return true; 1975 case ARM::VST2q16PseudoWB_fixed : return true; 1976 case ARM::VST2q32PseudoWB_fixed : return true; 1977 } 1978 } 1979 1980 // Get the register stride update opcode of a VLD/VST instruction that 1981 // is otherwise equivalent to the given fixed stride updating instruction. 1982 static unsigned getVLDSTRegisterUpdateOpcode(unsigned Opc) { 1983 assert((isVLDfixed(Opc) || isVSTfixed(Opc)) 1984 && "Incorrect fixed stride updating instruction."); 1985 switch (Opc) { 1986 default: break; 1987 case ARM::VLD1d8wb_fixed: return ARM::VLD1d8wb_register; 1988 case ARM::VLD1d16wb_fixed: return ARM::VLD1d16wb_register; 1989 case ARM::VLD1d32wb_fixed: return ARM::VLD1d32wb_register; 1990 case ARM::VLD1d64wb_fixed: return ARM::VLD1d64wb_register; 1991 case ARM::VLD1q8wb_fixed: return ARM::VLD1q8wb_register; 1992 case ARM::VLD1q16wb_fixed: return ARM::VLD1q16wb_register; 1993 case ARM::VLD1q32wb_fixed: return ARM::VLD1q32wb_register; 1994 case ARM::VLD1q64wb_fixed: return ARM::VLD1q64wb_register; 1995 case ARM::VLD1d64Twb_fixed: return ARM::VLD1d64Twb_register; 1996 case ARM::VLD1d64Qwb_fixed: return ARM::VLD1d64Qwb_register; 1997 case ARM::VLD1d64TPseudoWB_fixed: return ARM::VLD1d64TPseudoWB_register; 1998 case ARM::VLD1d64QPseudoWB_fixed: return ARM::VLD1d64QPseudoWB_register; 1999 case ARM::VLD1DUPd8wb_fixed : return ARM::VLD1DUPd8wb_register; 2000 case ARM::VLD1DUPd16wb_fixed : return ARM::VLD1DUPd16wb_register; 2001 case ARM::VLD1DUPd32wb_fixed : return ARM::VLD1DUPd32wb_register; 2002 case ARM::VLD1DUPq8wb_fixed : return ARM::VLD1DUPq8wb_register; 2003 case ARM::VLD1DUPq16wb_fixed : return ARM::VLD1DUPq16wb_register; 2004 case ARM::VLD1DUPq32wb_fixed : return ARM::VLD1DUPq32wb_register; 2005 2006 case ARM::VST1d8wb_fixed: return ARM::VST1d8wb_register; 2007 case ARM::VST1d16wb_fixed: return ARM::VST1d16wb_register; 2008 case ARM::VST1d32wb_fixed: return ARM::VST1d32wb_register; 2009 case ARM::VST1d64wb_fixed: return ARM::VST1d64wb_register; 2010 case ARM::VST1q8wb_fixed: return ARM::VST1q8wb_register; 2011 case ARM::VST1q16wb_fixed: return ARM::VST1q16wb_register; 2012 case ARM::VST1q32wb_fixed: return ARM::VST1q32wb_register; 2013 case ARM::VST1q64wb_fixed: return ARM::VST1q64wb_register; 2014 case ARM::VST1d64TPseudoWB_fixed: return ARM::VST1d64TPseudoWB_register; 2015 case ARM::VST1d64QPseudoWB_fixed: return ARM::VST1d64QPseudoWB_register; 2016 2017 case ARM::VLD2d8wb_fixed: return ARM::VLD2d8wb_register; 2018 case ARM::VLD2d16wb_fixed: return ARM::VLD2d16wb_register; 2019 case ARM::VLD2d32wb_fixed: return ARM::VLD2d32wb_register; 2020 case ARM::VLD2q8PseudoWB_fixed: return ARM::VLD2q8PseudoWB_register; 2021 case ARM::VLD2q16PseudoWB_fixed: return ARM::VLD2q16PseudoWB_register; 2022 case ARM::VLD2q32PseudoWB_fixed: return ARM::VLD2q32PseudoWB_register; 2023 2024 case ARM::VST2d8wb_fixed: return ARM::VST2d8wb_register; 2025 case ARM::VST2d16wb_fixed: return ARM::VST2d16wb_register; 2026 case ARM::VST2d32wb_fixed: return ARM::VST2d32wb_register; 2027 case ARM::VST2q8PseudoWB_fixed: return ARM::VST2q8PseudoWB_register; 2028 case ARM::VST2q16PseudoWB_fixed: return ARM::VST2q16PseudoWB_register; 2029 case ARM::VST2q32PseudoWB_fixed: return ARM::VST2q32PseudoWB_register; 2030 2031 case ARM::VLD2DUPd8wb_fixed: return ARM::VLD2DUPd8wb_register; 2032 case ARM::VLD2DUPd16wb_fixed: return ARM::VLD2DUPd16wb_register; 2033 case ARM::VLD2DUPd32wb_fixed: return ARM::VLD2DUPd32wb_register; 2034 } 2035 return Opc; // If not one we handle, return it unchanged. 2036 } 2037 2038 /// Returns true if the given increment is a Constant known to be equal to the 2039 /// access size performed by a NEON load/store. This means the "[rN]!" form can 2040 /// be used. 2041 static bool isPerfectIncrement(SDValue Inc, EVT VecTy, unsigned NumVecs) { 2042 auto C = dyn_cast<ConstantSDNode>(Inc); 2043 return C && C->getZExtValue() == VecTy.getSizeInBits() / 8 * NumVecs; 2044 } 2045 2046 void ARMDAGToDAGISel::SelectVLD(SDNode *N, bool isUpdating, unsigned NumVecs, 2047 const uint16_t *DOpcodes, 2048 const uint16_t *QOpcodes0, 2049 const uint16_t *QOpcodes1) { 2050 assert(Subtarget->hasNEON()); 2051 assert(NumVecs >= 1 && NumVecs <= 4 && "VLD NumVecs out-of-range"); 2052 SDLoc dl(N); 2053 2054 SDValue MemAddr, Align; 2055 bool IsIntrinsic = !isUpdating; // By coincidence, all supported updating 2056 // nodes are not intrinsics. 2057 unsigned AddrOpIdx = IsIntrinsic ? 2 : 1; 2058 if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align)) 2059 return; 2060 2061 SDValue Chain = N->getOperand(0); 2062 EVT VT = N->getValueType(0); 2063 bool is64BitVector = VT.is64BitVector(); 2064 Align = GetVLDSTAlign(Align, dl, NumVecs, is64BitVector); 2065 2066 unsigned OpcodeIndex; 2067 switch (VT.getSimpleVT().SimpleTy) { 2068 default: llvm_unreachable("unhandled vld type"); 2069 // Double-register operations: 2070 case MVT::v8i8: OpcodeIndex = 0; break; 2071 case MVT::v4f16: 2072 case MVT::v4i16: OpcodeIndex = 1; break; 2073 case MVT::v2f32: 2074 case MVT::v2i32: OpcodeIndex = 2; break; 2075 case MVT::v1i64: OpcodeIndex = 3; break; 2076 // Quad-register operations: 2077 case MVT::v16i8: OpcodeIndex = 0; break; 2078 case MVT::v8f16: 2079 case MVT::v8i16: OpcodeIndex = 1; break; 2080 case MVT::v4f32: 2081 case MVT::v4i32: OpcodeIndex = 2; break; 2082 case MVT::v2f64: 2083 case MVT::v2i64: OpcodeIndex = 3; break; 2084 } 2085 2086 EVT ResTy; 2087 if (NumVecs == 1) 2088 ResTy = VT; 2089 else { 2090 unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs; 2091 if (!is64BitVector) 2092 ResTyElts *= 2; 2093 ResTy = EVT::getVectorVT(*CurDAG->getContext(), MVT::i64, ResTyElts); 2094 } 2095 std::vector<EVT> ResTys; 2096 ResTys.push_back(ResTy); 2097 if (isUpdating) 2098 ResTys.push_back(MVT::i32); 2099 ResTys.push_back(MVT::Other); 2100 2101 SDValue Pred = getAL(CurDAG, dl); 2102 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2103 SDNode *VLd; 2104 SmallVector<SDValue, 7> Ops; 2105 2106 // Double registers and VLD1/VLD2 quad registers are directly supported. 2107 if (is64BitVector || NumVecs <= 2) { 2108 unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] : 2109 QOpcodes0[OpcodeIndex]); 2110 Ops.push_back(MemAddr); 2111 Ops.push_back(Align); 2112 if (isUpdating) { 2113 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2114 bool IsImmUpdate = isPerfectIncrement(Inc, VT, NumVecs); 2115 if (!IsImmUpdate) { 2116 // We use a VLD1 for v1i64 even if the pseudo says vld2/3/4, so 2117 // check for the opcode rather than the number of vector elements. 2118 if (isVLDfixed(Opc)) 2119 Opc = getVLDSTRegisterUpdateOpcode(Opc); 2120 Ops.push_back(Inc); 2121 // VLD1/VLD2 fixed increment does not need Reg0 so only include it in 2122 // the operands if not such an opcode. 2123 } else if (!isVLDfixed(Opc)) 2124 Ops.push_back(Reg0); 2125 } 2126 Ops.push_back(Pred); 2127 Ops.push_back(Reg0); 2128 Ops.push_back(Chain); 2129 VLd = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 2130 2131 } else { 2132 // Otherwise, quad registers are loaded with two separate instructions, 2133 // where one loads the even registers and the other loads the odd registers. 2134 EVT AddrTy = MemAddr.getValueType(); 2135 2136 // Load the even subregs. This is always an updating load, so that it 2137 // provides the address to the second load for the odd subregs. 2138 SDValue ImplDef = 2139 SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, ResTy), 0); 2140 const SDValue OpsA[] = { MemAddr, Align, Reg0, ImplDef, Pred, Reg0, Chain }; 2141 SDNode *VLdA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], dl, 2142 ResTy, AddrTy, MVT::Other, OpsA); 2143 Chain = SDValue(VLdA, 2); 2144 2145 // Load the odd subregs. 2146 Ops.push_back(SDValue(VLdA, 1)); 2147 Ops.push_back(Align); 2148 if (isUpdating) { 2149 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2150 assert(isa<ConstantSDNode>(Inc.getNode()) && 2151 "only constant post-increment update allowed for VLD3/4"); 2152 (void)Inc; 2153 Ops.push_back(Reg0); 2154 } 2155 Ops.push_back(SDValue(VLdA, 0)); 2156 Ops.push_back(Pred); 2157 Ops.push_back(Reg0); 2158 Ops.push_back(Chain); 2159 VLd = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys, Ops); 2160 } 2161 2162 // Transfer memoperands. 2163 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 2164 CurDAG->setNodeMemRefs(cast<MachineSDNode>(VLd), {MemOp}); 2165 2166 if (NumVecs == 1) { 2167 ReplaceNode(N, VLd); 2168 return; 2169 } 2170 2171 // Extract out the subregisters. 2172 SDValue SuperReg = SDValue(VLd, 0); 2173 static_assert(ARM::dsub_7 == ARM::dsub_0 + 7 && 2174 ARM::qsub_3 == ARM::qsub_0 + 3, 2175 "Unexpected subreg numbering"); 2176 unsigned Sub0 = (is64BitVector ? ARM::dsub_0 : ARM::qsub_0); 2177 for (unsigned Vec = 0; Vec < NumVecs; ++Vec) 2178 ReplaceUses(SDValue(N, Vec), 2179 CurDAG->getTargetExtractSubreg(Sub0 + Vec, dl, VT, SuperReg)); 2180 ReplaceUses(SDValue(N, NumVecs), SDValue(VLd, 1)); 2181 if (isUpdating) 2182 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLd, 2)); 2183 CurDAG->RemoveDeadNode(N); 2184 } 2185 2186 void ARMDAGToDAGISel::SelectVST(SDNode *N, bool isUpdating, unsigned NumVecs, 2187 const uint16_t *DOpcodes, 2188 const uint16_t *QOpcodes0, 2189 const uint16_t *QOpcodes1) { 2190 assert(Subtarget->hasNEON()); 2191 assert(NumVecs >= 1 && NumVecs <= 4 && "VST NumVecs out-of-range"); 2192 SDLoc dl(N); 2193 2194 SDValue MemAddr, Align; 2195 bool IsIntrinsic = !isUpdating; // By coincidence, all supported updating 2196 // nodes are not intrinsics. 2197 unsigned AddrOpIdx = IsIntrinsic ? 2 : 1; 2198 unsigned Vec0Idx = 3; // AddrOpIdx + (isUpdating ? 2 : 1) 2199 if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align)) 2200 return; 2201 2202 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 2203 2204 SDValue Chain = N->getOperand(0); 2205 EVT VT = N->getOperand(Vec0Idx).getValueType(); 2206 bool is64BitVector = VT.is64BitVector(); 2207 Align = GetVLDSTAlign(Align, dl, NumVecs, is64BitVector); 2208 2209 unsigned OpcodeIndex; 2210 switch (VT.getSimpleVT().SimpleTy) { 2211 default: llvm_unreachable("unhandled vst type"); 2212 // Double-register operations: 2213 case MVT::v8i8: OpcodeIndex = 0; break; 2214 case MVT::v4f16: 2215 case MVT::v4i16: OpcodeIndex = 1; break; 2216 case MVT::v2f32: 2217 case MVT::v2i32: OpcodeIndex = 2; break; 2218 case MVT::v1i64: OpcodeIndex = 3; break; 2219 // Quad-register operations: 2220 case MVT::v16i8: OpcodeIndex = 0; break; 2221 case MVT::v8f16: 2222 case MVT::v8i16: OpcodeIndex = 1; break; 2223 case MVT::v4f32: 2224 case MVT::v4i32: OpcodeIndex = 2; break; 2225 case MVT::v2f64: 2226 case MVT::v2i64: OpcodeIndex = 3; break; 2227 } 2228 2229 std::vector<EVT> ResTys; 2230 if (isUpdating) 2231 ResTys.push_back(MVT::i32); 2232 ResTys.push_back(MVT::Other); 2233 2234 SDValue Pred = getAL(CurDAG, dl); 2235 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2236 SmallVector<SDValue, 7> Ops; 2237 2238 // Double registers and VST1/VST2 quad registers are directly supported. 2239 if (is64BitVector || NumVecs <= 2) { 2240 SDValue SrcReg; 2241 if (NumVecs == 1) { 2242 SrcReg = N->getOperand(Vec0Idx); 2243 } else if (is64BitVector) { 2244 // Form a REG_SEQUENCE to force register allocation. 2245 SDValue V0 = N->getOperand(Vec0Idx + 0); 2246 SDValue V1 = N->getOperand(Vec0Idx + 1); 2247 if (NumVecs == 2) 2248 SrcReg = SDValue(createDRegPairNode(MVT::v2i64, V0, V1), 0); 2249 else { 2250 SDValue V2 = N->getOperand(Vec0Idx + 2); 2251 // If it's a vst3, form a quad D-register and leave the last part as 2252 // an undef. 2253 SDValue V3 = (NumVecs == 3) 2254 ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,dl,VT), 0) 2255 : N->getOperand(Vec0Idx + 3); 2256 SrcReg = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0); 2257 } 2258 } else { 2259 // Form a QQ register. 2260 SDValue Q0 = N->getOperand(Vec0Idx); 2261 SDValue Q1 = N->getOperand(Vec0Idx + 1); 2262 SrcReg = SDValue(createQRegPairNode(MVT::v4i64, Q0, Q1), 0); 2263 } 2264 2265 unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] : 2266 QOpcodes0[OpcodeIndex]); 2267 Ops.push_back(MemAddr); 2268 Ops.push_back(Align); 2269 if (isUpdating) { 2270 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2271 bool IsImmUpdate = isPerfectIncrement(Inc, VT, NumVecs); 2272 if (!IsImmUpdate) { 2273 // We use a VST1 for v1i64 even if the pseudo says VST2/3/4, so 2274 // check for the opcode rather than the number of vector elements. 2275 if (isVSTfixed(Opc)) 2276 Opc = getVLDSTRegisterUpdateOpcode(Opc); 2277 Ops.push_back(Inc); 2278 } 2279 // VST1/VST2 fixed increment does not need Reg0 so only include it in 2280 // the operands if not such an opcode. 2281 else if (!isVSTfixed(Opc)) 2282 Ops.push_back(Reg0); 2283 } 2284 Ops.push_back(SrcReg); 2285 Ops.push_back(Pred); 2286 Ops.push_back(Reg0); 2287 Ops.push_back(Chain); 2288 SDNode *VSt = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 2289 2290 // Transfer memoperands. 2291 CurDAG->setNodeMemRefs(cast<MachineSDNode>(VSt), {MemOp}); 2292 2293 ReplaceNode(N, VSt); 2294 return; 2295 } 2296 2297 // Otherwise, quad registers are stored with two separate instructions, 2298 // where one stores the even registers and the other stores the odd registers. 2299 2300 // Form the QQQQ REG_SEQUENCE. 2301 SDValue V0 = N->getOperand(Vec0Idx + 0); 2302 SDValue V1 = N->getOperand(Vec0Idx + 1); 2303 SDValue V2 = N->getOperand(Vec0Idx + 2); 2304 SDValue V3 = (NumVecs == 3) 2305 ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0) 2306 : N->getOperand(Vec0Idx + 3); 2307 SDValue RegSeq = SDValue(createQuadQRegsNode(MVT::v8i64, V0, V1, V2, V3), 0); 2308 2309 // Store the even D registers. This is always an updating store, so that it 2310 // provides the address to the second store for the odd subregs. 2311 const SDValue OpsA[] = { MemAddr, Align, Reg0, RegSeq, Pred, Reg0, Chain }; 2312 SDNode *VStA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], dl, 2313 MemAddr.getValueType(), 2314 MVT::Other, OpsA); 2315 CurDAG->setNodeMemRefs(cast<MachineSDNode>(VStA), {MemOp}); 2316 Chain = SDValue(VStA, 1); 2317 2318 // Store the odd D registers. 2319 Ops.push_back(SDValue(VStA, 0)); 2320 Ops.push_back(Align); 2321 if (isUpdating) { 2322 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2323 assert(isa<ConstantSDNode>(Inc.getNode()) && 2324 "only constant post-increment update allowed for VST3/4"); 2325 (void)Inc; 2326 Ops.push_back(Reg0); 2327 } 2328 Ops.push_back(RegSeq); 2329 Ops.push_back(Pred); 2330 Ops.push_back(Reg0); 2331 Ops.push_back(Chain); 2332 SDNode *VStB = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys, 2333 Ops); 2334 CurDAG->setNodeMemRefs(cast<MachineSDNode>(VStB), {MemOp}); 2335 ReplaceNode(N, VStB); 2336 } 2337 2338 void ARMDAGToDAGISel::SelectVLDSTLane(SDNode *N, bool IsLoad, bool isUpdating, 2339 unsigned NumVecs, 2340 const uint16_t *DOpcodes, 2341 const uint16_t *QOpcodes) { 2342 assert(Subtarget->hasNEON()); 2343 assert(NumVecs >=2 && NumVecs <= 4 && "VLDSTLane NumVecs out-of-range"); 2344 SDLoc dl(N); 2345 2346 SDValue MemAddr, Align; 2347 bool IsIntrinsic = !isUpdating; // By coincidence, all supported updating 2348 // nodes are not intrinsics. 2349 unsigned AddrOpIdx = IsIntrinsic ? 2 : 1; 2350 unsigned Vec0Idx = 3; // AddrOpIdx + (isUpdating ? 2 : 1) 2351 if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align)) 2352 return; 2353 2354 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 2355 2356 SDValue Chain = N->getOperand(0); 2357 unsigned Lane = 2358 cast<ConstantSDNode>(N->getOperand(Vec0Idx + NumVecs))->getZExtValue(); 2359 EVT VT = N->getOperand(Vec0Idx).getValueType(); 2360 bool is64BitVector = VT.is64BitVector(); 2361 2362 unsigned Alignment = 0; 2363 if (NumVecs != 3) { 2364 Alignment = cast<ConstantSDNode>(Align)->getZExtValue(); 2365 unsigned NumBytes = NumVecs * VT.getScalarSizeInBits() / 8; 2366 if (Alignment > NumBytes) 2367 Alignment = NumBytes; 2368 if (Alignment < 8 && Alignment < NumBytes) 2369 Alignment = 0; 2370 // Alignment must be a power of two; make sure of that. 2371 Alignment = (Alignment & -Alignment); 2372 if (Alignment == 1) 2373 Alignment = 0; 2374 } 2375 Align = CurDAG->getTargetConstant(Alignment, dl, MVT::i32); 2376 2377 unsigned OpcodeIndex; 2378 switch (VT.getSimpleVT().SimpleTy) { 2379 default: llvm_unreachable("unhandled vld/vst lane type"); 2380 // Double-register operations: 2381 case MVT::v8i8: OpcodeIndex = 0; break; 2382 case MVT::v4f16: 2383 case MVT::v4i16: OpcodeIndex = 1; break; 2384 case MVT::v2f32: 2385 case MVT::v2i32: OpcodeIndex = 2; break; 2386 // Quad-register operations: 2387 case MVT::v8f16: 2388 case MVT::v8i16: OpcodeIndex = 0; break; 2389 case MVT::v4f32: 2390 case MVT::v4i32: OpcodeIndex = 1; break; 2391 } 2392 2393 std::vector<EVT> ResTys; 2394 if (IsLoad) { 2395 unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs; 2396 if (!is64BitVector) 2397 ResTyElts *= 2; 2398 ResTys.push_back(EVT::getVectorVT(*CurDAG->getContext(), 2399 MVT::i64, ResTyElts)); 2400 } 2401 if (isUpdating) 2402 ResTys.push_back(MVT::i32); 2403 ResTys.push_back(MVT::Other); 2404 2405 SDValue Pred = getAL(CurDAG, dl); 2406 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2407 2408 SmallVector<SDValue, 8> Ops; 2409 Ops.push_back(MemAddr); 2410 Ops.push_back(Align); 2411 if (isUpdating) { 2412 SDValue Inc = N->getOperand(AddrOpIdx + 1); 2413 bool IsImmUpdate = 2414 isPerfectIncrement(Inc, VT.getVectorElementType(), NumVecs); 2415 Ops.push_back(IsImmUpdate ? Reg0 : Inc); 2416 } 2417 2418 SDValue SuperReg; 2419 SDValue V0 = N->getOperand(Vec0Idx + 0); 2420 SDValue V1 = N->getOperand(Vec0Idx + 1); 2421 if (NumVecs == 2) { 2422 if (is64BitVector) 2423 SuperReg = SDValue(createDRegPairNode(MVT::v2i64, V0, V1), 0); 2424 else 2425 SuperReg = SDValue(createQRegPairNode(MVT::v4i64, V0, V1), 0); 2426 } else { 2427 SDValue V2 = N->getOperand(Vec0Idx + 2); 2428 SDValue V3 = (NumVecs == 3) 2429 ? SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, VT), 0) 2430 : N->getOperand(Vec0Idx + 3); 2431 if (is64BitVector) 2432 SuperReg = SDValue(createQuadDRegsNode(MVT::v4i64, V0, V1, V2, V3), 0); 2433 else 2434 SuperReg = SDValue(createQuadQRegsNode(MVT::v8i64, V0, V1, V2, V3), 0); 2435 } 2436 Ops.push_back(SuperReg); 2437 Ops.push_back(getI32Imm(Lane, dl)); 2438 Ops.push_back(Pred); 2439 Ops.push_back(Reg0); 2440 Ops.push_back(Chain); 2441 2442 unsigned Opc = (is64BitVector ? DOpcodes[OpcodeIndex] : 2443 QOpcodes[OpcodeIndex]); 2444 SDNode *VLdLn = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 2445 CurDAG->setNodeMemRefs(cast<MachineSDNode>(VLdLn), {MemOp}); 2446 if (!IsLoad) { 2447 ReplaceNode(N, VLdLn); 2448 return; 2449 } 2450 2451 // Extract the subregisters. 2452 SuperReg = SDValue(VLdLn, 0); 2453 static_assert(ARM::dsub_7 == ARM::dsub_0 + 7 && 2454 ARM::qsub_3 == ARM::qsub_0 + 3, 2455 "Unexpected subreg numbering"); 2456 unsigned Sub0 = is64BitVector ? ARM::dsub_0 : ARM::qsub_0; 2457 for (unsigned Vec = 0; Vec < NumVecs; ++Vec) 2458 ReplaceUses(SDValue(N, Vec), 2459 CurDAG->getTargetExtractSubreg(Sub0 + Vec, dl, VT, SuperReg)); 2460 ReplaceUses(SDValue(N, NumVecs), SDValue(VLdLn, 1)); 2461 if (isUpdating) 2462 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLdLn, 2)); 2463 CurDAG->RemoveDeadNode(N); 2464 } 2465 2466 template <typename SDValueVector> 2467 void ARMDAGToDAGISel::AddMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc, 2468 SDValue PredicateMask) { 2469 Ops.push_back(CurDAG->getTargetConstant(ARMVCC::Then, Loc, MVT::i32)); 2470 Ops.push_back(PredicateMask); 2471 } 2472 2473 template <typename SDValueVector> 2474 void ARMDAGToDAGISel::AddMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc, 2475 SDValue PredicateMask, 2476 SDValue Inactive) { 2477 Ops.push_back(CurDAG->getTargetConstant(ARMVCC::Then, Loc, MVT::i32)); 2478 Ops.push_back(PredicateMask); 2479 Ops.push_back(Inactive); 2480 } 2481 2482 template <typename SDValueVector> 2483 void ARMDAGToDAGISel::AddEmptyMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc) { 2484 Ops.push_back(CurDAG->getTargetConstant(ARMVCC::None, Loc, MVT::i32)); 2485 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 2486 } 2487 2488 template <typename SDValueVector> 2489 void ARMDAGToDAGISel::AddEmptyMVEPredicateToOps(SDValueVector &Ops, SDLoc Loc, 2490 EVT InactiveTy) { 2491 Ops.push_back(CurDAG->getTargetConstant(ARMVCC::None, Loc, MVT::i32)); 2492 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 2493 Ops.push_back(SDValue( 2494 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, Loc, InactiveTy), 0)); 2495 } 2496 2497 void ARMDAGToDAGISel::SelectMVE_WB(SDNode *N, const uint16_t *Opcodes, 2498 bool Predicated) { 2499 SDLoc Loc(N); 2500 SmallVector<SDValue, 8> Ops; 2501 2502 uint16_t Opcode; 2503 switch (N->getValueType(1).getVectorElementType().getSizeInBits()) { 2504 case 32: 2505 Opcode = Opcodes[0]; 2506 break; 2507 case 64: 2508 Opcode = Opcodes[1]; 2509 break; 2510 default: 2511 llvm_unreachable("bad vector element size in SelectMVE_WB"); 2512 } 2513 2514 Ops.push_back(N->getOperand(2)); // vector of base addresses 2515 2516 int32_t ImmValue = cast<ConstantSDNode>(N->getOperand(3))->getZExtValue(); 2517 Ops.push_back(getI32Imm(ImmValue, Loc)); // immediate offset 2518 2519 if (Predicated) 2520 AddMVEPredicateToOps(Ops, Loc, N->getOperand(4)); 2521 else 2522 AddEmptyMVEPredicateToOps(Ops, Loc); 2523 2524 Ops.push_back(N->getOperand(0)); // chain 2525 2526 SmallVector<EVT, 8> VTs; 2527 VTs.push_back(N->getValueType(1)); 2528 VTs.push_back(N->getValueType(0)); 2529 VTs.push_back(N->getValueType(2)); 2530 2531 SDNode *New = CurDAG->getMachineNode(Opcode, SDLoc(N), VTs, Ops); 2532 ReplaceUses(SDValue(N, 0), SDValue(New, 1)); 2533 ReplaceUses(SDValue(N, 1), SDValue(New, 0)); 2534 ReplaceUses(SDValue(N, 2), SDValue(New, 2)); 2535 CurDAG->RemoveDeadNode(N); 2536 } 2537 2538 void ARMDAGToDAGISel::SelectMVE_LongShift(SDNode *N, uint16_t Opcode, 2539 bool Immediate, 2540 bool HasSaturationOperand) { 2541 SDLoc Loc(N); 2542 SmallVector<SDValue, 8> Ops; 2543 2544 // Two 32-bit halves of the value to be shifted 2545 Ops.push_back(N->getOperand(1)); 2546 Ops.push_back(N->getOperand(2)); 2547 2548 // The shift count 2549 if (Immediate) { 2550 int32_t ImmValue = cast<ConstantSDNode>(N->getOperand(3))->getZExtValue(); 2551 Ops.push_back(getI32Imm(ImmValue, Loc)); // immediate shift count 2552 } else { 2553 Ops.push_back(N->getOperand(3)); 2554 } 2555 2556 // The immediate saturation operand, if any 2557 if (HasSaturationOperand) { 2558 int32_t SatOp = cast<ConstantSDNode>(N->getOperand(4))->getZExtValue(); 2559 int SatBit = (SatOp == 64 ? 0 : 1); 2560 Ops.push_back(getI32Imm(SatBit, Loc)); 2561 } 2562 2563 // MVE scalar shifts are IT-predicable, so include the standard 2564 // predicate arguments. 2565 Ops.push_back(getAL(CurDAG, Loc)); 2566 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 2567 2568 CurDAG->SelectNodeTo(N, Opcode, N->getVTList(), makeArrayRef(Ops)); 2569 } 2570 2571 void ARMDAGToDAGISel::SelectMVE_VADCSBC(SDNode *N, uint16_t OpcodeWithCarry, 2572 uint16_t OpcodeWithNoCarry, 2573 bool Add, bool Predicated) { 2574 SDLoc Loc(N); 2575 SmallVector<SDValue, 8> Ops; 2576 uint16_t Opcode; 2577 2578 unsigned FirstInputOp = Predicated ? 2 : 1; 2579 2580 // Two input vectors and the input carry flag 2581 Ops.push_back(N->getOperand(FirstInputOp)); 2582 Ops.push_back(N->getOperand(FirstInputOp + 1)); 2583 SDValue CarryIn = N->getOperand(FirstInputOp + 2); 2584 ConstantSDNode *CarryInConstant = dyn_cast<ConstantSDNode>(CarryIn); 2585 uint32_t CarryMask = 1 << 29; 2586 uint32_t CarryExpected = Add ? 0 : CarryMask; 2587 if (CarryInConstant && 2588 (CarryInConstant->getZExtValue() & CarryMask) == CarryExpected) { 2589 Opcode = OpcodeWithNoCarry; 2590 } else { 2591 Ops.push_back(CarryIn); 2592 Opcode = OpcodeWithCarry; 2593 } 2594 2595 if (Predicated) 2596 AddMVEPredicateToOps(Ops, Loc, 2597 N->getOperand(FirstInputOp + 3), // predicate 2598 N->getOperand(FirstInputOp - 1)); // inactive 2599 else 2600 AddEmptyMVEPredicateToOps(Ops, Loc, N->getValueType(0)); 2601 2602 CurDAG->SelectNodeTo(N, Opcode, N->getVTList(), makeArrayRef(Ops)); 2603 } 2604 2605 void ARMDAGToDAGISel::SelectMVE_VSHLC(SDNode *N, bool Predicated) { 2606 SDLoc Loc(N); 2607 SmallVector<SDValue, 8> Ops; 2608 2609 // One vector input, followed by a 32-bit word of bits to shift in 2610 // and then an immediate shift count 2611 Ops.push_back(N->getOperand(1)); 2612 Ops.push_back(N->getOperand(2)); 2613 int32_t ImmValue = cast<ConstantSDNode>(N->getOperand(3))->getZExtValue(); 2614 Ops.push_back(getI32Imm(ImmValue, Loc)); // immediate shift count 2615 2616 if (Predicated) 2617 AddMVEPredicateToOps(Ops, Loc, N->getOperand(4)); 2618 else 2619 AddEmptyMVEPredicateToOps(Ops, Loc); 2620 2621 CurDAG->SelectNodeTo(N, ARM::MVE_VSHLC, N->getVTList(), makeArrayRef(Ops)); 2622 } 2623 2624 static bool SDValueToConstBool(SDValue SDVal) { 2625 assert(isa<ConstantSDNode>(SDVal) && "expected a compile-time constant"); 2626 ConstantSDNode *SDValConstant = dyn_cast<ConstantSDNode>(SDVal); 2627 uint64_t Value = SDValConstant->getZExtValue(); 2628 assert((Value == 0 || Value == 1) && "expected value 0 or 1"); 2629 return Value; 2630 } 2631 2632 void ARMDAGToDAGISel::SelectBaseMVE_VMLLDAV(SDNode *N, bool Predicated, 2633 const uint16_t *OpcodesS, 2634 const uint16_t *OpcodesU, 2635 size_t Stride, size_t TySize) { 2636 assert(TySize < Stride && "Invalid TySize"); 2637 bool IsUnsigned = SDValueToConstBool(N->getOperand(1)); 2638 bool IsSub = SDValueToConstBool(N->getOperand(2)); 2639 bool IsExchange = SDValueToConstBool(N->getOperand(3)); 2640 if (IsUnsigned) { 2641 assert(!IsSub && 2642 "Unsigned versions of vmlsldav[a]/vrmlsldavh[a] do not exist"); 2643 assert(!IsExchange && 2644 "Unsigned versions of vmlaldav[a]x/vrmlaldavh[a]x do not exist"); 2645 } 2646 2647 auto OpIsZero = [N](size_t OpNo) { 2648 if (ConstantSDNode *OpConst = dyn_cast<ConstantSDNode>(N->getOperand(OpNo))) 2649 if (OpConst->getZExtValue() == 0) 2650 return true; 2651 return false; 2652 }; 2653 2654 // If the input accumulator value is not zero, select an instruction with 2655 // accumulator, otherwise select an instruction without accumulator 2656 bool IsAccum = !(OpIsZero(4) && OpIsZero(5)); 2657 2658 const uint16_t *Opcodes = IsUnsigned ? OpcodesU : OpcodesS; 2659 if (IsSub) 2660 Opcodes += 4 * Stride; 2661 if (IsExchange) 2662 Opcodes += 2 * Stride; 2663 if (IsAccum) 2664 Opcodes += Stride; 2665 uint16_t Opcode = Opcodes[TySize]; 2666 2667 SDLoc Loc(N); 2668 SmallVector<SDValue, 8> Ops; 2669 // Push the accumulator operands, if they are used 2670 if (IsAccum) { 2671 Ops.push_back(N->getOperand(4)); 2672 Ops.push_back(N->getOperand(5)); 2673 } 2674 // Push the two vector operands 2675 Ops.push_back(N->getOperand(6)); 2676 Ops.push_back(N->getOperand(7)); 2677 2678 if (Predicated) 2679 AddMVEPredicateToOps(Ops, Loc, N->getOperand(8)); 2680 else 2681 AddEmptyMVEPredicateToOps(Ops, Loc); 2682 2683 CurDAG->SelectNodeTo(N, Opcode, N->getVTList(), makeArrayRef(Ops)); 2684 } 2685 2686 void ARMDAGToDAGISel::SelectMVE_VMLLDAV(SDNode *N, bool Predicated, 2687 const uint16_t *OpcodesS, 2688 const uint16_t *OpcodesU) { 2689 EVT VecTy = N->getOperand(6).getValueType(); 2690 size_t SizeIndex; 2691 switch (VecTy.getVectorElementType().getSizeInBits()) { 2692 case 16: 2693 SizeIndex = 0; 2694 break; 2695 case 32: 2696 SizeIndex = 1; 2697 break; 2698 default: 2699 llvm_unreachable("bad vector element size"); 2700 } 2701 2702 SelectBaseMVE_VMLLDAV(N, Predicated, OpcodesS, OpcodesU, 2, SizeIndex); 2703 } 2704 2705 void ARMDAGToDAGISel::SelectMVE_VRMLLDAVH(SDNode *N, bool Predicated, 2706 const uint16_t *OpcodesS, 2707 const uint16_t *OpcodesU) { 2708 assert( 2709 N->getOperand(6).getValueType().getVectorElementType().getSizeInBits() == 2710 32 && 2711 "bad vector element size"); 2712 SelectBaseMVE_VMLLDAV(N, Predicated, OpcodesS, OpcodesU, 1, 0); 2713 } 2714 2715 void ARMDAGToDAGISel::SelectMVE_VLD(SDNode *N, unsigned NumVecs, 2716 const uint16_t *const *Opcodes, 2717 bool HasWriteback) { 2718 EVT VT = N->getValueType(0); 2719 SDLoc Loc(N); 2720 2721 const uint16_t *OurOpcodes; 2722 switch (VT.getVectorElementType().getSizeInBits()) { 2723 case 8: 2724 OurOpcodes = Opcodes[0]; 2725 break; 2726 case 16: 2727 OurOpcodes = Opcodes[1]; 2728 break; 2729 case 32: 2730 OurOpcodes = Opcodes[2]; 2731 break; 2732 default: 2733 llvm_unreachable("bad vector element size in SelectMVE_VLD"); 2734 } 2735 2736 EVT DataTy = EVT::getVectorVT(*CurDAG->getContext(), MVT::i64, NumVecs * 2); 2737 SmallVector<EVT, 4> ResultTys = {DataTy, MVT::Other}; 2738 unsigned PtrOperand = HasWriteback ? 1 : 2; 2739 2740 auto Data = SDValue( 2741 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, Loc, DataTy), 0); 2742 SDValue Chain = N->getOperand(0); 2743 // Add a MVE_VLDn instruction for each Vec, except the last 2744 for (unsigned Stage = 0; Stage < NumVecs - 1; ++Stage) { 2745 SDValue Ops[] = {Data, N->getOperand(PtrOperand), Chain}; 2746 auto LoadInst = 2747 CurDAG->getMachineNode(OurOpcodes[Stage], Loc, ResultTys, Ops); 2748 Data = SDValue(LoadInst, 0); 2749 Chain = SDValue(LoadInst, 1); 2750 } 2751 // The last may need a writeback on it 2752 if (HasWriteback) 2753 ResultTys = {DataTy, MVT::i32, MVT::Other}; 2754 SDValue Ops[] = {Data, N->getOperand(PtrOperand), Chain}; 2755 auto LoadInst = 2756 CurDAG->getMachineNode(OurOpcodes[NumVecs - 1], Loc, ResultTys, Ops); 2757 2758 unsigned i; 2759 for (i = 0; i < NumVecs; i++) 2760 ReplaceUses(SDValue(N, i), 2761 CurDAG->getTargetExtractSubreg(ARM::qsub_0 + i, Loc, VT, 2762 SDValue(LoadInst, 0))); 2763 if (HasWriteback) 2764 ReplaceUses(SDValue(N, i++), SDValue(LoadInst, 1)); 2765 ReplaceUses(SDValue(N, i), SDValue(LoadInst, HasWriteback ? 2 : 1)); 2766 CurDAG->RemoveDeadNode(N); 2767 } 2768 2769 void ARMDAGToDAGISel::SelectMVE_VxDUP(SDNode *N, const uint16_t *Opcodes, 2770 bool Wrapping, bool Predicated) { 2771 EVT VT = N->getValueType(0); 2772 SDLoc Loc(N); 2773 2774 uint16_t Opcode; 2775 switch (VT.getScalarSizeInBits()) { 2776 case 8: 2777 Opcode = Opcodes[0]; 2778 break; 2779 case 16: 2780 Opcode = Opcodes[1]; 2781 break; 2782 case 32: 2783 Opcode = Opcodes[2]; 2784 break; 2785 default: 2786 llvm_unreachable("bad vector element size in SelectMVE_VxDUP"); 2787 } 2788 2789 SmallVector<SDValue, 8> Ops; 2790 unsigned OpIdx = 1; 2791 2792 SDValue Inactive; 2793 if (Predicated) 2794 Inactive = N->getOperand(OpIdx++); 2795 2796 Ops.push_back(N->getOperand(OpIdx++)); // base 2797 if (Wrapping) 2798 Ops.push_back(N->getOperand(OpIdx++)); // limit 2799 2800 SDValue ImmOp = N->getOperand(OpIdx++); // step 2801 int ImmValue = cast<ConstantSDNode>(ImmOp)->getZExtValue(); 2802 Ops.push_back(getI32Imm(ImmValue, Loc)); 2803 2804 if (Predicated) 2805 AddMVEPredicateToOps(Ops, Loc, N->getOperand(OpIdx), Inactive); 2806 else 2807 AddEmptyMVEPredicateToOps(Ops, Loc, N->getValueType(0)); 2808 2809 CurDAG->SelectNodeTo(N, Opcode, N->getVTList(), makeArrayRef(Ops)); 2810 } 2811 2812 void ARMDAGToDAGISel::SelectVLDDup(SDNode *N, bool IsIntrinsic, 2813 bool isUpdating, unsigned NumVecs, 2814 const uint16_t *DOpcodes, 2815 const uint16_t *QOpcodes0, 2816 const uint16_t *QOpcodes1) { 2817 assert(Subtarget->hasNEON()); 2818 assert(NumVecs >= 1 && NumVecs <= 4 && "VLDDup NumVecs out-of-range"); 2819 SDLoc dl(N); 2820 2821 SDValue MemAddr, Align; 2822 unsigned AddrOpIdx = IsIntrinsic ? 2 : 1; 2823 if (!SelectAddrMode6(N, N->getOperand(AddrOpIdx), MemAddr, Align)) 2824 return; 2825 2826 SDValue Chain = N->getOperand(0); 2827 EVT VT = N->getValueType(0); 2828 bool is64BitVector = VT.is64BitVector(); 2829 2830 unsigned Alignment = 0; 2831 if (NumVecs != 3) { 2832 Alignment = cast<ConstantSDNode>(Align)->getZExtValue(); 2833 unsigned NumBytes = NumVecs * VT.getScalarSizeInBits() / 8; 2834 if (Alignment > NumBytes) 2835 Alignment = NumBytes; 2836 if (Alignment < 8 && Alignment < NumBytes) 2837 Alignment = 0; 2838 // Alignment must be a power of two; make sure of that. 2839 Alignment = (Alignment & -Alignment); 2840 if (Alignment == 1) 2841 Alignment = 0; 2842 } 2843 Align = CurDAG->getTargetConstant(Alignment, dl, MVT::i32); 2844 2845 unsigned OpcodeIndex; 2846 switch (VT.getSimpleVT().SimpleTy) { 2847 default: llvm_unreachable("unhandled vld-dup type"); 2848 case MVT::v8i8: 2849 case MVT::v16i8: OpcodeIndex = 0; break; 2850 case MVT::v4i16: 2851 case MVT::v8i16: 2852 case MVT::v4f16: 2853 case MVT::v8f16: 2854 OpcodeIndex = 1; break; 2855 case MVT::v2f32: 2856 case MVT::v2i32: 2857 case MVT::v4f32: 2858 case MVT::v4i32: OpcodeIndex = 2; break; 2859 case MVT::v1f64: 2860 case MVT::v1i64: OpcodeIndex = 3; break; 2861 } 2862 2863 unsigned ResTyElts = (NumVecs == 3) ? 4 : NumVecs; 2864 if (!is64BitVector) 2865 ResTyElts *= 2; 2866 EVT ResTy = EVT::getVectorVT(*CurDAG->getContext(), MVT::i64, ResTyElts); 2867 2868 std::vector<EVT> ResTys; 2869 ResTys.push_back(ResTy); 2870 if (isUpdating) 2871 ResTys.push_back(MVT::i32); 2872 ResTys.push_back(MVT::Other); 2873 2874 SDValue Pred = getAL(CurDAG, dl); 2875 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2876 2877 SDNode *VLdDup; 2878 if (is64BitVector || NumVecs == 1) { 2879 SmallVector<SDValue, 6> Ops; 2880 Ops.push_back(MemAddr); 2881 Ops.push_back(Align); 2882 unsigned Opc = is64BitVector ? DOpcodes[OpcodeIndex] : 2883 QOpcodes0[OpcodeIndex]; 2884 if (isUpdating) { 2885 // fixed-stride update instructions don't have an explicit writeback 2886 // operand. It's implicit in the opcode itself. 2887 SDValue Inc = N->getOperand(2); 2888 bool IsImmUpdate = 2889 isPerfectIncrement(Inc, VT.getVectorElementType(), NumVecs); 2890 if (NumVecs <= 2 && !IsImmUpdate) 2891 Opc = getVLDSTRegisterUpdateOpcode(Opc); 2892 if (!IsImmUpdate) 2893 Ops.push_back(Inc); 2894 // FIXME: VLD3 and VLD4 haven't been updated to that form yet. 2895 else if (NumVecs > 2) 2896 Ops.push_back(Reg0); 2897 } 2898 Ops.push_back(Pred); 2899 Ops.push_back(Reg0); 2900 Ops.push_back(Chain); 2901 VLdDup = CurDAG->getMachineNode(Opc, dl, ResTys, Ops); 2902 } else if (NumVecs == 2) { 2903 const SDValue OpsA[] = { MemAddr, Align, Pred, Reg0, Chain }; 2904 SDNode *VLdA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], 2905 dl, ResTys, OpsA); 2906 2907 Chain = SDValue(VLdA, 1); 2908 const SDValue OpsB[] = { MemAddr, Align, Pred, Reg0, Chain }; 2909 VLdDup = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys, OpsB); 2910 } else { 2911 SDValue ImplDef = 2912 SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, ResTy), 0); 2913 const SDValue OpsA[] = { MemAddr, Align, ImplDef, Pred, Reg0, Chain }; 2914 SDNode *VLdA = CurDAG->getMachineNode(QOpcodes0[OpcodeIndex], 2915 dl, ResTys, OpsA); 2916 2917 SDValue SuperReg = SDValue(VLdA, 0); 2918 Chain = SDValue(VLdA, 1); 2919 const SDValue OpsB[] = { MemAddr, Align, SuperReg, Pred, Reg0, Chain }; 2920 VLdDup = CurDAG->getMachineNode(QOpcodes1[OpcodeIndex], dl, ResTys, OpsB); 2921 } 2922 2923 // Transfer memoperands. 2924 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 2925 CurDAG->setNodeMemRefs(cast<MachineSDNode>(VLdDup), {MemOp}); 2926 2927 // Extract the subregisters. 2928 if (NumVecs == 1) { 2929 ReplaceUses(SDValue(N, 0), SDValue(VLdDup, 0)); 2930 } else { 2931 SDValue SuperReg = SDValue(VLdDup, 0); 2932 static_assert(ARM::dsub_7 == ARM::dsub_0 + 7, "Unexpected subreg numbering"); 2933 unsigned SubIdx = is64BitVector ? ARM::dsub_0 : ARM::qsub_0; 2934 for (unsigned Vec = 0; Vec != NumVecs; ++Vec) { 2935 ReplaceUses(SDValue(N, Vec), 2936 CurDAG->getTargetExtractSubreg(SubIdx+Vec, dl, VT, SuperReg)); 2937 } 2938 } 2939 ReplaceUses(SDValue(N, NumVecs), SDValue(VLdDup, 1)); 2940 if (isUpdating) 2941 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(VLdDup, 2)); 2942 CurDAG->RemoveDeadNode(N); 2943 } 2944 2945 bool ARMDAGToDAGISel::tryV6T2BitfieldExtractOp(SDNode *N, bool isSigned) { 2946 if (!Subtarget->hasV6T2Ops()) 2947 return false; 2948 2949 unsigned Opc = isSigned 2950 ? (Subtarget->isThumb() ? ARM::t2SBFX : ARM::SBFX) 2951 : (Subtarget->isThumb() ? ARM::t2UBFX : ARM::UBFX); 2952 SDLoc dl(N); 2953 2954 // For unsigned extracts, check for a shift right and mask 2955 unsigned And_imm = 0; 2956 if (N->getOpcode() == ISD::AND) { 2957 if (isOpcWithIntImmediate(N, ISD::AND, And_imm)) { 2958 2959 // The immediate is a mask of the low bits iff imm & (imm+1) == 0 2960 if (And_imm & (And_imm + 1)) 2961 return false; 2962 2963 unsigned Srl_imm = 0; 2964 if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRL, 2965 Srl_imm)) { 2966 assert(Srl_imm > 0 && Srl_imm < 32 && "bad amount in shift node!"); 2967 2968 // Mask off the unnecessary bits of the AND immediate; normally 2969 // DAGCombine will do this, but that might not happen if 2970 // targetShrinkDemandedConstant chooses a different immediate. 2971 And_imm &= -1U >> Srl_imm; 2972 2973 // Note: The width operand is encoded as width-1. 2974 unsigned Width = countTrailingOnes(And_imm) - 1; 2975 unsigned LSB = Srl_imm; 2976 2977 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 2978 2979 if ((LSB + Width + 1) == N->getValueType(0).getSizeInBits()) { 2980 // It's cheaper to use a right shift to extract the top bits. 2981 if (Subtarget->isThumb()) { 2982 Opc = isSigned ? ARM::t2ASRri : ARM::t2LSRri; 2983 SDValue Ops[] = { N->getOperand(0).getOperand(0), 2984 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 2985 getAL(CurDAG, dl), Reg0, Reg0 }; 2986 CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 2987 return true; 2988 } 2989 2990 // ARM models shift instructions as MOVsi with shifter operand. 2991 ARM_AM::ShiftOpc ShOpcVal = ARM_AM::getShiftOpcForNode(ISD::SRL); 2992 SDValue ShOpc = 2993 CurDAG->getTargetConstant(ARM_AM::getSORegOpc(ShOpcVal, LSB), dl, 2994 MVT::i32); 2995 SDValue Ops[] = { N->getOperand(0).getOperand(0), ShOpc, 2996 getAL(CurDAG, dl), Reg0, Reg0 }; 2997 CurDAG->SelectNodeTo(N, ARM::MOVsi, MVT::i32, Ops); 2998 return true; 2999 } 3000 3001 assert(LSB + Width + 1 <= 32 && "Shouldn't create an invalid ubfx"); 3002 SDValue Ops[] = { N->getOperand(0).getOperand(0), 3003 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 3004 CurDAG->getTargetConstant(Width, dl, MVT::i32), 3005 getAL(CurDAG, dl), Reg0 }; 3006 CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 3007 return true; 3008 } 3009 } 3010 return false; 3011 } 3012 3013 // Otherwise, we're looking for a shift of a shift 3014 unsigned Shl_imm = 0; 3015 if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SHL, Shl_imm)) { 3016 assert(Shl_imm > 0 && Shl_imm < 32 && "bad amount in shift node!"); 3017 unsigned Srl_imm = 0; 3018 if (isInt32Immediate(N->getOperand(1), Srl_imm)) { 3019 assert(Srl_imm > 0 && Srl_imm < 32 && "bad amount in shift node!"); 3020 // Note: The width operand is encoded as width-1. 3021 unsigned Width = 32 - Srl_imm - 1; 3022 int LSB = Srl_imm - Shl_imm; 3023 if (LSB < 0) 3024 return false; 3025 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 3026 assert(LSB + Width + 1 <= 32 && "Shouldn't create an invalid ubfx"); 3027 SDValue Ops[] = { N->getOperand(0).getOperand(0), 3028 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 3029 CurDAG->getTargetConstant(Width, dl, MVT::i32), 3030 getAL(CurDAG, dl), Reg0 }; 3031 CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 3032 return true; 3033 } 3034 } 3035 3036 // Or we are looking for a shift of an and, with a mask operand 3037 if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::AND, And_imm) && 3038 isShiftedMask_32(And_imm)) { 3039 unsigned Srl_imm = 0; 3040 unsigned LSB = countTrailingZeros(And_imm); 3041 // Shift must be the same as the ands lsb 3042 if (isInt32Immediate(N->getOperand(1), Srl_imm) && Srl_imm == LSB) { 3043 assert(Srl_imm > 0 && Srl_imm < 32 && "bad amount in shift node!"); 3044 unsigned MSB = 31 - countLeadingZeros(And_imm); 3045 // Note: The width operand is encoded as width-1. 3046 unsigned Width = MSB - LSB; 3047 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 3048 assert(Srl_imm + Width + 1 <= 32 && "Shouldn't create an invalid ubfx"); 3049 SDValue Ops[] = { N->getOperand(0).getOperand(0), 3050 CurDAG->getTargetConstant(Srl_imm, dl, MVT::i32), 3051 CurDAG->getTargetConstant(Width, dl, MVT::i32), 3052 getAL(CurDAG, dl), Reg0 }; 3053 CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 3054 return true; 3055 } 3056 } 3057 3058 if (N->getOpcode() == ISD::SIGN_EXTEND_INREG) { 3059 unsigned Width = cast<VTSDNode>(N->getOperand(1))->getVT().getSizeInBits(); 3060 unsigned LSB = 0; 3061 if (!isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRL, LSB) && 3062 !isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SRA, LSB)) 3063 return false; 3064 3065 if (LSB + Width > 32) 3066 return false; 3067 3068 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 3069 assert(LSB + Width <= 32 && "Shouldn't create an invalid ubfx"); 3070 SDValue Ops[] = { N->getOperand(0).getOperand(0), 3071 CurDAG->getTargetConstant(LSB, dl, MVT::i32), 3072 CurDAG->getTargetConstant(Width - 1, dl, MVT::i32), 3073 getAL(CurDAG, dl), Reg0 }; 3074 CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 3075 return true; 3076 } 3077 3078 return false; 3079 } 3080 3081 /// Target-specific DAG combining for ISD::XOR. 3082 /// Target-independent combining lowers SELECT_CC nodes of the form 3083 /// select_cc setg[ge] X, 0, X, -X 3084 /// select_cc setgt X, -1, X, -X 3085 /// select_cc setl[te] X, 0, -X, X 3086 /// select_cc setlt X, 1, -X, X 3087 /// which represent Integer ABS into: 3088 /// Y = sra (X, size(X)-1); xor (add (X, Y), Y) 3089 /// ARM instruction selection detects the latter and matches it to 3090 /// ARM::ABS or ARM::t2ABS machine node. 3091 bool ARMDAGToDAGISel::tryABSOp(SDNode *N){ 3092 SDValue XORSrc0 = N->getOperand(0); 3093 SDValue XORSrc1 = N->getOperand(1); 3094 EVT VT = N->getValueType(0); 3095 3096 if (Subtarget->isThumb1Only()) 3097 return false; 3098 3099 if (XORSrc0.getOpcode() != ISD::ADD || XORSrc1.getOpcode() != ISD::SRA) 3100 return false; 3101 3102 SDValue ADDSrc0 = XORSrc0.getOperand(0); 3103 SDValue ADDSrc1 = XORSrc0.getOperand(1); 3104 SDValue SRASrc0 = XORSrc1.getOperand(0); 3105 SDValue SRASrc1 = XORSrc1.getOperand(1); 3106 ConstantSDNode *SRAConstant = dyn_cast<ConstantSDNode>(SRASrc1); 3107 EVT XType = SRASrc0.getValueType(); 3108 unsigned Size = XType.getSizeInBits() - 1; 3109 3110 if (ADDSrc1 == XORSrc1 && ADDSrc0 == SRASrc0 && 3111 XType.isInteger() && SRAConstant != nullptr && 3112 Size == SRAConstant->getZExtValue()) { 3113 unsigned Opcode = Subtarget->isThumb2() ? ARM::t2ABS : ARM::ABS; 3114 CurDAG->SelectNodeTo(N, Opcode, VT, ADDSrc0); 3115 return true; 3116 } 3117 3118 return false; 3119 } 3120 3121 /// We've got special pseudo-instructions for these 3122 void ARMDAGToDAGISel::SelectCMP_SWAP(SDNode *N) { 3123 unsigned Opcode; 3124 EVT MemTy = cast<MemSDNode>(N)->getMemoryVT(); 3125 if (MemTy == MVT::i8) 3126 Opcode = ARM::CMP_SWAP_8; 3127 else if (MemTy == MVT::i16) 3128 Opcode = ARM::CMP_SWAP_16; 3129 else if (MemTy == MVT::i32) 3130 Opcode = ARM::CMP_SWAP_32; 3131 else 3132 llvm_unreachable("Unknown AtomicCmpSwap type"); 3133 3134 SDValue Ops[] = {N->getOperand(1), N->getOperand(2), N->getOperand(3), 3135 N->getOperand(0)}; 3136 SDNode *CmpSwap = CurDAG->getMachineNode( 3137 Opcode, SDLoc(N), 3138 CurDAG->getVTList(MVT::i32, MVT::i32, MVT::Other), Ops); 3139 3140 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand(); 3141 CurDAG->setNodeMemRefs(cast<MachineSDNode>(CmpSwap), {MemOp}); 3142 3143 ReplaceUses(SDValue(N, 0), SDValue(CmpSwap, 0)); 3144 ReplaceUses(SDValue(N, 1), SDValue(CmpSwap, 2)); 3145 CurDAG->RemoveDeadNode(N); 3146 } 3147 3148 static Optional<std::pair<unsigned, unsigned>> 3149 getContiguousRangeOfSetBits(const APInt &A) { 3150 unsigned FirstOne = A.getBitWidth() - A.countLeadingZeros() - 1; 3151 unsigned LastOne = A.countTrailingZeros(); 3152 if (A.countPopulation() != (FirstOne - LastOne + 1)) 3153 return Optional<std::pair<unsigned,unsigned>>(); 3154 return std::make_pair(FirstOne, LastOne); 3155 } 3156 3157 void ARMDAGToDAGISel::SelectCMPZ(SDNode *N, bool &SwitchEQNEToPLMI) { 3158 assert(N->getOpcode() == ARMISD::CMPZ); 3159 SwitchEQNEToPLMI = false; 3160 3161 if (!Subtarget->isThumb()) 3162 // FIXME: Work out whether it is profitable to do this in A32 mode - LSL and 3163 // LSR don't exist as standalone instructions - they need the barrel shifter. 3164 return; 3165 3166 // select (cmpz (and X, C), #0) -> (LSLS X) or (LSRS X) or (LSRS (LSLS X)) 3167 SDValue And = N->getOperand(0); 3168 if (!And->hasOneUse()) 3169 return; 3170 3171 SDValue Zero = N->getOperand(1); 3172 if (!isa<ConstantSDNode>(Zero) || !cast<ConstantSDNode>(Zero)->isNullValue() || 3173 And->getOpcode() != ISD::AND) 3174 return; 3175 SDValue X = And.getOperand(0); 3176 auto C = dyn_cast<ConstantSDNode>(And.getOperand(1)); 3177 3178 if (!C) 3179 return; 3180 auto Range = getContiguousRangeOfSetBits(C->getAPIntValue()); 3181 if (!Range) 3182 return; 3183 3184 // There are several ways to lower this: 3185 SDNode *NewN; 3186 SDLoc dl(N); 3187 3188 auto EmitShift = [&](unsigned Opc, SDValue Src, unsigned Imm) -> SDNode* { 3189 if (Subtarget->isThumb2()) { 3190 Opc = (Opc == ARM::tLSLri) ? ARM::t2LSLri : ARM::t2LSRri; 3191 SDValue Ops[] = { Src, CurDAG->getTargetConstant(Imm, dl, MVT::i32), 3192 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32), 3193 CurDAG->getRegister(0, MVT::i32) }; 3194 return CurDAG->getMachineNode(Opc, dl, MVT::i32, Ops); 3195 } else { 3196 SDValue Ops[] = {CurDAG->getRegister(ARM::CPSR, MVT::i32), Src, 3197 CurDAG->getTargetConstant(Imm, dl, MVT::i32), 3198 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32)}; 3199 return CurDAG->getMachineNode(Opc, dl, MVT::i32, Ops); 3200 } 3201 }; 3202 3203 if (Range->second == 0) { 3204 // 1. Mask includes the LSB -> Simply shift the top N bits off 3205 NewN = EmitShift(ARM::tLSLri, X, 31 - Range->first); 3206 ReplaceNode(And.getNode(), NewN); 3207 } else if (Range->first == 31) { 3208 // 2. Mask includes the MSB -> Simply shift the bottom N bits off 3209 NewN = EmitShift(ARM::tLSRri, X, Range->second); 3210 ReplaceNode(And.getNode(), NewN); 3211 } else if (Range->first == Range->second) { 3212 // 3. Only one bit is set. We can shift this into the sign bit and use a 3213 // PL/MI comparison. 3214 NewN = EmitShift(ARM::tLSLri, X, 31 - Range->first); 3215 ReplaceNode(And.getNode(), NewN); 3216 3217 SwitchEQNEToPLMI = true; 3218 } else if (!Subtarget->hasV6T2Ops()) { 3219 // 4. Do a double shift to clear bottom and top bits, but only in 3220 // thumb-1 mode as in thumb-2 we can use UBFX. 3221 NewN = EmitShift(ARM::tLSLri, X, 31 - Range->first); 3222 NewN = EmitShift(ARM::tLSRri, SDValue(NewN, 0), 3223 Range->second + (31 - Range->first)); 3224 ReplaceNode(And.getNode(), NewN); 3225 } 3226 3227 } 3228 3229 void ARMDAGToDAGISel::Select(SDNode *N) { 3230 SDLoc dl(N); 3231 3232 if (N->isMachineOpcode()) { 3233 N->setNodeId(-1); 3234 return; // Already selected. 3235 } 3236 3237 switch (N->getOpcode()) { 3238 default: break; 3239 case ISD::STORE: { 3240 // For Thumb1, match an sp-relative store in C++. This is a little 3241 // unfortunate, but I don't think I can make the chain check work 3242 // otherwise. (The chain of the store has to be the same as the chain 3243 // of the CopyFromReg, or else we can't replace the CopyFromReg with 3244 // a direct reference to "SP".) 3245 // 3246 // This is only necessary on Thumb1 because Thumb1 sp-relative stores use 3247 // a different addressing mode from other four-byte stores. 3248 // 3249 // This pattern usually comes up with call arguments. 3250 StoreSDNode *ST = cast<StoreSDNode>(N); 3251 SDValue Ptr = ST->getBasePtr(); 3252 if (Subtarget->isThumb1Only() && ST->isUnindexed()) { 3253 int RHSC = 0; 3254 if (Ptr.getOpcode() == ISD::ADD && 3255 isScaledConstantInRange(Ptr.getOperand(1), /*Scale=*/4, 0, 256, RHSC)) 3256 Ptr = Ptr.getOperand(0); 3257 3258 if (Ptr.getOpcode() == ISD::CopyFromReg && 3259 cast<RegisterSDNode>(Ptr.getOperand(1))->getReg() == ARM::SP && 3260 Ptr.getOperand(0) == ST->getChain()) { 3261 SDValue Ops[] = {ST->getValue(), 3262 CurDAG->getRegister(ARM::SP, MVT::i32), 3263 CurDAG->getTargetConstant(RHSC, dl, MVT::i32), 3264 getAL(CurDAG, dl), 3265 CurDAG->getRegister(0, MVT::i32), 3266 ST->getChain()}; 3267 MachineSDNode *ResNode = 3268 CurDAG->getMachineNode(ARM::tSTRspi, dl, MVT::Other, Ops); 3269 MachineMemOperand *MemOp = ST->getMemOperand(); 3270 CurDAG->setNodeMemRefs(cast<MachineSDNode>(ResNode), {MemOp}); 3271 ReplaceNode(N, ResNode); 3272 return; 3273 } 3274 } 3275 break; 3276 } 3277 case ISD::WRITE_REGISTER: 3278 if (tryWriteRegister(N)) 3279 return; 3280 break; 3281 case ISD::READ_REGISTER: 3282 if (tryReadRegister(N)) 3283 return; 3284 break; 3285 case ISD::INLINEASM: 3286 case ISD::INLINEASM_BR: 3287 if (tryInlineAsm(N)) 3288 return; 3289 break; 3290 case ISD::XOR: 3291 // Select special operations if XOR node forms integer ABS pattern 3292 if (tryABSOp(N)) 3293 return; 3294 // Other cases are autogenerated. 3295 break; 3296 case ISD::Constant: { 3297 unsigned Val = cast<ConstantSDNode>(N)->getZExtValue(); 3298 // If we can't materialize the constant we need to use a literal pool 3299 if (ConstantMaterializationCost(Val, Subtarget) > 2) { 3300 SDValue CPIdx = CurDAG->getTargetConstantPool( 3301 ConstantInt::get(Type::getInt32Ty(*CurDAG->getContext()), Val), 3302 TLI->getPointerTy(CurDAG->getDataLayout())); 3303 3304 SDNode *ResNode; 3305 if (Subtarget->isThumb()) { 3306 SDValue Ops[] = { 3307 CPIdx, 3308 getAL(CurDAG, dl), 3309 CurDAG->getRegister(0, MVT::i32), 3310 CurDAG->getEntryNode() 3311 }; 3312 ResNode = CurDAG->getMachineNode(ARM::tLDRpci, dl, MVT::i32, MVT::Other, 3313 Ops); 3314 } else { 3315 SDValue Ops[] = { 3316 CPIdx, 3317 CurDAG->getTargetConstant(0, dl, MVT::i32), 3318 getAL(CurDAG, dl), 3319 CurDAG->getRegister(0, MVT::i32), 3320 CurDAG->getEntryNode() 3321 }; 3322 ResNode = CurDAG->getMachineNode(ARM::LDRcp, dl, MVT::i32, MVT::Other, 3323 Ops); 3324 } 3325 // Annotate the Node with memory operand information so that MachineInstr 3326 // queries work properly. This e.g. gives the register allocation the 3327 // required information for rematerialization. 3328 MachineFunction& MF = CurDAG->getMachineFunction(); 3329 MachineMemOperand *MemOp = 3330 MF.getMachineMemOperand(MachinePointerInfo::getConstantPool(MF), 3331 MachineMemOperand::MOLoad, 4, 4); 3332 3333 CurDAG->setNodeMemRefs(cast<MachineSDNode>(ResNode), {MemOp}); 3334 3335 ReplaceNode(N, ResNode); 3336 return; 3337 } 3338 3339 // Other cases are autogenerated. 3340 break; 3341 } 3342 case ISD::FrameIndex: { 3343 // Selects to ADDri FI, 0 which in turn will become ADDri SP, imm. 3344 int FI = cast<FrameIndexSDNode>(N)->getIndex(); 3345 SDValue TFI = CurDAG->getTargetFrameIndex( 3346 FI, TLI->getPointerTy(CurDAG->getDataLayout())); 3347 if (Subtarget->isThumb1Only()) { 3348 // Set the alignment of the frame object to 4, to avoid having to generate 3349 // more than one ADD 3350 MachineFrameInfo &MFI = MF->getFrameInfo(); 3351 if (MFI.getObjectAlignment(FI) < 4) 3352 MFI.setObjectAlignment(FI, 4); 3353 CurDAG->SelectNodeTo(N, ARM::tADDframe, MVT::i32, TFI, 3354 CurDAG->getTargetConstant(0, dl, MVT::i32)); 3355 return; 3356 } else { 3357 unsigned Opc = ((Subtarget->isThumb() && Subtarget->hasThumb2()) ? 3358 ARM::t2ADDri : ARM::ADDri); 3359 SDValue Ops[] = { TFI, CurDAG->getTargetConstant(0, dl, MVT::i32), 3360 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32), 3361 CurDAG->getRegister(0, MVT::i32) }; 3362 CurDAG->SelectNodeTo(N, Opc, MVT::i32, Ops); 3363 return; 3364 } 3365 } 3366 case ISD::SRL: 3367 if (tryV6T2BitfieldExtractOp(N, false)) 3368 return; 3369 break; 3370 case ISD::SIGN_EXTEND_INREG: 3371 case ISD::SRA: 3372 if (tryV6T2BitfieldExtractOp(N, true)) 3373 return; 3374 break; 3375 case ISD::MUL: 3376 if (Subtarget->isThumb1Only()) 3377 break; 3378 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(1))) { 3379 unsigned RHSV = C->getZExtValue(); 3380 if (!RHSV) break; 3381 if (isPowerOf2_32(RHSV-1)) { // 2^n+1? 3382 unsigned ShImm = Log2_32(RHSV-1); 3383 if (ShImm >= 32) 3384 break; 3385 SDValue V = N->getOperand(0); 3386 ShImm = ARM_AM::getSORegOpc(ARM_AM::lsl, ShImm); 3387 SDValue ShImmOp = CurDAG->getTargetConstant(ShImm, dl, MVT::i32); 3388 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 3389 if (Subtarget->isThumb()) { 3390 SDValue Ops[] = { V, V, ShImmOp, getAL(CurDAG, dl), Reg0, Reg0 }; 3391 CurDAG->SelectNodeTo(N, ARM::t2ADDrs, MVT::i32, Ops); 3392 return; 3393 } else { 3394 SDValue Ops[] = { V, V, Reg0, ShImmOp, getAL(CurDAG, dl), Reg0, 3395 Reg0 }; 3396 CurDAG->SelectNodeTo(N, ARM::ADDrsi, MVT::i32, Ops); 3397 return; 3398 } 3399 } 3400 if (isPowerOf2_32(RHSV+1)) { // 2^n-1? 3401 unsigned ShImm = Log2_32(RHSV+1); 3402 if (ShImm >= 32) 3403 break; 3404 SDValue V = N->getOperand(0); 3405 ShImm = ARM_AM::getSORegOpc(ARM_AM::lsl, ShImm); 3406 SDValue ShImmOp = CurDAG->getTargetConstant(ShImm, dl, MVT::i32); 3407 SDValue Reg0 = CurDAG->getRegister(0, MVT::i32); 3408 if (Subtarget->isThumb()) { 3409 SDValue Ops[] = { V, V, ShImmOp, getAL(CurDAG, dl), Reg0, Reg0 }; 3410 CurDAG->SelectNodeTo(N, ARM::t2RSBrs, MVT::i32, Ops); 3411 return; 3412 } else { 3413 SDValue Ops[] = { V, V, Reg0, ShImmOp, getAL(CurDAG, dl), Reg0, 3414 Reg0 }; 3415 CurDAG->SelectNodeTo(N, ARM::RSBrsi, MVT::i32, Ops); 3416 return; 3417 } 3418 } 3419 } 3420 break; 3421 case ISD::AND: { 3422 // Check for unsigned bitfield extract 3423 if (tryV6T2BitfieldExtractOp(N, false)) 3424 return; 3425 3426 // If an immediate is used in an AND node, it is possible that the immediate 3427 // can be more optimally materialized when negated. If this is the case we 3428 // can negate the immediate and use a BIC instead. 3429 auto *N1C = dyn_cast<ConstantSDNode>(N->getOperand(1)); 3430 if (N1C && N1C->hasOneUse() && Subtarget->isThumb()) { 3431 uint32_t Imm = (uint32_t) N1C->getZExtValue(); 3432 3433 // In Thumb2 mode, an AND can take a 12-bit immediate. If this 3434 // immediate can be negated and fit in the immediate operand of 3435 // a t2BIC, don't do any manual transform here as this can be 3436 // handled by the generic ISel machinery. 3437 bool PreferImmediateEncoding = 3438 Subtarget->hasThumb2() && (is_t2_so_imm(Imm) || is_t2_so_imm_not(Imm)); 3439 if (!PreferImmediateEncoding && 3440 ConstantMaterializationCost(Imm, Subtarget) > 3441 ConstantMaterializationCost(~Imm, Subtarget)) { 3442 // The current immediate costs more to materialize than a negated 3443 // immediate, so negate the immediate and use a BIC. 3444 SDValue NewImm = 3445 CurDAG->getConstant(~N1C->getZExtValue(), dl, MVT::i32); 3446 // If the new constant didn't exist before, reposition it in the topological 3447 // ordering so it is just before N. Otherwise, don't touch its location. 3448 if (NewImm->getNodeId() == -1) 3449 CurDAG->RepositionNode(N->getIterator(), NewImm.getNode()); 3450 3451 if (!Subtarget->hasThumb2()) { 3452 SDValue Ops[] = {CurDAG->getRegister(ARM::CPSR, MVT::i32), 3453 N->getOperand(0), NewImm, getAL(CurDAG, dl), 3454 CurDAG->getRegister(0, MVT::i32)}; 3455 ReplaceNode(N, CurDAG->getMachineNode(ARM::tBIC, dl, MVT::i32, Ops)); 3456 return; 3457 } else { 3458 SDValue Ops[] = {N->getOperand(0), NewImm, getAL(CurDAG, dl), 3459 CurDAG->getRegister(0, MVT::i32), 3460 CurDAG->getRegister(0, MVT::i32)}; 3461 ReplaceNode(N, 3462 CurDAG->getMachineNode(ARM::t2BICrr, dl, MVT::i32, Ops)); 3463 return; 3464 } 3465 } 3466 } 3467 3468 // (and (or x, c2), c1) and top 16-bits of c1 and c2 match, lower 16-bits 3469 // of c1 are 0xffff, and lower 16-bit of c2 are 0. That is, the top 16-bits 3470 // are entirely contributed by c2 and lower 16-bits are entirely contributed 3471 // by x. That's equal to (or (and x, 0xffff), (and c1, 0xffff0000)). 3472 // Select it to: "movt x, ((c1 & 0xffff) >> 16) 3473 EVT VT = N->getValueType(0); 3474 if (VT != MVT::i32) 3475 break; 3476 unsigned Opc = (Subtarget->isThumb() && Subtarget->hasThumb2()) 3477 ? ARM::t2MOVTi16 3478 : (Subtarget->hasV6T2Ops() ? ARM::MOVTi16 : 0); 3479 if (!Opc) 3480 break; 3481 SDValue N0 = N->getOperand(0), N1 = N->getOperand(1); 3482 N1C = dyn_cast<ConstantSDNode>(N1); 3483 if (!N1C) 3484 break; 3485 if (N0.getOpcode() == ISD::OR && N0.getNode()->hasOneUse()) { 3486 SDValue N2 = N0.getOperand(1); 3487 ConstantSDNode *N2C = dyn_cast<ConstantSDNode>(N2); 3488 if (!N2C) 3489 break; 3490 unsigned N1CVal = N1C->getZExtValue(); 3491 unsigned N2CVal = N2C->getZExtValue(); 3492 if ((N1CVal & 0xffff0000U) == (N2CVal & 0xffff0000U) && 3493 (N1CVal & 0xffffU) == 0xffffU && 3494 (N2CVal & 0xffffU) == 0x0U) { 3495 SDValue Imm16 = CurDAG->getTargetConstant((N2CVal & 0xFFFF0000U) >> 16, 3496 dl, MVT::i32); 3497 SDValue Ops[] = { N0.getOperand(0), Imm16, 3498 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32) }; 3499 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, VT, Ops)); 3500 return; 3501 } 3502 } 3503 3504 break; 3505 } 3506 case ARMISD::UMAAL: { 3507 unsigned Opc = Subtarget->isThumb() ? ARM::t2UMAAL : ARM::UMAAL; 3508 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), 3509 N->getOperand(2), N->getOperand(3), 3510 getAL(CurDAG, dl), 3511 CurDAG->getRegister(0, MVT::i32) }; 3512 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, MVT::i32, MVT::i32, Ops)); 3513 return; 3514 } 3515 case ARMISD::UMLAL:{ 3516 if (Subtarget->isThumb()) { 3517 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 3518 N->getOperand(3), getAL(CurDAG, dl), 3519 CurDAG->getRegister(0, MVT::i32)}; 3520 ReplaceNode( 3521 N, CurDAG->getMachineNode(ARM::t2UMLAL, dl, MVT::i32, MVT::i32, Ops)); 3522 return; 3523 }else{ 3524 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 3525 N->getOperand(3), getAL(CurDAG, dl), 3526 CurDAG->getRegister(0, MVT::i32), 3527 CurDAG->getRegister(0, MVT::i32) }; 3528 ReplaceNode(N, CurDAG->getMachineNode( 3529 Subtarget->hasV6Ops() ? ARM::UMLAL : ARM::UMLALv5, dl, 3530 MVT::i32, MVT::i32, Ops)); 3531 return; 3532 } 3533 } 3534 case ARMISD::SMLAL:{ 3535 if (Subtarget->isThumb()) { 3536 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 3537 N->getOperand(3), getAL(CurDAG, dl), 3538 CurDAG->getRegister(0, MVT::i32)}; 3539 ReplaceNode( 3540 N, CurDAG->getMachineNode(ARM::t2SMLAL, dl, MVT::i32, MVT::i32, Ops)); 3541 return; 3542 }else{ 3543 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2), 3544 N->getOperand(3), getAL(CurDAG, dl), 3545 CurDAG->getRegister(0, MVT::i32), 3546 CurDAG->getRegister(0, MVT::i32) }; 3547 ReplaceNode(N, CurDAG->getMachineNode( 3548 Subtarget->hasV6Ops() ? ARM::SMLAL : ARM::SMLALv5, dl, 3549 MVT::i32, MVT::i32, Ops)); 3550 return; 3551 } 3552 } 3553 case ARMISD::SUBE: { 3554 if (!Subtarget->hasV6Ops() || !Subtarget->hasDSP()) 3555 break; 3556 // Look for a pattern to match SMMLS 3557 // (sube a, (smul_loHi a, b), (subc 0, (smul_LOhi(a, b)))) 3558 if (N->getOperand(1).getOpcode() != ISD::SMUL_LOHI || 3559 N->getOperand(2).getOpcode() != ARMISD::SUBC || 3560 !SDValue(N, 1).use_empty()) 3561 break; 3562 3563 if (Subtarget->isThumb()) 3564 assert(Subtarget->hasThumb2() && 3565 "This pattern should not be generated for Thumb"); 3566 3567 SDValue SmulLoHi = N->getOperand(1); 3568 SDValue Subc = N->getOperand(2); 3569 auto *Zero = dyn_cast<ConstantSDNode>(Subc.getOperand(0)); 3570 3571 if (!Zero || Zero->getZExtValue() != 0 || 3572 Subc.getOperand(1) != SmulLoHi.getValue(0) || 3573 N->getOperand(1) != SmulLoHi.getValue(1) || 3574 N->getOperand(2) != Subc.getValue(1)) 3575 break; 3576 3577 unsigned Opc = Subtarget->isThumb2() ? ARM::t2SMMLS : ARM::SMMLS; 3578 SDValue Ops[] = { SmulLoHi.getOperand(0), SmulLoHi.getOperand(1), 3579 N->getOperand(0), getAL(CurDAG, dl), 3580 CurDAG->getRegister(0, MVT::i32) }; 3581 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, MVT::i32, Ops)); 3582 return; 3583 } 3584 case ISD::LOAD: { 3585 if (Subtarget->hasMVEIntegerOps() && tryMVEIndexedLoad(N)) 3586 return; 3587 if (Subtarget->isThumb() && Subtarget->hasThumb2()) { 3588 if (tryT2IndexedLoad(N)) 3589 return; 3590 } else if (Subtarget->isThumb()) { 3591 if (tryT1IndexedLoad(N)) 3592 return; 3593 } else if (tryARMIndexedLoad(N)) 3594 return; 3595 // Other cases are autogenerated. 3596 break; 3597 } 3598 case ISD::MLOAD: 3599 if (Subtarget->hasMVEIntegerOps() && tryMVEIndexedLoad(N)) 3600 return; 3601 // Other cases are autogenerated. 3602 break; 3603 case ARMISD::WLS: 3604 case ARMISD::LE: { 3605 SDValue Ops[] = { N->getOperand(1), 3606 N->getOperand(2), 3607 N->getOperand(0) }; 3608 unsigned Opc = N->getOpcode() == ARMISD::WLS ? 3609 ARM::t2WhileLoopStart : ARM::t2LoopEnd; 3610 SDNode *New = CurDAG->getMachineNode(Opc, dl, MVT::Other, Ops); 3611 ReplaceUses(N, New); 3612 CurDAG->RemoveDeadNode(N); 3613 return; 3614 } 3615 case ARMISD::LDRD: { 3616 if (Subtarget->isThumb2()) 3617 break; // TableGen handles isel in this case. 3618 SDValue Base, RegOffset, ImmOffset; 3619 const SDValue &Chain = N->getOperand(0); 3620 const SDValue &Addr = N->getOperand(1); 3621 SelectAddrMode3(Addr, Base, RegOffset, ImmOffset); 3622 if (RegOffset != CurDAG->getRegister(0, MVT::i32)) { 3623 // The register-offset variant of LDRD mandates that the register 3624 // allocated to RegOffset is not reused in any of the remaining operands. 3625 // This restriction is currently not enforced. Therefore emitting this 3626 // variant is explicitly avoided. 3627 Base = Addr; 3628 RegOffset = CurDAG->getRegister(0, MVT::i32); 3629 } 3630 SDValue Ops[] = {Base, RegOffset, ImmOffset, Chain}; 3631 SDNode *New = CurDAG->getMachineNode(ARM::LOADDUAL, dl, 3632 {MVT::Untyped, MVT::Other}, Ops); 3633 SDValue Lo = CurDAG->getTargetExtractSubreg(ARM::gsub_0, dl, MVT::i32, 3634 SDValue(New, 0)); 3635 SDValue Hi = CurDAG->getTargetExtractSubreg(ARM::gsub_1, dl, MVT::i32, 3636 SDValue(New, 0)); 3637 transferMemOperands(N, New); 3638 ReplaceUses(SDValue(N, 0), Lo); 3639 ReplaceUses(SDValue(N, 1), Hi); 3640 ReplaceUses(SDValue(N, 2), SDValue(New, 1)); 3641 CurDAG->RemoveDeadNode(N); 3642 return; 3643 } 3644 case ARMISD::STRD: { 3645 if (Subtarget->isThumb2()) 3646 break; // TableGen handles isel in this case. 3647 SDValue Base, RegOffset, ImmOffset; 3648 const SDValue &Chain = N->getOperand(0); 3649 const SDValue &Addr = N->getOperand(3); 3650 SelectAddrMode3(Addr, Base, RegOffset, ImmOffset); 3651 if (RegOffset != CurDAG->getRegister(0, MVT::i32)) { 3652 // The register-offset variant of STRD mandates that the register 3653 // allocated to RegOffset is not reused in any of the remaining operands. 3654 // This restriction is currently not enforced. Therefore emitting this 3655 // variant is explicitly avoided. 3656 Base = Addr; 3657 RegOffset = CurDAG->getRegister(0, MVT::i32); 3658 } 3659 SDNode *RegPair = 3660 createGPRPairNode(MVT::Untyped, N->getOperand(1), N->getOperand(2)); 3661 SDValue Ops[] = {SDValue(RegPair, 0), Base, RegOffset, ImmOffset, Chain}; 3662 SDNode *New = CurDAG->getMachineNode(ARM::STOREDUAL, dl, MVT::Other, Ops); 3663 transferMemOperands(N, New); 3664 ReplaceUses(SDValue(N, 0), SDValue(New, 0)); 3665 CurDAG->RemoveDeadNode(N); 3666 return; 3667 } 3668 case ARMISD::LOOP_DEC: { 3669 SDValue Ops[] = { N->getOperand(1), 3670 N->getOperand(2), 3671 N->getOperand(0) }; 3672 SDNode *Dec = 3673 CurDAG->getMachineNode(ARM::t2LoopDec, dl, 3674 CurDAG->getVTList(MVT::i32, MVT::Other), Ops); 3675 ReplaceUses(N, Dec); 3676 CurDAG->RemoveDeadNode(N); 3677 return; 3678 } 3679 case ARMISD::BRCOND: { 3680 // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc) 3681 // Emits: (Bcc:void (bb:Other):$dst, (imm:i32):$cc) 3682 // Pattern complexity = 6 cost = 1 size = 0 3683 3684 // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc) 3685 // Emits: (tBcc:void (bb:Other):$dst, (imm:i32):$cc) 3686 // Pattern complexity = 6 cost = 1 size = 0 3687 3688 // Pattern: (ARMbrcond:void (bb:Other):$dst, (imm:i32):$cc) 3689 // Emits: (t2Bcc:void (bb:Other):$dst, (imm:i32):$cc) 3690 // Pattern complexity = 6 cost = 1 size = 0 3691 3692 unsigned Opc = Subtarget->isThumb() ? 3693 ((Subtarget->hasThumb2()) ? ARM::t2Bcc : ARM::tBcc) : ARM::Bcc; 3694 SDValue Chain = N->getOperand(0); 3695 SDValue N1 = N->getOperand(1); 3696 SDValue N2 = N->getOperand(2); 3697 SDValue N3 = N->getOperand(3); 3698 SDValue InFlag = N->getOperand(4); 3699 assert(N1.getOpcode() == ISD::BasicBlock); 3700 assert(N2.getOpcode() == ISD::Constant); 3701 assert(N3.getOpcode() == ISD::Register); 3702 3703 unsigned CC = (unsigned) cast<ConstantSDNode>(N2)->getZExtValue(); 3704 3705 if (InFlag.getOpcode() == ARMISD::CMPZ) { 3706 if (InFlag.getOperand(0).getOpcode() == ISD::INTRINSIC_W_CHAIN) { 3707 SDValue Int = InFlag.getOperand(0); 3708 uint64_t ID = cast<ConstantSDNode>(Int->getOperand(1))->getZExtValue(); 3709 3710 // Handle low-overhead loops. 3711 if (ID == Intrinsic::loop_decrement_reg) { 3712 SDValue Elements = Int.getOperand(2); 3713 SDValue Size = CurDAG->getTargetConstant( 3714 cast<ConstantSDNode>(Int.getOperand(3))->getZExtValue(), dl, 3715 MVT::i32); 3716 3717 SDValue Args[] = { Elements, Size, Int.getOperand(0) }; 3718 SDNode *LoopDec = 3719 CurDAG->getMachineNode(ARM::t2LoopDec, dl, 3720 CurDAG->getVTList(MVT::i32, MVT::Other), 3721 Args); 3722 ReplaceUses(Int.getNode(), LoopDec); 3723 3724 SDValue EndArgs[] = { SDValue(LoopDec, 0), N1, Chain }; 3725 SDNode *LoopEnd = 3726 CurDAG->getMachineNode(ARM::t2LoopEnd, dl, MVT::Other, EndArgs); 3727 3728 ReplaceUses(N, LoopEnd); 3729 CurDAG->RemoveDeadNode(N); 3730 CurDAG->RemoveDeadNode(InFlag.getNode()); 3731 CurDAG->RemoveDeadNode(Int.getNode()); 3732 return; 3733 } 3734 } 3735 3736 bool SwitchEQNEToPLMI; 3737 SelectCMPZ(InFlag.getNode(), SwitchEQNEToPLMI); 3738 InFlag = N->getOperand(4); 3739 3740 if (SwitchEQNEToPLMI) { 3741 switch ((ARMCC::CondCodes)CC) { 3742 default: llvm_unreachable("CMPZ must be either NE or EQ!"); 3743 case ARMCC::NE: 3744 CC = (unsigned)ARMCC::MI; 3745 break; 3746 case ARMCC::EQ: 3747 CC = (unsigned)ARMCC::PL; 3748 break; 3749 } 3750 } 3751 } 3752 3753 SDValue Tmp2 = CurDAG->getTargetConstant(CC, dl, MVT::i32); 3754 SDValue Ops[] = { N1, Tmp2, N3, Chain, InFlag }; 3755 SDNode *ResNode = CurDAG->getMachineNode(Opc, dl, MVT::Other, 3756 MVT::Glue, Ops); 3757 Chain = SDValue(ResNode, 0); 3758 if (N->getNumValues() == 2) { 3759 InFlag = SDValue(ResNode, 1); 3760 ReplaceUses(SDValue(N, 1), InFlag); 3761 } 3762 ReplaceUses(SDValue(N, 0), 3763 SDValue(Chain.getNode(), Chain.getResNo())); 3764 CurDAG->RemoveDeadNode(N); 3765 return; 3766 } 3767 3768 case ARMISD::CMPZ: { 3769 // select (CMPZ X, #-C) -> (CMPZ (ADDS X, #C), #0) 3770 // This allows us to avoid materializing the expensive negative constant. 3771 // The CMPZ #0 is useless and will be peepholed away but we need to keep it 3772 // for its glue output. 3773 SDValue X = N->getOperand(0); 3774 auto *C = dyn_cast<ConstantSDNode>(N->getOperand(1).getNode()); 3775 if (C && C->getSExtValue() < 0 && Subtarget->isThumb()) { 3776 int64_t Addend = -C->getSExtValue(); 3777 3778 SDNode *Add = nullptr; 3779 // ADDS can be better than CMN if the immediate fits in a 3780 // 16-bit ADDS, which means either [0,256) for tADDi8 or [0,8) for tADDi3. 3781 // Outside that range we can just use a CMN which is 32-bit but has a 3782 // 12-bit immediate range. 3783 if (Addend < 1<<8) { 3784 if (Subtarget->isThumb2()) { 3785 SDValue Ops[] = { X, CurDAG->getTargetConstant(Addend, dl, MVT::i32), 3786 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32), 3787 CurDAG->getRegister(0, MVT::i32) }; 3788 Add = CurDAG->getMachineNode(ARM::t2ADDri, dl, MVT::i32, Ops); 3789 } else { 3790 unsigned Opc = (Addend < 1<<3) ? ARM::tADDi3 : ARM::tADDi8; 3791 SDValue Ops[] = {CurDAG->getRegister(ARM::CPSR, MVT::i32), X, 3792 CurDAG->getTargetConstant(Addend, dl, MVT::i32), 3793 getAL(CurDAG, dl), CurDAG->getRegister(0, MVT::i32)}; 3794 Add = CurDAG->getMachineNode(Opc, dl, MVT::i32, Ops); 3795 } 3796 } 3797 if (Add) { 3798 SDValue Ops2[] = {SDValue(Add, 0), CurDAG->getConstant(0, dl, MVT::i32)}; 3799 CurDAG->MorphNodeTo(N, ARMISD::CMPZ, CurDAG->getVTList(MVT::Glue), Ops2); 3800 } 3801 } 3802 // Other cases are autogenerated. 3803 break; 3804 } 3805 3806 case ARMISD::CMOV: { 3807 SDValue InFlag = N->getOperand(4); 3808 3809 if (InFlag.getOpcode() == ARMISD::CMPZ) { 3810 bool SwitchEQNEToPLMI; 3811 SelectCMPZ(InFlag.getNode(), SwitchEQNEToPLMI); 3812 3813 if (SwitchEQNEToPLMI) { 3814 SDValue ARMcc = N->getOperand(2); 3815 ARMCC::CondCodes CC = 3816 (ARMCC::CondCodes)cast<ConstantSDNode>(ARMcc)->getZExtValue(); 3817 3818 switch (CC) { 3819 default: llvm_unreachable("CMPZ must be either NE or EQ!"); 3820 case ARMCC::NE: 3821 CC = ARMCC::MI; 3822 break; 3823 case ARMCC::EQ: 3824 CC = ARMCC::PL; 3825 break; 3826 } 3827 SDValue NewARMcc = CurDAG->getConstant((unsigned)CC, dl, MVT::i32); 3828 SDValue Ops[] = {N->getOperand(0), N->getOperand(1), NewARMcc, 3829 N->getOperand(3), N->getOperand(4)}; 3830 CurDAG->MorphNodeTo(N, ARMISD::CMOV, N->getVTList(), Ops); 3831 } 3832 3833 } 3834 // Other cases are autogenerated. 3835 break; 3836 } 3837 3838 case ARMISD::VZIP: { 3839 unsigned Opc = 0; 3840 EVT VT = N->getValueType(0); 3841 switch (VT.getSimpleVT().SimpleTy) { 3842 default: return; 3843 case MVT::v8i8: Opc = ARM::VZIPd8; break; 3844 case MVT::v4f16: 3845 case MVT::v4i16: Opc = ARM::VZIPd16; break; 3846 case MVT::v2f32: 3847 // vzip.32 Dd, Dm is a pseudo-instruction expanded to vtrn.32 Dd, Dm. 3848 case MVT::v2i32: Opc = ARM::VTRNd32; break; 3849 case MVT::v16i8: Opc = ARM::VZIPq8; break; 3850 case MVT::v8f16: 3851 case MVT::v8i16: Opc = ARM::VZIPq16; break; 3852 case MVT::v4f32: 3853 case MVT::v4i32: Opc = ARM::VZIPq32; break; 3854 } 3855 SDValue Pred = getAL(CurDAG, dl); 3856 SDValue PredReg = CurDAG->getRegister(0, MVT::i32); 3857 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg }; 3858 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, VT, VT, Ops)); 3859 return; 3860 } 3861 case ARMISD::VUZP: { 3862 unsigned Opc = 0; 3863 EVT VT = N->getValueType(0); 3864 switch (VT.getSimpleVT().SimpleTy) { 3865 default: return; 3866 case MVT::v8i8: Opc = ARM::VUZPd8; break; 3867 case MVT::v4f16: 3868 case MVT::v4i16: Opc = ARM::VUZPd16; break; 3869 case MVT::v2f32: 3870 // vuzp.32 Dd, Dm is a pseudo-instruction expanded to vtrn.32 Dd, Dm. 3871 case MVT::v2i32: Opc = ARM::VTRNd32; break; 3872 case MVT::v16i8: Opc = ARM::VUZPq8; break; 3873 case MVT::v8f16: 3874 case MVT::v8i16: Opc = ARM::VUZPq16; break; 3875 case MVT::v4f32: 3876 case MVT::v4i32: Opc = ARM::VUZPq32; break; 3877 } 3878 SDValue Pred = getAL(CurDAG, dl); 3879 SDValue PredReg = CurDAG->getRegister(0, MVT::i32); 3880 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg }; 3881 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, VT, VT, Ops)); 3882 return; 3883 } 3884 case ARMISD::VTRN: { 3885 unsigned Opc = 0; 3886 EVT VT = N->getValueType(0); 3887 switch (VT.getSimpleVT().SimpleTy) { 3888 default: return; 3889 case MVT::v8i8: Opc = ARM::VTRNd8; break; 3890 case MVT::v4f16: 3891 case MVT::v4i16: Opc = ARM::VTRNd16; break; 3892 case MVT::v2f32: 3893 case MVT::v2i32: Opc = ARM::VTRNd32; break; 3894 case MVT::v16i8: Opc = ARM::VTRNq8; break; 3895 case MVT::v8f16: 3896 case MVT::v8i16: Opc = ARM::VTRNq16; break; 3897 case MVT::v4f32: 3898 case MVT::v4i32: Opc = ARM::VTRNq32; break; 3899 } 3900 SDValue Pred = getAL(CurDAG, dl); 3901 SDValue PredReg = CurDAG->getRegister(0, MVT::i32); 3902 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), Pred, PredReg }; 3903 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, VT, VT, Ops)); 3904 return; 3905 } 3906 case ARMISD::BUILD_VECTOR: { 3907 EVT VecVT = N->getValueType(0); 3908 EVT EltVT = VecVT.getVectorElementType(); 3909 unsigned NumElts = VecVT.getVectorNumElements(); 3910 if (EltVT == MVT::f64) { 3911 assert(NumElts == 2 && "unexpected type for BUILD_VECTOR"); 3912 ReplaceNode( 3913 N, createDRegPairNode(VecVT, N->getOperand(0), N->getOperand(1))); 3914 return; 3915 } 3916 assert(EltVT == MVT::f32 && "unexpected type for BUILD_VECTOR"); 3917 if (NumElts == 2) { 3918 ReplaceNode( 3919 N, createSRegPairNode(VecVT, N->getOperand(0), N->getOperand(1))); 3920 return; 3921 } 3922 assert(NumElts == 4 && "unexpected type for BUILD_VECTOR"); 3923 ReplaceNode(N, 3924 createQuadSRegsNode(VecVT, N->getOperand(0), N->getOperand(1), 3925 N->getOperand(2), N->getOperand(3))); 3926 return; 3927 } 3928 3929 case ARMISD::VLD1DUP: { 3930 static const uint16_t DOpcodes[] = { ARM::VLD1DUPd8, ARM::VLD1DUPd16, 3931 ARM::VLD1DUPd32 }; 3932 static const uint16_t QOpcodes[] = { ARM::VLD1DUPq8, ARM::VLD1DUPq16, 3933 ARM::VLD1DUPq32 }; 3934 SelectVLDDup(N, /* IsIntrinsic= */ false, false, 1, DOpcodes, QOpcodes); 3935 return; 3936 } 3937 3938 case ARMISD::VLD2DUP: { 3939 static const uint16_t Opcodes[] = { ARM::VLD2DUPd8, ARM::VLD2DUPd16, 3940 ARM::VLD2DUPd32 }; 3941 SelectVLDDup(N, /* IsIntrinsic= */ false, false, 2, Opcodes); 3942 return; 3943 } 3944 3945 case ARMISD::VLD3DUP: { 3946 static const uint16_t Opcodes[] = { ARM::VLD3DUPd8Pseudo, 3947 ARM::VLD3DUPd16Pseudo, 3948 ARM::VLD3DUPd32Pseudo }; 3949 SelectVLDDup(N, /* IsIntrinsic= */ false, false, 3, Opcodes); 3950 return; 3951 } 3952 3953 case ARMISD::VLD4DUP: { 3954 static const uint16_t Opcodes[] = { ARM::VLD4DUPd8Pseudo, 3955 ARM::VLD4DUPd16Pseudo, 3956 ARM::VLD4DUPd32Pseudo }; 3957 SelectVLDDup(N, /* IsIntrinsic= */ false, false, 4, Opcodes); 3958 return; 3959 } 3960 3961 case ARMISD::VLD1DUP_UPD: { 3962 static const uint16_t DOpcodes[] = { ARM::VLD1DUPd8wb_fixed, 3963 ARM::VLD1DUPd16wb_fixed, 3964 ARM::VLD1DUPd32wb_fixed }; 3965 static const uint16_t QOpcodes[] = { ARM::VLD1DUPq8wb_fixed, 3966 ARM::VLD1DUPq16wb_fixed, 3967 ARM::VLD1DUPq32wb_fixed }; 3968 SelectVLDDup(N, /* IsIntrinsic= */ false, true, 1, DOpcodes, QOpcodes); 3969 return; 3970 } 3971 3972 case ARMISD::VLD2DUP_UPD: { 3973 static const uint16_t Opcodes[] = { ARM::VLD2DUPd8wb_fixed, 3974 ARM::VLD2DUPd16wb_fixed, 3975 ARM::VLD2DUPd32wb_fixed }; 3976 SelectVLDDup(N, /* IsIntrinsic= */ false, true, 2, Opcodes); 3977 return; 3978 } 3979 3980 case ARMISD::VLD3DUP_UPD: { 3981 static const uint16_t Opcodes[] = { ARM::VLD3DUPd8Pseudo_UPD, 3982 ARM::VLD3DUPd16Pseudo_UPD, 3983 ARM::VLD3DUPd32Pseudo_UPD }; 3984 SelectVLDDup(N, /* IsIntrinsic= */ false, true, 3, Opcodes); 3985 return; 3986 } 3987 3988 case ARMISD::VLD4DUP_UPD: { 3989 static const uint16_t Opcodes[] = { ARM::VLD4DUPd8Pseudo_UPD, 3990 ARM::VLD4DUPd16Pseudo_UPD, 3991 ARM::VLD4DUPd32Pseudo_UPD }; 3992 SelectVLDDup(N, /* IsIntrinsic= */ false, true, 4, Opcodes); 3993 return; 3994 } 3995 3996 case ARMISD::VLD1_UPD: { 3997 static const uint16_t DOpcodes[] = { ARM::VLD1d8wb_fixed, 3998 ARM::VLD1d16wb_fixed, 3999 ARM::VLD1d32wb_fixed, 4000 ARM::VLD1d64wb_fixed }; 4001 static const uint16_t QOpcodes[] = { ARM::VLD1q8wb_fixed, 4002 ARM::VLD1q16wb_fixed, 4003 ARM::VLD1q32wb_fixed, 4004 ARM::VLD1q64wb_fixed }; 4005 SelectVLD(N, true, 1, DOpcodes, QOpcodes, nullptr); 4006 return; 4007 } 4008 4009 case ARMISD::VLD2_UPD: { 4010 if (Subtarget->hasNEON()) { 4011 static const uint16_t DOpcodes[] = { 4012 ARM::VLD2d8wb_fixed, ARM::VLD2d16wb_fixed, ARM::VLD2d32wb_fixed, 4013 ARM::VLD1q64wb_fixed}; 4014 static const uint16_t QOpcodes[] = {ARM::VLD2q8PseudoWB_fixed, 4015 ARM::VLD2q16PseudoWB_fixed, 4016 ARM::VLD2q32PseudoWB_fixed}; 4017 SelectVLD(N, true, 2, DOpcodes, QOpcodes, nullptr); 4018 } else { 4019 static const uint16_t Opcodes8[] = {ARM::MVE_VLD20_8, 4020 ARM::MVE_VLD21_8_wb}; 4021 static const uint16_t Opcodes16[] = {ARM::MVE_VLD20_16, 4022 ARM::MVE_VLD21_16_wb}; 4023 static const uint16_t Opcodes32[] = {ARM::MVE_VLD20_32, 4024 ARM::MVE_VLD21_32_wb}; 4025 static const uint16_t *const Opcodes[] = {Opcodes8, Opcodes16, Opcodes32}; 4026 SelectMVE_VLD(N, 2, Opcodes, true); 4027 } 4028 return; 4029 } 4030 4031 case ARMISD::VLD3_UPD: { 4032 static const uint16_t DOpcodes[] = { ARM::VLD3d8Pseudo_UPD, 4033 ARM::VLD3d16Pseudo_UPD, 4034 ARM::VLD3d32Pseudo_UPD, 4035 ARM::VLD1d64TPseudoWB_fixed}; 4036 static const uint16_t QOpcodes0[] = { ARM::VLD3q8Pseudo_UPD, 4037 ARM::VLD3q16Pseudo_UPD, 4038 ARM::VLD3q32Pseudo_UPD }; 4039 static const uint16_t QOpcodes1[] = { ARM::VLD3q8oddPseudo_UPD, 4040 ARM::VLD3q16oddPseudo_UPD, 4041 ARM::VLD3q32oddPseudo_UPD }; 4042 SelectVLD(N, true, 3, DOpcodes, QOpcodes0, QOpcodes1); 4043 return; 4044 } 4045 4046 case ARMISD::VLD4_UPD: { 4047 if (Subtarget->hasNEON()) { 4048 static const uint16_t DOpcodes[] = { 4049 ARM::VLD4d8Pseudo_UPD, ARM::VLD4d16Pseudo_UPD, ARM::VLD4d32Pseudo_UPD, 4050 ARM::VLD1d64QPseudoWB_fixed}; 4051 static const uint16_t QOpcodes0[] = {ARM::VLD4q8Pseudo_UPD, 4052 ARM::VLD4q16Pseudo_UPD, 4053 ARM::VLD4q32Pseudo_UPD}; 4054 static const uint16_t QOpcodes1[] = {ARM::VLD4q8oddPseudo_UPD, 4055 ARM::VLD4q16oddPseudo_UPD, 4056 ARM::VLD4q32oddPseudo_UPD}; 4057 SelectVLD(N, true, 4, DOpcodes, QOpcodes0, QOpcodes1); 4058 } else { 4059 static const uint16_t Opcodes8[] = {ARM::MVE_VLD40_8, ARM::MVE_VLD41_8, 4060 ARM::MVE_VLD42_8, 4061 ARM::MVE_VLD43_8_wb}; 4062 static const uint16_t Opcodes16[] = {ARM::MVE_VLD40_16, ARM::MVE_VLD41_16, 4063 ARM::MVE_VLD42_16, 4064 ARM::MVE_VLD43_16_wb}; 4065 static const uint16_t Opcodes32[] = {ARM::MVE_VLD40_32, ARM::MVE_VLD41_32, 4066 ARM::MVE_VLD42_32, 4067 ARM::MVE_VLD43_32_wb}; 4068 static const uint16_t *const Opcodes[] = {Opcodes8, Opcodes16, Opcodes32}; 4069 SelectMVE_VLD(N, 4, Opcodes, true); 4070 } 4071 return; 4072 } 4073 4074 case ARMISD::VLD2LN_UPD: { 4075 static const uint16_t DOpcodes[] = { ARM::VLD2LNd8Pseudo_UPD, 4076 ARM::VLD2LNd16Pseudo_UPD, 4077 ARM::VLD2LNd32Pseudo_UPD }; 4078 static const uint16_t QOpcodes[] = { ARM::VLD2LNq16Pseudo_UPD, 4079 ARM::VLD2LNq32Pseudo_UPD }; 4080 SelectVLDSTLane(N, true, true, 2, DOpcodes, QOpcodes); 4081 return; 4082 } 4083 4084 case ARMISD::VLD3LN_UPD: { 4085 static const uint16_t DOpcodes[] = { ARM::VLD3LNd8Pseudo_UPD, 4086 ARM::VLD3LNd16Pseudo_UPD, 4087 ARM::VLD3LNd32Pseudo_UPD }; 4088 static const uint16_t QOpcodes[] = { ARM::VLD3LNq16Pseudo_UPD, 4089 ARM::VLD3LNq32Pseudo_UPD }; 4090 SelectVLDSTLane(N, true, true, 3, DOpcodes, QOpcodes); 4091 return; 4092 } 4093 4094 case ARMISD::VLD4LN_UPD: { 4095 static const uint16_t DOpcodes[] = { ARM::VLD4LNd8Pseudo_UPD, 4096 ARM::VLD4LNd16Pseudo_UPD, 4097 ARM::VLD4LNd32Pseudo_UPD }; 4098 static const uint16_t QOpcodes[] = { ARM::VLD4LNq16Pseudo_UPD, 4099 ARM::VLD4LNq32Pseudo_UPD }; 4100 SelectVLDSTLane(N, true, true, 4, DOpcodes, QOpcodes); 4101 return; 4102 } 4103 4104 case ARMISD::VST1_UPD: { 4105 static const uint16_t DOpcodes[] = { ARM::VST1d8wb_fixed, 4106 ARM::VST1d16wb_fixed, 4107 ARM::VST1d32wb_fixed, 4108 ARM::VST1d64wb_fixed }; 4109 static const uint16_t QOpcodes[] = { ARM::VST1q8wb_fixed, 4110 ARM::VST1q16wb_fixed, 4111 ARM::VST1q32wb_fixed, 4112 ARM::VST1q64wb_fixed }; 4113 SelectVST(N, true, 1, DOpcodes, QOpcodes, nullptr); 4114 return; 4115 } 4116 4117 case ARMISD::VST2_UPD: { 4118 if (Subtarget->hasNEON()) { 4119 static const uint16_t DOpcodes[] = { 4120 ARM::VST2d8wb_fixed, ARM::VST2d16wb_fixed, ARM::VST2d32wb_fixed, 4121 ARM::VST1q64wb_fixed}; 4122 static const uint16_t QOpcodes[] = {ARM::VST2q8PseudoWB_fixed, 4123 ARM::VST2q16PseudoWB_fixed, 4124 ARM::VST2q32PseudoWB_fixed}; 4125 SelectVST(N, true, 2, DOpcodes, QOpcodes, nullptr); 4126 return; 4127 } 4128 break; 4129 } 4130 4131 case ARMISD::VST3_UPD: { 4132 static const uint16_t DOpcodes[] = { ARM::VST3d8Pseudo_UPD, 4133 ARM::VST3d16Pseudo_UPD, 4134 ARM::VST3d32Pseudo_UPD, 4135 ARM::VST1d64TPseudoWB_fixed}; 4136 static const uint16_t QOpcodes0[] = { ARM::VST3q8Pseudo_UPD, 4137 ARM::VST3q16Pseudo_UPD, 4138 ARM::VST3q32Pseudo_UPD }; 4139 static const uint16_t QOpcodes1[] = { ARM::VST3q8oddPseudo_UPD, 4140 ARM::VST3q16oddPseudo_UPD, 4141 ARM::VST3q32oddPseudo_UPD }; 4142 SelectVST(N, true, 3, DOpcodes, QOpcodes0, QOpcodes1); 4143 return; 4144 } 4145 4146 case ARMISD::VST4_UPD: { 4147 if (Subtarget->hasNEON()) { 4148 static const uint16_t DOpcodes[] = { 4149 ARM::VST4d8Pseudo_UPD, ARM::VST4d16Pseudo_UPD, ARM::VST4d32Pseudo_UPD, 4150 ARM::VST1d64QPseudoWB_fixed}; 4151 static const uint16_t QOpcodes0[] = {ARM::VST4q8Pseudo_UPD, 4152 ARM::VST4q16Pseudo_UPD, 4153 ARM::VST4q32Pseudo_UPD}; 4154 static const uint16_t QOpcodes1[] = {ARM::VST4q8oddPseudo_UPD, 4155 ARM::VST4q16oddPseudo_UPD, 4156 ARM::VST4q32oddPseudo_UPD}; 4157 SelectVST(N, true, 4, DOpcodes, QOpcodes0, QOpcodes1); 4158 return; 4159 } 4160 break; 4161 } 4162 4163 case ARMISD::VST2LN_UPD: { 4164 static const uint16_t DOpcodes[] = { ARM::VST2LNd8Pseudo_UPD, 4165 ARM::VST2LNd16Pseudo_UPD, 4166 ARM::VST2LNd32Pseudo_UPD }; 4167 static const uint16_t QOpcodes[] = { ARM::VST2LNq16Pseudo_UPD, 4168 ARM::VST2LNq32Pseudo_UPD }; 4169 SelectVLDSTLane(N, false, true, 2, DOpcodes, QOpcodes); 4170 return; 4171 } 4172 4173 case ARMISD::VST3LN_UPD: { 4174 static const uint16_t DOpcodes[] = { ARM::VST3LNd8Pseudo_UPD, 4175 ARM::VST3LNd16Pseudo_UPD, 4176 ARM::VST3LNd32Pseudo_UPD }; 4177 static const uint16_t QOpcodes[] = { ARM::VST3LNq16Pseudo_UPD, 4178 ARM::VST3LNq32Pseudo_UPD }; 4179 SelectVLDSTLane(N, false, true, 3, DOpcodes, QOpcodes); 4180 return; 4181 } 4182 4183 case ARMISD::VST4LN_UPD: { 4184 static const uint16_t DOpcodes[] = { ARM::VST4LNd8Pseudo_UPD, 4185 ARM::VST4LNd16Pseudo_UPD, 4186 ARM::VST4LNd32Pseudo_UPD }; 4187 static const uint16_t QOpcodes[] = { ARM::VST4LNq16Pseudo_UPD, 4188 ARM::VST4LNq32Pseudo_UPD }; 4189 SelectVLDSTLane(N, false, true, 4, DOpcodes, QOpcodes); 4190 return; 4191 } 4192 4193 case ISD::INTRINSIC_VOID: 4194 case ISD::INTRINSIC_W_CHAIN: { 4195 unsigned IntNo = cast<ConstantSDNode>(N->getOperand(1))->getZExtValue(); 4196 switch (IntNo) { 4197 default: 4198 break; 4199 4200 case Intrinsic::arm_mrrc: 4201 case Intrinsic::arm_mrrc2: { 4202 SDLoc dl(N); 4203 SDValue Chain = N->getOperand(0); 4204 unsigned Opc; 4205 4206 if (Subtarget->isThumb()) 4207 Opc = (IntNo == Intrinsic::arm_mrrc ? ARM::t2MRRC : ARM::t2MRRC2); 4208 else 4209 Opc = (IntNo == Intrinsic::arm_mrrc ? ARM::MRRC : ARM::MRRC2); 4210 4211 SmallVector<SDValue, 5> Ops; 4212 Ops.push_back(getI32Imm(cast<ConstantSDNode>(N->getOperand(2))->getZExtValue(), dl)); /* coproc */ 4213 Ops.push_back(getI32Imm(cast<ConstantSDNode>(N->getOperand(3))->getZExtValue(), dl)); /* opc */ 4214 Ops.push_back(getI32Imm(cast<ConstantSDNode>(N->getOperand(4))->getZExtValue(), dl)); /* CRm */ 4215 4216 // The mrrc2 instruction in ARM doesn't allow predicates, the top 4 bits of the encoded 4217 // instruction will always be '1111' but it is possible in assembly language to specify 4218 // AL as a predicate to mrrc2 but it doesn't make any difference to the encoded instruction. 4219 if (Opc != ARM::MRRC2) { 4220 Ops.push_back(getAL(CurDAG, dl)); 4221 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 4222 } 4223 4224 Ops.push_back(Chain); 4225 4226 // Writes to two registers. 4227 const EVT RetType[] = {MVT::i32, MVT::i32, MVT::Other}; 4228 4229 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, RetType, Ops)); 4230 return; 4231 } 4232 case Intrinsic::arm_ldaexd: 4233 case Intrinsic::arm_ldrexd: { 4234 SDLoc dl(N); 4235 SDValue Chain = N->getOperand(0); 4236 SDValue MemAddr = N->getOperand(2); 4237 bool isThumb = Subtarget->isThumb() && Subtarget->hasV8MBaselineOps(); 4238 4239 bool IsAcquire = IntNo == Intrinsic::arm_ldaexd; 4240 unsigned NewOpc = isThumb ? (IsAcquire ? ARM::t2LDAEXD : ARM::t2LDREXD) 4241 : (IsAcquire ? ARM::LDAEXD : ARM::LDREXD); 4242 4243 // arm_ldrexd returns a i64 value in {i32, i32} 4244 std::vector<EVT> ResTys; 4245 if (isThumb) { 4246 ResTys.push_back(MVT::i32); 4247 ResTys.push_back(MVT::i32); 4248 } else 4249 ResTys.push_back(MVT::Untyped); 4250 ResTys.push_back(MVT::Other); 4251 4252 // Place arguments in the right order. 4253 SDValue Ops[] = {MemAddr, getAL(CurDAG, dl), 4254 CurDAG->getRegister(0, MVT::i32), Chain}; 4255 SDNode *Ld = CurDAG->getMachineNode(NewOpc, dl, ResTys, Ops); 4256 // Transfer memoperands. 4257 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 4258 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Ld), {MemOp}); 4259 4260 // Remap uses. 4261 SDValue OutChain = isThumb ? SDValue(Ld, 2) : SDValue(Ld, 1); 4262 if (!SDValue(N, 0).use_empty()) { 4263 SDValue Result; 4264 if (isThumb) 4265 Result = SDValue(Ld, 0); 4266 else { 4267 SDValue SubRegIdx = 4268 CurDAG->getTargetConstant(ARM::gsub_0, dl, MVT::i32); 4269 SDNode *ResNode = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, 4270 dl, MVT::i32, SDValue(Ld, 0), SubRegIdx); 4271 Result = SDValue(ResNode,0); 4272 } 4273 ReplaceUses(SDValue(N, 0), Result); 4274 } 4275 if (!SDValue(N, 1).use_empty()) { 4276 SDValue Result; 4277 if (isThumb) 4278 Result = SDValue(Ld, 1); 4279 else { 4280 SDValue SubRegIdx = 4281 CurDAG->getTargetConstant(ARM::gsub_1, dl, MVT::i32); 4282 SDNode *ResNode = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, 4283 dl, MVT::i32, SDValue(Ld, 0), SubRegIdx); 4284 Result = SDValue(ResNode,0); 4285 } 4286 ReplaceUses(SDValue(N, 1), Result); 4287 } 4288 ReplaceUses(SDValue(N, 2), OutChain); 4289 CurDAG->RemoveDeadNode(N); 4290 return; 4291 } 4292 case Intrinsic::arm_stlexd: 4293 case Intrinsic::arm_strexd: { 4294 SDLoc dl(N); 4295 SDValue Chain = N->getOperand(0); 4296 SDValue Val0 = N->getOperand(2); 4297 SDValue Val1 = N->getOperand(3); 4298 SDValue MemAddr = N->getOperand(4); 4299 4300 // Store exclusive double return a i32 value which is the return status 4301 // of the issued store. 4302 const EVT ResTys[] = {MVT::i32, MVT::Other}; 4303 4304 bool isThumb = Subtarget->isThumb() && Subtarget->hasThumb2(); 4305 // Place arguments in the right order. 4306 SmallVector<SDValue, 7> Ops; 4307 if (isThumb) { 4308 Ops.push_back(Val0); 4309 Ops.push_back(Val1); 4310 } else 4311 // arm_strexd uses GPRPair. 4312 Ops.push_back(SDValue(createGPRPairNode(MVT::Untyped, Val0, Val1), 0)); 4313 Ops.push_back(MemAddr); 4314 Ops.push_back(getAL(CurDAG, dl)); 4315 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 4316 Ops.push_back(Chain); 4317 4318 bool IsRelease = IntNo == Intrinsic::arm_stlexd; 4319 unsigned NewOpc = isThumb ? (IsRelease ? ARM::t2STLEXD : ARM::t2STREXD) 4320 : (IsRelease ? ARM::STLEXD : ARM::STREXD); 4321 4322 SDNode *St = CurDAG->getMachineNode(NewOpc, dl, ResTys, Ops); 4323 // Transfer memoperands. 4324 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand(); 4325 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp}); 4326 4327 ReplaceNode(N, St); 4328 return; 4329 } 4330 4331 case Intrinsic::arm_neon_vld1: { 4332 static const uint16_t DOpcodes[] = { ARM::VLD1d8, ARM::VLD1d16, 4333 ARM::VLD1d32, ARM::VLD1d64 }; 4334 static const uint16_t QOpcodes[] = { ARM::VLD1q8, ARM::VLD1q16, 4335 ARM::VLD1q32, ARM::VLD1q64}; 4336 SelectVLD(N, false, 1, DOpcodes, QOpcodes, nullptr); 4337 return; 4338 } 4339 4340 case Intrinsic::arm_neon_vld1x2: { 4341 static const uint16_t DOpcodes[] = { ARM::VLD1q8, ARM::VLD1q16, 4342 ARM::VLD1q32, ARM::VLD1q64 }; 4343 static const uint16_t QOpcodes[] = { ARM::VLD1d8QPseudo, 4344 ARM::VLD1d16QPseudo, 4345 ARM::VLD1d32QPseudo, 4346 ARM::VLD1d64QPseudo }; 4347 SelectVLD(N, false, 2, DOpcodes, QOpcodes, nullptr); 4348 return; 4349 } 4350 4351 case Intrinsic::arm_neon_vld1x3: { 4352 static const uint16_t DOpcodes[] = { ARM::VLD1d8TPseudo, 4353 ARM::VLD1d16TPseudo, 4354 ARM::VLD1d32TPseudo, 4355 ARM::VLD1d64TPseudo }; 4356 static const uint16_t QOpcodes0[] = { ARM::VLD1q8LowTPseudo_UPD, 4357 ARM::VLD1q16LowTPseudo_UPD, 4358 ARM::VLD1q32LowTPseudo_UPD, 4359 ARM::VLD1q64LowTPseudo_UPD }; 4360 static const uint16_t QOpcodes1[] = { ARM::VLD1q8HighTPseudo, 4361 ARM::VLD1q16HighTPseudo, 4362 ARM::VLD1q32HighTPseudo, 4363 ARM::VLD1q64HighTPseudo }; 4364 SelectVLD(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1); 4365 return; 4366 } 4367 4368 case Intrinsic::arm_neon_vld1x4: { 4369 static const uint16_t DOpcodes[] = { ARM::VLD1d8QPseudo, 4370 ARM::VLD1d16QPseudo, 4371 ARM::VLD1d32QPseudo, 4372 ARM::VLD1d64QPseudo }; 4373 static const uint16_t QOpcodes0[] = { ARM::VLD1q8LowQPseudo_UPD, 4374 ARM::VLD1q16LowQPseudo_UPD, 4375 ARM::VLD1q32LowQPseudo_UPD, 4376 ARM::VLD1q64LowQPseudo_UPD }; 4377 static const uint16_t QOpcodes1[] = { ARM::VLD1q8HighQPseudo, 4378 ARM::VLD1q16HighQPseudo, 4379 ARM::VLD1q32HighQPseudo, 4380 ARM::VLD1q64HighQPseudo }; 4381 SelectVLD(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1); 4382 return; 4383 } 4384 4385 case Intrinsic::arm_neon_vld2: { 4386 static const uint16_t DOpcodes[] = { ARM::VLD2d8, ARM::VLD2d16, 4387 ARM::VLD2d32, ARM::VLD1q64 }; 4388 static const uint16_t QOpcodes[] = { ARM::VLD2q8Pseudo, ARM::VLD2q16Pseudo, 4389 ARM::VLD2q32Pseudo }; 4390 SelectVLD(N, false, 2, DOpcodes, QOpcodes, nullptr); 4391 return; 4392 } 4393 4394 case Intrinsic::arm_neon_vld3: { 4395 static const uint16_t DOpcodes[] = { ARM::VLD3d8Pseudo, 4396 ARM::VLD3d16Pseudo, 4397 ARM::VLD3d32Pseudo, 4398 ARM::VLD1d64TPseudo }; 4399 static const uint16_t QOpcodes0[] = { ARM::VLD3q8Pseudo_UPD, 4400 ARM::VLD3q16Pseudo_UPD, 4401 ARM::VLD3q32Pseudo_UPD }; 4402 static const uint16_t QOpcodes1[] = { ARM::VLD3q8oddPseudo, 4403 ARM::VLD3q16oddPseudo, 4404 ARM::VLD3q32oddPseudo }; 4405 SelectVLD(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1); 4406 return; 4407 } 4408 4409 case Intrinsic::arm_neon_vld4: { 4410 static const uint16_t DOpcodes[] = { ARM::VLD4d8Pseudo, 4411 ARM::VLD4d16Pseudo, 4412 ARM::VLD4d32Pseudo, 4413 ARM::VLD1d64QPseudo }; 4414 static const uint16_t QOpcodes0[] = { ARM::VLD4q8Pseudo_UPD, 4415 ARM::VLD4q16Pseudo_UPD, 4416 ARM::VLD4q32Pseudo_UPD }; 4417 static const uint16_t QOpcodes1[] = { ARM::VLD4q8oddPseudo, 4418 ARM::VLD4q16oddPseudo, 4419 ARM::VLD4q32oddPseudo }; 4420 SelectVLD(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1); 4421 return; 4422 } 4423 4424 case Intrinsic::arm_neon_vld2dup: { 4425 static const uint16_t DOpcodes[] = { ARM::VLD2DUPd8, ARM::VLD2DUPd16, 4426 ARM::VLD2DUPd32, ARM::VLD1q64 }; 4427 static const uint16_t QOpcodes0[] = { ARM::VLD2DUPq8EvenPseudo, 4428 ARM::VLD2DUPq16EvenPseudo, 4429 ARM::VLD2DUPq32EvenPseudo }; 4430 static const uint16_t QOpcodes1[] = { ARM::VLD2DUPq8OddPseudo, 4431 ARM::VLD2DUPq16OddPseudo, 4432 ARM::VLD2DUPq32OddPseudo }; 4433 SelectVLDDup(N, /* IsIntrinsic= */ true, false, 2, 4434 DOpcodes, QOpcodes0, QOpcodes1); 4435 return; 4436 } 4437 4438 case Intrinsic::arm_neon_vld3dup: { 4439 static const uint16_t DOpcodes[] = { ARM::VLD3DUPd8Pseudo, 4440 ARM::VLD3DUPd16Pseudo, 4441 ARM::VLD3DUPd32Pseudo, 4442 ARM::VLD1d64TPseudo }; 4443 static const uint16_t QOpcodes0[] = { ARM::VLD3DUPq8EvenPseudo, 4444 ARM::VLD3DUPq16EvenPseudo, 4445 ARM::VLD3DUPq32EvenPseudo }; 4446 static const uint16_t QOpcodes1[] = { ARM::VLD3DUPq8OddPseudo, 4447 ARM::VLD3DUPq16OddPseudo, 4448 ARM::VLD3DUPq32OddPseudo }; 4449 SelectVLDDup(N, /* IsIntrinsic= */ true, false, 3, 4450 DOpcodes, QOpcodes0, QOpcodes1); 4451 return; 4452 } 4453 4454 case Intrinsic::arm_neon_vld4dup: { 4455 static const uint16_t DOpcodes[] = { ARM::VLD4DUPd8Pseudo, 4456 ARM::VLD4DUPd16Pseudo, 4457 ARM::VLD4DUPd32Pseudo, 4458 ARM::VLD1d64QPseudo }; 4459 static const uint16_t QOpcodes0[] = { ARM::VLD4DUPq8EvenPseudo, 4460 ARM::VLD4DUPq16EvenPseudo, 4461 ARM::VLD4DUPq32EvenPseudo }; 4462 static const uint16_t QOpcodes1[] = { ARM::VLD4DUPq8OddPseudo, 4463 ARM::VLD4DUPq16OddPseudo, 4464 ARM::VLD4DUPq32OddPseudo }; 4465 SelectVLDDup(N, /* IsIntrinsic= */ true, false, 4, 4466 DOpcodes, QOpcodes0, QOpcodes1); 4467 return; 4468 } 4469 4470 case Intrinsic::arm_neon_vld2lane: { 4471 static const uint16_t DOpcodes[] = { ARM::VLD2LNd8Pseudo, 4472 ARM::VLD2LNd16Pseudo, 4473 ARM::VLD2LNd32Pseudo }; 4474 static const uint16_t QOpcodes[] = { ARM::VLD2LNq16Pseudo, 4475 ARM::VLD2LNq32Pseudo }; 4476 SelectVLDSTLane(N, true, false, 2, DOpcodes, QOpcodes); 4477 return; 4478 } 4479 4480 case Intrinsic::arm_neon_vld3lane: { 4481 static const uint16_t DOpcodes[] = { ARM::VLD3LNd8Pseudo, 4482 ARM::VLD3LNd16Pseudo, 4483 ARM::VLD3LNd32Pseudo }; 4484 static const uint16_t QOpcodes[] = { ARM::VLD3LNq16Pseudo, 4485 ARM::VLD3LNq32Pseudo }; 4486 SelectVLDSTLane(N, true, false, 3, DOpcodes, QOpcodes); 4487 return; 4488 } 4489 4490 case Intrinsic::arm_neon_vld4lane: { 4491 static const uint16_t DOpcodes[] = { ARM::VLD4LNd8Pseudo, 4492 ARM::VLD4LNd16Pseudo, 4493 ARM::VLD4LNd32Pseudo }; 4494 static const uint16_t QOpcodes[] = { ARM::VLD4LNq16Pseudo, 4495 ARM::VLD4LNq32Pseudo }; 4496 SelectVLDSTLane(N, true, false, 4, DOpcodes, QOpcodes); 4497 return; 4498 } 4499 4500 case Intrinsic::arm_neon_vst1: { 4501 static const uint16_t DOpcodes[] = { ARM::VST1d8, ARM::VST1d16, 4502 ARM::VST1d32, ARM::VST1d64 }; 4503 static const uint16_t QOpcodes[] = { ARM::VST1q8, ARM::VST1q16, 4504 ARM::VST1q32, ARM::VST1q64 }; 4505 SelectVST(N, false, 1, DOpcodes, QOpcodes, nullptr); 4506 return; 4507 } 4508 4509 case Intrinsic::arm_neon_vst1x2: { 4510 static const uint16_t DOpcodes[] = { ARM::VST1q8, ARM::VST1q16, 4511 ARM::VST1q32, ARM::VST1q64 }; 4512 static const uint16_t QOpcodes[] = { ARM::VST1d8QPseudo, 4513 ARM::VST1d16QPseudo, 4514 ARM::VST1d32QPseudo, 4515 ARM::VST1d64QPseudo }; 4516 SelectVST(N, false, 2, DOpcodes, QOpcodes, nullptr); 4517 return; 4518 } 4519 4520 case Intrinsic::arm_neon_vst1x3: { 4521 static const uint16_t DOpcodes[] = { ARM::VST1d8TPseudo, 4522 ARM::VST1d16TPseudo, 4523 ARM::VST1d32TPseudo, 4524 ARM::VST1d64TPseudo }; 4525 static const uint16_t QOpcodes0[] = { ARM::VST1q8LowTPseudo_UPD, 4526 ARM::VST1q16LowTPseudo_UPD, 4527 ARM::VST1q32LowTPseudo_UPD, 4528 ARM::VST1q64LowTPseudo_UPD }; 4529 static const uint16_t QOpcodes1[] = { ARM::VST1q8HighTPseudo, 4530 ARM::VST1q16HighTPseudo, 4531 ARM::VST1q32HighTPseudo, 4532 ARM::VST1q64HighTPseudo }; 4533 SelectVST(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1); 4534 return; 4535 } 4536 4537 case Intrinsic::arm_neon_vst1x4: { 4538 static const uint16_t DOpcodes[] = { ARM::VST1d8QPseudo, 4539 ARM::VST1d16QPseudo, 4540 ARM::VST1d32QPseudo, 4541 ARM::VST1d64QPseudo }; 4542 static const uint16_t QOpcodes0[] = { ARM::VST1q8LowQPseudo_UPD, 4543 ARM::VST1q16LowQPseudo_UPD, 4544 ARM::VST1q32LowQPseudo_UPD, 4545 ARM::VST1q64LowQPseudo_UPD }; 4546 static const uint16_t QOpcodes1[] = { ARM::VST1q8HighQPseudo, 4547 ARM::VST1q16HighQPseudo, 4548 ARM::VST1q32HighQPseudo, 4549 ARM::VST1q64HighQPseudo }; 4550 SelectVST(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1); 4551 return; 4552 } 4553 4554 case Intrinsic::arm_neon_vst2: { 4555 static const uint16_t DOpcodes[] = { ARM::VST2d8, ARM::VST2d16, 4556 ARM::VST2d32, ARM::VST1q64 }; 4557 static const uint16_t QOpcodes[] = { ARM::VST2q8Pseudo, ARM::VST2q16Pseudo, 4558 ARM::VST2q32Pseudo }; 4559 SelectVST(N, false, 2, DOpcodes, QOpcodes, nullptr); 4560 return; 4561 } 4562 4563 case Intrinsic::arm_neon_vst3: { 4564 static const uint16_t DOpcodes[] = { ARM::VST3d8Pseudo, 4565 ARM::VST3d16Pseudo, 4566 ARM::VST3d32Pseudo, 4567 ARM::VST1d64TPseudo }; 4568 static const uint16_t QOpcodes0[] = { ARM::VST3q8Pseudo_UPD, 4569 ARM::VST3q16Pseudo_UPD, 4570 ARM::VST3q32Pseudo_UPD }; 4571 static const uint16_t QOpcodes1[] = { ARM::VST3q8oddPseudo, 4572 ARM::VST3q16oddPseudo, 4573 ARM::VST3q32oddPseudo }; 4574 SelectVST(N, false, 3, DOpcodes, QOpcodes0, QOpcodes1); 4575 return; 4576 } 4577 4578 case Intrinsic::arm_neon_vst4: { 4579 static const uint16_t DOpcodes[] = { ARM::VST4d8Pseudo, 4580 ARM::VST4d16Pseudo, 4581 ARM::VST4d32Pseudo, 4582 ARM::VST1d64QPseudo }; 4583 static const uint16_t QOpcodes0[] = { ARM::VST4q8Pseudo_UPD, 4584 ARM::VST4q16Pseudo_UPD, 4585 ARM::VST4q32Pseudo_UPD }; 4586 static const uint16_t QOpcodes1[] = { ARM::VST4q8oddPseudo, 4587 ARM::VST4q16oddPseudo, 4588 ARM::VST4q32oddPseudo }; 4589 SelectVST(N, false, 4, DOpcodes, QOpcodes0, QOpcodes1); 4590 return; 4591 } 4592 4593 case Intrinsic::arm_neon_vst2lane: { 4594 static const uint16_t DOpcodes[] = { ARM::VST2LNd8Pseudo, 4595 ARM::VST2LNd16Pseudo, 4596 ARM::VST2LNd32Pseudo }; 4597 static const uint16_t QOpcodes[] = { ARM::VST2LNq16Pseudo, 4598 ARM::VST2LNq32Pseudo }; 4599 SelectVLDSTLane(N, false, false, 2, DOpcodes, QOpcodes); 4600 return; 4601 } 4602 4603 case Intrinsic::arm_neon_vst3lane: { 4604 static const uint16_t DOpcodes[] = { ARM::VST3LNd8Pseudo, 4605 ARM::VST3LNd16Pseudo, 4606 ARM::VST3LNd32Pseudo }; 4607 static const uint16_t QOpcodes[] = { ARM::VST3LNq16Pseudo, 4608 ARM::VST3LNq32Pseudo }; 4609 SelectVLDSTLane(N, false, false, 3, DOpcodes, QOpcodes); 4610 return; 4611 } 4612 4613 case Intrinsic::arm_neon_vst4lane: { 4614 static const uint16_t DOpcodes[] = { ARM::VST4LNd8Pseudo, 4615 ARM::VST4LNd16Pseudo, 4616 ARM::VST4LNd32Pseudo }; 4617 static const uint16_t QOpcodes[] = { ARM::VST4LNq16Pseudo, 4618 ARM::VST4LNq32Pseudo }; 4619 SelectVLDSTLane(N, false, false, 4, DOpcodes, QOpcodes); 4620 return; 4621 } 4622 4623 case Intrinsic::arm_mve_vldr_gather_base_wb: 4624 case Intrinsic::arm_mve_vldr_gather_base_wb_predicated: { 4625 static const uint16_t Opcodes[] = {ARM::MVE_VLDRWU32_qi_pre, 4626 ARM::MVE_VLDRDU64_qi_pre}; 4627 SelectMVE_WB(N, Opcodes, 4628 IntNo == Intrinsic::arm_mve_vldr_gather_base_wb_predicated); 4629 return; 4630 } 4631 4632 case Intrinsic::arm_mve_vld2q: { 4633 static const uint16_t Opcodes8[] = {ARM::MVE_VLD20_8, ARM::MVE_VLD21_8}; 4634 static const uint16_t Opcodes16[] = {ARM::MVE_VLD20_16, 4635 ARM::MVE_VLD21_16}; 4636 static const uint16_t Opcodes32[] = {ARM::MVE_VLD20_32, 4637 ARM::MVE_VLD21_32}; 4638 static const uint16_t *const Opcodes[] = {Opcodes8, Opcodes16, Opcodes32}; 4639 SelectMVE_VLD(N, 2, Opcodes, false); 4640 return; 4641 } 4642 4643 case Intrinsic::arm_mve_vld4q: { 4644 static const uint16_t Opcodes8[] = {ARM::MVE_VLD40_8, ARM::MVE_VLD41_8, 4645 ARM::MVE_VLD42_8, ARM::MVE_VLD43_8}; 4646 static const uint16_t Opcodes16[] = {ARM::MVE_VLD40_16, ARM::MVE_VLD41_16, 4647 ARM::MVE_VLD42_16, 4648 ARM::MVE_VLD43_16}; 4649 static const uint16_t Opcodes32[] = {ARM::MVE_VLD40_32, ARM::MVE_VLD41_32, 4650 ARM::MVE_VLD42_32, 4651 ARM::MVE_VLD43_32}; 4652 static const uint16_t *const Opcodes[] = {Opcodes8, Opcodes16, Opcodes32}; 4653 SelectMVE_VLD(N, 4, Opcodes, false); 4654 return; 4655 } 4656 } 4657 break; 4658 } 4659 4660 case ISD::INTRINSIC_WO_CHAIN: { 4661 unsigned IntNo = cast<ConstantSDNode>(N->getOperand(0))->getZExtValue(); 4662 switch (IntNo) { 4663 default: 4664 break; 4665 4666 case Intrinsic::arm_mve_urshrl: 4667 SelectMVE_LongShift(N, ARM::MVE_URSHRL, true, false); 4668 return; 4669 case Intrinsic::arm_mve_uqshll: 4670 SelectMVE_LongShift(N, ARM::MVE_UQSHLL, true, false); 4671 return; 4672 case Intrinsic::arm_mve_srshrl: 4673 SelectMVE_LongShift(N, ARM::MVE_SRSHRL, true, false); 4674 return; 4675 case Intrinsic::arm_mve_sqshll: 4676 SelectMVE_LongShift(N, ARM::MVE_SQSHLL, true, false); 4677 return; 4678 case Intrinsic::arm_mve_uqrshll: 4679 SelectMVE_LongShift(N, ARM::MVE_UQRSHLL, false, true); 4680 return; 4681 case Intrinsic::arm_mve_sqrshrl: 4682 SelectMVE_LongShift(N, ARM::MVE_SQRSHRL, false, true); 4683 return; 4684 4685 case Intrinsic::arm_mve_vadc: 4686 case Intrinsic::arm_mve_vadc_predicated: 4687 SelectMVE_VADCSBC(N, ARM::MVE_VADC, ARM::MVE_VADCI, true, 4688 IntNo == Intrinsic::arm_mve_vadc_predicated); 4689 return; 4690 case Intrinsic::arm_mve_vsbc: 4691 case Intrinsic::arm_mve_vsbc_predicated: 4692 SelectMVE_VADCSBC(N, ARM::MVE_VSBC, ARM::MVE_VSBCI, true, 4693 IntNo == Intrinsic::arm_mve_vsbc_predicated); 4694 return; 4695 case Intrinsic::arm_mve_vshlc: 4696 case Intrinsic::arm_mve_vshlc_predicated: 4697 SelectMVE_VSHLC(N, IntNo == Intrinsic::arm_mve_vshlc_predicated); 4698 return; 4699 4700 case Intrinsic::arm_mve_vmlldava: 4701 case Intrinsic::arm_mve_vmlldava_predicated: { 4702 static const uint16_t OpcodesU[] = { 4703 ARM::MVE_VMLALDAVu16, ARM::MVE_VMLALDAVu32, 4704 ARM::MVE_VMLALDAVau16, ARM::MVE_VMLALDAVau32, 4705 }; 4706 static const uint16_t OpcodesS[] = { 4707 ARM::MVE_VMLALDAVs16, ARM::MVE_VMLALDAVs32, 4708 ARM::MVE_VMLALDAVas16, ARM::MVE_VMLALDAVas32, 4709 ARM::MVE_VMLALDAVxs16, ARM::MVE_VMLALDAVxs32, 4710 ARM::MVE_VMLALDAVaxs16, ARM::MVE_VMLALDAVaxs32, 4711 ARM::MVE_VMLSLDAVs16, ARM::MVE_VMLSLDAVs32, 4712 ARM::MVE_VMLSLDAVas16, ARM::MVE_VMLSLDAVas32, 4713 ARM::MVE_VMLSLDAVxs16, ARM::MVE_VMLSLDAVxs32, 4714 ARM::MVE_VMLSLDAVaxs16, ARM::MVE_VMLSLDAVaxs32, 4715 }; 4716 SelectMVE_VMLLDAV(N, IntNo == Intrinsic::arm_mve_vmlldava_predicated, 4717 OpcodesS, OpcodesU); 4718 return; 4719 } 4720 4721 case Intrinsic::arm_mve_vrmlldavha: 4722 case Intrinsic::arm_mve_vrmlldavha_predicated: { 4723 static const uint16_t OpcodesU[] = { 4724 ARM::MVE_VRMLALDAVHu32, ARM::MVE_VRMLALDAVHau32, 4725 }; 4726 static const uint16_t OpcodesS[] = { 4727 ARM::MVE_VRMLALDAVHs32, ARM::MVE_VRMLALDAVHas32, 4728 ARM::MVE_VRMLALDAVHxs32, ARM::MVE_VRMLALDAVHaxs32, 4729 ARM::MVE_VRMLSLDAVHs32, ARM::MVE_VRMLSLDAVHas32, 4730 ARM::MVE_VRMLSLDAVHxs32, ARM::MVE_VRMLSLDAVHaxs32, 4731 }; 4732 SelectMVE_VRMLLDAVH(N, IntNo == Intrinsic::arm_mve_vrmlldavha_predicated, 4733 OpcodesS, OpcodesU); 4734 return; 4735 } 4736 4737 case Intrinsic::arm_mve_vidup: 4738 case Intrinsic::arm_mve_vidup_predicated: { 4739 static const uint16_t Opcodes[] = { 4740 ARM::MVE_VIDUPu8, ARM::MVE_VIDUPu16, ARM::MVE_VIDUPu32, 4741 }; 4742 SelectMVE_VxDUP(N, Opcodes, false, 4743 IntNo == Intrinsic::arm_mve_vidup_predicated); 4744 return; 4745 } 4746 4747 case Intrinsic::arm_mve_vddup: 4748 case Intrinsic::arm_mve_vddup_predicated: { 4749 static const uint16_t Opcodes[] = { 4750 ARM::MVE_VDDUPu8, ARM::MVE_VDDUPu16, ARM::MVE_VDDUPu32, 4751 }; 4752 SelectMVE_VxDUP(N, Opcodes, false, 4753 IntNo == Intrinsic::arm_mve_vddup_predicated); 4754 return; 4755 } 4756 4757 case Intrinsic::arm_mve_viwdup: 4758 case Intrinsic::arm_mve_viwdup_predicated: { 4759 static const uint16_t Opcodes[] = { 4760 ARM::MVE_VIWDUPu8, ARM::MVE_VIWDUPu16, ARM::MVE_VIWDUPu32, 4761 }; 4762 SelectMVE_VxDUP(N, Opcodes, true, 4763 IntNo == Intrinsic::arm_mve_viwdup_predicated); 4764 return; 4765 } 4766 4767 case Intrinsic::arm_mve_vdwdup: 4768 case Intrinsic::arm_mve_vdwdup_predicated: { 4769 static const uint16_t Opcodes[] = { 4770 ARM::MVE_VDWDUPu8, ARM::MVE_VDWDUPu16, ARM::MVE_VDWDUPu32, 4771 }; 4772 SelectMVE_VxDUP(N, Opcodes, true, 4773 IntNo == Intrinsic::arm_mve_vdwdup_predicated); 4774 return; 4775 } 4776 } 4777 break; 4778 } 4779 4780 case ISD::ATOMIC_CMP_SWAP: 4781 SelectCMP_SWAP(N); 4782 return; 4783 } 4784 4785 SelectCode(N); 4786 } 4787 4788 // Inspect a register string of the form 4789 // cp<coprocessor>:<opc1>:c<CRn>:c<CRm>:<opc2> (32bit) or 4790 // cp<coprocessor>:<opc1>:c<CRm> (64bit) inspect the fields of the string 4791 // and obtain the integer operands from them, adding these operands to the 4792 // provided vector. 4793 static void getIntOperandsFromRegisterString(StringRef RegString, 4794 SelectionDAG *CurDAG, 4795 const SDLoc &DL, 4796 std::vector<SDValue> &Ops) { 4797 SmallVector<StringRef, 5> Fields; 4798 RegString.split(Fields, ':'); 4799 4800 if (Fields.size() > 1) { 4801 bool AllIntFields = true; 4802 4803 for (StringRef Field : Fields) { 4804 // Need to trim out leading 'cp' characters and get the integer field. 4805 unsigned IntField; 4806 AllIntFields &= !Field.trim("CPcp").getAsInteger(10, IntField); 4807 Ops.push_back(CurDAG->getTargetConstant(IntField, DL, MVT::i32)); 4808 } 4809 4810 assert(AllIntFields && 4811 "Unexpected non-integer value in special register string."); 4812 } 4813 } 4814 4815 // Maps a Banked Register string to its mask value. The mask value returned is 4816 // for use in the MRSbanked / MSRbanked instruction nodes as the Banked Register 4817 // mask operand, which expresses which register is to be used, e.g. r8, and in 4818 // which mode it is to be used, e.g. usr. Returns -1 to signify that the string 4819 // was invalid. 4820 static inline int getBankedRegisterMask(StringRef RegString) { 4821 auto TheReg = ARMBankedReg::lookupBankedRegByName(RegString.lower()); 4822 if (!TheReg) 4823 return -1; 4824 return TheReg->Encoding; 4825 } 4826 4827 // The flags here are common to those allowed for apsr in the A class cores and 4828 // those allowed for the special registers in the M class cores. Returns a 4829 // value representing which flags were present, -1 if invalid. 4830 static inline int getMClassFlagsMask(StringRef Flags) { 4831 return StringSwitch<int>(Flags) 4832 .Case("", 0x2) // no flags means nzcvq for psr registers, and 0x2 is 4833 // correct when flags are not permitted 4834 .Case("g", 0x1) 4835 .Case("nzcvq", 0x2) 4836 .Case("nzcvqg", 0x3) 4837 .Default(-1); 4838 } 4839 4840 // Maps MClass special registers string to its value for use in the 4841 // t2MRS_M/t2MSR_M instruction nodes as the SYSm value operand. 4842 // Returns -1 to signify that the string was invalid. 4843 static int getMClassRegisterMask(StringRef Reg, const ARMSubtarget *Subtarget) { 4844 auto TheReg = ARMSysReg::lookupMClassSysRegByName(Reg); 4845 const FeatureBitset &FeatureBits = Subtarget->getFeatureBits(); 4846 if (!TheReg || !TheReg->hasRequiredFeatures(FeatureBits)) 4847 return -1; 4848 return (int)(TheReg->Encoding & 0xFFF); // SYSm value 4849 } 4850 4851 static int getARClassRegisterMask(StringRef Reg, StringRef Flags) { 4852 // The mask operand contains the special register (R Bit) in bit 4, whether 4853 // the register is spsr (R bit is 1) or one of cpsr/apsr (R bit is 0), and 4854 // bits 3-0 contains the fields to be accessed in the special register, set by 4855 // the flags provided with the register. 4856 int Mask = 0; 4857 if (Reg == "apsr") { 4858 // The flags permitted for apsr are the same flags that are allowed in 4859 // M class registers. We get the flag value and then shift the flags into 4860 // the correct place to combine with the mask. 4861 Mask = getMClassFlagsMask(Flags); 4862 if (Mask == -1) 4863 return -1; 4864 return Mask << 2; 4865 } 4866 4867 if (Reg != "cpsr" && Reg != "spsr") { 4868 return -1; 4869 } 4870 4871 // This is the same as if the flags were "fc" 4872 if (Flags.empty() || Flags == "all") 4873 return Mask | 0x9; 4874 4875 // Inspect the supplied flags string and set the bits in the mask for 4876 // the relevant and valid flags allowed for cpsr and spsr. 4877 for (char Flag : Flags) { 4878 int FlagVal; 4879 switch (Flag) { 4880 case 'c': 4881 FlagVal = 0x1; 4882 break; 4883 case 'x': 4884 FlagVal = 0x2; 4885 break; 4886 case 's': 4887 FlagVal = 0x4; 4888 break; 4889 case 'f': 4890 FlagVal = 0x8; 4891 break; 4892 default: 4893 FlagVal = 0; 4894 } 4895 4896 // This avoids allowing strings where the same flag bit appears twice. 4897 if (!FlagVal || (Mask & FlagVal)) 4898 return -1; 4899 Mask |= FlagVal; 4900 } 4901 4902 // If the register is spsr then we need to set the R bit. 4903 if (Reg == "spsr") 4904 Mask |= 0x10; 4905 4906 return Mask; 4907 } 4908 4909 // Lower the read_register intrinsic to ARM specific DAG nodes 4910 // using the supplied metadata string to select the instruction node to use 4911 // and the registers/masks to construct as operands for the node. 4912 bool ARMDAGToDAGISel::tryReadRegister(SDNode *N){ 4913 const MDNodeSDNode *MD = dyn_cast<MDNodeSDNode>(N->getOperand(1)); 4914 const MDString *RegString = dyn_cast<MDString>(MD->getMD()->getOperand(0)); 4915 bool IsThumb2 = Subtarget->isThumb2(); 4916 SDLoc DL(N); 4917 4918 std::vector<SDValue> Ops; 4919 getIntOperandsFromRegisterString(RegString->getString(), CurDAG, DL, Ops); 4920 4921 if (!Ops.empty()) { 4922 // If the special register string was constructed of fields (as defined 4923 // in the ACLE) then need to lower to MRC node (32 bit) or 4924 // MRRC node(64 bit), we can make the distinction based on the number of 4925 // operands we have. 4926 unsigned Opcode; 4927 SmallVector<EVT, 3> ResTypes; 4928 if (Ops.size() == 5){ 4929 Opcode = IsThumb2 ? ARM::t2MRC : ARM::MRC; 4930 ResTypes.append({ MVT::i32, MVT::Other }); 4931 } else { 4932 assert(Ops.size() == 3 && 4933 "Invalid number of fields in special register string."); 4934 Opcode = IsThumb2 ? ARM::t2MRRC : ARM::MRRC; 4935 ResTypes.append({ MVT::i32, MVT::i32, MVT::Other }); 4936 } 4937 4938 Ops.push_back(getAL(CurDAG, DL)); 4939 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 4940 Ops.push_back(N->getOperand(0)); 4941 ReplaceNode(N, CurDAG->getMachineNode(Opcode, DL, ResTypes, Ops)); 4942 return true; 4943 } 4944 4945 std::string SpecialReg = RegString->getString().lower(); 4946 4947 int BankedReg = getBankedRegisterMask(SpecialReg); 4948 if (BankedReg != -1) { 4949 Ops = { CurDAG->getTargetConstant(BankedReg, DL, MVT::i32), 4950 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 4951 N->getOperand(0) }; 4952 ReplaceNode( 4953 N, CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRSbanked : ARM::MRSbanked, 4954 DL, MVT::i32, MVT::Other, Ops)); 4955 return true; 4956 } 4957 4958 // The VFP registers are read by creating SelectionDAG nodes with opcodes 4959 // corresponding to the register that is being read from. So we switch on the 4960 // string to find which opcode we need to use. 4961 unsigned Opcode = StringSwitch<unsigned>(SpecialReg) 4962 .Case("fpscr", ARM::VMRS) 4963 .Case("fpexc", ARM::VMRS_FPEXC) 4964 .Case("fpsid", ARM::VMRS_FPSID) 4965 .Case("mvfr0", ARM::VMRS_MVFR0) 4966 .Case("mvfr1", ARM::VMRS_MVFR1) 4967 .Case("mvfr2", ARM::VMRS_MVFR2) 4968 .Case("fpinst", ARM::VMRS_FPINST) 4969 .Case("fpinst2", ARM::VMRS_FPINST2) 4970 .Default(0); 4971 4972 // If an opcode was found then we can lower the read to a VFP instruction. 4973 if (Opcode) { 4974 if (!Subtarget->hasVFP2Base()) 4975 return false; 4976 if (Opcode == ARM::VMRS_MVFR2 && !Subtarget->hasFPARMv8Base()) 4977 return false; 4978 4979 Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 4980 N->getOperand(0) }; 4981 ReplaceNode(N, 4982 CurDAG->getMachineNode(Opcode, DL, MVT::i32, MVT::Other, Ops)); 4983 return true; 4984 } 4985 4986 // If the target is M Class then need to validate that the register string 4987 // is an acceptable value, so check that a mask can be constructed from the 4988 // string. 4989 if (Subtarget->isMClass()) { 4990 int SYSmValue = getMClassRegisterMask(SpecialReg, Subtarget); 4991 if (SYSmValue == -1) 4992 return false; 4993 4994 SDValue Ops[] = { CurDAG->getTargetConstant(SYSmValue, DL, MVT::i32), 4995 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 4996 N->getOperand(0) }; 4997 ReplaceNode( 4998 N, CurDAG->getMachineNode(ARM::t2MRS_M, DL, MVT::i32, MVT::Other, Ops)); 4999 return true; 5000 } 5001 5002 // Here we know the target is not M Class so we need to check if it is one 5003 // of the remaining possible values which are apsr, cpsr or spsr. 5004 if (SpecialReg == "apsr" || SpecialReg == "cpsr") { 5005 Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 5006 N->getOperand(0) }; 5007 ReplaceNode(N, CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRS_AR : ARM::MRS, 5008 DL, MVT::i32, MVT::Other, Ops)); 5009 return true; 5010 } 5011 5012 if (SpecialReg == "spsr") { 5013 Ops = { getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 5014 N->getOperand(0) }; 5015 ReplaceNode( 5016 N, CurDAG->getMachineNode(IsThumb2 ? ARM::t2MRSsys_AR : ARM::MRSsys, DL, 5017 MVT::i32, MVT::Other, Ops)); 5018 return true; 5019 } 5020 5021 return false; 5022 } 5023 5024 // Lower the write_register intrinsic to ARM specific DAG nodes 5025 // using the supplied metadata string to select the instruction node to use 5026 // and the registers/masks to use in the nodes 5027 bool ARMDAGToDAGISel::tryWriteRegister(SDNode *N){ 5028 const MDNodeSDNode *MD = dyn_cast<MDNodeSDNode>(N->getOperand(1)); 5029 const MDString *RegString = dyn_cast<MDString>(MD->getMD()->getOperand(0)); 5030 bool IsThumb2 = Subtarget->isThumb2(); 5031 SDLoc DL(N); 5032 5033 std::vector<SDValue> Ops; 5034 getIntOperandsFromRegisterString(RegString->getString(), CurDAG, DL, Ops); 5035 5036 if (!Ops.empty()) { 5037 // If the special register string was constructed of fields (as defined 5038 // in the ACLE) then need to lower to MCR node (32 bit) or 5039 // MCRR node(64 bit), we can make the distinction based on the number of 5040 // operands we have. 5041 unsigned Opcode; 5042 if (Ops.size() == 5) { 5043 Opcode = IsThumb2 ? ARM::t2MCR : ARM::MCR; 5044 Ops.insert(Ops.begin()+2, N->getOperand(2)); 5045 } else { 5046 assert(Ops.size() == 3 && 5047 "Invalid number of fields in special register string."); 5048 Opcode = IsThumb2 ? ARM::t2MCRR : ARM::MCRR; 5049 SDValue WriteValue[] = { N->getOperand(2), N->getOperand(3) }; 5050 Ops.insert(Ops.begin()+2, WriteValue, WriteValue+2); 5051 } 5052 5053 Ops.push_back(getAL(CurDAG, DL)); 5054 Ops.push_back(CurDAG->getRegister(0, MVT::i32)); 5055 Ops.push_back(N->getOperand(0)); 5056 5057 ReplaceNode(N, CurDAG->getMachineNode(Opcode, DL, MVT::Other, Ops)); 5058 return true; 5059 } 5060 5061 std::string SpecialReg = RegString->getString().lower(); 5062 int BankedReg = getBankedRegisterMask(SpecialReg); 5063 if (BankedReg != -1) { 5064 Ops = { CurDAG->getTargetConstant(BankedReg, DL, MVT::i32), N->getOperand(2), 5065 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 5066 N->getOperand(0) }; 5067 ReplaceNode( 5068 N, CurDAG->getMachineNode(IsThumb2 ? ARM::t2MSRbanked : ARM::MSRbanked, 5069 DL, MVT::Other, Ops)); 5070 return true; 5071 } 5072 5073 // The VFP registers are written to by creating SelectionDAG nodes with 5074 // opcodes corresponding to the register that is being written. So we switch 5075 // on the string to find which opcode we need to use. 5076 unsigned Opcode = StringSwitch<unsigned>(SpecialReg) 5077 .Case("fpscr", ARM::VMSR) 5078 .Case("fpexc", ARM::VMSR_FPEXC) 5079 .Case("fpsid", ARM::VMSR_FPSID) 5080 .Case("fpinst", ARM::VMSR_FPINST) 5081 .Case("fpinst2", ARM::VMSR_FPINST2) 5082 .Default(0); 5083 5084 if (Opcode) { 5085 if (!Subtarget->hasVFP2Base()) 5086 return false; 5087 Ops = { N->getOperand(2), getAL(CurDAG, DL), 5088 CurDAG->getRegister(0, MVT::i32), N->getOperand(0) }; 5089 ReplaceNode(N, CurDAG->getMachineNode(Opcode, DL, MVT::Other, Ops)); 5090 return true; 5091 } 5092 5093 std::pair<StringRef, StringRef> Fields; 5094 Fields = StringRef(SpecialReg).rsplit('_'); 5095 std::string Reg = Fields.first.str(); 5096 StringRef Flags = Fields.second; 5097 5098 // If the target was M Class then need to validate the special register value 5099 // and retrieve the mask for use in the instruction node. 5100 if (Subtarget->isMClass()) { 5101 int SYSmValue = getMClassRegisterMask(SpecialReg, Subtarget); 5102 if (SYSmValue == -1) 5103 return false; 5104 5105 SDValue Ops[] = { CurDAG->getTargetConstant(SYSmValue, DL, MVT::i32), 5106 N->getOperand(2), getAL(CurDAG, DL), 5107 CurDAG->getRegister(0, MVT::i32), N->getOperand(0) }; 5108 ReplaceNode(N, CurDAG->getMachineNode(ARM::t2MSR_M, DL, MVT::Other, Ops)); 5109 return true; 5110 } 5111 5112 // We then check to see if a valid mask can be constructed for one of the 5113 // register string values permitted for the A and R class cores. These values 5114 // are apsr, spsr and cpsr; these are also valid on older cores. 5115 int Mask = getARClassRegisterMask(Reg, Flags); 5116 if (Mask != -1) { 5117 Ops = { CurDAG->getTargetConstant(Mask, DL, MVT::i32), N->getOperand(2), 5118 getAL(CurDAG, DL), CurDAG->getRegister(0, MVT::i32), 5119 N->getOperand(0) }; 5120 ReplaceNode(N, CurDAG->getMachineNode(IsThumb2 ? ARM::t2MSR_AR : ARM::MSR, 5121 DL, MVT::Other, Ops)); 5122 return true; 5123 } 5124 5125 return false; 5126 } 5127 5128 bool ARMDAGToDAGISel::tryInlineAsm(SDNode *N){ 5129 std::vector<SDValue> AsmNodeOperands; 5130 unsigned Flag, Kind; 5131 bool Changed = false; 5132 unsigned NumOps = N->getNumOperands(); 5133 5134 // Normally, i64 data is bounded to two arbitrary GRPs for "%r" constraint. 5135 // However, some instrstions (e.g. ldrexd/strexd in ARM mode) require 5136 // (even/even+1) GPRs and use %n and %Hn to refer to the individual regs 5137 // respectively. Since there is no constraint to explicitly specify a 5138 // reg pair, we use GPRPair reg class for "%r" for 64-bit data. For Thumb, 5139 // the 64-bit data may be referred by H, Q, R modifiers, so we still pack 5140 // them into a GPRPair. 5141 5142 SDLoc dl(N); 5143 SDValue Glue = N->getGluedNode() ? N->getOperand(NumOps-1) 5144 : SDValue(nullptr,0); 5145 5146 SmallVector<bool, 8> OpChanged; 5147 // Glue node will be appended late. 5148 for(unsigned i = 0, e = N->getGluedNode() ? NumOps - 1 : NumOps; i < e; ++i) { 5149 SDValue op = N->getOperand(i); 5150 AsmNodeOperands.push_back(op); 5151 5152 if (i < InlineAsm::Op_FirstOperand) 5153 continue; 5154 5155 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(i))) { 5156 Flag = C->getZExtValue(); 5157 Kind = InlineAsm::getKind(Flag); 5158 } 5159 else 5160 continue; 5161 5162 // Immediate operands to inline asm in the SelectionDAG are modeled with 5163 // two operands. The first is a constant of value InlineAsm::Kind_Imm, and 5164 // the second is a constant with the value of the immediate. If we get here 5165 // and we have a Kind_Imm, skip the next operand, and continue. 5166 if (Kind == InlineAsm::Kind_Imm) { 5167 SDValue op = N->getOperand(++i); 5168 AsmNodeOperands.push_back(op); 5169 continue; 5170 } 5171 5172 unsigned NumRegs = InlineAsm::getNumOperandRegisters(Flag); 5173 if (NumRegs) 5174 OpChanged.push_back(false); 5175 5176 unsigned DefIdx = 0; 5177 bool IsTiedToChangedOp = false; 5178 // If it's a use that is tied with a previous def, it has no 5179 // reg class constraint. 5180 if (Changed && InlineAsm::isUseOperandTiedToDef(Flag, DefIdx)) 5181 IsTiedToChangedOp = OpChanged[DefIdx]; 5182 5183 // Memory operands to inline asm in the SelectionDAG are modeled with two 5184 // operands: a constant of value InlineAsm::Kind_Mem followed by the input 5185 // operand. If we get here and we have a Kind_Mem, skip the next operand (so 5186 // it doesn't get misinterpreted), and continue. We do this here because 5187 // it's important to update the OpChanged array correctly before moving on. 5188 if (Kind == InlineAsm::Kind_Mem) { 5189 SDValue op = N->getOperand(++i); 5190 AsmNodeOperands.push_back(op); 5191 continue; 5192 } 5193 5194 if (Kind != InlineAsm::Kind_RegUse && Kind != InlineAsm::Kind_RegDef 5195 && Kind != InlineAsm::Kind_RegDefEarlyClobber) 5196 continue; 5197 5198 unsigned RC; 5199 bool HasRC = InlineAsm::hasRegClassConstraint(Flag, RC); 5200 if ((!IsTiedToChangedOp && (!HasRC || RC != ARM::GPRRegClassID)) 5201 || NumRegs != 2) 5202 continue; 5203 5204 assert((i+2 < NumOps) && "Invalid number of operands in inline asm"); 5205 SDValue V0 = N->getOperand(i+1); 5206 SDValue V1 = N->getOperand(i+2); 5207 unsigned Reg0 = cast<RegisterSDNode>(V0)->getReg(); 5208 unsigned Reg1 = cast<RegisterSDNode>(V1)->getReg(); 5209 SDValue PairedReg; 5210 MachineRegisterInfo &MRI = MF->getRegInfo(); 5211 5212 if (Kind == InlineAsm::Kind_RegDef || 5213 Kind == InlineAsm::Kind_RegDefEarlyClobber) { 5214 // Replace the two GPRs with 1 GPRPair and copy values from GPRPair to 5215 // the original GPRs. 5216 5217 Register GPVR = MRI.createVirtualRegister(&ARM::GPRPairRegClass); 5218 PairedReg = CurDAG->getRegister(GPVR, MVT::Untyped); 5219 SDValue Chain = SDValue(N,0); 5220 5221 SDNode *GU = N->getGluedUser(); 5222 SDValue RegCopy = CurDAG->getCopyFromReg(Chain, dl, GPVR, MVT::Untyped, 5223 Chain.getValue(1)); 5224 5225 // Extract values from a GPRPair reg and copy to the original GPR reg. 5226 SDValue Sub0 = CurDAG->getTargetExtractSubreg(ARM::gsub_0, dl, MVT::i32, 5227 RegCopy); 5228 SDValue Sub1 = CurDAG->getTargetExtractSubreg(ARM::gsub_1, dl, MVT::i32, 5229 RegCopy); 5230 SDValue T0 = CurDAG->getCopyToReg(Sub0, dl, Reg0, Sub0, 5231 RegCopy.getValue(1)); 5232 SDValue T1 = CurDAG->getCopyToReg(Sub1, dl, Reg1, Sub1, T0.getValue(1)); 5233 5234 // Update the original glue user. 5235 std::vector<SDValue> Ops(GU->op_begin(), GU->op_end()-1); 5236 Ops.push_back(T1.getValue(1)); 5237 CurDAG->UpdateNodeOperands(GU, Ops); 5238 } 5239 else { 5240 // For Kind == InlineAsm::Kind_RegUse, we first copy two GPRs into a 5241 // GPRPair and then pass the GPRPair to the inline asm. 5242 SDValue Chain = AsmNodeOperands[InlineAsm::Op_InputChain]; 5243 5244 // As REG_SEQ doesn't take RegisterSDNode, we copy them first. 5245 SDValue T0 = CurDAG->getCopyFromReg(Chain, dl, Reg0, MVT::i32, 5246 Chain.getValue(1)); 5247 SDValue T1 = CurDAG->getCopyFromReg(Chain, dl, Reg1, MVT::i32, 5248 T0.getValue(1)); 5249 SDValue Pair = SDValue(createGPRPairNode(MVT::Untyped, T0, T1), 0); 5250 5251 // Copy REG_SEQ into a GPRPair-typed VR and replace the original two 5252 // i32 VRs of inline asm with it. 5253 Register GPVR = MRI.createVirtualRegister(&ARM::GPRPairRegClass); 5254 PairedReg = CurDAG->getRegister(GPVR, MVT::Untyped); 5255 Chain = CurDAG->getCopyToReg(T1, dl, GPVR, Pair, T1.getValue(1)); 5256 5257 AsmNodeOperands[InlineAsm::Op_InputChain] = Chain; 5258 Glue = Chain.getValue(1); 5259 } 5260 5261 Changed = true; 5262 5263 if(PairedReg.getNode()) { 5264 OpChanged[OpChanged.size() -1 ] = true; 5265 Flag = InlineAsm::getFlagWord(Kind, 1 /* RegNum*/); 5266 if (IsTiedToChangedOp) 5267 Flag = InlineAsm::getFlagWordForMatchingOp(Flag, DefIdx); 5268 else 5269 Flag = InlineAsm::getFlagWordForRegClass(Flag, ARM::GPRPairRegClassID); 5270 // Replace the current flag. 5271 AsmNodeOperands[AsmNodeOperands.size() -1] = CurDAG->getTargetConstant( 5272 Flag, dl, MVT::i32); 5273 // Add the new register node and skip the original two GPRs. 5274 AsmNodeOperands.push_back(PairedReg); 5275 // Skip the next two GPRs. 5276 i += 2; 5277 } 5278 } 5279 5280 if (Glue.getNode()) 5281 AsmNodeOperands.push_back(Glue); 5282 if (!Changed) 5283 return false; 5284 5285 SDValue New = CurDAG->getNode(N->getOpcode(), SDLoc(N), 5286 CurDAG->getVTList(MVT::Other, MVT::Glue), AsmNodeOperands); 5287 New->setNodeId(-1); 5288 ReplaceNode(N, New.getNode()); 5289 return true; 5290 } 5291 5292 5293 bool ARMDAGToDAGISel:: 5294 SelectInlineAsmMemoryOperand(const SDValue &Op, unsigned ConstraintID, 5295 std::vector<SDValue> &OutOps) { 5296 switch(ConstraintID) { 5297 default: 5298 llvm_unreachable("Unexpected asm memory constraint"); 5299 case InlineAsm::Constraint_m: 5300 case InlineAsm::Constraint_o: 5301 case InlineAsm::Constraint_Q: 5302 case InlineAsm::Constraint_Um: 5303 case InlineAsm::Constraint_Un: 5304 case InlineAsm::Constraint_Uq: 5305 case InlineAsm::Constraint_Us: 5306 case InlineAsm::Constraint_Ut: 5307 case InlineAsm::Constraint_Uv: 5308 case InlineAsm::Constraint_Uy: 5309 // Require the address to be in a register. That is safe for all ARM 5310 // variants and it is hard to do anything much smarter without knowing 5311 // how the operand is used. 5312 OutOps.push_back(Op); 5313 return false; 5314 } 5315 return true; 5316 } 5317 5318 /// createARMISelDag - This pass converts a legalized DAG into a 5319 /// ARM-specific DAG, ready for instruction scheduling. 5320 /// 5321 FunctionPass *llvm::createARMISelDag(ARMBaseTargetMachine &TM, 5322 CodeGenOpt::Level OptLevel) { 5323 return new ARMDAGToDAGISel(TM, OptLevel); 5324 } 5325