1 //===- ARMISelLowering.cpp - ARM DAG Lowering Implementation --------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This file defines the interfaces that ARM uses to lower LLVM code into a 10 // selection DAG. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "ARMISelLowering.h" 15 #include "ARMBaseInstrInfo.h" 16 #include "ARMBaseRegisterInfo.h" 17 #include "ARMCallingConv.h" 18 #include "ARMConstantPoolValue.h" 19 #include "ARMMachineFunctionInfo.h" 20 #include "ARMPerfectShuffle.h" 21 #include "ARMRegisterInfo.h" 22 #include "ARMSelectionDAGInfo.h" 23 #include "ARMSubtarget.h" 24 #include "MCTargetDesc/ARMAddressingModes.h" 25 #include "MCTargetDesc/ARMBaseInfo.h" 26 #include "Utils/ARMBaseInfo.h" 27 #include "llvm/ADT/APFloat.h" 28 #include "llvm/ADT/APInt.h" 29 #include "llvm/ADT/ArrayRef.h" 30 #include "llvm/ADT/BitVector.h" 31 #include "llvm/ADT/DenseMap.h" 32 #include "llvm/ADT/STLExtras.h" 33 #include "llvm/ADT/SmallPtrSet.h" 34 #include "llvm/ADT/SmallVector.h" 35 #include "llvm/ADT/Statistic.h" 36 #include "llvm/ADT/StringExtras.h" 37 #include "llvm/ADT/StringRef.h" 38 #include "llvm/ADT/StringSwitch.h" 39 #include "llvm/ADT/Triple.h" 40 #include "llvm/ADT/Twine.h" 41 #include "llvm/Analysis/VectorUtils.h" 42 #include "llvm/CodeGen/CallingConvLower.h" 43 #include "llvm/CodeGen/ISDOpcodes.h" 44 #include "llvm/CodeGen/IntrinsicLowering.h" 45 #include "llvm/CodeGen/MachineBasicBlock.h" 46 #include "llvm/CodeGen/MachineConstantPool.h" 47 #include "llvm/CodeGen/MachineFrameInfo.h" 48 #include "llvm/CodeGen/MachineFunction.h" 49 #include "llvm/CodeGen/MachineInstr.h" 50 #include "llvm/CodeGen/MachineInstrBuilder.h" 51 #include "llvm/CodeGen/MachineJumpTableInfo.h" 52 #include "llvm/CodeGen/MachineMemOperand.h" 53 #include "llvm/CodeGen/MachineOperand.h" 54 #include "llvm/CodeGen/MachineRegisterInfo.h" 55 #include "llvm/CodeGen/RuntimeLibcalls.h" 56 #include "llvm/CodeGen/SelectionDAG.h" 57 #include "llvm/CodeGen/SelectionDAGNodes.h" 58 #include "llvm/CodeGen/TargetInstrInfo.h" 59 #include "llvm/CodeGen/TargetLowering.h" 60 #include "llvm/CodeGen/TargetOpcodes.h" 61 #include "llvm/CodeGen/TargetRegisterInfo.h" 62 #include "llvm/CodeGen/TargetSubtargetInfo.h" 63 #include "llvm/CodeGen/ValueTypes.h" 64 #include "llvm/IR/Attributes.h" 65 #include "llvm/IR/CallingConv.h" 66 #include "llvm/IR/Constant.h" 67 #include "llvm/IR/Constants.h" 68 #include "llvm/IR/DataLayout.h" 69 #include "llvm/IR/DebugLoc.h" 70 #include "llvm/IR/DerivedTypes.h" 71 #include "llvm/IR/Function.h" 72 #include "llvm/IR/GlobalAlias.h" 73 #include "llvm/IR/GlobalValue.h" 74 #include "llvm/IR/GlobalVariable.h" 75 #include "llvm/IR/IRBuilder.h" 76 #include "llvm/IR/InlineAsm.h" 77 #include "llvm/IR/Instruction.h" 78 #include "llvm/IR/Instructions.h" 79 #include "llvm/IR/IntrinsicInst.h" 80 #include "llvm/IR/Intrinsics.h" 81 #include "llvm/IR/Module.h" 82 #include "llvm/IR/PatternMatch.h" 83 #include "llvm/IR/Type.h" 84 #include "llvm/IR/User.h" 85 #include "llvm/IR/Value.h" 86 #include "llvm/MC/MCInstrDesc.h" 87 #include "llvm/MC/MCInstrItineraries.h" 88 #include "llvm/MC/MCRegisterInfo.h" 89 #include "llvm/MC/MCSchedule.h" 90 #include "llvm/Support/AtomicOrdering.h" 91 #include "llvm/Support/BranchProbability.h" 92 #include "llvm/Support/Casting.h" 93 #include "llvm/Support/CodeGen.h" 94 #include "llvm/Support/CommandLine.h" 95 #include "llvm/Support/Compiler.h" 96 #include "llvm/Support/Debug.h" 97 #include "llvm/Support/ErrorHandling.h" 98 #include "llvm/Support/KnownBits.h" 99 #include "llvm/Support/MachineValueType.h" 100 #include "llvm/Support/MathExtras.h" 101 #include "llvm/Support/raw_ostream.h" 102 #include "llvm/Target/TargetMachine.h" 103 #include "llvm/Target/TargetOptions.h" 104 #include <algorithm> 105 #include <cassert> 106 #include <cstdint> 107 #include <cstdlib> 108 #include <iterator> 109 #include <limits> 110 #include <string> 111 #include <tuple> 112 #include <utility> 113 #include <vector> 114 115 using namespace llvm; 116 using namespace llvm::PatternMatch; 117 118 #define DEBUG_TYPE "arm-isel" 119 120 STATISTIC(NumTailCalls, "Number of tail calls"); 121 STATISTIC(NumMovwMovt, "Number of GAs materialized with movw + movt"); 122 STATISTIC(NumLoopByVals, "Number of loops generated for byval arguments"); 123 STATISTIC(NumConstpoolPromoted, 124 "Number of constants with their storage promoted into constant pools"); 125 126 static cl::opt<bool> 127 ARMInterworking("arm-interworking", cl::Hidden, 128 cl::desc("Enable / disable ARM interworking (for debugging only)"), 129 cl::init(true)); 130 131 static cl::opt<bool> EnableConstpoolPromotion( 132 "arm-promote-constant", cl::Hidden, 133 cl::desc("Enable / disable promotion of unnamed_addr constants into " 134 "constant pools"), 135 cl::init(false)); // FIXME: set to true by default once PR32780 is fixed 136 static cl::opt<unsigned> ConstpoolPromotionMaxSize( 137 "arm-promote-constant-max-size", cl::Hidden, 138 cl::desc("Maximum size of constant to promote into a constant pool"), 139 cl::init(64)); 140 static cl::opt<unsigned> ConstpoolPromotionMaxTotal( 141 "arm-promote-constant-max-total", cl::Hidden, 142 cl::desc("Maximum size of ALL constants to promote into a constant pool"), 143 cl::init(128)); 144 145 // The APCS parameter registers. 146 static const MCPhysReg GPRArgRegs[] = { 147 ARM::R0, ARM::R1, ARM::R2, ARM::R3 148 }; 149 150 void ARMTargetLowering::addTypeForNEON(MVT VT, MVT PromotedLdStVT, 151 MVT PromotedBitwiseVT) { 152 if (VT != PromotedLdStVT) { 153 setOperationAction(ISD::LOAD, VT, Promote); 154 AddPromotedToType (ISD::LOAD, VT, PromotedLdStVT); 155 156 setOperationAction(ISD::STORE, VT, Promote); 157 AddPromotedToType (ISD::STORE, VT, PromotedLdStVT); 158 } 159 160 MVT ElemTy = VT.getVectorElementType(); 161 if (ElemTy != MVT::f64) 162 setOperationAction(ISD::SETCC, VT, Custom); 163 setOperationAction(ISD::INSERT_VECTOR_ELT, VT, Custom); 164 setOperationAction(ISD::EXTRACT_VECTOR_ELT, VT, Custom); 165 if (ElemTy == MVT::i32) { 166 setOperationAction(ISD::SINT_TO_FP, VT, Custom); 167 setOperationAction(ISD::UINT_TO_FP, VT, Custom); 168 setOperationAction(ISD::FP_TO_SINT, VT, Custom); 169 setOperationAction(ISD::FP_TO_UINT, VT, Custom); 170 } else { 171 setOperationAction(ISD::SINT_TO_FP, VT, Expand); 172 setOperationAction(ISD::UINT_TO_FP, VT, Expand); 173 setOperationAction(ISD::FP_TO_SINT, VT, Expand); 174 setOperationAction(ISD::FP_TO_UINT, VT, Expand); 175 } 176 setOperationAction(ISD::BUILD_VECTOR, VT, Custom); 177 setOperationAction(ISD::VECTOR_SHUFFLE, VT, Custom); 178 setOperationAction(ISD::CONCAT_VECTORS, VT, Legal); 179 setOperationAction(ISD::EXTRACT_SUBVECTOR, VT, Legal); 180 setOperationAction(ISD::SELECT, VT, Expand); 181 setOperationAction(ISD::SELECT_CC, VT, Expand); 182 setOperationAction(ISD::VSELECT, VT, Expand); 183 setOperationAction(ISD::SIGN_EXTEND_INREG, VT, Expand); 184 if (VT.isInteger()) { 185 setOperationAction(ISD::SHL, VT, Custom); 186 setOperationAction(ISD::SRA, VT, Custom); 187 setOperationAction(ISD::SRL, VT, Custom); 188 } 189 190 // Promote all bit-wise operations. 191 if (VT.isInteger() && VT != PromotedBitwiseVT) { 192 setOperationAction(ISD::AND, VT, Promote); 193 AddPromotedToType (ISD::AND, VT, PromotedBitwiseVT); 194 setOperationAction(ISD::OR, VT, Promote); 195 AddPromotedToType (ISD::OR, VT, PromotedBitwiseVT); 196 setOperationAction(ISD::XOR, VT, Promote); 197 AddPromotedToType (ISD::XOR, VT, PromotedBitwiseVT); 198 } 199 200 // Neon does not support vector divide/remainder operations. 201 setOperationAction(ISD::SDIV, VT, Expand); 202 setOperationAction(ISD::UDIV, VT, Expand); 203 setOperationAction(ISD::FDIV, VT, Expand); 204 setOperationAction(ISD::SREM, VT, Expand); 205 setOperationAction(ISD::UREM, VT, Expand); 206 setOperationAction(ISD::FREM, VT, Expand); 207 208 if (!VT.isFloatingPoint() && 209 VT != MVT::v2i64 && VT != MVT::v1i64) 210 for (auto Opcode : {ISD::ABS, ISD::SMIN, ISD::SMAX, ISD::UMIN, ISD::UMAX}) 211 setOperationAction(Opcode, VT, Legal); 212 } 213 214 void ARMTargetLowering::addDRTypeForNEON(MVT VT) { 215 addRegisterClass(VT, &ARM::DPRRegClass); 216 addTypeForNEON(VT, MVT::f64, MVT::v2i32); 217 } 218 219 void ARMTargetLowering::addQRTypeForNEON(MVT VT) { 220 addRegisterClass(VT, &ARM::DPairRegClass); 221 addTypeForNEON(VT, MVT::v2f64, MVT::v4i32); 222 } 223 224 void ARMTargetLowering::setAllExpand(MVT VT) { 225 for (unsigned Opc = 0; Opc < ISD::BUILTIN_OP_END; ++Opc) 226 setOperationAction(Opc, VT, Expand); 227 228 // We support these really simple operations even on types where all 229 // the actual arithmetic has to be broken down into simpler 230 // operations or turned into library calls. 231 setOperationAction(ISD::BITCAST, VT, Legal); 232 setOperationAction(ISD::LOAD, VT, Legal); 233 setOperationAction(ISD::STORE, VT, Legal); 234 setOperationAction(ISD::UNDEF, VT, Legal); 235 } 236 237 void ARMTargetLowering::addAllExtLoads(const MVT From, const MVT To, 238 LegalizeAction Action) { 239 setLoadExtAction(ISD::EXTLOAD, From, To, Action); 240 setLoadExtAction(ISD::ZEXTLOAD, From, To, Action); 241 setLoadExtAction(ISD::SEXTLOAD, From, To, Action); 242 } 243 244 void ARMTargetLowering::addMVEVectorTypes(bool HasMVEFP) { 245 const MVT IntTypes[] = { MVT::v16i8, MVT::v8i16, MVT::v4i32 }; 246 247 for (auto VT : IntTypes) { 248 addRegisterClass(VT, &ARM::MQPRRegClass); 249 setOperationAction(ISD::VECTOR_SHUFFLE, VT, Custom); 250 setOperationAction(ISD::INSERT_VECTOR_ELT, VT, Custom); 251 setOperationAction(ISD::EXTRACT_VECTOR_ELT, VT, Custom); 252 setOperationAction(ISD::BUILD_VECTOR, VT, Custom); 253 setOperationAction(ISD::SHL, VT, Custom); 254 setOperationAction(ISD::SRA, VT, Custom); 255 setOperationAction(ISD::SRL, VT, Custom); 256 setOperationAction(ISD::SMIN, VT, Legal); 257 setOperationAction(ISD::SMAX, VT, Legal); 258 setOperationAction(ISD::UMIN, VT, Legal); 259 setOperationAction(ISD::UMAX, VT, Legal); 260 setOperationAction(ISD::ABS, VT, Legal); 261 setOperationAction(ISD::SETCC, VT, Custom); 262 setOperationAction(ISD::MLOAD, VT, Custom); 263 setOperationAction(ISD::MSTORE, VT, Legal); 264 setOperationAction(ISD::CTLZ, VT, Legal); 265 setOperationAction(ISD::CTTZ, VT, Custom); 266 setOperationAction(ISD::BITREVERSE, VT, Legal); 267 setOperationAction(ISD::BSWAP, VT, Legal); 268 setOperationAction(ISD::SADDSAT, VT, Legal); 269 setOperationAction(ISD::UADDSAT, VT, Legal); 270 setOperationAction(ISD::SSUBSAT, VT, Legal); 271 setOperationAction(ISD::USUBSAT, VT, Legal); 272 273 // No native support for these. 274 setOperationAction(ISD::UDIV, VT, Expand); 275 setOperationAction(ISD::SDIV, VT, Expand); 276 setOperationAction(ISD::UREM, VT, Expand); 277 setOperationAction(ISD::SREM, VT, Expand); 278 setOperationAction(ISD::CTPOP, VT, Expand); 279 280 // Vector reductions 281 setOperationAction(ISD::VECREDUCE_ADD, VT, Legal); 282 setOperationAction(ISD::VECREDUCE_SMAX, VT, Legal); 283 setOperationAction(ISD::VECREDUCE_UMAX, VT, Legal); 284 setOperationAction(ISD::VECREDUCE_SMIN, VT, Legal); 285 setOperationAction(ISD::VECREDUCE_UMIN, VT, Legal); 286 287 if (!HasMVEFP) { 288 setOperationAction(ISD::SINT_TO_FP, VT, Expand); 289 setOperationAction(ISD::UINT_TO_FP, VT, Expand); 290 setOperationAction(ISD::FP_TO_SINT, VT, Expand); 291 setOperationAction(ISD::FP_TO_UINT, VT, Expand); 292 } 293 294 // Pre and Post inc are supported on loads and stores 295 for (unsigned im = (unsigned)ISD::PRE_INC; 296 im != (unsigned)ISD::LAST_INDEXED_MODE; ++im) { 297 setIndexedLoadAction(im, VT, Legal); 298 setIndexedStoreAction(im, VT, Legal); 299 } 300 } 301 302 const MVT FloatTypes[] = { MVT::v8f16, MVT::v4f32 }; 303 for (auto VT : FloatTypes) { 304 addRegisterClass(VT, &ARM::MQPRRegClass); 305 if (!HasMVEFP) 306 setAllExpand(VT); 307 308 // These are legal or custom whether we have MVE.fp or not 309 setOperationAction(ISD::VECTOR_SHUFFLE, VT, Custom); 310 setOperationAction(ISD::INSERT_VECTOR_ELT, VT, Custom); 311 setOperationAction(ISD::INSERT_VECTOR_ELT, VT.getVectorElementType(), Custom); 312 setOperationAction(ISD::EXTRACT_VECTOR_ELT, VT, Custom); 313 setOperationAction(ISD::BUILD_VECTOR, VT, Custom); 314 setOperationAction(ISD::BUILD_VECTOR, VT.getVectorElementType(), Custom); 315 setOperationAction(ISD::SCALAR_TO_VECTOR, VT, Legal); 316 setOperationAction(ISD::SETCC, VT, Custom); 317 setOperationAction(ISD::MLOAD, VT, Custom); 318 setOperationAction(ISD::MSTORE, VT, Legal); 319 320 // Pre and Post inc are supported on loads and stores 321 for (unsigned im = (unsigned)ISD::PRE_INC; 322 im != (unsigned)ISD::LAST_INDEXED_MODE; ++im) { 323 setIndexedLoadAction(im, VT, Legal); 324 setIndexedStoreAction(im, VT, Legal); 325 } 326 327 if (HasMVEFP) { 328 setOperationAction(ISD::FMINNUM, VT, Legal); 329 setOperationAction(ISD::FMAXNUM, VT, Legal); 330 setOperationAction(ISD::FROUND, VT, Legal); 331 332 // No native support for these. 333 setOperationAction(ISD::FDIV, VT, Expand); 334 setOperationAction(ISD::FREM, VT, Expand); 335 setOperationAction(ISD::FSQRT, VT, Expand); 336 setOperationAction(ISD::FSIN, VT, Expand); 337 setOperationAction(ISD::FCOS, VT, Expand); 338 setOperationAction(ISD::FPOW, VT, Expand); 339 setOperationAction(ISD::FLOG, VT, Expand); 340 setOperationAction(ISD::FLOG2, VT, Expand); 341 setOperationAction(ISD::FLOG10, VT, Expand); 342 setOperationAction(ISD::FEXP, VT, Expand); 343 setOperationAction(ISD::FEXP2, VT, Expand); 344 setOperationAction(ISD::FNEARBYINT, VT, Expand); 345 } 346 } 347 348 // We 'support' these types up to bitcast/load/store level, regardless of 349 // MVE integer-only / float support. Only doing FP data processing on the FP 350 // vector types is inhibited at integer-only level. 351 const MVT LongTypes[] = { MVT::v2i64, MVT::v2f64 }; 352 for (auto VT : LongTypes) { 353 addRegisterClass(VT, &ARM::MQPRRegClass); 354 setAllExpand(VT); 355 setOperationAction(ISD::INSERT_VECTOR_ELT, VT, Custom); 356 setOperationAction(ISD::EXTRACT_VECTOR_ELT, VT, Custom); 357 setOperationAction(ISD::BUILD_VECTOR, VT, Custom); 358 } 359 // We can do bitwise operations on v2i64 vectors 360 setOperationAction(ISD::AND, MVT::v2i64, Legal); 361 setOperationAction(ISD::OR, MVT::v2i64, Legal); 362 setOperationAction(ISD::XOR, MVT::v2i64, Legal); 363 364 // It is legal to extload from v4i8 to v4i16 or v4i32. 365 addAllExtLoads(MVT::v8i16, MVT::v8i8, Legal); 366 addAllExtLoads(MVT::v4i32, MVT::v4i16, Legal); 367 addAllExtLoads(MVT::v4i32, MVT::v4i8, Legal); 368 369 // Some truncating stores are legal too. 370 setTruncStoreAction(MVT::v4i32, MVT::v4i16, Legal); 371 setTruncStoreAction(MVT::v4i32, MVT::v4i8, Legal); 372 setTruncStoreAction(MVT::v8i16, MVT::v8i8, Legal); 373 374 // Pre and Post inc on these are legal, given the correct extends 375 for (unsigned im = (unsigned)ISD::PRE_INC; 376 im != (unsigned)ISD::LAST_INDEXED_MODE; ++im) { 377 setIndexedLoadAction(im, MVT::v8i8, Legal); 378 setIndexedStoreAction(im, MVT::v8i8, Legal); 379 setIndexedLoadAction(im, MVT::v4i8, Legal); 380 setIndexedStoreAction(im, MVT::v4i8, Legal); 381 setIndexedLoadAction(im, MVT::v4i16, Legal); 382 setIndexedStoreAction(im, MVT::v4i16, Legal); 383 } 384 385 // Predicate types 386 const MVT pTypes[] = {MVT::v16i1, MVT::v8i1, MVT::v4i1}; 387 for (auto VT : pTypes) { 388 addRegisterClass(VT, &ARM::VCCRRegClass); 389 setOperationAction(ISD::BUILD_VECTOR, VT, Custom); 390 setOperationAction(ISD::VECTOR_SHUFFLE, VT, Custom); 391 setOperationAction(ISD::EXTRACT_SUBVECTOR, VT, Custom); 392 setOperationAction(ISD::CONCAT_VECTORS, VT, Custom); 393 setOperationAction(ISD::INSERT_VECTOR_ELT, VT, Custom); 394 setOperationAction(ISD::EXTRACT_VECTOR_ELT, VT, Custom); 395 setOperationAction(ISD::SETCC, VT, Custom); 396 setOperationAction(ISD::SCALAR_TO_VECTOR, VT, Expand); 397 setOperationAction(ISD::LOAD, VT, Custom); 398 setOperationAction(ISD::STORE, VT, Custom); 399 } 400 } 401 402 ARMTargetLowering::ARMTargetLowering(const TargetMachine &TM, 403 const ARMSubtarget &STI) 404 : TargetLowering(TM), Subtarget(&STI) { 405 RegInfo = Subtarget->getRegisterInfo(); 406 Itins = Subtarget->getInstrItineraryData(); 407 408 setBooleanContents(ZeroOrOneBooleanContent); 409 setBooleanVectorContents(ZeroOrNegativeOneBooleanContent); 410 411 if (!Subtarget->isTargetDarwin() && !Subtarget->isTargetIOS() && 412 !Subtarget->isTargetWatchOS()) { 413 bool IsHFTarget = TM.Options.FloatABIType == FloatABI::Hard; 414 for (int LCID = 0; LCID < RTLIB::UNKNOWN_LIBCALL; ++LCID) 415 setLibcallCallingConv(static_cast<RTLIB::Libcall>(LCID), 416 IsHFTarget ? CallingConv::ARM_AAPCS_VFP 417 : CallingConv::ARM_AAPCS); 418 } 419 420 if (Subtarget->isTargetMachO()) { 421 // Uses VFP for Thumb libfuncs if available. 422 if (Subtarget->isThumb() && Subtarget->hasVFP2Base() && 423 Subtarget->hasARMOps() && !Subtarget->useSoftFloat()) { 424 static const struct { 425 const RTLIB::Libcall Op; 426 const char * const Name; 427 const ISD::CondCode Cond; 428 } LibraryCalls[] = { 429 // Single-precision floating-point arithmetic. 430 { RTLIB::ADD_F32, "__addsf3vfp", ISD::SETCC_INVALID }, 431 { RTLIB::SUB_F32, "__subsf3vfp", ISD::SETCC_INVALID }, 432 { RTLIB::MUL_F32, "__mulsf3vfp", ISD::SETCC_INVALID }, 433 { RTLIB::DIV_F32, "__divsf3vfp", ISD::SETCC_INVALID }, 434 435 // Double-precision floating-point arithmetic. 436 { RTLIB::ADD_F64, "__adddf3vfp", ISD::SETCC_INVALID }, 437 { RTLIB::SUB_F64, "__subdf3vfp", ISD::SETCC_INVALID }, 438 { RTLIB::MUL_F64, "__muldf3vfp", ISD::SETCC_INVALID }, 439 { RTLIB::DIV_F64, "__divdf3vfp", ISD::SETCC_INVALID }, 440 441 // Single-precision comparisons. 442 { RTLIB::OEQ_F32, "__eqsf2vfp", ISD::SETNE }, 443 { RTLIB::UNE_F32, "__nesf2vfp", ISD::SETNE }, 444 { RTLIB::OLT_F32, "__ltsf2vfp", ISD::SETNE }, 445 { RTLIB::OLE_F32, "__lesf2vfp", ISD::SETNE }, 446 { RTLIB::OGE_F32, "__gesf2vfp", ISD::SETNE }, 447 { RTLIB::OGT_F32, "__gtsf2vfp", ISD::SETNE }, 448 { RTLIB::UO_F32, "__unordsf2vfp", ISD::SETNE }, 449 { RTLIB::O_F32, "__unordsf2vfp", ISD::SETEQ }, 450 451 // Double-precision comparisons. 452 { RTLIB::OEQ_F64, "__eqdf2vfp", ISD::SETNE }, 453 { RTLIB::UNE_F64, "__nedf2vfp", ISD::SETNE }, 454 { RTLIB::OLT_F64, "__ltdf2vfp", ISD::SETNE }, 455 { RTLIB::OLE_F64, "__ledf2vfp", ISD::SETNE }, 456 { RTLIB::OGE_F64, "__gedf2vfp", ISD::SETNE }, 457 { RTLIB::OGT_F64, "__gtdf2vfp", ISD::SETNE }, 458 { RTLIB::UO_F64, "__unorddf2vfp", ISD::SETNE }, 459 { RTLIB::O_F64, "__unorddf2vfp", ISD::SETEQ }, 460 461 // Floating-point to integer conversions. 462 // i64 conversions are done via library routines even when generating VFP 463 // instructions, so use the same ones. 464 { RTLIB::FPTOSINT_F64_I32, "__fixdfsivfp", ISD::SETCC_INVALID }, 465 { RTLIB::FPTOUINT_F64_I32, "__fixunsdfsivfp", ISD::SETCC_INVALID }, 466 { RTLIB::FPTOSINT_F32_I32, "__fixsfsivfp", ISD::SETCC_INVALID }, 467 { RTLIB::FPTOUINT_F32_I32, "__fixunssfsivfp", ISD::SETCC_INVALID }, 468 469 // Conversions between floating types. 470 { RTLIB::FPROUND_F64_F32, "__truncdfsf2vfp", ISD::SETCC_INVALID }, 471 { RTLIB::FPEXT_F32_F64, "__extendsfdf2vfp", ISD::SETCC_INVALID }, 472 473 // Integer to floating-point conversions. 474 // i64 conversions are done via library routines even when generating VFP 475 // instructions, so use the same ones. 476 // FIXME: There appears to be some naming inconsistency in ARM libgcc: 477 // e.g., __floatunsidf vs. __floatunssidfvfp. 478 { RTLIB::SINTTOFP_I32_F64, "__floatsidfvfp", ISD::SETCC_INVALID }, 479 { RTLIB::UINTTOFP_I32_F64, "__floatunssidfvfp", ISD::SETCC_INVALID }, 480 { RTLIB::SINTTOFP_I32_F32, "__floatsisfvfp", ISD::SETCC_INVALID }, 481 { RTLIB::UINTTOFP_I32_F32, "__floatunssisfvfp", ISD::SETCC_INVALID }, 482 }; 483 484 for (const auto &LC : LibraryCalls) { 485 setLibcallName(LC.Op, LC.Name); 486 if (LC.Cond != ISD::SETCC_INVALID) 487 setCmpLibcallCC(LC.Op, LC.Cond); 488 } 489 } 490 } 491 492 // These libcalls are not available in 32-bit. 493 setLibcallName(RTLIB::SHL_I128, nullptr); 494 setLibcallName(RTLIB::SRL_I128, nullptr); 495 setLibcallName(RTLIB::SRA_I128, nullptr); 496 497 // RTLIB 498 if (Subtarget->isAAPCS_ABI() && 499 (Subtarget->isTargetAEABI() || Subtarget->isTargetGNUAEABI() || 500 Subtarget->isTargetMuslAEABI() || Subtarget->isTargetAndroid())) { 501 static const struct { 502 const RTLIB::Libcall Op; 503 const char * const Name; 504 const CallingConv::ID CC; 505 const ISD::CondCode Cond; 506 } LibraryCalls[] = { 507 // Double-precision floating-point arithmetic helper functions 508 // RTABI chapter 4.1.2, Table 2 509 { RTLIB::ADD_F64, "__aeabi_dadd", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 510 { RTLIB::DIV_F64, "__aeabi_ddiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 511 { RTLIB::MUL_F64, "__aeabi_dmul", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 512 { RTLIB::SUB_F64, "__aeabi_dsub", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 513 514 // Double-precision floating-point comparison helper functions 515 // RTABI chapter 4.1.2, Table 3 516 { RTLIB::OEQ_F64, "__aeabi_dcmpeq", CallingConv::ARM_AAPCS, ISD::SETNE }, 517 { RTLIB::UNE_F64, "__aeabi_dcmpeq", CallingConv::ARM_AAPCS, ISD::SETEQ }, 518 { RTLIB::OLT_F64, "__aeabi_dcmplt", CallingConv::ARM_AAPCS, ISD::SETNE }, 519 { RTLIB::OLE_F64, "__aeabi_dcmple", CallingConv::ARM_AAPCS, ISD::SETNE }, 520 { RTLIB::OGE_F64, "__aeabi_dcmpge", CallingConv::ARM_AAPCS, ISD::SETNE }, 521 { RTLIB::OGT_F64, "__aeabi_dcmpgt", CallingConv::ARM_AAPCS, ISD::SETNE }, 522 { RTLIB::UO_F64, "__aeabi_dcmpun", CallingConv::ARM_AAPCS, ISD::SETNE }, 523 { RTLIB::O_F64, "__aeabi_dcmpun", CallingConv::ARM_AAPCS, ISD::SETEQ }, 524 525 // Single-precision floating-point arithmetic helper functions 526 // RTABI chapter 4.1.2, Table 4 527 { RTLIB::ADD_F32, "__aeabi_fadd", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 528 { RTLIB::DIV_F32, "__aeabi_fdiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 529 { RTLIB::MUL_F32, "__aeabi_fmul", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 530 { RTLIB::SUB_F32, "__aeabi_fsub", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 531 532 // Single-precision floating-point comparison helper functions 533 // RTABI chapter 4.1.2, Table 5 534 { RTLIB::OEQ_F32, "__aeabi_fcmpeq", CallingConv::ARM_AAPCS, ISD::SETNE }, 535 { RTLIB::UNE_F32, "__aeabi_fcmpeq", CallingConv::ARM_AAPCS, ISD::SETEQ }, 536 { RTLIB::OLT_F32, "__aeabi_fcmplt", CallingConv::ARM_AAPCS, ISD::SETNE }, 537 { RTLIB::OLE_F32, "__aeabi_fcmple", CallingConv::ARM_AAPCS, ISD::SETNE }, 538 { RTLIB::OGE_F32, "__aeabi_fcmpge", CallingConv::ARM_AAPCS, ISD::SETNE }, 539 { RTLIB::OGT_F32, "__aeabi_fcmpgt", CallingConv::ARM_AAPCS, ISD::SETNE }, 540 { RTLIB::UO_F32, "__aeabi_fcmpun", CallingConv::ARM_AAPCS, ISD::SETNE }, 541 { RTLIB::O_F32, "__aeabi_fcmpun", CallingConv::ARM_AAPCS, ISD::SETEQ }, 542 543 // Floating-point to integer conversions. 544 // RTABI chapter 4.1.2, Table 6 545 { RTLIB::FPTOSINT_F64_I32, "__aeabi_d2iz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 546 { RTLIB::FPTOUINT_F64_I32, "__aeabi_d2uiz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 547 { RTLIB::FPTOSINT_F64_I64, "__aeabi_d2lz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 548 { RTLIB::FPTOUINT_F64_I64, "__aeabi_d2ulz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 549 { RTLIB::FPTOSINT_F32_I32, "__aeabi_f2iz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 550 { RTLIB::FPTOUINT_F32_I32, "__aeabi_f2uiz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 551 { RTLIB::FPTOSINT_F32_I64, "__aeabi_f2lz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 552 { RTLIB::FPTOUINT_F32_I64, "__aeabi_f2ulz", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 553 554 // Conversions between floating types. 555 // RTABI chapter 4.1.2, Table 7 556 { RTLIB::FPROUND_F64_F32, "__aeabi_d2f", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 557 { RTLIB::FPROUND_F64_F16, "__aeabi_d2h", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 558 { RTLIB::FPEXT_F32_F64, "__aeabi_f2d", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 559 560 // Integer to floating-point conversions. 561 // RTABI chapter 4.1.2, Table 8 562 { RTLIB::SINTTOFP_I32_F64, "__aeabi_i2d", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 563 { RTLIB::UINTTOFP_I32_F64, "__aeabi_ui2d", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 564 { RTLIB::SINTTOFP_I64_F64, "__aeabi_l2d", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 565 { RTLIB::UINTTOFP_I64_F64, "__aeabi_ul2d", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 566 { RTLIB::SINTTOFP_I32_F32, "__aeabi_i2f", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 567 { RTLIB::UINTTOFP_I32_F32, "__aeabi_ui2f", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 568 { RTLIB::SINTTOFP_I64_F32, "__aeabi_l2f", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 569 { RTLIB::UINTTOFP_I64_F32, "__aeabi_ul2f", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 570 571 // Long long helper functions 572 // RTABI chapter 4.2, Table 9 573 { RTLIB::MUL_I64, "__aeabi_lmul", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 574 { RTLIB::SHL_I64, "__aeabi_llsl", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 575 { RTLIB::SRL_I64, "__aeabi_llsr", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 576 { RTLIB::SRA_I64, "__aeabi_lasr", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 577 578 // Integer division functions 579 // RTABI chapter 4.3.1 580 { RTLIB::SDIV_I8, "__aeabi_idiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 581 { RTLIB::SDIV_I16, "__aeabi_idiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 582 { RTLIB::SDIV_I32, "__aeabi_idiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 583 { RTLIB::SDIV_I64, "__aeabi_ldivmod", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 584 { RTLIB::UDIV_I8, "__aeabi_uidiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 585 { RTLIB::UDIV_I16, "__aeabi_uidiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 586 { RTLIB::UDIV_I32, "__aeabi_uidiv", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 587 { RTLIB::UDIV_I64, "__aeabi_uldivmod", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 588 }; 589 590 for (const auto &LC : LibraryCalls) { 591 setLibcallName(LC.Op, LC.Name); 592 setLibcallCallingConv(LC.Op, LC.CC); 593 if (LC.Cond != ISD::SETCC_INVALID) 594 setCmpLibcallCC(LC.Op, LC.Cond); 595 } 596 597 // EABI dependent RTLIB 598 if (TM.Options.EABIVersion == EABI::EABI4 || 599 TM.Options.EABIVersion == EABI::EABI5) { 600 static const struct { 601 const RTLIB::Libcall Op; 602 const char *const Name; 603 const CallingConv::ID CC; 604 const ISD::CondCode Cond; 605 } MemOpsLibraryCalls[] = { 606 // Memory operations 607 // RTABI chapter 4.3.4 608 { RTLIB::MEMCPY, "__aeabi_memcpy", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 609 { RTLIB::MEMMOVE, "__aeabi_memmove", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 610 { RTLIB::MEMSET, "__aeabi_memset", CallingConv::ARM_AAPCS, ISD::SETCC_INVALID }, 611 }; 612 613 for (const auto &LC : MemOpsLibraryCalls) { 614 setLibcallName(LC.Op, LC.Name); 615 setLibcallCallingConv(LC.Op, LC.CC); 616 if (LC.Cond != ISD::SETCC_INVALID) 617 setCmpLibcallCC(LC.Op, LC.Cond); 618 } 619 } 620 } 621 622 if (Subtarget->isTargetWindows()) { 623 static const struct { 624 const RTLIB::Libcall Op; 625 const char * const Name; 626 const CallingConv::ID CC; 627 } LibraryCalls[] = { 628 { RTLIB::FPTOSINT_F32_I64, "__stoi64", CallingConv::ARM_AAPCS_VFP }, 629 { RTLIB::FPTOSINT_F64_I64, "__dtoi64", CallingConv::ARM_AAPCS_VFP }, 630 { RTLIB::FPTOUINT_F32_I64, "__stou64", CallingConv::ARM_AAPCS_VFP }, 631 { RTLIB::FPTOUINT_F64_I64, "__dtou64", CallingConv::ARM_AAPCS_VFP }, 632 { RTLIB::SINTTOFP_I64_F32, "__i64tos", CallingConv::ARM_AAPCS_VFP }, 633 { RTLIB::SINTTOFP_I64_F64, "__i64tod", CallingConv::ARM_AAPCS_VFP }, 634 { RTLIB::UINTTOFP_I64_F32, "__u64tos", CallingConv::ARM_AAPCS_VFP }, 635 { RTLIB::UINTTOFP_I64_F64, "__u64tod", CallingConv::ARM_AAPCS_VFP }, 636 }; 637 638 for (const auto &LC : LibraryCalls) { 639 setLibcallName(LC.Op, LC.Name); 640 setLibcallCallingConv(LC.Op, LC.CC); 641 } 642 } 643 644 // Use divmod compiler-rt calls for iOS 5.0 and later. 645 if (Subtarget->isTargetMachO() && 646 !(Subtarget->isTargetIOS() && 647 Subtarget->getTargetTriple().isOSVersionLT(5, 0))) { 648 setLibcallName(RTLIB::SDIVREM_I32, "__divmodsi4"); 649 setLibcallName(RTLIB::UDIVREM_I32, "__udivmodsi4"); 650 } 651 652 // The half <-> float conversion functions are always soft-float on 653 // non-watchos platforms, but are needed for some targets which use a 654 // hard-float calling convention by default. 655 if (!Subtarget->isTargetWatchABI()) { 656 if (Subtarget->isAAPCS_ABI()) { 657 setLibcallCallingConv(RTLIB::FPROUND_F32_F16, CallingConv::ARM_AAPCS); 658 setLibcallCallingConv(RTLIB::FPROUND_F64_F16, CallingConv::ARM_AAPCS); 659 setLibcallCallingConv(RTLIB::FPEXT_F16_F32, CallingConv::ARM_AAPCS); 660 } else { 661 setLibcallCallingConv(RTLIB::FPROUND_F32_F16, CallingConv::ARM_APCS); 662 setLibcallCallingConv(RTLIB::FPROUND_F64_F16, CallingConv::ARM_APCS); 663 setLibcallCallingConv(RTLIB::FPEXT_F16_F32, CallingConv::ARM_APCS); 664 } 665 } 666 667 // In EABI, these functions have an __aeabi_ prefix, but in GNUEABI they have 668 // a __gnu_ prefix (which is the default). 669 if (Subtarget->isTargetAEABI()) { 670 static const struct { 671 const RTLIB::Libcall Op; 672 const char * const Name; 673 const CallingConv::ID CC; 674 } LibraryCalls[] = { 675 { RTLIB::FPROUND_F32_F16, "__aeabi_f2h", CallingConv::ARM_AAPCS }, 676 { RTLIB::FPROUND_F64_F16, "__aeabi_d2h", CallingConv::ARM_AAPCS }, 677 { RTLIB::FPEXT_F16_F32, "__aeabi_h2f", CallingConv::ARM_AAPCS }, 678 }; 679 680 for (const auto &LC : LibraryCalls) { 681 setLibcallName(LC.Op, LC.Name); 682 setLibcallCallingConv(LC.Op, LC.CC); 683 } 684 } 685 686 if (Subtarget->isThumb1Only()) 687 addRegisterClass(MVT::i32, &ARM::tGPRRegClass); 688 else 689 addRegisterClass(MVT::i32, &ARM::GPRRegClass); 690 691 if (!Subtarget->useSoftFloat() && !Subtarget->isThumb1Only() && 692 Subtarget->hasFPRegs()) { 693 addRegisterClass(MVT::f32, &ARM::SPRRegClass); 694 addRegisterClass(MVT::f64, &ARM::DPRRegClass); 695 if (!Subtarget->hasVFP2Base()) 696 setAllExpand(MVT::f32); 697 if (!Subtarget->hasFP64()) 698 setAllExpand(MVT::f64); 699 } 700 701 if (Subtarget->hasFullFP16()) { 702 addRegisterClass(MVT::f16, &ARM::HPRRegClass); 703 setOperationAction(ISD::BITCAST, MVT::i16, Custom); 704 setOperationAction(ISD::BITCAST, MVT::i32, Custom); 705 setOperationAction(ISD::BITCAST, MVT::f16, Custom); 706 707 setOperationAction(ISD::FMINNUM, MVT::f16, Legal); 708 setOperationAction(ISD::FMAXNUM, MVT::f16, Legal); 709 } 710 711 for (MVT VT : MVT::fixedlen_vector_valuetypes()) { 712 for (MVT InnerVT : MVT::fixedlen_vector_valuetypes()) { 713 setTruncStoreAction(VT, InnerVT, Expand); 714 addAllExtLoads(VT, InnerVT, Expand); 715 } 716 717 setOperationAction(ISD::MULHS, VT, Expand); 718 setOperationAction(ISD::SMUL_LOHI, VT, Expand); 719 setOperationAction(ISD::MULHU, VT, Expand); 720 setOperationAction(ISD::UMUL_LOHI, VT, Expand); 721 722 setOperationAction(ISD::BSWAP, VT, Expand); 723 } 724 725 setOperationAction(ISD::ConstantFP, MVT::f32, Custom); 726 setOperationAction(ISD::ConstantFP, MVT::f64, Custom); 727 728 setOperationAction(ISD::READ_REGISTER, MVT::i64, Custom); 729 setOperationAction(ISD::WRITE_REGISTER, MVT::i64, Custom); 730 731 if (Subtarget->hasMVEIntegerOps()) 732 addMVEVectorTypes(Subtarget->hasMVEFloatOps()); 733 734 // Combine low-overhead loop intrinsics so that we can lower i1 types. 735 if (Subtarget->hasLOB()) { 736 setTargetDAGCombine(ISD::BRCOND); 737 setTargetDAGCombine(ISD::BR_CC); 738 } 739 740 if (Subtarget->hasNEON()) { 741 addDRTypeForNEON(MVT::v2f32); 742 addDRTypeForNEON(MVT::v8i8); 743 addDRTypeForNEON(MVT::v4i16); 744 addDRTypeForNEON(MVT::v2i32); 745 addDRTypeForNEON(MVT::v1i64); 746 747 addQRTypeForNEON(MVT::v4f32); 748 addQRTypeForNEON(MVT::v2f64); 749 addQRTypeForNEON(MVT::v16i8); 750 addQRTypeForNEON(MVT::v8i16); 751 addQRTypeForNEON(MVT::v4i32); 752 addQRTypeForNEON(MVT::v2i64); 753 754 if (Subtarget->hasFullFP16()) { 755 addQRTypeForNEON(MVT::v8f16); 756 addDRTypeForNEON(MVT::v4f16); 757 } 758 } 759 760 if (Subtarget->hasMVEIntegerOps() || Subtarget->hasNEON()) { 761 // v2f64 is legal so that QR subregs can be extracted as f64 elements, but 762 // none of Neon, MVE or VFP supports any arithmetic operations on it. 763 setOperationAction(ISD::FADD, MVT::v2f64, Expand); 764 setOperationAction(ISD::FSUB, MVT::v2f64, Expand); 765 setOperationAction(ISD::FMUL, MVT::v2f64, Expand); 766 // FIXME: Code duplication: FDIV and FREM are expanded always, see 767 // ARMTargetLowering::addTypeForNEON method for details. 768 setOperationAction(ISD::FDIV, MVT::v2f64, Expand); 769 setOperationAction(ISD::FREM, MVT::v2f64, Expand); 770 // FIXME: Create unittest. 771 // In another words, find a way when "copysign" appears in DAG with vector 772 // operands. 773 setOperationAction(ISD::FCOPYSIGN, MVT::v2f64, Expand); 774 // FIXME: Code duplication: SETCC has custom operation action, see 775 // ARMTargetLowering::addTypeForNEON method for details. 776 setOperationAction(ISD::SETCC, MVT::v2f64, Expand); 777 // FIXME: Create unittest for FNEG and for FABS. 778 setOperationAction(ISD::FNEG, MVT::v2f64, Expand); 779 setOperationAction(ISD::FABS, MVT::v2f64, Expand); 780 setOperationAction(ISD::FSQRT, MVT::v2f64, Expand); 781 setOperationAction(ISD::FSIN, MVT::v2f64, Expand); 782 setOperationAction(ISD::FCOS, MVT::v2f64, Expand); 783 setOperationAction(ISD::FPOW, MVT::v2f64, Expand); 784 setOperationAction(ISD::FLOG, MVT::v2f64, Expand); 785 setOperationAction(ISD::FLOG2, MVT::v2f64, Expand); 786 setOperationAction(ISD::FLOG10, MVT::v2f64, Expand); 787 setOperationAction(ISD::FEXP, MVT::v2f64, Expand); 788 setOperationAction(ISD::FEXP2, MVT::v2f64, Expand); 789 // FIXME: Create unittest for FCEIL, FTRUNC, FRINT, FNEARBYINT, FFLOOR. 790 setOperationAction(ISD::FCEIL, MVT::v2f64, Expand); 791 setOperationAction(ISD::FTRUNC, MVT::v2f64, Expand); 792 setOperationAction(ISD::FRINT, MVT::v2f64, Expand); 793 setOperationAction(ISD::FNEARBYINT, MVT::v2f64, Expand); 794 setOperationAction(ISD::FFLOOR, MVT::v2f64, Expand); 795 setOperationAction(ISD::FMA, MVT::v2f64, Expand); 796 } 797 798 if (Subtarget->hasNEON()) { 799 // The same with v4f32. But keep in mind that vadd, vsub, vmul are natively 800 // supported for v4f32. 801 setOperationAction(ISD::FSQRT, MVT::v4f32, Expand); 802 setOperationAction(ISD::FSIN, MVT::v4f32, Expand); 803 setOperationAction(ISD::FCOS, MVT::v4f32, Expand); 804 setOperationAction(ISD::FPOW, MVT::v4f32, Expand); 805 setOperationAction(ISD::FLOG, MVT::v4f32, Expand); 806 setOperationAction(ISD::FLOG2, MVT::v4f32, Expand); 807 setOperationAction(ISD::FLOG10, MVT::v4f32, Expand); 808 setOperationAction(ISD::FEXP, MVT::v4f32, Expand); 809 setOperationAction(ISD::FEXP2, MVT::v4f32, Expand); 810 setOperationAction(ISD::FCEIL, MVT::v4f32, Expand); 811 setOperationAction(ISD::FTRUNC, MVT::v4f32, Expand); 812 setOperationAction(ISD::FRINT, MVT::v4f32, Expand); 813 setOperationAction(ISD::FNEARBYINT, MVT::v4f32, Expand); 814 setOperationAction(ISD::FFLOOR, MVT::v4f32, Expand); 815 816 // Mark v2f32 intrinsics. 817 setOperationAction(ISD::FSQRT, MVT::v2f32, Expand); 818 setOperationAction(ISD::FSIN, MVT::v2f32, Expand); 819 setOperationAction(ISD::FCOS, MVT::v2f32, Expand); 820 setOperationAction(ISD::FPOW, MVT::v2f32, Expand); 821 setOperationAction(ISD::FLOG, MVT::v2f32, Expand); 822 setOperationAction(ISD::FLOG2, MVT::v2f32, Expand); 823 setOperationAction(ISD::FLOG10, MVT::v2f32, Expand); 824 setOperationAction(ISD::FEXP, MVT::v2f32, Expand); 825 setOperationAction(ISD::FEXP2, MVT::v2f32, Expand); 826 setOperationAction(ISD::FCEIL, MVT::v2f32, Expand); 827 setOperationAction(ISD::FTRUNC, MVT::v2f32, Expand); 828 setOperationAction(ISD::FRINT, MVT::v2f32, Expand); 829 setOperationAction(ISD::FNEARBYINT, MVT::v2f32, Expand); 830 setOperationAction(ISD::FFLOOR, MVT::v2f32, Expand); 831 832 // Neon does not support some operations on v1i64 and v2i64 types. 833 setOperationAction(ISD::MUL, MVT::v1i64, Expand); 834 // Custom handling for some quad-vector types to detect VMULL. 835 setOperationAction(ISD::MUL, MVT::v8i16, Custom); 836 setOperationAction(ISD::MUL, MVT::v4i32, Custom); 837 setOperationAction(ISD::MUL, MVT::v2i64, Custom); 838 // Custom handling for some vector types to avoid expensive expansions 839 setOperationAction(ISD::SDIV, MVT::v4i16, Custom); 840 setOperationAction(ISD::SDIV, MVT::v8i8, Custom); 841 setOperationAction(ISD::UDIV, MVT::v4i16, Custom); 842 setOperationAction(ISD::UDIV, MVT::v8i8, Custom); 843 // Neon does not have single instruction SINT_TO_FP and UINT_TO_FP with 844 // a destination type that is wider than the source, and nor does 845 // it have a FP_TO_[SU]INT instruction with a narrower destination than 846 // source. 847 setOperationAction(ISD::SINT_TO_FP, MVT::v4i16, Custom); 848 setOperationAction(ISD::SINT_TO_FP, MVT::v8i16, Custom); 849 setOperationAction(ISD::UINT_TO_FP, MVT::v4i16, Custom); 850 setOperationAction(ISD::UINT_TO_FP, MVT::v8i16, Custom); 851 setOperationAction(ISD::FP_TO_UINT, MVT::v4i16, Custom); 852 setOperationAction(ISD::FP_TO_UINT, MVT::v8i16, Custom); 853 setOperationAction(ISD::FP_TO_SINT, MVT::v4i16, Custom); 854 setOperationAction(ISD::FP_TO_SINT, MVT::v8i16, Custom); 855 856 setOperationAction(ISD::FP_ROUND, MVT::v2f32, Expand); 857 setOperationAction(ISD::FP_EXTEND, MVT::v2f64, Expand); 858 859 // NEON does not have single instruction CTPOP for vectors with element 860 // types wider than 8-bits. However, custom lowering can leverage the 861 // v8i8/v16i8 vcnt instruction. 862 setOperationAction(ISD::CTPOP, MVT::v2i32, Custom); 863 setOperationAction(ISD::CTPOP, MVT::v4i32, Custom); 864 setOperationAction(ISD::CTPOP, MVT::v4i16, Custom); 865 setOperationAction(ISD::CTPOP, MVT::v8i16, Custom); 866 setOperationAction(ISD::CTPOP, MVT::v1i64, Custom); 867 setOperationAction(ISD::CTPOP, MVT::v2i64, Custom); 868 869 setOperationAction(ISD::CTLZ, MVT::v1i64, Expand); 870 setOperationAction(ISD::CTLZ, MVT::v2i64, Expand); 871 872 // NEON does not have single instruction CTTZ for vectors. 873 setOperationAction(ISD::CTTZ, MVT::v8i8, Custom); 874 setOperationAction(ISD::CTTZ, MVT::v4i16, Custom); 875 setOperationAction(ISD::CTTZ, MVT::v2i32, Custom); 876 setOperationAction(ISD::CTTZ, MVT::v1i64, Custom); 877 878 setOperationAction(ISD::CTTZ, MVT::v16i8, Custom); 879 setOperationAction(ISD::CTTZ, MVT::v8i16, Custom); 880 setOperationAction(ISD::CTTZ, MVT::v4i32, Custom); 881 setOperationAction(ISD::CTTZ, MVT::v2i64, Custom); 882 883 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v8i8, Custom); 884 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v4i16, Custom); 885 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v2i32, Custom); 886 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v1i64, Custom); 887 888 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v16i8, Custom); 889 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v8i16, Custom); 890 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v4i32, Custom); 891 setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::v2i64, Custom); 892 893 // NEON only has FMA instructions as of VFP4. 894 if (!Subtarget->hasVFP4Base()) { 895 setOperationAction(ISD::FMA, MVT::v2f32, Expand); 896 setOperationAction(ISD::FMA, MVT::v4f32, Expand); 897 } 898 899 setTargetDAGCombine(ISD::INTRINSIC_VOID); 900 setTargetDAGCombine(ISD::INTRINSIC_W_CHAIN); 901 setTargetDAGCombine(ISD::INTRINSIC_WO_CHAIN); 902 setTargetDAGCombine(ISD::SHL); 903 setTargetDAGCombine(ISD::SRL); 904 setTargetDAGCombine(ISD::SRA); 905 setTargetDAGCombine(ISD::FP_TO_SINT); 906 setTargetDAGCombine(ISD::FP_TO_UINT); 907 setTargetDAGCombine(ISD::FDIV); 908 setTargetDAGCombine(ISD::LOAD); 909 910 // It is legal to extload from v4i8 to v4i16 or v4i32. 911 for (MVT Ty : {MVT::v8i8, MVT::v4i8, MVT::v2i8, MVT::v4i16, MVT::v2i16, 912 MVT::v2i32}) { 913 for (MVT VT : MVT::integer_fixedlen_vector_valuetypes()) { 914 setLoadExtAction(ISD::EXTLOAD, VT, Ty, Legal); 915 setLoadExtAction(ISD::ZEXTLOAD, VT, Ty, Legal); 916 setLoadExtAction(ISD::SEXTLOAD, VT, Ty, Legal); 917 } 918 } 919 } 920 921 if (Subtarget->hasNEON() || Subtarget->hasMVEIntegerOps()) { 922 setTargetDAGCombine(ISD::BUILD_VECTOR); 923 setTargetDAGCombine(ISD::VECTOR_SHUFFLE); 924 setTargetDAGCombine(ISD::INSERT_VECTOR_ELT); 925 setTargetDAGCombine(ISD::STORE); 926 setTargetDAGCombine(ISD::SIGN_EXTEND); 927 setTargetDAGCombine(ISD::ZERO_EXTEND); 928 setTargetDAGCombine(ISD::ANY_EXTEND); 929 } 930 931 if (!Subtarget->hasFP64()) { 932 // When targeting a floating-point unit with only single-precision 933 // operations, f64 is legal for the few double-precision instructions which 934 // are present However, no double-precision operations other than moves, 935 // loads and stores are provided by the hardware. 936 setOperationAction(ISD::FADD, MVT::f64, Expand); 937 setOperationAction(ISD::FSUB, MVT::f64, Expand); 938 setOperationAction(ISD::FMUL, MVT::f64, Expand); 939 setOperationAction(ISD::FMA, MVT::f64, Expand); 940 setOperationAction(ISD::FDIV, MVT::f64, Expand); 941 setOperationAction(ISD::FREM, MVT::f64, Expand); 942 setOperationAction(ISD::FCOPYSIGN, MVT::f64, Expand); 943 setOperationAction(ISD::FGETSIGN, MVT::f64, Expand); 944 setOperationAction(ISD::FNEG, MVT::f64, Expand); 945 setOperationAction(ISD::FABS, MVT::f64, Expand); 946 setOperationAction(ISD::FSQRT, MVT::f64, Expand); 947 setOperationAction(ISD::FSIN, MVT::f64, Expand); 948 setOperationAction(ISD::FCOS, MVT::f64, Expand); 949 setOperationAction(ISD::FPOW, MVT::f64, Expand); 950 setOperationAction(ISD::FLOG, MVT::f64, Expand); 951 setOperationAction(ISD::FLOG2, MVT::f64, Expand); 952 setOperationAction(ISD::FLOG10, MVT::f64, Expand); 953 setOperationAction(ISD::FEXP, MVT::f64, Expand); 954 setOperationAction(ISD::FEXP2, MVT::f64, Expand); 955 setOperationAction(ISD::FCEIL, MVT::f64, Expand); 956 setOperationAction(ISD::FTRUNC, MVT::f64, Expand); 957 setOperationAction(ISD::FRINT, MVT::f64, Expand); 958 setOperationAction(ISD::FNEARBYINT, MVT::f64, Expand); 959 setOperationAction(ISD::FFLOOR, MVT::f64, Expand); 960 setOperationAction(ISD::SINT_TO_FP, MVT::i32, Custom); 961 setOperationAction(ISD::UINT_TO_FP, MVT::i32, Custom); 962 setOperationAction(ISD::FP_TO_SINT, MVT::i32, Custom); 963 setOperationAction(ISD::FP_TO_UINT, MVT::i32, Custom); 964 setOperationAction(ISD::FP_TO_SINT, MVT::f64, Custom); 965 setOperationAction(ISD::FP_TO_UINT, MVT::f64, Custom); 966 setOperationAction(ISD::FP_ROUND, MVT::f32, Custom); 967 } 968 969 if (!Subtarget->hasFP64() || !Subtarget->hasFPARMv8Base()) { 970 setOperationAction(ISD::FP_EXTEND, MVT::f64, Custom); 971 if (Subtarget->hasFullFP16()) 972 setOperationAction(ISD::FP_ROUND, MVT::f16, Custom); 973 } 974 975 if (!Subtarget->hasFP16()) 976 setOperationAction(ISD::FP_EXTEND, MVT::f32, Custom); 977 978 if (!Subtarget->hasFP64()) 979 setOperationAction(ISD::FP_ROUND, MVT::f32, Custom); 980 981 computeRegisterProperties(Subtarget->getRegisterInfo()); 982 983 // ARM does not have floating-point extending loads. 984 for (MVT VT : MVT::fp_valuetypes()) { 985 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f32, Expand); 986 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f16, Expand); 987 } 988 989 // ... or truncating stores 990 setTruncStoreAction(MVT::f64, MVT::f32, Expand); 991 setTruncStoreAction(MVT::f32, MVT::f16, Expand); 992 setTruncStoreAction(MVT::f64, MVT::f16, Expand); 993 994 // ARM does not have i1 sign extending load. 995 for (MVT VT : MVT::integer_valuetypes()) 996 setLoadExtAction(ISD::SEXTLOAD, VT, MVT::i1, Promote); 997 998 // ARM supports all 4 flavors of integer indexed load / store. 999 if (!Subtarget->isThumb1Only()) { 1000 for (unsigned im = (unsigned)ISD::PRE_INC; 1001 im != (unsigned)ISD::LAST_INDEXED_MODE; ++im) { 1002 setIndexedLoadAction(im, MVT::i1, Legal); 1003 setIndexedLoadAction(im, MVT::i8, Legal); 1004 setIndexedLoadAction(im, MVT::i16, Legal); 1005 setIndexedLoadAction(im, MVT::i32, Legal); 1006 setIndexedStoreAction(im, MVT::i1, Legal); 1007 setIndexedStoreAction(im, MVT::i8, Legal); 1008 setIndexedStoreAction(im, MVT::i16, Legal); 1009 setIndexedStoreAction(im, MVT::i32, Legal); 1010 } 1011 } else { 1012 // Thumb-1 has limited post-inc load/store support - LDM r0!, {r1}. 1013 setIndexedLoadAction(ISD::POST_INC, MVT::i32, Legal); 1014 setIndexedStoreAction(ISD::POST_INC, MVT::i32, Legal); 1015 } 1016 1017 setOperationAction(ISD::SADDO, MVT::i32, Custom); 1018 setOperationAction(ISD::UADDO, MVT::i32, Custom); 1019 setOperationAction(ISD::SSUBO, MVT::i32, Custom); 1020 setOperationAction(ISD::USUBO, MVT::i32, Custom); 1021 1022 setOperationAction(ISD::ADDCARRY, MVT::i32, Custom); 1023 setOperationAction(ISD::SUBCARRY, MVT::i32, Custom); 1024 if (Subtarget->hasDSP()) { 1025 setOperationAction(ISD::SADDSAT, MVT::i8, Custom); 1026 setOperationAction(ISD::SSUBSAT, MVT::i8, Custom); 1027 setOperationAction(ISD::SADDSAT, MVT::i16, Custom); 1028 setOperationAction(ISD::SSUBSAT, MVT::i16, Custom); 1029 } 1030 if (Subtarget->hasBaseDSP()) { 1031 setOperationAction(ISD::SADDSAT, MVT::i32, Legal); 1032 setOperationAction(ISD::SSUBSAT, MVT::i32, Legal); 1033 } 1034 1035 // i64 operation support. 1036 setOperationAction(ISD::MUL, MVT::i64, Expand); 1037 setOperationAction(ISD::MULHU, MVT::i32, Expand); 1038 if (Subtarget->isThumb1Only()) { 1039 setOperationAction(ISD::UMUL_LOHI, MVT::i32, Expand); 1040 setOperationAction(ISD::SMUL_LOHI, MVT::i32, Expand); 1041 } 1042 if (Subtarget->isThumb1Only() || !Subtarget->hasV6Ops() 1043 || (Subtarget->isThumb2() && !Subtarget->hasDSP())) 1044 setOperationAction(ISD::MULHS, MVT::i32, Expand); 1045 1046 setOperationAction(ISD::SHL_PARTS, MVT::i32, Custom); 1047 setOperationAction(ISD::SRA_PARTS, MVT::i32, Custom); 1048 setOperationAction(ISD::SRL_PARTS, MVT::i32, Custom); 1049 setOperationAction(ISD::SRL, MVT::i64, Custom); 1050 setOperationAction(ISD::SRA, MVT::i64, Custom); 1051 setOperationAction(ISD::INTRINSIC_VOID, MVT::Other, Custom); 1052 setOperationAction(ISD::INTRINSIC_WO_CHAIN, MVT::i64, Custom); 1053 1054 // MVE lowers 64 bit shifts to lsll and lsrl 1055 // assuming that ISD::SRL and SRA of i64 are already marked custom 1056 if (Subtarget->hasMVEIntegerOps()) 1057 setOperationAction(ISD::SHL, MVT::i64, Custom); 1058 1059 // Expand to __aeabi_l{lsl,lsr,asr} calls for Thumb1. 1060 if (Subtarget->isThumb1Only()) { 1061 setOperationAction(ISD::SHL_PARTS, MVT::i32, Expand); 1062 setOperationAction(ISD::SRA_PARTS, MVT::i32, Expand); 1063 setOperationAction(ISD::SRL_PARTS, MVT::i32, Expand); 1064 } 1065 1066 if (!Subtarget->isThumb1Only() && Subtarget->hasV6T2Ops()) 1067 setOperationAction(ISD::BITREVERSE, MVT::i32, Legal); 1068 1069 // ARM does not have ROTL. 1070 setOperationAction(ISD::ROTL, MVT::i32, Expand); 1071 for (MVT VT : MVT::fixedlen_vector_valuetypes()) { 1072 setOperationAction(ISD::ROTL, VT, Expand); 1073 setOperationAction(ISD::ROTR, VT, Expand); 1074 } 1075 setOperationAction(ISD::CTTZ, MVT::i32, Custom); 1076 setOperationAction(ISD::CTPOP, MVT::i32, Expand); 1077 if (!Subtarget->hasV5TOps() || Subtarget->isThumb1Only()) { 1078 setOperationAction(ISD::CTLZ, MVT::i32, Expand); 1079 setOperationAction(ISD::CTLZ_ZERO_UNDEF, MVT::i32, LibCall); 1080 } 1081 1082 // @llvm.readcyclecounter requires the Performance Monitors extension. 1083 // Default to the 0 expansion on unsupported platforms. 1084 // FIXME: Technically there are older ARM CPUs that have 1085 // implementation-specific ways of obtaining this information. 1086 if (Subtarget->hasPerfMon()) 1087 setOperationAction(ISD::READCYCLECOUNTER, MVT::i64, Custom); 1088 1089 // Only ARMv6 has BSWAP. 1090 if (!Subtarget->hasV6Ops()) 1091 setOperationAction(ISD::BSWAP, MVT::i32, Expand); 1092 1093 bool hasDivide = Subtarget->isThumb() ? Subtarget->hasDivideInThumbMode() 1094 : Subtarget->hasDivideInARMMode(); 1095 if (!hasDivide) { 1096 // These are expanded into libcalls if the cpu doesn't have HW divider. 1097 setOperationAction(ISD::SDIV, MVT::i32, LibCall); 1098 setOperationAction(ISD::UDIV, MVT::i32, LibCall); 1099 } 1100 1101 if (Subtarget->isTargetWindows() && !Subtarget->hasDivideInThumbMode()) { 1102 setOperationAction(ISD::SDIV, MVT::i32, Custom); 1103 setOperationAction(ISD::UDIV, MVT::i32, Custom); 1104 1105 setOperationAction(ISD::SDIV, MVT::i64, Custom); 1106 setOperationAction(ISD::UDIV, MVT::i64, Custom); 1107 } 1108 1109 setOperationAction(ISD::SREM, MVT::i32, Expand); 1110 setOperationAction(ISD::UREM, MVT::i32, Expand); 1111 1112 // Register based DivRem for AEABI (RTABI 4.2) 1113 if (Subtarget->isTargetAEABI() || Subtarget->isTargetAndroid() || 1114 Subtarget->isTargetGNUAEABI() || Subtarget->isTargetMuslAEABI() || 1115 Subtarget->isTargetWindows()) { 1116 setOperationAction(ISD::SREM, MVT::i64, Custom); 1117 setOperationAction(ISD::UREM, MVT::i64, Custom); 1118 HasStandaloneRem = false; 1119 1120 if (Subtarget->isTargetWindows()) { 1121 const struct { 1122 const RTLIB::Libcall Op; 1123 const char * const Name; 1124 const CallingConv::ID CC; 1125 } LibraryCalls[] = { 1126 { RTLIB::SDIVREM_I8, "__rt_sdiv", CallingConv::ARM_AAPCS }, 1127 { RTLIB::SDIVREM_I16, "__rt_sdiv", CallingConv::ARM_AAPCS }, 1128 { RTLIB::SDIVREM_I32, "__rt_sdiv", CallingConv::ARM_AAPCS }, 1129 { RTLIB::SDIVREM_I64, "__rt_sdiv64", CallingConv::ARM_AAPCS }, 1130 1131 { RTLIB::UDIVREM_I8, "__rt_udiv", CallingConv::ARM_AAPCS }, 1132 { RTLIB::UDIVREM_I16, "__rt_udiv", CallingConv::ARM_AAPCS }, 1133 { RTLIB::UDIVREM_I32, "__rt_udiv", CallingConv::ARM_AAPCS }, 1134 { RTLIB::UDIVREM_I64, "__rt_udiv64", CallingConv::ARM_AAPCS }, 1135 }; 1136 1137 for (const auto &LC : LibraryCalls) { 1138 setLibcallName(LC.Op, LC.Name); 1139 setLibcallCallingConv(LC.Op, LC.CC); 1140 } 1141 } else { 1142 const struct { 1143 const RTLIB::Libcall Op; 1144 const char * const Name; 1145 const CallingConv::ID CC; 1146 } LibraryCalls[] = { 1147 { RTLIB::SDIVREM_I8, "__aeabi_idivmod", CallingConv::ARM_AAPCS }, 1148 { RTLIB::SDIVREM_I16, "__aeabi_idivmod", CallingConv::ARM_AAPCS }, 1149 { RTLIB::SDIVREM_I32, "__aeabi_idivmod", CallingConv::ARM_AAPCS }, 1150 { RTLIB::SDIVREM_I64, "__aeabi_ldivmod", CallingConv::ARM_AAPCS }, 1151 1152 { RTLIB::UDIVREM_I8, "__aeabi_uidivmod", CallingConv::ARM_AAPCS }, 1153 { RTLIB::UDIVREM_I16, "__aeabi_uidivmod", CallingConv::ARM_AAPCS }, 1154 { RTLIB::UDIVREM_I32, "__aeabi_uidivmod", CallingConv::ARM_AAPCS }, 1155 { RTLIB::UDIVREM_I64, "__aeabi_uldivmod", CallingConv::ARM_AAPCS }, 1156 }; 1157 1158 for (const auto &LC : LibraryCalls) { 1159 setLibcallName(LC.Op, LC.Name); 1160 setLibcallCallingConv(LC.Op, LC.CC); 1161 } 1162 } 1163 1164 setOperationAction(ISD::SDIVREM, MVT::i32, Custom); 1165 setOperationAction(ISD::UDIVREM, MVT::i32, Custom); 1166 setOperationAction(ISD::SDIVREM, MVT::i64, Custom); 1167 setOperationAction(ISD::UDIVREM, MVT::i64, Custom); 1168 } else { 1169 setOperationAction(ISD::SDIVREM, MVT::i32, Expand); 1170 setOperationAction(ISD::UDIVREM, MVT::i32, Expand); 1171 } 1172 1173 if (Subtarget->isTargetWindows() && Subtarget->getTargetTriple().isOSMSVCRT()) 1174 for (auto &VT : {MVT::f32, MVT::f64}) 1175 setOperationAction(ISD::FPOWI, VT, Custom); 1176 1177 setOperationAction(ISD::GlobalAddress, MVT::i32, Custom); 1178 setOperationAction(ISD::ConstantPool, MVT::i32, Custom); 1179 setOperationAction(ISD::GlobalTLSAddress, MVT::i32, Custom); 1180 setOperationAction(ISD::BlockAddress, MVT::i32, Custom); 1181 1182 setOperationAction(ISD::TRAP, MVT::Other, Legal); 1183 setOperationAction(ISD::DEBUGTRAP, MVT::Other, Legal); 1184 1185 // Use the default implementation. 1186 setOperationAction(ISD::VASTART, MVT::Other, Custom); 1187 setOperationAction(ISD::VAARG, MVT::Other, Expand); 1188 setOperationAction(ISD::VACOPY, MVT::Other, Expand); 1189 setOperationAction(ISD::VAEND, MVT::Other, Expand); 1190 setOperationAction(ISD::STACKSAVE, MVT::Other, Expand); 1191 setOperationAction(ISD::STACKRESTORE, MVT::Other, Expand); 1192 1193 if (Subtarget->isTargetWindows()) 1194 setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i32, Custom); 1195 else 1196 setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i32, Expand); 1197 1198 // ARMv6 Thumb1 (except for CPUs that support dmb / dsb) and earlier use 1199 // the default expansion. 1200 InsertFencesForAtomic = false; 1201 if (Subtarget->hasAnyDataBarrier() && 1202 (!Subtarget->isThumb() || Subtarget->hasV8MBaselineOps())) { 1203 // ATOMIC_FENCE needs custom lowering; the others should have been expanded 1204 // to ldrex/strex loops already. 1205 setOperationAction(ISD::ATOMIC_FENCE, MVT::Other, Custom); 1206 if (!Subtarget->isThumb() || !Subtarget->isMClass()) 1207 setOperationAction(ISD::ATOMIC_CMP_SWAP, MVT::i64, Custom); 1208 1209 // On v8, we have particularly efficient implementations of atomic fences 1210 // if they can be combined with nearby atomic loads and stores. 1211 if (!Subtarget->hasAcquireRelease() || 1212 getTargetMachine().getOptLevel() == 0) { 1213 // Automatically insert fences (dmb ish) around ATOMIC_SWAP etc. 1214 InsertFencesForAtomic = true; 1215 } 1216 } else { 1217 // If there's anything we can use as a barrier, go through custom lowering 1218 // for ATOMIC_FENCE. 1219 // If target has DMB in thumb, Fences can be inserted. 1220 if (Subtarget->hasDataBarrier()) 1221 InsertFencesForAtomic = true; 1222 1223 setOperationAction(ISD::ATOMIC_FENCE, MVT::Other, 1224 Subtarget->hasAnyDataBarrier() ? Custom : Expand); 1225 1226 // Set them all for expansion, which will force libcalls. 1227 setOperationAction(ISD::ATOMIC_CMP_SWAP, MVT::i32, Expand); 1228 setOperationAction(ISD::ATOMIC_SWAP, MVT::i32, Expand); 1229 setOperationAction(ISD::ATOMIC_LOAD_ADD, MVT::i32, Expand); 1230 setOperationAction(ISD::ATOMIC_LOAD_SUB, MVT::i32, Expand); 1231 setOperationAction(ISD::ATOMIC_LOAD_AND, MVT::i32, Expand); 1232 setOperationAction(ISD::ATOMIC_LOAD_OR, MVT::i32, Expand); 1233 setOperationAction(ISD::ATOMIC_LOAD_XOR, MVT::i32, Expand); 1234 setOperationAction(ISD::ATOMIC_LOAD_NAND, MVT::i32, Expand); 1235 setOperationAction(ISD::ATOMIC_LOAD_MIN, MVT::i32, Expand); 1236 setOperationAction(ISD::ATOMIC_LOAD_MAX, MVT::i32, Expand); 1237 setOperationAction(ISD::ATOMIC_LOAD_UMIN, MVT::i32, Expand); 1238 setOperationAction(ISD::ATOMIC_LOAD_UMAX, MVT::i32, Expand); 1239 // Mark ATOMIC_LOAD and ATOMIC_STORE custom so we can handle the 1240 // Unordered/Monotonic case. 1241 if (!InsertFencesForAtomic) { 1242 setOperationAction(ISD::ATOMIC_LOAD, MVT::i32, Custom); 1243 setOperationAction(ISD::ATOMIC_STORE, MVT::i32, Custom); 1244 } 1245 } 1246 1247 setOperationAction(ISD::PREFETCH, MVT::Other, Custom); 1248 1249 // Requires SXTB/SXTH, available on v6 and up in both ARM and Thumb modes. 1250 if (!Subtarget->hasV6Ops()) { 1251 setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::i16, Expand); 1252 setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::i8, Expand); 1253 } 1254 setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::i1, Expand); 1255 1256 if (!Subtarget->useSoftFloat() && Subtarget->hasFPRegs() && 1257 !Subtarget->isThumb1Only()) { 1258 // Turn f64->i64 into VMOVRRD, i64 -> f64 to VMOVDRR 1259 // iff target supports vfp2. 1260 setOperationAction(ISD::BITCAST, MVT::i64, Custom); 1261 setOperationAction(ISD::FLT_ROUNDS_, MVT::i32, Custom); 1262 } 1263 1264 // We want to custom lower some of our intrinsics. 1265 setOperationAction(ISD::INTRINSIC_WO_CHAIN, MVT::Other, Custom); 1266 setOperationAction(ISD::EH_SJLJ_SETJMP, MVT::i32, Custom); 1267 setOperationAction(ISD::EH_SJLJ_LONGJMP, MVT::Other, Custom); 1268 setOperationAction(ISD::EH_SJLJ_SETUP_DISPATCH, MVT::Other, Custom); 1269 if (Subtarget->useSjLjEH()) 1270 setLibcallName(RTLIB::UNWIND_RESUME, "_Unwind_SjLj_Resume"); 1271 1272 setOperationAction(ISD::SETCC, MVT::i32, Expand); 1273 setOperationAction(ISD::SETCC, MVT::f32, Expand); 1274 setOperationAction(ISD::SETCC, MVT::f64, Expand); 1275 setOperationAction(ISD::SELECT, MVT::i32, Custom); 1276 setOperationAction(ISD::SELECT, MVT::f32, Custom); 1277 setOperationAction(ISD::SELECT, MVT::f64, Custom); 1278 setOperationAction(ISD::SELECT_CC, MVT::i32, Custom); 1279 setOperationAction(ISD::SELECT_CC, MVT::f32, Custom); 1280 setOperationAction(ISD::SELECT_CC, MVT::f64, Custom); 1281 if (Subtarget->hasFullFP16()) { 1282 setOperationAction(ISD::SETCC, MVT::f16, Expand); 1283 setOperationAction(ISD::SELECT, MVT::f16, Custom); 1284 setOperationAction(ISD::SELECT_CC, MVT::f16, Custom); 1285 } 1286 1287 setOperationAction(ISD::SETCCCARRY, MVT::i32, Custom); 1288 1289 setOperationAction(ISD::BRCOND, MVT::Other, Custom); 1290 setOperationAction(ISD::BR_CC, MVT::i32, Custom); 1291 if (Subtarget->hasFullFP16()) 1292 setOperationAction(ISD::BR_CC, MVT::f16, Custom); 1293 setOperationAction(ISD::BR_CC, MVT::f32, Custom); 1294 setOperationAction(ISD::BR_CC, MVT::f64, Custom); 1295 setOperationAction(ISD::BR_JT, MVT::Other, Custom); 1296 1297 // We don't support sin/cos/fmod/copysign/pow 1298 setOperationAction(ISD::FSIN, MVT::f64, Expand); 1299 setOperationAction(ISD::FSIN, MVT::f32, Expand); 1300 setOperationAction(ISD::FCOS, MVT::f32, Expand); 1301 setOperationAction(ISD::FCOS, MVT::f64, Expand); 1302 setOperationAction(ISD::FSINCOS, MVT::f64, Expand); 1303 setOperationAction(ISD::FSINCOS, MVT::f32, Expand); 1304 setOperationAction(ISD::FREM, MVT::f64, Expand); 1305 setOperationAction(ISD::FREM, MVT::f32, Expand); 1306 if (!Subtarget->useSoftFloat() && Subtarget->hasVFP2Base() && 1307 !Subtarget->isThumb1Only()) { 1308 setOperationAction(ISD::FCOPYSIGN, MVT::f64, Custom); 1309 setOperationAction(ISD::FCOPYSIGN, MVT::f32, Custom); 1310 } 1311 setOperationAction(ISD::FPOW, MVT::f64, Expand); 1312 setOperationAction(ISD::FPOW, MVT::f32, Expand); 1313 1314 if (!Subtarget->hasVFP4Base()) { 1315 setOperationAction(ISD::FMA, MVT::f64, Expand); 1316 setOperationAction(ISD::FMA, MVT::f32, Expand); 1317 } 1318 1319 // Various VFP goodness 1320 if (!Subtarget->useSoftFloat() && !Subtarget->isThumb1Only()) { 1321 // FP-ARMv8 adds f64 <-> f16 conversion. Before that it should be expanded. 1322 if (!Subtarget->hasFPARMv8Base() || !Subtarget->hasFP64()) { 1323 setOperationAction(ISD::FP16_TO_FP, MVT::f64, Expand); 1324 setOperationAction(ISD::FP_TO_FP16, MVT::f64, Expand); 1325 } 1326 1327 // fp16 is a special v7 extension that adds f16 <-> f32 conversions. 1328 if (!Subtarget->hasFP16()) { 1329 setOperationAction(ISD::FP16_TO_FP, MVT::f32, Expand); 1330 setOperationAction(ISD::FP_TO_FP16, MVT::f32, Expand); 1331 } 1332 } 1333 1334 // Use __sincos_stret if available. 1335 if (getLibcallName(RTLIB::SINCOS_STRET_F32) != nullptr && 1336 getLibcallName(RTLIB::SINCOS_STRET_F64) != nullptr) { 1337 setOperationAction(ISD::FSINCOS, MVT::f64, Custom); 1338 setOperationAction(ISD::FSINCOS, MVT::f32, Custom); 1339 } 1340 1341 // FP-ARMv8 implements a lot of rounding-like FP operations. 1342 if (Subtarget->hasFPARMv8Base()) { 1343 setOperationAction(ISD::FFLOOR, MVT::f32, Legal); 1344 setOperationAction(ISD::FCEIL, MVT::f32, Legal); 1345 setOperationAction(ISD::FROUND, MVT::f32, Legal); 1346 setOperationAction(ISD::FTRUNC, MVT::f32, Legal); 1347 setOperationAction(ISD::FNEARBYINT, MVT::f32, Legal); 1348 setOperationAction(ISD::FRINT, MVT::f32, Legal); 1349 setOperationAction(ISD::FMINNUM, MVT::f32, Legal); 1350 setOperationAction(ISD::FMAXNUM, MVT::f32, Legal); 1351 if (Subtarget->hasNEON()) { 1352 setOperationAction(ISD::FMINNUM, MVT::v2f32, Legal); 1353 setOperationAction(ISD::FMAXNUM, MVT::v2f32, Legal); 1354 setOperationAction(ISD::FMINNUM, MVT::v4f32, Legal); 1355 setOperationAction(ISD::FMAXNUM, MVT::v4f32, Legal); 1356 } 1357 1358 if (Subtarget->hasFP64()) { 1359 setOperationAction(ISD::FFLOOR, MVT::f64, Legal); 1360 setOperationAction(ISD::FCEIL, MVT::f64, Legal); 1361 setOperationAction(ISD::FROUND, MVT::f64, Legal); 1362 setOperationAction(ISD::FTRUNC, MVT::f64, Legal); 1363 setOperationAction(ISD::FNEARBYINT, MVT::f64, Legal); 1364 setOperationAction(ISD::FRINT, MVT::f64, Legal); 1365 setOperationAction(ISD::FMINNUM, MVT::f64, Legal); 1366 setOperationAction(ISD::FMAXNUM, MVT::f64, Legal); 1367 } 1368 } 1369 1370 // FP16 often need to be promoted to call lib functions 1371 if (Subtarget->hasFullFP16()) { 1372 setOperationAction(ISD::FREM, MVT::f16, Promote); 1373 setOperationAction(ISD::FCOPYSIGN, MVT::f16, Expand); 1374 setOperationAction(ISD::FSIN, MVT::f16, Promote); 1375 setOperationAction(ISD::FCOS, MVT::f16, Promote); 1376 setOperationAction(ISD::FSINCOS, MVT::f16, Promote); 1377 setOperationAction(ISD::FPOWI, MVT::f16, Promote); 1378 setOperationAction(ISD::FPOW, MVT::f16, Promote); 1379 setOperationAction(ISD::FEXP, MVT::f16, Promote); 1380 setOperationAction(ISD::FEXP2, MVT::f16, Promote); 1381 setOperationAction(ISD::FLOG, MVT::f16, Promote); 1382 setOperationAction(ISD::FLOG10, MVT::f16, Promote); 1383 setOperationAction(ISD::FLOG2, MVT::f16, Promote); 1384 1385 setOperationAction(ISD::FROUND, MVT::f16, Legal); 1386 } 1387 1388 if (Subtarget->hasNEON()) { 1389 // vmin and vmax aren't available in a scalar form, so we use 1390 // a NEON instruction with an undef lane instead. 1391 setOperationAction(ISD::FMINIMUM, MVT::f16, Legal); 1392 setOperationAction(ISD::FMAXIMUM, MVT::f16, Legal); 1393 setOperationAction(ISD::FMINIMUM, MVT::f32, Legal); 1394 setOperationAction(ISD::FMAXIMUM, MVT::f32, Legal); 1395 setOperationAction(ISD::FMINIMUM, MVT::v2f32, Legal); 1396 setOperationAction(ISD::FMAXIMUM, MVT::v2f32, Legal); 1397 setOperationAction(ISD::FMINIMUM, MVT::v4f32, Legal); 1398 setOperationAction(ISD::FMAXIMUM, MVT::v4f32, Legal); 1399 1400 if (Subtarget->hasFullFP16()) { 1401 setOperationAction(ISD::FMINNUM, MVT::v4f16, Legal); 1402 setOperationAction(ISD::FMAXNUM, MVT::v4f16, Legal); 1403 setOperationAction(ISD::FMINNUM, MVT::v8f16, Legal); 1404 setOperationAction(ISD::FMAXNUM, MVT::v8f16, Legal); 1405 1406 setOperationAction(ISD::FMINIMUM, MVT::v4f16, Legal); 1407 setOperationAction(ISD::FMAXIMUM, MVT::v4f16, Legal); 1408 setOperationAction(ISD::FMINIMUM, MVT::v8f16, Legal); 1409 setOperationAction(ISD::FMAXIMUM, MVT::v8f16, Legal); 1410 } 1411 } 1412 1413 // We have target-specific dag combine patterns for the following nodes: 1414 // ARMISD::VMOVRRD - No need to call setTargetDAGCombine 1415 setTargetDAGCombine(ISD::ADD); 1416 setTargetDAGCombine(ISD::SUB); 1417 setTargetDAGCombine(ISD::MUL); 1418 setTargetDAGCombine(ISD::AND); 1419 setTargetDAGCombine(ISD::OR); 1420 setTargetDAGCombine(ISD::XOR); 1421 1422 if (Subtarget->hasV6Ops()) 1423 setTargetDAGCombine(ISD::SRL); 1424 if (Subtarget->isThumb1Only()) 1425 setTargetDAGCombine(ISD::SHL); 1426 1427 setStackPointerRegisterToSaveRestore(ARM::SP); 1428 1429 if (Subtarget->useSoftFloat() || Subtarget->isThumb1Only() || 1430 !Subtarget->hasVFP2Base() || Subtarget->hasMinSize()) 1431 setSchedulingPreference(Sched::RegPressure); 1432 else 1433 setSchedulingPreference(Sched::Hybrid); 1434 1435 //// temporary - rewrite interface to use type 1436 MaxStoresPerMemset = 8; 1437 MaxStoresPerMemsetOptSize = 4; 1438 MaxStoresPerMemcpy = 4; // For @llvm.memcpy -> sequence of stores 1439 MaxStoresPerMemcpyOptSize = 2; 1440 MaxStoresPerMemmove = 4; // For @llvm.memmove -> sequence of stores 1441 MaxStoresPerMemmoveOptSize = 2; 1442 1443 // On ARM arguments smaller than 4 bytes are extended, so all arguments 1444 // are at least 4 bytes aligned. 1445 setMinStackArgumentAlignment(Align(4)); 1446 1447 // Prefer likely predicted branches to selects on out-of-order cores. 1448 PredictableSelectIsExpensive = Subtarget->getSchedModel().isOutOfOrder(); 1449 1450 setPrefLoopAlignment(Align(1ULL << Subtarget->getPrefLoopLogAlignment())); 1451 1452 setMinFunctionAlignment(Subtarget->isThumb() ? Align(2) : Align(4)); 1453 1454 if (Subtarget->isThumb() || Subtarget->isThumb2()) 1455 setTargetDAGCombine(ISD::ABS); 1456 } 1457 1458 bool ARMTargetLowering::useSoftFloat() const { 1459 return Subtarget->useSoftFloat(); 1460 } 1461 1462 // FIXME: It might make sense to define the representative register class as the 1463 // nearest super-register that has a non-null superset. For example, DPR_VFP2 is 1464 // a super-register of SPR, and DPR is a superset if DPR_VFP2. Consequently, 1465 // SPR's representative would be DPR_VFP2. This should work well if register 1466 // pressure tracking were modified such that a register use would increment the 1467 // pressure of the register class's representative and all of it's super 1468 // classes' representatives transitively. We have not implemented this because 1469 // of the difficulty prior to coalescing of modeling operand register classes 1470 // due to the common occurrence of cross class copies and subregister insertions 1471 // and extractions. 1472 std::pair<const TargetRegisterClass *, uint8_t> 1473 ARMTargetLowering::findRepresentativeClass(const TargetRegisterInfo *TRI, 1474 MVT VT) const { 1475 const TargetRegisterClass *RRC = nullptr; 1476 uint8_t Cost = 1; 1477 switch (VT.SimpleTy) { 1478 default: 1479 return TargetLowering::findRepresentativeClass(TRI, VT); 1480 // Use DPR as representative register class for all floating point 1481 // and vector types. Since there are 32 SPR registers and 32 DPR registers so 1482 // the cost is 1 for both f32 and f64. 1483 case MVT::f32: case MVT::f64: case MVT::v8i8: case MVT::v4i16: 1484 case MVT::v2i32: case MVT::v1i64: case MVT::v2f32: 1485 RRC = &ARM::DPRRegClass; 1486 // When NEON is used for SP, only half of the register file is available 1487 // because operations that define both SP and DP results will be constrained 1488 // to the VFP2 class (D0-D15). We currently model this constraint prior to 1489 // coalescing by double-counting the SP regs. See the FIXME above. 1490 if (Subtarget->useNEONForSinglePrecisionFP()) 1491 Cost = 2; 1492 break; 1493 case MVT::v16i8: case MVT::v8i16: case MVT::v4i32: case MVT::v2i64: 1494 case MVT::v4f32: case MVT::v2f64: 1495 RRC = &ARM::DPRRegClass; 1496 Cost = 2; 1497 break; 1498 case MVT::v4i64: 1499 RRC = &ARM::DPRRegClass; 1500 Cost = 4; 1501 break; 1502 case MVT::v8i64: 1503 RRC = &ARM::DPRRegClass; 1504 Cost = 8; 1505 break; 1506 } 1507 return std::make_pair(RRC, Cost); 1508 } 1509 1510 const char *ARMTargetLowering::getTargetNodeName(unsigned Opcode) const { 1511 switch ((ARMISD::NodeType)Opcode) { 1512 case ARMISD::FIRST_NUMBER: break; 1513 case ARMISD::Wrapper: return "ARMISD::Wrapper"; 1514 case ARMISD::WrapperPIC: return "ARMISD::WrapperPIC"; 1515 case ARMISD::WrapperJT: return "ARMISD::WrapperJT"; 1516 case ARMISD::COPY_STRUCT_BYVAL: return "ARMISD::COPY_STRUCT_BYVAL"; 1517 case ARMISD::CALL: return "ARMISD::CALL"; 1518 case ARMISD::CALL_PRED: return "ARMISD::CALL_PRED"; 1519 case ARMISD::CALL_NOLINK: return "ARMISD::CALL_NOLINK"; 1520 case ARMISD::BRCOND: return "ARMISD::BRCOND"; 1521 case ARMISD::BR_JT: return "ARMISD::BR_JT"; 1522 case ARMISD::BR2_JT: return "ARMISD::BR2_JT"; 1523 case ARMISD::RET_FLAG: return "ARMISD::RET_FLAG"; 1524 case ARMISD::INTRET_FLAG: return "ARMISD::INTRET_FLAG"; 1525 case ARMISD::PIC_ADD: return "ARMISD::PIC_ADD"; 1526 case ARMISD::CMP: return "ARMISD::CMP"; 1527 case ARMISD::CMN: return "ARMISD::CMN"; 1528 case ARMISD::CMPZ: return "ARMISD::CMPZ"; 1529 case ARMISD::CMPFP: return "ARMISD::CMPFP"; 1530 case ARMISD::CMPFPw0: return "ARMISD::CMPFPw0"; 1531 case ARMISD::BCC_i64: return "ARMISD::BCC_i64"; 1532 case ARMISD::FMSTAT: return "ARMISD::FMSTAT"; 1533 1534 case ARMISD::CMOV: return "ARMISD::CMOV"; 1535 case ARMISD::SUBS: return "ARMISD::SUBS"; 1536 1537 case ARMISD::SSAT: return "ARMISD::SSAT"; 1538 case ARMISD::USAT: return "ARMISD::USAT"; 1539 1540 case ARMISD::ASRL: return "ARMISD::ASRL"; 1541 case ARMISD::LSRL: return "ARMISD::LSRL"; 1542 case ARMISD::LSLL: return "ARMISD::LSLL"; 1543 1544 case ARMISD::SRL_FLAG: return "ARMISD::SRL_FLAG"; 1545 case ARMISD::SRA_FLAG: return "ARMISD::SRA_FLAG"; 1546 case ARMISD::RRX: return "ARMISD::RRX"; 1547 1548 case ARMISD::ADDC: return "ARMISD::ADDC"; 1549 case ARMISD::ADDE: return "ARMISD::ADDE"; 1550 case ARMISD::SUBC: return "ARMISD::SUBC"; 1551 case ARMISD::SUBE: return "ARMISD::SUBE"; 1552 case ARMISD::LSLS: return "ARMISD::LSLS"; 1553 1554 case ARMISD::VMOVRRD: return "ARMISD::VMOVRRD"; 1555 case ARMISD::VMOVDRR: return "ARMISD::VMOVDRR"; 1556 case ARMISD::VMOVhr: return "ARMISD::VMOVhr"; 1557 case ARMISD::VMOVrh: return "ARMISD::VMOVrh"; 1558 case ARMISD::VMOVSR: return "ARMISD::VMOVSR"; 1559 1560 case ARMISD::EH_SJLJ_SETJMP: return "ARMISD::EH_SJLJ_SETJMP"; 1561 case ARMISD::EH_SJLJ_LONGJMP: return "ARMISD::EH_SJLJ_LONGJMP"; 1562 case ARMISD::EH_SJLJ_SETUP_DISPATCH: return "ARMISD::EH_SJLJ_SETUP_DISPATCH"; 1563 1564 case ARMISD::TC_RETURN: return "ARMISD::TC_RETURN"; 1565 1566 case ARMISD::THREAD_POINTER:return "ARMISD::THREAD_POINTER"; 1567 1568 case ARMISD::DYN_ALLOC: return "ARMISD::DYN_ALLOC"; 1569 1570 case ARMISD::MEMBARRIER_MCR: return "ARMISD::MEMBARRIER_MCR"; 1571 1572 case ARMISD::PRELOAD: return "ARMISD::PRELOAD"; 1573 1574 case ARMISD::WIN__CHKSTK: return "ARMISD::WIN__CHKSTK"; 1575 case ARMISD::WIN__DBZCHK: return "ARMISD::WIN__DBZCHK"; 1576 1577 case ARMISD::PREDICATE_CAST: return "ARMISD::PREDICATE_CAST"; 1578 case ARMISD::VCMP: return "ARMISD::VCMP"; 1579 case ARMISD::VCMPZ: return "ARMISD::VCMPZ"; 1580 case ARMISD::VTST: return "ARMISD::VTST"; 1581 1582 case ARMISD::VSHLs: return "ARMISD::VSHLs"; 1583 case ARMISD::VSHLu: return "ARMISD::VSHLu"; 1584 case ARMISD::VSHLIMM: return "ARMISD::VSHLIMM"; 1585 case ARMISD::VSHRsIMM: return "ARMISD::VSHRsIMM"; 1586 case ARMISD::VSHRuIMM: return "ARMISD::VSHRuIMM"; 1587 case ARMISD::VRSHRsIMM: return "ARMISD::VRSHRsIMM"; 1588 case ARMISD::VRSHRuIMM: return "ARMISD::VRSHRuIMM"; 1589 case ARMISD::VRSHRNIMM: return "ARMISD::VRSHRNIMM"; 1590 case ARMISD::VQSHLsIMM: return "ARMISD::VQSHLsIMM"; 1591 case ARMISD::VQSHLuIMM: return "ARMISD::VQSHLuIMM"; 1592 case ARMISD::VQSHLsuIMM: return "ARMISD::VQSHLsuIMM"; 1593 case ARMISD::VQSHRNsIMM: return "ARMISD::VQSHRNsIMM"; 1594 case ARMISD::VQSHRNuIMM: return "ARMISD::VQSHRNuIMM"; 1595 case ARMISD::VQSHRNsuIMM: return "ARMISD::VQSHRNsuIMM"; 1596 case ARMISD::VQRSHRNsIMM: return "ARMISD::VQRSHRNsIMM"; 1597 case ARMISD::VQRSHRNuIMM: return "ARMISD::VQRSHRNuIMM"; 1598 case ARMISD::VQRSHRNsuIMM: return "ARMISD::VQRSHRNsuIMM"; 1599 case ARMISD::VSLIIMM: return "ARMISD::VSLIIMM"; 1600 case ARMISD::VSRIIMM: return "ARMISD::VSRIIMM"; 1601 case ARMISD::VGETLANEu: return "ARMISD::VGETLANEu"; 1602 case ARMISD::VGETLANEs: return "ARMISD::VGETLANEs"; 1603 case ARMISD::VMOVIMM: return "ARMISD::VMOVIMM"; 1604 case ARMISD::VMVNIMM: return "ARMISD::VMVNIMM"; 1605 case ARMISD::VMOVFPIMM: return "ARMISD::VMOVFPIMM"; 1606 case ARMISD::VDUP: return "ARMISD::VDUP"; 1607 case ARMISD::VDUPLANE: return "ARMISD::VDUPLANE"; 1608 case ARMISD::VEXT: return "ARMISD::VEXT"; 1609 case ARMISD::VREV64: return "ARMISD::VREV64"; 1610 case ARMISD::VREV32: return "ARMISD::VREV32"; 1611 case ARMISD::VREV16: return "ARMISD::VREV16"; 1612 case ARMISD::VZIP: return "ARMISD::VZIP"; 1613 case ARMISD::VUZP: return "ARMISD::VUZP"; 1614 case ARMISD::VTRN: return "ARMISD::VTRN"; 1615 case ARMISD::VTBL1: return "ARMISD::VTBL1"; 1616 case ARMISD::VTBL2: return "ARMISD::VTBL2"; 1617 case ARMISD::VMOVN: return "ARMISD::VMOVN"; 1618 case ARMISD::VMULLs: return "ARMISD::VMULLs"; 1619 case ARMISD::VMULLu: return "ARMISD::VMULLu"; 1620 case ARMISD::UMAAL: return "ARMISD::UMAAL"; 1621 case ARMISD::UMLAL: return "ARMISD::UMLAL"; 1622 case ARMISD::SMLAL: return "ARMISD::SMLAL"; 1623 case ARMISD::SMLALBB: return "ARMISD::SMLALBB"; 1624 case ARMISD::SMLALBT: return "ARMISD::SMLALBT"; 1625 case ARMISD::SMLALTB: return "ARMISD::SMLALTB"; 1626 case ARMISD::SMLALTT: return "ARMISD::SMLALTT"; 1627 case ARMISD::SMULWB: return "ARMISD::SMULWB"; 1628 case ARMISD::SMULWT: return "ARMISD::SMULWT"; 1629 case ARMISD::SMLALD: return "ARMISD::SMLALD"; 1630 case ARMISD::SMLALDX: return "ARMISD::SMLALDX"; 1631 case ARMISD::SMLSLD: return "ARMISD::SMLSLD"; 1632 case ARMISD::SMLSLDX: return "ARMISD::SMLSLDX"; 1633 case ARMISD::SMMLAR: return "ARMISD::SMMLAR"; 1634 case ARMISD::SMMLSR: return "ARMISD::SMMLSR"; 1635 case ARMISD::QADD16b: return "ARMISD::QADD16b"; 1636 case ARMISD::QSUB16b: return "ARMISD::QSUB16b"; 1637 case ARMISD::QADD8b: return "ARMISD::QADD8b"; 1638 case ARMISD::QSUB8b: return "ARMISD::QSUB8b"; 1639 case ARMISD::BUILD_VECTOR: return "ARMISD::BUILD_VECTOR"; 1640 case ARMISD::BFI: return "ARMISD::BFI"; 1641 case ARMISD::VORRIMM: return "ARMISD::VORRIMM"; 1642 case ARMISD::VBICIMM: return "ARMISD::VBICIMM"; 1643 case ARMISD::VBSL: return "ARMISD::VBSL"; 1644 case ARMISD::MEMCPY: return "ARMISD::MEMCPY"; 1645 case ARMISD::VLD1DUP: return "ARMISD::VLD1DUP"; 1646 case ARMISD::VLD2DUP: return "ARMISD::VLD2DUP"; 1647 case ARMISD::VLD3DUP: return "ARMISD::VLD3DUP"; 1648 case ARMISD::VLD4DUP: return "ARMISD::VLD4DUP"; 1649 case ARMISD::VLD1_UPD: return "ARMISD::VLD1_UPD"; 1650 case ARMISD::VLD2_UPD: return "ARMISD::VLD2_UPD"; 1651 case ARMISD::VLD3_UPD: return "ARMISD::VLD3_UPD"; 1652 case ARMISD::VLD4_UPD: return "ARMISD::VLD4_UPD"; 1653 case ARMISD::VLD2LN_UPD: return "ARMISD::VLD2LN_UPD"; 1654 case ARMISD::VLD3LN_UPD: return "ARMISD::VLD3LN_UPD"; 1655 case ARMISD::VLD4LN_UPD: return "ARMISD::VLD4LN_UPD"; 1656 case ARMISD::VLD1DUP_UPD: return "ARMISD::VLD1DUP_UPD"; 1657 case ARMISD::VLD2DUP_UPD: return "ARMISD::VLD2DUP_UPD"; 1658 case ARMISD::VLD3DUP_UPD: return "ARMISD::VLD3DUP_UPD"; 1659 case ARMISD::VLD4DUP_UPD: return "ARMISD::VLD4DUP_UPD"; 1660 case ARMISD::VST1_UPD: return "ARMISD::VST1_UPD"; 1661 case ARMISD::VST2_UPD: return "ARMISD::VST2_UPD"; 1662 case ARMISD::VST3_UPD: return "ARMISD::VST3_UPD"; 1663 case ARMISD::VST4_UPD: return "ARMISD::VST4_UPD"; 1664 case ARMISD::VST2LN_UPD: return "ARMISD::VST2LN_UPD"; 1665 case ARMISD::VST3LN_UPD: return "ARMISD::VST3LN_UPD"; 1666 case ARMISD::VST4LN_UPD: return "ARMISD::VST4LN_UPD"; 1667 case ARMISD::WLS: return "ARMISD::WLS"; 1668 case ARMISD::LE: return "ARMISD::LE"; 1669 case ARMISD::LOOP_DEC: return "ARMISD::LOOP_DEC"; 1670 case ARMISD::CSINV: return "ARMISD::CSINV"; 1671 case ARMISD::CSNEG: return "ARMISD::CSNEG"; 1672 case ARMISD::CSINC: return "ARMISD::CSINC"; 1673 } 1674 return nullptr; 1675 } 1676 1677 EVT ARMTargetLowering::getSetCCResultType(const DataLayout &DL, LLVMContext &, 1678 EVT VT) const { 1679 if (!VT.isVector()) 1680 return getPointerTy(DL); 1681 1682 // MVE has a predicate register. 1683 if (Subtarget->hasMVEIntegerOps() && 1684 (VT == MVT::v4i32 || VT == MVT::v8i16 || VT == MVT::v16i8)) 1685 return MVT::getVectorVT(MVT::i1, VT.getVectorElementCount()); 1686 return VT.changeVectorElementTypeToInteger(); 1687 } 1688 1689 /// getRegClassFor - Return the register class that should be used for the 1690 /// specified value type. 1691 const TargetRegisterClass * 1692 ARMTargetLowering::getRegClassFor(MVT VT, bool isDivergent) const { 1693 (void)isDivergent; 1694 // Map v4i64 to QQ registers but do not make the type legal. Similarly map 1695 // v8i64 to QQQQ registers. v4i64 and v8i64 are only used for REG_SEQUENCE to 1696 // load / store 4 to 8 consecutive NEON D registers, or 2 to 4 consecutive 1697 // MVE Q registers. 1698 if (Subtarget->hasNEON() || Subtarget->hasMVEIntegerOps()) { 1699 if (VT == MVT::v4i64) 1700 return &ARM::QQPRRegClass; 1701 if (VT == MVT::v8i64) 1702 return &ARM::QQQQPRRegClass; 1703 } 1704 return TargetLowering::getRegClassFor(VT); 1705 } 1706 1707 // memcpy, and other memory intrinsics, typically tries to use LDM/STM if the 1708 // source/dest is aligned and the copy size is large enough. We therefore want 1709 // to align such objects passed to memory intrinsics. 1710 bool ARMTargetLowering::shouldAlignPointerArgs(CallInst *CI, unsigned &MinSize, 1711 unsigned &PrefAlign) const { 1712 if (!isa<MemIntrinsic>(CI)) 1713 return false; 1714 MinSize = 8; 1715 // On ARM11 onwards (excluding M class) 8-byte aligned LDM is typically 1 1716 // cycle faster than 4-byte aligned LDM. 1717 PrefAlign = (Subtarget->hasV6Ops() && !Subtarget->isMClass() ? 8 : 4); 1718 return true; 1719 } 1720 1721 // Create a fast isel object. 1722 FastISel * 1723 ARMTargetLowering::createFastISel(FunctionLoweringInfo &funcInfo, 1724 const TargetLibraryInfo *libInfo) const { 1725 return ARM::createFastISel(funcInfo, libInfo); 1726 } 1727 1728 Sched::Preference ARMTargetLowering::getSchedulingPreference(SDNode *N) const { 1729 unsigned NumVals = N->getNumValues(); 1730 if (!NumVals) 1731 return Sched::RegPressure; 1732 1733 for (unsigned i = 0; i != NumVals; ++i) { 1734 EVT VT = N->getValueType(i); 1735 if (VT == MVT::Glue || VT == MVT::Other) 1736 continue; 1737 if (VT.isFloatingPoint() || VT.isVector()) 1738 return Sched::ILP; 1739 } 1740 1741 if (!N->isMachineOpcode()) 1742 return Sched::RegPressure; 1743 1744 // Load are scheduled for latency even if there instruction itinerary 1745 // is not available. 1746 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 1747 const MCInstrDesc &MCID = TII->get(N->getMachineOpcode()); 1748 1749 if (MCID.getNumDefs() == 0) 1750 return Sched::RegPressure; 1751 if (!Itins->isEmpty() && 1752 Itins->getOperandCycle(MCID.getSchedClass(), 0) > 2) 1753 return Sched::ILP; 1754 1755 return Sched::RegPressure; 1756 } 1757 1758 //===----------------------------------------------------------------------===// 1759 // Lowering Code 1760 //===----------------------------------------------------------------------===// 1761 1762 static bool isSRL16(const SDValue &Op) { 1763 if (Op.getOpcode() != ISD::SRL) 1764 return false; 1765 if (auto Const = dyn_cast<ConstantSDNode>(Op.getOperand(1))) 1766 return Const->getZExtValue() == 16; 1767 return false; 1768 } 1769 1770 static bool isSRA16(const SDValue &Op) { 1771 if (Op.getOpcode() != ISD::SRA) 1772 return false; 1773 if (auto Const = dyn_cast<ConstantSDNode>(Op.getOperand(1))) 1774 return Const->getZExtValue() == 16; 1775 return false; 1776 } 1777 1778 static bool isSHL16(const SDValue &Op) { 1779 if (Op.getOpcode() != ISD::SHL) 1780 return false; 1781 if (auto Const = dyn_cast<ConstantSDNode>(Op.getOperand(1))) 1782 return Const->getZExtValue() == 16; 1783 return false; 1784 } 1785 1786 // Check for a signed 16-bit value. We special case SRA because it makes it 1787 // more simple when also looking for SRAs that aren't sign extending a 1788 // smaller value. Without the check, we'd need to take extra care with 1789 // checking order for some operations. 1790 static bool isS16(const SDValue &Op, SelectionDAG &DAG) { 1791 if (isSRA16(Op)) 1792 return isSHL16(Op.getOperand(0)); 1793 return DAG.ComputeNumSignBits(Op) == 17; 1794 } 1795 1796 /// IntCCToARMCC - Convert a DAG integer condition code to an ARM CC 1797 static ARMCC::CondCodes IntCCToARMCC(ISD::CondCode CC) { 1798 switch (CC) { 1799 default: llvm_unreachable("Unknown condition code!"); 1800 case ISD::SETNE: return ARMCC::NE; 1801 case ISD::SETEQ: return ARMCC::EQ; 1802 case ISD::SETGT: return ARMCC::GT; 1803 case ISD::SETGE: return ARMCC::GE; 1804 case ISD::SETLT: return ARMCC::LT; 1805 case ISD::SETLE: return ARMCC::LE; 1806 case ISD::SETUGT: return ARMCC::HI; 1807 case ISD::SETUGE: return ARMCC::HS; 1808 case ISD::SETULT: return ARMCC::LO; 1809 case ISD::SETULE: return ARMCC::LS; 1810 } 1811 } 1812 1813 /// FPCCToARMCC - Convert a DAG fp condition code to an ARM CC. 1814 static void FPCCToARMCC(ISD::CondCode CC, ARMCC::CondCodes &CondCode, 1815 ARMCC::CondCodes &CondCode2) { 1816 CondCode2 = ARMCC::AL; 1817 switch (CC) { 1818 default: llvm_unreachable("Unknown FP condition!"); 1819 case ISD::SETEQ: 1820 case ISD::SETOEQ: CondCode = ARMCC::EQ; break; 1821 case ISD::SETGT: 1822 case ISD::SETOGT: CondCode = ARMCC::GT; break; 1823 case ISD::SETGE: 1824 case ISD::SETOGE: CondCode = ARMCC::GE; break; 1825 case ISD::SETOLT: CondCode = ARMCC::MI; break; 1826 case ISD::SETOLE: CondCode = ARMCC::LS; break; 1827 case ISD::SETONE: CondCode = ARMCC::MI; CondCode2 = ARMCC::GT; break; 1828 case ISD::SETO: CondCode = ARMCC::VC; break; 1829 case ISD::SETUO: CondCode = ARMCC::VS; break; 1830 case ISD::SETUEQ: CondCode = ARMCC::EQ; CondCode2 = ARMCC::VS; break; 1831 case ISD::SETUGT: CondCode = ARMCC::HI; break; 1832 case ISD::SETUGE: CondCode = ARMCC::PL; break; 1833 case ISD::SETLT: 1834 case ISD::SETULT: CondCode = ARMCC::LT; break; 1835 case ISD::SETLE: 1836 case ISD::SETULE: CondCode = ARMCC::LE; break; 1837 case ISD::SETNE: 1838 case ISD::SETUNE: CondCode = ARMCC::NE; break; 1839 } 1840 } 1841 1842 //===----------------------------------------------------------------------===// 1843 // Calling Convention Implementation 1844 //===----------------------------------------------------------------------===// 1845 1846 /// getEffectiveCallingConv - Get the effective calling convention, taking into 1847 /// account presence of floating point hardware and calling convention 1848 /// limitations, such as support for variadic functions. 1849 CallingConv::ID 1850 ARMTargetLowering::getEffectiveCallingConv(CallingConv::ID CC, 1851 bool isVarArg) const { 1852 switch (CC) { 1853 default: 1854 report_fatal_error("Unsupported calling convention"); 1855 case CallingConv::ARM_AAPCS: 1856 case CallingConv::ARM_APCS: 1857 case CallingConv::GHC: 1858 return CC; 1859 case CallingConv::PreserveMost: 1860 return CallingConv::PreserveMost; 1861 case CallingConv::ARM_AAPCS_VFP: 1862 case CallingConv::Swift: 1863 return isVarArg ? CallingConv::ARM_AAPCS : CallingConv::ARM_AAPCS_VFP; 1864 case CallingConv::C: 1865 if (!Subtarget->isAAPCS_ABI()) 1866 return CallingConv::ARM_APCS; 1867 else if (Subtarget->hasVFP2Base() && !Subtarget->isThumb1Only() && 1868 getTargetMachine().Options.FloatABIType == FloatABI::Hard && 1869 !isVarArg) 1870 return CallingConv::ARM_AAPCS_VFP; 1871 else 1872 return CallingConv::ARM_AAPCS; 1873 case CallingConv::Fast: 1874 case CallingConv::CXX_FAST_TLS: 1875 if (!Subtarget->isAAPCS_ABI()) { 1876 if (Subtarget->hasVFP2Base() && !Subtarget->isThumb1Only() && !isVarArg) 1877 return CallingConv::Fast; 1878 return CallingConv::ARM_APCS; 1879 } else if (Subtarget->hasVFP2Base() && 1880 !Subtarget->isThumb1Only() && !isVarArg) 1881 return CallingConv::ARM_AAPCS_VFP; 1882 else 1883 return CallingConv::ARM_AAPCS; 1884 } 1885 } 1886 1887 CCAssignFn *ARMTargetLowering::CCAssignFnForCall(CallingConv::ID CC, 1888 bool isVarArg) const { 1889 return CCAssignFnForNode(CC, false, isVarArg); 1890 } 1891 1892 CCAssignFn *ARMTargetLowering::CCAssignFnForReturn(CallingConv::ID CC, 1893 bool isVarArg) const { 1894 return CCAssignFnForNode(CC, true, isVarArg); 1895 } 1896 1897 /// CCAssignFnForNode - Selects the correct CCAssignFn for the given 1898 /// CallingConvention. 1899 CCAssignFn *ARMTargetLowering::CCAssignFnForNode(CallingConv::ID CC, 1900 bool Return, 1901 bool isVarArg) const { 1902 switch (getEffectiveCallingConv(CC, isVarArg)) { 1903 default: 1904 report_fatal_error("Unsupported calling convention"); 1905 case CallingConv::ARM_APCS: 1906 return (Return ? RetCC_ARM_APCS : CC_ARM_APCS); 1907 case CallingConv::ARM_AAPCS: 1908 return (Return ? RetCC_ARM_AAPCS : CC_ARM_AAPCS); 1909 case CallingConv::ARM_AAPCS_VFP: 1910 return (Return ? RetCC_ARM_AAPCS_VFP : CC_ARM_AAPCS_VFP); 1911 case CallingConv::Fast: 1912 return (Return ? RetFastCC_ARM_APCS : FastCC_ARM_APCS); 1913 case CallingConv::GHC: 1914 return (Return ? RetCC_ARM_APCS : CC_ARM_APCS_GHC); 1915 case CallingConv::PreserveMost: 1916 return (Return ? RetCC_ARM_AAPCS : CC_ARM_AAPCS); 1917 } 1918 } 1919 1920 /// LowerCallResult - Lower the result values of a call into the 1921 /// appropriate copies out of appropriate physical registers. 1922 SDValue ARMTargetLowering::LowerCallResult( 1923 SDValue Chain, SDValue InFlag, CallingConv::ID CallConv, bool isVarArg, 1924 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &dl, 1925 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals, bool isThisReturn, 1926 SDValue ThisVal) const { 1927 // Assign locations to each value returned by this call. 1928 SmallVector<CCValAssign, 16> RVLocs; 1929 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), RVLocs, 1930 *DAG.getContext()); 1931 CCInfo.AnalyzeCallResult(Ins, CCAssignFnForReturn(CallConv, isVarArg)); 1932 1933 // Copy all of the result registers out of their specified physreg. 1934 for (unsigned i = 0; i != RVLocs.size(); ++i) { 1935 CCValAssign VA = RVLocs[i]; 1936 1937 // Pass 'this' value directly from the argument to return value, to avoid 1938 // reg unit interference 1939 if (i == 0 && isThisReturn) { 1940 assert(!VA.needsCustom() && VA.getLocVT() == MVT::i32 && 1941 "unexpected return calling convention register assignment"); 1942 InVals.push_back(ThisVal); 1943 continue; 1944 } 1945 1946 SDValue Val; 1947 if (VA.needsCustom()) { 1948 // Handle f64 or half of a v2f64. 1949 SDValue Lo = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32, 1950 InFlag); 1951 Chain = Lo.getValue(1); 1952 InFlag = Lo.getValue(2); 1953 VA = RVLocs[++i]; // skip ahead to next loc 1954 SDValue Hi = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32, 1955 InFlag); 1956 Chain = Hi.getValue(1); 1957 InFlag = Hi.getValue(2); 1958 if (!Subtarget->isLittle()) 1959 std::swap (Lo, Hi); 1960 Val = DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi); 1961 1962 if (VA.getLocVT() == MVT::v2f64) { 1963 SDValue Vec = DAG.getNode(ISD::UNDEF, dl, MVT::v2f64); 1964 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Vec, Val, 1965 DAG.getConstant(0, dl, MVT::i32)); 1966 1967 VA = RVLocs[++i]; // skip ahead to next loc 1968 Lo = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32, InFlag); 1969 Chain = Lo.getValue(1); 1970 InFlag = Lo.getValue(2); 1971 VA = RVLocs[++i]; // skip ahead to next loc 1972 Hi = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32, InFlag); 1973 Chain = Hi.getValue(1); 1974 InFlag = Hi.getValue(2); 1975 if (!Subtarget->isLittle()) 1976 std::swap (Lo, Hi); 1977 Val = DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi); 1978 Val = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Vec, Val, 1979 DAG.getConstant(1, dl, MVT::i32)); 1980 } 1981 } else { 1982 Val = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), VA.getLocVT(), 1983 InFlag); 1984 Chain = Val.getValue(1); 1985 InFlag = Val.getValue(2); 1986 } 1987 1988 switch (VA.getLocInfo()) { 1989 default: llvm_unreachable("Unknown loc info!"); 1990 case CCValAssign::Full: break; 1991 case CCValAssign::BCvt: 1992 Val = DAG.getNode(ISD::BITCAST, dl, VA.getValVT(), Val); 1993 break; 1994 } 1995 1996 InVals.push_back(Val); 1997 } 1998 1999 return Chain; 2000 } 2001 2002 /// LowerMemOpCallTo - Store the argument to the stack. 2003 SDValue ARMTargetLowering::LowerMemOpCallTo(SDValue Chain, SDValue StackPtr, 2004 SDValue Arg, const SDLoc &dl, 2005 SelectionDAG &DAG, 2006 const CCValAssign &VA, 2007 ISD::ArgFlagsTy Flags) const { 2008 unsigned LocMemOffset = VA.getLocMemOffset(); 2009 SDValue PtrOff = DAG.getIntPtrConstant(LocMemOffset, dl); 2010 PtrOff = DAG.getNode(ISD::ADD, dl, getPointerTy(DAG.getDataLayout()), 2011 StackPtr, PtrOff); 2012 return DAG.getStore( 2013 Chain, dl, Arg, PtrOff, 2014 MachinePointerInfo::getStack(DAG.getMachineFunction(), LocMemOffset)); 2015 } 2016 2017 void ARMTargetLowering::PassF64ArgInRegs(const SDLoc &dl, SelectionDAG &DAG, 2018 SDValue Chain, SDValue &Arg, 2019 RegsToPassVector &RegsToPass, 2020 CCValAssign &VA, CCValAssign &NextVA, 2021 SDValue &StackPtr, 2022 SmallVectorImpl<SDValue> &MemOpChains, 2023 ISD::ArgFlagsTy Flags) const { 2024 SDValue fmrrd = DAG.getNode(ARMISD::VMOVRRD, dl, 2025 DAG.getVTList(MVT::i32, MVT::i32), Arg); 2026 unsigned id = Subtarget->isLittle() ? 0 : 1; 2027 RegsToPass.push_back(std::make_pair(VA.getLocReg(), fmrrd.getValue(id))); 2028 2029 if (NextVA.isRegLoc()) 2030 RegsToPass.push_back(std::make_pair(NextVA.getLocReg(), fmrrd.getValue(1-id))); 2031 else { 2032 assert(NextVA.isMemLoc()); 2033 if (!StackPtr.getNode()) 2034 StackPtr = DAG.getCopyFromReg(Chain, dl, ARM::SP, 2035 getPointerTy(DAG.getDataLayout())); 2036 2037 MemOpChains.push_back(LowerMemOpCallTo(Chain, StackPtr, fmrrd.getValue(1-id), 2038 dl, DAG, NextVA, 2039 Flags)); 2040 } 2041 } 2042 2043 /// LowerCall - Lowering a call into a callseq_start <- 2044 /// ARMISD:CALL <- callseq_end chain. Also add input and output parameter 2045 /// nodes. 2046 SDValue 2047 ARMTargetLowering::LowerCall(TargetLowering::CallLoweringInfo &CLI, 2048 SmallVectorImpl<SDValue> &InVals) const { 2049 SelectionDAG &DAG = CLI.DAG; 2050 SDLoc &dl = CLI.DL; 2051 SmallVectorImpl<ISD::OutputArg> &Outs = CLI.Outs; 2052 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals; 2053 SmallVectorImpl<ISD::InputArg> &Ins = CLI.Ins; 2054 SDValue Chain = CLI.Chain; 2055 SDValue Callee = CLI.Callee; 2056 bool &isTailCall = CLI.IsTailCall; 2057 CallingConv::ID CallConv = CLI.CallConv; 2058 bool doesNotRet = CLI.DoesNotReturn; 2059 bool isVarArg = CLI.IsVarArg; 2060 2061 MachineFunction &MF = DAG.getMachineFunction(); 2062 MachineFunction::CallSiteInfo CSInfo; 2063 bool isStructRet = (Outs.empty()) ? false : Outs[0].Flags.isSRet(); 2064 bool isThisReturn = false; 2065 auto Attr = MF.getFunction().getFnAttribute("disable-tail-calls"); 2066 bool PreferIndirect = false; 2067 2068 // Disable tail calls if they're not supported. 2069 if (!Subtarget->supportsTailCall() || Attr.getValueAsString() == "true") 2070 isTailCall = false; 2071 2072 if (isa<GlobalAddressSDNode>(Callee)) { 2073 // If we're optimizing for minimum size and the function is called three or 2074 // more times in this block, we can improve codesize by calling indirectly 2075 // as BLXr has a 16-bit encoding. 2076 auto *GV = cast<GlobalAddressSDNode>(Callee)->getGlobal(); 2077 if (CLI.CS) { 2078 auto *BB = CLI.CS.getParent(); 2079 PreferIndirect = Subtarget->isThumb() && Subtarget->hasMinSize() && 2080 count_if(GV->users(), [&BB](const User *U) { 2081 return isa<Instruction>(U) && 2082 cast<Instruction>(U)->getParent() == BB; 2083 }) > 2; 2084 } 2085 } 2086 if (isTailCall) { 2087 // Check if it's really possible to do a tail call. 2088 isTailCall = IsEligibleForTailCallOptimization( 2089 Callee, CallConv, isVarArg, isStructRet, 2090 MF.getFunction().hasStructRetAttr(), Outs, OutVals, Ins, DAG, 2091 PreferIndirect); 2092 if (!isTailCall && CLI.CS && CLI.CS.isMustTailCall()) 2093 report_fatal_error("failed to perform tail call elimination on a call " 2094 "site marked musttail"); 2095 // We don't support GuaranteedTailCallOpt for ARM, only automatically 2096 // detected sibcalls. 2097 if (isTailCall) 2098 ++NumTailCalls; 2099 } 2100 2101 // Analyze operands of the call, assigning locations to each operand. 2102 SmallVector<CCValAssign, 16> ArgLocs; 2103 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), ArgLocs, 2104 *DAG.getContext()); 2105 CCInfo.AnalyzeCallOperands(Outs, CCAssignFnForCall(CallConv, isVarArg)); 2106 2107 // Get a count of how many bytes are to be pushed on the stack. 2108 unsigned NumBytes = CCInfo.getNextStackOffset(); 2109 2110 if (isTailCall) { 2111 // For tail calls, memory operands are available in our caller's stack. 2112 NumBytes = 0; 2113 } else { 2114 // Adjust the stack pointer for the new arguments... 2115 // These operations are automatically eliminated by the prolog/epilog pass 2116 Chain = DAG.getCALLSEQ_START(Chain, NumBytes, 0, dl); 2117 } 2118 2119 SDValue StackPtr = 2120 DAG.getCopyFromReg(Chain, dl, ARM::SP, getPointerTy(DAG.getDataLayout())); 2121 2122 RegsToPassVector RegsToPass; 2123 SmallVector<SDValue, 8> MemOpChains; 2124 2125 // Walk the register/memloc assignments, inserting copies/loads. In the case 2126 // of tail call optimization, arguments are handled later. 2127 for (unsigned i = 0, realArgIdx = 0, e = ArgLocs.size(); 2128 i != e; 2129 ++i, ++realArgIdx) { 2130 CCValAssign &VA = ArgLocs[i]; 2131 SDValue Arg = OutVals[realArgIdx]; 2132 ISD::ArgFlagsTy Flags = Outs[realArgIdx].Flags; 2133 bool isByVal = Flags.isByVal(); 2134 2135 // Promote the value if needed. 2136 switch (VA.getLocInfo()) { 2137 default: llvm_unreachable("Unknown loc info!"); 2138 case CCValAssign::Full: break; 2139 case CCValAssign::SExt: 2140 Arg = DAG.getNode(ISD::SIGN_EXTEND, dl, VA.getLocVT(), Arg); 2141 break; 2142 case CCValAssign::ZExt: 2143 Arg = DAG.getNode(ISD::ZERO_EXTEND, dl, VA.getLocVT(), Arg); 2144 break; 2145 case CCValAssign::AExt: 2146 Arg = DAG.getNode(ISD::ANY_EXTEND, dl, VA.getLocVT(), Arg); 2147 break; 2148 case CCValAssign::BCvt: 2149 Arg = DAG.getNode(ISD::BITCAST, dl, VA.getLocVT(), Arg); 2150 break; 2151 } 2152 2153 // f64 and v2f64 might be passed in i32 pairs and must be split into pieces 2154 if (VA.needsCustom()) { 2155 if (VA.getLocVT() == MVT::v2f64) { 2156 SDValue Op0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg, 2157 DAG.getConstant(0, dl, MVT::i32)); 2158 SDValue Op1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg, 2159 DAG.getConstant(1, dl, MVT::i32)); 2160 2161 PassF64ArgInRegs(dl, DAG, Chain, Op0, RegsToPass, 2162 VA, ArgLocs[++i], StackPtr, MemOpChains, Flags); 2163 2164 VA = ArgLocs[++i]; // skip ahead to next loc 2165 if (VA.isRegLoc()) { 2166 PassF64ArgInRegs(dl, DAG, Chain, Op1, RegsToPass, 2167 VA, ArgLocs[++i], StackPtr, MemOpChains, Flags); 2168 } else { 2169 assert(VA.isMemLoc()); 2170 2171 MemOpChains.push_back(LowerMemOpCallTo(Chain, StackPtr, Op1, 2172 dl, DAG, VA, Flags)); 2173 } 2174 } else { 2175 PassF64ArgInRegs(dl, DAG, Chain, Arg, RegsToPass, VA, ArgLocs[++i], 2176 StackPtr, MemOpChains, Flags); 2177 } 2178 } else if (VA.isRegLoc()) { 2179 if (realArgIdx == 0 && Flags.isReturned() && !Flags.isSwiftSelf() && 2180 Outs[0].VT == MVT::i32) { 2181 assert(VA.getLocVT() == MVT::i32 && 2182 "unexpected calling convention register assignment"); 2183 assert(!Ins.empty() && Ins[0].VT == MVT::i32 && 2184 "unexpected use of 'returned'"); 2185 isThisReturn = true; 2186 } 2187 const TargetOptions &Options = DAG.getTarget().Options; 2188 if (Options.EnableDebugEntryValues) 2189 CSInfo.emplace_back(VA.getLocReg(), i); 2190 RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg)); 2191 } else if (isByVal) { 2192 assert(VA.isMemLoc()); 2193 unsigned offset = 0; 2194 2195 // True if this byval aggregate will be split between registers 2196 // and memory. 2197 unsigned ByValArgsCount = CCInfo.getInRegsParamsCount(); 2198 unsigned CurByValIdx = CCInfo.getInRegsParamsProcessed(); 2199 2200 if (CurByValIdx < ByValArgsCount) { 2201 2202 unsigned RegBegin, RegEnd; 2203 CCInfo.getInRegsParamInfo(CurByValIdx, RegBegin, RegEnd); 2204 2205 EVT PtrVT = 2206 DAG.getTargetLoweringInfo().getPointerTy(DAG.getDataLayout()); 2207 unsigned int i, j; 2208 for (i = 0, j = RegBegin; j < RegEnd; i++, j++) { 2209 SDValue Const = DAG.getConstant(4*i, dl, MVT::i32); 2210 SDValue AddArg = DAG.getNode(ISD::ADD, dl, PtrVT, Arg, Const); 2211 SDValue Load = DAG.getLoad(PtrVT, dl, Chain, AddArg, 2212 MachinePointerInfo(), 2213 DAG.InferPtrAlignment(AddArg)); 2214 MemOpChains.push_back(Load.getValue(1)); 2215 RegsToPass.push_back(std::make_pair(j, Load)); 2216 } 2217 2218 // If parameter size outsides register area, "offset" value 2219 // helps us to calculate stack slot for remained part properly. 2220 offset = RegEnd - RegBegin; 2221 2222 CCInfo.nextInRegsParam(); 2223 } 2224 2225 if (Flags.getByValSize() > 4*offset) { 2226 auto PtrVT = getPointerTy(DAG.getDataLayout()); 2227 unsigned LocMemOffset = VA.getLocMemOffset(); 2228 SDValue StkPtrOff = DAG.getIntPtrConstant(LocMemOffset, dl); 2229 SDValue Dst = DAG.getNode(ISD::ADD, dl, PtrVT, StackPtr, StkPtrOff); 2230 SDValue SrcOffset = DAG.getIntPtrConstant(4*offset, dl); 2231 SDValue Src = DAG.getNode(ISD::ADD, dl, PtrVT, Arg, SrcOffset); 2232 SDValue SizeNode = DAG.getConstant(Flags.getByValSize() - 4*offset, dl, 2233 MVT::i32); 2234 SDValue AlignNode = DAG.getConstant(Flags.getByValAlign(), dl, 2235 MVT::i32); 2236 2237 SDVTList VTs = DAG.getVTList(MVT::Other, MVT::Glue); 2238 SDValue Ops[] = { Chain, Dst, Src, SizeNode, AlignNode}; 2239 MemOpChains.push_back(DAG.getNode(ARMISD::COPY_STRUCT_BYVAL, dl, VTs, 2240 Ops)); 2241 } 2242 } else if (!isTailCall) { 2243 assert(VA.isMemLoc()); 2244 2245 MemOpChains.push_back(LowerMemOpCallTo(Chain, StackPtr, Arg, 2246 dl, DAG, VA, Flags)); 2247 } 2248 } 2249 2250 if (!MemOpChains.empty()) 2251 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, MemOpChains); 2252 2253 // Build a sequence of copy-to-reg nodes chained together with token chain 2254 // and flag operands which copy the outgoing args into the appropriate regs. 2255 SDValue InFlag; 2256 for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) { 2257 Chain = DAG.getCopyToReg(Chain, dl, RegsToPass[i].first, 2258 RegsToPass[i].second, InFlag); 2259 InFlag = Chain.getValue(1); 2260 } 2261 2262 // If the callee is a GlobalAddress/ExternalSymbol node (quite common, every 2263 // direct call is) turn it into a TargetGlobalAddress/TargetExternalSymbol 2264 // node so that legalize doesn't hack it. 2265 bool isDirect = false; 2266 2267 const TargetMachine &TM = getTargetMachine(); 2268 const Module *Mod = MF.getFunction().getParent(); 2269 const GlobalValue *GV = nullptr; 2270 if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee)) 2271 GV = G->getGlobal(); 2272 bool isStub = 2273 !TM.shouldAssumeDSOLocal(*Mod, GV) && Subtarget->isTargetMachO(); 2274 2275 bool isARMFunc = !Subtarget->isThumb() || (isStub && !Subtarget->isMClass()); 2276 bool isLocalARMFunc = false; 2277 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 2278 auto PtrVt = getPointerTy(DAG.getDataLayout()); 2279 2280 if (Subtarget->genLongCalls()) { 2281 assert((!isPositionIndependent() || Subtarget->isTargetWindows()) && 2282 "long-calls codegen is not position independent!"); 2283 // Handle a global address or an external symbol. If it's not one of 2284 // those, the target's already in a register, so we don't need to do 2285 // anything extra. 2286 if (isa<GlobalAddressSDNode>(Callee)) { 2287 // Create a constant pool entry for the callee address 2288 unsigned ARMPCLabelIndex = AFI->createPICLabelUId(); 2289 ARMConstantPoolValue *CPV = 2290 ARMConstantPoolConstant::Create(GV, ARMPCLabelIndex, ARMCP::CPValue, 0); 2291 2292 // Get the address of the callee into a register 2293 SDValue CPAddr = DAG.getTargetConstantPool(CPV, PtrVt, 4); 2294 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr); 2295 Callee = DAG.getLoad( 2296 PtrVt, dl, DAG.getEntryNode(), CPAddr, 2297 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 2298 } else if (ExternalSymbolSDNode *S=dyn_cast<ExternalSymbolSDNode>(Callee)) { 2299 const char *Sym = S->getSymbol(); 2300 2301 // Create a constant pool entry for the callee address 2302 unsigned ARMPCLabelIndex = AFI->createPICLabelUId(); 2303 ARMConstantPoolValue *CPV = 2304 ARMConstantPoolSymbol::Create(*DAG.getContext(), Sym, 2305 ARMPCLabelIndex, 0); 2306 // Get the address of the callee into a register 2307 SDValue CPAddr = DAG.getTargetConstantPool(CPV, PtrVt, 4); 2308 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr); 2309 Callee = DAG.getLoad( 2310 PtrVt, dl, DAG.getEntryNode(), CPAddr, 2311 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 2312 } 2313 } else if (isa<GlobalAddressSDNode>(Callee)) { 2314 if (!PreferIndirect) { 2315 isDirect = true; 2316 bool isDef = GV->isStrongDefinitionForLinker(); 2317 2318 // ARM call to a local ARM function is predicable. 2319 isLocalARMFunc = !Subtarget->isThumb() && (isDef || !ARMInterworking); 2320 // tBX takes a register source operand. 2321 if (isStub && Subtarget->isThumb1Only() && !Subtarget->hasV5TOps()) { 2322 assert(Subtarget->isTargetMachO() && "WrapperPIC use on non-MachO?"); 2323 Callee = DAG.getNode( 2324 ARMISD::WrapperPIC, dl, PtrVt, 2325 DAG.getTargetGlobalAddress(GV, dl, PtrVt, 0, ARMII::MO_NONLAZY)); 2326 Callee = DAG.getLoad( 2327 PtrVt, dl, DAG.getEntryNode(), Callee, 2328 MachinePointerInfo::getGOT(DAG.getMachineFunction()), 2329 /* Alignment = */ 0, MachineMemOperand::MODereferenceable | 2330 MachineMemOperand::MOInvariant); 2331 } else if (Subtarget->isTargetCOFF()) { 2332 assert(Subtarget->isTargetWindows() && 2333 "Windows is the only supported COFF target"); 2334 unsigned TargetFlags = GV->hasDLLImportStorageClass() 2335 ? ARMII::MO_DLLIMPORT 2336 : ARMII::MO_NO_FLAG; 2337 Callee = DAG.getTargetGlobalAddress(GV, dl, PtrVt, /*offset=*/0, 2338 TargetFlags); 2339 if (GV->hasDLLImportStorageClass()) 2340 Callee = 2341 DAG.getLoad(PtrVt, dl, DAG.getEntryNode(), 2342 DAG.getNode(ARMISD::Wrapper, dl, PtrVt, Callee), 2343 MachinePointerInfo::getGOT(DAG.getMachineFunction())); 2344 } else { 2345 Callee = DAG.getTargetGlobalAddress(GV, dl, PtrVt, 0, 0); 2346 } 2347 } 2348 } else if (ExternalSymbolSDNode *S = dyn_cast<ExternalSymbolSDNode>(Callee)) { 2349 isDirect = true; 2350 // tBX takes a register source operand. 2351 const char *Sym = S->getSymbol(); 2352 if (isARMFunc && Subtarget->isThumb1Only() && !Subtarget->hasV5TOps()) { 2353 unsigned ARMPCLabelIndex = AFI->createPICLabelUId(); 2354 ARMConstantPoolValue *CPV = 2355 ARMConstantPoolSymbol::Create(*DAG.getContext(), Sym, 2356 ARMPCLabelIndex, 4); 2357 SDValue CPAddr = DAG.getTargetConstantPool(CPV, PtrVt, 4); 2358 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr); 2359 Callee = DAG.getLoad( 2360 PtrVt, dl, DAG.getEntryNode(), CPAddr, 2361 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 2362 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32); 2363 Callee = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVt, Callee, PICLabel); 2364 } else { 2365 Callee = DAG.getTargetExternalSymbol(Sym, PtrVt, 0); 2366 } 2367 } 2368 2369 // FIXME: handle tail calls differently. 2370 unsigned CallOpc; 2371 if (Subtarget->isThumb()) { 2372 if ((!isDirect || isARMFunc) && !Subtarget->hasV5TOps()) 2373 CallOpc = ARMISD::CALL_NOLINK; 2374 else 2375 CallOpc = ARMISD::CALL; 2376 } else { 2377 if (!isDirect && !Subtarget->hasV5TOps()) 2378 CallOpc = ARMISD::CALL_NOLINK; 2379 else if (doesNotRet && isDirect && Subtarget->hasRetAddrStack() && 2380 // Emit regular call when code size is the priority 2381 !Subtarget->hasMinSize()) 2382 // "mov lr, pc; b _foo" to avoid confusing the RSP 2383 CallOpc = ARMISD::CALL_NOLINK; 2384 else 2385 CallOpc = isLocalARMFunc ? ARMISD::CALL_PRED : ARMISD::CALL; 2386 } 2387 2388 std::vector<SDValue> Ops; 2389 Ops.push_back(Chain); 2390 Ops.push_back(Callee); 2391 2392 // Add argument registers to the end of the list so that they are known live 2393 // into the call. 2394 for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) 2395 Ops.push_back(DAG.getRegister(RegsToPass[i].first, 2396 RegsToPass[i].second.getValueType())); 2397 2398 // Add a register mask operand representing the call-preserved registers. 2399 if (!isTailCall) { 2400 const uint32_t *Mask; 2401 const ARMBaseRegisterInfo *ARI = Subtarget->getRegisterInfo(); 2402 if (isThisReturn) { 2403 // For 'this' returns, use the R0-preserving mask if applicable 2404 Mask = ARI->getThisReturnPreservedMask(MF, CallConv); 2405 if (!Mask) { 2406 // Set isThisReturn to false if the calling convention is not one that 2407 // allows 'returned' to be modeled in this way, so LowerCallResult does 2408 // not try to pass 'this' straight through 2409 isThisReturn = false; 2410 Mask = ARI->getCallPreservedMask(MF, CallConv); 2411 } 2412 } else 2413 Mask = ARI->getCallPreservedMask(MF, CallConv); 2414 2415 assert(Mask && "Missing call preserved mask for calling convention"); 2416 Ops.push_back(DAG.getRegisterMask(Mask)); 2417 } 2418 2419 if (InFlag.getNode()) 2420 Ops.push_back(InFlag); 2421 2422 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue); 2423 if (isTailCall) { 2424 MF.getFrameInfo().setHasTailCall(); 2425 SDValue Ret = DAG.getNode(ARMISD::TC_RETURN, dl, NodeTys, Ops); 2426 DAG.addCallSiteInfo(Ret.getNode(), std::move(CSInfo)); 2427 return Ret; 2428 } 2429 2430 // Returns a chain and a flag for retval copy to use. 2431 Chain = DAG.getNode(CallOpc, dl, NodeTys, Ops); 2432 InFlag = Chain.getValue(1); 2433 DAG.addCallSiteInfo(Chain.getNode(), std::move(CSInfo)); 2434 2435 Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(NumBytes, dl, true), 2436 DAG.getIntPtrConstant(0, dl, true), InFlag, dl); 2437 if (!Ins.empty()) 2438 InFlag = Chain.getValue(1); 2439 2440 // Handle result values, copying them out of physregs into vregs that we 2441 // return. 2442 return LowerCallResult(Chain, InFlag, CallConv, isVarArg, Ins, dl, DAG, 2443 InVals, isThisReturn, 2444 isThisReturn ? OutVals[0] : SDValue()); 2445 } 2446 2447 /// HandleByVal - Every parameter *after* a byval parameter is passed 2448 /// on the stack. Remember the next parameter register to allocate, 2449 /// and then confiscate the rest of the parameter registers to insure 2450 /// this. 2451 void ARMTargetLowering::HandleByVal(CCState *State, unsigned &Size, 2452 unsigned Align) const { 2453 // Byval (as with any stack) slots are always at least 4 byte aligned. 2454 Align = std::max(Align, 4U); 2455 2456 unsigned Reg = State->AllocateReg(GPRArgRegs); 2457 if (!Reg) 2458 return; 2459 2460 unsigned AlignInRegs = Align / 4; 2461 unsigned Waste = (ARM::R4 - Reg) % AlignInRegs; 2462 for (unsigned i = 0; i < Waste; ++i) 2463 Reg = State->AllocateReg(GPRArgRegs); 2464 2465 if (!Reg) 2466 return; 2467 2468 unsigned Excess = 4 * (ARM::R4 - Reg); 2469 2470 // Special case when NSAA != SP and parameter size greater than size of 2471 // all remained GPR regs. In that case we can't split parameter, we must 2472 // send it to stack. We also must set NCRN to R4, so waste all 2473 // remained registers. 2474 const unsigned NSAAOffset = State->getNextStackOffset(); 2475 if (NSAAOffset != 0 && Size > Excess) { 2476 while (State->AllocateReg(GPRArgRegs)) 2477 ; 2478 return; 2479 } 2480 2481 // First register for byval parameter is the first register that wasn't 2482 // allocated before this method call, so it would be "reg". 2483 // If parameter is small enough to be saved in range [reg, r4), then 2484 // the end (first after last) register would be reg + param-size-in-regs, 2485 // else parameter would be splitted between registers and stack, 2486 // end register would be r4 in this case. 2487 unsigned ByValRegBegin = Reg; 2488 unsigned ByValRegEnd = std::min<unsigned>(Reg + Size / 4, ARM::R4); 2489 State->addInRegsParamInfo(ByValRegBegin, ByValRegEnd); 2490 // Note, first register is allocated in the beginning of function already, 2491 // allocate remained amount of registers we need. 2492 for (unsigned i = Reg + 1; i != ByValRegEnd; ++i) 2493 State->AllocateReg(GPRArgRegs); 2494 // A byval parameter that is split between registers and memory needs its 2495 // size truncated here. 2496 // In the case where the entire structure fits in registers, we set the 2497 // size in memory to zero. 2498 Size = std::max<int>(Size - Excess, 0); 2499 } 2500 2501 /// MatchingStackOffset - Return true if the given stack call argument is 2502 /// already available in the same position (relatively) of the caller's 2503 /// incoming argument stack. 2504 static 2505 bool MatchingStackOffset(SDValue Arg, unsigned Offset, ISD::ArgFlagsTy Flags, 2506 MachineFrameInfo &MFI, const MachineRegisterInfo *MRI, 2507 const TargetInstrInfo *TII) { 2508 unsigned Bytes = Arg.getValueSizeInBits() / 8; 2509 int FI = std::numeric_limits<int>::max(); 2510 if (Arg.getOpcode() == ISD::CopyFromReg) { 2511 unsigned VR = cast<RegisterSDNode>(Arg.getOperand(1))->getReg(); 2512 if (!Register::isVirtualRegister(VR)) 2513 return false; 2514 MachineInstr *Def = MRI->getVRegDef(VR); 2515 if (!Def) 2516 return false; 2517 if (!Flags.isByVal()) { 2518 if (!TII->isLoadFromStackSlot(*Def, FI)) 2519 return false; 2520 } else { 2521 return false; 2522 } 2523 } else if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Arg)) { 2524 if (Flags.isByVal()) 2525 // ByVal argument is passed in as a pointer but it's now being 2526 // dereferenced. e.g. 2527 // define @foo(%struct.X* %A) { 2528 // tail call @bar(%struct.X* byval %A) 2529 // } 2530 return false; 2531 SDValue Ptr = Ld->getBasePtr(); 2532 FrameIndexSDNode *FINode = dyn_cast<FrameIndexSDNode>(Ptr); 2533 if (!FINode) 2534 return false; 2535 FI = FINode->getIndex(); 2536 } else 2537 return false; 2538 2539 assert(FI != std::numeric_limits<int>::max()); 2540 if (!MFI.isFixedObjectIndex(FI)) 2541 return false; 2542 return Offset == MFI.getObjectOffset(FI) && Bytes == MFI.getObjectSize(FI); 2543 } 2544 2545 /// IsEligibleForTailCallOptimization - Check whether the call is eligible 2546 /// for tail call optimization. Targets which want to do tail call 2547 /// optimization should implement this function. 2548 bool ARMTargetLowering::IsEligibleForTailCallOptimization( 2549 SDValue Callee, CallingConv::ID CalleeCC, bool isVarArg, 2550 bool isCalleeStructRet, bool isCallerStructRet, 2551 const SmallVectorImpl<ISD::OutputArg> &Outs, 2552 const SmallVectorImpl<SDValue> &OutVals, 2553 const SmallVectorImpl<ISD::InputArg> &Ins, SelectionDAG &DAG, 2554 const bool isIndirect) const { 2555 MachineFunction &MF = DAG.getMachineFunction(); 2556 const Function &CallerF = MF.getFunction(); 2557 CallingConv::ID CallerCC = CallerF.getCallingConv(); 2558 2559 assert(Subtarget->supportsTailCall()); 2560 2561 // Indirect tail calls cannot be optimized for Thumb1 if the args 2562 // to the call take up r0-r3. The reason is that there are no legal registers 2563 // left to hold the pointer to the function to be called. 2564 if (Subtarget->isThumb1Only() && Outs.size() >= 4 && 2565 (!isa<GlobalAddressSDNode>(Callee.getNode()) || isIndirect)) 2566 return false; 2567 2568 // Look for obvious safe cases to perform tail call optimization that do not 2569 // require ABI changes. This is what gcc calls sibcall. 2570 2571 // Exception-handling functions need a special set of instructions to indicate 2572 // a return to the hardware. Tail-calling another function would probably 2573 // break this. 2574 if (CallerF.hasFnAttribute("interrupt")) 2575 return false; 2576 2577 // Also avoid sibcall optimization if either caller or callee uses struct 2578 // return semantics. 2579 if (isCalleeStructRet || isCallerStructRet) 2580 return false; 2581 2582 // Externally-defined functions with weak linkage should not be 2583 // tail-called on ARM when the OS does not support dynamic 2584 // pre-emption of symbols, as the AAELF spec requires normal calls 2585 // to undefined weak functions to be replaced with a NOP or jump to the 2586 // next instruction. The behaviour of branch instructions in this 2587 // situation (as used for tail calls) is implementation-defined, so we 2588 // cannot rely on the linker replacing the tail call with a return. 2589 if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee)) { 2590 const GlobalValue *GV = G->getGlobal(); 2591 const Triple &TT = getTargetMachine().getTargetTriple(); 2592 if (GV->hasExternalWeakLinkage() && 2593 (!TT.isOSWindows() || TT.isOSBinFormatELF() || TT.isOSBinFormatMachO())) 2594 return false; 2595 } 2596 2597 // Check that the call results are passed in the same way. 2598 LLVMContext &C = *DAG.getContext(); 2599 if (!CCState::resultsCompatible(CalleeCC, CallerCC, MF, C, Ins, 2600 CCAssignFnForReturn(CalleeCC, isVarArg), 2601 CCAssignFnForReturn(CallerCC, isVarArg))) 2602 return false; 2603 // The callee has to preserve all registers the caller needs to preserve. 2604 const ARMBaseRegisterInfo *TRI = Subtarget->getRegisterInfo(); 2605 const uint32_t *CallerPreserved = TRI->getCallPreservedMask(MF, CallerCC); 2606 if (CalleeCC != CallerCC) { 2607 const uint32_t *CalleePreserved = TRI->getCallPreservedMask(MF, CalleeCC); 2608 if (!TRI->regmaskSubsetEqual(CallerPreserved, CalleePreserved)) 2609 return false; 2610 } 2611 2612 // If Caller's vararg or byval argument has been split between registers and 2613 // stack, do not perform tail call, since part of the argument is in caller's 2614 // local frame. 2615 const ARMFunctionInfo *AFI_Caller = MF.getInfo<ARMFunctionInfo>(); 2616 if (AFI_Caller->getArgRegsSaveSize()) 2617 return false; 2618 2619 // If the callee takes no arguments then go on to check the results of the 2620 // call. 2621 if (!Outs.empty()) { 2622 // Check if stack adjustment is needed. For now, do not do this if any 2623 // argument is passed on the stack. 2624 SmallVector<CCValAssign, 16> ArgLocs; 2625 CCState CCInfo(CalleeCC, isVarArg, MF, ArgLocs, C); 2626 CCInfo.AnalyzeCallOperands(Outs, CCAssignFnForCall(CalleeCC, isVarArg)); 2627 if (CCInfo.getNextStackOffset()) { 2628 // Check if the arguments are already laid out in the right way as 2629 // the caller's fixed stack objects. 2630 MachineFrameInfo &MFI = MF.getFrameInfo(); 2631 const MachineRegisterInfo *MRI = &MF.getRegInfo(); 2632 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 2633 for (unsigned i = 0, realArgIdx = 0, e = ArgLocs.size(); 2634 i != e; 2635 ++i, ++realArgIdx) { 2636 CCValAssign &VA = ArgLocs[i]; 2637 EVT RegVT = VA.getLocVT(); 2638 SDValue Arg = OutVals[realArgIdx]; 2639 ISD::ArgFlagsTy Flags = Outs[realArgIdx].Flags; 2640 if (VA.getLocInfo() == CCValAssign::Indirect) 2641 return false; 2642 if (VA.needsCustom()) { 2643 // f64 and vector types are split into multiple registers or 2644 // register/stack-slot combinations. The types will not match 2645 // the registers; give up on memory f64 refs until we figure 2646 // out what to do about this. 2647 if (!VA.isRegLoc()) 2648 return false; 2649 if (!ArgLocs[++i].isRegLoc()) 2650 return false; 2651 if (RegVT == MVT::v2f64) { 2652 if (!ArgLocs[++i].isRegLoc()) 2653 return false; 2654 if (!ArgLocs[++i].isRegLoc()) 2655 return false; 2656 } 2657 } else if (!VA.isRegLoc()) { 2658 if (!MatchingStackOffset(Arg, VA.getLocMemOffset(), Flags, 2659 MFI, MRI, TII)) 2660 return false; 2661 } 2662 } 2663 } 2664 2665 const MachineRegisterInfo &MRI = MF.getRegInfo(); 2666 if (!parametersInCSRMatch(MRI, CallerPreserved, ArgLocs, OutVals)) 2667 return false; 2668 } 2669 2670 return true; 2671 } 2672 2673 bool 2674 ARMTargetLowering::CanLowerReturn(CallingConv::ID CallConv, 2675 MachineFunction &MF, bool isVarArg, 2676 const SmallVectorImpl<ISD::OutputArg> &Outs, 2677 LLVMContext &Context) const { 2678 SmallVector<CCValAssign, 16> RVLocs; 2679 CCState CCInfo(CallConv, isVarArg, MF, RVLocs, Context); 2680 return CCInfo.CheckReturn(Outs, CCAssignFnForReturn(CallConv, isVarArg)); 2681 } 2682 2683 static SDValue LowerInterruptReturn(SmallVectorImpl<SDValue> &RetOps, 2684 const SDLoc &DL, SelectionDAG &DAG) { 2685 const MachineFunction &MF = DAG.getMachineFunction(); 2686 const Function &F = MF.getFunction(); 2687 2688 StringRef IntKind = F.getFnAttribute("interrupt").getValueAsString(); 2689 2690 // See ARM ARM v7 B1.8.3. On exception entry LR is set to a possibly offset 2691 // version of the "preferred return address". These offsets affect the return 2692 // instruction if this is a return from PL1 without hypervisor extensions. 2693 // IRQ/FIQ: +4 "subs pc, lr, #4" 2694 // SWI: 0 "subs pc, lr, #0" 2695 // ABORT: +4 "subs pc, lr, #4" 2696 // UNDEF: +4/+2 "subs pc, lr, #0" 2697 // UNDEF varies depending on where the exception came from ARM or Thumb 2698 // mode. Alongside GCC, we throw our hands up in disgust and pretend it's 0. 2699 2700 int64_t LROffset; 2701 if (IntKind == "" || IntKind == "IRQ" || IntKind == "FIQ" || 2702 IntKind == "ABORT") 2703 LROffset = 4; 2704 else if (IntKind == "SWI" || IntKind == "UNDEF") 2705 LROffset = 0; 2706 else 2707 report_fatal_error("Unsupported interrupt attribute. If present, value " 2708 "must be one of: IRQ, FIQ, SWI, ABORT or UNDEF"); 2709 2710 RetOps.insert(RetOps.begin() + 1, 2711 DAG.getConstant(LROffset, DL, MVT::i32, false)); 2712 2713 return DAG.getNode(ARMISD::INTRET_FLAG, DL, MVT::Other, RetOps); 2714 } 2715 2716 SDValue 2717 ARMTargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv, 2718 bool isVarArg, 2719 const SmallVectorImpl<ISD::OutputArg> &Outs, 2720 const SmallVectorImpl<SDValue> &OutVals, 2721 const SDLoc &dl, SelectionDAG &DAG) const { 2722 // CCValAssign - represent the assignment of the return value to a location. 2723 SmallVector<CCValAssign, 16> RVLocs; 2724 2725 // CCState - Info about the registers and stack slots. 2726 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), RVLocs, 2727 *DAG.getContext()); 2728 2729 // Analyze outgoing return values. 2730 CCInfo.AnalyzeReturn(Outs, CCAssignFnForReturn(CallConv, isVarArg)); 2731 2732 SDValue Flag; 2733 SmallVector<SDValue, 4> RetOps; 2734 RetOps.push_back(Chain); // Operand #0 = Chain (updated below) 2735 bool isLittleEndian = Subtarget->isLittle(); 2736 2737 MachineFunction &MF = DAG.getMachineFunction(); 2738 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 2739 AFI->setReturnRegsCount(RVLocs.size()); 2740 2741 // Copy the result values into the output registers. 2742 for (unsigned i = 0, realRVLocIdx = 0; 2743 i != RVLocs.size(); 2744 ++i, ++realRVLocIdx) { 2745 CCValAssign &VA = RVLocs[i]; 2746 assert(VA.isRegLoc() && "Can only return in registers!"); 2747 2748 SDValue Arg = OutVals[realRVLocIdx]; 2749 bool ReturnF16 = false; 2750 2751 if (Subtarget->hasFullFP16() && Subtarget->isTargetHardFloat()) { 2752 // Half-precision return values can be returned like this: 2753 // 2754 // t11 f16 = fadd ... 2755 // t12: i16 = bitcast t11 2756 // t13: i32 = zero_extend t12 2757 // t14: f32 = bitcast t13 <~~~~~~~ Arg 2758 // 2759 // to avoid code generation for bitcasts, we simply set Arg to the node 2760 // that produces the f16 value, t11 in this case. 2761 // 2762 if (Arg.getValueType() == MVT::f32 && Arg.getOpcode() == ISD::BITCAST) { 2763 SDValue ZE = Arg.getOperand(0); 2764 if (ZE.getOpcode() == ISD::ZERO_EXTEND && ZE.getValueType() == MVT::i32) { 2765 SDValue BC = ZE.getOperand(0); 2766 if (BC.getOpcode() == ISD::BITCAST && BC.getValueType() == MVT::i16) { 2767 Arg = BC.getOperand(0); 2768 ReturnF16 = true; 2769 } 2770 } 2771 } 2772 } 2773 2774 switch (VA.getLocInfo()) { 2775 default: llvm_unreachable("Unknown loc info!"); 2776 case CCValAssign::Full: break; 2777 case CCValAssign::BCvt: 2778 if (!ReturnF16) 2779 Arg = DAG.getNode(ISD::BITCAST, dl, VA.getLocVT(), Arg); 2780 break; 2781 } 2782 2783 if (VA.needsCustom()) { 2784 if (VA.getLocVT() == MVT::v2f64) { 2785 // Extract the first half and return it in two registers. 2786 SDValue Half = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg, 2787 DAG.getConstant(0, dl, MVT::i32)); 2788 SDValue HalfGPRs = DAG.getNode(ARMISD::VMOVRRD, dl, 2789 DAG.getVTList(MVT::i32, MVT::i32), Half); 2790 2791 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(), 2792 HalfGPRs.getValue(isLittleEndian ? 0 : 1), 2793 Flag); 2794 Flag = Chain.getValue(1); 2795 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT())); 2796 VA = RVLocs[++i]; // skip ahead to next loc 2797 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(), 2798 HalfGPRs.getValue(isLittleEndian ? 1 : 0), 2799 Flag); 2800 Flag = Chain.getValue(1); 2801 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT())); 2802 VA = RVLocs[++i]; // skip ahead to next loc 2803 2804 // Extract the 2nd half and fall through to handle it as an f64 value. 2805 Arg = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg, 2806 DAG.getConstant(1, dl, MVT::i32)); 2807 } 2808 // Legalize ret f64 -> ret 2 x i32. We always have fmrrd if f64 is 2809 // available. 2810 SDValue fmrrd = DAG.getNode(ARMISD::VMOVRRD, dl, 2811 DAG.getVTList(MVT::i32, MVT::i32), Arg); 2812 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(), 2813 fmrrd.getValue(isLittleEndian ? 0 : 1), 2814 Flag); 2815 Flag = Chain.getValue(1); 2816 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT())); 2817 VA = RVLocs[++i]; // skip ahead to next loc 2818 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(), 2819 fmrrd.getValue(isLittleEndian ? 1 : 0), 2820 Flag); 2821 } else 2822 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(), Arg, Flag); 2823 2824 // Guarantee that all emitted copies are 2825 // stuck together, avoiding something bad. 2826 Flag = Chain.getValue(1); 2827 RetOps.push_back(DAG.getRegister(VA.getLocReg(), 2828 ReturnF16 ? MVT::f16 : VA.getLocVT())); 2829 } 2830 const ARMBaseRegisterInfo *TRI = Subtarget->getRegisterInfo(); 2831 const MCPhysReg *I = 2832 TRI->getCalleeSavedRegsViaCopy(&DAG.getMachineFunction()); 2833 if (I) { 2834 for (; *I; ++I) { 2835 if (ARM::GPRRegClass.contains(*I)) 2836 RetOps.push_back(DAG.getRegister(*I, MVT::i32)); 2837 else if (ARM::DPRRegClass.contains(*I)) 2838 RetOps.push_back(DAG.getRegister(*I, MVT::getFloatingPointVT(64))); 2839 else 2840 llvm_unreachable("Unexpected register class in CSRsViaCopy!"); 2841 } 2842 } 2843 2844 // Update chain and glue. 2845 RetOps[0] = Chain; 2846 if (Flag.getNode()) 2847 RetOps.push_back(Flag); 2848 2849 // CPUs which aren't M-class use a special sequence to return from 2850 // exceptions (roughly, any instruction setting pc and cpsr simultaneously, 2851 // though we use "subs pc, lr, #N"). 2852 // 2853 // M-class CPUs actually use a normal return sequence with a special 2854 // (hardware-provided) value in LR, so the normal code path works. 2855 if (DAG.getMachineFunction().getFunction().hasFnAttribute("interrupt") && 2856 !Subtarget->isMClass()) { 2857 if (Subtarget->isThumb1Only()) 2858 report_fatal_error("interrupt attribute is not supported in Thumb1"); 2859 return LowerInterruptReturn(RetOps, dl, DAG); 2860 } 2861 2862 return DAG.getNode(ARMISD::RET_FLAG, dl, MVT::Other, RetOps); 2863 } 2864 2865 bool ARMTargetLowering::isUsedByReturnOnly(SDNode *N, SDValue &Chain) const { 2866 if (N->getNumValues() != 1) 2867 return false; 2868 if (!N->hasNUsesOfValue(1, 0)) 2869 return false; 2870 2871 SDValue TCChain = Chain; 2872 SDNode *Copy = *N->use_begin(); 2873 if (Copy->getOpcode() == ISD::CopyToReg) { 2874 // If the copy has a glue operand, we conservatively assume it isn't safe to 2875 // perform a tail call. 2876 if (Copy->getOperand(Copy->getNumOperands()-1).getValueType() == MVT::Glue) 2877 return false; 2878 TCChain = Copy->getOperand(0); 2879 } else if (Copy->getOpcode() == ARMISD::VMOVRRD) { 2880 SDNode *VMov = Copy; 2881 // f64 returned in a pair of GPRs. 2882 SmallPtrSet<SDNode*, 2> Copies; 2883 for (SDNode::use_iterator UI = VMov->use_begin(), UE = VMov->use_end(); 2884 UI != UE; ++UI) { 2885 if (UI->getOpcode() != ISD::CopyToReg) 2886 return false; 2887 Copies.insert(*UI); 2888 } 2889 if (Copies.size() > 2) 2890 return false; 2891 2892 for (SDNode::use_iterator UI = VMov->use_begin(), UE = VMov->use_end(); 2893 UI != UE; ++UI) { 2894 SDValue UseChain = UI->getOperand(0); 2895 if (Copies.count(UseChain.getNode())) 2896 // Second CopyToReg 2897 Copy = *UI; 2898 else { 2899 // We are at the top of this chain. 2900 // If the copy has a glue operand, we conservatively assume it 2901 // isn't safe to perform a tail call. 2902 if (UI->getOperand(UI->getNumOperands()-1).getValueType() == MVT::Glue) 2903 return false; 2904 // First CopyToReg 2905 TCChain = UseChain; 2906 } 2907 } 2908 } else if (Copy->getOpcode() == ISD::BITCAST) { 2909 // f32 returned in a single GPR. 2910 if (!Copy->hasOneUse()) 2911 return false; 2912 Copy = *Copy->use_begin(); 2913 if (Copy->getOpcode() != ISD::CopyToReg || !Copy->hasNUsesOfValue(1, 0)) 2914 return false; 2915 // If the copy has a glue operand, we conservatively assume it isn't safe to 2916 // perform a tail call. 2917 if (Copy->getOperand(Copy->getNumOperands()-1).getValueType() == MVT::Glue) 2918 return false; 2919 TCChain = Copy->getOperand(0); 2920 } else { 2921 return false; 2922 } 2923 2924 bool HasRet = false; 2925 for (SDNode::use_iterator UI = Copy->use_begin(), UE = Copy->use_end(); 2926 UI != UE; ++UI) { 2927 if (UI->getOpcode() != ARMISD::RET_FLAG && 2928 UI->getOpcode() != ARMISD::INTRET_FLAG) 2929 return false; 2930 HasRet = true; 2931 } 2932 2933 if (!HasRet) 2934 return false; 2935 2936 Chain = TCChain; 2937 return true; 2938 } 2939 2940 bool ARMTargetLowering::mayBeEmittedAsTailCall(const CallInst *CI) const { 2941 if (!Subtarget->supportsTailCall()) 2942 return false; 2943 2944 auto Attr = 2945 CI->getParent()->getParent()->getFnAttribute("disable-tail-calls"); 2946 if (!CI->isTailCall() || Attr.getValueAsString() == "true") 2947 return false; 2948 2949 return true; 2950 } 2951 2952 // Trying to write a 64 bit value so need to split into two 32 bit values first, 2953 // and pass the lower and high parts through. 2954 static SDValue LowerWRITE_REGISTER(SDValue Op, SelectionDAG &DAG) { 2955 SDLoc DL(Op); 2956 SDValue WriteValue = Op->getOperand(2); 2957 2958 // This function is only supposed to be called for i64 type argument. 2959 assert(WriteValue.getValueType() == MVT::i64 2960 && "LowerWRITE_REGISTER called for non-i64 type argument."); 2961 2962 SDValue Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i32, WriteValue, 2963 DAG.getConstant(0, DL, MVT::i32)); 2964 SDValue Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i32, WriteValue, 2965 DAG.getConstant(1, DL, MVT::i32)); 2966 SDValue Ops[] = { Op->getOperand(0), Op->getOperand(1), Lo, Hi }; 2967 return DAG.getNode(ISD::WRITE_REGISTER, DL, MVT::Other, Ops); 2968 } 2969 2970 // ConstantPool, JumpTable, GlobalAddress, and ExternalSymbol are lowered as 2971 // their target counterpart wrapped in the ARMISD::Wrapper node. Suppose N is 2972 // one of the above mentioned nodes. It has to be wrapped because otherwise 2973 // Select(N) returns N. So the raw TargetGlobalAddress nodes, etc. can only 2974 // be used to form addressing mode. These wrapped nodes will be selected 2975 // into MOVi. 2976 SDValue ARMTargetLowering::LowerConstantPool(SDValue Op, 2977 SelectionDAG &DAG) const { 2978 EVT PtrVT = Op.getValueType(); 2979 // FIXME there is no actual debug info here 2980 SDLoc dl(Op); 2981 ConstantPoolSDNode *CP = cast<ConstantPoolSDNode>(Op); 2982 SDValue Res; 2983 2984 // When generating execute-only code Constant Pools must be promoted to the 2985 // global data section. It's a bit ugly that we can't share them across basic 2986 // blocks, but this way we guarantee that execute-only behaves correct with 2987 // position-independent addressing modes. 2988 if (Subtarget->genExecuteOnly()) { 2989 auto AFI = DAG.getMachineFunction().getInfo<ARMFunctionInfo>(); 2990 auto T = const_cast<Type*>(CP->getType()); 2991 auto C = const_cast<Constant*>(CP->getConstVal()); 2992 auto M = const_cast<Module*>(DAG.getMachineFunction(). 2993 getFunction().getParent()); 2994 auto GV = new GlobalVariable( 2995 *M, T, /*isConstant=*/true, GlobalVariable::InternalLinkage, C, 2996 Twine(DAG.getDataLayout().getPrivateGlobalPrefix()) + "CP" + 2997 Twine(DAG.getMachineFunction().getFunctionNumber()) + "_" + 2998 Twine(AFI->createPICLabelUId()) 2999 ); 3000 SDValue GA = DAG.getTargetGlobalAddress(dyn_cast<GlobalValue>(GV), 3001 dl, PtrVT); 3002 return LowerGlobalAddress(GA, DAG); 3003 } 3004 3005 if (CP->isMachineConstantPoolEntry()) 3006 Res = DAG.getTargetConstantPool(CP->getMachineCPVal(), PtrVT, 3007 CP->getAlignment()); 3008 else 3009 Res = DAG.getTargetConstantPool(CP->getConstVal(), PtrVT, 3010 CP->getAlignment()); 3011 return DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Res); 3012 } 3013 3014 unsigned ARMTargetLowering::getJumpTableEncoding() const { 3015 return MachineJumpTableInfo::EK_Inline; 3016 } 3017 3018 SDValue ARMTargetLowering::LowerBlockAddress(SDValue Op, 3019 SelectionDAG &DAG) const { 3020 MachineFunction &MF = DAG.getMachineFunction(); 3021 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3022 unsigned ARMPCLabelIndex = 0; 3023 SDLoc DL(Op); 3024 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3025 const BlockAddress *BA = cast<BlockAddressSDNode>(Op)->getBlockAddress(); 3026 SDValue CPAddr; 3027 bool IsPositionIndependent = isPositionIndependent() || Subtarget->isROPI(); 3028 if (!IsPositionIndependent) { 3029 CPAddr = DAG.getTargetConstantPool(BA, PtrVT, 4); 3030 } else { 3031 unsigned PCAdj = Subtarget->isThumb() ? 4 : 8; 3032 ARMPCLabelIndex = AFI->createPICLabelUId(); 3033 ARMConstantPoolValue *CPV = 3034 ARMConstantPoolConstant::Create(BA, ARMPCLabelIndex, 3035 ARMCP::CPBlockAddress, PCAdj); 3036 CPAddr = DAG.getTargetConstantPool(CPV, PtrVT, 4); 3037 } 3038 CPAddr = DAG.getNode(ARMISD::Wrapper, DL, PtrVT, CPAddr); 3039 SDValue Result = DAG.getLoad( 3040 PtrVT, DL, DAG.getEntryNode(), CPAddr, 3041 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3042 if (!IsPositionIndependent) 3043 return Result; 3044 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, DL, MVT::i32); 3045 return DAG.getNode(ARMISD::PIC_ADD, DL, PtrVT, Result, PICLabel); 3046 } 3047 3048 /// Convert a TLS address reference into the correct sequence of loads 3049 /// and calls to compute the variable's address for Darwin, and return an 3050 /// SDValue containing the final node. 3051 3052 /// Darwin only has one TLS scheme which must be capable of dealing with the 3053 /// fully general situation, in the worst case. This means: 3054 /// + "extern __thread" declaration. 3055 /// + Defined in a possibly unknown dynamic library. 3056 /// 3057 /// The general system is that each __thread variable has a [3 x i32] descriptor 3058 /// which contains information used by the runtime to calculate the address. The 3059 /// only part of this the compiler needs to know about is the first word, which 3060 /// contains a function pointer that must be called with the address of the 3061 /// entire descriptor in "r0". 3062 /// 3063 /// Since this descriptor may be in a different unit, in general access must 3064 /// proceed along the usual ARM rules. A common sequence to produce is: 3065 /// 3066 /// movw rT1, :lower16:_var$non_lazy_ptr 3067 /// movt rT1, :upper16:_var$non_lazy_ptr 3068 /// ldr r0, [rT1] 3069 /// ldr rT2, [r0] 3070 /// blx rT2 3071 /// [...address now in r0...] 3072 SDValue 3073 ARMTargetLowering::LowerGlobalTLSAddressDarwin(SDValue Op, 3074 SelectionDAG &DAG) const { 3075 assert(Subtarget->isTargetDarwin() && 3076 "This function expects a Darwin target"); 3077 SDLoc DL(Op); 3078 3079 // First step is to get the address of the actua global symbol. This is where 3080 // the TLS descriptor lives. 3081 SDValue DescAddr = LowerGlobalAddressDarwin(Op, DAG); 3082 3083 // The first entry in the descriptor is a function pointer that we must call 3084 // to obtain the address of the variable. 3085 SDValue Chain = DAG.getEntryNode(); 3086 SDValue FuncTLVGet = DAG.getLoad( 3087 MVT::i32, DL, Chain, DescAddr, 3088 MachinePointerInfo::getGOT(DAG.getMachineFunction()), 3089 /* Alignment = */ 4, 3090 MachineMemOperand::MONonTemporal | MachineMemOperand::MODereferenceable | 3091 MachineMemOperand::MOInvariant); 3092 Chain = FuncTLVGet.getValue(1); 3093 3094 MachineFunction &F = DAG.getMachineFunction(); 3095 MachineFrameInfo &MFI = F.getFrameInfo(); 3096 MFI.setAdjustsStack(true); 3097 3098 // TLS calls preserve all registers except those that absolutely must be 3099 // trashed: R0 (it takes an argument), LR (it's a call) and CPSR (let's not be 3100 // silly). 3101 auto TRI = 3102 getTargetMachine().getSubtargetImpl(F.getFunction())->getRegisterInfo(); 3103 auto ARI = static_cast<const ARMRegisterInfo *>(TRI); 3104 const uint32_t *Mask = ARI->getTLSCallPreservedMask(DAG.getMachineFunction()); 3105 3106 // Finally, we can make the call. This is just a degenerate version of a 3107 // normal AArch64 call node: r0 takes the address of the descriptor, and 3108 // returns the address of the variable in this thread. 3109 Chain = DAG.getCopyToReg(Chain, DL, ARM::R0, DescAddr, SDValue()); 3110 Chain = 3111 DAG.getNode(ARMISD::CALL, DL, DAG.getVTList(MVT::Other, MVT::Glue), 3112 Chain, FuncTLVGet, DAG.getRegister(ARM::R0, MVT::i32), 3113 DAG.getRegisterMask(Mask), Chain.getValue(1)); 3114 return DAG.getCopyFromReg(Chain, DL, ARM::R0, MVT::i32, Chain.getValue(1)); 3115 } 3116 3117 SDValue 3118 ARMTargetLowering::LowerGlobalTLSAddressWindows(SDValue Op, 3119 SelectionDAG &DAG) const { 3120 assert(Subtarget->isTargetWindows() && "Windows specific TLS lowering"); 3121 3122 SDValue Chain = DAG.getEntryNode(); 3123 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3124 SDLoc DL(Op); 3125 3126 // Load the current TEB (thread environment block) 3127 SDValue Ops[] = {Chain, 3128 DAG.getTargetConstant(Intrinsic::arm_mrc, DL, MVT::i32), 3129 DAG.getTargetConstant(15, DL, MVT::i32), 3130 DAG.getTargetConstant(0, DL, MVT::i32), 3131 DAG.getTargetConstant(13, DL, MVT::i32), 3132 DAG.getTargetConstant(0, DL, MVT::i32), 3133 DAG.getTargetConstant(2, DL, MVT::i32)}; 3134 SDValue CurrentTEB = DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL, 3135 DAG.getVTList(MVT::i32, MVT::Other), Ops); 3136 3137 SDValue TEB = CurrentTEB.getValue(0); 3138 Chain = CurrentTEB.getValue(1); 3139 3140 // Load the ThreadLocalStoragePointer from the TEB 3141 // A pointer to the TLS array is located at offset 0x2c from the TEB. 3142 SDValue TLSArray = 3143 DAG.getNode(ISD::ADD, DL, PtrVT, TEB, DAG.getIntPtrConstant(0x2c, DL)); 3144 TLSArray = DAG.getLoad(PtrVT, DL, Chain, TLSArray, MachinePointerInfo()); 3145 3146 // The pointer to the thread's TLS data area is at the TLS Index scaled by 4 3147 // offset into the TLSArray. 3148 3149 // Load the TLS index from the C runtime 3150 SDValue TLSIndex = 3151 DAG.getTargetExternalSymbol("_tls_index", PtrVT, ARMII::MO_NO_FLAG); 3152 TLSIndex = DAG.getNode(ARMISD::Wrapper, DL, PtrVT, TLSIndex); 3153 TLSIndex = DAG.getLoad(PtrVT, DL, Chain, TLSIndex, MachinePointerInfo()); 3154 3155 SDValue Slot = DAG.getNode(ISD::SHL, DL, PtrVT, TLSIndex, 3156 DAG.getConstant(2, DL, MVT::i32)); 3157 SDValue TLS = DAG.getLoad(PtrVT, DL, Chain, 3158 DAG.getNode(ISD::ADD, DL, PtrVT, TLSArray, Slot), 3159 MachinePointerInfo()); 3160 3161 // Get the offset of the start of the .tls section (section base) 3162 const auto *GA = cast<GlobalAddressSDNode>(Op); 3163 auto *CPV = ARMConstantPoolConstant::Create(GA->getGlobal(), ARMCP::SECREL); 3164 SDValue Offset = DAG.getLoad( 3165 PtrVT, DL, Chain, DAG.getNode(ARMISD::Wrapper, DL, MVT::i32, 3166 DAG.getTargetConstantPool(CPV, PtrVT, 4)), 3167 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3168 3169 return DAG.getNode(ISD::ADD, DL, PtrVT, TLS, Offset); 3170 } 3171 3172 // Lower ISD::GlobalTLSAddress using the "general dynamic" model 3173 SDValue 3174 ARMTargetLowering::LowerToTLSGeneralDynamicModel(GlobalAddressSDNode *GA, 3175 SelectionDAG &DAG) const { 3176 SDLoc dl(GA); 3177 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3178 unsigned char PCAdj = Subtarget->isThumb() ? 4 : 8; 3179 MachineFunction &MF = DAG.getMachineFunction(); 3180 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3181 unsigned ARMPCLabelIndex = AFI->createPICLabelUId(); 3182 ARMConstantPoolValue *CPV = 3183 ARMConstantPoolConstant::Create(GA->getGlobal(), ARMPCLabelIndex, 3184 ARMCP::CPValue, PCAdj, ARMCP::TLSGD, true); 3185 SDValue Argument = DAG.getTargetConstantPool(CPV, PtrVT, 4); 3186 Argument = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Argument); 3187 Argument = DAG.getLoad( 3188 PtrVT, dl, DAG.getEntryNode(), Argument, 3189 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3190 SDValue Chain = Argument.getValue(1); 3191 3192 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32); 3193 Argument = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVT, Argument, PICLabel); 3194 3195 // call __tls_get_addr. 3196 ArgListTy Args; 3197 ArgListEntry Entry; 3198 Entry.Node = Argument; 3199 Entry.Ty = (Type *) Type::getInt32Ty(*DAG.getContext()); 3200 Args.push_back(Entry); 3201 3202 // FIXME: is there useful debug info available here? 3203 TargetLowering::CallLoweringInfo CLI(DAG); 3204 CLI.setDebugLoc(dl).setChain(Chain).setLibCallee( 3205 CallingConv::C, Type::getInt32Ty(*DAG.getContext()), 3206 DAG.getExternalSymbol("__tls_get_addr", PtrVT), std::move(Args)); 3207 3208 std::pair<SDValue, SDValue> CallResult = LowerCallTo(CLI); 3209 return CallResult.first; 3210 } 3211 3212 // Lower ISD::GlobalTLSAddress using the "initial exec" or 3213 // "local exec" model. 3214 SDValue 3215 ARMTargetLowering::LowerToTLSExecModels(GlobalAddressSDNode *GA, 3216 SelectionDAG &DAG, 3217 TLSModel::Model model) const { 3218 const GlobalValue *GV = GA->getGlobal(); 3219 SDLoc dl(GA); 3220 SDValue Offset; 3221 SDValue Chain = DAG.getEntryNode(); 3222 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3223 // Get the Thread Pointer 3224 SDValue ThreadPointer = DAG.getNode(ARMISD::THREAD_POINTER, dl, PtrVT); 3225 3226 if (model == TLSModel::InitialExec) { 3227 MachineFunction &MF = DAG.getMachineFunction(); 3228 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3229 unsigned ARMPCLabelIndex = AFI->createPICLabelUId(); 3230 // Initial exec model. 3231 unsigned char PCAdj = Subtarget->isThumb() ? 4 : 8; 3232 ARMConstantPoolValue *CPV = 3233 ARMConstantPoolConstant::Create(GA->getGlobal(), ARMPCLabelIndex, 3234 ARMCP::CPValue, PCAdj, ARMCP::GOTTPOFF, 3235 true); 3236 Offset = DAG.getTargetConstantPool(CPV, PtrVT, 4); 3237 Offset = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Offset); 3238 Offset = DAG.getLoad( 3239 PtrVT, dl, Chain, Offset, 3240 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3241 Chain = Offset.getValue(1); 3242 3243 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32); 3244 Offset = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVT, Offset, PICLabel); 3245 3246 Offset = DAG.getLoad( 3247 PtrVT, dl, Chain, Offset, 3248 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3249 } else { 3250 // local exec model 3251 assert(model == TLSModel::LocalExec); 3252 ARMConstantPoolValue *CPV = 3253 ARMConstantPoolConstant::Create(GV, ARMCP::TPOFF); 3254 Offset = DAG.getTargetConstantPool(CPV, PtrVT, 4); 3255 Offset = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Offset); 3256 Offset = DAG.getLoad( 3257 PtrVT, dl, Chain, Offset, 3258 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3259 } 3260 3261 // The address of the thread local variable is the add of the thread 3262 // pointer with the offset of the variable. 3263 return DAG.getNode(ISD::ADD, dl, PtrVT, ThreadPointer, Offset); 3264 } 3265 3266 SDValue 3267 ARMTargetLowering::LowerGlobalTLSAddress(SDValue Op, SelectionDAG &DAG) const { 3268 GlobalAddressSDNode *GA = cast<GlobalAddressSDNode>(Op); 3269 if (DAG.getTarget().useEmulatedTLS()) 3270 return LowerToTLSEmulatedModel(GA, DAG); 3271 3272 if (Subtarget->isTargetDarwin()) 3273 return LowerGlobalTLSAddressDarwin(Op, DAG); 3274 3275 if (Subtarget->isTargetWindows()) 3276 return LowerGlobalTLSAddressWindows(Op, DAG); 3277 3278 // TODO: implement the "local dynamic" model 3279 assert(Subtarget->isTargetELF() && "Only ELF implemented here"); 3280 TLSModel::Model model = getTargetMachine().getTLSModel(GA->getGlobal()); 3281 3282 switch (model) { 3283 case TLSModel::GeneralDynamic: 3284 case TLSModel::LocalDynamic: 3285 return LowerToTLSGeneralDynamicModel(GA, DAG); 3286 case TLSModel::InitialExec: 3287 case TLSModel::LocalExec: 3288 return LowerToTLSExecModels(GA, DAG, model); 3289 } 3290 llvm_unreachable("bogus TLS model"); 3291 } 3292 3293 /// Return true if all users of V are within function F, looking through 3294 /// ConstantExprs. 3295 static bool allUsersAreInFunction(const Value *V, const Function *F) { 3296 SmallVector<const User*,4> Worklist; 3297 for (auto *U : V->users()) 3298 Worklist.push_back(U); 3299 while (!Worklist.empty()) { 3300 auto *U = Worklist.pop_back_val(); 3301 if (isa<ConstantExpr>(U)) { 3302 for (auto *UU : U->users()) 3303 Worklist.push_back(UU); 3304 continue; 3305 } 3306 3307 auto *I = dyn_cast<Instruction>(U); 3308 if (!I || I->getParent()->getParent() != F) 3309 return false; 3310 } 3311 return true; 3312 } 3313 3314 static SDValue promoteToConstantPool(const ARMTargetLowering *TLI, 3315 const GlobalValue *GV, SelectionDAG &DAG, 3316 EVT PtrVT, const SDLoc &dl) { 3317 // If we're creating a pool entry for a constant global with unnamed address, 3318 // and the global is small enough, we can emit it inline into the constant pool 3319 // to save ourselves an indirection. 3320 // 3321 // This is a win if the constant is only used in one function (so it doesn't 3322 // need to be duplicated) or duplicating the constant wouldn't increase code 3323 // size (implying the constant is no larger than 4 bytes). 3324 const Function &F = DAG.getMachineFunction().getFunction(); 3325 3326 // We rely on this decision to inline being idemopotent and unrelated to the 3327 // use-site. We know that if we inline a variable at one use site, we'll 3328 // inline it elsewhere too (and reuse the constant pool entry). Fast-isel 3329 // doesn't know about this optimization, so bail out if it's enabled else 3330 // we could decide to inline here (and thus never emit the GV) but require 3331 // the GV from fast-isel generated code. 3332 if (!EnableConstpoolPromotion || 3333 DAG.getMachineFunction().getTarget().Options.EnableFastISel) 3334 return SDValue(); 3335 3336 auto *GVar = dyn_cast<GlobalVariable>(GV); 3337 if (!GVar || !GVar->hasInitializer() || 3338 !GVar->isConstant() || !GVar->hasGlobalUnnamedAddr() || 3339 !GVar->hasLocalLinkage()) 3340 return SDValue(); 3341 3342 // If we inline a value that contains relocations, we move the relocations 3343 // from .data to .text. This is not allowed in position-independent code. 3344 auto *Init = GVar->getInitializer(); 3345 if ((TLI->isPositionIndependent() || TLI->getSubtarget()->isROPI()) && 3346 Init->needsRelocation()) 3347 return SDValue(); 3348 3349 // The constant islands pass can only really deal with alignment requests 3350 // <= 4 bytes and cannot pad constants itself. Therefore we cannot promote 3351 // any type wanting greater alignment requirements than 4 bytes. We also 3352 // can only promote constants that are multiples of 4 bytes in size or 3353 // are paddable to a multiple of 4. Currently we only try and pad constants 3354 // that are strings for simplicity. 3355 auto *CDAInit = dyn_cast<ConstantDataArray>(Init); 3356 unsigned Size = DAG.getDataLayout().getTypeAllocSize(Init->getType()); 3357 unsigned Align = DAG.getDataLayout().getPreferredAlignment(GVar); 3358 unsigned RequiredPadding = 4 - (Size % 4); 3359 bool PaddingPossible = 3360 RequiredPadding == 4 || (CDAInit && CDAInit->isString()); 3361 if (!PaddingPossible || Align > 4 || Size > ConstpoolPromotionMaxSize || 3362 Size == 0) 3363 return SDValue(); 3364 3365 unsigned PaddedSize = Size + ((RequiredPadding == 4) ? 0 : RequiredPadding); 3366 MachineFunction &MF = DAG.getMachineFunction(); 3367 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3368 3369 // We can't bloat the constant pool too much, else the ConstantIslands pass 3370 // may fail to converge. If we haven't promoted this global yet (it may have 3371 // multiple uses), and promoting it would increase the constant pool size (Sz 3372 // > 4), ensure we have space to do so up to MaxTotal. 3373 if (!AFI->getGlobalsPromotedToConstantPool().count(GVar) && Size > 4) 3374 if (AFI->getPromotedConstpoolIncrease() + PaddedSize - 4 >= 3375 ConstpoolPromotionMaxTotal) 3376 return SDValue(); 3377 3378 // This is only valid if all users are in a single function; we can't clone 3379 // the constant in general. The LLVM IR unnamed_addr allows merging 3380 // constants, but not cloning them. 3381 // 3382 // We could potentially allow cloning if we could prove all uses of the 3383 // constant in the current function don't care about the address, like 3384 // printf format strings. But that isn't implemented for now. 3385 if (!allUsersAreInFunction(GVar, &F)) 3386 return SDValue(); 3387 3388 // We're going to inline this global. Pad it out if needed. 3389 if (RequiredPadding != 4) { 3390 StringRef S = CDAInit->getAsString(); 3391 3392 SmallVector<uint8_t,16> V(S.size()); 3393 std::copy(S.bytes_begin(), S.bytes_end(), V.begin()); 3394 while (RequiredPadding--) 3395 V.push_back(0); 3396 Init = ConstantDataArray::get(*DAG.getContext(), V); 3397 } 3398 3399 auto CPVal = ARMConstantPoolConstant::Create(GVar, Init); 3400 SDValue CPAddr = 3401 DAG.getTargetConstantPool(CPVal, PtrVT, /*Align=*/4); 3402 if (!AFI->getGlobalsPromotedToConstantPool().count(GVar)) { 3403 AFI->markGlobalAsPromotedToConstantPool(GVar); 3404 AFI->setPromotedConstpoolIncrease(AFI->getPromotedConstpoolIncrease() + 3405 PaddedSize - 4); 3406 } 3407 ++NumConstpoolPromoted; 3408 return DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr); 3409 } 3410 3411 bool ARMTargetLowering::isReadOnly(const GlobalValue *GV) const { 3412 if (const GlobalAlias *GA = dyn_cast<GlobalAlias>(GV)) 3413 if (!(GV = GA->getBaseObject())) 3414 return false; 3415 if (const auto *V = dyn_cast<GlobalVariable>(GV)) 3416 return V->isConstant(); 3417 return isa<Function>(GV); 3418 } 3419 3420 SDValue ARMTargetLowering::LowerGlobalAddress(SDValue Op, 3421 SelectionDAG &DAG) const { 3422 switch (Subtarget->getTargetTriple().getObjectFormat()) { 3423 default: llvm_unreachable("unknown object format"); 3424 case Triple::COFF: 3425 return LowerGlobalAddressWindows(Op, DAG); 3426 case Triple::ELF: 3427 return LowerGlobalAddressELF(Op, DAG); 3428 case Triple::MachO: 3429 return LowerGlobalAddressDarwin(Op, DAG); 3430 } 3431 } 3432 3433 SDValue ARMTargetLowering::LowerGlobalAddressELF(SDValue Op, 3434 SelectionDAG &DAG) const { 3435 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3436 SDLoc dl(Op); 3437 const GlobalValue *GV = cast<GlobalAddressSDNode>(Op)->getGlobal(); 3438 const TargetMachine &TM = getTargetMachine(); 3439 bool IsRO = isReadOnly(GV); 3440 3441 // promoteToConstantPool only if not generating XO text section 3442 if (TM.shouldAssumeDSOLocal(*GV->getParent(), GV) && !Subtarget->genExecuteOnly()) 3443 if (SDValue V = promoteToConstantPool(this, GV, DAG, PtrVT, dl)) 3444 return V; 3445 3446 if (isPositionIndependent()) { 3447 bool UseGOT_PREL = !TM.shouldAssumeDSOLocal(*GV->getParent(), GV); 3448 SDValue G = DAG.getTargetGlobalAddress(GV, dl, PtrVT, 0, 3449 UseGOT_PREL ? ARMII::MO_GOT : 0); 3450 SDValue Result = DAG.getNode(ARMISD::WrapperPIC, dl, PtrVT, G); 3451 if (UseGOT_PREL) 3452 Result = 3453 DAG.getLoad(PtrVT, dl, DAG.getEntryNode(), Result, 3454 MachinePointerInfo::getGOT(DAG.getMachineFunction())); 3455 return Result; 3456 } else if (Subtarget->isROPI() && IsRO) { 3457 // PC-relative. 3458 SDValue G = DAG.getTargetGlobalAddress(GV, dl, PtrVT); 3459 SDValue Result = DAG.getNode(ARMISD::WrapperPIC, dl, PtrVT, G); 3460 return Result; 3461 } else if (Subtarget->isRWPI() && !IsRO) { 3462 // SB-relative. 3463 SDValue RelAddr; 3464 if (Subtarget->useMovt()) { 3465 ++NumMovwMovt; 3466 SDValue G = DAG.getTargetGlobalAddress(GV, dl, PtrVT, 0, ARMII::MO_SBREL); 3467 RelAddr = DAG.getNode(ARMISD::Wrapper, dl, PtrVT, G); 3468 } else { // use literal pool for address constant 3469 ARMConstantPoolValue *CPV = 3470 ARMConstantPoolConstant::Create(GV, ARMCP::SBREL); 3471 SDValue CPAddr = DAG.getTargetConstantPool(CPV, PtrVT, 4); 3472 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr); 3473 RelAddr = DAG.getLoad( 3474 PtrVT, dl, DAG.getEntryNode(), CPAddr, 3475 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3476 } 3477 SDValue SB = DAG.getCopyFromReg(DAG.getEntryNode(), dl, ARM::R9, PtrVT); 3478 SDValue Result = DAG.getNode(ISD::ADD, dl, PtrVT, SB, RelAddr); 3479 return Result; 3480 } 3481 3482 // If we have T2 ops, we can materialize the address directly via movt/movw 3483 // pair. This is always cheaper. 3484 if (Subtarget->useMovt()) { 3485 ++NumMovwMovt; 3486 // FIXME: Once remat is capable of dealing with instructions with register 3487 // operands, expand this into two nodes. 3488 return DAG.getNode(ARMISD::Wrapper, dl, PtrVT, 3489 DAG.getTargetGlobalAddress(GV, dl, PtrVT)); 3490 } else { 3491 SDValue CPAddr = DAG.getTargetConstantPool(GV, PtrVT, 4); 3492 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr); 3493 return DAG.getLoad( 3494 PtrVT, dl, DAG.getEntryNode(), CPAddr, 3495 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3496 } 3497 } 3498 3499 SDValue ARMTargetLowering::LowerGlobalAddressDarwin(SDValue Op, 3500 SelectionDAG &DAG) const { 3501 assert(!Subtarget->isROPI() && !Subtarget->isRWPI() && 3502 "ROPI/RWPI not currently supported for Darwin"); 3503 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3504 SDLoc dl(Op); 3505 const GlobalValue *GV = cast<GlobalAddressSDNode>(Op)->getGlobal(); 3506 3507 if (Subtarget->useMovt()) 3508 ++NumMovwMovt; 3509 3510 // FIXME: Once remat is capable of dealing with instructions with register 3511 // operands, expand this into multiple nodes 3512 unsigned Wrapper = 3513 isPositionIndependent() ? ARMISD::WrapperPIC : ARMISD::Wrapper; 3514 3515 SDValue G = DAG.getTargetGlobalAddress(GV, dl, PtrVT, 0, ARMII::MO_NONLAZY); 3516 SDValue Result = DAG.getNode(Wrapper, dl, PtrVT, G); 3517 3518 if (Subtarget->isGVIndirectSymbol(GV)) 3519 Result = DAG.getLoad(PtrVT, dl, DAG.getEntryNode(), Result, 3520 MachinePointerInfo::getGOT(DAG.getMachineFunction())); 3521 return Result; 3522 } 3523 3524 SDValue ARMTargetLowering::LowerGlobalAddressWindows(SDValue Op, 3525 SelectionDAG &DAG) const { 3526 assert(Subtarget->isTargetWindows() && "non-Windows COFF is not supported"); 3527 assert(Subtarget->useMovt() && 3528 "Windows on ARM expects to use movw/movt"); 3529 assert(!Subtarget->isROPI() && !Subtarget->isRWPI() && 3530 "ROPI/RWPI not currently supported for Windows"); 3531 3532 const TargetMachine &TM = getTargetMachine(); 3533 const GlobalValue *GV = cast<GlobalAddressSDNode>(Op)->getGlobal(); 3534 ARMII::TOF TargetFlags = ARMII::MO_NO_FLAG; 3535 if (GV->hasDLLImportStorageClass()) 3536 TargetFlags = ARMII::MO_DLLIMPORT; 3537 else if (!TM.shouldAssumeDSOLocal(*GV->getParent(), GV)) 3538 TargetFlags = ARMII::MO_COFFSTUB; 3539 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3540 SDValue Result; 3541 SDLoc DL(Op); 3542 3543 ++NumMovwMovt; 3544 3545 // FIXME: Once remat is capable of dealing with instructions with register 3546 // operands, expand this into two nodes. 3547 Result = DAG.getNode(ARMISD::Wrapper, DL, PtrVT, 3548 DAG.getTargetGlobalAddress(GV, DL, PtrVT, /*offset=*/0, 3549 TargetFlags)); 3550 if (TargetFlags & (ARMII::MO_DLLIMPORT | ARMII::MO_COFFSTUB)) 3551 Result = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Result, 3552 MachinePointerInfo::getGOT(DAG.getMachineFunction())); 3553 return Result; 3554 } 3555 3556 SDValue 3557 ARMTargetLowering::LowerEH_SJLJ_SETJMP(SDValue Op, SelectionDAG &DAG) const { 3558 SDLoc dl(Op); 3559 SDValue Val = DAG.getConstant(0, dl, MVT::i32); 3560 return DAG.getNode(ARMISD::EH_SJLJ_SETJMP, dl, 3561 DAG.getVTList(MVT::i32, MVT::Other), Op.getOperand(0), 3562 Op.getOperand(1), Val); 3563 } 3564 3565 SDValue 3566 ARMTargetLowering::LowerEH_SJLJ_LONGJMP(SDValue Op, SelectionDAG &DAG) const { 3567 SDLoc dl(Op); 3568 return DAG.getNode(ARMISD::EH_SJLJ_LONGJMP, dl, MVT::Other, Op.getOperand(0), 3569 Op.getOperand(1), DAG.getConstant(0, dl, MVT::i32)); 3570 } 3571 3572 SDValue ARMTargetLowering::LowerEH_SJLJ_SETUP_DISPATCH(SDValue Op, 3573 SelectionDAG &DAG) const { 3574 SDLoc dl(Op); 3575 return DAG.getNode(ARMISD::EH_SJLJ_SETUP_DISPATCH, dl, MVT::Other, 3576 Op.getOperand(0)); 3577 } 3578 3579 SDValue ARMTargetLowering::LowerINTRINSIC_VOID( 3580 SDValue Op, SelectionDAG &DAG, const ARMSubtarget *Subtarget) const { 3581 unsigned IntNo = 3582 cast<ConstantSDNode>( 3583 Op.getOperand(Op.getOperand(0).getValueType() == MVT::Other)) 3584 ->getZExtValue(); 3585 switch (IntNo) { 3586 default: 3587 return SDValue(); // Don't custom lower most intrinsics. 3588 case Intrinsic::arm_gnu_eabi_mcount: { 3589 MachineFunction &MF = DAG.getMachineFunction(); 3590 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3591 SDLoc dl(Op); 3592 SDValue Chain = Op.getOperand(0); 3593 // call "\01__gnu_mcount_nc" 3594 const ARMBaseRegisterInfo *ARI = Subtarget->getRegisterInfo(); 3595 const uint32_t *Mask = 3596 ARI->getCallPreservedMask(DAG.getMachineFunction(), CallingConv::C); 3597 assert(Mask && "Missing call preserved mask for calling convention"); 3598 // Mark LR an implicit live-in. 3599 unsigned Reg = MF.addLiveIn(ARM::LR, getRegClassFor(MVT::i32)); 3600 SDValue ReturnAddress = 3601 DAG.getCopyFromReg(DAG.getEntryNode(), dl, Reg, PtrVT); 3602 std::vector<EVT> ResultTys = {MVT::Other, MVT::Glue}; 3603 SDValue Callee = 3604 DAG.getTargetExternalSymbol("\01__gnu_mcount_nc", PtrVT, 0); 3605 SDValue RegisterMask = DAG.getRegisterMask(Mask); 3606 if (Subtarget->isThumb()) 3607 return SDValue( 3608 DAG.getMachineNode( 3609 ARM::tBL_PUSHLR, dl, ResultTys, 3610 {ReturnAddress, DAG.getTargetConstant(ARMCC::AL, dl, PtrVT), 3611 DAG.getRegister(0, PtrVT), Callee, RegisterMask, Chain}), 3612 0); 3613 return SDValue( 3614 DAG.getMachineNode(ARM::BL_PUSHLR, dl, ResultTys, 3615 {ReturnAddress, Callee, RegisterMask, Chain}), 3616 0); 3617 } 3618 } 3619 } 3620 3621 SDValue 3622 ARMTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op, SelectionDAG &DAG, 3623 const ARMSubtarget *Subtarget) const { 3624 unsigned IntNo = cast<ConstantSDNode>(Op.getOperand(0))->getZExtValue(); 3625 SDLoc dl(Op); 3626 switch (IntNo) { 3627 default: return SDValue(); // Don't custom lower most intrinsics. 3628 case Intrinsic::thread_pointer: { 3629 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3630 return DAG.getNode(ARMISD::THREAD_POINTER, dl, PtrVT); 3631 } 3632 case Intrinsic::eh_sjlj_lsda: { 3633 MachineFunction &MF = DAG.getMachineFunction(); 3634 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3635 unsigned ARMPCLabelIndex = AFI->createPICLabelUId(); 3636 EVT PtrVT = getPointerTy(DAG.getDataLayout()); 3637 SDValue CPAddr; 3638 bool IsPositionIndependent = isPositionIndependent(); 3639 unsigned PCAdj = IsPositionIndependent ? (Subtarget->isThumb() ? 4 : 8) : 0; 3640 ARMConstantPoolValue *CPV = 3641 ARMConstantPoolConstant::Create(&MF.getFunction(), ARMPCLabelIndex, 3642 ARMCP::CPLSDA, PCAdj); 3643 CPAddr = DAG.getTargetConstantPool(CPV, PtrVT, 4); 3644 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr); 3645 SDValue Result = DAG.getLoad( 3646 PtrVT, dl, DAG.getEntryNode(), CPAddr, 3647 MachinePointerInfo::getConstantPool(DAG.getMachineFunction())); 3648 3649 if (IsPositionIndependent) { 3650 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32); 3651 Result = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVT, Result, PICLabel); 3652 } 3653 return Result; 3654 } 3655 case Intrinsic::arm_neon_vabs: 3656 return DAG.getNode(ISD::ABS, SDLoc(Op), Op.getValueType(), 3657 Op.getOperand(1)); 3658 case Intrinsic::arm_neon_vmulls: 3659 case Intrinsic::arm_neon_vmullu: { 3660 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vmulls) 3661 ? ARMISD::VMULLs : ARMISD::VMULLu; 3662 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(), 3663 Op.getOperand(1), Op.getOperand(2)); 3664 } 3665 case Intrinsic::arm_neon_vminnm: 3666 case Intrinsic::arm_neon_vmaxnm: { 3667 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vminnm) 3668 ? ISD::FMINNUM : ISD::FMAXNUM; 3669 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(), 3670 Op.getOperand(1), Op.getOperand(2)); 3671 } 3672 case Intrinsic::arm_neon_vminu: 3673 case Intrinsic::arm_neon_vmaxu: { 3674 if (Op.getValueType().isFloatingPoint()) 3675 return SDValue(); 3676 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vminu) 3677 ? ISD::UMIN : ISD::UMAX; 3678 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(), 3679 Op.getOperand(1), Op.getOperand(2)); 3680 } 3681 case Intrinsic::arm_neon_vmins: 3682 case Intrinsic::arm_neon_vmaxs: { 3683 // v{min,max}s is overloaded between signed integers and floats. 3684 if (!Op.getValueType().isFloatingPoint()) { 3685 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vmins) 3686 ? ISD::SMIN : ISD::SMAX; 3687 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(), 3688 Op.getOperand(1), Op.getOperand(2)); 3689 } 3690 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vmins) 3691 ? ISD::FMINIMUM : ISD::FMAXIMUM; 3692 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(), 3693 Op.getOperand(1), Op.getOperand(2)); 3694 } 3695 case Intrinsic::arm_neon_vtbl1: 3696 return DAG.getNode(ARMISD::VTBL1, SDLoc(Op), Op.getValueType(), 3697 Op.getOperand(1), Op.getOperand(2)); 3698 case Intrinsic::arm_neon_vtbl2: 3699 return DAG.getNode(ARMISD::VTBL2, SDLoc(Op), Op.getValueType(), 3700 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3)); 3701 case Intrinsic::arm_mve_pred_i2v: 3702 case Intrinsic::arm_mve_pred_v2i: 3703 return DAG.getNode(ARMISD::PREDICATE_CAST, SDLoc(Op), Op.getValueType(), 3704 Op.getOperand(1)); 3705 } 3706 } 3707 3708 static SDValue LowerATOMIC_FENCE(SDValue Op, SelectionDAG &DAG, 3709 const ARMSubtarget *Subtarget) { 3710 SDLoc dl(Op); 3711 ConstantSDNode *SSIDNode = cast<ConstantSDNode>(Op.getOperand(2)); 3712 auto SSID = static_cast<SyncScope::ID>(SSIDNode->getZExtValue()); 3713 if (SSID == SyncScope::SingleThread) 3714 return Op; 3715 3716 if (!Subtarget->hasDataBarrier()) { 3717 // Some ARMv6 cpus can support data barriers with an mcr instruction. 3718 // Thumb1 and pre-v6 ARM mode use a libcall instead and should never get 3719 // here. 3720 assert(Subtarget->hasV6Ops() && !Subtarget->isThumb() && 3721 "Unexpected ISD::ATOMIC_FENCE encountered. Should be libcall!"); 3722 return DAG.getNode(ARMISD::MEMBARRIER_MCR, dl, MVT::Other, Op.getOperand(0), 3723 DAG.getConstant(0, dl, MVT::i32)); 3724 } 3725 3726 ConstantSDNode *OrdN = cast<ConstantSDNode>(Op.getOperand(1)); 3727 AtomicOrdering Ord = static_cast<AtomicOrdering>(OrdN->getZExtValue()); 3728 ARM_MB::MemBOpt Domain = ARM_MB::ISH; 3729 if (Subtarget->isMClass()) { 3730 // Only a full system barrier exists in the M-class architectures. 3731 Domain = ARM_MB::SY; 3732 } else if (Subtarget->preferISHSTBarriers() && 3733 Ord == AtomicOrdering::Release) { 3734 // Swift happens to implement ISHST barriers in a way that's compatible with 3735 // Release semantics but weaker than ISH so we'd be fools not to use 3736 // it. Beware: other processors probably don't! 3737 Domain = ARM_MB::ISHST; 3738 } 3739 3740 return DAG.getNode(ISD::INTRINSIC_VOID, dl, MVT::Other, Op.getOperand(0), 3741 DAG.getConstant(Intrinsic::arm_dmb, dl, MVT::i32), 3742 DAG.getConstant(Domain, dl, MVT::i32)); 3743 } 3744 3745 static SDValue LowerPREFETCH(SDValue Op, SelectionDAG &DAG, 3746 const ARMSubtarget *Subtarget) { 3747 // ARM pre v5TE and Thumb1 does not have preload instructions. 3748 if (!(Subtarget->isThumb2() || 3749 (!Subtarget->isThumb1Only() && Subtarget->hasV5TEOps()))) 3750 // Just preserve the chain. 3751 return Op.getOperand(0); 3752 3753 SDLoc dl(Op); 3754 unsigned isRead = ~cast<ConstantSDNode>(Op.getOperand(2))->getZExtValue() & 1; 3755 if (!isRead && 3756 (!Subtarget->hasV7Ops() || !Subtarget->hasMPExtension())) 3757 // ARMv7 with MP extension has PLDW. 3758 return Op.getOperand(0); 3759 3760 unsigned isData = cast<ConstantSDNode>(Op.getOperand(4))->getZExtValue(); 3761 if (Subtarget->isThumb()) { 3762 // Invert the bits. 3763 isRead = ~isRead & 1; 3764 isData = ~isData & 1; 3765 } 3766 3767 return DAG.getNode(ARMISD::PRELOAD, dl, MVT::Other, Op.getOperand(0), 3768 Op.getOperand(1), DAG.getConstant(isRead, dl, MVT::i32), 3769 DAG.getConstant(isData, dl, MVT::i32)); 3770 } 3771 3772 static SDValue LowerVASTART(SDValue Op, SelectionDAG &DAG) { 3773 MachineFunction &MF = DAG.getMachineFunction(); 3774 ARMFunctionInfo *FuncInfo = MF.getInfo<ARMFunctionInfo>(); 3775 3776 // vastart just stores the address of the VarArgsFrameIndex slot into the 3777 // memory location argument. 3778 SDLoc dl(Op); 3779 EVT PtrVT = DAG.getTargetLoweringInfo().getPointerTy(DAG.getDataLayout()); 3780 SDValue FR = DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(), PtrVT); 3781 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue(); 3782 return DAG.getStore(Op.getOperand(0), dl, FR, Op.getOperand(1), 3783 MachinePointerInfo(SV)); 3784 } 3785 3786 SDValue ARMTargetLowering::GetF64FormalArgument(CCValAssign &VA, 3787 CCValAssign &NextVA, 3788 SDValue &Root, 3789 SelectionDAG &DAG, 3790 const SDLoc &dl) const { 3791 MachineFunction &MF = DAG.getMachineFunction(); 3792 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3793 3794 const TargetRegisterClass *RC; 3795 if (AFI->isThumb1OnlyFunction()) 3796 RC = &ARM::tGPRRegClass; 3797 else 3798 RC = &ARM::GPRRegClass; 3799 3800 // Transform the arguments stored in physical registers into virtual ones. 3801 unsigned Reg = MF.addLiveIn(VA.getLocReg(), RC); 3802 SDValue ArgValue = DAG.getCopyFromReg(Root, dl, Reg, MVT::i32); 3803 3804 SDValue ArgValue2; 3805 if (NextVA.isMemLoc()) { 3806 MachineFrameInfo &MFI = MF.getFrameInfo(); 3807 int FI = MFI.CreateFixedObject(4, NextVA.getLocMemOffset(), true); 3808 3809 // Create load node to retrieve arguments from the stack. 3810 SDValue FIN = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout())); 3811 ArgValue2 = DAG.getLoad( 3812 MVT::i32, dl, Root, FIN, 3813 MachinePointerInfo::getFixedStack(DAG.getMachineFunction(), FI)); 3814 } else { 3815 Reg = MF.addLiveIn(NextVA.getLocReg(), RC); 3816 ArgValue2 = DAG.getCopyFromReg(Root, dl, Reg, MVT::i32); 3817 } 3818 if (!Subtarget->isLittle()) 3819 std::swap (ArgValue, ArgValue2); 3820 return DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, ArgValue, ArgValue2); 3821 } 3822 3823 // The remaining GPRs hold either the beginning of variable-argument 3824 // data, or the beginning of an aggregate passed by value (usually 3825 // byval). Either way, we allocate stack slots adjacent to the data 3826 // provided by our caller, and store the unallocated registers there. 3827 // If this is a variadic function, the va_list pointer will begin with 3828 // these values; otherwise, this reassembles a (byval) structure that 3829 // was split between registers and memory. 3830 // Return: The frame index registers were stored into. 3831 int ARMTargetLowering::StoreByValRegs(CCState &CCInfo, SelectionDAG &DAG, 3832 const SDLoc &dl, SDValue &Chain, 3833 const Value *OrigArg, 3834 unsigned InRegsParamRecordIdx, 3835 int ArgOffset, unsigned ArgSize) const { 3836 // Currently, two use-cases possible: 3837 // Case #1. Non-var-args function, and we meet first byval parameter. 3838 // Setup first unallocated register as first byval register; 3839 // eat all remained registers 3840 // (these two actions are performed by HandleByVal method). 3841 // Then, here, we initialize stack frame with 3842 // "store-reg" instructions. 3843 // Case #2. Var-args function, that doesn't contain byval parameters. 3844 // The same: eat all remained unallocated registers, 3845 // initialize stack frame. 3846 3847 MachineFunction &MF = DAG.getMachineFunction(); 3848 MachineFrameInfo &MFI = MF.getFrameInfo(); 3849 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3850 unsigned RBegin, REnd; 3851 if (InRegsParamRecordIdx < CCInfo.getInRegsParamsCount()) { 3852 CCInfo.getInRegsParamInfo(InRegsParamRecordIdx, RBegin, REnd); 3853 } else { 3854 unsigned RBeginIdx = CCInfo.getFirstUnallocated(GPRArgRegs); 3855 RBegin = RBeginIdx == 4 ? (unsigned)ARM::R4 : GPRArgRegs[RBeginIdx]; 3856 REnd = ARM::R4; 3857 } 3858 3859 if (REnd != RBegin) 3860 ArgOffset = -4 * (ARM::R4 - RBegin); 3861 3862 auto PtrVT = getPointerTy(DAG.getDataLayout()); 3863 int FrameIndex = MFI.CreateFixedObject(ArgSize, ArgOffset, false); 3864 SDValue FIN = DAG.getFrameIndex(FrameIndex, PtrVT); 3865 3866 SmallVector<SDValue, 4> MemOps; 3867 const TargetRegisterClass *RC = 3868 AFI->isThumb1OnlyFunction() ? &ARM::tGPRRegClass : &ARM::GPRRegClass; 3869 3870 for (unsigned Reg = RBegin, i = 0; Reg < REnd; ++Reg, ++i) { 3871 unsigned VReg = MF.addLiveIn(Reg, RC); 3872 SDValue Val = DAG.getCopyFromReg(Chain, dl, VReg, MVT::i32); 3873 SDValue Store = DAG.getStore(Val.getValue(1), dl, Val, FIN, 3874 MachinePointerInfo(OrigArg, 4 * i)); 3875 MemOps.push_back(Store); 3876 FIN = DAG.getNode(ISD::ADD, dl, PtrVT, FIN, DAG.getConstant(4, dl, PtrVT)); 3877 } 3878 3879 if (!MemOps.empty()) 3880 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, MemOps); 3881 return FrameIndex; 3882 } 3883 3884 // Setup stack frame, the va_list pointer will start from. 3885 void ARMTargetLowering::VarArgStyleRegisters(CCState &CCInfo, SelectionDAG &DAG, 3886 const SDLoc &dl, SDValue &Chain, 3887 unsigned ArgOffset, 3888 unsigned TotalArgRegsSaveSize, 3889 bool ForceMutable) const { 3890 MachineFunction &MF = DAG.getMachineFunction(); 3891 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3892 3893 // Try to store any remaining integer argument regs 3894 // to their spots on the stack so that they may be loaded by dereferencing 3895 // the result of va_next. 3896 // If there is no regs to be stored, just point address after last 3897 // argument passed via stack. 3898 int FrameIndex = StoreByValRegs(CCInfo, DAG, dl, Chain, nullptr, 3899 CCInfo.getInRegsParamsCount(), 3900 CCInfo.getNextStackOffset(), 3901 std::max(4U, TotalArgRegsSaveSize)); 3902 AFI->setVarArgsFrameIndex(FrameIndex); 3903 } 3904 3905 SDValue ARMTargetLowering::LowerFormalArguments( 3906 SDValue Chain, CallingConv::ID CallConv, bool isVarArg, 3907 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &dl, 3908 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const { 3909 MachineFunction &MF = DAG.getMachineFunction(); 3910 MachineFrameInfo &MFI = MF.getFrameInfo(); 3911 3912 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>(); 3913 3914 // Assign locations to all of the incoming arguments. 3915 SmallVector<CCValAssign, 16> ArgLocs; 3916 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), ArgLocs, 3917 *DAG.getContext()); 3918 CCInfo.AnalyzeFormalArguments(Ins, CCAssignFnForCall(CallConv, isVarArg)); 3919 3920 SmallVector<SDValue, 16> ArgValues; 3921 SDValue ArgValue; 3922 Function::const_arg_iterator CurOrigArg = MF.getFunction().arg_begin(); 3923 unsigned CurArgIdx = 0; 3924 3925 // Initially ArgRegsSaveSize is zero. 3926 // Then we increase this value each time we meet byval parameter. 3927 // We also increase this value in case of varargs function. 3928 AFI->setArgRegsSaveSize(0); 3929 3930 // Calculate the amount of stack space that we need to allocate to store 3931 // byval and variadic arguments that are passed in registers. 3932 // We need to know this before we allocate the first byval or variadic 3933 // argument, as they will be allocated a stack slot below the CFA (Canonical 3934 // Frame Address, the stack pointer at entry to the function). 3935 unsigned ArgRegBegin = ARM::R4; 3936 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { 3937 if (CCInfo.getInRegsParamsProcessed() >= CCInfo.getInRegsParamsCount()) 3938 break; 3939 3940 CCValAssign &VA = ArgLocs[i]; 3941 unsigned Index = VA.getValNo(); 3942 ISD::ArgFlagsTy Flags = Ins[Index].Flags; 3943 if (!Flags.isByVal()) 3944 continue; 3945 3946 assert(VA.isMemLoc() && "unexpected byval pointer in reg"); 3947 unsigned RBegin, REnd; 3948 CCInfo.getInRegsParamInfo(CCInfo.getInRegsParamsProcessed(), RBegin, REnd); 3949 ArgRegBegin = std::min(ArgRegBegin, RBegin); 3950 3951 CCInfo.nextInRegsParam(); 3952 } 3953 CCInfo.rewindByValRegsInfo(); 3954 3955 int lastInsIndex = -1; 3956 if (isVarArg && MFI.hasVAStart()) { 3957 unsigned RegIdx = CCInfo.getFirstUnallocated(GPRArgRegs); 3958 if (RegIdx != array_lengthof(GPRArgRegs)) 3959 ArgRegBegin = std::min(ArgRegBegin, (unsigned)GPRArgRegs[RegIdx]); 3960 } 3961 3962 unsigned TotalArgRegsSaveSize = 4 * (ARM::R4 - ArgRegBegin); 3963 AFI->setArgRegsSaveSize(TotalArgRegsSaveSize); 3964 auto PtrVT = getPointerTy(DAG.getDataLayout()); 3965 3966 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { 3967 CCValAssign &VA = ArgLocs[i]; 3968 if (Ins[VA.getValNo()].isOrigArg()) { 3969 std::advance(CurOrigArg, 3970 Ins[VA.getValNo()].getOrigArgIndex() - CurArgIdx); 3971 CurArgIdx = Ins[VA.getValNo()].getOrigArgIndex(); 3972 } 3973 // Arguments stored in registers. 3974 if (VA.isRegLoc()) { 3975 EVT RegVT = VA.getLocVT(); 3976 3977 if (VA.needsCustom()) { 3978 // f64 and vector types are split up into multiple registers or 3979 // combinations of registers and stack slots. 3980 if (VA.getLocVT() == MVT::v2f64) { 3981 SDValue ArgValue1 = GetF64FormalArgument(VA, ArgLocs[++i], 3982 Chain, DAG, dl); 3983 VA = ArgLocs[++i]; // skip ahead to next loc 3984 SDValue ArgValue2; 3985 if (VA.isMemLoc()) { 3986 int FI = MFI.CreateFixedObject(8, VA.getLocMemOffset(), true); 3987 SDValue FIN = DAG.getFrameIndex(FI, PtrVT); 3988 ArgValue2 = DAG.getLoad(MVT::f64, dl, Chain, FIN, 3989 MachinePointerInfo::getFixedStack( 3990 DAG.getMachineFunction(), FI)); 3991 } else { 3992 ArgValue2 = GetF64FormalArgument(VA, ArgLocs[++i], 3993 Chain, DAG, dl); 3994 } 3995 ArgValue = DAG.getNode(ISD::UNDEF, dl, MVT::v2f64); 3996 ArgValue = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, 3997 ArgValue, ArgValue1, 3998 DAG.getIntPtrConstant(0, dl)); 3999 ArgValue = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, 4000 ArgValue, ArgValue2, 4001 DAG.getIntPtrConstant(1, dl)); 4002 } else 4003 ArgValue = GetF64FormalArgument(VA, ArgLocs[++i], Chain, DAG, dl); 4004 } else { 4005 const TargetRegisterClass *RC; 4006 4007 4008 if (RegVT == MVT::f16) 4009 RC = &ARM::HPRRegClass; 4010 else if (RegVT == MVT::f32) 4011 RC = &ARM::SPRRegClass; 4012 else if (RegVT == MVT::f64 || RegVT == MVT::v4f16) 4013 RC = &ARM::DPRRegClass; 4014 else if (RegVT == MVT::v2f64 || RegVT == MVT::v8f16) 4015 RC = &ARM::QPRRegClass; 4016 else if (RegVT == MVT::i32) 4017 RC = AFI->isThumb1OnlyFunction() ? &ARM::tGPRRegClass 4018 : &ARM::GPRRegClass; 4019 else 4020 llvm_unreachable("RegVT not supported by FORMAL_ARGUMENTS Lowering"); 4021 4022 // Transform the arguments in physical registers into virtual ones. 4023 unsigned Reg = MF.addLiveIn(VA.getLocReg(), RC); 4024 ArgValue = DAG.getCopyFromReg(Chain, dl, Reg, RegVT); 4025 4026 // If this value is passed in r0 and has the returned attribute (e.g. 4027 // C++ 'structors), record this fact for later use. 4028 if (VA.getLocReg() == ARM::R0 && Ins[VA.getValNo()].Flags.isReturned()) { 4029 AFI->setPreservesR0(); 4030 } 4031 } 4032 4033 // If this is an 8 or 16-bit value, it is really passed promoted 4034 // to 32 bits. Insert an assert[sz]ext to capture this, then 4035 // truncate to the right size. 4036 switch (VA.getLocInfo()) { 4037 default: llvm_unreachable("Unknown loc info!"); 4038 case CCValAssign::Full: break; 4039 case CCValAssign::BCvt: 4040 ArgValue = DAG.getNode(ISD::BITCAST, dl, VA.getValVT(), ArgValue); 4041 break; 4042 case CCValAssign::SExt: 4043 ArgValue = DAG.getNode(ISD::AssertSext, dl, RegVT, ArgValue, 4044 DAG.getValueType(VA.getValVT())); 4045 ArgValue = DAG.getNode(ISD::TRUNCATE, dl, VA.getValVT(), ArgValue); 4046 break; 4047 case CCValAssign::ZExt: 4048 ArgValue = DAG.getNode(ISD::AssertZext, dl, RegVT, ArgValue, 4049 DAG.getValueType(VA.getValVT())); 4050 ArgValue = DAG.getNode(ISD::TRUNCATE, dl, VA.getValVT(), ArgValue); 4051 break; 4052 } 4053 4054 InVals.push_back(ArgValue); 4055 } else { // VA.isRegLoc() 4056 // sanity check 4057 assert(VA.isMemLoc()); 4058 assert(VA.getValVT() != MVT::i64 && "i64 should already be lowered"); 4059 4060 int index = VA.getValNo(); 4061 4062 // Some Ins[] entries become multiple ArgLoc[] entries. 4063 // Process them only once. 4064 if (index != lastInsIndex) 4065 { 4066 ISD::ArgFlagsTy Flags = Ins[index].Flags; 4067 // FIXME: For now, all byval parameter objects are marked mutable. 4068 // This can be changed with more analysis. 4069 // In case of tail call optimization mark all arguments mutable. 4070 // Since they could be overwritten by lowering of arguments in case of 4071 // a tail call. 4072 if (Flags.isByVal()) { 4073 assert(Ins[index].isOrigArg() && 4074 "Byval arguments cannot be implicit"); 4075 unsigned CurByValIndex = CCInfo.getInRegsParamsProcessed(); 4076 4077 int FrameIndex = StoreByValRegs( 4078 CCInfo, DAG, dl, Chain, &*CurOrigArg, CurByValIndex, 4079 VA.getLocMemOffset(), Flags.getByValSize()); 4080 InVals.push_back(DAG.getFrameIndex(FrameIndex, PtrVT)); 4081 CCInfo.nextInRegsParam(); 4082 } else { 4083 unsigned FIOffset = VA.getLocMemOffset(); 4084 int FI = MFI.CreateFixedObject(VA.getLocVT().getSizeInBits()/8, 4085 FIOffset, true); 4086 4087 // Create load nodes to retrieve arguments from the stack. 4088 SDValue FIN = DAG.getFrameIndex(FI, PtrVT); 4089 InVals.push_back(DAG.getLoad(VA.getValVT(), dl, Chain, FIN, 4090 MachinePointerInfo::getFixedStack( 4091 DAG.getMachineFunction(), FI))); 4092 } 4093 lastInsIndex = index; 4094 } 4095 } 4096 } 4097 4098 // varargs 4099 if (isVarArg && MFI.hasVAStart()) 4100 VarArgStyleRegisters(CCInfo, DAG, dl, Chain, 4101 CCInfo.getNextStackOffset(), 4102 TotalArgRegsSaveSize); 4103 4104 AFI->setArgumentStackSize(CCInfo.getNextStackOffset()); 4105 4106 return Chain; 4107 } 4108 4109 /// isFloatingPointZero - Return true if this is +0.0. 4110 static bool isFloatingPointZero(SDValue Op) { 4111 if (ConstantFPSDNode *CFP = dyn_cast<ConstantFPSDNode>(Op)) 4112 return CFP->getValueAPF().isPosZero(); 4113 else if (ISD::isEXTLoad(Op.getNode()) || ISD::isNON_EXTLoad(Op.getNode())) { 4114 // Maybe this has already been legalized into the constant pool? 4115 if (Op.getOperand(1).getOpcode() == ARMISD::Wrapper) { 4116 SDValue WrapperOp = Op.getOperand(1).getOperand(0); 4117 if (ConstantPoolSDNode *CP = dyn_cast<ConstantPoolSDNode>(WrapperOp)) 4118 if (const ConstantFP *CFP = dyn_cast<ConstantFP>(CP->getConstVal())) 4119 return CFP->getValueAPF().isPosZero(); 4120 } 4121 } else if (Op->getOpcode() == ISD::BITCAST && 4122 Op->getValueType(0) == MVT::f64) { 4123 // Handle (ISD::BITCAST (ARMISD::VMOVIMM (ISD::TargetConstant 0)) MVT::f64) 4124 // created by LowerConstantFP(). 4125 SDValue BitcastOp = Op->getOperand(0); 4126 if (BitcastOp->getOpcode() == ARMISD::VMOVIMM && 4127 isNullConstant(BitcastOp->getOperand(0))) 4128 return true; 4129 } 4130 return false; 4131 } 4132 4133 /// Returns appropriate ARM CMP (cmp) and corresponding condition code for 4134 /// the given operands. 4135 SDValue ARMTargetLowering::getARMCmp(SDValue LHS, SDValue RHS, ISD::CondCode CC, 4136 SDValue &ARMcc, SelectionDAG &DAG, 4137 const SDLoc &dl) const { 4138 if (ConstantSDNode *RHSC = dyn_cast<ConstantSDNode>(RHS.getNode())) { 4139 unsigned C = RHSC->getZExtValue(); 4140 if (!isLegalICmpImmediate((int32_t)C)) { 4141 // Constant does not fit, try adjusting it by one. 4142 switch (CC) { 4143 default: break; 4144 case ISD::SETLT: 4145 case ISD::SETGE: 4146 if (C != 0x80000000 && isLegalICmpImmediate(C-1)) { 4147 CC = (CC == ISD::SETLT) ? ISD::SETLE : ISD::SETGT; 4148 RHS = DAG.getConstant(C - 1, dl, MVT::i32); 4149 } 4150 break; 4151 case ISD::SETULT: 4152 case ISD::SETUGE: 4153 if (C != 0 && isLegalICmpImmediate(C-1)) { 4154 CC = (CC == ISD::SETULT) ? ISD::SETULE : ISD::SETUGT; 4155 RHS = DAG.getConstant(C - 1, dl, MVT::i32); 4156 } 4157 break; 4158 case ISD::SETLE: 4159 case ISD::SETGT: 4160 if (C != 0x7fffffff && isLegalICmpImmediate(C+1)) { 4161 CC = (CC == ISD::SETLE) ? ISD::SETLT : ISD::SETGE; 4162 RHS = DAG.getConstant(C + 1, dl, MVT::i32); 4163 } 4164 break; 4165 case ISD::SETULE: 4166 case ISD::SETUGT: 4167 if (C != 0xffffffff && isLegalICmpImmediate(C+1)) { 4168 CC = (CC == ISD::SETULE) ? ISD::SETULT : ISD::SETUGE; 4169 RHS = DAG.getConstant(C + 1, dl, MVT::i32); 4170 } 4171 break; 4172 } 4173 } 4174 } else if ((ARM_AM::getShiftOpcForNode(LHS.getOpcode()) != ARM_AM::no_shift) && 4175 (ARM_AM::getShiftOpcForNode(RHS.getOpcode()) == ARM_AM::no_shift)) { 4176 // In ARM and Thumb-2, the compare instructions can shift their second 4177 // operand. 4178 CC = ISD::getSetCCSwappedOperands(CC); 4179 std::swap(LHS, RHS); 4180 } 4181 4182 // Thumb1 has very limited immediate modes, so turning an "and" into a 4183 // shift can save multiple instructions. 4184 // 4185 // If we have (x & C1), and C1 is an appropriate mask, we can transform it 4186 // into "((x << n) >> n)". But that isn't necessarily profitable on its 4187 // own. If it's the operand to an unsigned comparison with an immediate, 4188 // we can eliminate one of the shifts: we transform 4189 // "((x << n) >> n) == C2" to "(x << n) == (C2 << n)". 4190 // 4191 // We avoid transforming cases which aren't profitable due to encoding 4192 // details: 4193 // 4194 // 1. C2 fits into the immediate field of a cmp, and the transformed version 4195 // would not; in that case, we're essentially trading one immediate load for 4196 // another. 4197 // 2. C1 is 255 or 65535, so we can use uxtb or uxth. 4198 // 3. C2 is zero; we have other code for this special case. 4199 // 4200 // FIXME: Figure out profitability for Thumb2; we usually can't save an 4201 // instruction, since the AND is always one instruction anyway, but we could 4202 // use narrow instructions in some cases. 4203 if (Subtarget->isThumb1Only() && LHS->getOpcode() == ISD::AND && 4204 LHS->hasOneUse() && isa<ConstantSDNode>(LHS.getOperand(1)) && 4205 LHS.getValueType() == MVT::i32 && isa<ConstantSDNode>(RHS) && 4206 !isSignedIntSetCC(CC)) { 4207 unsigned Mask = cast<ConstantSDNode>(LHS.getOperand(1))->getZExtValue(); 4208 auto *RHSC = cast<ConstantSDNode>(RHS.getNode()); 4209 uint64_t RHSV = RHSC->getZExtValue(); 4210 if (isMask_32(Mask) && (RHSV & ~Mask) == 0 && Mask != 255 && Mask != 65535) { 4211 unsigned ShiftBits = countLeadingZeros(Mask); 4212 if (RHSV && (RHSV > 255 || (RHSV << ShiftBits) <= 255)) { 4213 SDValue ShiftAmt = DAG.getConstant(ShiftBits, dl, MVT::i32); 4214 LHS = DAG.getNode(ISD::SHL, dl, MVT::i32, LHS.getOperand(0), ShiftAmt); 4215 RHS = DAG.getConstant(RHSV << ShiftBits, dl, MVT::i32); 4216 } 4217 } 4218 } 4219 4220 // The specific comparison "(x<<c) > 0x80000000U" can be optimized to a 4221 // single "lsls x, c+1". The shift sets the "C" and "Z" flags the same 4222 // way a cmp would. 4223 // FIXME: Add support for ARM/Thumb2; this would need isel patterns, and 4224 // some tweaks to the heuristics for the previous and->shift transform. 4225 // FIXME: Optimize cases where the LHS isn't a shift. 4226 if (Subtarget->isThumb1Only() && LHS->getOpcode() == ISD::SHL && 4227 isa<ConstantSDNode>(RHS) && 4228 cast<ConstantSDNode>(RHS)->getZExtValue() == 0x80000000U && 4229 CC == ISD::SETUGT && isa<ConstantSDNode>(LHS.getOperand(1)) && 4230 cast<ConstantSDNode>(LHS.getOperand(1))->getZExtValue() < 31) { 4231 unsigned ShiftAmt = 4232 cast<ConstantSDNode>(LHS.getOperand(1))->getZExtValue() + 1; 4233 SDValue Shift = DAG.getNode(ARMISD::LSLS, dl, 4234 DAG.getVTList(MVT::i32, MVT::i32), 4235 LHS.getOperand(0), 4236 DAG.getConstant(ShiftAmt, dl, MVT::i32)); 4237 SDValue Chain = DAG.getCopyToReg(DAG.getEntryNode(), dl, ARM::CPSR, 4238 Shift.getValue(1), SDValue()); 4239 ARMcc = DAG.getConstant(ARMCC::HI, dl, MVT::i32); 4240 return Chain.getValue(1); 4241 } 4242 4243 ARMCC::CondCodes CondCode = IntCCToARMCC(CC); 4244 4245 // If the RHS is a constant zero then the V (overflow) flag will never be 4246 // set. This can allow us to simplify GE to PL or LT to MI, which can be 4247 // simpler for other passes (like the peephole optimiser) to deal with. 4248 if (isNullConstant(RHS)) { 4249 switch (CondCode) { 4250 default: break; 4251 case ARMCC::GE: 4252 CondCode = ARMCC::PL; 4253 break; 4254 case ARMCC::LT: 4255 CondCode = ARMCC::MI; 4256 break; 4257 } 4258 } 4259 4260 ARMISD::NodeType CompareType; 4261 switch (CondCode) { 4262 default: 4263 CompareType = ARMISD::CMP; 4264 break; 4265 case ARMCC::EQ: 4266 case ARMCC::NE: 4267 // Uses only Z Flag 4268 CompareType = ARMISD::CMPZ; 4269 break; 4270 } 4271 ARMcc = DAG.getConstant(CondCode, dl, MVT::i32); 4272 return DAG.getNode(CompareType, dl, MVT::Glue, LHS, RHS); 4273 } 4274 4275 /// Returns a appropriate VFP CMP (fcmp{s|d}+fmstat) for the given operands. 4276 SDValue ARMTargetLowering::getVFPCmp(SDValue LHS, SDValue RHS, 4277 SelectionDAG &DAG, const SDLoc &dl) const { 4278 assert(Subtarget->hasFP64() || RHS.getValueType() != MVT::f64); 4279 SDValue Cmp; 4280 if (!isFloatingPointZero(RHS)) 4281 Cmp = DAG.getNode(ARMISD::CMPFP, dl, MVT::Glue, LHS, RHS); 4282 else 4283 Cmp = DAG.getNode(ARMISD::CMPFPw0, dl, MVT::Glue, LHS); 4284 return DAG.getNode(ARMISD::FMSTAT, dl, MVT::Glue, Cmp); 4285 } 4286 4287 /// duplicateCmp - Glue values can have only one use, so this function 4288 /// duplicates a comparison node. 4289 SDValue 4290 ARMTargetLowering::duplicateCmp(SDValue Cmp, SelectionDAG &DAG) const { 4291 unsigned Opc = Cmp.getOpcode(); 4292 SDLoc DL(Cmp); 4293 if (Opc == ARMISD::CMP || Opc == ARMISD::CMPZ) 4294 return DAG.getNode(Opc, DL, MVT::Glue, Cmp.getOperand(0),Cmp.getOperand(1)); 4295 4296 assert(Opc == ARMISD::FMSTAT && "unexpected comparison operation"); 4297 Cmp = Cmp.getOperand(0); 4298 Opc = Cmp.getOpcode(); 4299 if (Opc == ARMISD::CMPFP) 4300 Cmp = DAG.getNode(Opc, DL, MVT::Glue, Cmp.getOperand(0),Cmp.getOperand(1)); 4301 else { 4302 assert(Opc == ARMISD::CMPFPw0 && "unexpected operand of FMSTAT"); 4303 Cmp = DAG.getNode(Opc, DL, MVT::Glue, Cmp.getOperand(0)); 4304 } 4305 return DAG.getNode(ARMISD::FMSTAT, DL, MVT::Glue, Cmp); 4306 } 4307 4308 // This function returns three things: the arithmetic computation itself 4309 // (Value), a comparison (OverflowCmp), and a condition code (ARMcc). The 4310 // comparison and the condition code define the case in which the arithmetic 4311 // computation *does not* overflow. 4312 std::pair<SDValue, SDValue> 4313 ARMTargetLowering::getARMXALUOOp(SDValue Op, SelectionDAG &DAG, 4314 SDValue &ARMcc) const { 4315 assert(Op.getValueType() == MVT::i32 && "Unsupported value type"); 4316 4317 SDValue Value, OverflowCmp; 4318 SDValue LHS = Op.getOperand(0); 4319 SDValue RHS = Op.getOperand(1); 4320 SDLoc dl(Op); 4321 4322 // FIXME: We are currently always generating CMPs because we don't support 4323 // generating CMN through the backend. This is not as good as the natural 4324 // CMP case because it causes a register dependency and cannot be folded 4325 // later. 4326 4327 switch (Op.getOpcode()) { 4328 default: 4329 llvm_unreachable("Unknown overflow instruction!"); 4330 case ISD::SADDO: 4331 ARMcc = DAG.getConstant(ARMCC::VC, dl, MVT::i32); 4332 Value = DAG.getNode(ISD::ADD, dl, Op.getValueType(), LHS, RHS); 4333 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, MVT::Glue, Value, LHS); 4334 break; 4335 case ISD::UADDO: 4336 ARMcc = DAG.getConstant(ARMCC::HS, dl, MVT::i32); 4337 // We use ADDC here to correspond to its use in LowerUnsignedALUO. 4338 // We do not use it in the USUBO case as Value may not be used. 4339 Value = DAG.getNode(ARMISD::ADDC, dl, 4340 DAG.getVTList(Op.getValueType(), MVT::i32), LHS, RHS) 4341 .getValue(0); 4342 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, MVT::Glue, Value, LHS); 4343 break; 4344 case ISD::SSUBO: 4345 ARMcc = DAG.getConstant(ARMCC::VC, dl, MVT::i32); 4346 Value = DAG.getNode(ISD::SUB, dl, Op.getValueType(), LHS, RHS); 4347 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, MVT::Glue, LHS, RHS); 4348 break; 4349 case ISD::USUBO: 4350 ARMcc = DAG.getConstant(ARMCC::HS, dl, MVT::i32); 4351 Value = DAG.getNode(ISD::SUB, dl, Op.getValueType(), LHS, RHS); 4352 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, MVT::Glue, LHS, RHS); 4353 break; 4354 case ISD::UMULO: 4355 // We generate a UMUL_LOHI and then check if the high word is 0. 4356 ARMcc = DAG.getConstant(ARMCC::EQ, dl, MVT::i32); 4357 Value = DAG.getNode(ISD::UMUL_LOHI, dl, 4358 DAG.getVTList(Op.getValueType(), Op.getValueType()), 4359 LHS, RHS); 4360 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, MVT::Glue, Value.getValue(1), 4361 DAG.getConstant(0, dl, MVT::i32)); 4362 Value = Value.getValue(0); // We only want the low 32 bits for the result. 4363 break; 4364 case ISD::SMULO: 4365 // We generate a SMUL_LOHI and then check if all the bits of the high word 4366 // are the same as the sign bit of the low word. 4367 ARMcc = DAG.getConstant(ARMCC::EQ, dl, MVT::i32); 4368 Value = DAG.getNode(ISD::SMUL_LOHI, dl, 4369 DAG.getVTList(Op.getValueType(), Op.getValueType()), 4370 LHS, RHS); 4371 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, MVT::Glue, Value.getValue(1), 4372 DAG.getNode(ISD::SRA, dl, Op.getValueType(), 4373 Value.getValue(0), 4374 DAG.getConstant(31, dl, MVT::i32))); 4375 Value = Value.getValue(0); // We only want the low 32 bits for the result. 4376 break; 4377 } // switch (...) 4378 4379 return std::make_pair(Value, OverflowCmp); 4380 } 4381 4382 SDValue 4383 ARMTargetLowering::LowerSignedALUO(SDValue Op, SelectionDAG &DAG) const { 4384 // Let legalize expand this if it isn't a legal type yet. 4385 if (!DAG.getTargetLoweringInfo().isTypeLegal(Op.getValueType())) 4386 return SDValue(); 4387 4388 SDValue Value, OverflowCmp; 4389 SDValue ARMcc; 4390 std::tie(Value, OverflowCmp) = getARMXALUOOp(Op, DAG, ARMcc); 4391 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 4392 SDLoc dl(Op); 4393 // We use 0 and 1 as false and true values. 4394 SDValue TVal = DAG.getConstant(1, dl, MVT::i32); 4395 SDValue FVal = DAG.getConstant(0, dl, MVT::i32); 4396 EVT VT = Op.getValueType(); 4397 4398 SDValue Overflow = DAG.getNode(ARMISD::CMOV, dl, VT, TVal, FVal, 4399 ARMcc, CCR, OverflowCmp); 4400 4401 SDVTList VTs = DAG.getVTList(Op.getValueType(), MVT::i32); 4402 return DAG.getNode(ISD::MERGE_VALUES, dl, VTs, Value, Overflow); 4403 } 4404 4405 static SDValue ConvertBooleanCarryToCarryFlag(SDValue BoolCarry, 4406 SelectionDAG &DAG) { 4407 SDLoc DL(BoolCarry); 4408 EVT CarryVT = BoolCarry.getValueType(); 4409 4410 // This converts the boolean value carry into the carry flag by doing 4411 // ARMISD::SUBC Carry, 1 4412 SDValue Carry = DAG.getNode(ARMISD::SUBC, DL, 4413 DAG.getVTList(CarryVT, MVT::i32), 4414 BoolCarry, DAG.getConstant(1, DL, CarryVT)); 4415 return Carry.getValue(1); 4416 } 4417 4418 static SDValue ConvertCarryFlagToBooleanCarry(SDValue Flags, EVT VT, 4419 SelectionDAG &DAG) { 4420 SDLoc DL(Flags); 4421 4422 // Now convert the carry flag into a boolean carry. We do this 4423 // using ARMISD:ADDE 0, 0, Carry 4424 return DAG.getNode(ARMISD::ADDE, DL, DAG.getVTList(VT, MVT::i32), 4425 DAG.getConstant(0, DL, MVT::i32), 4426 DAG.getConstant(0, DL, MVT::i32), Flags); 4427 } 4428 4429 SDValue ARMTargetLowering::LowerUnsignedALUO(SDValue Op, 4430 SelectionDAG &DAG) const { 4431 // Let legalize expand this if it isn't a legal type yet. 4432 if (!DAG.getTargetLoweringInfo().isTypeLegal(Op.getValueType())) 4433 return SDValue(); 4434 4435 SDValue LHS = Op.getOperand(0); 4436 SDValue RHS = Op.getOperand(1); 4437 SDLoc dl(Op); 4438 4439 EVT VT = Op.getValueType(); 4440 SDVTList VTs = DAG.getVTList(VT, MVT::i32); 4441 SDValue Value; 4442 SDValue Overflow; 4443 switch (Op.getOpcode()) { 4444 default: 4445 llvm_unreachable("Unknown overflow instruction!"); 4446 case ISD::UADDO: 4447 Value = DAG.getNode(ARMISD::ADDC, dl, VTs, LHS, RHS); 4448 // Convert the carry flag into a boolean value. 4449 Overflow = ConvertCarryFlagToBooleanCarry(Value.getValue(1), VT, DAG); 4450 break; 4451 case ISD::USUBO: { 4452 Value = DAG.getNode(ARMISD::SUBC, dl, VTs, LHS, RHS); 4453 // Convert the carry flag into a boolean value. 4454 Overflow = ConvertCarryFlagToBooleanCarry(Value.getValue(1), VT, DAG); 4455 // ARMISD::SUBC returns 0 when we have to borrow, so make it an overflow 4456 // value. So compute 1 - C. 4457 Overflow = DAG.getNode(ISD::SUB, dl, MVT::i32, 4458 DAG.getConstant(1, dl, MVT::i32), Overflow); 4459 break; 4460 } 4461 } 4462 4463 return DAG.getNode(ISD::MERGE_VALUES, dl, VTs, Value, Overflow); 4464 } 4465 4466 static SDValue LowerSADDSUBSAT(SDValue Op, SelectionDAG &DAG, 4467 const ARMSubtarget *Subtarget) { 4468 EVT VT = Op.getValueType(); 4469 if (!Subtarget->hasDSP()) 4470 return SDValue(); 4471 if (!VT.isSimple()) 4472 return SDValue(); 4473 4474 unsigned NewOpcode; 4475 bool IsAdd = Op->getOpcode() == ISD::SADDSAT; 4476 switch (VT.getSimpleVT().SimpleTy) { 4477 default: 4478 return SDValue(); 4479 case MVT::i8: 4480 NewOpcode = IsAdd ? ARMISD::QADD8b : ARMISD::QSUB8b; 4481 break; 4482 case MVT::i16: 4483 NewOpcode = IsAdd ? ARMISD::QADD16b : ARMISD::QSUB16b; 4484 break; 4485 } 4486 4487 SDLoc dl(Op); 4488 SDValue Add = 4489 DAG.getNode(NewOpcode, dl, MVT::i32, 4490 DAG.getSExtOrTrunc(Op->getOperand(0), dl, MVT::i32), 4491 DAG.getSExtOrTrunc(Op->getOperand(1), dl, MVT::i32)); 4492 return DAG.getNode(ISD::TRUNCATE, dl, VT, Add); 4493 } 4494 4495 SDValue ARMTargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const { 4496 SDValue Cond = Op.getOperand(0); 4497 SDValue SelectTrue = Op.getOperand(1); 4498 SDValue SelectFalse = Op.getOperand(2); 4499 SDLoc dl(Op); 4500 unsigned Opc = Cond.getOpcode(); 4501 4502 if (Cond.getResNo() == 1 && 4503 (Opc == ISD::SADDO || Opc == ISD::UADDO || Opc == ISD::SSUBO || 4504 Opc == ISD::USUBO)) { 4505 if (!DAG.getTargetLoweringInfo().isTypeLegal(Cond->getValueType(0))) 4506 return SDValue(); 4507 4508 SDValue Value, OverflowCmp; 4509 SDValue ARMcc; 4510 std::tie(Value, OverflowCmp) = getARMXALUOOp(Cond, DAG, ARMcc); 4511 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 4512 EVT VT = Op.getValueType(); 4513 4514 return getCMOV(dl, VT, SelectTrue, SelectFalse, ARMcc, CCR, 4515 OverflowCmp, DAG); 4516 } 4517 4518 // Convert: 4519 // 4520 // (select (cmov 1, 0, cond), t, f) -> (cmov t, f, cond) 4521 // (select (cmov 0, 1, cond), t, f) -> (cmov f, t, cond) 4522 // 4523 if (Cond.getOpcode() == ARMISD::CMOV && Cond.hasOneUse()) { 4524 const ConstantSDNode *CMOVTrue = 4525 dyn_cast<ConstantSDNode>(Cond.getOperand(0)); 4526 const ConstantSDNode *CMOVFalse = 4527 dyn_cast<ConstantSDNode>(Cond.getOperand(1)); 4528 4529 if (CMOVTrue && CMOVFalse) { 4530 unsigned CMOVTrueVal = CMOVTrue->getZExtValue(); 4531 unsigned CMOVFalseVal = CMOVFalse->getZExtValue(); 4532 4533 SDValue True; 4534 SDValue False; 4535 if (CMOVTrueVal == 1 && CMOVFalseVal == 0) { 4536 True = SelectTrue; 4537 False = SelectFalse; 4538 } else if (CMOVTrueVal == 0 && CMOVFalseVal == 1) { 4539 True = SelectFalse; 4540 False = SelectTrue; 4541 } 4542 4543 if (True.getNode() && False.getNode()) { 4544 EVT VT = Op.getValueType(); 4545 SDValue ARMcc = Cond.getOperand(2); 4546 SDValue CCR = Cond.getOperand(3); 4547 SDValue Cmp = duplicateCmp(Cond.getOperand(4), DAG); 4548 assert(True.getValueType() == VT); 4549 return getCMOV(dl, VT, True, False, ARMcc, CCR, Cmp, DAG); 4550 } 4551 } 4552 } 4553 4554 // ARM's BooleanContents value is UndefinedBooleanContent. Mask out the 4555 // undefined bits before doing a full-word comparison with zero. 4556 Cond = DAG.getNode(ISD::AND, dl, Cond.getValueType(), Cond, 4557 DAG.getConstant(1, dl, Cond.getValueType())); 4558 4559 return DAG.getSelectCC(dl, Cond, 4560 DAG.getConstant(0, dl, Cond.getValueType()), 4561 SelectTrue, SelectFalse, ISD::SETNE); 4562 } 4563 4564 static void checkVSELConstraints(ISD::CondCode CC, ARMCC::CondCodes &CondCode, 4565 bool &swpCmpOps, bool &swpVselOps) { 4566 // Start by selecting the GE condition code for opcodes that return true for 4567 // 'equality' 4568 if (CC == ISD::SETUGE || CC == ISD::SETOGE || CC == ISD::SETOLE || 4569 CC == ISD::SETULE || CC == ISD::SETGE || CC == ISD::SETLE) 4570 CondCode = ARMCC::GE; 4571 4572 // and GT for opcodes that return false for 'equality'. 4573 else if (CC == ISD::SETUGT || CC == ISD::SETOGT || CC == ISD::SETOLT || 4574 CC == ISD::SETULT || CC == ISD::SETGT || CC == ISD::SETLT) 4575 CondCode = ARMCC::GT; 4576 4577 // Since we are constrained to GE/GT, if the opcode contains 'less', we need 4578 // to swap the compare operands. 4579 if (CC == ISD::SETOLE || CC == ISD::SETULE || CC == ISD::SETOLT || 4580 CC == ISD::SETULT || CC == ISD::SETLE || CC == ISD::SETLT) 4581 swpCmpOps = true; 4582 4583 // Both GT and GE are ordered comparisons, and return false for 'unordered'. 4584 // If we have an unordered opcode, we need to swap the operands to the VSEL 4585 // instruction (effectively negating the condition). 4586 // 4587 // This also has the effect of swapping which one of 'less' or 'greater' 4588 // returns true, so we also swap the compare operands. It also switches 4589 // whether we return true for 'equality', so we compensate by picking the 4590 // opposite condition code to our original choice. 4591 if (CC == ISD::SETULE || CC == ISD::SETULT || CC == ISD::SETUGE || 4592 CC == ISD::SETUGT) { 4593 swpCmpOps = !swpCmpOps; 4594 swpVselOps = !swpVselOps; 4595 CondCode = CondCode == ARMCC::GT ? ARMCC::GE : ARMCC::GT; 4596 } 4597 4598 // 'ordered' is 'anything but unordered', so use the VS condition code and 4599 // swap the VSEL operands. 4600 if (CC == ISD::SETO) { 4601 CondCode = ARMCC::VS; 4602 swpVselOps = true; 4603 } 4604 4605 // 'unordered or not equal' is 'anything but equal', so use the EQ condition 4606 // code and swap the VSEL operands. Also do this if we don't care about the 4607 // unordered case. 4608 if (CC == ISD::SETUNE || CC == ISD::SETNE) { 4609 CondCode = ARMCC::EQ; 4610 swpVselOps = true; 4611 } 4612 } 4613 4614 SDValue ARMTargetLowering::getCMOV(const SDLoc &dl, EVT VT, SDValue FalseVal, 4615 SDValue TrueVal, SDValue ARMcc, SDValue CCR, 4616 SDValue Cmp, SelectionDAG &DAG) const { 4617 if (!Subtarget->hasFP64() && VT == MVT::f64) { 4618 FalseVal = DAG.getNode(ARMISD::VMOVRRD, dl, 4619 DAG.getVTList(MVT::i32, MVT::i32), FalseVal); 4620 TrueVal = DAG.getNode(ARMISD::VMOVRRD, dl, 4621 DAG.getVTList(MVT::i32, MVT::i32), TrueVal); 4622 4623 SDValue TrueLow = TrueVal.getValue(0); 4624 SDValue TrueHigh = TrueVal.getValue(1); 4625 SDValue FalseLow = FalseVal.getValue(0); 4626 SDValue FalseHigh = FalseVal.getValue(1); 4627 4628 SDValue Low = DAG.getNode(ARMISD::CMOV, dl, MVT::i32, FalseLow, TrueLow, 4629 ARMcc, CCR, Cmp); 4630 SDValue High = DAG.getNode(ARMISD::CMOV, dl, MVT::i32, FalseHigh, TrueHigh, 4631 ARMcc, CCR, duplicateCmp(Cmp, DAG)); 4632 4633 return DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Low, High); 4634 } else { 4635 return DAG.getNode(ARMISD::CMOV, dl, VT, FalseVal, TrueVal, ARMcc, CCR, 4636 Cmp); 4637 } 4638 } 4639 4640 static bool isGTorGE(ISD::CondCode CC) { 4641 return CC == ISD::SETGT || CC == ISD::SETGE; 4642 } 4643 4644 static bool isLTorLE(ISD::CondCode CC) { 4645 return CC == ISD::SETLT || CC == ISD::SETLE; 4646 } 4647 4648 // See if a conditional (LHS CC RHS ? TrueVal : FalseVal) is lower-saturating. 4649 // All of these conditions (and their <= and >= counterparts) will do: 4650 // x < k ? k : x 4651 // x > k ? x : k 4652 // k < x ? x : k 4653 // k > x ? k : x 4654 static bool isLowerSaturate(const SDValue LHS, const SDValue RHS, 4655 const SDValue TrueVal, const SDValue FalseVal, 4656 const ISD::CondCode CC, const SDValue K) { 4657 return (isGTorGE(CC) && 4658 ((K == LHS && K == TrueVal) || (K == RHS && K == FalseVal))) || 4659 (isLTorLE(CC) && 4660 ((K == RHS && K == TrueVal) || (K == LHS && K == FalseVal))); 4661 } 4662 4663 // Similar to isLowerSaturate(), but checks for upper-saturating conditions. 4664 static bool isUpperSaturate(const SDValue LHS, const SDValue RHS, 4665 const SDValue TrueVal, const SDValue FalseVal, 4666 const ISD::CondCode CC, const SDValue K) { 4667 return (isGTorGE(CC) && 4668 ((K == RHS && K == TrueVal) || (K == LHS && K == FalseVal))) || 4669 (isLTorLE(CC) && 4670 ((K == LHS && K == TrueVal) || (K == RHS && K == FalseVal))); 4671 } 4672 4673 // Check if two chained conditionals could be converted into SSAT or USAT. 4674 // 4675 // SSAT can replace a set of two conditional selectors that bound a number to an 4676 // interval of type [k, ~k] when k + 1 is a power of 2. Here are some examples: 4677 // 4678 // x < -k ? -k : (x > k ? k : x) 4679 // x < -k ? -k : (x < k ? x : k) 4680 // x > -k ? (x > k ? k : x) : -k 4681 // x < k ? (x < -k ? -k : x) : k 4682 // etc. 4683 // 4684 // USAT works similarily to SSAT but bounds on the interval [0, k] where k + 1 is 4685 // a power of 2. 4686 // 4687 // It returns true if the conversion can be done, false otherwise. 4688 // Additionally, the variable is returned in parameter V, the constant in K and 4689 // usat is set to true if the conditional represents an unsigned saturation 4690 static bool isSaturatingConditional(const SDValue &Op, SDValue &V, 4691 uint64_t &K, bool &usat) { 4692 SDValue LHS1 = Op.getOperand(0); 4693 SDValue RHS1 = Op.getOperand(1); 4694 SDValue TrueVal1 = Op.getOperand(2); 4695 SDValue FalseVal1 = Op.getOperand(3); 4696 ISD::CondCode CC1 = cast<CondCodeSDNode>(Op.getOperand(4))->get(); 4697 4698 const SDValue Op2 = isa<ConstantSDNode>(TrueVal1) ? FalseVal1 : TrueVal1; 4699 if (Op2.getOpcode() != ISD::SELECT_CC) 4700 return false; 4701 4702 SDValue LHS2 = Op2.getOperand(0); 4703 SDValue RHS2 = Op2.getOperand(1); 4704 SDValue TrueVal2 = Op2.getOperand(2); 4705 SDValue FalseVal2 = Op2.getOperand(3); 4706 ISD::CondCode CC2 = cast<CondCodeSDNode>(Op2.getOperand(4))->get(); 4707 4708 // Find out which are the constants and which are the variables 4709 // in each conditional 4710 SDValue *K1 = isa<ConstantSDNode>(LHS1) ? &LHS1 : isa<ConstantSDNode>(RHS1) 4711 ? &RHS1 4712 : nullptr; 4713 SDValue *K2 = isa<ConstantSDNode>(LHS2) ? &LHS2 : isa<ConstantSDNode>(RHS2) 4714 ? &RHS2 4715 : nullptr; 4716 SDValue K2Tmp = isa<ConstantSDNode>(TrueVal2) ? TrueVal2 : FalseVal2; 4717 SDValue V1Tmp = (K1 && *K1 == LHS1) ? RHS1 : LHS1; 4718 SDValue V2Tmp = (K2 && *K2 == LHS2) ? RHS2 : LHS2; 4719 SDValue V2 = (K2Tmp == TrueVal2) ? FalseVal2 : TrueVal2; 4720 4721 // We must detect cases where the original operations worked with 16- or 4722 // 8-bit values. In such case, V2Tmp != V2 because the comparison operations 4723 // must work with sign-extended values but the select operations return 4724 // the original non-extended value. 4725 SDValue V2TmpReg = V2Tmp; 4726 if (V2Tmp->getOpcode() == ISD::SIGN_EXTEND_INREG) 4727 V2TmpReg = V2Tmp->getOperand(0); 4728 4729 // Check that the registers and the constants have the correct values 4730 // in both conditionals 4731 if (!K1 || !K2 || *K1 == Op2 || *K2 != K2Tmp || V1Tmp != V2Tmp || 4732 V2TmpReg != V2) 4733 return false; 4734 4735 // Figure out which conditional is saturating the lower/upper bound. 4736 const SDValue *LowerCheckOp = 4737 isLowerSaturate(LHS1, RHS1, TrueVal1, FalseVal1, CC1, *K1) 4738 ? &Op 4739 : isLowerSaturate(LHS2, RHS2, TrueVal2, FalseVal2, CC2, *K2) 4740 ? &Op2 4741 : nullptr; 4742 const SDValue *UpperCheckOp = 4743 isUpperSaturate(LHS1, RHS1, TrueVal1, FalseVal1, CC1, *K1) 4744 ? &Op 4745 : isUpperSaturate(LHS2, RHS2, TrueVal2, FalseVal2, CC2, *K2) 4746 ? &Op2 4747 : nullptr; 4748 4749 if (!UpperCheckOp || !LowerCheckOp || LowerCheckOp == UpperCheckOp) 4750 return false; 4751 4752 // Check that the constant in the lower-bound check is 4753 // the opposite of the constant in the upper-bound check 4754 // in 1's complement. 4755 int64_t Val1 = cast<ConstantSDNode>(*K1)->getSExtValue(); 4756 int64_t Val2 = cast<ConstantSDNode>(*K2)->getSExtValue(); 4757 int64_t PosVal = std::max(Val1, Val2); 4758 int64_t NegVal = std::min(Val1, Val2); 4759 4760 if (((Val1 > Val2 && UpperCheckOp == &Op) || 4761 (Val1 < Val2 && UpperCheckOp == &Op2)) && 4762 isPowerOf2_64(PosVal + 1)) { 4763 4764 // Handle the difference between USAT (unsigned) and SSAT (signed) saturation 4765 if (Val1 == ~Val2) 4766 usat = false; 4767 else if (NegVal == 0) 4768 usat = true; 4769 else 4770 return false; 4771 4772 V = V2; 4773 K = (uint64_t)PosVal; // At this point, PosVal is guaranteed to be positive 4774 4775 return true; 4776 } 4777 4778 return false; 4779 } 4780 4781 // Check if a condition of the type x < k ? k : x can be converted into a 4782 // bit operation instead of conditional moves. 4783 // Currently this is allowed given: 4784 // - The conditions and values match up 4785 // - k is 0 or -1 (all ones) 4786 // This function will not check the last condition, thats up to the caller 4787 // It returns true if the transformation can be made, and in such case 4788 // returns x in V, and k in SatK. 4789 static bool isLowerSaturatingConditional(const SDValue &Op, SDValue &V, 4790 SDValue &SatK) 4791 { 4792 SDValue LHS = Op.getOperand(0); 4793 SDValue RHS = Op.getOperand(1); 4794 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(4))->get(); 4795 SDValue TrueVal = Op.getOperand(2); 4796 SDValue FalseVal = Op.getOperand(3); 4797 4798 SDValue *K = isa<ConstantSDNode>(LHS) ? &LHS : isa<ConstantSDNode>(RHS) 4799 ? &RHS 4800 : nullptr; 4801 4802 // No constant operation in comparison, early out 4803 if (!K) 4804 return false; 4805 4806 SDValue KTmp = isa<ConstantSDNode>(TrueVal) ? TrueVal : FalseVal; 4807 V = (KTmp == TrueVal) ? FalseVal : TrueVal; 4808 SDValue VTmp = (K && *K == LHS) ? RHS : LHS; 4809 4810 // If the constant on left and right side, or variable on left and right, 4811 // does not match, early out 4812 if (*K != KTmp || V != VTmp) 4813 return false; 4814 4815 if (isLowerSaturate(LHS, RHS, TrueVal, FalseVal, CC, *K)) { 4816 SatK = *K; 4817 return true; 4818 } 4819 4820 return false; 4821 } 4822 4823 bool ARMTargetLowering::isUnsupportedFloatingType(EVT VT) const { 4824 if (VT == MVT::f32) 4825 return !Subtarget->hasVFP2Base(); 4826 if (VT == MVT::f64) 4827 return !Subtarget->hasFP64(); 4828 if (VT == MVT::f16) 4829 return !Subtarget->hasFullFP16(); 4830 return false; 4831 } 4832 4833 SDValue ARMTargetLowering::LowerSELECT_CC(SDValue Op, SelectionDAG &DAG) const { 4834 EVT VT = Op.getValueType(); 4835 SDLoc dl(Op); 4836 4837 // Try to convert two saturating conditional selects into a single SSAT 4838 SDValue SatValue; 4839 uint64_t SatConstant; 4840 bool SatUSat; 4841 if (((!Subtarget->isThumb() && Subtarget->hasV6Ops()) || Subtarget->isThumb2()) && 4842 isSaturatingConditional(Op, SatValue, SatConstant, SatUSat)) { 4843 if (SatUSat) 4844 return DAG.getNode(ARMISD::USAT, dl, VT, SatValue, 4845 DAG.getConstant(countTrailingOnes(SatConstant), dl, VT)); 4846 else 4847 return DAG.getNode(ARMISD::SSAT, dl, VT, SatValue, 4848 DAG.getConstant(countTrailingOnes(SatConstant), dl, VT)); 4849 } 4850 4851 // Try to convert expressions of the form x < k ? k : x (and similar forms) 4852 // into more efficient bit operations, which is possible when k is 0 or -1 4853 // On ARM and Thumb-2 which have flexible operand 2 this will result in 4854 // single instructions. On Thumb the shift and the bit operation will be two 4855 // instructions. 4856 // Only allow this transformation on full-width (32-bit) operations 4857 SDValue LowerSatConstant; 4858 if (VT == MVT::i32 && 4859 isLowerSaturatingConditional(Op, SatValue, LowerSatConstant)) { 4860 SDValue ShiftV = DAG.getNode(ISD::SRA, dl, VT, SatValue, 4861 DAG.getConstant(31, dl, VT)); 4862 if (isNullConstant(LowerSatConstant)) { 4863 SDValue NotShiftV = DAG.getNode(ISD::XOR, dl, VT, ShiftV, 4864 DAG.getAllOnesConstant(dl, VT)); 4865 return DAG.getNode(ISD::AND, dl, VT, SatValue, NotShiftV); 4866 } else if (isAllOnesConstant(LowerSatConstant)) 4867 return DAG.getNode(ISD::OR, dl, VT, SatValue, ShiftV); 4868 } 4869 4870 SDValue LHS = Op.getOperand(0); 4871 SDValue RHS = Op.getOperand(1); 4872 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(4))->get(); 4873 SDValue TrueVal = Op.getOperand(2); 4874 SDValue FalseVal = Op.getOperand(3); 4875 ConstantSDNode *CFVal = dyn_cast<ConstantSDNode>(FalseVal); 4876 ConstantSDNode *CTVal = dyn_cast<ConstantSDNode>(TrueVal); 4877 4878 if (Subtarget->hasV8_1MMainlineOps() && CFVal && CTVal && 4879 LHS.getValueType() == MVT::i32 && RHS.getValueType() == MVT::i32) { 4880 unsigned TVal = CTVal->getZExtValue(); 4881 unsigned FVal = CFVal->getZExtValue(); 4882 unsigned Opcode = 0; 4883 4884 if (TVal == ~FVal) { 4885 Opcode = ARMISD::CSINV; 4886 } else if (TVal == ~FVal + 1) { 4887 Opcode = ARMISD::CSNEG; 4888 } else if (TVal + 1 == FVal) { 4889 Opcode = ARMISD::CSINC; 4890 } else if (TVal == FVal + 1) { 4891 Opcode = ARMISD::CSINC; 4892 std::swap(TrueVal, FalseVal); 4893 std::swap(TVal, FVal); 4894 CC = ISD::getSetCCInverse(CC, true); 4895 } 4896 4897 if (Opcode) { 4898 // If one of the constants is cheaper than another, materialise the 4899 // cheaper one and let the csel generate the other. 4900 if (Opcode != ARMISD::CSINC && 4901 HasLowerConstantMaterializationCost(FVal, TVal, Subtarget)) { 4902 std::swap(TrueVal, FalseVal); 4903 std::swap(TVal, FVal); 4904 CC = ISD::getSetCCInverse(CC, true); 4905 } 4906 4907 // Attempt to use ZR checking TVal is 0, possibly inverting the condition 4908 // to get there. CSINC not is invertable like the other two (~(~a) == a, 4909 // -(-a) == a, but (a+1)+1 != a). 4910 if (FVal == 0 && Opcode != ARMISD::CSINC) { 4911 std::swap(TrueVal, FalseVal); 4912 std::swap(TVal, FVal); 4913 CC = ISD::getSetCCInverse(CC, true); 4914 } 4915 if (TVal == 0) 4916 TrueVal = DAG.getRegister(ARM::ZR, MVT::i32); 4917 4918 // Drops F's value because we can get it by inverting/negating TVal. 4919 FalseVal = TrueVal; 4920 4921 SDValue ARMcc; 4922 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl); 4923 EVT VT = TrueVal.getValueType(); 4924 return DAG.getNode(Opcode, dl, VT, TrueVal, FalseVal, ARMcc, Cmp); 4925 } 4926 } 4927 4928 if (isUnsupportedFloatingType(LHS.getValueType())) { 4929 DAG.getTargetLoweringInfo().softenSetCCOperands( 4930 DAG, LHS.getValueType(), LHS, RHS, CC, dl, LHS, RHS); 4931 4932 // If softenSetCCOperands only returned one value, we should compare it to 4933 // zero. 4934 if (!RHS.getNode()) { 4935 RHS = DAG.getConstant(0, dl, LHS.getValueType()); 4936 CC = ISD::SETNE; 4937 } 4938 } 4939 4940 if (LHS.getValueType() == MVT::i32) { 4941 // Try to generate VSEL on ARMv8. 4942 // The VSEL instruction can't use all the usual ARM condition 4943 // codes: it only has two bits to select the condition code, so it's 4944 // constrained to use only GE, GT, VS and EQ. 4945 // 4946 // To implement all the various ISD::SETXXX opcodes, we sometimes need to 4947 // swap the operands of the previous compare instruction (effectively 4948 // inverting the compare condition, swapping 'less' and 'greater') and 4949 // sometimes need to swap the operands to the VSEL (which inverts the 4950 // condition in the sense of firing whenever the previous condition didn't) 4951 if (Subtarget->hasFPARMv8Base() && (TrueVal.getValueType() == MVT::f16 || 4952 TrueVal.getValueType() == MVT::f32 || 4953 TrueVal.getValueType() == MVT::f64)) { 4954 ARMCC::CondCodes CondCode = IntCCToARMCC(CC); 4955 if (CondCode == ARMCC::LT || CondCode == ARMCC::LE || 4956 CondCode == ARMCC::VC || CondCode == ARMCC::NE) { 4957 CC = ISD::getSetCCInverse(CC, true); 4958 std::swap(TrueVal, FalseVal); 4959 } 4960 } 4961 4962 SDValue ARMcc; 4963 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 4964 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl); 4965 // Choose GE over PL, which vsel does now support 4966 if (cast<ConstantSDNode>(ARMcc)->getZExtValue() == ARMCC::PL) 4967 ARMcc = DAG.getConstant(ARMCC::GE, dl, MVT::i32); 4968 return getCMOV(dl, VT, FalseVal, TrueVal, ARMcc, CCR, Cmp, DAG); 4969 } 4970 4971 ARMCC::CondCodes CondCode, CondCode2; 4972 FPCCToARMCC(CC, CondCode, CondCode2); 4973 4974 // Normalize the fp compare. If RHS is zero we prefer to keep it there so we 4975 // match CMPFPw0 instead of CMPFP, though we don't do this for f16 because we 4976 // must use VSEL (limited condition codes), due to not having conditional f16 4977 // moves. 4978 if (Subtarget->hasFPARMv8Base() && 4979 !(isFloatingPointZero(RHS) && TrueVal.getValueType() != MVT::f16) && 4980 (TrueVal.getValueType() == MVT::f16 || 4981 TrueVal.getValueType() == MVT::f32 || 4982 TrueVal.getValueType() == MVT::f64)) { 4983 bool swpCmpOps = false; 4984 bool swpVselOps = false; 4985 checkVSELConstraints(CC, CondCode, swpCmpOps, swpVselOps); 4986 4987 if (CondCode == ARMCC::GT || CondCode == ARMCC::GE || 4988 CondCode == ARMCC::VS || CondCode == ARMCC::EQ) { 4989 if (swpCmpOps) 4990 std::swap(LHS, RHS); 4991 if (swpVselOps) 4992 std::swap(TrueVal, FalseVal); 4993 } 4994 } 4995 4996 SDValue ARMcc = DAG.getConstant(CondCode, dl, MVT::i32); 4997 SDValue Cmp = getVFPCmp(LHS, RHS, DAG, dl); 4998 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 4999 SDValue Result = getCMOV(dl, VT, FalseVal, TrueVal, ARMcc, CCR, Cmp, DAG); 5000 if (CondCode2 != ARMCC::AL) { 5001 SDValue ARMcc2 = DAG.getConstant(CondCode2, dl, MVT::i32); 5002 // FIXME: Needs another CMP because flag can have but one use. 5003 SDValue Cmp2 = getVFPCmp(LHS, RHS, DAG, dl); 5004 Result = getCMOV(dl, VT, Result, TrueVal, ARMcc2, CCR, Cmp2, DAG); 5005 } 5006 return Result; 5007 } 5008 5009 /// canChangeToInt - Given the fp compare operand, return true if it is suitable 5010 /// to morph to an integer compare sequence. 5011 static bool canChangeToInt(SDValue Op, bool &SeenZero, 5012 const ARMSubtarget *Subtarget) { 5013 SDNode *N = Op.getNode(); 5014 if (!N->hasOneUse()) 5015 // Otherwise it requires moving the value from fp to integer registers. 5016 return false; 5017 if (!N->getNumValues()) 5018 return false; 5019 EVT VT = Op.getValueType(); 5020 if (VT != MVT::f32 && !Subtarget->isFPBrccSlow()) 5021 // f32 case is generally profitable. f64 case only makes sense when vcmpe + 5022 // vmrs are very slow, e.g. cortex-a8. 5023 return false; 5024 5025 if (isFloatingPointZero(Op)) { 5026 SeenZero = true; 5027 return true; 5028 } 5029 return ISD::isNormalLoad(N); 5030 } 5031 5032 static SDValue bitcastf32Toi32(SDValue Op, SelectionDAG &DAG) { 5033 if (isFloatingPointZero(Op)) 5034 return DAG.getConstant(0, SDLoc(Op), MVT::i32); 5035 5036 if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Op)) 5037 return DAG.getLoad(MVT::i32, SDLoc(Op), Ld->getChain(), Ld->getBasePtr(), 5038 Ld->getPointerInfo(), Ld->getAlignment(), 5039 Ld->getMemOperand()->getFlags()); 5040 5041 llvm_unreachable("Unknown VFP cmp argument!"); 5042 } 5043 5044 static void expandf64Toi32(SDValue Op, SelectionDAG &DAG, 5045 SDValue &RetVal1, SDValue &RetVal2) { 5046 SDLoc dl(Op); 5047 5048 if (isFloatingPointZero(Op)) { 5049 RetVal1 = DAG.getConstant(0, dl, MVT::i32); 5050 RetVal2 = DAG.getConstant(0, dl, MVT::i32); 5051 return; 5052 } 5053 5054 if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Op)) { 5055 SDValue Ptr = Ld->getBasePtr(); 5056 RetVal1 = 5057 DAG.getLoad(MVT::i32, dl, Ld->getChain(), Ptr, Ld->getPointerInfo(), 5058 Ld->getAlignment(), Ld->getMemOperand()->getFlags()); 5059 5060 EVT PtrType = Ptr.getValueType(); 5061 unsigned NewAlign = MinAlign(Ld->getAlignment(), 4); 5062 SDValue NewPtr = DAG.getNode(ISD::ADD, dl, 5063 PtrType, Ptr, DAG.getConstant(4, dl, PtrType)); 5064 RetVal2 = DAG.getLoad(MVT::i32, dl, Ld->getChain(), NewPtr, 5065 Ld->getPointerInfo().getWithOffset(4), NewAlign, 5066 Ld->getMemOperand()->getFlags()); 5067 return; 5068 } 5069 5070 llvm_unreachable("Unknown VFP cmp argument!"); 5071 } 5072 5073 /// OptimizeVFPBrcond - With -enable-unsafe-fp-math, it's legal to optimize some 5074 /// f32 and even f64 comparisons to integer ones. 5075 SDValue 5076 ARMTargetLowering::OptimizeVFPBrcond(SDValue Op, SelectionDAG &DAG) const { 5077 SDValue Chain = Op.getOperand(0); 5078 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(1))->get(); 5079 SDValue LHS = Op.getOperand(2); 5080 SDValue RHS = Op.getOperand(3); 5081 SDValue Dest = Op.getOperand(4); 5082 SDLoc dl(Op); 5083 5084 bool LHSSeenZero = false; 5085 bool LHSOk = canChangeToInt(LHS, LHSSeenZero, Subtarget); 5086 bool RHSSeenZero = false; 5087 bool RHSOk = canChangeToInt(RHS, RHSSeenZero, Subtarget); 5088 if (LHSOk && RHSOk && (LHSSeenZero || RHSSeenZero)) { 5089 // If unsafe fp math optimization is enabled and there are no other uses of 5090 // the CMP operands, and the condition code is EQ or NE, we can optimize it 5091 // to an integer comparison. 5092 if (CC == ISD::SETOEQ) 5093 CC = ISD::SETEQ; 5094 else if (CC == ISD::SETUNE) 5095 CC = ISD::SETNE; 5096 5097 SDValue Mask = DAG.getConstant(0x7fffffff, dl, MVT::i32); 5098 SDValue ARMcc; 5099 if (LHS.getValueType() == MVT::f32) { 5100 LHS = DAG.getNode(ISD::AND, dl, MVT::i32, 5101 bitcastf32Toi32(LHS, DAG), Mask); 5102 RHS = DAG.getNode(ISD::AND, dl, MVT::i32, 5103 bitcastf32Toi32(RHS, DAG), Mask); 5104 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl); 5105 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 5106 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, 5107 Chain, Dest, ARMcc, CCR, Cmp); 5108 } 5109 5110 SDValue LHS1, LHS2; 5111 SDValue RHS1, RHS2; 5112 expandf64Toi32(LHS, DAG, LHS1, LHS2); 5113 expandf64Toi32(RHS, DAG, RHS1, RHS2); 5114 LHS2 = DAG.getNode(ISD::AND, dl, MVT::i32, LHS2, Mask); 5115 RHS2 = DAG.getNode(ISD::AND, dl, MVT::i32, RHS2, Mask); 5116 ARMCC::CondCodes CondCode = IntCCToARMCC(CC); 5117 ARMcc = DAG.getConstant(CondCode, dl, MVT::i32); 5118 SDVTList VTList = DAG.getVTList(MVT::Other, MVT::Glue); 5119 SDValue Ops[] = { Chain, ARMcc, LHS1, LHS2, RHS1, RHS2, Dest }; 5120 return DAG.getNode(ARMISD::BCC_i64, dl, VTList, Ops); 5121 } 5122 5123 return SDValue(); 5124 } 5125 5126 SDValue ARMTargetLowering::LowerBRCOND(SDValue Op, SelectionDAG &DAG) const { 5127 SDValue Chain = Op.getOperand(0); 5128 SDValue Cond = Op.getOperand(1); 5129 SDValue Dest = Op.getOperand(2); 5130 SDLoc dl(Op); 5131 5132 // Optimize {s|u}{add|sub|mul}.with.overflow feeding into a branch 5133 // instruction. 5134 unsigned Opc = Cond.getOpcode(); 5135 bool OptimizeMul = (Opc == ISD::SMULO || Opc == ISD::UMULO) && 5136 !Subtarget->isThumb1Only(); 5137 if (Cond.getResNo() == 1 && 5138 (Opc == ISD::SADDO || Opc == ISD::UADDO || Opc == ISD::SSUBO || 5139 Opc == ISD::USUBO || OptimizeMul)) { 5140 // Only lower legal XALUO ops. 5141 if (!DAG.getTargetLoweringInfo().isTypeLegal(Cond->getValueType(0))) 5142 return SDValue(); 5143 5144 // The actual operation with overflow check. 5145 SDValue Value, OverflowCmp; 5146 SDValue ARMcc; 5147 std::tie(Value, OverflowCmp) = getARMXALUOOp(Cond, DAG, ARMcc); 5148 5149 // Reverse the condition code. 5150 ARMCC::CondCodes CondCode = 5151 (ARMCC::CondCodes)cast<const ConstantSDNode>(ARMcc)->getZExtValue(); 5152 CondCode = ARMCC::getOppositeCondition(CondCode); 5153 ARMcc = DAG.getConstant(CondCode, SDLoc(ARMcc), MVT::i32); 5154 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 5155 5156 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Chain, Dest, ARMcc, CCR, 5157 OverflowCmp); 5158 } 5159 5160 return SDValue(); 5161 } 5162 5163 SDValue ARMTargetLowering::LowerBR_CC(SDValue Op, SelectionDAG &DAG) const { 5164 SDValue Chain = Op.getOperand(0); 5165 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(1))->get(); 5166 SDValue LHS = Op.getOperand(2); 5167 SDValue RHS = Op.getOperand(3); 5168 SDValue Dest = Op.getOperand(4); 5169 SDLoc dl(Op); 5170 5171 if (isUnsupportedFloatingType(LHS.getValueType())) { 5172 DAG.getTargetLoweringInfo().softenSetCCOperands( 5173 DAG, LHS.getValueType(), LHS, RHS, CC, dl, LHS, RHS); 5174 5175 // If softenSetCCOperands only returned one value, we should compare it to 5176 // zero. 5177 if (!RHS.getNode()) { 5178 RHS = DAG.getConstant(0, dl, LHS.getValueType()); 5179 CC = ISD::SETNE; 5180 } 5181 } 5182 5183 // Optimize {s|u}{add|sub|mul}.with.overflow feeding into a branch 5184 // instruction. 5185 unsigned Opc = LHS.getOpcode(); 5186 bool OptimizeMul = (Opc == ISD::SMULO || Opc == ISD::UMULO) && 5187 !Subtarget->isThumb1Only(); 5188 if (LHS.getResNo() == 1 && (isOneConstant(RHS) || isNullConstant(RHS)) && 5189 (Opc == ISD::SADDO || Opc == ISD::UADDO || Opc == ISD::SSUBO || 5190 Opc == ISD::USUBO || OptimizeMul) && 5191 (CC == ISD::SETEQ || CC == ISD::SETNE)) { 5192 // Only lower legal XALUO ops. 5193 if (!DAG.getTargetLoweringInfo().isTypeLegal(LHS->getValueType(0))) 5194 return SDValue(); 5195 5196 // The actual operation with overflow check. 5197 SDValue Value, OverflowCmp; 5198 SDValue ARMcc; 5199 std::tie(Value, OverflowCmp) = getARMXALUOOp(LHS.getValue(0), DAG, ARMcc); 5200 5201 if ((CC == ISD::SETNE) != isOneConstant(RHS)) { 5202 // Reverse the condition code. 5203 ARMCC::CondCodes CondCode = 5204 (ARMCC::CondCodes)cast<const ConstantSDNode>(ARMcc)->getZExtValue(); 5205 CondCode = ARMCC::getOppositeCondition(CondCode); 5206 ARMcc = DAG.getConstant(CondCode, SDLoc(ARMcc), MVT::i32); 5207 } 5208 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 5209 5210 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Chain, Dest, ARMcc, CCR, 5211 OverflowCmp); 5212 } 5213 5214 if (LHS.getValueType() == MVT::i32) { 5215 SDValue ARMcc; 5216 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl); 5217 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 5218 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, 5219 Chain, Dest, ARMcc, CCR, Cmp); 5220 } 5221 5222 if (getTargetMachine().Options.UnsafeFPMath && 5223 (CC == ISD::SETEQ || CC == ISD::SETOEQ || 5224 CC == ISD::SETNE || CC == ISD::SETUNE)) { 5225 if (SDValue Result = OptimizeVFPBrcond(Op, DAG)) 5226 return Result; 5227 } 5228 5229 ARMCC::CondCodes CondCode, CondCode2; 5230 FPCCToARMCC(CC, CondCode, CondCode2); 5231 5232 SDValue ARMcc = DAG.getConstant(CondCode, dl, MVT::i32); 5233 SDValue Cmp = getVFPCmp(LHS, RHS, DAG, dl); 5234 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 5235 SDVTList VTList = DAG.getVTList(MVT::Other, MVT::Glue); 5236 SDValue Ops[] = { Chain, Dest, ARMcc, CCR, Cmp }; 5237 SDValue Res = DAG.getNode(ARMISD::BRCOND, dl, VTList, Ops); 5238 if (CondCode2 != ARMCC::AL) { 5239 ARMcc = DAG.getConstant(CondCode2, dl, MVT::i32); 5240 SDValue Ops[] = { Res, Dest, ARMcc, CCR, Res.getValue(1) }; 5241 Res = DAG.getNode(ARMISD::BRCOND, dl, VTList, Ops); 5242 } 5243 return Res; 5244 } 5245 5246 SDValue ARMTargetLowering::LowerBR_JT(SDValue Op, SelectionDAG &DAG) const { 5247 SDValue Chain = Op.getOperand(0); 5248 SDValue Table = Op.getOperand(1); 5249 SDValue Index = Op.getOperand(2); 5250 SDLoc dl(Op); 5251 5252 EVT PTy = getPointerTy(DAG.getDataLayout()); 5253 JumpTableSDNode *JT = cast<JumpTableSDNode>(Table); 5254 SDValue JTI = DAG.getTargetJumpTable(JT->getIndex(), PTy); 5255 Table = DAG.getNode(ARMISD::WrapperJT, dl, MVT::i32, JTI); 5256 Index = DAG.getNode(ISD::MUL, dl, PTy, Index, DAG.getConstant(4, dl, PTy)); 5257 SDValue Addr = DAG.getNode(ISD::ADD, dl, PTy, Table, Index); 5258 if (Subtarget->isThumb2() || (Subtarget->hasV8MBaselineOps() && Subtarget->isThumb())) { 5259 // Thumb2 and ARMv8-M use a two-level jump. That is, it jumps into the jump table 5260 // which does another jump to the destination. This also makes it easier 5261 // to translate it to TBB / TBH later (Thumb2 only). 5262 // FIXME: This might not work if the function is extremely large. 5263 return DAG.getNode(ARMISD::BR2_JT, dl, MVT::Other, Chain, 5264 Addr, Op.getOperand(2), JTI); 5265 } 5266 if (isPositionIndependent() || Subtarget->isROPI()) { 5267 Addr = 5268 DAG.getLoad((EVT)MVT::i32, dl, Chain, Addr, 5269 MachinePointerInfo::getJumpTable(DAG.getMachineFunction())); 5270 Chain = Addr.getValue(1); 5271 Addr = DAG.getNode(ISD::ADD, dl, PTy, Table, Addr); 5272 return DAG.getNode(ARMISD::BR_JT, dl, MVT::Other, Chain, Addr, JTI); 5273 } else { 5274 Addr = 5275 DAG.getLoad(PTy, dl, Chain, Addr, 5276 MachinePointerInfo::getJumpTable(DAG.getMachineFunction())); 5277 Chain = Addr.getValue(1); 5278 return DAG.getNode(ARMISD::BR_JT, dl, MVT::Other, Chain, Addr, JTI); 5279 } 5280 } 5281 5282 static SDValue LowerVectorFP_TO_INT(SDValue Op, SelectionDAG &DAG) { 5283 EVT VT = Op.getValueType(); 5284 SDLoc dl(Op); 5285 5286 if (Op.getValueType().getVectorElementType() == MVT::i32) { 5287 if (Op.getOperand(0).getValueType().getVectorElementType() == MVT::f32) 5288 return Op; 5289 return DAG.UnrollVectorOp(Op.getNode()); 5290 } 5291 5292 const bool HasFullFP16 = 5293 static_cast<const ARMSubtarget&>(DAG.getSubtarget()).hasFullFP16(); 5294 5295 EVT NewTy; 5296 const EVT OpTy = Op.getOperand(0).getValueType(); 5297 if (OpTy == MVT::v4f32) 5298 NewTy = MVT::v4i32; 5299 else if (OpTy == MVT::v4f16 && HasFullFP16) 5300 NewTy = MVT::v4i16; 5301 else if (OpTy == MVT::v8f16 && HasFullFP16) 5302 NewTy = MVT::v8i16; 5303 else 5304 llvm_unreachable("Invalid type for custom lowering!"); 5305 5306 if (VT != MVT::v4i16 && VT != MVT::v8i16) 5307 return DAG.UnrollVectorOp(Op.getNode()); 5308 5309 Op = DAG.getNode(Op.getOpcode(), dl, NewTy, Op.getOperand(0)); 5310 return DAG.getNode(ISD::TRUNCATE, dl, VT, Op); 5311 } 5312 5313 SDValue ARMTargetLowering::LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const { 5314 EVT VT = Op.getValueType(); 5315 if (VT.isVector()) 5316 return LowerVectorFP_TO_INT(Op, DAG); 5317 if (isUnsupportedFloatingType(Op.getOperand(0).getValueType())) { 5318 RTLIB::Libcall LC; 5319 if (Op.getOpcode() == ISD::FP_TO_SINT) 5320 LC = RTLIB::getFPTOSINT(Op.getOperand(0).getValueType(), 5321 Op.getValueType()); 5322 else 5323 LC = RTLIB::getFPTOUINT(Op.getOperand(0).getValueType(), 5324 Op.getValueType()); 5325 MakeLibCallOptions CallOptions; 5326 return makeLibCall(DAG, LC, Op.getValueType(), Op.getOperand(0), 5327 CallOptions, SDLoc(Op)).first; 5328 } 5329 5330 return Op; 5331 } 5332 5333 static SDValue LowerVectorINT_TO_FP(SDValue Op, SelectionDAG &DAG) { 5334 EVT VT = Op.getValueType(); 5335 SDLoc dl(Op); 5336 5337 if (Op.getOperand(0).getValueType().getVectorElementType() == MVT::i32) { 5338 if (VT.getVectorElementType() == MVT::f32) 5339 return Op; 5340 return DAG.UnrollVectorOp(Op.getNode()); 5341 } 5342 5343 assert((Op.getOperand(0).getValueType() == MVT::v4i16 || 5344 Op.getOperand(0).getValueType() == MVT::v8i16) && 5345 "Invalid type for custom lowering!"); 5346 5347 const bool HasFullFP16 = 5348 static_cast<const ARMSubtarget&>(DAG.getSubtarget()).hasFullFP16(); 5349 5350 EVT DestVecType; 5351 if (VT == MVT::v4f32) 5352 DestVecType = MVT::v4i32; 5353 else if (VT == MVT::v4f16 && HasFullFP16) 5354 DestVecType = MVT::v4i16; 5355 else if (VT == MVT::v8f16 && HasFullFP16) 5356 DestVecType = MVT::v8i16; 5357 else 5358 return DAG.UnrollVectorOp(Op.getNode()); 5359 5360 unsigned CastOpc; 5361 unsigned Opc; 5362 switch (Op.getOpcode()) { 5363 default: llvm_unreachable("Invalid opcode!"); 5364 case ISD::SINT_TO_FP: 5365 CastOpc = ISD::SIGN_EXTEND; 5366 Opc = ISD::SINT_TO_FP; 5367 break; 5368 case ISD::UINT_TO_FP: 5369 CastOpc = ISD::ZERO_EXTEND; 5370 Opc = ISD::UINT_TO_FP; 5371 break; 5372 } 5373 5374 Op = DAG.getNode(CastOpc, dl, DestVecType, Op.getOperand(0)); 5375 return DAG.getNode(Opc, dl, VT, Op); 5376 } 5377 5378 SDValue ARMTargetLowering::LowerINT_TO_FP(SDValue Op, SelectionDAG &DAG) const { 5379 EVT VT = Op.getValueType(); 5380 if (VT.isVector()) 5381 return LowerVectorINT_TO_FP(Op, DAG); 5382 if (isUnsupportedFloatingType(VT)) { 5383 RTLIB::Libcall LC; 5384 if (Op.getOpcode() == ISD::SINT_TO_FP) 5385 LC = RTLIB::getSINTTOFP(Op.getOperand(0).getValueType(), 5386 Op.getValueType()); 5387 else 5388 LC = RTLIB::getUINTTOFP(Op.getOperand(0).getValueType(), 5389 Op.getValueType()); 5390 MakeLibCallOptions CallOptions; 5391 return makeLibCall(DAG, LC, Op.getValueType(), Op.getOperand(0), 5392 CallOptions, SDLoc(Op)).first; 5393 } 5394 5395 return Op; 5396 } 5397 5398 SDValue ARMTargetLowering::LowerFCOPYSIGN(SDValue Op, SelectionDAG &DAG) const { 5399 // Implement fcopysign with a fabs and a conditional fneg. 5400 SDValue Tmp0 = Op.getOperand(0); 5401 SDValue Tmp1 = Op.getOperand(1); 5402 SDLoc dl(Op); 5403 EVT VT = Op.getValueType(); 5404 EVT SrcVT = Tmp1.getValueType(); 5405 bool InGPR = Tmp0.getOpcode() == ISD::BITCAST || 5406 Tmp0.getOpcode() == ARMISD::VMOVDRR; 5407 bool UseNEON = !InGPR && Subtarget->hasNEON(); 5408 5409 if (UseNEON) { 5410 // Use VBSL to copy the sign bit. 5411 unsigned EncodedVal = ARM_AM::createVMOVModImm(0x6, 0x80); 5412 SDValue Mask = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v2i32, 5413 DAG.getTargetConstant(EncodedVal, dl, MVT::i32)); 5414 EVT OpVT = (VT == MVT::f32) ? MVT::v2i32 : MVT::v1i64; 5415 if (VT == MVT::f64) 5416 Mask = DAG.getNode(ARMISD::VSHLIMM, dl, OpVT, 5417 DAG.getNode(ISD::BITCAST, dl, OpVT, Mask), 5418 DAG.getConstant(32, dl, MVT::i32)); 5419 else /*if (VT == MVT::f32)*/ 5420 Tmp0 = DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, MVT::v2f32, Tmp0); 5421 if (SrcVT == MVT::f32) { 5422 Tmp1 = DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, MVT::v2f32, Tmp1); 5423 if (VT == MVT::f64) 5424 Tmp1 = DAG.getNode(ARMISD::VSHLIMM, dl, OpVT, 5425 DAG.getNode(ISD::BITCAST, dl, OpVT, Tmp1), 5426 DAG.getConstant(32, dl, MVT::i32)); 5427 } else if (VT == MVT::f32) 5428 Tmp1 = DAG.getNode(ARMISD::VSHRuIMM, dl, MVT::v1i64, 5429 DAG.getNode(ISD::BITCAST, dl, MVT::v1i64, Tmp1), 5430 DAG.getConstant(32, dl, MVT::i32)); 5431 Tmp0 = DAG.getNode(ISD::BITCAST, dl, OpVT, Tmp0); 5432 Tmp1 = DAG.getNode(ISD::BITCAST, dl, OpVT, Tmp1); 5433 5434 SDValue AllOnes = DAG.getTargetConstant(ARM_AM::createVMOVModImm(0xe, 0xff), 5435 dl, MVT::i32); 5436 AllOnes = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v8i8, AllOnes); 5437 SDValue MaskNot = DAG.getNode(ISD::XOR, dl, OpVT, Mask, 5438 DAG.getNode(ISD::BITCAST, dl, OpVT, AllOnes)); 5439 5440 SDValue Res = DAG.getNode(ISD::OR, dl, OpVT, 5441 DAG.getNode(ISD::AND, dl, OpVT, Tmp1, Mask), 5442 DAG.getNode(ISD::AND, dl, OpVT, Tmp0, MaskNot)); 5443 if (VT == MVT::f32) { 5444 Res = DAG.getNode(ISD::BITCAST, dl, MVT::v2f32, Res); 5445 Res = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f32, Res, 5446 DAG.getConstant(0, dl, MVT::i32)); 5447 } else { 5448 Res = DAG.getNode(ISD::BITCAST, dl, MVT::f64, Res); 5449 } 5450 5451 return Res; 5452 } 5453 5454 // Bitcast operand 1 to i32. 5455 if (SrcVT == MVT::f64) 5456 Tmp1 = DAG.getNode(ARMISD::VMOVRRD, dl, DAG.getVTList(MVT::i32, MVT::i32), 5457 Tmp1).getValue(1); 5458 Tmp1 = DAG.getNode(ISD::BITCAST, dl, MVT::i32, Tmp1); 5459 5460 // Or in the signbit with integer operations. 5461 SDValue Mask1 = DAG.getConstant(0x80000000, dl, MVT::i32); 5462 SDValue Mask2 = DAG.getConstant(0x7fffffff, dl, MVT::i32); 5463 Tmp1 = DAG.getNode(ISD::AND, dl, MVT::i32, Tmp1, Mask1); 5464 if (VT == MVT::f32) { 5465 Tmp0 = DAG.getNode(ISD::AND, dl, MVT::i32, 5466 DAG.getNode(ISD::BITCAST, dl, MVT::i32, Tmp0), Mask2); 5467 return DAG.getNode(ISD::BITCAST, dl, MVT::f32, 5468 DAG.getNode(ISD::OR, dl, MVT::i32, Tmp0, Tmp1)); 5469 } 5470 5471 // f64: Or the high part with signbit and then combine two parts. 5472 Tmp0 = DAG.getNode(ARMISD::VMOVRRD, dl, DAG.getVTList(MVT::i32, MVT::i32), 5473 Tmp0); 5474 SDValue Lo = Tmp0.getValue(0); 5475 SDValue Hi = DAG.getNode(ISD::AND, dl, MVT::i32, Tmp0.getValue(1), Mask2); 5476 Hi = DAG.getNode(ISD::OR, dl, MVT::i32, Hi, Tmp1); 5477 return DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi); 5478 } 5479 5480 SDValue ARMTargetLowering::LowerRETURNADDR(SDValue Op, SelectionDAG &DAG) const{ 5481 MachineFunction &MF = DAG.getMachineFunction(); 5482 MachineFrameInfo &MFI = MF.getFrameInfo(); 5483 MFI.setReturnAddressIsTaken(true); 5484 5485 if (verifyReturnAddressArgumentIsConstant(Op, DAG)) 5486 return SDValue(); 5487 5488 EVT VT = Op.getValueType(); 5489 SDLoc dl(Op); 5490 unsigned Depth = cast<ConstantSDNode>(Op.getOperand(0))->getZExtValue(); 5491 if (Depth) { 5492 SDValue FrameAddr = LowerFRAMEADDR(Op, DAG); 5493 SDValue Offset = DAG.getConstant(4, dl, MVT::i32); 5494 return DAG.getLoad(VT, dl, DAG.getEntryNode(), 5495 DAG.getNode(ISD::ADD, dl, VT, FrameAddr, Offset), 5496 MachinePointerInfo()); 5497 } 5498 5499 // Return LR, which contains the return address. Mark it an implicit live-in. 5500 unsigned Reg = MF.addLiveIn(ARM::LR, getRegClassFor(MVT::i32)); 5501 return DAG.getCopyFromReg(DAG.getEntryNode(), dl, Reg, VT); 5502 } 5503 5504 SDValue ARMTargetLowering::LowerFRAMEADDR(SDValue Op, SelectionDAG &DAG) const { 5505 const ARMBaseRegisterInfo &ARI = 5506 *static_cast<const ARMBaseRegisterInfo*>(RegInfo); 5507 MachineFunction &MF = DAG.getMachineFunction(); 5508 MachineFrameInfo &MFI = MF.getFrameInfo(); 5509 MFI.setFrameAddressIsTaken(true); 5510 5511 EVT VT = Op.getValueType(); 5512 SDLoc dl(Op); // FIXME probably not meaningful 5513 unsigned Depth = cast<ConstantSDNode>(Op.getOperand(0))->getZExtValue(); 5514 Register FrameReg = ARI.getFrameRegister(MF); 5515 SDValue FrameAddr = DAG.getCopyFromReg(DAG.getEntryNode(), dl, FrameReg, VT); 5516 while (Depth--) 5517 FrameAddr = DAG.getLoad(VT, dl, DAG.getEntryNode(), FrameAddr, 5518 MachinePointerInfo()); 5519 return FrameAddr; 5520 } 5521 5522 // FIXME? Maybe this could be a TableGen attribute on some registers and 5523 // this table could be generated automatically from RegInfo. 5524 Register ARMTargetLowering::getRegisterByName(const char* RegName, EVT VT, 5525 const MachineFunction &MF) const { 5526 Register Reg = StringSwitch<unsigned>(RegName) 5527 .Case("sp", ARM::SP) 5528 .Default(0); 5529 if (Reg) 5530 return Reg; 5531 report_fatal_error(Twine("Invalid register name \"" 5532 + StringRef(RegName) + "\".")); 5533 } 5534 5535 // Result is 64 bit value so split into two 32 bit values and return as a 5536 // pair of values. 5537 static void ExpandREAD_REGISTER(SDNode *N, SmallVectorImpl<SDValue> &Results, 5538 SelectionDAG &DAG) { 5539 SDLoc DL(N); 5540 5541 // This function is only supposed to be called for i64 type destination. 5542 assert(N->getValueType(0) == MVT::i64 5543 && "ExpandREAD_REGISTER called for non-i64 type result."); 5544 5545 SDValue Read = DAG.getNode(ISD::READ_REGISTER, DL, 5546 DAG.getVTList(MVT::i32, MVT::i32, MVT::Other), 5547 N->getOperand(0), 5548 N->getOperand(1)); 5549 5550 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, Read.getValue(0), 5551 Read.getValue(1))); 5552 Results.push_back(Read.getOperand(0)); 5553 } 5554 5555 /// \p BC is a bitcast that is about to be turned into a VMOVDRR. 5556 /// When \p DstVT, the destination type of \p BC, is on the vector 5557 /// register bank and the source of bitcast, \p Op, operates on the same bank, 5558 /// it might be possible to combine them, such that everything stays on the 5559 /// vector register bank. 5560 /// \p return The node that would replace \p BT, if the combine 5561 /// is possible. 5562 static SDValue CombineVMOVDRRCandidateWithVecOp(const SDNode *BC, 5563 SelectionDAG &DAG) { 5564 SDValue Op = BC->getOperand(0); 5565 EVT DstVT = BC->getValueType(0); 5566 5567 // The only vector instruction that can produce a scalar (remember, 5568 // since the bitcast was about to be turned into VMOVDRR, the source 5569 // type is i64) from a vector is EXTRACT_VECTOR_ELT. 5570 // Moreover, we can do this combine only if there is one use. 5571 // Finally, if the destination type is not a vector, there is not 5572 // much point on forcing everything on the vector bank. 5573 if (!DstVT.isVector() || Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT || 5574 !Op.hasOneUse()) 5575 return SDValue(); 5576 5577 // If the index is not constant, we will introduce an additional 5578 // multiply that will stick. 5579 // Give up in that case. 5580 ConstantSDNode *Index = dyn_cast<ConstantSDNode>(Op.getOperand(1)); 5581 if (!Index) 5582 return SDValue(); 5583 unsigned DstNumElt = DstVT.getVectorNumElements(); 5584 5585 // Compute the new index. 5586 const APInt &APIntIndex = Index->getAPIntValue(); 5587 APInt NewIndex(APIntIndex.getBitWidth(), DstNumElt); 5588 NewIndex *= APIntIndex; 5589 // Check if the new constant index fits into i32. 5590 if (NewIndex.getBitWidth() > 32) 5591 return SDValue(); 5592 5593 // vMTy bitcast(i64 extractelt vNi64 src, i32 index) -> 5594 // vMTy extractsubvector vNxMTy (bitcast vNi64 src), i32 index*M) 5595 SDLoc dl(Op); 5596 SDValue ExtractSrc = Op.getOperand(0); 5597 EVT VecVT = EVT::getVectorVT( 5598 *DAG.getContext(), DstVT.getScalarType(), 5599 ExtractSrc.getValueType().getVectorNumElements() * DstNumElt); 5600 SDValue BitCast = DAG.getNode(ISD::BITCAST, dl, VecVT, ExtractSrc); 5601 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DstVT, BitCast, 5602 DAG.getConstant(NewIndex.getZExtValue(), dl, MVT::i32)); 5603 } 5604 5605 /// ExpandBITCAST - If the target supports VFP, this function is called to 5606 /// expand a bit convert where either the source or destination type is i64 to 5607 /// use a VMOVDRR or VMOVRRD node. This should not be done when the non-i64 5608 /// operand type is illegal (e.g., v2f32 for a target that doesn't support 5609 /// vectors), since the legalizer won't know what to do with that. 5610 static SDValue ExpandBITCAST(SDNode *N, SelectionDAG &DAG, 5611 const ARMSubtarget *Subtarget) { 5612 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 5613 SDLoc dl(N); 5614 SDValue Op = N->getOperand(0); 5615 5616 // This function is only supposed to be called for i64 types, either as the 5617 // source or destination of the bit convert. 5618 EVT SrcVT = Op.getValueType(); 5619 EVT DstVT = N->getValueType(0); 5620 const bool HasFullFP16 = Subtarget->hasFullFP16(); 5621 5622 if (SrcVT == MVT::f32 && DstVT == MVT::i32) { 5623 // FullFP16: half values are passed in S-registers, and we don't 5624 // need any of the bitcast and moves: 5625 // 5626 // t2: f32,ch = CopyFromReg t0, Register:f32 %0 5627 // t5: i32 = bitcast t2 5628 // t18: f16 = ARMISD::VMOVhr t5 5629 if (Op.getOpcode() != ISD::CopyFromReg || 5630 Op.getValueType() != MVT::f32) 5631 return SDValue(); 5632 5633 auto Move = N->use_begin(); 5634 if (Move->getOpcode() != ARMISD::VMOVhr) 5635 return SDValue(); 5636 5637 SDValue Ops[] = { Op.getOperand(0), Op.getOperand(1) }; 5638 SDValue Copy = DAG.getNode(ISD::CopyFromReg, SDLoc(Op), MVT::f16, Ops); 5639 DAG.ReplaceAllUsesWith(*Move, &Copy); 5640 return Copy; 5641 } 5642 5643 if (SrcVT == MVT::i16 && DstVT == MVT::f16) { 5644 if (!HasFullFP16) 5645 return SDValue(); 5646 // SoftFP: read half-precision arguments: 5647 // 5648 // t2: i32,ch = ... 5649 // t7: i16 = truncate t2 <~~~~ Op 5650 // t8: f16 = bitcast t7 <~~~~ N 5651 // 5652 if (Op.getOperand(0).getValueType() == MVT::i32) 5653 return DAG.getNode(ARMISD::VMOVhr, SDLoc(Op), 5654 MVT::f16, Op.getOperand(0)); 5655 5656 return SDValue(); 5657 } 5658 5659 // Half-precision return values 5660 if (SrcVT == MVT::f16 && DstVT == MVT::i16) { 5661 if (!HasFullFP16) 5662 return SDValue(); 5663 // 5664 // t11: f16 = fadd t8, t10 5665 // t12: i16 = bitcast t11 <~~~ SDNode N 5666 // t13: i32 = zero_extend t12 5667 // t16: ch,glue = CopyToReg t0, Register:i32 %r0, t13 5668 // t17: ch = ARMISD::RET_FLAG t16, Register:i32 %r0, t16:1 5669 // 5670 // transform this into: 5671 // 5672 // t20: i32 = ARMISD::VMOVrh t11 5673 // t16: ch,glue = CopyToReg t0, Register:i32 %r0, t20 5674 // 5675 auto ZeroExtend = N->use_begin(); 5676 if (N->use_size() != 1 || ZeroExtend->getOpcode() != ISD::ZERO_EXTEND || 5677 ZeroExtend->getValueType(0) != MVT::i32) 5678 return SDValue(); 5679 5680 auto Copy = ZeroExtend->use_begin(); 5681 if (Copy->getOpcode() == ISD::CopyToReg && 5682 Copy->use_begin()->getOpcode() == ARMISD::RET_FLAG) { 5683 SDValue Cvt = DAG.getNode(ARMISD::VMOVrh, SDLoc(Op), MVT::i32, Op); 5684 DAG.ReplaceAllUsesWith(*ZeroExtend, &Cvt); 5685 return Cvt; 5686 } 5687 return SDValue(); 5688 } 5689 5690 if (!(SrcVT == MVT::i64 || DstVT == MVT::i64)) 5691 return SDValue(); 5692 5693 // Turn i64->f64 into VMOVDRR. 5694 if (SrcVT == MVT::i64 && TLI.isTypeLegal(DstVT)) { 5695 // Do not force values to GPRs (this is what VMOVDRR does for the inputs) 5696 // if we can combine the bitcast with its source. 5697 if (SDValue Val = CombineVMOVDRRCandidateWithVecOp(N, DAG)) 5698 return Val; 5699 5700 SDValue Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, Op, 5701 DAG.getConstant(0, dl, MVT::i32)); 5702 SDValue Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, Op, 5703 DAG.getConstant(1, dl, MVT::i32)); 5704 return DAG.getNode(ISD::BITCAST, dl, DstVT, 5705 DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi)); 5706 } 5707 5708 // Turn f64->i64 into VMOVRRD. 5709 if (DstVT == MVT::i64 && TLI.isTypeLegal(SrcVT)) { 5710 SDValue Cvt; 5711 if (DAG.getDataLayout().isBigEndian() && SrcVT.isVector() && 5712 SrcVT.getVectorNumElements() > 1) 5713 Cvt = DAG.getNode(ARMISD::VMOVRRD, dl, 5714 DAG.getVTList(MVT::i32, MVT::i32), 5715 DAG.getNode(ARMISD::VREV64, dl, SrcVT, Op)); 5716 else 5717 Cvt = DAG.getNode(ARMISD::VMOVRRD, dl, 5718 DAG.getVTList(MVT::i32, MVT::i32), Op); 5719 // Merge the pieces into a single i64 value. 5720 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Cvt, Cvt.getValue(1)); 5721 } 5722 5723 return SDValue(); 5724 } 5725 5726 /// getZeroVector - Returns a vector of specified type with all zero elements. 5727 /// Zero vectors are used to represent vector negation and in those cases 5728 /// will be implemented with the NEON VNEG instruction. However, VNEG does 5729 /// not support i64 elements, so sometimes the zero vectors will need to be 5730 /// explicitly constructed. Regardless, use a canonical VMOV to create the 5731 /// zero vector. 5732 static SDValue getZeroVector(EVT VT, SelectionDAG &DAG, const SDLoc &dl) { 5733 assert(VT.isVector() && "Expected a vector type"); 5734 // The canonical modified immediate encoding of a zero vector is....0! 5735 SDValue EncodedVal = DAG.getTargetConstant(0, dl, MVT::i32); 5736 EVT VmovVT = VT.is128BitVector() ? MVT::v4i32 : MVT::v2i32; 5737 SDValue Vmov = DAG.getNode(ARMISD::VMOVIMM, dl, VmovVT, EncodedVal); 5738 return DAG.getNode(ISD::BITCAST, dl, VT, Vmov); 5739 } 5740 5741 /// LowerShiftRightParts - Lower SRA_PARTS, which returns two 5742 /// i32 values and take a 2 x i32 value to shift plus a shift amount. 5743 SDValue ARMTargetLowering::LowerShiftRightParts(SDValue Op, 5744 SelectionDAG &DAG) const { 5745 assert(Op.getNumOperands() == 3 && "Not a double-shift!"); 5746 EVT VT = Op.getValueType(); 5747 unsigned VTBits = VT.getSizeInBits(); 5748 SDLoc dl(Op); 5749 SDValue ShOpLo = Op.getOperand(0); 5750 SDValue ShOpHi = Op.getOperand(1); 5751 SDValue ShAmt = Op.getOperand(2); 5752 SDValue ARMcc; 5753 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 5754 unsigned Opc = (Op.getOpcode() == ISD::SRA_PARTS) ? ISD::SRA : ISD::SRL; 5755 5756 assert(Op.getOpcode() == ISD::SRA_PARTS || Op.getOpcode() == ISD::SRL_PARTS); 5757 5758 SDValue RevShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32, 5759 DAG.getConstant(VTBits, dl, MVT::i32), ShAmt); 5760 SDValue Tmp1 = DAG.getNode(ISD::SRL, dl, VT, ShOpLo, ShAmt); 5761 SDValue ExtraShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32, ShAmt, 5762 DAG.getConstant(VTBits, dl, MVT::i32)); 5763 SDValue Tmp2 = DAG.getNode(ISD::SHL, dl, VT, ShOpHi, RevShAmt); 5764 SDValue LoSmallShift = DAG.getNode(ISD::OR, dl, VT, Tmp1, Tmp2); 5765 SDValue LoBigShift = DAG.getNode(Opc, dl, VT, ShOpHi, ExtraShAmt); 5766 SDValue CmpLo = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32), 5767 ISD::SETGE, ARMcc, DAG, dl); 5768 SDValue Lo = DAG.getNode(ARMISD::CMOV, dl, VT, LoSmallShift, LoBigShift, 5769 ARMcc, CCR, CmpLo); 5770 5771 SDValue HiSmallShift = DAG.getNode(Opc, dl, VT, ShOpHi, ShAmt); 5772 SDValue HiBigShift = Opc == ISD::SRA 5773 ? DAG.getNode(Opc, dl, VT, ShOpHi, 5774 DAG.getConstant(VTBits - 1, dl, VT)) 5775 : DAG.getConstant(0, dl, VT); 5776 SDValue CmpHi = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32), 5777 ISD::SETGE, ARMcc, DAG, dl); 5778 SDValue Hi = DAG.getNode(ARMISD::CMOV, dl, VT, HiSmallShift, HiBigShift, 5779 ARMcc, CCR, CmpHi); 5780 5781 SDValue Ops[2] = { Lo, Hi }; 5782 return DAG.getMergeValues(Ops, dl); 5783 } 5784 5785 /// LowerShiftLeftParts - Lower SHL_PARTS, which returns two 5786 /// i32 values and take a 2 x i32 value to shift plus a shift amount. 5787 SDValue ARMTargetLowering::LowerShiftLeftParts(SDValue Op, 5788 SelectionDAG &DAG) const { 5789 assert(Op.getNumOperands() == 3 && "Not a double-shift!"); 5790 EVT VT = Op.getValueType(); 5791 unsigned VTBits = VT.getSizeInBits(); 5792 SDLoc dl(Op); 5793 SDValue ShOpLo = Op.getOperand(0); 5794 SDValue ShOpHi = Op.getOperand(1); 5795 SDValue ShAmt = Op.getOperand(2); 5796 SDValue ARMcc; 5797 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 5798 5799 assert(Op.getOpcode() == ISD::SHL_PARTS); 5800 SDValue RevShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32, 5801 DAG.getConstant(VTBits, dl, MVT::i32), ShAmt); 5802 SDValue Tmp1 = DAG.getNode(ISD::SRL, dl, VT, ShOpLo, RevShAmt); 5803 SDValue Tmp2 = DAG.getNode(ISD::SHL, dl, VT, ShOpHi, ShAmt); 5804 SDValue HiSmallShift = DAG.getNode(ISD::OR, dl, VT, Tmp1, Tmp2); 5805 5806 SDValue ExtraShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32, ShAmt, 5807 DAG.getConstant(VTBits, dl, MVT::i32)); 5808 SDValue HiBigShift = DAG.getNode(ISD::SHL, dl, VT, ShOpLo, ExtraShAmt); 5809 SDValue CmpHi = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32), 5810 ISD::SETGE, ARMcc, DAG, dl); 5811 SDValue Hi = DAG.getNode(ARMISD::CMOV, dl, VT, HiSmallShift, HiBigShift, 5812 ARMcc, CCR, CmpHi); 5813 5814 SDValue CmpLo = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32), 5815 ISD::SETGE, ARMcc, DAG, dl); 5816 SDValue LoSmallShift = DAG.getNode(ISD::SHL, dl, VT, ShOpLo, ShAmt); 5817 SDValue Lo = DAG.getNode(ARMISD::CMOV, dl, VT, LoSmallShift, 5818 DAG.getConstant(0, dl, VT), ARMcc, CCR, CmpLo); 5819 5820 SDValue Ops[2] = { Lo, Hi }; 5821 return DAG.getMergeValues(Ops, dl); 5822 } 5823 5824 SDValue ARMTargetLowering::LowerFLT_ROUNDS_(SDValue Op, 5825 SelectionDAG &DAG) const { 5826 // The rounding mode is in bits 23:22 of the FPSCR. 5827 // The ARM rounding mode value to FLT_ROUNDS mapping is 0->1, 1->2, 2->3, 3->0 5828 // The formula we use to implement this is (((FPSCR + 1 << 22) >> 22) & 3) 5829 // so that the shift + and get folded into a bitfield extract. 5830 SDLoc dl(Op); 5831 SDValue Ops[] = { DAG.getEntryNode(), 5832 DAG.getConstant(Intrinsic::arm_get_fpscr, dl, MVT::i32) }; 5833 5834 SDValue FPSCR = DAG.getNode(ISD::INTRINSIC_W_CHAIN, dl, MVT::i32, Ops); 5835 SDValue FltRounds = DAG.getNode(ISD::ADD, dl, MVT::i32, FPSCR, 5836 DAG.getConstant(1U << 22, dl, MVT::i32)); 5837 SDValue RMODE = DAG.getNode(ISD::SRL, dl, MVT::i32, FltRounds, 5838 DAG.getConstant(22, dl, MVT::i32)); 5839 return DAG.getNode(ISD::AND, dl, MVT::i32, RMODE, 5840 DAG.getConstant(3, dl, MVT::i32)); 5841 } 5842 5843 static SDValue LowerCTTZ(SDNode *N, SelectionDAG &DAG, 5844 const ARMSubtarget *ST) { 5845 SDLoc dl(N); 5846 EVT VT = N->getValueType(0); 5847 if (VT.isVector() && ST->hasNEON()) { 5848 5849 // Compute the least significant set bit: LSB = X & -X 5850 SDValue X = N->getOperand(0); 5851 SDValue NX = DAG.getNode(ISD::SUB, dl, VT, getZeroVector(VT, DAG, dl), X); 5852 SDValue LSB = DAG.getNode(ISD::AND, dl, VT, X, NX); 5853 5854 EVT ElemTy = VT.getVectorElementType(); 5855 5856 if (ElemTy == MVT::i8) { 5857 // Compute with: cttz(x) = ctpop(lsb - 1) 5858 SDValue One = DAG.getNode(ARMISD::VMOVIMM, dl, VT, 5859 DAG.getTargetConstant(1, dl, ElemTy)); 5860 SDValue Bits = DAG.getNode(ISD::SUB, dl, VT, LSB, One); 5861 return DAG.getNode(ISD::CTPOP, dl, VT, Bits); 5862 } 5863 5864 if ((ElemTy == MVT::i16 || ElemTy == MVT::i32) && 5865 (N->getOpcode() == ISD::CTTZ_ZERO_UNDEF)) { 5866 // Compute with: cttz(x) = (width - 1) - ctlz(lsb), if x != 0 5867 unsigned NumBits = ElemTy.getSizeInBits(); 5868 SDValue WidthMinus1 = 5869 DAG.getNode(ARMISD::VMOVIMM, dl, VT, 5870 DAG.getTargetConstant(NumBits - 1, dl, ElemTy)); 5871 SDValue CTLZ = DAG.getNode(ISD::CTLZ, dl, VT, LSB); 5872 return DAG.getNode(ISD::SUB, dl, VT, WidthMinus1, CTLZ); 5873 } 5874 5875 // Compute with: cttz(x) = ctpop(lsb - 1) 5876 5877 // Compute LSB - 1. 5878 SDValue Bits; 5879 if (ElemTy == MVT::i64) { 5880 // Load constant 0xffff'ffff'ffff'ffff to register. 5881 SDValue FF = DAG.getNode(ARMISD::VMOVIMM, dl, VT, 5882 DAG.getTargetConstant(0x1eff, dl, MVT::i32)); 5883 Bits = DAG.getNode(ISD::ADD, dl, VT, LSB, FF); 5884 } else { 5885 SDValue One = DAG.getNode(ARMISD::VMOVIMM, dl, VT, 5886 DAG.getTargetConstant(1, dl, ElemTy)); 5887 Bits = DAG.getNode(ISD::SUB, dl, VT, LSB, One); 5888 } 5889 return DAG.getNode(ISD::CTPOP, dl, VT, Bits); 5890 } 5891 5892 if (!ST->hasV6T2Ops()) 5893 return SDValue(); 5894 5895 SDValue rbit = DAG.getNode(ISD::BITREVERSE, dl, VT, N->getOperand(0)); 5896 return DAG.getNode(ISD::CTLZ, dl, VT, rbit); 5897 } 5898 5899 static SDValue LowerCTPOP(SDNode *N, SelectionDAG &DAG, 5900 const ARMSubtarget *ST) { 5901 EVT VT = N->getValueType(0); 5902 SDLoc DL(N); 5903 5904 assert(ST->hasNEON() && "Custom ctpop lowering requires NEON."); 5905 assert((VT == MVT::v1i64 || VT == MVT::v2i64 || VT == MVT::v2i32 || 5906 VT == MVT::v4i32 || VT == MVT::v4i16 || VT == MVT::v8i16) && 5907 "Unexpected type for custom ctpop lowering"); 5908 5909 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 5910 EVT VT8Bit = VT.is64BitVector() ? MVT::v8i8 : MVT::v16i8; 5911 SDValue Res = DAG.getBitcast(VT8Bit, N->getOperand(0)); 5912 Res = DAG.getNode(ISD::CTPOP, DL, VT8Bit, Res); 5913 5914 // Widen v8i8/v16i8 CTPOP result to VT by repeatedly widening pairwise adds. 5915 unsigned EltSize = 8; 5916 unsigned NumElts = VT.is64BitVector() ? 8 : 16; 5917 while (EltSize != VT.getScalarSizeInBits()) { 5918 SmallVector<SDValue, 8> Ops; 5919 Ops.push_back(DAG.getConstant(Intrinsic::arm_neon_vpaddlu, DL, 5920 TLI.getPointerTy(DAG.getDataLayout()))); 5921 Ops.push_back(Res); 5922 5923 EltSize *= 2; 5924 NumElts /= 2; 5925 MVT WidenVT = MVT::getVectorVT(MVT::getIntegerVT(EltSize), NumElts); 5926 Res = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, WidenVT, Ops); 5927 } 5928 5929 return Res; 5930 } 5931 5932 /// Getvshiftimm - Check if this is a valid build_vector for the immediate 5933 /// operand of a vector shift operation, where all the elements of the 5934 /// build_vector must have the same constant integer value. 5935 static bool getVShiftImm(SDValue Op, unsigned ElementBits, int64_t &Cnt) { 5936 // Ignore bit_converts. 5937 while (Op.getOpcode() == ISD::BITCAST) 5938 Op = Op.getOperand(0); 5939 BuildVectorSDNode *BVN = dyn_cast<BuildVectorSDNode>(Op.getNode()); 5940 APInt SplatBits, SplatUndef; 5941 unsigned SplatBitSize; 5942 bool HasAnyUndefs; 5943 if (!BVN || 5944 !BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs, 5945 ElementBits) || 5946 SplatBitSize > ElementBits) 5947 return false; 5948 Cnt = SplatBits.getSExtValue(); 5949 return true; 5950 } 5951 5952 /// isVShiftLImm - Check if this is a valid build_vector for the immediate 5953 /// operand of a vector shift left operation. That value must be in the range: 5954 /// 0 <= Value < ElementBits for a left shift; or 5955 /// 0 <= Value <= ElementBits for a long left shift. 5956 static bool isVShiftLImm(SDValue Op, EVT VT, bool isLong, int64_t &Cnt) { 5957 assert(VT.isVector() && "vector shift count is not a vector type"); 5958 int64_t ElementBits = VT.getScalarSizeInBits(); 5959 if (!getVShiftImm(Op, ElementBits, Cnt)) 5960 return false; 5961 return (Cnt >= 0 && (isLong ? Cnt - 1 : Cnt) < ElementBits); 5962 } 5963 5964 /// isVShiftRImm - Check if this is a valid build_vector for the immediate 5965 /// operand of a vector shift right operation. For a shift opcode, the value 5966 /// is positive, but for an intrinsic the value count must be negative. The 5967 /// absolute value must be in the range: 5968 /// 1 <= |Value| <= ElementBits for a right shift; or 5969 /// 1 <= |Value| <= ElementBits/2 for a narrow right shift. 5970 static bool isVShiftRImm(SDValue Op, EVT VT, bool isNarrow, bool isIntrinsic, 5971 int64_t &Cnt) { 5972 assert(VT.isVector() && "vector shift count is not a vector type"); 5973 int64_t ElementBits = VT.getScalarSizeInBits(); 5974 if (!getVShiftImm(Op, ElementBits, Cnt)) 5975 return false; 5976 if (!isIntrinsic) 5977 return (Cnt >= 1 && Cnt <= (isNarrow ? ElementBits / 2 : ElementBits)); 5978 if (Cnt >= -(isNarrow ? ElementBits / 2 : ElementBits) && Cnt <= -1) { 5979 Cnt = -Cnt; 5980 return true; 5981 } 5982 return false; 5983 } 5984 5985 static SDValue LowerShift(SDNode *N, SelectionDAG &DAG, 5986 const ARMSubtarget *ST) { 5987 EVT VT = N->getValueType(0); 5988 SDLoc dl(N); 5989 int64_t Cnt; 5990 5991 if (!VT.isVector()) 5992 return SDValue(); 5993 5994 // We essentially have two forms here. Shift by an immediate and shift by a 5995 // vector register (there are also shift by a gpr, but that is just handled 5996 // with a tablegen pattern). We cannot easily match shift by an immediate in 5997 // tablegen so we do that here and generate a VSHLIMM/VSHRsIMM/VSHRuIMM. 5998 // For shifting by a vector, we don't have VSHR, only VSHL (which can be 5999 // signed or unsigned, and a negative shift indicates a shift right). 6000 if (N->getOpcode() == ISD::SHL) { 6001 if (isVShiftLImm(N->getOperand(1), VT, false, Cnt)) 6002 return DAG.getNode(ARMISD::VSHLIMM, dl, VT, N->getOperand(0), 6003 DAG.getConstant(Cnt, dl, MVT::i32)); 6004 return DAG.getNode(ARMISD::VSHLu, dl, VT, N->getOperand(0), 6005 N->getOperand(1)); 6006 } 6007 6008 assert((N->getOpcode() == ISD::SRA || N->getOpcode() == ISD::SRL) && 6009 "unexpected vector shift opcode"); 6010 6011 if (isVShiftRImm(N->getOperand(1), VT, false, false, Cnt)) { 6012 unsigned VShiftOpc = 6013 (N->getOpcode() == ISD::SRA ? ARMISD::VSHRsIMM : ARMISD::VSHRuIMM); 6014 return DAG.getNode(VShiftOpc, dl, VT, N->getOperand(0), 6015 DAG.getConstant(Cnt, dl, MVT::i32)); 6016 } 6017 6018 // Other right shifts we don't have operations for (we use a shift left by a 6019 // negative number). 6020 EVT ShiftVT = N->getOperand(1).getValueType(); 6021 SDValue NegatedCount = DAG.getNode( 6022 ISD::SUB, dl, ShiftVT, getZeroVector(ShiftVT, DAG, dl), N->getOperand(1)); 6023 unsigned VShiftOpc = 6024 (N->getOpcode() == ISD::SRA ? ARMISD::VSHLs : ARMISD::VSHLu); 6025 return DAG.getNode(VShiftOpc, dl, VT, N->getOperand(0), NegatedCount); 6026 } 6027 6028 static SDValue Expand64BitShift(SDNode *N, SelectionDAG &DAG, 6029 const ARMSubtarget *ST) { 6030 EVT VT = N->getValueType(0); 6031 SDLoc dl(N); 6032 6033 // We can get here for a node like i32 = ISD::SHL i32, i64 6034 if (VT != MVT::i64) 6035 return SDValue(); 6036 6037 assert((N->getOpcode() == ISD::SRL || N->getOpcode() == ISD::SRA || 6038 N->getOpcode() == ISD::SHL) && 6039 "Unknown shift to lower!"); 6040 6041 unsigned ShOpc = N->getOpcode(); 6042 if (ST->hasMVEIntegerOps()) { 6043 SDValue ShAmt = N->getOperand(1); 6044 unsigned ShPartsOpc = ARMISD::LSLL; 6045 ConstantSDNode *Con = dyn_cast<ConstantSDNode>(ShAmt); 6046 6047 // If the shift amount is greater than 32 or has a greater bitwidth than 64 6048 // then do the default optimisation 6049 if (ShAmt->getValueType(0).getSizeInBits() > 64 || 6050 (Con && (Con->getZExtValue() == 0 || Con->getZExtValue() >= 32))) 6051 return SDValue(); 6052 6053 // Extract the lower 32 bits of the shift amount if it's not an i32 6054 if (ShAmt->getValueType(0) != MVT::i32) 6055 ShAmt = DAG.getZExtOrTrunc(ShAmt, dl, MVT::i32); 6056 6057 if (ShOpc == ISD::SRL) { 6058 if (!Con) 6059 // There is no t2LSRLr instruction so negate and perform an lsll if the 6060 // shift amount is in a register, emulating a right shift. 6061 ShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32, 6062 DAG.getConstant(0, dl, MVT::i32), ShAmt); 6063 else 6064 // Else generate an lsrl on the immediate shift amount 6065 ShPartsOpc = ARMISD::LSRL; 6066 } else if (ShOpc == ISD::SRA) 6067 ShPartsOpc = ARMISD::ASRL; 6068 6069 // Lower 32 bits of the destination/source 6070 SDValue Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, N->getOperand(0), 6071 DAG.getConstant(0, dl, MVT::i32)); 6072 // Upper 32 bits of the destination/source 6073 SDValue Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, N->getOperand(0), 6074 DAG.getConstant(1, dl, MVT::i32)); 6075 6076 // Generate the shift operation as computed above 6077 Lo = DAG.getNode(ShPartsOpc, dl, DAG.getVTList(MVT::i32, MVT::i32), Lo, Hi, 6078 ShAmt); 6079 // The upper 32 bits come from the second return value of lsll 6080 Hi = SDValue(Lo.getNode(), 1); 6081 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Lo, Hi); 6082 } 6083 6084 // We only lower SRA, SRL of 1 here, all others use generic lowering. 6085 if (!isOneConstant(N->getOperand(1)) || N->getOpcode() == ISD::SHL) 6086 return SDValue(); 6087 6088 // If we are in thumb mode, we don't have RRX. 6089 if (ST->isThumb1Only()) 6090 return SDValue(); 6091 6092 // Okay, we have a 64-bit SRA or SRL of 1. Lower this to an RRX expr. 6093 SDValue Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, N->getOperand(0), 6094 DAG.getConstant(0, dl, MVT::i32)); 6095 SDValue Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, N->getOperand(0), 6096 DAG.getConstant(1, dl, MVT::i32)); 6097 6098 // First, build a SRA_FLAG/SRL_FLAG op, which shifts the top part by one and 6099 // captures the result into a carry flag. 6100 unsigned Opc = N->getOpcode() == ISD::SRL ? ARMISD::SRL_FLAG:ARMISD::SRA_FLAG; 6101 Hi = DAG.getNode(Opc, dl, DAG.getVTList(MVT::i32, MVT::Glue), Hi); 6102 6103 // The low part is an ARMISD::RRX operand, which shifts the carry in. 6104 Lo = DAG.getNode(ARMISD::RRX, dl, MVT::i32, Lo, Hi.getValue(1)); 6105 6106 // Merge the pieces into a single i64 value. 6107 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Lo, Hi); 6108 } 6109 6110 static SDValue LowerVSETCC(SDValue Op, SelectionDAG &DAG, 6111 const ARMSubtarget *ST) { 6112 bool Invert = false; 6113 bool Swap = false; 6114 unsigned Opc = ARMCC::AL; 6115 6116 SDValue Op0 = Op.getOperand(0); 6117 SDValue Op1 = Op.getOperand(1); 6118 SDValue CC = Op.getOperand(2); 6119 EVT VT = Op.getValueType(); 6120 ISD::CondCode SetCCOpcode = cast<CondCodeSDNode>(CC)->get(); 6121 SDLoc dl(Op); 6122 6123 EVT CmpVT; 6124 if (ST->hasNEON()) 6125 CmpVT = Op0.getValueType().changeVectorElementTypeToInteger(); 6126 else { 6127 assert(ST->hasMVEIntegerOps() && 6128 "No hardware support for integer vector comparison!"); 6129 6130 if (Op.getValueType().getVectorElementType() != MVT::i1) 6131 return SDValue(); 6132 6133 // Make sure we expand floating point setcc to scalar if we do not have 6134 // mve.fp, so that we can handle them from there. 6135 if (Op0.getValueType().isFloatingPoint() && !ST->hasMVEFloatOps()) 6136 return SDValue(); 6137 6138 CmpVT = VT; 6139 } 6140 6141 if (Op0.getValueType().getVectorElementType() == MVT::i64 && 6142 (SetCCOpcode == ISD::SETEQ || SetCCOpcode == ISD::SETNE)) { 6143 // Special-case integer 64-bit equality comparisons. They aren't legal, 6144 // but they can be lowered with a few vector instructions. 6145 unsigned CmpElements = CmpVT.getVectorNumElements() * 2; 6146 EVT SplitVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32, CmpElements); 6147 SDValue CastOp0 = DAG.getNode(ISD::BITCAST, dl, SplitVT, Op0); 6148 SDValue CastOp1 = DAG.getNode(ISD::BITCAST, dl, SplitVT, Op1); 6149 SDValue Cmp = DAG.getNode(ISD::SETCC, dl, SplitVT, CastOp0, CastOp1, 6150 DAG.getCondCode(ISD::SETEQ)); 6151 SDValue Reversed = DAG.getNode(ARMISD::VREV64, dl, SplitVT, Cmp); 6152 SDValue Merged = DAG.getNode(ISD::AND, dl, SplitVT, Cmp, Reversed); 6153 Merged = DAG.getNode(ISD::BITCAST, dl, CmpVT, Merged); 6154 if (SetCCOpcode == ISD::SETNE) 6155 Merged = DAG.getNOT(dl, Merged, CmpVT); 6156 Merged = DAG.getSExtOrTrunc(Merged, dl, VT); 6157 return Merged; 6158 } 6159 6160 if (CmpVT.getVectorElementType() == MVT::i64) 6161 // 64-bit comparisons are not legal in general. 6162 return SDValue(); 6163 6164 if (Op1.getValueType().isFloatingPoint()) { 6165 switch (SetCCOpcode) { 6166 default: llvm_unreachable("Illegal FP comparison"); 6167 case ISD::SETUNE: 6168 case ISD::SETNE: 6169 if (ST->hasMVEFloatOps()) { 6170 Opc = ARMCC::NE; break; 6171 } else { 6172 Invert = true; LLVM_FALLTHROUGH; 6173 } 6174 case ISD::SETOEQ: 6175 case ISD::SETEQ: Opc = ARMCC::EQ; break; 6176 case ISD::SETOLT: 6177 case ISD::SETLT: Swap = true; LLVM_FALLTHROUGH; 6178 case ISD::SETOGT: 6179 case ISD::SETGT: Opc = ARMCC::GT; break; 6180 case ISD::SETOLE: 6181 case ISD::SETLE: Swap = true; LLVM_FALLTHROUGH; 6182 case ISD::SETOGE: 6183 case ISD::SETGE: Opc = ARMCC::GE; break; 6184 case ISD::SETUGE: Swap = true; LLVM_FALLTHROUGH; 6185 case ISD::SETULE: Invert = true; Opc = ARMCC::GT; break; 6186 case ISD::SETUGT: Swap = true; LLVM_FALLTHROUGH; 6187 case ISD::SETULT: Invert = true; Opc = ARMCC::GE; break; 6188 case ISD::SETUEQ: Invert = true; LLVM_FALLTHROUGH; 6189 case ISD::SETONE: { 6190 // Expand this to (OLT | OGT). 6191 SDValue TmpOp0 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op1, Op0, 6192 DAG.getConstant(ARMCC::GT, dl, MVT::i32)); 6193 SDValue TmpOp1 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op0, Op1, 6194 DAG.getConstant(ARMCC::GT, dl, MVT::i32)); 6195 SDValue Result = DAG.getNode(ISD::OR, dl, CmpVT, TmpOp0, TmpOp1); 6196 if (Invert) 6197 Result = DAG.getNOT(dl, Result, VT); 6198 return Result; 6199 } 6200 case ISD::SETUO: Invert = true; LLVM_FALLTHROUGH; 6201 case ISD::SETO: { 6202 // Expand this to (OLT | OGE). 6203 SDValue TmpOp0 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op1, Op0, 6204 DAG.getConstant(ARMCC::GT, dl, MVT::i32)); 6205 SDValue TmpOp1 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op0, Op1, 6206 DAG.getConstant(ARMCC::GE, dl, MVT::i32)); 6207 SDValue Result = DAG.getNode(ISD::OR, dl, CmpVT, TmpOp0, TmpOp1); 6208 if (Invert) 6209 Result = DAG.getNOT(dl, Result, VT); 6210 return Result; 6211 } 6212 } 6213 } else { 6214 // Integer comparisons. 6215 switch (SetCCOpcode) { 6216 default: llvm_unreachable("Illegal integer comparison"); 6217 case ISD::SETNE: 6218 if (ST->hasMVEIntegerOps()) { 6219 Opc = ARMCC::NE; break; 6220 } else { 6221 Invert = true; LLVM_FALLTHROUGH; 6222 } 6223 case ISD::SETEQ: Opc = ARMCC::EQ; break; 6224 case ISD::SETLT: Swap = true; LLVM_FALLTHROUGH; 6225 case ISD::SETGT: Opc = ARMCC::GT; break; 6226 case ISD::SETLE: Swap = true; LLVM_FALLTHROUGH; 6227 case ISD::SETGE: Opc = ARMCC::GE; break; 6228 case ISD::SETULT: Swap = true; LLVM_FALLTHROUGH; 6229 case ISD::SETUGT: Opc = ARMCC::HI; break; 6230 case ISD::SETULE: Swap = true; LLVM_FALLTHROUGH; 6231 case ISD::SETUGE: Opc = ARMCC::HS; break; 6232 } 6233 6234 // Detect VTST (Vector Test Bits) = icmp ne (and (op0, op1), zero). 6235 if (ST->hasNEON() && Opc == ARMCC::EQ) { 6236 SDValue AndOp; 6237 if (ISD::isBuildVectorAllZeros(Op1.getNode())) 6238 AndOp = Op0; 6239 else if (ISD::isBuildVectorAllZeros(Op0.getNode())) 6240 AndOp = Op1; 6241 6242 // Ignore bitconvert. 6243 if (AndOp.getNode() && AndOp.getOpcode() == ISD::BITCAST) 6244 AndOp = AndOp.getOperand(0); 6245 6246 if (AndOp.getNode() && AndOp.getOpcode() == ISD::AND) { 6247 Op0 = DAG.getNode(ISD::BITCAST, dl, CmpVT, AndOp.getOperand(0)); 6248 Op1 = DAG.getNode(ISD::BITCAST, dl, CmpVT, AndOp.getOperand(1)); 6249 SDValue Result = DAG.getNode(ARMISD::VTST, dl, CmpVT, Op0, Op1); 6250 if (!Invert) 6251 Result = DAG.getNOT(dl, Result, VT); 6252 return Result; 6253 } 6254 } 6255 } 6256 6257 if (Swap) 6258 std::swap(Op0, Op1); 6259 6260 // If one of the operands is a constant vector zero, attempt to fold the 6261 // comparison to a specialized compare-against-zero form. 6262 SDValue SingleOp; 6263 if (ISD::isBuildVectorAllZeros(Op1.getNode())) 6264 SingleOp = Op0; 6265 else if (ISD::isBuildVectorAllZeros(Op0.getNode())) { 6266 if (Opc == ARMCC::GE) 6267 Opc = ARMCC::LE; 6268 else if (Opc == ARMCC::GT) 6269 Opc = ARMCC::LT; 6270 SingleOp = Op1; 6271 } 6272 6273 SDValue Result; 6274 if (SingleOp.getNode()) { 6275 Result = DAG.getNode(ARMISD::VCMPZ, dl, CmpVT, SingleOp, 6276 DAG.getConstant(Opc, dl, MVT::i32)); 6277 } else { 6278 Result = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op0, Op1, 6279 DAG.getConstant(Opc, dl, MVT::i32)); 6280 } 6281 6282 Result = DAG.getSExtOrTrunc(Result, dl, VT); 6283 6284 if (Invert) 6285 Result = DAG.getNOT(dl, Result, VT); 6286 6287 return Result; 6288 } 6289 6290 static SDValue LowerSETCCCARRY(SDValue Op, SelectionDAG &DAG) { 6291 SDValue LHS = Op.getOperand(0); 6292 SDValue RHS = Op.getOperand(1); 6293 SDValue Carry = Op.getOperand(2); 6294 SDValue Cond = Op.getOperand(3); 6295 SDLoc DL(Op); 6296 6297 assert(LHS.getSimpleValueType().isInteger() && "SETCCCARRY is integer only."); 6298 6299 // ARMISD::SUBE expects a carry not a borrow like ISD::SUBCARRY so we 6300 // have to invert the carry first. 6301 Carry = DAG.getNode(ISD::SUB, DL, MVT::i32, 6302 DAG.getConstant(1, DL, MVT::i32), Carry); 6303 // This converts the boolean value carry into the carry flag. 6304 Carry = ConvertBooleanCarryToCarryFlag(Carry, DAG); 6305 6306 SDVTList VTs = DAG.getVTList(LHS.getValueType(), MVT::i32); 6307 SDValue Cmp = DAG.getNode(ARMISD::SUBE, DL, VTs, LHS, RHS, Carry); 6308 6309 SDValue FVal = DAG.getConstant(0, DL, MVT::i32); 6310 SDValue TVal = DAG.getConstant(1, DL, MVT::i32); 6311 SDValue ARMcc = DAG.getConstant( 6312 IntCCToARMCC(cast<CondCodeSDNode>(Cond)->get()), DL, MVT::i32); 6313 SDValue CCR = DAG.getRegister(ARM::CPSR, MVT::i32); 6314 SDValue Chain = DAG.getCopyToReg(DAG.getEntryNode(), DL, ARM::CPSR, 6315 Cmp.getValue(1), SDValue()); 6316 return DAG.getNode(ARMISD::CMOV, DL, Op.getValueType(), FVal, TVal, ARMcc, 6317 CCR, Chain.getValue(1)); 6318 } 6319 6320 /// isVMOVModifiedImm - Check if the specified splat value corresponds to a 6321 /// valid vector constant for a NEON or MVE instruction with a "modified 6322 /// immediate" operand (e.g., VMOV). If so, return the encoded value. 6323 static SDValue isVMOVModifiedImm(uint64_t SplatBits, uint64_t SplatUndef, 6324 unsigned SplatBitSize, SelectionDAG &DAG, 6325 const SDLoc &dl, EVT &VT, bool is128Bits, 6326 VMOVModImmType type) { 6327 unsigned OpCmode, Imm; 6328 6329 // SplatBitSize is set to the smallest size that splats the vector, so a 6330 // zero vector will always have SplatBitSize == 8. However, NEON modified 6331 // immediate instructions others than VMOV do not support the 8-bit encoding 6332 // of a zero vector, and the default encoding of zero is supposed to be the 6333 // 32-bit version. 6334 if (SplatBits == 0) 6335 SplatBitSize = 32; 6336 6337 switch (SplatBitSize) { 6338 case 8: 6339 if (type != VMOVModImm) 6340 return SDValue(); 6341 // Any 1-byte value is OK. Op=0, Cmode=1110. 6342 assert((SplatBits & ~0xff) == 0 && "one byte splat value is too big"); 6343 OpCmode = 0xe; 6344 Imm = SplatBits; 6345 VT = is128Bits ? MVT::v16i8 : MVT::v8i8; 6346 break; 6347 6348 case 16: 6349 // NEON's 16-bit VMOV supports splat values where only one byte is nonzero. 6350 VT = is128Bits ? MVT::v8i16 : MVT::v4i16; 6351 if ((SplatBits & ~0xff) == 0) { 6352 // Value = 0x00nn: Op=x, Cmode=100x. 6353 OpCmode = 0x8; 6354 Imm = SplatBits; 6355 break; 6356 } 6357 if ((SplatBits & ~0xff00) == 0) { 6358 // Value = 0xnn00: Op=x, Cmode=101x. 6359 OpCmode = 0xa; 6360 Imm = SplatBits >> 8; 6361 break; 6362 } 6363 return SDValue(); 6364 6365 case 32: 6366 // NEON's 32-bit VMOV supports splat values where: 6367 // * only one byte is nonzero, or 6368 // * the least significant byte is 0xff and the second byte is nonzero, or 6369 // * the least significant 2 bytes are 0xff and the third is nonzero. 6370 VT = is128Bits ? MVT::v4i32 : MVT::v2i32; 6371 if ((SplatBits & ~0xff) == 0) { 6372 // Value = 0x000000nn: Op=x, Cmode=000x. 6373 OpCmode = 0; 6374 Imm = SplatBits; 6375 break; 6376 } 6377 if ((SplatBits & ~0xff00) == 0) { 6378 // Value = 0x0000nn00: Op=x, Cmode=001x. 6379 OpCmode = 0x2; 6380 Imm = SplatBits >> 8; 6381 break; 6382 } 6383 if ((SplatBits & ~0xff0000) == 0) { 6384 // Value = 0x00nn0000: Op=x, Cmode=010x. 6385 OpCmode = 0x4; 6386 Imm = SplatBits >> 16; 6387 break; 6388 } 6389 if ((SplatBits & ~0xff000000) == 0) { 6390 // Value = 0xnn000000: Op=x, Cmode=011x. 6391 OpCmode = 0x6; 6392 Imm = SplatBits >> 24; 6393 break; 6394 } 6395 6396 // cmode == 0b1100 and cmode == 0b1101 are not supported for VORR or VBIC 6397 if (type == OtherModImm) return SDValue(); 6398 6399 if ((SplatBits & ~0xffff) == 0 && 6400 ((SplatBits | SplatUndef) & 0xff) == 0xff) { 6401 // Value = 0x0000nnff: Op=x, Cmode=1100. 6402 OpCmode = 0xc; 6403 Imm = SplatBits >> 8; 6404 break; 6405 } 6406 6407 // cmode == 0b1101 is not supported for MVE VMVN 6408 if (type == MVEVMVNModImm) 6409 return SDValue(); 6410 6411 if ((SplatBits & ~0xffffff) == 0 && 6412 ((SplatBits | SplatUndef) & 0xffff) == 0xffff) { 6413 // Value = 0x00nnffff: Op=x, Cmode=1101. 6414 OpCmode = 0xd; 6415 Imm = SplatBits >> 16; 6416 break; 6417 } 6418 6419 // Note: there are a few 32-bit splat values (specifically: 00ffff00, 6420 // ff000000, ff0000ff, and ffff00ff) that are valid for VMOV.I64 but not 6421 // VMOV.I32. A (very) minor optimization would be to replicate the value 6422 // and fall through here to test for a valid 64-bit splat. But, then the 6423 // caller would also need to check and handle the change in size. 6424 return SDValue(); 6425 6426 case 64: { 6427 if (type != VMOVModImm) 6428 return SDValue(); 6429 // NEON has a 64-bit VMOV splat where each byte is either 0 or 0xff. 6430 uint64_t BitMask = 0xff; 6431 uint64_t Val = 0; 6432 unsigned ImmMask = 1; 6433 Imm = 0; 6434 for (int ByteNum = 0; ByteNum < 8; ++ByteNum) { 6435 if (((SplatBits | SplatUndef) & BitMask) == BitMask) { 6436 Val |= BitMask; 6437 Imm |= ImmMask; 6438 } else if ((SplatBits & BitMask) != 0) { 6439 return SDValue(); 6440 } 6441 BitMask <<= 8; 6442 ImmMask <<= 1; 6443 } 6444 6445 if (DAG.getDataLayout().isBigEndian()) 6446 // swap higher and lower 32 bit word 6447 Imm = ((Imm & 0xf) << 4) | ((Imm & 0xf0) >> 4); 6448 6449 // Op=1, Cmode=1110. 6450 OpCmode = 0x1e; 6451 VT = is128Bits ? MVT::v2i64 : MVT::v1i64; 6452 break; 6453 } 6454 6455 default: 6456 llvm_unreachable("unexpected size for isVMOVModifiedImm"); 6457 } 6458 6459 unsigned EncodedVal = ARM_AM::createVMOVModImm(OpCmode, Imm); 6460 return DAG.getTargetConstant(EncodedVal, dl, MVT::i32); 6461 } 6462 6463 SDValue ARMTargetLowering::LowerConstantFP(SDValue Op, SelectionDAG &DAG, 6464 const ARMSubtarget *ST) const { 6465 EVT VT = Op.getValueType(); 6466 bool IsDouble = (VT == MVT::f64); 6467 ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Op); 6468 const APFloat &FPVal = CFP->getValueAPF(); 6469 6470 // Prevent floating-point constants from using literal loads 6471 // when execute-only is enabled. 6472 if (ST->genExecuteOnly()) { 6473 // If we can represent the constant as an immediate, don't lower it 6474 if (isFPImmLegal(FPVal, VT)) 6475 return Op; 6476 // Otherwise, construct as integer, and move to float register 6477 APInt INTVal = FPVal.bitcastToAPInt(); 6478 SDLoc DL(CFP); 6479 switch (VT.getSimpleVT().SimpleTy) { 6480 default: 6481 llvm_unreachable("Unknown floating point type!"); 6482 break; 6483 case MVT::f64: { 6484 SDValue Lo = DAG.getConstant(INTVal.trunc(32), DL, MVT::i32); 6485 SDValue Hi = DAG.getConstant(INTVal.lshr(32).trunc(32), DL, MVT::i32); 6486 if (!ST->isLittle()) 6487 std::swap(Lo, Hi); 6488 return DAG.getNode(ARMISD::VMOVDRR, DL, MVT::f64, Lo, Hi); 6489 } 6490 case MVT::f32: 6491 return DAG.getNode(ARMISD::VMOVSR, DL, VT, 6492 DAG.getConstant(INTVal, DL, MVT::i32)); 6493 } 6494 } 6495 6496 if (!ST->hasVFP3Base()) 6497 return SDValue(); 6498 6499 // Use the default (constant pool) lowering for double constants when we have 6500 // an SP-only FPU 6501 if (IsDouble && !Subtarget->hasFP64()) 6502 return SDValue(); 6503 6504 // Try splatting with a VMOV.f32... 6505 int ImmVal = IsDouble ? ARM_AM::getFP64Imm(FPVal) : ARM_AM::getFP32Imm(FPVal); 6506 6507 if (ImmVal != -1) { 6508 if (IsDouble || !ST->useNEONForSinglePrecisionFP()) { 6509 // We have code in place to select a valid ConstantFP already, no need to 6510 // do any mangling. 6511 return Op; 6512 } 6513 6514 // It's a float and we are trying to use NEON operations where 6515 // possible. Lower it to a splat followed by an extract. 6516 SDLoc DL(Op); 6517 SDValue NewVal = DAG.getTargetConstant(ImmVal, DL, MVT::i32); 6518 SDValue VecConstant = DAG.getNode(ARMISD::VMOVFPIMM, DL, MVT::v2f32, 6519 NewVal); 6520 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, VecConstant, 6521 DAG.getConstant(0, DL, MVT::i32)); 6522 } 6523 6524 // The rest of our options are NEON only, make sure that's allowed before 6525 // proceeding.. 6526 if (!ST->hasNEON() || (!IsDouble && !ST->useNEONForSinglePrecisionFP())) 6527 return SDValue(); 6528 6529 EVT VMovVT; 6530 uint64_t iVal = FPVal.bitcastToAPInt().getZExtValue(); 6531 6532 // It wouldn't really be worth bothering for doubles except for one very 6533 // important value, which does happen to match: 0.0. So make sure we don't do 6534 // anything stupid. 6535 if (IsDouble && (iVal & 0xffffffff) != (iVal >> 32)) 6536 return SDValue(); 6537 6538 // Try a VMOV.i32 (FIXME: i8, i16, or i64 could work too). 6539 SDValue NewVal = isVMOVModifiedImm(iVal & 0xffffffffU, 0, 32, DAG, SDLoc(Op), 6540 VMovVT, false, VMOVModImm); 6541 if (NewVal != SDValue()) { 6542 SDLoc DL(Op); 6543 SDValue VecConstant = DAG.getNode(ARMISD::VMOVIMM, DL, VMovVT, 6544 NewVal); 6545 if (IsDouble) 6546 return DAG.getNode(ISD::BITCAST, DL, MVT::f64, VecConstant); 6547 6548 // It's a float: cast and extract a vector element. 6549 SDValue VecFConstant = DAG.getNode(ISD::BITCAST, DL, MVT::v2f32, 6550 VecConstant); 6551 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, VecFConstant, 6552 DAG.getConstant(0, DL, MVT::i32)); 6553 } 6554 6555 // Finally, try a VMVN.i32 6556 NewVal = isVMOVModifiedImm(~iVal & 0xffffffffU, 0, 32, DAG, SDLoc(Op), VMovVT, 6557 false, VMVNModImm); 6558 if (NewVal != SDValue()) { 6559 SDLoc DL(Op); 6560 SDValue VecConstant = DAG.getNode(ARMISD::VMVNIMM, DL, VMovVT, NewVal); 6561 6562 if (IsDouble) 6563 return DAG.getNode(ISD::BITCAST, DL, MVT::f64, VecConstant); 6564 6565 // It's a float: cast and extract a vector element. 6566 SDValue VecFConstant = DAG.getNode(ISD::BITCAST, DL, MVT::v2f32, 6567 VecConstant); 6568 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, VecFConstant, 6569 DAG.getConstant(0, DL, MVT::i32)); 6570 } 6571 6572 return SDValue(); 6573 } 6574 6575 // check if an VEXT instruction can handle the shuffle mask when the 6576 // vector sources of the shuffle are the same. 6577 static bool isSingletonVEXTMask(ArrayRef<int> M, EVT VT, unsigned &Imm) { 6578 unsigned NumElts = VT.getVectorNumElements(); 6579 6580 // Assume that the first shuffle index is not UNDEF. Fail if it is. 6581 if (M[0] < 0) 6582 return false; 6583 6584 Imm = M[0]; 6585 6586 // If this is a VEXT shuffle, the immediate value is the index of the first 6587 // element. The other shuffle indices must be the successive elements after 6588 // the first one. 6589 unsigned ExpectedElt = Imm; 6590 for (unsigned i = 1; i < NumElts; ++i) { 6591 // Increment the expected index. If it wraps around, just follow it 6592 // back to index zero and keep going. 6593 ++ExpectedElt; 6594 if (ExpectedElt == NumElts) 6595 ExpectedElt = 0; 6596 6597 if (M[i] < 0) continue; // ignore UNDEF indices 6598 if (ExpectedElt != static_cast<unsigned>(M[i])) 6599 return false; 6600 } 6601 6602 return true; 6603 } 6604 6605 static bool isVEXTMask(ArrayRef<int> M, EVT VT, 6606 bool &ReverseVEXT, unsigned &Imm) { 6607 unsigned NumElts = VT.getVectorNumElements(); 6608 ReverseVEXT = false; 6609 6610 // Assume that the first shuffle index is not UNDEF. Fail if it is. 6611 if (M[0] < 0) 6612 return false; 6613 6614 Imm = M[0]; 6615 6616 // If this is a VEXT shuffle, the immediate value is the index of the first 6617 // element. The other shuffle indices must be the successive elements after 6618 // the first one. 6619 unsigned ExpectedElt = Imm; 6620 for (unsigned i = 1; i < NumElts; ++i) { 6621 // Increment the expected index. If it wraps around, it may still be 6622 // a VEXT but the source vectors must be swapped. 6623 ExpectedElt += 1; 6624 if (ExpectedElt == NumElts * 2) { 6625 ExpectedElt = 0; 6626 ReverseVEXT = true; 6627 } 6628 6629 if (M[i] < 0) continue; // ignore UNDEF indices 6630 if (ExpectedElt != static_cast<unsigned>(M[i])) 6631 return false; 6632 } 6633 6634 // Adjust the index value if the source operands will be swapped. 6635 if (ReverseVEXT) 6636 Imm -= NumElts; 6637 6638 return true; 6639 } 6640 6641 /// isVREVMask - Check if a vector shuffle corresponds to a VREV 6642 /// instruction with the specified blocksize. (The order of the elements 6643 /// within each block of the vector is reversed.) 6644 static bool isVREVMask(ArrayRef<int> M, EVT VT, unsigned BlockSize) { 6645 assert((BlockSize==16 || BlockSize==32 || BlockSize==64) && 6646 "Only possible block sizes for VREV are: 16, 32, 64"); 6647 6648 unsigned EltSz = VT.getScalarSizeInBits(); 6649 if (EltSz == 64) 6650 return false; 6651 6652 unsigned NumElts = VT.getVectorNumElements(); 6653 unsigned BlockElts = M[0] + 1; 6654 // If the first shuffle index is UNDEF, be optimistic. 6655 if (M[0] < 0) 6656 BlockElts = BlockSize / EltSz; 6657 6658 if (BlockSize <= EltSz || BlockSize != BlockElts * EltSz) 6659 return false; 6660 6661 for (unsigned i = 0; i < NumElts; ++i) { 6662 if (M[i] < 0) continue; // ignore UNDEF indices 6663 if ((unsigned) M[i] != (i - i%BlockElts) + (BlockElts - 1 - i%BlockElts)) 6664 return false; 6665 } 6666 6667 return true; 6668 } 6669 6670 static bool isVTBLMask(ArrayRef<int> M, EVT VT) { 6671 // We can handle <8 x i8> vector shuffles. If the index in the mask is out of 6672 // range, then 0 is placed into the resulting vector. So pretty much any mask 6673 // of 8 elements can work here. 6674 return VT == MVT::v8i8 && M.size() == 8; 6675 } 6676 6677 static unsigned SelectPairHalf(unsigned Elements, ArrayRef<int> Mask, 6678 unsigned Index) { 6679 if (Mask.size() == Elements * 2) 6680 return Index / Elements; 6681 return Mask[Index] == 0 ? 0 : 1; 6682 } 6683 6684 // Checks whether the shuffle mask represents a vector transpose (VTRN) by 6685 // checking that pairs of elements in the shuffle mask represent the same index 6686 // in each vector, incrementing the expected index by 2 at each step. 6687 // e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 2, 6] 6688 // v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,c,g} 6689 // v2={e,f,g,h} 6690 // WhichResult gives the offset for each element in the mask based on which 6691 // of the two results it belongs to. 6692 // 6693 // The transpose can be represented either as: 6694 // result1 = shufflevector v1, v2, result1_shuffle_mask 6695 // result2 = shufflevector v1, v2, result2_shuffle_mask 6696 // where v1/v2 and the shuffle masks have the same number of elements 6697 // (here WhichResult (see below) indicates which result is being checked) 6698 // 6699 // or as: 6700 // results = shufflevector v1, v2, shuffle_mask 6701 // where both results are returned in one vector and the shuffle mask has twice 6702 // as many elements as v1/v2 (here WhichResult will always be 0 if true) here we 6703 // want to check the low half and high half of the shuffle mask as if it were 6704 // the other case 6705 static bool isVTRNMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) { 6706 unsigned EltSz = VT.getScalarSizeInBits(); 6707 if (EltSz == 64) 6708 return false; 6709 6710 unsigned NumElts = VT.getVectorNumElements(); 6711 if (M.size() != NumElts && M.size() != NumElts*2) 6712 return false; 6713 6714 // If the mask is twice as long as the input vector then we need to check the 6715 // upper and lower parts of the mask with a matching value for WhichResult 6716 // FIXME: A mask with only even values will be rejected in case the first 6717 // element is undefined, e.g. [-1, 4, 2, 6] will be rejected, because only 6718 // M[0] is used to determine WhichResult 6719 for (unsigned i = 0; i < M.size(); i += NumElts) { 6720 WhichResult = SelectPairHalf(NumElts, M, i); 6721 for (unsigned j = 0; j < NumElts; j += 2) { 6722 if ((M[i+j] >= 0 && (unsigned) M[i+j] != j + WhichResult) || 6723 (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != j + NumElts + WhichResult)) 6724 return false; 6725 } 6726 } 6727 6728 if (M.size() == NumElts*2) 6729 WhichResult = 0; 6730 6731 return true; 6732 } 6733 6734 /// isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of 6735 /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef". 6736 /// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>. 6737 static bool isVTRN_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){ 6738 unsigned EltSz = VT.getScalarSizeInBits(); 6739 if (EltSz == 64) 6740 return false; 6741 6742 unsigned NumElts = VT.getVectorNumElements(); 6743 if (M.size() != NumElts && M.size() != NumElts*2) 6744 return false; 6745 6746 for (unsigned i = 0; i < M.size(); i += NumElts) { 6747 WhichResult = SelectPairHalf(NumElts, M, i); 6748 for (unsigned j = 0; j < NumElts; j += 2) { 6749 if ((M[i+j] >= 0 && (unsigned) M[i+j] != j + WhichResult) || 6750 (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != j + WhichResult)) 6751 return false; 6752 } 6753 } 6754 6755 if (M.size() == NumElts*2) 6756 WhichResult = 0; 6757 6758 return true; 6759 } 6760 6761 // Checks whether the shuffle mask represents a vector unzip (VUZP) by checking 6762 // that the mask elements are either all even and in steps of size 2 or all odd 6763 // and in steps of size 2. 6764 // e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 2, 4, 6] 6765 // v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,c,e,g} 6766 // v2={e,f,g,h} 6767 // Requires similar checks to that of isVTRNMask with 6768 // respect the how results are returned. 6769 static bool isVUZPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) { 6770 unsigned EltSz = VT.getScalarSizeInBits(); 6771 if (EltSz == 64) 6772 return false; 6773 6774 unsigned NumElts = VT.getVectorNumElements(); 6775 if (M.size() != NumElts && M.size() != NumElts*2) 6776 return false; 6777 6778 for (unsigned i = 0; i < M.size(); i += NumElts) { 6779 WhichResult = SelectPairHalf(NumElts, M, i); 6780 for (unsigned j = 0; j < NumElts; ++j) { 6781 if (M[i+j] >= 0 && (unsigned) M[i+j] != 2 * j + WhichResult) 6782 return false; 6783 } 6784 } 6785 6786 if (M.size() == NumElts*2) 6787 WhichResult = 0; 6788 6789 // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. 6790 if (VT.is64BitVector() && EltSz == 32) 6791 return false; 6792 6793 return true; 6794 } 6795 6796 /// isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of 6797 /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef". 6798 /// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>, 6799 static bool isVUZP_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){ 6800 unsigned EltSz = VT.getScalarSizeInBits(); 6801 if (EltSz == 64) 6802 return false; 6803 6804 unsigned NumElts = VT.getVectorNumElements(); 6805 if (M.size() != NumElts && M.size() != NumElts*2) 6806 return false; 6807 6808 unsigned Half = NumElts / 2; 6809 for (unsigned i = 0; i < M.size(); i += NumElts) { 6810 WhichResult = SelectPairHalf(NumElts, M, i); 6811 for (unsigned j = 0; j < NumElts; j += Half) { 6812 unsigned Idx = WhichResult; 6813 for (unsigned k = 0; k < Half; ++k) { 6814 int MIdx = M[i + j + k]; 6815 if (MIdx >= 0 && (unsigned) MIdx != Idx) 6816 return false; 6817 Idx += 2; 6818 } 6819 } 6820 } 6821 6822 if (M.size() == NumElts*2) 6823 WhichResult = 0; 6824 6825 // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. 6826 if (VT.is64BitVector() && EltSz == 32) 6827 return false; 6828 6829 return true; 6830 } 6831 6832 // Checks whether the shuffle mask represents a vector zip (VZIP) by checking 6833 // that pairs of elements of the shufflemask represent the same index in each 6834 // vector incrementing sequentially through the vectors. 6835 // e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 1, 5] 6836 // v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,b,f} 6837 // v2={e,f,g,h} 6838 // Requires similar checks to that of isVTRNMask with respect the how results 6839 // are returned. 6840 static bool isVZIPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) { 6841 unsigned EltSz = VT.getScalarSizeInBits(); 6842 if (EltSz == 64) 6843 return false; 6844 6845 unsigned NumElts = VT.getVectorNumElements(); 6846 if (M.size() != NumElts && M.size() != NumElts*2) 6847 return false; 6848 6849 for (unsigned i = 0; i < M.size(); i += NumElts) { 6850 WhichResult = SelectPairHalf(NumElts, M, i); 6851 unsigned Idx = WhichResult * NumElts / 2; 6852 for (unsigned j = 0; j < NumElts; j += 2) { 6853 if ((M[i+j] >= 0 && (unsigned) M[i+j] != Idx) || 6854 (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != Idx + NumElts)) 6855 return false; 6856 Idx += 1; 6857 } 6858 } 6859 6860 if (M.size() == NumElts*2) 6861 WhichResult = 0; 6862 6863 // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. 6864 if (VT.is64BitVector() && EltSz == 32) 6865 return false; 6866 6867 return true; 6868 } 6869 6870 /// isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of 6871 /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef". 6872 /// Mask is e.g., <0, 0, 1, 1> instead of <0, 4, 1, 5>. 6873 static bool isVZIP_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){ 6874 unsigned EltSz = VT.getScalarSizeInBits(); 6875 if (EltSz == 64) 6876 return false; 6877 6878 unsigned NumElts = VT.getVectorNumElements(); 6879 if (M.size() != NumElts && M.size() != NumElts*2) 6880 return false; 6881 6882 for (unsigned i = 0; i < M.size(); i += NumElts) { 6883 WhichResult = SelectPairHalf(NumElts, M, i); 6884 unsigned Idx = WhichResult * NumElts / 2; 6885 for (unsigned j = 0; j < NumElts; j += 2) { 6886 if ((M[i+j] >= 0 && (unsigned) M[i+j] != Idx) || 6887 (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != Idx)) 6888 return false; 6889 Idx += 1; 6890 } 6891 } 6892 6893 if (M.size() == NumElts*2) 6894 WhichResult = 0; 6895 6896 // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. 6897 if (VT.is64BitVector() && EltSz == 32) 6898 return false; 6899 6900 return true; 6901 } 6902 6903 /// Check if \p ShuffleMask is a NEON two-result shuffle (VZIP, VUZP, VTRN), 6904 /// and return the corresponding ARMISD opcode if it is, or 0 if it isn't. 6905 static unsigned isNEONTwoResultShuffleMask(ArrayRef<int> ShuffleMask, EVT VT, 6906 unsigned &WhichResult, 6907 bool &isV_UNDEF) { 6908 isV_UNDEF = false; 6909 if (isVTRNMask(ShuffleMask, VT, WhichResult)) 6910 return ARMISD::VTRN; 6911 if (isVUZPMask(ShuffleMask, VT, WhichResult)) 6912 return ARMISD::VUZP; 6913 if (isVZIPMask(ShuffleMask, VT, WhichResult)) 6914 return ARMISD::VZIP; 6915 6916 isV_UNDEF = true; 6917 if (isVTRN_v_undef_Mask(ShuffleMask, VT, WhichResult)) 6918 return ARMISD::VTRN; 6919 if (isVUZP_v_undef_Mask(ShuffleMask, VT, WhichResult)) 6920 return ARMISD::VUZP; 6921 if (isVZIP_v_undef_Mask(ShuffleMask, VT, WhichResult)) 6922 return ARMISD::VZIP; 6923 6924 return 0; 6925 } 6926 6927 /// \return true if this is a reverse operation on an vector. 6928 static bool isReverseMask(ArrayRef<int> M, EVT VT) { 6929 unsigned NumElts = VT.getVectorNumElements(); 6930 // Make sure the mask has the right size. 6931 if (NumElts != M.size()) 6932 return false; 6933 6934 // Look for <15, ..., 3, -1, 1, 0>. 6935 for (unsigned i = 0; i != NumElts; ++i) 6936 if (M[i] >= 0 && M[i] != (int) (NumElts - 1 - i)) 6937 return false; 6938 6939 return true; 6940 } 6941 6942 static bool isVMOVNMask(ArrayRef<int> M, EVT VT, bool Top) { 6943 unsigned NumElts = VT.getVectorNumElements(); 6944 // Make sure the mask has the right size. 6945 if (NumElts != M.size() || (VT != MVT::v8i16 && VT != MVT::v16i8)) 6946 return false; 6947 6948 // If Top 6949 // Look for <0, N, 2, N+2, 4, N+4, ..>. 6950 // This inserts Input2 into Input1 6951 // else if not Top 6952 // Look for <0, N+1, 2, N+3, 4, N+5, ..> 6953 // This inserts Input1 into Input2 6954 unsigned Offset = Top ? 0 : 1; 6955 for (unsigned i = 0; i < NumElts; i+=2) { 6956 if (M[i] >= 0 && M[i] != (int)i) 6957 return false; 6958 if (M[i+1] >= 0 && M[i+1] != (int)(NumElts + i + Offset)) 6959 return false; 6960 } 6961 6962 return true; 6963 } 6964 6965 // If N is an integer constant that can be moved into a register in one 6966 // instruction, return an SDValue of such a constant (will become a MOV 6967 // instruction). Otherwise return null. 6968 static SDValue IsSingleInstrConstant(SDValue N, SelectionDAG &DAG, 6969 const ARMSubtarget *ST, const SDLoc &dl) { 6970 uint64_t Val; 6971 if (!isa<ConstantSDNode>(N)) 6972 return SDValue(); 6973 Val = cast<ConstantSDNode>(N)->getZExtValue(); 6974 6975 if (ST->isThumb1Only()) { 6976 if (Val <= 255 || ~Val <= 255) 6977 return DAG.getConstant(Val, dl, MVT::i32); 6978 } else { 6979 if (ARM_AM::getSOImmVal(Val) != -1 || ARM_AM::getSOImmVal(~Val) != -1) 6980 return DAG.getConstant(Val, dl, MVT::i32); 6981 } 6982 return SDValue(); 6983 } 6984 6985 static SDValue LowerBUILD_VECTOR_i1(SDValue Op, SelectionDAG &DAG, 6986 const ARMSubtarget *ST) { 6987 SDLoc dl(Op); 6988 EVT VT = Op.getValueType(); 6989 6990 assert(ST->hasMVEIntegerOps() && "LowerBUILD_VECTOR_i1 called without MVE!"); 6991 6992 unsigned NumElts = VT.getVectorNumElements(); 6993 unsigned BoolMask; 6994 unsigned BitsPerBool; 6995 if (NumElts == 4) { 6996 BitsPerBool = 4; 6997 BoolMask = 0xf; 6998 } else if (NumElts == 8) { 6999 BitsPerBool = 2; 7000 BoolMask = 0x3; 7001 } else if (NumElts == 16) { 7002 BitsPerBool = 1; 7003 BoolMask = 0x1; 7004 } else 7005 return SDValue(); 7006 7007 // If this is a single value copied into all lanes (a splat), we can just sign 7008 // extend that single value 7009 SDValue FirstOp = Op.getOperand(0); 7010 if (!isa<ConstantSDNode>(FirstOp) && 7011 std::all_of(std::next(Op->op_begin()), Op->op_end(), 7012 [&FirstOp](SDUse &U) { 7013 return U.get().isUndef() || U.get() == FirstOp; 7014 })) { 7015 SDValue Ext = DAG.getNode(ISD::SIGN_EXTEND_INREG, dl, MVT::i32, FirstOp, 7016 DAG.getValueType(MVT::i1)); 7017 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, Op.getValueType(), Ext); 7018 } 7019 7020 // First create base with bits set where known 7021 unsigned Bits32 = 0; 7022 for (unsigned i = 0; i < NumElts; ++i) { 7023 SDValue V = Op.getOperand(i); 7024 if (!isa<ConstantSDNode>(V) && !V.isUndef()) 7025 continue; 7026 bool BitSet = V.isUndef() ? false : cast<ConstantSDNode>(V)->getZExtValue(); 7027 if (BitSet) 7028 Bits32 |= BoolMask << (i * BitsPerBool); 7029 } 7030 7031 // Add in unknown nodes 7032 SDValue Base = DAG.getNode(ARMISD::PREDICATE_CAST, dl, VT, 7033 DAG.getConstant(Bits32, dl, MVT::i32)); 7034 for (unsigned i = 0; i < NumElts; ++i) { 7035 SDValue V = Op.getOperand(i); 7036 if (isa<ConstantSDNode>(V) || V.isUndef()) 7037 continue; 7038 Base = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, Base, V, 7039 DAG.getConstant(i, dl, MVT::i32)); 7040 } 7041 7042 return Base; 7043 } 7044 7045 // If this is a case we can't handle, return null and let the default 7046 // expansion code take care of it. 7047 SDValue ARMTargetLowering::LowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG, 7048 const ARMSubtarget *ST) const { 7049 BuildVectorSDNode *BVN = cast<BuildVectorSDNode>(Op.getNode()); 7050 SDLoc dl(Op); 7051 EVT VT = Op.getValueType(); 7052 7053 if (ST->hasMVEIntegerOps() && VT.getScalarSizeInBits() == 1) 7054 return LowerBUILD_VECTOR_i1(Op, DAG, ST); 7055 7056 APInt SplatBits, SplatUndef; 7057 unsigned SplatBitSize; 7058 bool HasAnyUndefs; 7059 if (BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs)) { 7060 if (SplatUndef.isAllOnesValue()) 7061 return DAG.getUNDEF(VT); 7062 7063 if ((ST->hasNEON() && SplatBitSize <= 64) || 7064 (ST->hasMVEIntegerOps() && SplatBitSize <= 32)) { 7065 // Check if an immediate VMOV works. 7066 EVT VmovVT; 7067 SDValue Val = isVMOVModifiedImm(SplatBits.getZExtValue(), 7068 SplatUndef.getZExtValue(), SplatBitSize, 7069 DAG, dl, VmovVT, VT.is128BitVector(), 7070 VMOVModImm); 7071 7072 if (Val.getNode()) { 7073 SDValue Vmov = DAG.getNode(ARMISD::VMOVIMM, dl, VmovVT, Val); 7074 return DAG.getNode(ISD::BITCAST, dl, VT, Vmov); 7075 } 7076 7077 // Try an immediate VMVN. 7078 uint64_t NegatedImm = (~SplatBits).getZExtValue(); 7079 Val = isVMOVModifiedImm( 7080 NegatedImm, SplatUndef.getZExtValue(), SplatBitSize, 7081 DAG, dl, VmovVT, VT.is128BitVector(), 7082 ST->hasMVEIntegerOps() ? MVEVMVNModImm : VMVNModImm); 7083 if (Val.getNode()) { 7084 SDValue Vmov = DAG.getNode(ARMISD::VMVNIMM, dl, VmovVT, Val); 7085 return DAG.getNode(ISD::BITCAST, dl, VT, Vmov); 7086 } 7087 7088 // Use vmov.f32 to materialize other v2f32 and v4f32 splats. 7089 if ((VT == MVT::v2f32 || VT == MVT::v4f32) && SplatBitSize == 32) { 7090 int ImmVal = ARM_AM::getFP32Imm(SplatBits); 7091 if (ImmVal != -1) { 7092 SDValue Val = DAG.getTargetConstant(ImmVal, dl, MVT::i32); 7093 return DAG.getNode(ARMISD::VMOVFPIMM, dl, VT, Val); 7094 } 7095 } 7096 } 7097 } 7098 7099 // Scan through the operands to see if only one value is used. 7100 // 7101 // As an optimisation, even if more than one value is used it may be more 7102 // profitable to splat with one value then change some lanes. 7103 // 7104 // Heuristically we decide to do this if the vector has a "dominant" value, 7105 // defined as splatted to more than half of the lanes. 7106 unsigned NumElts = VT.getVectorNumElements(); 7107 bool isOnlyLowElement = true; 7108 bool usesOnlyOneValue = true; 7109 bool hasDominantValue = false; 7110 bool isConstant = true; 7111 7112 // Map of the number of times a particular SDValue appears in the 7113 // element list. 7114 DenseMap<SDValue, unsigned> ValueCounts; 7115 SDValue Value; 7116 for (unsigned i = 0; i < NumElts; ++i) { 7117 SDValue V = Op.getOperand(i); 7118 if (V.isUndef()) 7119 continue; 7120 if (i > 0) 7121 isOnlyLowElement = false; 7122 if (!isa<ConstantFPSDNode>(V) && !isa<ConstantSDNode>(V)) 7123 isConstant = false; 7124 7125 ValueCounts.insert(std::make_pair(V, 0)); 7126 unsigned &Count = ValueCounts[V]; 7127 7128 // Is this value dominant? (takes up more than half of the lanes) 7129 if (++Count > (NumElts / 2)) { 7130 hasDominantValue = true; 7131 Value = V; 7132 } 7133 } 7134 if (ValueCounts.size() != 1) 7135 usesOnlyOneValue = false; 7136 if (!Value.getNode() && !ValueCounts.empty()) 7137 Value = ValueCounts.begin()->first; 7138 7139 if (ValueCounts.empty()) 7140 return DAG.getUNDEF(VT); 7141 7142 // Loads are better lowered with insert_vector_elt/ARMISD::BUILD_VECTOR. 7143 // Keep going if we are hitting this case. 7144 if (isOnlyLowElement && !ISD::isNormalLoad(Value.getNode())) 7145 return DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, VT, Value); 7146 7147 unsigned EltSize = VT.getScalarSizeInBits(); 7148 7149 // Use VDUP for non-constant splats. For f32 constant splats, reduce to 7150 // i32 and try again. 7151 if (hasDominantValue && EltSize <= 32) { 7152 if (!isConstant) { 7153 SDValue N; 7154 7155 // If we are VDUPing a value that comes directly from a vector, that will 7156 // cause an unnecessary move to and from a GPR, where instead we could 7157 // just use VDUPLANE. We can only do this if the lane being extracted 7158 // is at a constant index, as the VDUP from lane instructions only have 7159 // constant-index forms. 7160 ConstantSDNode *constIndex; 7161 if (Value->getOpcode() == ISD::EXTRACT_VECTOR_ELT && 7162 (constIndex = dyn_cast<ConstantSDNode>(Value->getOperand(1)))) { 7163 // We need to create a new undef vector to use for the VDUPLANE if the 7164 // size of the vector from which we get the value is different than the 7165 // size of the vector that we need to create. We will insert the element 7166 // such that the register coalescer will remove unnecessary copies. 7167 if (VT != Value->getOperand(0).getValueType()) { 7168 unsigned index = constIndex->getAPIntValue().getLimitedValue() % 7169 VT.getVectorNumElements(); 7170 N = DAG.getNode(ARMISD::VDUPLANE, dl, VT, 7171 DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, DAG.getUNDEF(VT), 7172 Value, DAG.getConstant(index, dl, MVT::i32)), 7173 DAG.getConstant(index, dl, MVT::i32)); 7174 } else 7175 N = DAG.getNode(ARMISD::VDUPLANE, dl, VT, 7176 Value->getOperand(0), Value->getOperand(1)); 7177 } else 7178 N = DAG.getNode(ARMISD::VDUP, dl, VT, Value); 7179 7180 if (!usesOnlyOneValue) { 7181 // The dominant value was splatted as 'N', but we now have to insert 7182 // all differing elements. 7183 for (unsigned I = 0; I < NumElts; ++I) { 7184 if (Op.getOperand(I) == Value) 7185 continue; 7186 SmallVector<SDValue, 3> Ops; 7187 Ops.push_back(N); 7188 Ops.push_back(Op.getOperand(I)); 7189 Ops.push_back(DAG.getConstant(I, dl, MVT::i32)); 7190 N = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, Ops); 7191 } 7192 } 7193 return N; 7194 } 7195 if (VT.getVectorElementType().isFloatingPoint()) { 7196 SmallVector<SDValue, 8> Ops; 7197 MVT FVT = VT.getVectorElementType().getSimpleVT(); 7198 assert(FVT == MVT::f32 || FVT == MVT::f16); 7199 MVT IVT = (FVT == MVT::f32) ? MVT::i32 : MVT::i16; 7200 for (unsigned i = 0; i < NumElts; ++i) 7201 Ops.push_back(DAG.getNode(ISD::BITCAST, dl, IVT, 7202 Op.getOperand(i))); 7203 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), IVT, NumElts); 7204 SDValue Val = DAG.getBuildVector(VecVT, dl, Ops); 7205 Val = LowerBUILD_VECTOR(Val, DAG, ST); 7206 if (Val.getNode()) 7207 return DAG.getNode(ISD::BITCAST, dl, VT, Val); 7208 } 7209 if (usesOnlyOneValue) { 7210 SDValue Val = IsSingleInstrConstant(Value, DAG, ST, dl); 7211 if (isConstant && Val.getNode()) 7212 return DAG.getNode(ARMISD::VDUP, dl, VT, Val); 7213 } 7214 } 7215 7216 // If all elements are constants and the case above didn't get hit, fall back 7217 // to the default expansion, which will generate a load from the constant 7218 // pool. 7219 if (isConstant) 7220 return SDValue(); 7221 7222 // Empirical tests suggest this is rarely worth it for vectors of length <= 2. 7223 if (NumElts >= 4) { 7224 SDValue shuffle = ReconstructShuffle(Op, DAG); 7225 if (shuffle != SDValue()) 7226 return shuffle; 7227 } 7228 7229 if (ST->hasNEON() && VT.is128BitVector() && VT != MVT::v2f64 && VT != MVT::v4f32) { 7230 // If we haven't found an efficient lowering, try splitting a 128-bit vector 7231 // into two 64-bit vectors; we might discover a better way to lower it. 7232 SmallVector<SDValue, 64> Ops(Op->op_begin(), Op->op_begin() + NumElts); 7233 EVT ExtVT = VT.getVectorElementType(); 7234 EVT HVT = EVT::getVectorVT(*DAG.getContext(), ExtVT, NumElts / 2); 7235 SDValue Lower = 7236 DAG.getBuildVector(HVT, dl, makeArrayRef(&Ops[0], NumElts / 2)); 7237 if (Lower.getOpcode() == ISD::BUILD_VECTOR) 7238 Lower = LowerBUILD_VECTOR(Lower, DAG, ST); 7239 SDValue Upper = DAG.getBuildVector( 7240 HVT, dl, makeArrayRef(&Ops[NumElts / 2], NumElts / 2)); 7241 if (Upper.getOpcode() == ISD::BUILD_VECTOR) 7242 Upper = LowerBUILD_VECTOR(Upper, DAG, ST); 7243 if (Lower && Upper) 7244 return DAG.getNode(ISD::CONCAT_VECTORS, dl, VT, Lower, Upper); 7245 } 7246 7247 // Vectors with 32- or 64-bit elements can be built by directly assigning 7248 // the subregisters. Lower it to an ARMISD::BUILD_VECTOR so the operands 7249 // will be legalized. 7250 if (EltSize >= 32) { 7251 // Do the expansion with floating-point types, since that is what the VFP 7252 // registers are defined to use, and since i64 is not legal. 7253 EVT EltVT = EVT::getFloatingPointVT(EltSize); 7254 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), EltVT, NumElts); 7255 SmallVector<SDValue, 8> Ops; 7256 for (unsigned i = 0; i < NumElts; ++i) 7257 Ops.push_back(DAG.getNode(ISD::BITCAST, dl, EltVT, Op.getOperand(i))); 7258 SDValue Val = DAG.getNode(ARMISD::BUILD_VECTOR, dl, VecVT, Ops); 7259 return DAG.getNode(ISD::BITCAST, dl, VT, Val); 7260 } 7261 7262 // If all else fails, just use a sequence of INSERT_VECTOR_ELT when we 7263 // know the default expansion would otherwise fall back on something even 7264 // worse. For a vector with one or two non-undef values, that's 7265 // scalar_to_vector for the elements followed by a shuffle (provided the 7266 // shuffle is valid for the target) and materialization element by element 7267 // on the stack followed by a load for everything else. 7268 if (!isConstant && !usesOnlyOneValue) { 7269 SDValue Vec = DAG.getUNDEF(VT); 7270 for (unsigned i = 0 ; i < NumElts; ++i) { 7271 SDValue V = Op.getOperand(i); 7272 if (V.isUndef()) 7273 continue; 7274 SDValue LaneIdx = DAG.getConstant(i, dl, MVT::i32); 7275 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, Vec, V, LaneIdx); 7276 } 7277 return Vec; 7278 } 7279 7280 return SDValue(); 7281 } 7282 7283 // Gather data to see if the operation can be modelled as a 7284 // shuffle in combination with VEXTs. 7285 SDValue ARMTargetLowering::ReconstructShuffle(SDValue Op, 7286 SelectionDAG &DAG) const { 7287 assert(Op.getOpcode() == ISD::BUILD_VECTOR && "Unknown opcode!"); 7288 SDLoc dl(Op); 7289 EVT VT = Op.getValueType(); 7290 unsigned NumElts = VT.getVectorNumElements(); 7291 7292 struct ShuffleSourceInfo { 7293 SDValue Vec; 7294 unsigned MinElt = std::numeric_limits<unsigned>::max(); 7295 unsigned MaxElt = 0; 7296 7297 // We may insert some combination of BITCASTs and VEXT nodes to force Vec to 7298 // be compatible with the shuffle we intend to construct. As a result 7299 // ShuffleVec will be some sliding window into the original Vec. 7300 SDValue ShuffleVec; 7301 7302 // Code should guarantee that element i in Vec starts at element "WindowBase 7303 // + i * WindowScale in ShuffleVec". 7304 int WindowBase = 0; 7305 int WindowScale = 1; 7306 7307 ShuffleSourceInfo(SDValue Vec) : Vec(Vec), ShuffleVec(Vec) {} 7308 7309 bool operator ==(SDValue OtherVec) { return Vec == OtherVec; } 7310 }; 7311 7312 // First gather all vectors used as an immediate source for this BUILD_VECTOR 7313 // node. 7314 SmallVector<ShuffleSourceInfo, 2> Sources; 7315 for (unsigned i = 0; i < NumElts; ++i) { 7316 SDValue V = Op.getOperand(i); 7317 if (V.isUndef()) 7318 continue; 7319 else if (V.getOpcode() != ISD::EXTRACT_VECTOR_ELT) { 7320 // A shuffle can only come from building a vector from various 7321 // elements of other vectors. 7322 return SDValue(); 7323 } else if (!isa<ConstantSDNode>(V.getOperand(1))) { 7324 // Furthermore, shuffles require a constant mask, whereas extractelts 7325 // accept variable indices. 7326 return SDValue(); 7327 } 7328 7329 // Add this element source to the list if it's not already there. 7330 SDValue SourceVec = V.getOperand(0); 7331 auto Source = llvm::find(Sources, SourceVec); 7332 if (Source == Sources.end()) 7333 Source = Sources.insert(Sources.end(), ShuffleSourceInfo(SourceVec)); 7334 7335 // Update the minimum and maximum lane number seen. 7336 unsigned EltNo = cast<ConstantSDNode>(V.getOperand(1))->getZExtValue(); 7337 Source->MinElt = std::min(Source->MinElt, EltNo); 7338 Source->MaxElt = std::max(Source->MaxElt, EltNo); 7339 } 7340 7341 // Currently only do something sane when at most two source vectors 7342 // are involved. 7343 if (Sources.size() > 2) 7344 return SDValue(); 7345 7346 // Find out the smallest element size among result and two sources, and use 7347 // it as element size to build the shuffle_vector. 7348 EVT SmallestEltTy = VT.getVectorElementType(); 7349 for (auto &Source : Sources) { 7350 EVT SrcEltTy = Source.Vec.getValueType().getVectorElementType(); 7351 if (SrcEltTy.bitsLT(SmallestEltTy)) 7352 SmallestEltTy = SrcEltTy; 7353 } 7354 unsigned ResMultiplier = 7355 VT.getScalarSizeInBits() / SmallestEltTy.getSizeInBits(); 7356 NumElts = VT.getSizeInBits() / SmallestEltTy.getSizeInBits(); 7357 EVT ShuffleVT = EVT::getVectorVT(*DAG.getContext(), SmallestEltTy, NumElts); 7358 7359 // If the source vector is too wide or too narrow, we may nevertheless be able 7360 // to construct a compatible shuffle either by concatenating it with UNDEF or 7361 // extracting a suitable range of elements. 7362 for (auto &Src : Sources) { 7363 EVT SrcVT = Src.ShuffleVec.getValueType(); 7364 7365 if (SrcVT.getSizeInBits() == VT.getSizeInBits()) 7366 continue; 7367 7368 // This stage of the search produces a source with the same element type as 7369 // the original, but with a total width matching the BUILD_VECTOR output. 7370 EVT EltVT = SrcVT.getVectorElementType(); 7371 unsigned NumSrcElts = VT.getSizeInBits() / EltVT.getSizeInBits(); 7372 EVT DestVT = EVT::getVectorVT(*DAG.getContext(), EltVT, NumSrcElts); 7373 7374 if (SrcVT.getSizeInBits() < VT.getSizeInBits()) { 7375 if (2 * SrcVT.getSizeInBits() != VT.getSizeInBits()) 7376 return SDValue(); 7377 // We can pad out the smaller vector for free, so if it's part of a 7378 // shuffle... 7379 Src.ShuffleVec = 7380 DAG.getNode(ISD::CONCAT_VECTORS, dl, DestVT, Src.ShuffleVec, 7381 DAG.getUNDEF(Src.ShuffleVec.getValueType())); 7382 continue; 7383 } 7384 7385 if (SrcVT.getSizeInBits() != 2 * VT.getSizeInBits()) 7386 return SDValue(); 7387 7388 if (Src.MaxElt - Src.MinElt >= NumSrcElts) { 7389 // Span too large for a VEXT to cope 7390 return SDValue(); 7391 } 7392 7393 if (Src.MinElt >= NumSrcElts) { 7394 // The extraction can just take the second half 7395 Src.ShuffleVec = 7396 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec, 7397 DAG.getConstant(NumSrcElts, dl, MVT::i32)); 7398 Src.WindowBase = -NumSrcElts; 7399 } else if (Src.MaxElt < NumSrcElts) { 7400 // The extraction can just take the first half 7401 Src.ShuffleVec = 7402 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec, 7403 DAG.getConstant(0, dl, MVT::i32)); 7404 } else { 7405 // An actual VEXT is needed 7406 SDValue VEXTSrc1 = 7407 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec, 7408 DAG.getConstant(0, dl, MVT::i32)); 7409 SDValue VEXTSrc2 = 7410 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec, 7411 DAG.getConstant(NumSrcElts, dl, MVT::i32)); 7412 7413 Src.ShuffleVec = DAG.getNode(ARMISD::VEXT, dl, DestVT, VEXTSrc1, 7414 VEXTSrc2, 7415 DAG.getConstant(Src.MinElt, dl, MVT::i32)); 7416 Src.WindowBase = -Src.MinElt; 7417 } 7418 } 7419 7420 // Another possible incompatibility occurs from the vector element types. We 7421 // can fix this by bitcasting the source vectors to the same type we intend 7422 // for the shuffle. 7423 for (auto &Src : Sources) { 7424 EVT SrcEltTy = Src.ShuffleVec.getValueType().getVectorElementType(); 7425 if (SrcEltTy == SmallestEltTy) 7426 continue; 7427 assert(ShuffleVT.getVectorElementType() == SmallestEltTy); 7428 Src.ShuffleVec = DAG.getNode(ISD::BITCAST, dl, ShuffleVT, Src.ShuffleVec); 7429 Src.WindowScale = SrcEltTy.getSizeInBits() / SmallestEltTy.getSizeInBits(); 7430 Src.WindowBase *= Src.WindowScale; 7431 } 7432 7433 // Final sanity check before we try to actually produce a shuffle. 7434 LLVM_DEBUG(for (auto Src 7435 : Sources) 7436 assert(Src.ShuffleVec.getValueType() == ShuffleVT);); 7437 7438 // The stars all align, our next step is to produce the mask for the shuffle. 7439 SmallVector<int, 8> Mask(ShuffleVT.getVectorNumElements(), -1); 7440 int BitsPerShuffleLane = ShuffleVT.getScalarSizeInBits(); 7441 for (unsigned i = 0; i < VT.getVectorNumElements(); ++i) { 7442 SDValue Entry = Op.getOperand(i); 7443 if (Entry.isUndef()) 7444 continue; 7445 7446 auto Src = llvm::find(Sources, Entry.getOperand(0)); 7447 int EltNo = cast<ConstantSDNode>(Entry.getOperand(1))->getSExtValue(); 7448 7449 // EXTRACT_VECTOR_ELT performs an implicit any_ext; BUILD_VECTOR an implicit 7450 // trunc. So only std::min(SrcBits, DestBits) actually get defined in this 7451 // segment. 7452 EVT OrigEltTy = Entry.getOperand(0).getValueType().getVectorElementType(); 7453 int BitsDefined = std::min(OrigEltTy.getSizeInBits(), 7454 VT.getScalarSizeInBits()); 7455 int LanesDefined = BitsDefined / BitsPerShuffleLane; 7456 7457 // This source is expected to fill ResMultiplier lanes of the final shuffle, 7458 // starting at the appropriate offset. 7459 int *LaneMask = &Mask[i * ResMultiplier]; 7460 7461 int ExtractBase = EltNo * Src->WindowScale + Src->WindowBase; 7462 ExtractBase += NumElts * (Src - Sources.begin()); 7463 for (int j = 0; j < LanesDefined; ++j) 7464 LaneMask[j] = ExtractBase + j; 7465 } 7466 7467 7468 // We can't handle more than two sources. This should have already 7469 // been checked before this point. 7470 assert(Sources.size() <= 2 && "Too many sources!"); 7471 7472 SDValue ShuffleOps[] = { DAG.getUNDEF(ShuffleVT), DAG.getUNDEF(ShuffleVT) }; 7473 for (unsigned i = 0; i < Sources.size(); ++i) 7474 ShuffleOps[i] = Sources[i].ShuffleVec; 7475 7476 SDValue Shuffle = buildLegalVectorShuffle(ShuffleVT, dl, ShuffleOps[0], 7477 ShuffleOps[1], Mask, DAG); 7478 if (!Shuffle) 7479 return SDValue(); 7480 return DAG.getNode(ISD::BITCAST, dl, VT, Shuffle); 7481 } 7482 7483 enum ShuffleOpCodes { 7484 OP_COPY = 0, // Copy, used for things like <u,u,u,3> to say it is <0,1,2,3> 7485 OP_VREV, 7486 OP_VDUP0, 7487 OP_VDUP1, 7488 OP_VDUP2, 7489 OP_VDUP3, 7490 OP_VEXT1, 7491 OP_VEXT2, 7492 OP_VEXT3, 7493 OP_VUZPL, // VUZP, left result 7494 OP_VUZPR, // VUZP, right result 7495 OP_VZIPL, // VZIP, left result 7496 OP_VZIPR, // VZIP, right result 7497 OP_VTRNL, // VTRN, left result 7498 OP_VTRNR // VTRN, right result 7499 }; 7500 7501 static bool isLegalMVEShuffleOp(unsigned PFEntry) { 7502 unsigned OpNum = (PFEntry >> 26) & 0x0F; 7503 switch (OpNum) { 7504 case OP_COPY: 7505 case OP_VREV: 7506 case OP_VDUP0: 7507 case OP_VDUP1: 7508 case OP_VDUP2: 7509 case OP_VDUP3: 7510 return true; 7511 } 7512 return false; 7513 } 7514 7515 /// isShuffleMaskLegal - Targets can use this to indicate that they only 7516 /// support *some* VECTOR_SHUFFLE operations, those with specific masks. 7517 /// By default, if a target supports the VECTOR_SHUFFLE node, all mask values 7518 /// are assumed to be legal. 7519 bool ARMTargetLowering::isShuffleMaskLegal(ArrayRef<int> M, EVT VT) const { 7520 if (VT.getVectorNumElements() == 4 && 7521 (VT.is128BitVector() || VT.is64BitVector())) { 7522 unsigned PFIndexes[4]; 7523 for (unsigned i = 0; i != 4; ++i) { 7524 if (M[i] < 0) 7525 PFIndexes[i] = 8; 7526 else 7527 PFIndexes[i] = M[i]; 7528 } 7529 7530 // Compute the index in the perfect shuffle table. 7531 unsigned PFTableIndex = 7532 PFIndexes[0]*9*9*9+PFIndexes[1]*9*9+PFIndexes[2]*9+PFIndexes[3]; 7533 unsigned PFEntry = PerfectShuffleTable[PFTableIndex]; 7534 unsigned Cost = (PFEntry >> 30); 7535 7536 if (Cost <= 4 && (Subtarget->hasNEON() || isLegalMVEShuffleOp(PFEntry))) 7537 return true; 7538 } 7539 7540 bool ReverseVEXT, isV_UNDEF; 7541 unsigned Imm, WhichResult; 7542 7543 unsigned EltSize = VT.getScalarSizeInBits(); 7544 if (EltSize >= 32 || 7545 ShuffleVectorSDNode::isSplatMask(&M[0], VT) || 7546 ShuffleVectorInst::isIdentityMask(M) || 7547 isVREVMask(M, VT, 64) || 7548 isVREVMask(M, VT, 32) || 7549 isVREVMask(M, VT, 16)) 7550 return true; 7551 else if (Subtarget->hasNEON() && 7552 (isVEXTMask(M, VT, ReverseVEXT, Imm) || 7553 isVTBLMask(M, VT) || 7554 isNEONTwoResultShuffleMask(M, VT, WhichResult, isV_UNDEF))) 7555 return true; 7556 else if (Subtarget->hasNEON() && (VT == MVT::v8i16 || VT == MVT::v16i8) && 7557 isReverseMask(M, VT)) 7558 return true; 7559 else if (Subtarget->hasMVEIntegerOps() && 7560 (isVMOVNMask(M, VT, 0) || isVMOVNMask(M, VT, 1))) 7561 return true; 7562 else 7563 return false; 7564 } 7565 7566 /// GeneratePerfectShuffle - Given an entry in the perfect-shuffle table, emit 7567 /// the specified operations to build the shuffle. 7568 static SDValue GeneratePerfectShuffle(unsigned PFEntry, SDValue LHS, 7569 SDValue RHS, SelectionDAG &DAG, 7570 const SDLoc &dl) { 7571 unsigned OpNum = (PFEntry >> 26) & 0x0F; 7572 unsigned LHSID = (PFEntry >> 13) & ((1 << 13)-1); 7573 unsigned RHSID = (PFEntry >> 0) & ((1 << 13)-1); 7574 7575 if (OpNum == OP_COPY) { 7576 if (LHSID == (1*9+2)*9+3) return LHS; 7577 assert(LHSID == ((4*9+5)*9+6)*9+7 && "Illegal OP_COPY!"); 7578 return RHS; 7579 } 7580 7581 SDValue OpLHS, OpRHS; 7582 OpLHS = GeneratePerfectShuffle(PerfectShuffleTable[LHSID], LHS, RHS, DAG, dl); 7583 OpRHS = GeneratePerfectShuffle(PerfectShuffleTable[RHSID], LHS, RHS, DAG, dl); 7584 EVT VT = OpLHS.getValueType(); 7585 7586 switch (OpNum) { 7587 default: llvm_unreachable("Unknown shuffle opcode!"); 7588 case OP_VREV: 7589 // VREV divides the vector in half and swaps within the half. 7590 if (VT.getVectorElementType() == MVT::i32 || 7591 VT.getVectorElementType() == MVT::f32) 7592 return DAG.getNode(ARMISD::VREV64, dl, VT, OpLHS); 7593 // vrev <4 x i16> -> VREV32 7594 if (VT.getVectorElementType() == MVT::i16) 7595 return DAG.getNode(ARMISD::VREV32, dl, VT, OpLHS); 7596 // vrev <4 x i8> -> VREV16 7597 assert(VT.getVectorElementType() == MVT::i8); 7598 return DAG.getNode(ARMISD::VREV16, dl, VT, OpLHS); 7599 case OP_VDUP0: 7600 case OP_VDUP1: 7601 case OP_VDUP2: 7602 case OP_VDUP3: 7603 return DAG.getNode(ARMISD::VDUPLANE, dl, VT, 7604 OpLHS, DAG.getConstant(OpNum-OP_VDUP0, dl, MVT::i32)); 7605 case OP_VEXT1: 7606 case OP_VEXT2: 7607 case OP_VEXT3: 7608 return DAG.getNode(ARMISD::VEXT, dl, VT, 7609 OpLHS, OpRHS, 7610 DAG.getConstant(OpNum - OP_VEXT1 + 1, dl, MVT::i32)); 7611 case OP_VUZPL: 7612 case OP_VUZPR: 7613 return DAG.getNode(ARMISD::VUZP, dl, DAG.getVTList(VT, VT), 7614 OpLHS, OpRHS).getValue(OpNum-OP_VUZPL); 7615 case OP_VZIPL: 7616 case OP_VZIPR: 7617 return DAG.getNode(ARMISD::VZIP, dl, DAG.getVTList(VT, VT), 7618 OpLHS, OpRHS).getValue(OpNum-OP_VZIPL); 7619 case OP_VTRNL: 7620 case OP_VTRNR: 7621 return DAG.getNode(ARMISD::VTRN, dl, DAG.getVTList(VT, VT), 7622 OpLHS, OpRHS).getValue(OpNum-OP_VTRNL); 7623 } 7624 } 7625 7626 static SDValue LowerVECTOR_SHUFFLEv8i8(SDValue Op, 7627 ArrayRef<int> ShuffleMask, 7628 SelectionDAG &DAG) { 7629 // Check to see if we can use the VTBL instruction. 7630 SDValue V1 = Op.getOperand(0); 7631 SDValue V2 = Op.getOperand(1); 7632 SDLoc DL(Op); 7633 7634 SmallVector<SDValue, 8> VTBLMask; 7635 for (ArrayRef<int>::iterator 7636 I = ShuffleMask.begin(), E = ShuffleMask.end(); I != E; ++I) 7637 VTBLMask.push_back(DAG.getConstant(*I, DL, MVT::i32)); 7638 7639 if (V2.getNode()->isUndef()) 7640 return DAG.getNode(ARMISD::VTBL1, DL, MVT::v8i8, V1, 7641 DAG.getBuildVector(MVT::v8i8, DL, VTBLMask)); 7642 7643 return DAG.getNode(ARMISD::VTBL2, DL, MVT::v8i8, V1, V2, 7644 DAG.getBuildVector(MVT::v8i8, DL, VTBLMask)); 7645 } 7646 7647 static SDValue LowerReverse_VECTOR_SHUFFLEv16i8_v8i16(SDValue Op, 7648 SelectionDAG &DAG) { 7649 SDLoc DL(Op); 7650 SDValue OpLHS = Op.getOperand(0); 7651 EVT VT = OpLHS.getValueType(); 7652 7653 assert((VT == MVT::v8i16 || VT == MVT::v16i8) && 7654 "Expect an v8i16/v16i8 type"); 7655 OpLHS = DAG.getNode(ARMISD::VREV64, DL, VT, OpLHS); 7656 // For a v16i8 type: After the VREV, we have got <8, ...15, 8, ..., 0>. Now, 7657 // extract the first 8 bytes into the top double word and the last 8 bytes 7658 // into the bottom double word. The v8i16 case is similar. 7659 unsigned ExtractNum = (VT == MVT::v16i8) ? 8 : 4; 7660 return DAG.getNode(ARMISD::VEXT, DL, VT, OpLHS, OpLHS, 7661 DAG.getConstant(ExtractNum, DL, MVT::i32)); 7662 } 7663 7664 static EVT getVectorTyFromPredicateVector(EVT VT) { 7665 switch (VT.getSimpleVT().SimpleTy) { 7666 case MVT::v4i1: 7667 return MVT::v4i32; 7668 case MVT::v8i1: 7669 return MVT::v8i16; 7670 case MVT::v16i1: 7671 return MVT::v16i8; 7672 default: 7673 llvm_unreachable("Unexpected vector predicate type"); 7674 } 7675 } 7676 7677 static SDValue PromoteMVEPredVector(SDLoc dl, SDValue Pred, EVT VT, 7678 SelectionDAG &DAG) { 7679 // Converting from boolean predicates to integers involves creating a vector 7680 // of all ones or all zeroes and selecting the lanes based upon the real 7681 // predicate. 7682 SDValue AllOnes = 7683 DAG.getTargetConstant(ARM_AM::createVMOVModImm(0xe, 0xff), dl, MVT::i32); 7684 AllOnes = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v16i8, AllOnes); 7685 7686 SDValue AllZeroes = 7687 DAG.getTargetConstant(ARM_AM::createVMOVModImm(0xe, 0x0), dl, MVT::i32); 7688 AllZeroes = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v16i8, AllZeroes); 7689 7690 // Get full vector type from predicate type 7691 EVT NewVT = getVectorTyFromPredicateVector(VT); 7692 7693 SDValue RecastV1; 7694 // If the real predicate is an v8i1 or v4i1 (not v16i1) then we need to recast 7695 // this to a v16i1. This cannot be done with an ordinary bitcast because the 7696 // sizes are not the same. We have to use a MVE specific PREDICATE_CAST node, 7697 // since we know in hardware the sizes are really the same. 7698 if (VT != MVT::v16i1) 7699 RecastV1 = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::v16i1, Pred); 7700 else 7701 RecastV1 = Pred; 7702 7703 // Select either all ones or zeroes depending upon the real predicate bits. 7704 SDValue PredAsVector = 7705 DAG.getNode(ISD::VSELECT, dl, MVT::v16i8, RecastV1, AllOnes, AllZeroes); 7706 7707 // Recast our new predicate-as-integer v16i8 vector into something 7708 // appropriate for the shuffle, i.e. v4i32 for a real v4i1 predicate. 7709 return DAG.getNode(ISD::BITCAST, dl, NewVT, PredAsVector); 7710 } 7711 7712 static SDValue LowerVECTOR_SHUFFLE_i1(SDValue Op, SelectionDAG &DAG, 7713 const ARMSubtarget *ST) { 7714 EVT VT = Op.getValueType(); 7715 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(Op.getNode()); 7716 ArrayRef<int> ShuffleMask = SVN->getMask(); 7717 7718 assert(ST->hasMVEIntegerOps() && 7719 "No support for vector shuffle of boolean predicates"); 7720 7721 SDValue V1 = Op.getOperand(0); 7722 SDLoc dl(Op); 7723 if (isReverseMask(ShuffleMask, VT)) { 7724 SDValue cast = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, V1); 7725 SDValue rbit = DAG.getNode(ISD::BITREVERSE, dl, MVT::i32, cast); 7726 SDValue srl = DAG.getNode(ISD::SRL, dl, MVT::i32, rbit, 7727 DAG.getConstant(16, dl, MVT::i32)); 7728 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, VT, srl); 7729 } 7730 7731 // Until we can come up with optimised cases for every single vector 7732 // shuffle in existence we have chosen the least painful strategy. This is 7733 // to essentially promote the boolean predicate to a 8-bit integer, where 7734 // each predicate represents a byte. Then we fall back on a normal integer 7735 // vector shuffle and convert the result back into a predicate vector. In 7736 // many cases the generated code might be even better than scalar code 7737 // operating on bits. Just imagine trying to shuffle 8 arbitrary 2-bit 7738 // fields in a register into 8 other arbitrary 2-bit fields! 7739 SDValue PredAsVector = PromoteMVEPredVector(dl, V1, VT, DAG); 7740 EVT NewVT = PredAsVector.getValueType(); 7741 7742 // Do the shuffle! 7743 SDValue Shuffled = DAG.getVectorShuffle(NewVT, dl, PredAsVector, 7744 DAG.getUNDEF(NewVT), ShuffleMask); 7745 7746 // Now return the result of comparing the shuffled vector with zero, 7747 // which will generate a real predicate, i.e. v4i1, v8i1 or v16i1. 7748 return DAG.getNode(ARMISD::VCMPZ, dl, VT, Shuffled, 7749 DAG.getConstant(ARMCC::NE, dl, MVT::i32)); 7750 } 7751 7752 static SDValue LowerVECTOR_SHUFFLE(SDValue Op, SelectionDAG &DAG, 7753 const ARMSubtarget *ST) { 7754 SDValue V1 = Op.getOperand(0); 7755 SDValue V2 = Op.getOperand(1); 7756 SDLoc dl(Op); 7757 EVT VT = Op.getValueType(); 7758 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(Op.getNode()); 7759 unsigned EltSize = VT.getScalarSizeInBits(); 7760 7761 if (ST->hasMVEIntegerOps() && EltSize == 1) 7762 return LowerVECTOR_SHUFFLE_i1(Op, DAG, ST); 7763 7764 // Convert shuffles that are directly supported on NEON to target-specific 7765 // DAG nodes, instead of keeping them as shuffles and matching them again 7766 // during code selection. This is more efficient and avoids the possibility 7767 // of inconsistencies between legalization and selection. 7768 // FIXME: floating-point vectors should be canonicalized to integer vectors 7769 // of the same time so that they get CSEd properly. 7770 ArrayRef<int> ShuffleMask = SVN->getMask(); 7771 7772 if (EltSize <= 32) { 7773 if (SVN->isSplat()) { 7774 int Lane = SVN->getSplatIndex(); 7775 // If this is undef splat, generate it via "just" vdup, if possible. 7776 if (Lane == -1) Lane = 0; 7777 7778 // Test if V1 is a SCALAR_TO_VECTOR. 7779 if (Lane == 0 && V1.getOpcode() == ISD::SCALAR_TO_VECTOR) { 7780 return DAG.getNode(ARMISD::VDUP, dl, VT, V1.getOperand(0)); 7781 } 7782 // Test if V1 is a BUILD_VECTOR which is equivalent to a SCALAR_TO_VECTOR 7783 // (and probably will turn into a SCALAR_TO_VECTOR once legalization 7784 // reaches it). 7785 if (Lane == 0 && V1.getOpcode() == ISD::BUILD_VECTOR && 7786 !isa<ConstantSDNode>(V1.getOperand(0))) { 7787 bool IsScalarToVector = true; 7788 for (unsigned i = 1, e = V1.getNumOperands(); i != e; ++i) 7789 if (!V1.getOperand(i).isUndef()) { 7790 IsScalarToVector = false; 7791 break; 7792 } 7793 if (IsScalarToVector) 7794 return DAG.getNode(ARMISD::VDUP, dl, VT, V1.getOperand(0)); 7795 } 7796 return DAG.getNode(ARMISD::VDUPLANE, dl, VT, V1, 7797 DAG.getConstant(Lane, dl, MVT::i32)); 7798 } 7799 7800 bool ReverseVEXT = false; 7801 unsigned Imm = 0; 7802 if (ST->hasNEON() && isVEXTMask(ShuffleMask, VT, ReverseVEXT, Imm)) { 7803 if (ReverseVEXT) 7804 std::swap(V1, V2); 7805 return DAG.getNode(ARMISD::VEXT, dl, VT, V1, V2, 7806 DAG.getConstant(Imm, dl, MVT::i32)); 7807 } 7808 7809 if (isVREVMask(ShuffleMask, VT, 64)) 7810 return DAG.getNode(ARMISD::VREV64, dl, VT, V1); 7811 if (isVREVMask(ShuffleMask, VT, 32)) 7812 return DAG.getNode(ARMISD::VREV32, dl, VT, V1); 7813 if (isVREVMask(ShuffleMask, VT, 16)) 7814 return DAG.getNode(ARMISD::VREV16, dl, VT, V1); 7815 7816 if (ST->hasNEON() && V2->isUndef() && isSingletonVEXTMask(ShuffleMask, VT, Imm)) { 7817 return DAG.getNode(ARMISD::VEXT, dl, VT, V1, V1, 7818 DAG.getConstant(Imm, dl, MVT::i32)); 7819 } 7820 7821 // Check for Neon shuffles that modify both input vectors in place. 7822 // If both results are used, i.e., if there are two shuffles with the same 7823 // source operands and with masks corresponding to both results of one of 7824 // these operations, DAG memoization will ensure that a single node is 7825 // used for both shuffles. 7826 unsigned WhichResult = 0; 7827 bool isV_UNDEF = false; 7828 if (ST->hasNEON()) { 7829 if (unsigned ShuffleOpc = isNEONTwoResultShuffleMask( 7830 ShuffleMask, VT, WhichResult, isV_UNDEF)) { 7831 if (isV_UNDEF) 7832 V2 = V1; 7833 return DAG.getNode(ShuffleOpc, dl, DAG.getVTList(VT, VT), V1, V2) 7834 .getValue(WhichResult); 7835 } 7836 } 7837 if (ST->hasMVEIntegerOps()) { 7838 if (isVMOVNMask(ShuffleMask, VT, 0)) 7839 return DAG.getNode(ARMISD::VMOVN, dl, VT, V2, V1, 7840 DAG.getConstant(0, dl, MVT::i32)); 7841 if (isVMOVNMask(ShuffleMask, VT, 1)) 7842 return DAG.getNode(ARMISD::VMOVN, dl, VT, V1, V2, 7843 DAG.getConstant(1, dl, MVT::i32)); 7844 } 7845 7846 // Also check for these shuffles through CONCAT_VECTORS: we canonicalize 7847 // shuffles that produce a result larger than their operands with: 7848 // shuffle(concat(v1, undef), concat(v2, undef)) 7849 // -> 7850 // shuffle(concat(v1, v2), undef) 7851 // because we can access quad vectors (see PerformVECTOR_SHUFFLECombine). 7852 // 7853 // This is useful in the general case, but there are special cases where 7854 // native shuffles produce larger results: the two-result ops. 7855 // 7856 // Look through the concat when lowering them: 7857 // shuffle(concat(v1, v2), undef) 7858 // -> 7859 // concat(VZIP(v1, v2):0, :1) 7860 // 7861 if (ST->hasNEON() && V1->getOpcode() == ISD::CONCAT_VECTORS && V2->isUndef()) { 7862 SDValue SubV1 = V1->getOperand(0); 7863 SDValue SubV2 = V1->getOperand(1); 7864 EVT SubVT = SubV1.getValueType(); 7865 7866 // We expect these to have been canonicalized to -1. 7867 assert(llvm::all_of(ShuffleMask, [&](int i) { 7868 return i < (int)VT.getVectorNumElements(); 7869 }) && "Unexpected shuffle index into UNDEF operand!"); 7870 7871 if (unsigned ShuffleOpc = isNEONTwoResultShuffleMask( 7872 ShuffleMask, SubVT, WhichResult, isV_UNDEF)) { 7873 if (isV_UNDEF) 7874 SubV2 = SubV1; 7875 assert((WhichResult == 0) && 7876 "In-place shuffle of concat can only have one result!"); 7877 SDValue Res = DAG.getNode(ShuffleOpc, dl, DAG.getVTList(SubVT, SubVT), 7878 SubV1, SubV2); 7879 return DAG.getNode(ISD::CONCAT_VECTORS, dl, VT, Res.getValue(0), 7880 Res.getValue(1)); 7881 } 7882 } 7883 } 7884 7885 // If the shuffle is not directly supported and it has 4 elements, use 7886 // the PerfectShuffle-generated table to synthesize it from other shuffles. 7887 unsigned NumElts = VT.getVectorNumElements(); 7888 if (NumElts == 4) { 7889 unsigned PFIndexes[4]; 7890 for (unsigned i = 0; i != 4; ++i) { 7891 if (ShuffleMask[i] < 0) 7892 PFIndexes[i] = 8; 7893 else 7894 PFIndexes[i] = ShuffleMask[i]; 7895 } 7896 7897 // Compute the index in the perfect shuffle table. 7898 unsigned PFTableIndex = 7899 PFIndexes[0]*9*9*9+PFIndexes[1]*9*9+PFIndexes[2]*9+PFIndexes[3]; 7900 unsigned PFEntry = PerfectShuffleTable[PFTableIndex]; 7901 unsigned Cost = (PFEntry >> 30); 7902 7903 if (Cost <= 4) { 7904 if (ST->hasNEON()) 7905 return GeneratePerfectShuffle(PFEntry, V1, V2, DAG, dl); 7906 else if (isLegalMVEShuffleOp(PFEntry)) { 7907 unsigned LHSID = (PFEntry >> 13) & ((1 << 13)-1); 7908 unsigned RHSID = (PFEntry >> 0) & ((1 << 13)-1); 7909 unsigned PFEntryLHS = PerfectShuffleTable[LHSID]; 7910 unsigned PFEntryRHS = PerfectShuffleTable[RHSID]; 7911 if (isLegalMVEShuffleOp(PFEntryLHS) && isLegalMVEShuffleOp(PFEntryRHS)) 7912 return GeneratePerfectShuffle(PFEntry, V1, V2, DAG, dl); 7913 } 7914 } 7915 } 7916 7917 // Implement shuffles with 32- or 64-bit elements as ARMISD::BUILD_VECTORs. 7918 if (EltSize >= 32) { 7919 // Do the expansion with floating-point types, since that is what the VFP 7920 // registers are defined to use, and since i64 is not legal. 7921 EVT EltVT = EVT::getFloatingPointVT(EltSize); 7922 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), EltVT, NumElts); 7923 V1 = DAG.getNode(ISD::BITCAST, dl, VecVT, V1); 7924 V2 = DAG.getNode(ISD::BITCAST, dl, VecVT, V2); 7925 SmallVector<SDValue, 8> Ops; 7926 for (unsigned i = 0; i < NumElts; ++i) { 7927 if (ShuffleMask[i] < 0) 7928 Ops.push_back(DAG.getUNDEF(EltVT)); 7929 else 7930 Ops.push_back(DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, 7931 ShuffleMask[i] < (int)NumElts ? V1 : V2, 7932 DAG.getConstant(ShuffleMask[i] & (NumElts-1), 7933 dl, MVT::i32))); 7934 } 7935 SDValue Val = DAG.getNode(ARMISD::BUILD_VECTOR, dl, VecVT, Ops); 7936 return DAG.getNode(ISD::BITCAST, dl, VT, Val); 7937 } 7938 7939 if (ST->hasNEON() && (VT == MVT::v8i16 || VT == MVT::v16i8) && isReverseMask(ShuffleMask, VT)) 7940 return LowerReverse_VECTOR_SHUFFLEv16i8_v8i16(Op, DAG); 7941 7942 if (ST->hasNEON() && VT == MVT::v8i8) 7943 if (SDValue NewOp = LowerVECTOR_SHUFFLEv8i8(Op, ShuffleMask, DAG)) 7944 return NewOp; 7945 7946 return SDValue(); 7947 } 7948 7949 static SDValue LowerINSERT_VECTOR_ELT_i1(SDValue Op, SelectionDAG &DAG, 7950 const ARMSubtarget *ST) { 7951 EVT VecVT = Op.getOperand(0).getValueType(); 7952 SDLoc dl(Op); 7953 7954 assert(ST->hasMVEIntegerOps() && 7955 "LowerINSERT_VECTOR_ELT_i1 called without MVE!"); 7956 7957 SDValue Conv = 7958 DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, Op->getOperand(0)); 7959 unsigned Lane = cast<ConstantSDNode>(Op.getOperand(2))->getZExtValue(); 7960 unsigned LaneWidth = 7961 getVectorTyFromPredicateVector(VecVT).getScalarSizeInBits() / 8; 7962 unsigned Mask = ((1 << LaneWidth) - 1) << Lane * LaneWidth; 7963 SDValue Ext = DAG.getNode(ISD::SIGN_EXTEND_INREG, dl, MVT::i32, 7964 Op.getOperand(1), DAG.getValueType(MVT::i1)); 7965 SDValue BFI = DAG.getNode(ARMISD::BFI, dl, MVT::i32, Conv, Ext, 7966 DAG.getConstant(~Mask, dl, MVT::i32)); 7967 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, Op.getValueType(), BFI); 7968 } 7969 7970 SDValue ARMTargetLowering::LowerINSERT_VECTOR_ELT(SDValue Op, 7971 SelectionDAG &DAG) const { 7972 // INSERT_VECTOR_ELT is legal only for immediate indexes. 7973 SDValue Lane = Op.getOperand(2); 7974 if (!isa<ConstantSDNode>(Lane)) 7975 return SDValue(); 7976 7977 SDValue Elt = Op.getOperand(1); 7978 EVT EltVT = Elt.getValueType(); 7979 7980 if (Subtarget->hasMVEIntegerOps() && 7981 Op.getValueType().getScalarSizeInBits() == 1) 7982 return LowerINSERT_VECTOR_ELT_i1(Op, DAG, Subtarget); 7983 7984 if (getTypeAction(*DAG.getContext(), EltVT) == 7985 TargetLowering::TypePromoteFloat) { 7986 // INSERT_VECTOR_ELT doesn't want f16 operands promoting to f32, 7987 // but the type system will try to do that if we don't intervene. 7988 // Reinterpret any such vector-element insertion as one with the 7989 // corresponding integer types. 7990 7991 SDLoc dl(Op); 7992 7993 EVT IEltVT = MVT::getIntegerVT(EltVT.getScalarSizeInBits()); 7994 assert(getTypeAction(*DAG.getContext(), IEltVT) != 7995 TargetLowering::TypePromoteFloat); 7996 7997 SDValue VecIn = Op.getOperand(0); 7998 EVT VecVT = VecIn.getValueType(); 7999 EVT IVecVT = EVT::getVectorVT(*DAG.getContext(), IEltVT, 8000 VecVT.getVectorNumElements()); 8001 8002 SDValue IElt = DAG.getNode(ISD::BITCAST, dl, IEltVT, Elt); 8003 SDValue IVecIn = DAG.getNode(ISD::BITCAST, dl, IVecVT, VecIn); 8004 SDValue IVecOut = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, IVecVT, 8005 IVecIn, IElt, Lane); 8006 return DAG.getNode(ISD::BITCAST, dl, VecVT, IVecOut); 8007 } 8008 8009 return Op; 8010 } 8011 8012 static SDValue LowerEXTRACT_VECTOR_ELT_i1(SDValue Op, SelectionDAG &DAG, 8013 const ARMSubtarget *ST) { 8014 EVT VecVT = Op.getOperand(0).getValueType(); 8015 SDLoc dl(Op); 8016 8017 assert(ST->hasMVEIntegerOps() && 8018 "LowerINSERT_VECTOR_ELT_i1 called without MVE!"); 8019 8020 SDValue Conv = 8021 DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, Op->getOperand(0)); 8022 unsigned Lane = cast<ConstantSDNode>(Op.getOperand(1))->getZExtValue(); 8023 unsigned LaneWidth = 8024 getVectorTyFromPredicateVector(VecVT).getScalarSizeInBits() / 8; 8025 SDValue Shift = DAG.getNode(ISD::SRL, dl, MVT::i32, Conv, 8026 DAG.getConstant(Lane * LaneWidth, dl, MVT::i32)); 8027 return Shift; 8028 } 8029 8030 static SDValue LowerEXTRACT_VECTOR_ELT(SDValue Op, SelectionDAG &DAG, 8031 const ARMSubtarget *ST) { 8032 // EXTRACT_VECTOR_ELT is legal only for immediate indexes. 8033 SDValue Lane = Op.getOperand(1); 8034 if (!isa<ConstantSDNode>(Lane)) 8035 return SDValue(); 8036 8037 SDValue Vec = Op.getOperand(0); 8038 EVT VT = Vec.getValueType(); 8039 8040 if (ST->hasMVEIntegerOps() && VT.getScalarSizeInBits() == 1) 8041 return LowerEXTRACT_VECTOR_ELT_i1(Op, DAG, ST); 8042 8043 if (Op.getValueType() == MVT::i32 && Vec.getScalarValueSizeInBits() < 32) { 8044 SDLoc dl(Op); 8045 return DAG.getNode(ARMISD::VGETLANEu, dl, MVT::i32, Vec, Lane); 8046 } 8047 8048 return Op; 8049 } 8050 8051 static SDValue LowerCONCAT_VECTORS_i1(SDValue Op, SelectionDAG &DAG, 8052 const ARMSubtarget *ST) { 8053 SDValue V1 = Op.getOperand(0); 8054 SDValue V2 = Op.getOperand(1); 8055 SDLoc dl(Op); 8056 EVT VT = Op.getValueType(); 8057 EVT Op1VT = V1.getValueType(); 8058 EVT Op2VT = V2.getValueType(); 8059 unsigned NumElts = VT.getVectorNumElements(); 8060 8061 assert(Op1VT == Op2VT && "Operand types don't match!"); 8062 assert(VT.getScalarSizeInBits() == 1 && 8063 "Unexpected custom CONCAT_VECTORS lowering"); 8064 assert(ST->hasMVEIntegerOps() && 8065 "CONCAT_VECTORS lowering only supported for MVE"); 8066 8067 SDValue NewV1 = PromoteMVEPredVector(dl, V1, Op1VT, DAG); 8068 SDValue NewV2 = PromoteMVEPredVector(dl, V2, Op2VT, DAG); 8069 8070 // We now have Op1 + Op2 promoted to vectors of integers, where v8i1 gets 8071 // promoted to v8i16, etc. 8072 8073 MVT ElType = getVectorTyFromPredicateVector(VT).getScalarType().getSimpleVT(); 8074 8075 // Extract the vector elements from Op1 and Op2 one by one and truncate them 8076 // to be the right size for the destination. For example, if Op1 is v4i1 then 8077 // the promoted vector is v4i32. The result of concatentation gives a v8i1, 8078 // which when promoted is v8i16. That means each i32 element from Op1 needs 8079 // truncating to i16 and inserting in the result. 8080 EVT ConcatVT = MVT::getVectorVT(ElType, NumElts); 8081 SDValue ConVec = DAG.getNode(ISD::UNDEF, dl, ConcatVT); 8082 auto ExractInto = [&DAG, &dl](SDValue NewV, SDValue ConVec, unsigned &j) { 8083 EVT NewVT = NewV.getValueType(); 8084 EVT ConcatVT = ConVec.getValueType(); 8085 for (unsigned i = 0, e = NewVT.getVectorNumElements(); i < e; i++, j++) { 8086 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::i32, NewV, 8087 DAG.getIntPtrConstant(i, dl)); 8088 ConVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, ConcatVT, ConVec, Elt, 8089 DAG.getConstant(j, dl, MVT::i32)); 8090 } 8091 return ConVec; 8092 }; 8093 unsigned j = 0; 8094 ConVec = ExractInto(NewV1, ConVec, j); 8095 ConVec = ExractInto(NewV2, ConVec, j); 8096 8097 // Now return the result of comparing the subvector with zero, 8098 // which will generate a real predicate, i.e. v4i1, v8i1 or v16i1. 8099 return DAG.getNode(ARMISD::VCMPZ, dl, VT, ConVec, 8100 DAG.getConstant(ARMCC::NE, dl, MVT::i32)); 8101 } 8102 8103 static SDValue LowerCONCAT_VECTORS(SDValue Op, SelectionDAG &DAG, 8104 const ARMSubtarget *ST) { 8105 EVT VT = Op->getValueType(0); 8106 if (ST->hasMVEIntegerOps() && VT.getScalarSizeInBits() == 1) 8107 return LowerCONCAT_VECTORS_i1(Op, DAG, ST); 8108 8109 // The only time a CONCAT_VECTORS operation can have legal types is when 8110 // two 64-bit vectors are concatenated to a 128-bit vector. 8111 assert(Op.getValueType().is128BitVector() && Op.getNumOperands() == 2 && 8112 "unexpected CONCAT_VECTORS"); 8113 SDLoc dl(Op); 8114 SDValue Val = DAG.getUNDEF(MVT::v2f64); 8115 SDValue Op0 = Op.getOperand(0); 8116 SDValue Op1 = Op.getOperand(1); 8117 if (!Op0.isUndef()) 8118 Val = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Val, 8119 DAG.getNode(ISD::BITCAST, dl, MVT::f64, Op0), 8120 DAG.getIntPtrConstant(0, dl)); 8121 if (!Op1.isUndef()) 8122 Val = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Val, 8123 DAG.getNode(ISD::BITCAST, dl, MVT::f64, Op1), 8124 DAG.getIntPtrConstant(1, dl)); 8125 return DAG.getNode(ISD::BITCAST, dl, Op.getValueType(), Val); 8126 } 8127 8128 static SDValue LowerEXTRACT_SUBVECTOR(SDValue Op, SelectionDAG &DAG, 8129 const ARMSubtarget *ST) { 8130 SDValue V1 = Op.getOperand(0); 8131 SDValue V2 = Op.getOperand(1); 8132 SDLoc dl(Op); 8133 EVT VT = Op.getValueType(); 8134 EVT Op1VT = V1.getValueType(); 8135 unsigned NumElts = VT.getVectorNumElements(); 8136 unsigned Index = cast<ConstantSDNode>(V2)->getZExtValue(); 8137 8138 assert(VT.getScalarSizeInBits() == 1 && 8139 "Unexpected custom EXTRACT_SUBVECTOR lowering"); 8140 assert(ST->hasMVEIntegerOps() && 8141 "EXTRACT_SUBVECTOR lowering only supported for MVE"); 8142 8143 SDValue NewV1 = PromoteMVEPredVector(dl, V1, Op1VT, DAG); 8144 8145 // We now have Op1 promoted to a vector of integers, where v8i1 gets 8146 // promoted to v8i16, etc. 8147 8148 MVT ElType = getVectorTyFromPredicateVector(VT).getScalarType().getSimpleVT(); 8149 8150 EVT SubVT = MVT::getVectorVT(ElType, NumElts); 8151 SDValue SubVec = DAG.getNode(ISD::UNDEF, dl, SubVT); 8152 for (unsigned i = Index, j = 0; i < (Index + NumElts); i++, j++) { 8153 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::i32, NewV1, 8154 DAG.getIntPtrConstant(i, dl)); 8155 SubVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, SubVT, SubVec, Elt, 8156 DAG.getConstant(j, dl, MVT::i32)); 8157 } 8158 8159 // Now return the result of comparing the subvector with zero, 8160 // which will generate a real predicate, i.e. v4i1, v8i1 or v16i1. 8161 return DAG.getNode(ARMISD::VCMPZ, dl, VT, SubVec, 8162 DAG.getConstant(ARMCC::NE, dl, MVT::i32)); 8163 } 8164 8165 /// isExtendedBUILD_VECTOR - Check if N is a constant BUILD_VECTOR where each 8166 /// element has been zero/sign-extended, depending on the isSigned parameter, 8167 /// from an integer type half its size. 8168 static bool isExtendedBUILD_VECTOR(SDNode *N, SelectionDAG &DAG, 8169 bool isSigned) { 8170 // A v2i64 BUILD_VECTOR will have been legalized to a BITCAST from v4i32. 8171 EVT VT = N->getValueType(0); 8172 if (VT == MVT::v2i64 && N->getOpcode() == ISD::BITCAST) { 8173 SDNode *BVN = N->getOperand(0).getNode(); 8174 if (BVN->getValueType(0) != MVT::v4i32 || 8175 BVN->getOpcode() != ISD::BUILD_VECTOR) 8176 return false; 8177 unsigned LoElt = DAG.getDataLayout().isBigEndian() ? 1 : 0; 8178 unsigned HiElt = 1 - LoElt; 8179 ConstantSDNode *Lo0 = dyn_cast<ConstantSDNode>(BVN->getOperand(LoElt)); 8180 ConstantSDNode *Hi0 = dyn_cast<ConstantSDNode>(BVN->getOperand(HiElt)); 8181 ConstantSDNode *Lo1 = dyn_cast<ConstantSDNode>(BVN->getOperand(LoElt+2)); 8182 ConstantSDNode *Hi1 = dyn_cast<ConstantSDNode>(BVN->getOperand(HiElt+2)); 8183 if (!Lo0 || !Hi0 || !Lo1 || !Hi1) 8184 return false; 8185 if (isSigned) { 8186 if (Hi0->getSExtValue() == Lo0->getSExtValue() >> 32 && 8187 Hi1->getSExtValue() == Lo1->getSExtValue() >> 32) 8188 return true; 8189 } else { 8190 if (Hi0->isNullValue() && Hi1->isNullValue()) 8191 return true; 8192 } 8193 return false; 8194 } 8195 8196 if (N->getOpcode() != ISD::BUILD_VECTOR) 8197 return false; 8198 8199 for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) { 8200 SDNode *Elt = N->getOperand(i).getNode(); 8201 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Elt)) { 8202 unsigned EltSize = VT.getScalarSizeInBits(); 8203 unsigned HalfSize = EltSize / 2; 8204 if (isSigned) { 8205 if (!isIntN(HalfSize, C->getSExtValue())) 8206 return false; 8207 } else { 8208 if (!isUIntN(HalfSize, C->getZExtValue())) 8209 return false; 8210 } 8211 continue; 8212 } 8213 return false; 8214 } 8215 8216 return true; 8217 } 8218 8219 /// isSignExtended - Check if a node is a vector value that is sign-extended 8220 /// or a constant BUILD_VECTOR with sign-extended elements. 8221 static bool isSignExtended(SDNode *N, SelectionDAG &DAG) { 8222 if (N->getOpcode() == ISD::SIGN_EXTEND || ISD::isSEXTLoad(N)) 8223 return true; 8224 if (isExtendedBUILD_VECTOR(N, DAG, true)) 8225 return true; 8226 return false; 8227 } 8228 8229 /// isZeroExtended - Check if a node is a vector value that is zero-extended 8230 /// or a constant BUILD_VECTOR with zero-extended elements. 8231 static bool isZeroExtended(SDNode *N, SelectionDAG &DAG) { 8232 if (N->getOpcode() == ISD::ZERO_EXTEND || ISD::isZEXTLoad(N)) 8233 return true; 8234 if (isExtendedBUILD_VECTOR(N, DAG, false)) 8235 return true; 8236 return false; 8237 } 8238 8239 static EVT getExtensionTo64Bits(const EVT &OrigVT) { 8240 if (OrigVT.getSizeInBits() >= 64) 8241 return OrigVT; 8242 8243 assert(OrigVT.isSimple() && "Expecting a simple value type"); 8244 8245 MVT::SimpleValueType OrigSimpleTy = OrigVT.getSimpleVT().SimpleTy; 8246 switch (OrigSimpleTy) { 8247 default: llvm_unreachable("Unexpected Vector Type"); 8248 case MVT::v2i8: 8249 case MVT::v2i16: 8250 return MVT::v2i32; 8251 case MVT::v4i8: 8252 return MVT::v4i16; 8253 } 8254 } 8255 8256 /// AddRequiredExtensionForVMULL - Add a sign/zero extension to extend the total 8257 /// value size to 64 bits. We need a 64-bit D register as an operand to VMULL. 8258 /// We insert the required extension here to get the vector to fill a D register. 8259 static SDValue AddRequiredExtensionForVMULL(SDValue N, SelectionDAG &DAG, 8260 const EVT &OrigTy, 8261 const EVT &ExtTy, 8262 unsigned ExtOpcode) { 8263 // The vector originally had a size of OrigTy. It was then extended to ExtTy. 8264 // We expect the ExtTy to be 128-bits total. If the OrigTy is less than 8265 // 64-bits we need to insert a new extension so that it will be 64-bits. 8266 assert(ExtTy.is128BitVector() && "Unexpected extension size"); 8267 if (OrigTy.getSizeInBits() >= 64) 8268 return N; 8269 8270 // Must extend size to at least 64 bits to be used as an operand for VMULL. 8271 EVT NewVT = getExtensionTo64Bits(OrigTy); 8272 8273 return DAG.getNode(ExtOpcode, SDLoc(N), NewVT, N); 8274 } 8275 8276 /// SkipLoadExtensionForVMULL - return a load of the original vector size that 8277 /// does not do any sign/zero extension. If the original vector is less 8278 /// than 64 bits, an appropriate extension will be added after the load to 8279 /// reach a total size of 64 bits. We have to add the extension separately 8280 /// because ARM does not have a sign/zero extending load for vectors. 8281 static SDValue SkipLoadExtensionForVMULL(LoadSDNode *LD, SelectionDAG& DAG) { 8282 EVT ExtendedTy = getExtensionTo64Bits(LD->getMemoryVT()); 8283 8284 // The load already has the right type. 8285 if (ExtendedTy == LD->getMemoryVT()) 8286 return DAG.getLoad(LD->getMemoryVT(), SDLoc(LD), LD->getChain(), 8287 LD->getBasePtr(), LD->getPointerInfo(), 8288 LD->getAlignment(), LD->getMemOperand()->getFlags()); 8289 8290 // We need to create a zextload/sextload. We cannot just create a load 8291 // followed by a zext/zext node because LowerMUL is also run during normal 8292 // operation legalization where we can't create illegal types. 8293 return DAG.getExtLoad(LD->getExtensionType(), SDLoc(LD), ExtendedTy, 8294 LD->getChain(), LD->getBasePtr(), LD->getPointerInfo(), 8295 LD->getMemoryVT(), LD->getAlignment(), 8296 LD->getMemOperand()->getFlags()); 8297 } 8298 8299 /// SkipExtensionForVMULL - For a node that is a SIGN_EXTEND, ZERO_EXTEND, 8300 /// extending load, or BUILD_VECTOR with extended elements, return the 8301 /// unextended value. The unextended vector should be 64 bits so that it can 8302 /// be used as an operand to a VMULL instruction. If the original vector size 8303 /// before extension is less than 64 bits we add a an extension to resize 8304 /// the vector to 64 bits. 8305 static SDValue SkipExtensionForVMULL(SDNode *N, SelectionDAG &DAG) { 8306 if (N->getOpcode() == ISD::SIGN_EXTEND || N->getOpcode() == ISD::ZERO_EXTEND) 8307 return AddRequiredExtensionForVMULL(N->getOperand(0), DAG, 8308 N->getOperand(0)->getValueType(0), 8309 N->getValueType(0), 8310 N->getOpcode()); 8311 8312 if (LoadSDNode *LD = dyn_cast<LoadSDNode>(N)) { 8313 assert((ISD::isSEXTLoad(LD) || ISD::isZEXTLoad(LD)) && 8314 "Expected extending load"); 8315 8316 SDValue newLoad = SkipLoadExtensionForVMULL(LD, DAG); 8317 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), newLoad.getValue(1)); 8318 unsigned Opcode = ISD::isSEXTLoad(LD) ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND; 8319 SDValue extLoad = 8320 DAG.getNode(Opcode, SDLoc(newLoad), LD->getValueType(0), newLoad); 8321 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 0), extLoad); 8322 8323 return newLoad; 8324 } 8325 8326 // Otherwise, the value must be a BUILD_VECTOR. For v2i64, it will 8327 // have been legalized as a BITCAST from v4i32. 8328 if (N->getOpcode() == ISD::BITCAST) { 8329 SDNode *BVN = N->getOperand(0).getNode(); 8330 assert(BVN->getOpcode() == ISD::BUILD_VECTOR && 8331 BVN->getValueType(0) == MVT::v4i32 && "expected v4i32 BUILD_VECTOR"); 8332 unsigned LowElt = DAG.getDataLayout().isBigEndian() ? 1 : 0; 8333 return DAG.getBuildVector( 8334 MVT::v2i32, SDLoc(N), 8335 {BVN->getOperand(LowElt), BVN->getOperand(LowElt + 2)}); 8336 } 8337 // Construct a new BUILD_VECTOR with elements truncated to half the size. 8338 assert(N->getOpcode() == ISD::BUILD_VECTOR && "expected BUILD_VECTOR"); 8339 EVT VT = N->getValueType(0); 8340 unsigned EltSize = VT.getScalarSizeInBits() / 2; 8341 unsigned NumElts = VT.getVectorNumElements(); 8342 MVT TruncVT = MVT::getIntegerVT(EltSize); 8343 SmallVector<SDValue, 8> Ops; 8344 SDLoc dl(N); 8345 for (unsigned i = 0; i != NumElts; ++i) { 8346 ConstantSDNode *C = cast<ConstantSDNode>(N->getOperand(i)); 8347 const APInt &CInt = C->getAPIntValue(); 8348 // Element types smaller than 32 bits are not legal, so use i32 elements. 8349 // The values are implicitly truncated so sext vs. zext doesn't matter. 8350 Ops.push_back(DAG.getConstant(CInt.zextOrTrunc(32), dl, MVT::i32)); 8351 } 8352 return DAG.getBuildVector(MVT::getVectorVT(TruncVT, NumElts), dl, Ops); 8353 } 8354 8355 static bool isAddSubSExt(SDNode *N, SelectionDAG &DAG) { 8356 unsigned Opcode = N->getOpcode(); 8357 if (Opcode == ISD::ADD || Opcode == ISD::SUB) { 8358 SDNode *N0 = N->getOperand(0).getNode(); 8359 SDNode *N1 = N->getOperand(1).getNode(); 8360 return N0->hasOneUse() && N1->hasOneUse() && 8361 isSignExtended(N0, DAG) && isSignExtended(N1, DAG); 8362 } 8363 return false; 8364 } 8365 8366 static bool isAddSubZExt(SDNode *N, SelectionDAG &DAG) { 8367 unsigned Opcode = N->getOpcode(); 8368 if (Opcode == ISD::ADD || Opcode == ISD::SUB) { 8369 SDNode *N0 = N->getOperand(0).getNode(); 8370 SDNode *N1 = N->getOperand(1).getNode(); 8371 return N0->hasOneUse() && N1->hasOneUse() && 8372 isZeroExtended(N0, DAG) && isZeroExtended(N1, DAG); 8373 } 8374 return false; 8375 } 8376 8377 static SDValue LowerMUL(SDValue Op, SelectionDAG &DAG) { 8378 // Multiplications are only custom-lowered for 128-bit vectors so that 8379 // VMULL can be detected. Otherwise v2i64 multiplications are not legal. 8380 EVT VT = Op.getValueType(); 8381 assert(VT.is128BitVector() && VT.isInteger() && 8382 "unexpected type for custom-lowering ISD::MUL"); 8383 SDNode *N0 = Op.getOperand(0).getNode(); 8384 SDNode *N1 = Op.getOperand(1).getNode(); 8385 unsigned NewOpc = 0; 8386 bool isMLA = false; 8387 bool isN0SExt = isSignExtended(N0, DAG); 8388 bool isN1SExt = isSignExtended(N1, DAG); 8389 if (isN0SExt && isN1SExt) 8390 NewOpc = ARMISD::VMULLs; 8391 else { 8392 bool isN0ZExt = isZeroExtended(N0, DAG); 8393 bool isN1ZExt = isZeroExtended(N1, DAG); 8394 if (isN0ZExt && isN1ZExt) 8395 NewOpc = ARMISD::VMULLu; 8396 else if (isN1SExt || isN1ZExt) { 8397 // Look for (s/zext A + s/zext B) * (s/zext C). We want to turn these 8398 // into (s/zext A * s/zext C) + (s/zext B * s/zext C) 8399 if (isN1SExt && isAddSubSExt(N0, DAG)) { 8400 NewOpc = ARMISD::VMULLs; 8401 isMLA = true; 8402 } else if (isN1ZExt && isAddSubZExt(N0, DAG)) { 8403 NewOpc = ARMISD::VMULLu; 8404 isMLA = true; 8405 } else if (isN0ZExt && isAddSubZExt(N1, DAG)) { 8406 std::swap(N0, N1); 8407 NewOpc = ARMISD::VMULLu; 8408 isMLA = true; 8409 } 8410 } 8411 8412 if (!NewOpc) { 8413 if (VT == MVT::v2i64) 8414 // Fall through to expand this. It is not legal. 8415 return SDValue(); 8416 else 8417 // Other vector multiplications are legal. 8418 return Op; 8419 } 8420 } 8421 8422 // Legalize to a VMULL instruction. 8423 SDLoc DL(Op); 8424 SDValue Op0; 8425 SDValue Op1 = SkipExtensionForVMULL(N1, DAG); 8426 if (!isMLA) { 8427 Op0 = SkipExtensionForVMULL(N0, DAG); 8428 assert(Op0.getValueType().is64BitVector() && 8429 Op1.getValueType().is64BitVector() && 8430 "unexpected types for extended operands to VMULL"); 8431 return DAG.getNode(NewOpc, DL, VT, Op0, Op1); 8432 } 8433 8434 // Optimizing (zext A + zext B) * C, to (VMULL A, C) + (VMULL B, C) during 8435 // isel lowering to take advantage of no-stall back to back vmul + vmla. 8436 // vmull q0, d4, d6 8437 // vmlal q0, d5, d6 8438 // is faster than 8439 // vaddl q0, d4, d5 8440 // vmovl q1, d6 8441 // vmul q0, q0, q1 8442 SDValue N00 = SkipExtensionForVMULL(N0->getOperand(0).getNode(), DAG); 8443 SDValue N01 = SkipExtensionForVMULL(N0->getOperand(1).getNode(), DAG); 8444 EVT Op1VT = Op1.getValueType(); 8445 return DAG.getNode(N0->getOpcode(), DL, VT, 8446 DAG.getNode(NewOpc, DL, VT, 8447 DAG.getNode(ISD::BITCAST, DL, Op1VT, N00), Op1), 8448 DAG.getNode(NewOpc, DL, VT, 8449 DAG.getNode(ISD::BITCAST, DL, Op1VT, N01), Op1)); 8450 } 8451 8452 static SDValue LowerSDIV_v4i8(SDValue X, SDValue Y, const SDLoc &dl, 8453 SelectionDAG &DAG) { 8454 // TODO: Should this propagate fast-math-flags? 8455 8456 // Convert to float 8457 // float4 xf = vcvt_f32_s32(vmovl_s16(a.lo)); 8458 // float4 yf = vcvt_f32_s32(vmovl_s16(b.lo)); 8459 X = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, X); 8460 Y = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, Y); 8461 X = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, X); 8462 Y = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, Y); 8463 // Get reciprocal estimate. 8464 // float4 recip = vrecpeq_f32(yf); 8465 Y = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32, 8466 DAG.getConstant(Intrinsic::arm_neon_vrecpe, dl, MVT::i32), 8467 Y); 8468 // Because char has a smaller range than uchar, we can actually get away 8469 // without any newton steps. This requires that we use a weird bias 8470 // of 0xb000, however (again, this has been exhaustively tested). 8471 // float4 result = as_float4(as_int4(xf*recip) + 0xb000); 8472 X = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, X, Y); 8473 X = DAG.getNode(ISD::BITCAST, dl, MVT::v4i32, X); 8474 Y = DAG.getConstant(0xb000, dl, MVT::v4i32); 8475 X = DAG.getNode(ISD::ADD, dl, MVT::v4i32, X, Y); 8476 X = DAG.getNode(ISD::BITCAST, dl, MVT::v4f32, X); 8477 // Convert back to short. 8478 X = DAG.getNode(ISD::FP_TO_SINT, dl, MVT::v4i32, X); 8479 X = DAG.getNode(ISD::TRUNCATE, dl, MVT::v4i16, X); 8480 return X; 8481 } 8482 8483 static SDValue LowerSDIV_v4i16(SDValue N0, SDValue N1, const SDLoc &dl, 8484 SelectionDAG &DAG) { 8485 // TODO: Should this propagate fast-math-flags? 8486 8487 SDValue N2; 8488 // Convert to float. 8489 // float4 yf = vcvt_f32_s32(vmovl_s16(y)); 8490 // float4 xf = vcvt_f32_s32(vmovl_s16(x)); 8491 N0 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, N0); 8492 N1 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, N1); 8493 N0 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N0); 8494 N1 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N1); 8495 8496 // Use reciprocal estimate and one refinement step. 8497 // float4 recip = vrecpeq_f32(yf); 8498 // recip *= vrecpsq_f32(yf, recip); 8499 N2 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32, 8500 DAG.getConstant(Intrinsic::arm_neon_vrecpe, dl, MVT::i32), 8501 N1); 8502 N1 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32, 8503 DAG.getConstant(Intrinsic::arm_neon_vrecps, dl, MVT::i32), 8504 N1, N2); 8505 N2 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N1, N2); 8506 // Because short has a smaller range than ushort, we can actually get away 8507 // with only a single newton step. This requires that we use a weird bias 8508 // of 89, however (again, this has been exhaustively tested). 8509 // float4 result = as_float4(as_int4(xf*recip) + 0x89); 8510 N0 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N0, N2); 8511 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4i32, N0); 8512 N1 = DAG.getConstant(0x89, dl, MVT::v4i32); 8513 N0 = DAG.getNode(ISD::ADD, dl, MVT::v4i32, N0, N1); 8514 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4f32, N0); 8515 // Convert back to integer and return. 8516 // return vmovn_s32(vcvt_s32_f32(result)); 8517 N0 = DAG.getNode(ISD::FP_TO_SINT, dl, MVT::v4i32, N0); 8518 N0 = DAG.getNode(ISD::TRUNCATE, dl, MVT::v4i16, N0); 8519 return N0; 8520 } 8521 8522 static SDValue LowerSDIV(SDValue Op, SelectionDAG &DAG, 8523 const ARMSubtarget *ST) { 8524 EVT VT = Op.getValueType(); 8525 assert((VT == MVT::v4i16 || VT == MVT::v8i8) && 8526 "unexpected type for custom-lowering ISD::SDIV"); 8527 8528 SDLoc dl(Op); 8529 SDValue N0 = Op.getOperand(0); 8530 SDValue N1 = Op.getOperand(1); 8531 SDValue N2, N3; 8532 8533 if (VT == MVT::v8i8) { 8534 N0 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v8i16, N0); 8535 N1 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v8i16, N1); 8536 8537 N2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0, 8538 DAG.getIntPtrConstant(4, dl)); 8539 N3 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1, 8540 DAG.getIntPtrConstant(4, dl)); 8541 N0 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0, 8542 DAG.getIntPtrConstant(0, dl)); 8543 N1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1, 8544 DAG.getIntPtrConstant(0, dl)); 8545 8546 N0 = LowerSDIV_v4i8(N0, N1, dl, DAG); // v4i16 8547 N2 = LowerSDIV_v4i8(N2, N3, dl, DAG); // v4i16 8548 8549 N0 = DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v8i16, N0, N2); 8550 N0 = LowerCONCAT_VECTORS(N0, DAG, ST); 8551 8552 N0 = DAG.getNode(ISD::TRUNCATE, dl, MVT::v8i8, N0); 8553 return N0; 8554 } 8555 return LowerSDIV_v4i16(N0, N1, dl, DAG); 8556 } 8557 8558 static SDValue LowerUDIV(SDValue Op, SelectionDAG &DAG, 8559 const ARMSubtarget *ST) { 8560 // TODO: Should this propagate fast-math-flags? 8561 EVT VT = Op.getValueType(); 8562 assert((VT == MVT::v4i16 || VT == MVT::v8i8) && 8563 "unexpected type for custom-lowering ISD::UDIV"); 8564 8565 SDLoc dl(Op); 8566 SDValue N0 = Op.getOperand(0); 8567 SDValue N1 = Op.getOperand(1); 8568 SDValue N2, N3; 8569 8570 if (VT == MVT::v8i8) { 8571 N0 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v8i16, N0); 8572 N1 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v8i16, N1); 8573 8574 N2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0, 8575 DAG.getIntPtrConstant(4, dl)); 8576 N3 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1, 8577 DAG.getIntPtrConstant(4, dl)); 8578 N0 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0, 8579 DAG.getIntPtrConstant(0, dl)); 8580 N1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1, 8581 DAG.getIntPtrConstant(0, dl)); 8582 8583 N0 = LowerSDIV_v4i16(N0, N1, dl, DAG); // v4i16 8584 N2 = LowerSDIV_v4i16(N2, N3, dl, DAG); // v4i16 8585 8586 N0 = DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v8i16, N0, N2); 8587 N0 = LowerCONCAT_VECTORS(N0, DAG, ST); 8588 8589 N0 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v8i8, 8590 DAG.getConstant(Intrinsic::arm_neon_vqmovnsu, dl, 8591 MVT::i32), 8592 N0); 8593 return N0; 8594 } 8595 8596 // v4i16 sdiv ... Convert to float. 8597 // float4 yf = vcvt_f32_s32(vmovl_u16(y)); 8598 // float4 xf = vcvt_f32_s32(vmovl_u16(x)); 8599 N0 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v4i32, N0); 8600 N1 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v4i32, N1); 8601 N0 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N0); 8602 SDValue BN1 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N1); 8603 8604 // Use reciprocal estimate and two refinement steps. 8605 // float4 recip = vrecpeq_f32(yf); 8606 // recip *= vrecpsq_f32(yf, recip); 8607 // recip *= vrecpsq_f32(yf, recip); 8608 N2 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32, 8609 DAG.getConstant(Intrinsic::arm_neon_vrecpe, dl, MVT::i32), 8610 BN1); 8611 N1 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32, 8612 DAG.getConstant(Intrinsic::arm_neon_vrecps, dl, MVT::i32), 8613 BN1, N2); 8614 N2 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N1, N2); 8615 N1 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32, 8616 DAG.getConstant(Intrinsic::arm_neon_vrecps, dl, MVT::i32), 8617 BN1, N2); 8618 N2 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N1, N2); 8619 // Simply multiplying by the reciprocal estimate can leave us a few ulps 8620 // too low, so we add 2 ulps (exhaustive testing shows that this is enough, 8621 // and that it will never cause us to return an answer too large). 8622 // float4 result = as_float4(as_int4(xf*recip) + 2); 8623 N0 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N0, N2); 8624 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4i32, N0); 8625 N1 = DAG.getConstant(2, dl, MVT::v4i32); 8626 N0 = DAG.getNode(ISD::ADD, dl, MVT::v4i32, N0, N1); 8627 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4f32, N0); 8628 // Convert back to integer and return. 8629 // return vmovn_u32(vcvt_s32_f32(result)); 8630 N0 = DAG.getNode(ISD::FP_TO_SINT, dl, MVT::v4i32, N0); 8631 N0 = DAG.getNode(ISD::TRUNCATE, dl, MVT::v4i16, N0); 8632 return N0; 8633 } 8634 8635 static SDValue LowerADDSUBCARRY(SDValue Op, SelectionDAG &DAG) { 8636 SDNode *N = Op.getNode(); 8637 EVT VT = N->getValueType(0); 8638 SDVTList VTs = DAG.getVTList(VT, MVT::i32); 8639 8640 SDValue Carry = Op.getOperand(2); 8641 8642 SDLoc DL(Op); 8643 8644 SDValue Result; 8645 if (Op.getOpcode() == ISD::ADDCARRY) { 8646 // This converts the boolean value carry into the carry flag. 8647 Carry = ConvertBooleanCarryToCarryFlag(Carry, DAG); 8648 8649 // Do the addition proper using the carry flag we wanted. 8650 Result = DAG.getNode(ARMISD::ADDE, DL, VTs, Op.getOperand(0), 8651 Op.getOperand(1), Carry); 8652 8653 // Now convert the carry flag into a boolean value. 8654 Carry = ConvertCarryFlagToBooleanCarry(Result.getValue(1), VT, DAG); 8655 } else { 8656 // ARMISD::SUBE expects a carry not a borrow like ISD::SUBCARRY so we 8657 // have to invert the carry first. 8658 Carry = DAG.getNode(ISD::SUB, DL, MVT::i32, 8659 DAG.getConstant(1, DL, MVT::i32), Carry); 8660 // This converts the boolean value carry into the carry flag. 8661 Carry = ConvertBooleanCarryToCarryFlag(Carry, DAG); 8662 8663 // Do the subtraction proper using the carry flag we wanted. 8664 Result = DAG.getNode(ARMISD::SUBE, DL, VTs, Op.getOperand(0), 8665 Op.getOperand(1), Carry); 8666 8667 // Now convert the carry flag into a boolean value. 8668 Carry = ConvertCarryFlagToBooleanCarry(Result.getValue(1), VT, DAG); 8669 // But the carry returned by ARMISD::SUBE is not a borrow as expected 8670 // by ISD::SUBCARRY, so compute 1 - C. 8671 Carry = DAG.getNode(ISD::SUB, DL, MVT::i32, 8672 DAG.getConstant(1, DL, MVT::i32), Carry); 8673 } 8674 8675 // Return both values. 8676 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, Carry); 8677 } 8678 8679 SDValue ARMTargetLowering::LowerFSINCOS(SDValue Op, SelectionDAG &DAG) const { 8680 assert(Subtarget->isTargetDarwin()); 8681 8682 // For iOS, we want to call an alternative entry point: __sincos_stret, 8683 // return values are passed via sret. 8684 SDLoc dl(Op); 8685 SDValue Arg = Op.getOperand(0); 8686 EVT ArgVT = Arg.getValueType(); 8687 Type *ArgTy = ArgVT.getTypeForEVT(*DAG.getContext()); 8688 auto PtrVT = getPointerTy(DAG.getDataLayout()); 8689 8690 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); 8691 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 8692 8693 // Pair of floats / doubles used to pass the result. 8694 Type *RetTy = StructType::get(ArgTy, ArgTy); 8695 auto &DL = DAG.getDataLayout(); 8696 8697 ArgListTy Args; 8698 bool ShouldUseSRet = Subtarget->isAPCS_ABI(); 8699 SDValue SRet; 8700 if (ShouldUseSRet) { 8701 // Create stack object for sret. 8702 const uint64_t ByteSize = DL.getTypeAllocSize(RetTy); 8703 const unsigned StackAlign = DL.getPrefTypeAlignment(RetTy); 8704 int FrameIdx = MFI.CreateStackObject(ByteSize, StackAlign, false); 8705 SRet = DAG.getFrameIndex(FrameIdx, TLI.getPointerTy(DL)); 8706 8707 ArgListEntry Entry; 8708 Entry.Node = SRet; 8709 Entry.Ty = RetTy->getPointerTo(); 8710 Entry.IsSExt = false; 8711 Entry.IsZExt = false; 8712 Entry.IsSRet = true; 8713 Args.push_back(Entry); 8714 RetTy = Type::getVoidTy(*DAG.getContext()); 8715 } 8716 8717 ArgListEntry Entry; 8718 Entry.Node = Arg; 8719 Entry.Ty = ArgTy; 8720 Entry.IsSExt = false; 8721 Entry.IsZExt = false; 8722 Args.push_back(Entry); 8723 8724 RTLIB::Libcall LC = 8725 (ArgVT == MVT::f64) ? RTLIB::SINCOS_STRET_F64 : RTLIB::SINCOS_STRET_F32; 8726 const char *LibcallName = getLibcallName(LC); 8727 CallingConv::ID CC = getLibcallCallingConv(LC); 8728 SDValue Callee = DAG.getExternalSymbol(LibcallName, getPointerTy(DL)); 8729 8730 TargetLowering::CallLoweringInfo CLI(DAG); 8731 CLI.setDebugLoc(dl) 8732 .setChain(DAG.getEntryNode()) 8733 .setCallee(CC, RetTy, Callee, std::move(Args)) 8734 .setDiscardResult(ShouldUseSRet); 8735 std::pair<SDValue, SDValue> CallResult = LowerCallTo(CLI); 8736 8737 if (!ShouldUseSRet) 8738 return CallResult.first; 8739 8740 SDValue LoadSin = 8741 DAG.getLoad(ArgVT, dl, CallResult.second, SRet, MachinePointerInfo()); 8742 8743 // Address of cos field. 8744 SDValue Add = DAG.getNode(ISD::ADD, dl, PtrVT, SRet, 8745 DAG.getIntPtrConstant(ArgVT.getStoreSize(), dl)); 8746 SDValue LoadCos = 8747 DAG.getLoad(ArgVT, dl, LoadSin.getValue(1), Add, MachinePointerInfo()); 8748 8749 SDVTList Tys = DAG.getVTList(ArgVT, ArgVT); 8750 return DAG.getNode(ISD::MERGE_VALUES, dl, Tys, 8751 LoadSin.getValue(0), LoadCos.getValue(0)); 8752 } 8753 8754 SDValue ARMTargetLowering::LowerWindowsDIVLibCall(SDValue Op, SelectionDAG &DAG, 8755 bool Signed, 8756 SDValue &Chain) const { 8757 EVT VT = Op.getValueType(); 8758 assert((VT == MVT::i32 || VT == MVT::i64) && 8759 "unexpected type for custom lowering DIV"); 8760 SDLoc dl(Op); 8761 8762 const auto &DL = DAG.getDataLayout(); 8763 const auto &TLI = DAG.getTargetLoweringInfo(); 8764 8765 const char *Name = nullptr; 8766 if (Signed) 8767 Name = (VT == MVT::i32) ? "__rt_sdiv" : "__rt_sdiv64"; 8768 else 8769 Name = (VT == MVT::i32) ? "__rt_udiv" : "__rt_udiv64"; 8770 8771 SDValue ES = DAG.getExternalSymbol(Name, TLI.getPointerTy(DL)); 8772 8773 ARMTargetLowering::ArgListTy Args; 8774 8775 for (auto AI : {1, 0}) { 8776 ArgListEntry Arg; 8777 Arg.Node = Op.getOperand(AI); 8778 Arg.Ty = Arg.Node.getValueType().getTypeForEVT(*DAG.getContext()); 8779 Args.push_back(Arg); 8780 } 8781 8782 CallLoweringInfo CLI(DAG); 8783 CLI.setDebugLoc(dl) 8784 .setChain(Chain) 8785 .setCallee(CallingConv::ARM_AAPCS_VFP, VT.getTypeForEVT(*DAG.getContext()), 8786 ES, std::move(Args)); 8787 8788 return LowerCallTo(CLI).first; 8789 } 8790 8791 // This is a code size optimisation: return the original SDIV node to 8792 // DAGCombiner when we don't want to expand SDIV into a sequence of 8793 // instructions, and an empty node otherwise which will cause the 8794 // SDIV to be expanded in DAGCombine. 8795 SDValue 8796 ARMTargetLowering::BuildSDIVPow2(SDNode *N, const APInt &Divisor, 8797 SelectionDAG &DAG, 8798 SmallVectorImpl<SDNode *> &Created) const { 8799 // TODO: Support SREM 8800 if (N->getOpcode() != ISD::SDIV) 8801 return SDValue(); 8802 8803 const auto &ST = static_cast<const ARMSubtarget&>(DAG.getSubtarget()); 8804 const bool MinSize = ST.hasMinSize(); 8805 const bool HasDivide = ST.isThumb() ? ST.hasDivideInThumbMode() 8806 : ST.hasDivideInARMMode(); 8807 8808 // Don't touch vector types; rewriting this may lead to scalarizing 8809 // the int divs. 8810 if (N->getOperand(0).getValueType().isVector()) 8811 return SDValue(); 8812 8813 // Bail if MinSize is not set, and also for both ARM and Thumb mode we need 8814 // hwdiv support for this to be really profitable. 8815 if (!(MinSize && HasDivide)) 8816 return SDValue(); 8817 8818 // ARM mode is a bit simpler than Thumb: we can handle large power 8819 // of 2 immediates with 1 mov instruction; no further checks required, 8820 // just return the sdiv node. 8821 if (!ST.isThumb()) 8822 return SDValue(N, 0); 8823 8824 // In Thumb mode, immediates larger than 128 need a wide 4-byte MOV, 8825 // and thus lose the code size benefits of a MOVS that requires only 2. 8826 // TargetTransformInfo and 'getIntImmCodeSizeCost' could be helpful here, 8827 // but as it's doing exactly this, it's not worth the trouble to get TTI. 8828 if (Divisor.sgt(128)) 8829 return SDValue(); 8830 8831 return SDValue(N, 0); 8832 } 8833 8834 SDValue ARMTargetLowering::LowerDIV_Windows(SDValue Op, SelectionDAG &DAG, 8835 bool Signed) const { 8836 assert(Op.getValueType() == MVT::i32 && 8837 "unexpected type for custom lowering DIV"); 8838 SDLoc dl(Op); 8839 8840 SDValue DBZCHK = DAG.getNode(ARMISD::WIN__DBZCHK, dl, MVT::Other, 8841 DAG.getEntryNode(), Op.getOperand(1)); 8842 8843 return LowerWindowsDIVLibCall(Op, DAG, Signed, DBZCHK); 8844 } 8845 8846 static SDValue WinDBZCheckDenominator(SelectionDAG &DAG, SDNode *N, SDValue InChain) { 8847 SDLoc DL(N); 8848 SDValue Op = N->getOperand(1); 8849 if (N->getValueType(0) == MVT::i32) 8850 return DAG.getNode(ARMISD::WIN__DBZCHK, DL, MVT::Other, InChain, Op); 8851 SDValue Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i32, Op, 8852 DAG.getConstant(0, DL, MVT::i32)); 8853 SDValue Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i32, Op, 8854 DAG.getConstant(1, DL, MVT::i32)); 8855 return DAG.getNode(ARMISD::WIN__DBZCHK, DL, MVT::Other, InChain, 8856 DAG.getNode(ISD::OR, DL, MVT::i32, Lo, Hi)); 8857 } 8858 8859 void ARMTargetLowering::ExpandDIV_Windows( 8860 SDValue Op, SelectionDAG &DAG, bool Signed, 8861 SmallVectorImpl<SDValue> &Results) const { 8862 const auto &DL = DAG.getDataLayout(); 8863 const auto &TLI = DAG.getTargetLoweringInfo(); 8864 8865 assert(Op.getValueType() == MVT::i64 && 8866 "unexpected type for custom lowering DIV"); 8867 SDLoc dl(Op); 8868 8869 SDValue DBZCHK = WinDBZCheckDenominator(DAG, Op.getNode(), DAG.getEntryNode()); 8870 8871 SDValue Result = LowerWindowsDIVLibCall(Op, DAG, Signed, DBZCHK); 8872 8873 SDValue Lower = DAG.getNode(ISD::TRUNCATE, dl, MVT::i32, Result); 8874 SDValue Upper = DAG.getNode(ISD::SRL, dl, MVT::i64, Result, 8875 DAG.getConstant(32, dl, TLI.getPointerTy(DL))); 8876 Upper = DAG.getNode(ISD::TRUNCATE, dl, MVT::i32, Upper); 8877 8878 Results.push_back(Lower); 8879 Results.push_back(Upper); 8880 } 8881 8882 static SDValue LowerPredicateLoad(SDValue Op, SelectionDAG &DAG) { 8883 LoadSDNode *LD = cast<LoadSDNode>(Op.getNode()); 8884 EVT MemVT = LD->getMemoryVT(); 8885 assert((MemVT == MVT::v4i1 || MemVT == MVT::v8i1 || MemVT == MVT::v16i1) && 8886 "Expected a predicate type!"); 8887 assert(MemVT == Op.getValueType()); 8888 assert(LD->getExtensionType() == ISD::NON_EXTLOAD && 8889 "Expected a non-extending load"); 8890 assert(LD->isUnindexed() && "Expected a unindexed load"); 8891 8892 // The basic MVE VLDR on a v4i1/v8i1 actually loads the entire 16bit 8893 // predicate, with the "v4i1" bits spread out over the 16 bits loaded. We 8894 // need to make sure that 8/4 bits are actually loaded into the correct 8895 // place, which means loading the value and then shuffling the values into 8896 // the bottom bits of the predicate. 8897 // Equally, VLDR for an v16i1 will actually load 32bits (so will be incorrect 8898 // for BE). 8899 8900 SDLoc dl(Op); 8901 SDValue Load = DAG.getExtLoad( 8902 ISD::EXTLOAD, dl, MVT::i32, LD->getChain(), LD->getBasePtr(), 8903 EVT::getIntegerVT(*DAG.getContext(), MemVT.getSizeInBits()), 8904 LD->getMemOperand()); 8905 SDValue Pred = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::v16i1, Load); 8906 if (MemVT != MVT::v16i1) 8907 Pred = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MemVT, Pred, 8908 DAG.getConstant(0, dl, MVT::i32)); 8909 return DAG.getMergeValues({Pred, Load.getValue(1)}, dl); 8910 } 8911 8912 static SDValue LowerPredicateStore(SDValue Op, SelectionDAG &DAG) { 8913 StoreSDNode *ST = cast<StoreSDNode>(Op.getNode()); 8914 EVT MemVT = ST->getMemoryVT(); 8915 assert((MemVT == MVT::v4i1 || MemVT == MVT::v8i1 || MemVT == MVT::v16i1) && 8916 "Expected a predicate type!"); 8917 assert(MemVT == ST->getValue().getValueType()); 8918 assert(!ST->isTruncatingStore() && "Expected a non-extending store"); 8919 assert(ST->isUnindexed() && "Expected a unindexed store"); 8920 8921 // Only store the v4i1 or v8i1 worth of bits, via a buildvector with top bits 8922 // unset and a scalar store. 8923 SDLoc dl(Op); 8924 SDValue Build = ST->getValue(); 8925 if (MemVT != MVT::v16i1) { 8926 SmallVector<SDValue, 16> Ops; 8927 for (unsigned I = 0; I < MemVT.getVectorNumElements(); I++) 8928 Ops.push_back(DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::i32, Build, 8929 DAG.getConstant(I, dl, MVT::i32))); 8930 for (unsigned I = MemVT.getVectorNumElements(); I < 16; I++) 8931 Ops.push_back(DAG.getUNDEF(MVT::i32)); 8932 Build = DAG.getNode(ISD::BUILD_VECTOR, dl, MVT::v16i1, Ops); 8933 } 8934 SDValue GRP = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, Build); 8935 return DAG.getTruncStore( 8936 ST->getChain(), dl, GRP, ST->getBasePtr(), 8937 EVT::getIntegerVT(*DAG.getContext(), MemVT.getSizeInBits()), 8938 ST->getMemOperand()); 8939 } 8940 8941 static SDValue LowerMLOAD(SDValue Op, SelectionDAG &DAG) { 8942 MaskedLoadSDNode *N = cast<MaskedLoadSDNode>(Op.getNode()); 8943 MVT VT = Op.getSimpleValueType(); 8944 SDValue Mask = N->getMask(); 8945 SDValue PassThru = N->getPassThru(); 8946 SDLoc dl(Op); 8947 8948 auto IsZero = [](SDValue PassThru) { 8949 return (ISD::isBuildVectorAllZeros(PassThru.getNode()) || 8950 (PassThru->getOpcode() == ARMISD::VMOVIMM && 8951 isNullConstant(PassThru->getOperand(0)))); 8952 }; 8953 8954 if (IsZero(PassThru)) 8955 return Op; 8956 8957 // MVE Masked loads use zero as the passthru value. Here we convert undef to 8958 // zero too, and other values are lowered to a select. 8959 SDValue ZeroVec = DAG.getNode(ARMISD::VMOVIMM, dl, VT, 8960 DAG.getTargetConstant(0, dl, MVT::i32)); 8961 SDValue NewLoad = DAG.getMaskedLoad( 8962 VT, dl, N->getChain(), N->getBasePtr(), Mask, ZeroVec, N->getMemoryVT(), 8963 N->getMemOperand(), N->getExtensionType(), N->isExpandingLoad()); 8964 SDValue Combo = NewLoad; 8965 if (!PassThru.isUndef() && 8966 (PassThru.getOpcode() != ISD::BITCAST || 8967 !IsZero(PassThru->getOperand(0)))) 8968 Combo = DAG.getNode(ISD::VSELECT, dl, VT, Mask, NewLoad, PassThru); 8969 return DAG.getMergeValues({Combo, NewLoad.getValue(1)}, dl); 8970 } 8971 8972 static SDValue LowerAtomicLoadStore(SDValue Op, SelectionDAG &DAG) { 8973 if (isStrongerThanMonotonic(cast<AtomicSDNode>(Op)->getOrdering())) 8974 // Acquire/Release load/store is not legal for targets without a dmb or 8975 // equivalent available. 8976 return SDValue(); 8977 8978 // Monotonic load/store is legal for all targets. 8979 return Op; 8980 } 8981 8982 static void ReplaceREADCYCLECOUNTER(SDNode *N, 8983 SmallVectorImpl<SDValue> &Results, 8984 SelectionDAG &DAG, 8985 const ARMSubtarget *Subtarget) { 8986 SDLoc DL(N); 8987 // Under Power Management extensions, the cycle-count is: 8988 // mrc p15, #0, <Rt>, c9, c13, #0 8989 SDValue Ops[] = { N->getOperand(0), // Chain 8990 DAG.getTargetConstant(Intrinsic::arm_mrc, DL, MVT::i32), 8991 DAG.getTargetConstant(15, DL, MVT::i32), 8992 DAG.getTargetConstant(0, DL, MVT::i32), 8993 DAG.getTargetConstant(9, DL, MVT::i32), 8994 DAG.getTargetConstant(13, DL, MVT::i32), 8995 DAG.getTargetConstant(0, DL, MVT::i32) 8996 }; 8997 8998 SDValue Cycles32 = DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL, 8999 DAG.getVTList(MVT::i32, MVT::Other), Ops); 9000 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, Cycles32, 9001 DAG.getConstant(0, DL, MVT::i32))); 9002 Results.push_back(Cycles32.getValue(1)); 9003 } 9004 9005 static SDValue createGPRPairNode(SelectionDAG &DAG, SDValue V) { 9006 SDLoc dl(V.getNode()); 9007 SDValue VLo = DAG.getAnyExtOrTrunc(V, dl, MVT::i32); 9008 SDValue VHi = DAG.getAnyExtOrTrunc( 9009 DAG.getNode(ISD::SRL, dl, MVT::i64, V, DAG.getConstant(32, dl, MVT::i32)), 9010 dl, MVT::i32); 9011 bool isBigEndian = DAG.getDataLayout().isBigEndian(); 9012 if (isBigEndian) 9013 std::swap (VLo, VHi); 9014 SDValue RegClass = 9015 DAG.getTargetConstant(ARM::GPRPairRegClassID, dl, MVT::i32); 9016 SDValue SubReg0 = DAG.getTargetConstant(ARM::gsub_0, dl, MVT::i32); 9017 SDValue SubReg1 = DAG.getTargetConstant(ARM::gsub_1, dl, MVT::i32); 9018 const SDValue Ops[] = { RegClass, VLo, SubReg0, VHi, SubReg1 }; 9019 return SDValue( 9020 DAG.getMachineNode(TargetOpcode::REG_SEQUENCE, dl, MVT::Untyped, Ops), 0); 9021 } 9022 9023 static void ReplaceCMP_SWAP_64Results(SDNode *N, 9024 SmallVectorImpl<SDValue> & Results, 9025 SelectionDAG &DAG) { 9026 assert(N->getValueType(0) == MVT::i64 && 9027 "AtomicCmpSwap on types less than 64 should be legal"); 9028 SDValue Ops[] = {N->getOperand(1), 9029 createGPRPairNode(DAG, N->getOperand(2)), 9030 createGPRPairNode(DAG, N->getOperand(3)), 9031 N->getOperand(0)}; 9032 SDNode *CmpSwap = DAG.getMachineNode( 9033 ARM::CMP_SWAP_64, SDLoc(N), 9034 DAG.getVTList(MVT::Untyped, MVT::i32, MVT::Other), Ops); 9035 9036 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand(); 9037 DAG.setNodeMemRefs(cast<MachineSDNode>(CmpSwap), {MemOp}); 9038 9039 bool isBigEndian = DAG.getDataLayout().isBigEndian(); 9040 9041 Results.push_back( 9042 DAG.getTargetExtractSubreg(isBigEndian ? ARM::gsub_1 : ARM::gsub_0, 9043 SDLoc(N), MVT::i32, SDValue(CmpSwap, 0))); 9044 Results.push_back( 9045 DAG.getTargetExtractSubreg(isBigEndian ? ARM::gsub_0 : ARM::gsub_1, 9046 SDLoc(N), MVT::i32, SDValue(CmpSwap, 0))); 9047 Results.push_back(SDValue(CmpSwap, 2)); 9048 } 9049 9050 static SDValue LowerFPOWI(SDValue Op, const ARMSubtarget &Subtarget, 9051 SelectionDAG &DAG) { 9052 const auto &TLI = DAG.getTargetLoweringInfo(); 9053 9054 assert(Subtarget.getTargetTriple().isOSMSVCRT() && 9055 "Custom lowering is MSVCRT specific!"); 9056 9057 SDLoc dl(Op); 9058 SDValue Val = Op.getOperand(0); 9059 MVT Ty = Val->getSimpleValueType(0); 9060 SDValue Exponent = DAG.getNode(ISD::SINT_TO_FP, dl, Ty, Op.getOperand(1)); 9061 SDValue Callee = DAG.getExternalSymbol(Ty == MVT::f32 ? "powf" : "pow", 9062 TLI.getPointerTy(DAG.getDataLayout())); 9063 9064 TargetLowering::ArgListTy Args; 9065 TargetLowering::ArgListEntry Entry; 9066 9067 Entry.Node = Val; 9068 Entry.Ty = Val.getValueType().getTypeForEVT(*DAG.getContext()); 9069 Entry.IsZExt = true; 9070 Args.push_back(Entry); 9071 9072 Entry.Node = Exponent; 9073 Entry.Ty = Exponent.getValueType().getTypeForEVT(*DAG.getContext()); 9074 Entry.IsZExt = true; 9075 Args.push_back(Entry); 9076 9077 Type *LCRTy = Val.getValueType().getTypeForEVT(*DAG.getContext()); 9078 9079 // In the in-chain to the call is the entry node If we are emitting a 9080 // tailcall, the chain will be mutated if the node has a non-entry input 9081 // chain. 9082 SDValue InChain = DAG.getEntryNode(); 9083 SDValue TCChain = InChain; 9084 9085 const Function &F = DAG.getMachineFunction().getFunction(); 9086 bool IsTC = TLI.isInTailCallPosition(DAG, Op.getNode(), TCChain) && 9087 F.getReturnType() == LCRTy; 9088 if (IsTC) 9089 InChain = TCChain; 9090 9091 TargetLowering::CallLoweringInfo CLI(DAG); 9092 CLI.setDebugLoc(dl) 9093 .setChain(InChain) 9094 .setCallee(CallingConv::ARM_AAPCS_VFP, LCRTy, Callee, std::move(Args)) 9095 .setTailCall(IsTC); 9096 std::pair<SDValue, SDValue> CI = TLI.LowerCallTo(CLI); 9097 9098 // Return the chain (the DAG root) if it is a tail call 9099 return !CI.second.getNode() ? DAG.getRoot() : CI.first; 9100 } 9101 9102 SDValue ARMTargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const { 9103 LLVM_DEBUG(dbgs() << "Lowering node: "; Op.dump()); 9104 switch (Op.getOpcode()) { 9105 default: llvm_unreachable("Don't know how to custom lower this!"); 9106 case ISD::WRITE_REGISTER: return LowerWRITE_REGISTER(Op, DAG); 9107 case ISD::ConstantPool: return LowerConstantPool(Op, DAG); 9108 case ISD::BlockAddress: return LowerBlockAddress(Op, DAG); 9109 case ISD::GlobalAddress: return LowerGlobalAddress(Op, DAG); 9110 case ISD::GlobalTLSAddress: return LowerGlobalTLSAddress(Op, DAG); 9111 case ISD::SELECT: return LowerSELECT(Op, DAG); 9112 case ISD::SELECT_CC: return LowerSELECT_CC(Op, DAG); 9113 case ISD::BRCOND: return LowerBRCOND(Op, DAG); 9114 case ISD::BR_CC: return LowerBR_CC(Op, DAG); 9115 case ISD::BR_JT: return LowerBR_JT(Op, DAG); 9116 case ISD::VASTART: return LowerVASTART(Op, DAG); 9117 case ISD::ATOMIC_FENCE: return LowerATOMIC_FENCE(Op, DAG, Subtarget); 9118 case ISD::PREFETCH: return LowerPREFETCH(Op, DAG, Subtarget); 9119 case ISD::SINT_TO_FP: 9120 case ISD::UINT_TO_FP: return LowerINT_TO_FP(Op, DAG); 9121 case ISD::FP_TO_SINT: 9122 case ISD::FP_TO_UINT: return LowerFP_TO_INT(Op, DAG); 9123 case ISD::FCOPYSIGN: return LowerFCOPYSIGN(Op, DAG); 9124 case ISD::RETURNADDR: return LowerRETURNADDR(Op, DAG); 9125 case ISD::FRAMEADDR: return LowerFRAMEADDR(Op, DAG); 9126 case ISD::EH_SJLJ_SETJMP: return LowerEH_SJLJ_SETJMP(Op, DAG); 9127 case ISD::EH_SJLJ_LONGJMP: return LowerEH_SJLJ_LONGJMP(Op, DAG); 9128 case ISD::EH_SJLJ_SETUP_DISPATCH: return LowerEH_SJLJ_SETUP_DISPATCH(Op, DAG); 9129 case ISD::INTRINSIC_VOID: return LowerINTRINSIC_VOID(Op, DAG, Subtarget); 9130 case ISD::INTRINSIC_WO_CHAIN: return LowerINTRINSIC_WO_CHAIN(Op, DAG, 9131 Subtarget); 9132 case ISD::BITCAST: return ExpandBITCAST(Op.getNode(), DAG, Subtarget); 9133 case ISD::SHL: 9134 case ISD::SRL: 9135 case ISD::SRA: return LowerShift(Op.getNode(), DAG, Subtarget); 9136 case ISD::SREM: return LowerREM(Op.getNode(), DAG); 9137 case ISD::UREM: return LowerREM(Op.getNode(), DAG); 9138 case ISD::SHL_PARTS: return LowerShiftLeftParts(Op, DAG); 9139 case ISD::SRL_PARTS: 9140 case ISD::SRA_PARTS: return LowerShiftRightParts(Op, DAG); 9141 case ISD::CTTZ: 9142 case ISD::CTTZ_ZERO_UNDEF: return LowerCTTZ(Op.getNode(), DAG, Subtarget); 9143 case ISD::CTPOP: return LowerCTPOP(Op.getNode(), DAG, Subtarget); 9144 case ISD::SETCC: return LowerVSETCC(Op, DAG, Subtarget); 9145 case ISD::SETCCCARRY: return LowerSETCCCARRY(Op, DAG); 9146 case ISD::ConstantFP: return LowerConstantFP(Op, DAG, Subtarget); 9147 case ISD::BUILD_VECTOR: return LowerBUILD_VECTOR(Op, DAG, Subtarget); 9148 case ISD::VECTOR_SHUFFLE: return LowerVECTOR_SHUFFLE(Op, DAG, Subtarget); 9149 case ISD::EXTRACT_SUBVECTOR: return LowerEXTRACT_SUBVECTOR(Op, DAG, Subtarget); 9150 case ISD::INSERT_VECTOR_ELT: return LowerINSERT_VECTOR_ELT(Op, DAG); 9151 case ISD::EXTRACT_VECTOR_ELT: return LowerEXTRACT_VECTOR_ELT(Op, DAG, Subtarget); 9152 case ISD::CONCAT_VECTORS: return LowerCONCAT_VECTORS(Op, DAG, Subtarget); 9153 case ISD::FLT_ROUNDS_: return LowerFLT_ROUNDS_(Op, DAG); 9154 case ISD::MUL: return LowerMUL(Op, DAG); 9155 case ISD::SDIV: 9156 if (Subtarget->isTargetWindows() && !Op.getValueType().isVector()) 9157 return LowerDIV_Windows(Op, DAG, /* Signed */ true); 9158 return LowerSDIV(Op, DAG, Subtarget); 9159 case ISD::UDIV: 9160 if (Subtarget->isTargetWindows() && !Op.getValueType().isVector()) 9161 return LowerDIV_Windows(Op, DAG, /* Signed */ false); 9162 return LowerUDIV(Op, DAG, Subtarget); 9163 case ISD::ADDCARRY: 9164 case ISD::SUBCARRY: return LowerADDSUBCARRY(Op, DAG); 9165 case ISD::SADDO: 9166 case ISD::SSUBO: 9167 return LowerSignedALUO(Op, DAG); 9168 case ISD::UADDO: 9169 case ISD::USUBO: 9170 return LowerUnsignedALUO(Op, DAG); 9171 case ISD::SADDSAT: 9172 case ISD::SSUBSAT: 9173 return LowerSADDSUBSAT(Op, DAG, Subtarget); 9174 case ISD::LOAD: 9175 return LowerPredicateLoad(Op, DAG); 9176 case ISD::STORE: 9177 return LowerPredicateStore(Op, DAG); 9178 case ISD::MLOAD: 9179 return LowerMLOAD(Op, DAG); 9180 case ISD::ATOMIC_LOAD: 9181 case ISD::ATOMIC_STORE: return LowerAtomicLoadStore(Op, DAG); 9182 case ISD::FSINCOS: return LowerFSINCOS(Op, DAG); 9183 case ISD::SDIVREM: 9184 case ISD::UDIVREM: return LowerDivRem(Op, DAG); 9185 case ISD::DYNAMIC_STACKALLOC: 9186 if (Subtarget->isTargetWindows()) 9187 return LowerDYNAMIC_STACKALLOC(Op, DAG); 9188 llvm_unreachable("Don't know how to custom lower this!"); 9189 case ISD::FP_ROUND: return LowerFP_ROUND(Op, DAG); 9190 case ISD::FP_EXTEND: return LowerFP_EXTEND(Op, DAG); 9191 case ISD::FPOWI: return LowerFPOWI(Op, *Subtarget, DAG); 9192 case ARMISD::WIN__DBZCHK: return SDValue(); 9193 } 9194 } 9195 9196 static void ReplaceLongIntrinsic(SDNode *N, SmallVectorImpl<SDValue> &Results, 9197 SelectionDAG &DAG) { 9198 unsigned IntNo = cast<ConstantSDNode>(N->getOperand(0))->getZExtValue(); 9199 unsigned Opc = 0; 9200 if (IntNo == Intrinsic::arm_smlald) 9201 Opc = ARMISD::SMLALD; 9202 else if (IntNo == Intrinsic::arm_smlaldx) 9203 Opc = ARMISD::SMLALDX; 9204 else if (IntNo == Intrinsic::arm_smlsld) 9205 Opc = ARMISD::SMLSLD; 9206 else if (IntNo == Intrinsic::arm_smlsldx) 9207 Opc = ARMISD::SMLSLDX; 9208 else 9209 return; 9210 9211 SDLoc dl(N); 9212 SDValue Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, 9213 N->getOperand(3), 9214 DAG.getConstant(0, dl, MVT::i32)); 9215 SDValue Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, MVT::i32, 9216 N->getOperand(3), 9217 DAG.getConstant(1, dl, MVT::i32)); 9218 9219 SDValue LongMul = DAG.getNode(Opc, dl, 9220 DAG.getVTList(MVT::i32, MVT::i32), 9221 N->getOperand(1), N->getOperand(2), 9222 Lo, Hi); 9223 Results.push_back(LongMul.getValue(0)); 9224 Results.push_back(LongMul.getValue(1)); 9225 } 9226 9227 /// ReplaceNodeResults - Replace the results of node with an illegal result 9228 /// type with new values built out of custom code. 9229 void ARMTargetLowering::ReplaceNodeResults(SDNode *N, 9230 SmallVectorImpl<SDValue> &Results, 9231 SelectionDAG &DAG) const { 9232 SDValue Res; 9233 switch (N->getOpcode()) { 9234 default: 9235 llvm_unreachable("Don't know how to custom expand this!"); 9236 case ISD::READ_REGISTER: 9237 ExpandREAD_REGISTER(N, Results, DAG); 9238 break; 9239 case ISD::BITCAST: 9240 Res = ExpandBITCAST(N, DAG, Subtarget); 9241 break; 9242 case ISD::SRL: 9243 case ISD::SRA: 9244 case ISD::SHL: 9245 Res = Expand64BitShift(N, DAG, Subtarget); 9246 break; 9247 case ISD::SREM: 9248 case ISD::UREM: 9249 Res = LowerREM(N, DAG); 9250 break; 9251 case ISD::SDIVREM: 9252 case ISD::UDIVREM: 9253 Res = LowerDivRem(SDValue(N, 0), DAG); 9254 assert(Res.getNumOperands() == 2 && "DivRem needs two values"); 9255 Results.push_back(Res.getValue(0)); 9256 Results.push_back(Res.getValue(1)); 9257 return; 9258 case ISD::SADDSAT: 9259 case ISD::SSUBSAT: 9260 Res = LowerSADDSUBSAT(SDValue(N, 0), DAG, Subtarget); 9261 break; 9262 case ISD::READCYCLECOUNTER: 9263 ReplaceREADCYCLECOUNTER(N, Results, DAG, Subtarget); 9264 return; 9265 case ISD::UDIV: 9266 case ISD::SDIV: 9267 assert(Subtarget->isTargetWindows() && "can only expand DIV on Windows"); 9268 return ExpandDIV_Windows(SDValue(N, 0), DAG, N->getOpcode() == ISD::SDIV, 9269 Results); 9270 case ISD::ATOMIC_CMP_SWAP: 9271 ReplaceCMP_SWAP_64Results(N, Results, DAG); 9272 return; 9273 case ISD::INTRINSIC_WO_CHAIN: 9274 return ReplaceLongIntrinsic(N, Results, DAG); 9275 case ISD::ABS: 9276 lowerABS(N, Results, DAG); 9277 return ; 9278 9279 } 9280 if (Res.getNode()) 9281 Results.push_back(Res); 9282 } 9283 9284 //===----------------------------------------------------------------------===// 9285 // ARM Scheduler Hooks 9286 //===----------------------------------------------------------------------===// 9287 9288 /// SetupEntryBlockForSjLj - Insert code into the entry block that creates and 9289 /// registers the function context. 9290 void ARMTargetLowering::SetupEntryBlockForSjLj(MachineInstr &MI, 9291 MachineBasicBlock *MBB, 9292 MachineBasicBlock *DispatchBB, 9293 int FI) const { 9294 assert(!Subtarget->isROPI() && !Subtarget->isRWPI() && 9295 "ROPI/RWPI not currently supported with SjLj"); 9296 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 9297 DebugLoc dl = MI.getDebugLoc(); 9298 MachineFunction *MF = MBB->getParent(); 9299 MachineRegisterInfo *MRI = &MF->getRegInfo(); 9300 MachineConstantPool *MCP = MF->getConstantPool(); 9301 ARMFunctionInfo *AFI = MF->getInfo<ARMFunctionInfo>(); 9302 const Function &F = MF->getFunction(); 9303 9304 bool isThumb = Subtarget->isThumb(); 9305 bool isThumb2 = Subtarget->isThumb2(); 9306 9307 unsigned PCLabelId = AFI->createPICLabelUId(); 9308 unsigned PCAdj = (isThumb || isThumb2) ? 4 : 8; 9309 ARMConstantPoolValue *CPV = 9310 ARMConstantPoolMBB::Create(F.getContext(), DispatchBB, PCLabelId, PCAdj); 9311 unsigned CPI = MCP->getConstantPoolIndex(CPV, 4); 9312 9313 const TargetRegisterClass *TRC = isThumb ? &ARM::tGPRRegClass 9314 : &ARM::GPRRegClass; 9315 9316 // Grab constant pool and fixed stack memory operands. 9317 MachineMemOperand *CPMMO = 9318 MF->getMachineMemOperand(MachinePointerInfo::getConstantPool(*MF), 9319 MachineMemOperand::MOLoad, 4, 4); 9320 9321 MachineMemOperand *FIMMOSt = 9322 MF->getMachineMemOperand(MachinePointerInfo::getFixedStack(*MF, FI), 9323 MachineMemOperand::MOStore, 4, 4); 9324 9325 // Load the address of the dispatch MBB into the jump buffer. 9326 if (isThumb2) { 9327 // Incoming value: jbuf 9328 // ldr.n r5, LCPI1_1 9329 // orr r5, r5, #1 9330 // add r5, pc 9331 // str r5, [$jbuf, #+4] ; &jbuf[1] 9332 Register NewVReg1 = MRI->createVirtualRegister(TRC); 9333 BuildMI(*MBB, MI, dl, TII->get(ARM::t2LDRpci), NewVReg1) 9334 .addConstantPoolIndex(CPI) 9335 .addMemOperand(CPMMO) 9336 .add(predOps(ARMCC::AL)); 9337 // Set the low bit because of thumb mode. 9338 Register NewVReg2 = MRI->createVirtualRegister(TRC); 9339 BuildMI(*MBB, MI, dl, TII->get(ARM::t2ORRri), NewVReg2) 9340 .addReg(NewVReg1, RegState::Kill) 9341 .addImm(0x01) 9342 .add(predOps(ARMCC::AL)) 9343 .add(condCodeOp()); 9344 Register NewVReg3 = MRI->createVirtualRegister(TRC); 9345 BuildMI(*MBB, MI, dl, TII->get(ARM::tPICADD), NewVReg3) 9346 .addReg(NewVReg2, RegState::Kill) 9347 .addImm(PCLabelId); 9348 BuildMI(*MBB, MI, dl, TII->get(ARM::t2STRi12)) 9349 .addReg(NewVReg3, RegState::Kill) 9350 .addFrameIndex(FI) 9351 .addImm(36) // &jbuf[1] :: pc 9352 .addMemOperand(FIMMOSt) 9353 .add(predOps(ARMCC::AL)); 9354 } else if (isThumb) { 9355 // Incoming value: jbuf 9356 // ldr.n r1, LCPI1_4 9357 // add r1, pc 9358 // mov r2, #1 9359 // orrs r1, r2 9360 // add r2, $jbuf, #+4 ; &jbuf[1] 9361 // str r1, [r2] 9362 Register NewVReg1 = MRI->createVirtualRegister(TRC); 9363 BuildMI(*MBB, MI, dl, TII->get(ARM::tLDRpci), NewVReg1) 9364 .addConstantPoolIndex(CPI) 9365 .addMemOperand(CPMMO) 9366 .add(predOps(ARMCC::AL)); 9367 Register NewVReg2 = MRI->createVirtualRegister(TRC); 9368 BuildMI(*MBB, MI, dl, TII->get(ARM::tPICADD), NewVReg2) 9369 .addReg(NewVReg1, RegState::Kill) 9370 .addImm(PCLabelId); 9371 // Set the low bit because of thumb mode. 9372 Register NewVReg3 = MRI->createVirtualRegister(TRC); 9373 BuildMI(*MBB, MI, dl, TII->get(ARM::tMOVi8), NewVReg3) 9374 .addReg(ARM::CPSR, RegState::Define) 9375 .addImm(1) 9376 .add(predOps(ARMCC::AL)); 9377 Register NewVReg4 = MRI->createVirtualRegister(TRC); 9378 BuildMI(*MBB, MI, dl, TII->get(ARM::tORR), NewVReg4) 9379 .addReg(ARM::CPSR, RegState::Define) 9380 .addReg(NewVReg2, RegState::Kill) 9381 .addReg(NewVReg3, RegState::Kill) 9382 .add(predOps(ARMCC::AL)); 9383 Register NewVReg5 = MRI->createVirtualRegister(TRC); 9384 BuildMI(*MBB, MI, dl, TII->get(ARM::tADDframe), NewVReg5) 9385 .addFrameIndex(FI) 9386 .addImm(36); // &jbuf[1] :: pc 9387 BuildMI(*MBB, MI, dl, TII->get(ARM::tSTRi)) 9388 .addReg(NewVReg4, RegState::Kill) 9389 .addReg(NewVReg5, RegState::Kill) 9390 .addImm(0) 9391 .addMemOperand(FIMMOSt) 9392 .add(predOps(ARMCC::AL)); 9393 } else { 9394 // Incoming value: jbuf 9395 // ldr r1, LCPI1_1 9396 // add r1, pc, r1 9397 // str r1, [$jbuf, #+4] ; &jbuf[1] 9398 Register NewVReg1 = MRI->createVirtualRegister(TRC); 9399 BuildMI(*MBB, MI, dl, TII->get(ARM::LDRi12), NewVReg1) 9400 .addConstantPoolIndex(CPI) 9401 .addImm(0) 9402 .addMemOperand(CPMMO) 9403 .add(predOps(ARMCC::AL)); 9404 Register NewVReg2 = MRI->createVirtualRegister(TRC); 9405 BuildMI(*MBB, MI, dl, TII->get(ARM::PICADD), NewVReg2) 9406 .addReg(NewVReg1, RegState::Kill) 9407 .addImm(PCLabelId) 9408 .add(predOps(ARMCC::AL)); 9409 BuildMI(*MBB, MI, dl, TII->get(ARM::STRi12)) 9410 .addReg(NewVReg2, RegState::Kill) 9411 .addFrameIndex(FI) 9412 .addImm(36) // &jbuf[1] :: pc 9413 .addMemOperand(FIMMOSt) 9414 .add(predOps(ARMCC::AL)); 9415 } 9416 } 9417 9418 void ARMTargetLowering::EmitSjLjDispatchBlock(MachineInstr &MI, 9419 MachineBasicBlock *MBB) const { 9420 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 9421 DebugLoc dl = MI.getDebugLoc(); 9422 MachineFunction *MF = MBB->getParent(); 9423 MachineRegisterInfo *MRI = &MF->getRegInfo(); 9424 MachineFrameInfo &MFI = MF->getFrameInfo(); 9425 int FI = MFI.getFunctionContextIndex(); 9426 9427 const TargetRegisterClass *TRC = Subtarget->isThumb() ? &ARM::tGPRRegClass 9428 : &ARM::GPRnopcRegClass; 9429 9430 // Get a mapping of the call site numbers to all of the landing pads they're 9431 // associated with. 9432 DenseMap<unsigned, SmallVector<MachineBasicBlock*, 2>> CallSiteNumToLPad; 9433 unsigned MaxCSNum = 0; 9434 for (MachineFunction::iterator BB = MF->begin(), E = MF->end(); BB != E; 9435 ++BB) { 9436 if (!BB->isEHPad()) continue; 9437 9438 // FIXME: We should assert that the EH_LABEL is the first MI in the landing 9439 // pad. 9440 for (MachineBasicBlock::iterator 9441 II = BB->begin(), IE = BB->end(); II != IE; ++II) { 9442 if (!II->isEHLabel()) continue; 9443 9444 MCSymbol *Sym = II->getOperand(0).getMCSymbol(); 9445 if (!MF->hasCallSiteLandingPad(Sym)) continue; 9446 9447 SmallVectorImpl<unsigned> &CallSiteIdxs = MF->getCallSiteLandingPad(Sym); 9448 for (SmallVectorImpl<unsigned>::iterator 9449 CSI = CallSiteIdxs.begin(), CSE = CallSiteIdxs.end(); 9450 CSI != CSE; ++CSI) { 9451 CallSiteNumToLPad[*CSI].push_back(&*BB); 9452 MaxCSNum = std::max(MaxCSNum, *CSI); 9453 } 9454 break; 9455 } 9456 } 9457 9458 // Get an ordered list of the machine basic blocks for the jump table. 9459 std::vector<MachineBasicBlock*> LPadList; 9460 SmallPtrSet<MachineBasicBlock*, 32> InvokeBBs; 9461 LPadList.reserve(CallSiteNumToLPad.size()); 9462 for (unsigned I = 1; I <= MaxCSNum; ++I) { 9463 SmallVectorImpl<MachineBasicBlock*> &MBBList = CallSiteNumToLPad[I]; 9464 for (SmallVectorImpl<MachineBasicBlock*>::iterator 9465 II = MBBList.begin(), IE = MBBList.end(); II != IE; ++II) { 9466 LPadList.push_back(*II); 9467 InvokeBBs.insert((*II)->pred_begin(), (*II)->pred_end()); 9468 } 9469 } 9470 9471 assert(!LPadList.empty() && 9472 "No landing pad destinations for the dispatch jump table!"); 9473 9474 // Create the jump table and associated information. 9475 MachineJumpTableInfo *JTI = 9476 MF->getOrCreateJumpTableInfo(MachineJumpTableInfo::EK_Inline); 9477 unsigned MJTI = JTI->createJumpTableIndex(LPadList); 9478 9479 // Create the MBBs for the dispatch code. 9480 9481 // Shove the dispatch's address into the return slot in the function context. 9482 MachineBasicBlock *DispatchBB = MF->CreateMachineBasicBlock(); 9483 DispatchBB->setIsEHPad(); 9484 9485 MachineBasicBlock *TrapBB = MF->CreateMachineBasicBlock(); 9486 unsigned trap_opcode; 9487 if (Subtarget->isThumb()) 9488 trap_opcode = ARM::tTRAP; 9489 else 9490 trap_opcode = Subtarget->useNaClTrap() ? ARM::TRAPNaCl : ARM::TRAP; 9491 9492 BuildMI(TrapBB, dl, TII->get(trap_opcode)); 9493 DispatchBB->addSuccessor(TrapBB); 9494 9495 MachineBasicBlock *DispContBB = MF->CreateMachineBasicBlock(); 9496 DispatchBB->addSuccessor(DispContBB); 9497 9498 // Insert and MBBs. 9499 MF->insert(MF->end(), DispatchBB); 9500 MF->insert(MF->end(), DispContBB); 9501 MF->insert(MF->end(), TrapBB); 9502 9503 // Insert code into the entry block that creates and registers the function 9504 // context. 9505 SetupEntryBlockForSjLj(MI, MBB, DispatchBB, FI); 9506 9507 MachineMemOperand *FIMMOLd = MF->getMachineMemOperand( 9508 MachinePointerInfo::getFixedStack(*MF, FI), 9509 MachineMemOperand::MOLoad | MachineMemOperand::MOVolatile, 4, 4); 9510 9511 MachineInstrBuilder MIB; 9512 MIB = BuildMI(DispatchBB, dl, TII->get(ARM::Int_eh_sjlj_dispatchsetup)); 9513 9514 const ARMBaseInstrInfo *AII = static_cast<const ARMBaseInstrInfo*>(TII); 9515 const ARMBaseRegisterInfo &RI = AII->getRegisterInfo(); 9516 9517 // Add a register mask with no preserved registers. This results in all 9518 // registers being marked as clobbered. This can't work if the dispatch block 9519 // is in a Thumb1 function and is linked with ARM code which uses the FP 9520 // registers, as there is no way to preserve the FP registers in Thumb1 mode. 9521 MIB.addRegMask(RI.getSjLjDispatchPreservedMask(*MF)); 9522 9523 bool IsPositionIndependent = isPositionIndependent(); 9524 unsigned NumLPads = LPadList.size(); 9525 if (Subtarget->isThumb2()) { 9526 Register NewVReg1 = MRI->createVirtualRegister(TRC); 9527 BuildMI(DispatchBB, dl, TII->get(ARM::t2LDRi12), NewVReg1) 9528 .addFrameIndex(FI) 9529 .addImm(4) 9530 .addMemOperand(FIMMOLd) 9531 .add(predOps(ARMCC::AL)); 9532 9533 if (NumLPads < 256) { 9534 BuildMI(DispatchBB, dl, TII->get(ARM::t2CMPri)) 9535 .addReg(NewVReg1) 9536 .addImm(LPadList.size()) 9537 .add(predOps(ARMCC::AL)); 9538 } else { 9539 Register VReg1 = MRI->createVirtualRegister(TRC); 9540 BuildMI(DispatchBB, dl, TII->get(ARM::t2MOVi16), VReg1) 9541 .addImm(NumLPads & 0xFFFF) 9542 .add(predOps(ARMCC::AL)); 9543 9544 unsigned VReg2 = VReg1; 9545 if ((NumLPads & 0xFFFF0000) != 0) { 9546 VReg2 = MRI->createVirtualRegister(TRC); 9547 BuildMI(DispatchBB, dl, TII->get(ARM::t2MOVTi16), VReg2) 9548 .addReg(VReg1) 9549 .addImm(NumLPads >> 16) 9550 .add(predOps(ARMCC::AL)); 9551 } 9552 9553 BuildMI(DispatchBB, dl, TII->get(ARM::t2CMPrr)) 9554 .addReg(NewVReg1) 9555 .addReg(VReg2) 9556 .add(predOps(ARMCC::AL)); 9557 } 9558 9559 BuildMI(DispatchBB, dl, TII->get(ARM::t2Bcc)) 9560 .addMBB(TrapBB) 9561 .addImm(ARMCC::HI) 9562 .addReg(ARM::CPSR); 9563 9564 Register NewVReg3 = MRI->createVirtualRegister(TRC); 9565 BuildMI(DispContBB, dl, TII->get(ARM::t2LEApcrelJT), NewVReg3) 9566 .addJumpTableIndex(MJTI) 9567 .add(predOps(ARMCC::AL)); 9568 9569 Register NewVReg4 = MRI->createVirtualRegister(TRC); 9570 BuildMI(DispContBB, dl, TII->get(ARM::t2ADDrs), NewVReg4) 9571 .addReg(NewVReg3, RegState::Kill) 9572 .addReg(NewVReg1) 9573 .addImm(ARM_AM::getSORegOpc(ARM_AM::lsl, 2)) 9574 .add(predOps(ARMCC::AL)) 9575 .add(condCodeOp()); 9576 9577 BuildMI(DispContBB, dl, TII->get(ARM::t2BR_JT)) 9578 .addReg(NewVReg4, RegState::Kill) 9579 .addReg(NewVReg1) 9580 .addJumpTableIndex(MJTI); 9581 } else if (Subtarget->isThumb()) { 9582 Register NewVReg1 = MRI->createVirtualRegister(TRC); 9583 BuildMI(DispatchBB, dl, TII->get(ARM::tLDRspi), NewVReg1) 9584 .addFrameIndex(FI) 9585 .addImm(1) 9586 .addMemOperand(FIMMOLd) 9587 .add(predOps(ARMCC::AL)); 9588 9589 if (NumLPads < 256) { 9590 BuildMI(DispatchBB, dl, TII->get(ARM::tCMPi8)) 9591 .addReg(NewVReg1) 9592 .addImm(NumLPads) 9593 .add(predOps(ARMCC::AL)); 9594 } else { 9595 MachineConstantPool *ConstantPool = MF->getConstantPool(); 9596 Type *Int32Ty = Type::getInt32Ty(MF->getFunction().getContext()); 9597 const Constant *C = ConstantInt::get(Int32Ty, NumLPads); 9598 9599 // MachineConstantPool wants an explicit alignment. 9600 unsigned Align = MF->getDataLayout().getPrefTypeAlignment(Int32Ty); 9601 if (Align == 0) 9602 Align = MF->getDataLayout().getTypeAllocSize(C->getType()); 9603 unsigned Idx = ConstantPool->getConstantPoolIndex(C, Align); 9604 9605 Register VReg1 = MRI->createVirtualRegister(TRC); 9606 BuildMI(DispatchBB, dl, TII->get(ARM::tLDRpci)) 9607 .addReg(VReg1, RegState::Define) 9608 .addConstantPoolIndex(Idx) 9609 .add(predOps(ARMCC::AL)); 9610 BuildMI(DispatchBB, dl, TII->get(ARM::tCMPr)) 9611 .addReg(NewVReg1) 9612 .addReg(VReg1) 9613 .add(predOps(ARMCC::AL)); 9614 } 9615 9616 BuildMI(DispatchBB, dl, TII->get(ARM::tBcc)) 9617 .addMBB(TrapBB) 9618 .addImm(ARMCC::HI) 9619 .addReg(ARM::CPSR); 9620 9621 Register NewVReg2 = MRI->createVirtualRegister(TRC); 9622 BuildMI(DispContBB, dl, TII->get(ARM::tLSLri), NewVReg2) 9623 .addReg(ARM::CPSR, RegState::Define) 9624 .addReg(NewVReg1) 9625 .addImm(2) 9626 .add(predOps(ARMCC::AL)); 9627 9628 Register NewVReg3 = MRI->createVirtualRegister(TRC); 9629 BuildMI(DispContBB, dl, TII->get(ARM::tLEApcrelJT), NewVReg3) 9630 .addJumpTableIndex(MJTI) 9631 .add(predOps(ARMCC::AL)); 9632 9633 Register NewVReg4 = MRI->createVirtualRegister(TRC); 9634 BuildMI(DispContBB, dl, TII->get(ARM::tADDrr), NewVReg4) 9635 .addReg(ARM::CPSR, RegState::Define) 9636 .addReg(NewVReg2, RegState::Kill) 9637 .addReg(NewVReg3) 9638 .add(predOps(ARMCC::AL)); 9639 9640 MachineMemOperand *JTMMOLd = MF->getMachineMemOperand( 9641 MachinePointerInfo::getJumpTable(*MF), MachineMemOperand::MOLoad, 4, 4); 9642 9643 Register NewVReg5 = MRI->createVirtualRegister(TRC); 9644 BuildMI(DispContBB, dl, TII->get(ARM::tLDRi), NewVReg5) 9645 .addReg(NewVReg4, RegState::Kill) 9646 .addImm(0) 9647 .addMemOperand(JTMMOLd) 9648 .add(predOps(ARMCC::AL)); 9649 9650 unsigned NewVReg6 = NewVReg5; 9651 if (IsPositionIndependent) { 9652 NewVReg6 = MRI->createVirtualRegister(TRC); 9653 BuildMI(DispContBB, dl, TII->get(ARM::tADDrr), NewVReg6) 9654 .addReg(ARM::CPSR, RegState::Define) 9655 .addReg(NewVReg5, RegState::Kill) 9656 .addReg(NewVReg3) 9657 .add(predOps(ARMCC::AL)); 9658 } 9659 9660 BuildMI(DispContBB, dl, TII->get(ARM::tBR_JTr)) 9661 .addReg(NewVReg6, RegState::Kill) 9662 .addJumpTableIndex(MJTI); 9663 } else { 9664 Register NewVReg1 = MRI->createVirtualRegister(TRC); 9665 BuildMI(DispatchBB, dl, TII->get(ARM::LDRi12), NewVReg1) 9666 .addFrameIndex(FI) 9667 .addImm(4) 9668 .addMemOperand(FIMMOLd) 9669 .add(predOps(ARMCC::AL)); 9670 9671 if (NumLPads < 256) { 9672 BuildMI(DispatchBB, dl, TII->get(ARM::CMPri)) 9673 .addReg(NewVReg1) 9674 .addImm(NumLPads) 9675 .add(predOps(ARMCC::AL)); 9676 } else if (Subtarget->hasV6T2Ops() && isUInt<16>(NumLPads)) { 9677 Register VReg1 = MRI->createVirtualRegister(TRC); 9678 BuildMI(DispatchBB, dl, TII->get(ARM::MOVi16), VReg1) 9679 .addImm(NumLPads & 0xFFFF) 9680 .add(predOps(ARMCC::AL)); 9681 9682 unsigned VReg2 = VReg1; 9683 if ((NumLPads & 0xFFFF0000) != 0) { 9684 VReg2 = MRI->createVirtualRegister(TRC); 9685 BuildMI(DispatchBB, dl, TII->get(ARM::MOVTi16), VReg2) 9686 .addReg(VReg1) 9687 .addImm(NumLPads >> 16) 9688 .add(predOps(ARMCC::AL)); 9689 } 9690 9691 BuildMI(DispatchBB, dl, TII->get(ARM::CMPrr)) 9692 .addReg(NewVReg1) 9693 .addReg(VReg2) 9694 .add(predOps(ARMCC::AL)); 9695 } else { 9696 MachineConstantPool *ConstantPool = MF->getConstantPool(); 9697 Type *Int32Ty = Type::getInt32Ty(MF->getFunction().getContext()); 9698 const Constant *C = ConstantInt::get(Int32Ty, NumLPads); 9699 9700 // MachineConstantPool wants an explicit alignment. 9701 unsigned Align = MF->getDataLayout().getPrefTypeAlignment(Int32Ty); 9702 if (Align == 0) 9703 Align = MF->getDataLayout().getTypeAllocSize(C->getType()); 9704 unsigned Idx = ConstantPool->getConstantPoolIndex(C, Align); 9705 9706 Register VReg1 = MRI->createVirtualRegister(TRC); 9707 BuildMI(DispatchBB, dl, TII->get(ARM::LDRcp)) 9708 .addReg(VReg1, RegState::Define) 9709 .addConstantPoolIndex(Idx) 9710 .addImm(0) 9711 .add(predOps(ARMCC::AL)); 9712 BuildMI(DispatchBB, dl, TII->get(ARM::CMPrr)) 9713 .addReg(NewVReg1) 9714 .addReg(VReg1, RegState::Kill) 9715 .add(predOps(ARMCC::AL)); 9716 } 9717 9718 BuildMI(DispatchBB, dl, TII->get(ARM::Bcc)) 9719 .addMBB(TrapBB) 9720 .addImm(ARMCC::HI) 9721 .addReg(ARM::CPSR); 9722 9723 Register NewVReg3 = MRI->createVirtualRegister(TRC); 9724 BuildMI(DispContBB, dl, TII->get(ARM::MOVsi), NewVReg3) 9725 .addReg(NewVReg1) 9726 .addImm(ARM_AM::getSORegOpc(ARM_AM::lsl, 2)) 9727 .add(predOps(ARMCC::AL)) 9728 .add(condCodeOp()); 9729 Register NewVReg4 = MRI->createVirtualRegister(TRC); 9730 BuildMI(DispContBB, dl, TII->get(ARM::LEApcrelJT), NewVReg4) 9731 .addJumpTableIndex(MJTI) 9732 .add(predOps(ARMCC::AL)); 9733 9734 MachineMemOperand *JTMMOLd = MF->getMachineMemOperand( 9735 MachinePointerInfo::getJumpTable(*MF), MachineMemOperand::MOLoad, 4, 4); 9736 Register NewVReg5 = MRI->createVirtualRegister(TRC); 9737 BuildMI(DispContBB, dl, TII->get(ARM::LDRrs), NewVReg5) 9738 .addReg(NewVReg3, RegState::Kill) 9739 .addReg(NewVReg4) 9740 .addImm(0) 9741 .addMemOperand(JTMMOLd) 9742 .add(predOps(ARMCC::AL)); 9743 9744 if (IsPositionIndependent) { 9745 BuildMI(DispContBB, dl, TII->get(ARM::BR_JTadd)) 9746 .addReg(NewVReg5, RegState::Kill) 9747 .addReg(NewVReg4) 9748 .addJumpTableIndex(MJTI); 9749 } else { 9750 BuildMI(DispContBB, dl, TII->get(ARM::BR_JTr)) 9751 .addReg(NewVReg5, RegState::Kill) 9752 .addJumpTableIndex(MJTI); 9753 } 9754 } 9755 9756 // Add the jump table entries as successors to the MBB. 9757 SmallPtrSet<MachineBasicBlock*, 8> SeenMBBs; 9758 for (std::vector<MachineBasicBlock*>::iterator 9759 I = LPadList.begin(), E = LPadList.end(); I != E; ++I) { 9760 MachineBasicBlock *CurMBB = *I; 9761 if (SeenMBBs.insert(CurMBB).second) 9762 DispContBB->addSuccessor(CurMBB); 9763 } 9764 9765 // N.B. the order the invoke BBs are processed in doesn't matter here. 9766 const MCPhysReg *SavedRegs = RI.getCalleeSavedRegs(MF); 9767 SmallVector<MachineBasicBlock*, 64> MBBLPads; 9768 for (MachineBasicBlock *BB : InvokeBBs) { 9769 9770 // Remove the landing pad successor from the invoke block and replace it 9771 // with the new dispatch block. 9772 SmallVector<MachineBasicBlock*, 4> Successors(BB->succ_begin(), 9773 BB->succ_end()); 9774 while (!Successors.empty()) { 9775 MachineBasicBlock *SMBB = Successors.pop_back_val(); 9776 if (SMBB->isEHPad()) { 9777 BB->removeSuccessor(SMBB); 9778 MBBLPads.push_back(SMBB); 9779 } 9780 } 9781 9782 BB->addSuccessor(DispatchBB, BranchProbability::getZero()); 9783 BB->normalizeSuccProbs(); 9784 9785 // Find the invoke call and mark all of the callee-saved registers as 9786 // 'implicit defined' so that they're spilled. This prevents code from 9787 // moving instructions to before the EH block, where they will never be 9788 // executed. 9789 for (MachineBasicBlock::reverse_iterator 9790 II = BB->rbegin(), IE = BB->rend(); II != IE; ++II) { 9791 if (!II->isCall()) continue; 9792 9793 DenseMap<unsigned, bool> DefRegs; 9794 for (MachineInstr::mop_iterator 9795 OI = II->operands_begin(), OE = II->operands_end(); 9796 OI != OE; ++OI) { 9797 if (!OI->isReg()) continue; 9798 DefRegs[OI->getReg()] = true; 9799 } 9800 9801 MachineInstrBuilder MIB(*MF, &*II); 9802 9803 for (unsigned i = 0; SavedRegs[i] != 0; ++i) { 9804 unsigned Reg = SavedRegs[i]; 9805 if (Subtarget->isThumb2() && 9806 !ARM::tGPRRegClass.contains(Reg) && 9807 !ARM::hGPRRegClass.contains(Reg)) 9808 continue; 9809 if (Subtarget->isThumb1Only() && !ARM::tGPRRegClass.contains(Reg)) 9810 continue; 9811 if (!Subtarget->isThumb() && !ARM::GPRRegClass.contains(Reg)) 9812 continue; 9813 if (!DefRegs[Reg]) 9814 MIB.addReg(Reg, RegState::ImplicitDefine | RegState::Dead); 9815 } 9816 9817 break; 9818 } 9819 } 9820 9821 // Mark all former landing pads as non-landing pads. The dispatch is the only 9822 // landing pad now. 9823 for (SmallVectorImpl<MachineBasicBlock*>::iterator 9824 I = MBBLPads.begin(), E = MBBLPads.end(); I != E; ++I) 9825 (*I)->setIsEHPad(false); 9826 9827 // The instruction is gone now. 9828 MI.eraseFromParent(); 9829 } 9830 9831 static 9832 MachineBasicBlock *OtherSucc(MachineBasicBlock *MBB, MachineBasicBlock *Succ) { 9833 for (MachineBasicBlock::succ_iterator I = MBB->succ_begin(), 9834 E = MBB->succ_end(); I != E; ++I) 9835 if (*I != Succ) 9836 return *I; 9837 llvm_unreachable("Expecting a BB with two successors!"); 9838 } 9839 9840 /// Return the load opcode for a given load size. If load size >= 8, 9841 /// neon opcode will be returned. 9842 static unsigned getLdOpcode(unsigned LdSize, bool IsThumb1, bool IsThumb2) { 9843 if (LdSize >= 8) 9844 return LdSize == 16 ? ARM::VLD1q32wb_fixed 9845 : LdSize == 8 ? ARM::VLD1d32wb_fixed : 0; 9846 if (IsThumb1) 9847 return LdSize == 4 ? ARM::tLDRi 9848 : LdSize == 2 ? ARM::tLDRHi 9849 : LdSize == 1 ? ARM::tLDRBi : 0; 9850 if (IsThumb2) 9851 return LdSize == 4 ? ARM::t2LDR_POST 9852 : LdSize == 2 ? ARM::t2LDRH_POST 9853 : LdSize == 1 ? ARM::t2LDRB_POST : 0; 9854 return LdSize == 4 ? ARM::LDR_POST_IMM 9855 : LdSize == 2 ? ARM::LDRH_POST 9856 : LdSize == 1 ? ARM::LDRB_POST_IMM : 0; 9857 } 9858 9859 /// Return the store opcode for a given store size. If store size >= 8, 9860 /// neon opcode will be returned. 9861 static unsigned getStOpcode(unsigned StSize, bool IsThumb1, bool IsThumb2) { 9862 if (StSize >= 8) 9863 return StSize == 16 ? ARM::VST1q32wb_fixed 9864 : StSize == 8 ? ARM::VST1d32wb_fixed : 0; 9865 if (IsThumb1) 9866 return StSize == 4 ? ARM::tSTRi 9867 : StSize == 2 ? ARM::tSTRHi 9868 : StSize == 1 ? ARM::tSTRBi : 0; 9869 if (IsThumb2) 9870 return StSize == 4 ? ARM::t2STR_POST 9871 : StSize == 2 ? ARM::t2STRH_POST 9872 : StSize == 1 ? ARM::t2STRB_POST : 0; 9873 return StSize == 4 ? ARM::STR_POST_IMM 9874 : StSize == 2 ? ARM::STRH_POST 9875 : StSize == 1 ? ARM::STRB_POST_IMM : 0; 9876 } 9877 9878 /// Emit a post-increment load operation with given size. The instructions 9879 /// will be added to BB at Pos. 9880 static void emitPostLd(MachineBasicBlock *BB, MachineBasicBlock::iterator Pos, 9881 const TargetInstrInfo *TII, const DebugLoc &dl, 9882 unsigned LdSize, unsigned Data, unsigned AddrIn, 9883 unsigned AddrOut, bool IsThumb1, bool IsThumb2) { 9884 unsigned LdOpc = getLdOpcode(LdSize, IsThumb1, IsThumb2); 9885 assert(LdOpc != 0 && "Should have a load opcode"); 9886 if (LdSize >= 8) { 9887 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data) 9888 .addReg(AddrOut, RegState::Define) 9889 .addReg(AddrIn) 9890 .addImm(0) 9891 .add(predOps(ARMCC::AL)); 9892 } else if (IsThumb1) { 9893 // load + update AddrIn 9894 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data) 9895 .addReg(AddrIn) 9896 .addImm(0) 9897 .add(predOps(ARMCC::AL)); 9898 BuildMI(*BB, Pos, dl, TII->get(ARM::tADDi8), AddrOut) 9899 .add(t1CondCodeOp()) 9900 .addReg(AddrIn) 9901 .addImm(LdSize) 9902 .add(predOps(ARMCC::AL)); 9903 } else if (IsThumb2) { 9904 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data) 9905 .addReg(AddrOut, RegState::Define) 9906 .addReg(AddrIn) 9907 .addImm(LdSize) 9908 .add(predOps(ARMCC::AL)); 9909 } else { // arm 9910 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data) 9911 .addReg(AddrOut, RegState::Define) 9912 .addReg(AddrIn) 9913 .addReg(0) 9914 .addImm(LdSize) 9915 .add(predOps(ARMCC::AL)); 9916 } 9917 } 9918 9919 /// Emit a post-increment store operation with given size. The instructions 9920 /// will be added to BB at Pos. 9921 static void emitPostSt(MachineBasicBlock *BB, MachineBasicBlock::iterator Pos, 9922 const TargetInstrInfo *TII, const DebugLoc &dl, 9923 unsigned StSize, unsigned Data, unsigned AddrIn, 9924 unsigned AddrOut, bool IsThumb1, bool IsThumb2) { 9925 unsigned StOpc = getStOpcode(StSize, IsThumb1, IsThumb2); 9926 assert(StOpc != 0 && "Should have a store opcode"); 9927 if (StSize >= 8) { 9928 BuildMI(*BB, Pos, dl, TII->get(StOpc), AddrOut) 9929 .addReg(AddrIn) 9930 .addImm(0) 9931 .addReg(Data) 9932 .add(predOps(ARMCC::AL)); 9933 } else if (IsThumb1) { 9934 // store + update AddrIn 9935 BuildMI(*BB, Pos, dl, TII->get(StOpc)) 9936 .addReg(Data) 9937 .addReg(AddrIn) 9938 .addImm(0) 9939 .add(predOps(ARMCC::AL)); 9940 BuildMI(*BB, Pos, dl, TII->get(ARM::tADDi8), AddrOut) 9941 .add(t1CondCodeOp()) 9942 .addReg(AddrIn) 9943 .addImm(StSize) 9944 .add(predOps(ARMCC::AL)); 9945 } else if (IsThumb2) { 9946 BuildMI(*BB, Pos, dl, TII->get(StOpc), AddrOut) 9947 .addReg(Data) 9948 .addReg(AddrIn) 9949 .addImm(StSize) 9950 .add(predOps(ARMCC::AL)); 9951 } else { // arm 9952 BuildMI(*BB, Pos, dl, TII->get(StOpc), AddrOut) 9953 .addReg(Data) 9954 .addReg(AddrIn) 9955 .addReg(0) 9956 .addImm(StSize) 9957 .add(predOps(ARMCC::AL)); 9958 } 9959 } 9960 9961 MachineBasicBlock * 9962 ARMTargetLowering::EmitStructByval(MachineInstr &MI, 9963 MachineBasicBlock *BB) const { 9964 // This pseudo instruction has 3 operands: dst, src, size 9965 // We expand it to a loop if size > Subtarget->getMaxInlineSizeThreshold(). 9966 // Otherwise, we will generate unrolled scalar copies. 9967 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 9968 const BasicBlock *LLVM_BB = BB->getBasicBlock(); 9969 MachineFunction::iterator It = ++BB->getIterator(); 9970 9971 Register dest = MI.getOperand(0).getReg(); 9972 Register src = MI.getOperand(1).getReg(); 9973 unsigned SizeVal = MI.getOperand(2).getImm(); 9974 unsigned Align = MI.getOperand(3).getImm(); 9975 DebugLoc dl = MI.getDebugLoc(); 9976 9977 MachineFunction *MF = BB->getParent(); 9978 MachineRegisterInfo &MRI = MF->getRegInfo(); 9979 unsigned UnitSize = 0; 9980 const TargetRegisterClass *TRC = nullptr; 9981 const TargetRegisterClass *VecTRC = nullptr; 9982 9983 bool IsThumb1 = Subtarget->isThumb1Only(); 9984 bool IsThumb2 = Subtarget->isThumb2(); 9985 bool IsThumb = Subtarget->isThumb(); 9986 9987 if (Align & 1) { 9988 UnitSize = 1; 9989 } else if (Align & 2) { 9990 UnitSize = 2; 9991 } else { 9992 // Check whether we can use NEON instructions. 9993 if (!MF->getFunction().hasFnAttribute(Attribute::NoImplicitFloat) && 9994 Subtarget->hasNEON()) { 9995 if ((Align % 16 == 0) && SizeVal >= 16) 9996 UnitSize = 16; 9997 else if ((Align % 8 == 0) && SizeVal >= 8) 9998 UnitSize = 8; 9999 } 10000 // Can't use NEON instructions. 10001 if (UnitSize == 0) 10002 UnitSize = 4; 10003 } 10004 10005 // Select the correct opcode and register class for unit size load/store 10006 bool IsNeon = UnitSize >= 8; 10007 TRC = IsThumb ? &ARM::tGPRRegClass : &ARM::GPRRegClass; 10008 if (IsNeon) 10009 VecTRC = UnitSize == 16 ? &ARM::DPairRegClass 10010 : UnitSize == 8 ? &ARM::DPRRegClass 10011 : nullptr; 10012 10013 unsigned BytesLeft = SizeVal % UnitSize; 10014 unsigned LoopSize = SizeVal - BytesLeft; 10015 10016 if (SizeVal <= Subtarget->getMaxInlineSizeThreshold()) { 10017 // Use LDR and STR to copy. 10018 // [scratch, srcOut] = LDR_POST(srcIn, UnitSize) 10019 // [destOut] = STR_POST(scratch, destIn, UnitSize) 10020 unsigned srcIn = src; 10021 unsigned destIn = dest; 10022 for (unsigned i = 0; i < LoopSize; i+=UnitSize) { 10023 Register srcOut = MRI.createVirtualRegister(TRC); 10024 Register destOut = MRI.createVirtualRegister(TRC); 10025 Register scratch = MRI.createVirtualRegister(IsNeon ? VecTRC : TRC); 10026 emitPostLd(BB, MI, TII, dl, UnitSize, scratch, srcIn, srcOut, 10027 IsThumb1, IsThumb2); 10028 emitPostSt(BB, MI, TII, dl, UnitSize, scratch, destIn, destOut, 10029 IsThumb1, IsThumb2); 10030 srcIn = srcOut; 10031 destIn = destOut; 10032 } 10033 10034 // Handle the leftover bytes with LDRB and STRB. 10035 // [scratch, srcOut] = LDRB_POST(srcIn, 1) 10036 // [destOut] = STRB_POST(scratch, destIn, 1) 10037 for (unsigned i = 0; i < BytesLeft; i++) { 10038 Register srcOut = MRI.createVirtualRegister(TRC); 10039 Register destOut = MRI.createVirtualRegister(TRC); 10040 Register scratch = MRI.createVirtualRegister(TRC); 10041 emitPostLd(BB, MI, TII, dl, 1, scratch, srcIn, srcOut, 10042 IsThumb1, IsThumb2); 10043 emitPostSt(BB, MI, TII, dl, 1, scratch, destIn, destOut, 10044 IsThumb1, IsThumb2); 10045 srcIn = srcOut; 10046 destIn = destOut; 10047 } 10048 MI.eraseFromParent(); // The instruction is gone now. 10049 return BB; 10050 } 10051 10052 // Expand the pseudo op to a loop. 10053 // thisMBB: 10054 // ... 10055 // movw varEnd, # --> with thumb2 10056 // movt varEnd, # 10057 // ldrcp varEnd, idx --> without thumb2 10058 // fallthrough --> loopMBB 10059 // loopMBB: 10060 // PHI varPhi, varEnd, varLoop 10061 // PHI srcPhi, src, srcLoop 10062 // PHI destPhi, dst, destLoop 10063 // [scratch, srcLoop] = LDR_POST(srcPhi, UnitSize) 10064 // [destLoop] = STR_POST(scratch, destPhi, UnitSize) 10065 // subs varLoop, varPhi, #UnitSize 10066 // bne loopMBB 10067 // fallthrough --> exitMBB 10068 // exitMBB: 10069 // epilogue to handle left-over bytes 10070 // [scratch, srcOut] = LDRB_POST(srcLoop, 1) 10071 // [destOut] = STRB_POST(scratch, destLoop, 1) 10072 MachineBasicBlock *loopMBB = MF->CreateMachineBasicBlock(LLVM_BB); 10073 MachineBasicBlock *exitMBB = MF->CreateMachineBasicBlock(LLVM_BB); 10074 MF->insert(It, loopMBB); 10075 MF->insert(It, exitMBB); 10076 10077 // Transfer the remainder of BB and its successor edges to exitMBB. 10078 exitMBB->splice(exitMBB->begin(), BB, 10079 std::next(MachineBasicBlock::iterator(MI)), BB->end()); 10080 exitMBB->transferSuccessorsAndUpdatePHIs(BB); 10081 10082 // Load an immediate to varEnd. 10083 Register varEnd = MRI.createVirtualRegister(TRC); 10084 if (Subtarget->useMovt()) { 10085 unsigned Vtmp = varEnd; 10086 if ((LoopSize & 0xFFFF0000) != 0) 10087 Vtmp = MRI.createVirtualRegister(TRC); 10088 BuildMI(BB, dl, TII->get(IsThumb ? ARM::t2MOVi16 : ARM::MOVi16), Vtmp) 10089 .addImm(LoopSize & 0xFFFF) 10090 .add(predOps(ARMCC::AL)); 10091 10092 if ((LoopSize & 0xFFFF0000) != 0) 10093 BuildMI(BB, dl, TII->get(IsThumb ? ARM::t2MOVTi16 : ARM::MOVTi16), varEnd) 10094 .addReg(Vtmp) 10095 .addImm(LoopSize >> 16) 10096 .add(predOps(ARMCC::AL)); 10097 } else { 10098 MachineConstantPool *ConstantPool = MF->getConstantPool(); 10099 Type *Int32Ty = Type::getInt32Ty(MF->getFunction().getContext()); 10100 const Constant *C = ConstantInt::get(Int32Ty, LoopSize); 10101 10102 // MachineConstantPool wants an explicit alignment. 10103 unsigned Align = MF->getDataLayout().getPrefTypeAlignment(Int32Ty); 10104 if (Align == 0) 10105 Align = MF->getDataLayout().getTypeAllocSize(C->getType()); 10106 unsigned Idx = ConstantPool->getConstantPoolIndex(C, Align); 10107 MachineMemOperand *CPMMO = 10108 MF->getMachineMemOperand(MachinePointerInfo::getConstantPool(*MF), 10109 MachineMemOperand::MOLoad, 4, 4); 10110 10111 if (IsThumb) 10112 BuildMI(*BB, MI, dl, TII->get(ARM::tLDRpci)) 10113 .addReg(varEnd, RegState::Define) 10114 .addConstantPoolIndex(Idx) 10115 .add(predOps(ARMCC::AL)) 10116 .addMemOperand(CPMMO); 10117 else 10118 BuildMI(*BB, MI, dl, TII->get(ARM::LDRcp)) 10119 .addReg(varEnd, RegState::Define) 10120 .addConstantPoolIndex(Idx) 10121 .addImm(0) 10122 .add(predOps(ARMCC::AL)) 10123 .addMemOperand(CPMMO); 10124 } 10125 BB->addSuccessor(loopMBB); 10126 10127 // Generate the loop body: 10128 // varPhi = PHI(varLoop, varEnd) 10129 // srcPhi = PHI(srcLoop, src) 10130 // destPhi = PHI(destLoop, dst) 10131 MachineBasicBlock *entryBB = BB; 10132 BB = loopMBB; 10133 Register varLoop = MRI.createVirtualRegister(TRC); 10134 Register varPhi = MRI.createVirtualRegister(TRC); 10135 Register srcLoop = MRI.createVirtualRegister(TRC); 10136 Register srcPhi = MRI.createVirtualRegister(TRC); 10137 Register destLoop = MRI.createVirtualRegister(TRC); 10138 Register destPhi = MRI.createVirtualRegister(TRC); 10139 10140 BuildMI(*BB, BB->begin(), dl, TII->get(ARM::PHI), varPhi) 10141 .addReg(varLoop).addMBB(loopMBB) 10142 .addReg(varEnd).addMBB(entryBB); 10143 BuildMI(BB, dl, TII->get(ARM::PHI), srcPhi) 10144 .addReg(srcLoop).addMBB(loopMBB) 10145 .addReg(src).addMBB(entryBB); 10146 BuildMI(BB, dl, TII->get(ARM::PHI), destPhi) 10147 .addReg(destLoop).addMBB(loopMBB) 10148 .addReg(dest).addMBB(entryBB); 10149 10150 // [scratch, srcLoop] = LDR_POST(srcPhi, UnitSize) 10151 // [destLoop] = STR_POST(scratch, destPhi, UnitSiz) 10152 Register scratch = MRI.createVirtualRegister(IsNeon ? VecTRC : TRC); 10153 emitPostLd(BB, BB->end(), TII, dl, UnitSize, scratch, srcPhi, srcLoop, 10154 IsThumb1, IsThumb2); 10155 emitPostSt(BB, BB->end(), TII, dl, UnitSize, scratch, destPhi, destLoop, 10156 IsThumb1, IsThumb2); 10157 10158 // Decrement loop variable by UnitSize. 10159 if (IsThumb1) { 10160 BuildMI(*BB, BB->end(), dl, TII->get(ARM::tSUBi8), varLoop) 10161 .add(t1CondCodeOp()) 10162 .addReg(varPhi) 10163 .addImm(UnitSize) 10164 .add(predOps(ARMCC::AL)); 10165 } else { 10166 MachineInstrBuilder MIB = 10167 BuildMI(*BB, BB->end(), dl, 10168 TII->get(IsThumb2 ? ARM::t2SUBri : ARM::SUBri), varLoop); 10169 MIB.addReg(varPhi) 10170 .addImm(UnitSize) 10171 .add(predOps(ARMCC::AL)) 10172 .add(condCodeOp()); 10173 MIB->getOperand(5).setReg(ARM::CPSR); 10174 MIB->getOperand(5).setIsDef(true); 10175 } 10176 BuildMI(*BB, BB->end(), dl, 10177 TII->get(IsThumb1 ? ARM::tBcc : IsThumb2 ? ARM::t2Bcc : ARM::Bcc)) 10178 .addMBB(loopMBB).addImm(ARMCC::NE).addReg(ARM::CPSR); 10179 10180 // loopMBB can loop back to loopMBB or fall through to exitMBB. 10181 BB->addSuccessor(loopMBB); 10182 BB->addSuccessor(exitMBB); 10183 10184 // Add epilogue to handle BytesLeft. 10185 BB = exitMBB; 10186 auto StartOfExit = exitMBB->begin(); 10187 10188 // [scratch, srcOut] = LDRB_POST(srcLoop, 1) 10189 // [destOut] = STRB_POST(scratch, destLoop, 1) 10190 unsigned srcIn = srcLoop; 10191 unsigned destIn = destLoop; 10192 for (unsigned i = 0; i < BytesLeft; i++) { 10193 Register srcOut = MRI.createVirtualRegister(TRC); 10194 Register destOut = MRI.createVirtualRegister(TRC); 10195 Register scratch = MRI.createVirtualRegister(TRC); 10196 emitPostLd(BB, StartOfExit, TII, dl, 1, scratch, srcIn, srcOut, 10197 IsThumb1, IsThumb2); 10198 emitPostSt(BB, StartOfExit, TII, dl, 1, scratch, destIn, destOut, 10199 IsThumb1, IsThumb2); 10200 srcIn = srcOut; 10201 destIn = destOut; 10202 } 10203 10204 MI.eraseFromParent(); // The instruction is gone now. 10205 return BB; 10206 } 10207 10208 MachineBasicBlock * 10209 ARMTargetLowering::EmitLowered__chkstk(MachineInstr &MI, 10210 MachineBasicBlock *MBB) const { 10211 const TargetMachine &TM = getTargetMachine(); 10212 const TargetInstrInfo &TII = *Subtarget->getInstrInfo(); 10213 DebugLoc DL = MI.getDebugLoc(); 10214 10215 assert(Subtarget->isTargetWindows() && 10216 "__chkstk is only supported on Windows"); 10217 assert(Subtarget->isThumb2() && "Windows on ARM requires Thumb-2 mode"); 10218 10219 // __chkstk takes the number of words to allocate on the stack in R4, and 10220 // returns the stack adjustment in number of bytes in R4. This will not 10221 // clober any other registers (other than the obvious lr). 10222 // 10223 // Although, technically, IP should be considered a register which may be 10224 // clobbered, the call itself will not touch it. Windows on ARM is a pure 10225 // thumb-2 environment, so there is no interworking required. As a result, we 10226 // do not expect a veneer to be emitted by the linker, clobbering IP. 10227 // 10228 // Each module receives its own copy of __chkstk, so no import thunk is 10229 // required, again, ensuring that IP is not clobbered. 10230 // 10231 // Finally, although some linkers may theoretically provide a trampoline for 10232 // out of range calls (which is quite common due to a 32M range limitation of 10233 // branches for Thumb), we can generate the long-call version via 10234 // -mcmodel=large, alleviating the need for the trampoline which may clobber 10235 // IP. 10236 10237 switch (TM.getCodeModel()) { 10238 case CodeModel::Tiny: 10239 llvm_unreachable("Tiny code model not available on ARM."); 10240 case CodeModel::Small: 10241 case CodeModel::Medium: 10242 case CodeModel::Kernel: 10243 BuildMI(*MBB, MI, DL, TII.get(ARM::tBL)) 10244 .add(predOps(ARMCC::AL)) 10245 .addExternalSymbol("__chkstk") 10246 .addReg(ARM::R4, RegState::Implicit | RegState::Kill) 10247 .addReg(ARM::R4, RegState::Implicit | RegState::Define) 10248 .addReg(ARM::R12, 10249 RegState::Implicit | RegState::Define | RegState::Dead) 10250 .addReg(ARM::CPSR, 10251 RegState::Implicit | RegState::Define | RegState::Dead); 10252 break; 10253 case CodeModel::Large: { 10254 MachineRegisterInfo &MRI = MBB->getParent()->getRegInfo(); 10255 Register Reg = MRI.createVirtualRegister(&ARM::rGPRRegClass); 10256 10257 BuildMI(*MBB, MI, DL, TII.get(ARM::t2MOVi32imm), Reg) 10258 .addExternalSymbol("__chkstk"); 10259 BuildMI(*MBB, MI, DL, TII.get(ARM::tBLXr)) 10260 .add(predOps(ARMCC::AL)) 10261 .addReg(Reg, RegState::Kill) 10262 .addReg(ARM::R4, RegState::Implicit | RegState::Kill) 10263 .addReg(ARM::R4, RegState::Implicit | RegState::Define) 10264 .addReg(ARM::R12, 10265 RegState::Implicit | RegState::Define | RegState::Dead) 10266 .addReg(ARM::CPSR, 10267 RegState::Implicit | RegState::Define | RegState::Dead); 10268 break; 10269 } 10270 } 10271 10272 BuildMI(*MBB, MI, DL, TII.get(ARM::t2SUBrr), ARM::SP) 10273 .addReg(ARM::SP, RegState::Kill) 10274 .addReg(ARM::R4, RegState::Kill) 10275 .setMIFlags(MachineInstr::FrameSetup) 10276 .add(predOps(ARMCC::AL)) 10277 .add(condCodeOp()); 10278 10279 MI.eraseFromParent(); 10280 return MBB; 10281 } 10282 10283 MachineBasicBlock * 10284 ARMTargetLowering::EmitLowered__dbzchk(MachineInstr &MI, 10285 MachineBasicBlock *MBB) const { 10286 DebugLoc DL = MI.getDebugLoc(); 10287 MachineFunction *MF = MBB->getParent(); 10288 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 10289 10290 MachineBasicBlock *ContBB = MF->CreateMachineBasicBlock(); 10291 MF->insert(++MBB->getIterator(), ContBB); 10292 ContBB->splice(ContBB->begin(), MBB, 10293 std::next(MachineBasicBlock::iterator(MI)), MBB->end()); 10294 ContBB->transferSuccessorsAndUpdatePHIs(MBB); 10295 MBB->addSuccessor(ContBB); 10296 10297 MachineBasicBlock *TrapBB = MF->CreateMachineBasicBlock(); 10298 BuildMI(TrapBB, DL, TII->get(ARM::t__brkdiv0)); 10299 MF->push_back(TrapBB); 10300 MBB->addSuccessor(TrapBB); 10301 10302 BuildMI(*MBB, MI, DL, TII->get(ARM::tCMPi8)) 10303 .addReg(MI.getOperand(0).getReg()) 10304 .addImm(0) 10305 .add(predOps(ARMCC::AL)); 10306 BuildMI(*MBB, MI, DL, TII->get(ARM::t2Bcc)) 10307 .addMBB(TrapBB) 10308 .addImm(ARMCC::EQ) 10309 .addReg(ARM::CPSR); 10310 10311 MI.eraseFromParent(); 10312 return ContBB; 10313 } 10314 10315 // The CPSR operand of SelectItr might be missing a kill marker 10316 // because there were multiple uses of CPSR, and ISel didn't know 10317 // which to mark. Figure out whether SelectItr should have had a 10318 // kill marker, and set it if it should. Returns the correct kill 10319 // marker value. 10320 static bool checkAndUpdateCPSRKill(MachineBasicBlock::iterator SelectItr, 10321 MachineBasicBlock* BB, 10322 const TargetRegisterInfo* TRI) { 10323 // Scan forward through BB for a use/def of CPSR. 10324 MachineBasicBlock::iterator miI(std::next(SelectItr)); 10325 for (MachineBasicBlock::iterator miE = BB->end(); miI != miE; ++miI) { 10326 const MachineInstr& mi = *miI; 10327 if (mi.readsRegister(ARM::CPSR)) 10328 return false; 10329 if (mi.definesRegister(ARM::CPSR)) 10330 break; // Should have kill-flag - update below. 10331 } 10332 10333 // If we hit the end of the block, check whether CPSR is live into a 10334 // successor. 10335 if (miI == BB->end()) { 10336 for (MachineBasicBlock::succ_iterator sItr = BB->succ_begin(), 10337 sEnd = BB->succ_end(); 10338 sItr != sEnd; ++sItr) { 10339 MachineBasicBlock* succ = *sItr; 10340 if (succ->isLiveIn(ARM::CPSR)) 10341 return false; 10342 } 10343 } 10344 10345 // We found a def, or hit the end of the basic block and CPSR wasn't live 10346 // out. SelectMI should have a kill flag on CPSR. 10347 SelectItr->addRegisterKilled(ARM::CPSR, TRI); 10348 return true; 10349 } 10350 10351 MachineBasicBlock * 10352 ARMTargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI, 10353 MachineBasicBlock *BB) const { 10354 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 10355 DebugLoc dl = MI.getDebugLoc(); 10356 bool isThumb2 = Subtarget->isThumb2(); 10357 switch (MI.getOpcode()) { 10358 default: { 10359 MI.print(errs()); 10360 llvm_unreachable("Unexpected instr type to insert"); 10361 } 10362 10363 // Thumb1 post-indexed loads are really just single-register LDMs. 10364 case ARM::tLDR_postidx: { 10365 MachineOperand Def(MI.getOperand(1)); 10366 BuildMI(*BB, MI, dl, TII->get(ARM::tLDMIA_UPD)) 10367 .add(Def) // Rn_wb 10368 .add(MI.getOperand(2)) // Rn 10369 .add(MI.getOperand(3)) // PredImm 10370 .add(MI.getOperand(4)) // PredReg 10371 .add(MI.getOperand(0)) // Rt 10372 .cloneMemRefs(MI); 10373 MI.eraseFromParent(); 10374 return BB; 10375 } 10376 10377 // The Thumb2 pre-indexed stores have the same MI operands, they just 10378 // define them differently in the .td files from the isel patterns, so 10379 // they need pseudos. 10380 case ARM::t2STR_preidx: 10381 MI.setDesc(TII->get(ARM::t2STR_PRE)); 10382 return BB; 10383 case ARM::t2STRB_preidx: 10384 MI.setDesc(TII->get(ARM::t2STRB_PRE)); 10385 return BB; 10386 case ARM::t2STRH_preidx: 10387 MI.setDesc(TII->get(ARM::t2STRH_PRE)); 10388 return BB; 10389 10390 case ARM::STRi_preidx: 10391 case ARM::STRBi_preidx: { 10392 unsigned NewOpc = MI.getOpcode() == ARM::STRi_preidx ? ARM::STR_PRE_IMM 10393 : ARM::STRB_PRE_IMM; 10394 // Decode the offset. 10395 unsigned Offset = MI.getOperand(4).getImm(); 10396 bool isSub = ARM_AM::getAM2Op(Offset) == ARM_AM::sub; 10397 Offset = ARM_AM::getAM2Offset(Offset); 10398 if (isSub) 10399 Offset = -Offset; 10400 10401 MachineMemOperand *MMO = *MI.memoperands_begin(); 10402 BuildMI(*BB, MI, dl, TII->get(NewOpc)) 10403 .add(MI.getOperand(0)) // Rn_wb 10404 .add(MI.getOperand(1)) // Rt 10405 .add(MI.getOperand(2)) // Rn 10406 .addImm(Offset) // offset (skip GPR==zero_reg) 10407 .add(MI.getOperand(5)) // pred 10408 .add(MI.getOperand(6)) 10409 .addMemOperand(MMO); 10410 MI.eraseFromParent(); 10411 return BB; 10412 } 10413 case ARM::STRr_preidx: 10414 case ARM::STRBr_preidx: 10415 case ARM::STRH_preidx: { 10416 unsigned NewOpc; 10417 switch (MI.getOpcode()) { 10418 default: llvm_unreachable("unexpected opcode!"); 10419 case ARM::STRr_preidx: NewOpc = ARM::STR_PRE_REG; break; 10420 case ARM::STRBr_preidx: NewOpc = ARM::STRB_PRE_REG; break; 10421 case ARM::STRH_preidx: NewOpc = ARM::STRH_PRE; break; 10422 } 10423 MachineInstrBuilder MIB = BuildMI(*BB, MI, dl, TII->get(NewOpc)); 10424 for (unsigned i = 0; i < MI.getNumOperands(); ++i) 10425 MIB.add(MI.getOperand(i)); 10426 MI.eraseFromParent(); 10427 return BB; 10428 } 10429 10430 case ARM::tMOVCCr_pseudo: { 10431 // To "insert" a SELECT_CC instruction, we actually have to insert the 10432 // diamond control-flow pattern. The incoming instruction knows the 10433 // destination vreg to set, the condition code register to branch on, the 10434 // true/false values to select between, and a branch opcode to use. 10435 const BasicBlock *LLVM_BB = BB->getBasicBlock(); 10436 MachineFunction::iterator It = ++BB->getIterator(); 10437 10438 // thisMBB: 10439 // ... 10440 // TrueVal = ... 10441 // cmpTY ccX, r1, r2 10442 // bCC copy1MBB 10443 // fallthrough --> copy0MBB 10444 MachineBasicBlock *thisMBB = BB; 10445 MachineFunction *F = BB->getParent(); 10446 MachineBasicBlock *copy0MBB = F->CreateMachineBasicBlock(LLVM_BB); 10447 MachineBasicBlock *sinkMBB = F->CreateMachineBasicBlock(LLVM_BB); 10448 F->insert(It, copy0MBB); 10449 F->insert(It, sinkMBB); 10450 10451 // Check whether CPSR is live past the tMOVCCr_pseudo. 10452 const TargetRegisterInfo *TRI = Subtarget->getRegisterInfo(); 10453 if (!MI.killsRegister(ARM::CPSR) && 10454 !checkAndUpdateCPSRKill(MI, thisMBB, TRI)) { 10455 copy0MBB->addLiveIn(ARM::CPSR); 10456 sinkMBB->addLiveIn(ARM::CPSR); 10457 } 10458 10459 // Transfer the remainder of BB and its successor edges to sinkMBB. 10460 sinkMBB->splice(sinkMBB->begin(), BB, 10461 std::next(MachineBasicBlock::iterator(MI)), BB->end()); 10462 sinkMBB->transferSuccessorsAndUpdatePHIs(BB); 10463 10464 BB->addSuccessor(copy0MBB); 10465 BB->addSuccessor(sinkMBB); 10466 10467 BuildMI(BB, dl, TII->get(ARM::tBcc)) 10468 .addMBB(sinkMBB) 10469 .addImm(MI.getOperand(3).getImm()) 10470 .addReg(MI.getOperand(4).getReg()); 10471 10472 // copy0MBB: 10473 // %FalseValue = ... 10474 // # fallthrough to sinkMBB 10475 BB = copy0MBB; 10476 10477 // Update machine-CFG edges 10478 BB->addSuccessor(sinkMBB); 10479 10480 // sinkMBB: 10481 // %Result = phi [ %FalseValue, copy0MBB ], [ %TrueValue, thisMBB ] 10482 // ... 10483 BB = sinkMBB; 10484 BuildMI(*BB, BB->begin(), dl, TII->get(ARM::PHI), MI.getOperand(0).getReg()) 10485 .addReg(MI.getOperand(1).getReg()) 10486 .addMBB(copy0MBB) 10487 .addReg(MI.getOperand(2).getReg()) 10488 .addMBB(thisMBB); 10489 10490 MI.eraseFromParent(); // The pseudo instruction is gone now. 10491 return BB; 10492 } 10493 10494 case ARM::BCCi64: 10495 case ARM::BCCZi64: { 10496 // If there is an unconditional branch to the other successor, remove it. 10497 BB->erase(std::next(MachineBasicBlock::iterator(MI)), BB->end()); 10498 10499 // Compare both parts that make up the double comparison separately for 10500 // equality. 10501 bool RHSisZero = MI.getOpcode() == ARM::BCCZi64; 10502 10503 Register LHS1 = MI.getOperand(1).getReg(); 10504 Register LHS2 = MI.getOperand(2).getReg(); 10505 if (RHSisZero) { 10506 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPri : ARM::CMPri)) 10507 .addReg(LHS1) 10508 .addImm(0) 10509 .add(predOps(ARMCC::AL)); 10510 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPri : ARM::CMPri)) 10511 .addReg(LHS2).addImm(0) 10512 .addImm(ARMCC::EQ).addReg(ARM::CPSR); 10513 } else { 10514 Register RHS1 = MI.getOperand(3).getReg(); 10515 Register RHS2 = MI.getOperand(4).getReg(); 10516 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPrr : ARM::CMPrr)) 10517 .addReg(LHS1) 10518 .addReg(RHS1) 10519 .add(predOps(ARMCC::AL)); 10520 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPrr : ARM::CMPrr)) 10521 .addReg(LHS2).addReg(RHS2) 10522 .addImm(ARMCC::EQ).addReg(ARM::CPSR); 10523 } 10524 10525 MachineBasicBlock *destMBB = MI.getOperand(RHSisZero ? 3 : 5).getMBB(); 10526 MachineBasicBlock *exitMBB = OtherSucc(BB, destMBB); 10527 if (MI.getOperand(0).getImm() == ARMCC::NE) 10528 std::swap(destMBB, exitMBB); 10529 10530 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2Bcc : ARM::Bcc)) 10531 .addMBB(destMBB).addImm(ARMCC::EQ).addReg(ARM::CPSR); 10532 if (isThumb2) 10533 BuildMI(BB, dl, TII->get(ARM::t2B)) 10534 .addMBB(exitMBB) 10535 .add(predOps(ARMCC::AL)); 10536 else 10537 BuildMI(BB, dl, TII->get(ARM::B)) .addMBB(exitMBB); 10538 10539 MI.eraseFromParent(); // The pseudo instruction is gone now. 10540 return BB; 10541 } 10542 10543 case ARM::Int_eh_sjlj_setjmp: 10544 case ARM::Int_eh_sjlj_setjmp_nofp: 10545 case ARM::tInt_eh_sjlj_setjmp: 10546 case ARM::t2Int_eh_sjlj_setjmp: 10547 case ARM::t2Int_eh_sjlj_setjmp_nofp: 10548 return BB; 10549 10550 case ARM::Int_eh_sjlj_setup_dispatch: 10551 EmitSjLjDispatchBlock(MI, BB); 10552 return BB; 10553 10554 case ARM::ABS: 10555 case ARM::t2ABS: { 10556 // To insert an ABS instruction, we have to insert the 10557 // diamond control-flow pattern. The incoming instruction knows the 10558 // source vreg to test against 0, the destination vreg to set, 10559 // the condition code register to branch on, the 10560 // true/false values to select between, and a branch opcode to use. 10561 // It transforms 10562 // V1 = ABS V0 10563 // into 10564 // V2 = MOVS V0 10565 // BCC (branch to SinkBB if V0 >= 0) 10566 // RSBBB: V3 = RSBri V2, 0 (compute ABS if V2 < 0) 10567 // SinkBB: V1 = PHI(V2, V3) 10568 const BasicBlock *LLVM_BB = BB->getBasicBlock(); 10569 MachineFunction::iterator BBI = ++BB->getIterator(); 10570 MachineFunction *Fn = BB->getParent(); 10571 MachineBasicBlock *RSBBB = Fn->CreateMachineBasicBlock(LLVM_BB); 10572 MachineBasicBlock *SinkBB = Fn->CreateMachineBasicBlock(LLVM_BB); 10573 Fn->insert(BBI, RSBBB); 10574 Fn->insert(BBI, SinkBB); 10575 10576 Register ABSSrcReg = MI.getOperand(1).getReg(); 10577 Register ABSDstReg = MI.getOperand(0).getReg(); 10578 bool ABSSrcKIll = MI.getOperand(1).isKill(); 10579 bool isThumb2 = Subtarget->isThumb2(); 10580 MachineRegisterInfo &MRI = Fn->getRegInfo(); 10581 // In Thumb mode S must not be specified if source register is the SP or 10582 // PC and if destination register is the SP, so restrict register class 10583 Register NewRsbDstReg = MRI.createVirtualRegister( 10584 isThumb2 ? &ARM::rGPRRegClass : &ARM::GPRRegClass); 10585 10586 // Transfer the remainder of BB and its successor edges to sinkMBB. 10587 SinkBB->splice(SinkBB->begin(), BB, 10588 std::next(MachineBasicBlock::iterator(MI)), BB->end()); 10589 SinkBB->transferSuccessorsAndUpdatePHIs(BB); 10590 10591 BB->addSuccessor(RSBBB); 10592 BB->addSuccessor(SinkBB); 10593 10594 // fall through to SinkMBB 10595 RSBBB->addSuccessor(SinkBB); 10596 10597 // insert a cmp at the end of BB 10598 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPri : ARM::CMPri)) 10599 .addReg(ABSSrcReg) 10600 .addImm(0) 10601 .add(predOps(ARMCC::AL)); 10602 10603 // insert a bcc with opposite CC to ARMCC::MI at the end of BB 10604 BuildMI(BB, dl, 10605 TII->get(isThumb2 ? ARM::t2Bcc : ARM::Bcc)).addMBB(SinkBB) 10606 .addImm(ARMCC::getOppositeCondition(ARMCC::MI)).addReg(ARM::CPSR); 10607 10608 // insert rsbri in RSBBB 10609 // Note: BCC and rsbri will be converted into predicated rsbmi 10610 // by if-conversion pass 10611 BuildMI(*RSBBB, RSBBB->begin(), dl, 10612 TII->get(isThumb2 ? ARM::t2RSBri : ARM::RSBri), NewRsbDstReg) 10613 .addReg(ABSSrcReg, ABSSrcKIll ? RegState::Kill : 0) 10614 .addImm(0) 10615 .add(predOps(ARMCC::AL)) 10616 .add(condCodeOp()); 10617 10618 // insert PHI in SinkBB, 10619 // reuse ABSDstReg to not change uses of ABS instruction 10620 BuildMI(*SinkBB, SinkBB->begin(), dl, 10621 TII->get(ARM::PHI), ABSDstReg) 10622 .addReg(NewRsbDstReg).addMBB(RSBBB) 10623 .addReg(ABSSrcReg).addMBB(BB); 10624 10625 // remove ABS instruction 10626 MI.eraseFromParent(); 10627 10628 // return last added BB 10629 return SinkBB; 10630 } 10631 case ARM::COPY_STRUCT_BYVAL_I32: 10632 ++NumLoopByVals; 10633 return EmitStructByval(MI, BB); 10634 case ARM::WIN__CHKSTK: 10635 return EmitLowered__chkstk(MI, BB); 10636 case ARM::WIN__DBZCHK: 10637 return EmitLowered__dbzchk(MI, BB); 10638 } 10639 } 10640 10641 /// Attaches vregs to MEMCPY that it will use as scratch registers 10642 /// when it is expanded into LDM/STM. This is done as a post-isel lowering 10643 /// instead of as a custom inserter because we need the use list from the SDNode. 10644 static void attachMEMCPYScratchRegs(const ARMSubtarget *Subtarget, 10645 MachineInstr &MI, const SDNode *Node) { 10646 bool isThumb1 = Subtarget->isThumb1Only(); 10647 10648 DebugLoc DL = MI.getDebugLoc(); 10649 MachineFunction *MF = MI.getParent()->getParent(); 10650 MachineRegisterInfo &MRI = MF->getRegInfo(); 10651 MachineInstrBuilder MIB(*MF, MI); 10652 10653 // If the new dst/src is unused mark it as dead. 10654 if (!Node->hasAnyUseOfValue(0)) { 10655 MI.getOperand(0).setIsDead(true); 10656 } 10657 if (!Node->hasAnyUseOfValue(1)) { 10658 MI.getOperand(1).setIsDead(true); 10659 } 10660 10661 // The MEMCPY both defines and kills the scratch registers. 10662 for (unsigned I = 0; I != MI.getOperand(4).getImm(); ++I) { 10663 Register TmpReg = MRI.createVirtualRegister(isThumb1 ? &ARM::tGPRRegClass 10664 : &ARM::GPRRegClass); 10665 MIB.addReg(TmpReg, RegState::Define|RegState::Dead); 10666 } 10667 } 10668 10669 void ARMTargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI, 10670 SDNode *Node) const { 10671 if (MI.getOpcode() == ARM::MEMCPY) { 10672 attachMEMCPYScratchRegs(Subtarget, MI, Node); 10673 return; 10674 } 10675 10676 const MCInstrDesc *MCID = &MI.getDesc(); 10677 // Adjust potentially 's' setting instructions after isel, i.e. ADC, SBC, RSB, 10678 // RSC. Coming out of isel, they have an implicit CPSR def, but the optional 10679 // operand is still set to noreg. If needed, set the optional operand's 10680 // register to CPSR, and remove the redundant implicit def. 10681 // 10682 // e.g. ADCS (..., implicit-def CPSR) -> ADC (... opt:def CPSR). 10683 10684 // Rename pseudo opcodes. 10685 unsigned NewOpc = convertAddSubFlagsOpcode(MI.getOpcode()); 10686 unsigned ccOutIdx; 10687 if (NewOpc) { 10688 const ARMBaseInstrInfo *TII = Subtarget->getInstrInfo(); 10689 MCID = &TII->get(NewOpc); 10690 10691 assert(MCID->getNumOperands() == 10692 MI.getDesc().getNumOperands() + 5 - MI.getDesc().getSize() 10693 && "converted opcode should be the same except for cc_out" 10694 " (and, on Thumb1, pred)"); 10695 10696 MI.setDesc(*MCID); 10697 10698 // Add the optional cc_out operand 10699 MI.addOperand(MachineOperand::CreateReg(0, /*isDef=*/true)); 10700 10701 // On Thumb1, move all input operands to the end, then add the predicate 10702 if (Subtarget->isThumb1Only()) { 10703 for (unsigned c = MCID->getNumOperands() - 4; c--;) { 10704 MI.addOperand(MI.getOperand(1)); 10705 MI.RemoveOperand(1); 10706 } 10707 10708 // Restore the ties 10709 for (unsigned i = MI.getNumOperands(); i--;) { 10710 const MachineOperand& op = MI.getOperand(i); 10711 if (op.isReg() && op.isUse()) { 10712 int DefIdx = MCID->getOperandConstraint(i, MCOI::TIED_TO); 10713 if (DefIdx != -1) 10714 MI.tieOperands(DefIdx, i); 10715 } 10716 } 10717 10718 MI.addOperand(MachineOperand::CreateImm(ARMCC::AL)); 10719 MI.addOperand(MachineOperand::CreateReg(0, /*isDef=*/false)); 10720 ccOutIdx = 1; 10721 } else 10722 ccOutIdx = MCID->getNumOperands() - 1; 10723 } else 10724 ccOutIdx = MCID->getNumOperands() - 1; 10725 10726 // Any ARM instruction that sets the 's' bit should specify an optional 10727 // "cc_out" operand in the last operand position. 10728 if (!MI.hasOptionalDef() || !MCID->OpInfo[ccOutIdx].isOptionalDef()) { 10729 assert(!NewOpc && "Optional cc_out operand required"); 10730 return; 10731 } 10732 // Look for an implicit def of CPSR added by MachineInstr ctor. Remove it 10733 // since we already have an optional CPSR def. 10734 bool definesCPSR = false; 10735 bool deadCPSR = false; 10736 for (unsigned i = MCID->getNumOperands(), e = MI.getNumOperands(); i != e; 10737 ++i) { 10738 const MachineOperand &MO = MI.getOperand(i); 10739 if (MO.isReg() && MO.isDef() && MO.getReg() == ARM::CPSR) { 10740 definesCPSR = true; 10741 if (MO.isDead()) 10742 deadCPSR = true; 10743 MI.RemoveOperand(i); 10744 break; 10745 } 10746 } 10747 if (!definesCPSR) { 10748 assert(!NewOpc && "Optional cc_out operand required"); 10749 return; 10750 } 10751 assert(deadCPSR == !Node->hasAnyUseOfValue(1) && "inconsistent dead flag"); 10752 if (deadCPSR) { 10753 assert(!MI.getOperand(ccOutIdx).getReg() && 10754 "expect uninitialized optional cc_out operand"); 10755 // Thumb1 instructions must have the S bit even if the CPSR is dead. 10756 if (!Subtarget->isThumb1Only()) 10757 return; 10758 } 10759 10760 // If this instruction was defined with an optional CPSR def and its dag node 10761 // had a live implicit CPSR def, then activate the optional CPSR def. 10762 MachineOperand &MO = MI.getOperand(ccOutIdx); 10763 MO.setReg(ARM::CPSR); 10764 MO.setIsDef(true); 10765 } 10766 10767 //===----------------------------------------------------------------------===// 10768 // ARM Optimization Hooks 10769 //===----------------------------------------------------------------------===// 10770 10771 // Helper function that checks if N is a null or all ones constant. 10772 static inline bool isZeroOrAllOnes(SDValue N, bool AllOnes) { 10773 return AllOnes ? isAllOnesConstant(N) : isNullConstant(N); 10774 } 10775 10776 // Return true if N is conditionally 0 or all ones. 10777 // Detects these expressions where cc is an i1 value: 10778 // 10779 // (select cc 0, y) [AllOnes=0] 10780 // (select cc y, 0) [AllOnes=0] 10781 // (zext cc) [AllOnes=0] 10782 // (sext cc) [AllOnes=0/1] 10783 // (select cc -1, y) [AllOnes=1] 10784 // (select cc y, -1) [AllOnes=1] 10785 // 10786 // Invert is set when N is the null/all ones constant when CC is false. 10787 // OtherOp is set to the alternative value of N. 10788 static bool isConditionalZeroOrAllOnes(SDNode *N, bool AllOnes, 10789 SDValue &CC, bool &Invert, 10790 SDValue &OtherOp, 10791 SelectionDAG &DAG) { 10792 switch (N->getOpcode()) { 10793 default: return false; 10794 case ISD::SELECT: { 10795 CC = N->getOperand(0); 10796 SDValue N1 = N->getOperand(1); 10797 SDValue N2 = N->getOperand(2); 10798 if (isZeroOrAllOnes(N1, AllOnes)) { 10799 Invert = false; 10800 OtherOp = N2; 10801 return true; 10802 } 10803 if (isZeroOrAllOnes(N2, AllOnes)) { 10804 Invert = true; 10805 OtherOp = N1; 10806 return true; 10807 } 10808 return false; 10809 } 10810 case ISD::ZERO_EXTEND: 10811 // (zext cc) can never be the all ones value. 10812 if (AllOnes) 10813 return false; 10814 LLVM_FALLTHROUGH; 10815 case ISD::SIGN_EXTEND: { 10816 SDLoc dl(N); 10817 EVT VT = N->getValueType(0); 10818 CC = N->getOperand(0); 10819 if (CC.getValueType() != MVT::i1 || CC.getOpcode() != ISD::SETCC) 10820 return false; 10821 Invert = !AllOnes; 10822 if (AllOnes) 10823 // When looking for an AllOnes constant, N is an sext, and the 'other' 10824 // value is 0. 10825 OtherOp = DAG.getConstant(0, dl, VT); 10826 else if (N->getOpcode() == ISD::ZERO_EXTEND) 10827 // When looking for a 0 constant, N can be zext or sext. 10828 OtherOp = DAG.getConstant(1, dl, VT); 10829 else 10830 OtherOp = DAG.getConstant(APInt::getAllOnesValue(VT.getSizeInBits()), dl, 10831 VT); 10832 return true; 10833 } 10834 } 10835 } 10836 10837 // Combine a constant select operand into its use: 10838 // 10839 // (add (select cc, 0, c), x) -> (select cc, x, (add, x, c)) 10840 // (sub x, (select cc, 0, c)) -> (select cc, x, (sub, x, c)) 10841 // (and (select cc, -1, c), x) -> (select cc, x, (and, x, c)) [AllOnes=1] 10842 // (or (select cc, 0, c), x) -> (select cc, x, (or, x, c)) 10843 // (xor (select cc, 0, c), x) -> (select cc, x, (xor, x, c)) 10844 // 10845 // The transform is rejected if the select doesn't have a constant operand that 10846 // is null, or all ones when AllOnes is set. 10847 // 10848 // Also recognize sext/zext from i1: 10849 // 10850 // (add (zext cc), x) -> (select cc (add x, 1), x) 10851 // (add (sext cc), x) -> (select cc (add x, -1), x) 10852 // 10853 // These transformations eventually create predicated instructions. 10854 // 10855 // @param N The node to transform. 10856 // @param Slct The N operand that is a select. 10857 // @param OtherOp The other N operand (x above). 10858 // @param DCI Context. 10859 // @param AllOnes Require the select constant to be all ones instead of null. 10860 // @returns The new node, or SDValue() on failure. 10861 static 10862 SDValue combineSelectAndUse(SDNode *N, SDValue Slct, SDValue OtherOp, 10863 TargetLowering::DAGCombinerInfo &DCI, 10864 bool AllOnes = false) { 10865 SelectionDAG &DAG = DCI.DAG; 10866 EVT VT = N->getValueType(0); 10867 SDValue NonConstantVal; 10868 SDValue CCOp; 10869 bool SwapSelectOps; 10870 if (!isConditionalZeroOrAllOnes(Slct.getNode(), AllOnes, CCOp, SwapSelectOps, 10871 NonConstantVal, DAG)) 10872 return SDValue(); 10873 10874 // Slct is now know to be the desired identity constant when CC is true. 10875 SDValue TrueVal = OtherOp; 10876 SDValue FalseVal = DAG.getNode(N->getOpcode(), SDLoc(N), VT, 10877 OtherOp, NonConstantVal); 10878 // Unless SwapSelectOps says CC should be false. 10879 if (SwapSelectOps) 10880 std::swap(TrueVal, FalseVal); 10881 10882 return DAG.getNode(ISD::SELECT, SDLoc(N), VT, 10883 CCOp, TrueVal, FalseVal); 10884 } 10885 10886 // Attempt combineSelectAndUse on each operand of a commutative operator N. 10887 static 10888 SDValue combineSelectAndUseCommutative(SDNode *N, bool AllOnes, 10889 TargetLowering::DAGCombinerInfo &DCI) { 10890 SDValue N0 = N->getOperand(0); 10891 SDValue N1 = N->getOperand(1); 10892 if (N0.getNode()->hasOneUse()) 10893 if (SDValue Result = combineSelectAndUse(N, N0, N1, DCI, AllOnes)) 10894 return Result; 10895 if (N1.getNode()->hasOneUse()) 10896 if (SDValue Result = combineSelectAndUse(N, N1, N0, DCI, AllOnes)) 10897 return Result; 10898 return SDValue(); 10899 } 10900 10901 static bool IsVUZPShuffleNode(SDNode *N) { 10902 // VUZP shuffle node. 10903 if (N->getOpcode() == ARMISD::VUZP) 10904 return true; 10905 10906 // "VUZP" on i32 is an alias for VTRN. 10907 if (N->getOpcode() == ARMISD::VTRN && N->getValueType(0) == MVT::v2i32) 10908 return true; 10909 10910 return false; 10911 } 10912 10913 static SDValue AddCombineToVPADD(SDNode *N, SDValue N0, SDValue N1, 10914 TargetLowering::DAGCombinerInfo &DCI, 10915 const ARMSubtarget *Subtarget) { 10916 // Look for ADD(VUZP.0, VUZP.1). 10917 if (!IsVUZPShuffleNode(N0.getNode()) || N0.getNode() != N1.getNode() || 10918 N0 == N1) 10919 return SDValue(); 10920 10921 // Make sure the ADD is a 64-bit add; there is no 128-bit VPADD. 10922 if (!N->getValueType(0).is64BitVector()) 10923 return SDValue(); 10924 10925 // Generate vpadd. 10926 SelectionDAG &DAG = DCI.DAG; 10927 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 10928 SDLoc dl(N); 10929 SDNode *Unzip = N0.getNode(); 10930 EVT VT = N->getValueType(0); 10931 10932 SmallVector<SDValue, 8> Ops; 10933 Ops.push_back(DAG.getConstant(Intrinsic::arm_neon_vpadd, dl, 10934 TLI.getPointerTy(DAG.getDataLayout()))); 10935 Ops.push_back(Unzip->getOperand(0)); 10936 Ops.push_back(Unzip->getOperand(1)); 10937 10938 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, VT, Ops); 10939 } 10940 10941 static SDValue AddCombineVUZPToVPADDL(SDNode *N, SDValue N0, SDValue N1, 10942 TargetLowering::DAGCombinerInfo &DCI, 10943 const ARMSubtarget *Subtarget) { 10944 // Check for two extended operands. 10945 if (!(N0.getOpcode() == ISD::SIGN_EXTEND && 10946 N1.getOpcode() == ISD::SIGN_EXTEND) && 10947 !(N0.getOpcode() == ISD::ZERO_EXTEND && 10948 N1.getOpcode() == ISD::ZERO_EXTEND)) 10949 return SDValue(); 10950 10951 SDValue N00 = N0.getOperand(0); 10952 SDValue N10 = N1.getOperand(0); 10953 10954 // Look for ADD(SEXT(VUZP.0), SEXT(VUZP.1)) 10955 if (!IsVUZPShuffleNode(N00.getNode()) || N00.getNode() != N10.getNode() || 10956 N00 == N10) 10957 return SDValue(); 10958 10959 // We only recognize Q register paddl here; this can't be reached until 10960 // after type legalization. 10961 if (!N00.getValueType().is64BitVector() || 10962 !N0.getValueType().is128BitVector()) 10963 return SDValue(); 10964 10965 // Generate vpaddl. 10966 SelectionDAG &DAG = DCI.DAG; 10967 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 10968 SDLoc dl(N); 10969 EVT VT = N->getValueType(0); 10970 10971 SmallVector<SDValue, 8> Ops; 10972 // Form vpaddl.sN or vpaddl.uN depending on the kind of extension. 10973 unsigned Opcode; 10974 if (N0.getOpcode() == ISD::SIGN_EXTEND) 10975 Opcode = Intrinsic::arm_neon_vpaddls; 10976 else 10977 Opcode = Intrinsic::arm_neon_vpaddlu; 10978 Ops.push_back(DAG.getConstant(Opcode, dl, 10979 TLI.getPointerTy(DAG.getDataLayout()))); 10980 EVT ElemTy = N00.getValueType().getVectorElementType(); 10981 unsigned NumElts = VT.getVectorNumElements(); 10982 EVT ConcatVT = EVT::getVectorVT(*DAG.getContext(), ElemTy, NumElts * 2); 10983 SDValue Concat = DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), ConcatVT, 10984 N00.getOperand(0), N00.getOperand(1)); 10985 Ops.push_back(Concat); 10986 10987 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, VT, Ops); 10988 } 10989 10990 // FIXME: This function shouldn't be necessary; if we lower BUILD_VECTOR in 10991 // an appropriate manner, we end up with ADD(VUZP(ZEXT(N))), which is 10992 // much easier to match. 10993 static SDValue 10994 AddCombineBUILD_VECTORToVPADDL(SDNode *N, SDValue N0, SDValue N1, 10995 TargetLowering::DAGCombinerInfo &DCI, 10996 const ARMSubtarget *Subtarget) { 10997 // Only perform optimization if after legalize, and if NEON is available. We 10998 // also expected both operands to be BUILD_VECTORs. 10999 if (DCI.isBeforeLegalize() || !Subtarget->hasNEON() 11000 || N0.getOpcode() != ISD::BUILD_VECTOR 11001 || N1.getOpcode() != ISD::BUILD_VECTOR) 11002 return SDValue(); 11003 11004 // Check output type since VPADDL operand elements can only be 8, 16, or 32. 11005 EVT VT = N->getValueType(0); 11006 if (!VT.isInteger() || VT.getVectorElementType() == MVT::i64) 11007 return SDValue(); 11008 11009 // Check that the vector operands are of the right form. 11010 // N0 and N1 are BUILD_VECTOR nodes with N number of EXTRACT_VECTOR 11011 // operands, where N is the size of the formed vector. 11012 // Each EXTRACT_VECTOR should have the same input vector and odd or even 11013 // index such that we have a pair wise add pattern. 11014 11015 // Grab the vector that all EXTRACT_VECTOR nodes should be referencing. 11016 if (N0->getOperand(0)->getOpcode() != ISD::EXTRACT_VECTOR_ELT) 11017 return SDValue(); 11018 SDValue Vec = N0->getOperand(0)->getOperand(0); 11019 SDNode *V = Vec.getNode(); 11020 unsigned nextIndex = 0; 11021 11022 // For each operands to the ADD which are BUILD_VECTORs, 11023 // check to see if each of their operands are an EXTRACT_VECTOR with 11024 // the same vector and appropriate index. 11025 for (unsigned i = 0, e = N0->getNumOperands(); i != e; ++i) { 11026 if (N0->getOperand(i)->getOpcode() == ISD::EXTRACT_VECTOR_ELT 11027 && N1->getOperand(i)->getOpcode() == ISD::EXTRACT_VECTOR_ELT) { 11028 11029 SDValue ExtVec0 = N0->getOperand(i); 11030 SDValue ExtVec1 = N1->getOperand(i); 11031 11032 // First operand is the vector, verify its the same. 11033 if (V != ExtVec0->getOperand(0).getNode() || 11034 V != ExtVec1->getOperand(0).getNode()) 11035 return SDValue(); 11036 11037 // Second is the constant, verify its correct. 11038 ConstantSDNode *C0 = dyn_cast<ConstantSDNode>(ExtVec0->getOperand(1)); 11039 ConstantSDNode *C1 = dyn_cast<ConstantSDNode>(ExtVec1->getOperand(1)); 11040 11041 // For the constant, we want to see all the even or all the odd. 11042 if (!C0 || !C1 || C0->getZExtValue() != nextIndex 11043 || C1->getZExtValue() != nextIndex+1) 11044 return SDValue(); 11045 11046 // Increment index. 11047 nextIndex+=2; 11048 } else 11049 return SDValue(); 11050 } 11051 11052 // Don't generate vpaddl+vmovn; we'll match it to vpadd later. Also make sure 11053 // we're using the entire input vector, otherwise there's a size/legality 11054 // mismatch somewhere. 11055 if (nextIndex != Vec.getValueType().getVectorNumElements() || 11056 Vec.getValueType().getVectorElementType() == VT.getVectorElementType()) 11057 return SDValue(); 11058 11059 // Create VPADDL node. 11060 SelectionDAG &DAG = DCI.DAG; 11061 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 11062 11063 SDLoc dl(N); 11064 11065 // Build operand list. 11066 SmallVector<SDValue, 8> Ops; 11067 Ops.push_back(DAG.getConstant(Intrinsic::arm_neon_vpaddls, dl, 11068 TLI.getPointerTy(DAG.getDataLayout()))); 11069 11070 // Input is the vector. 11071 Ops.push_back(Vec); 11072 11073 // Get widened type and narrowed type. 11074 MVT widenType; 11075 unsigned numElem = VT.getVectorNumElements(); 11076 11077 EVT inputLaneType = Vec.getValueType().getVectorElementType(); 11078 switch (inputLaneType.getSimpleVT().SimpleTy) { 11079 case MVT::i8: widenType = MVT::getVectorVT(MVT::i16, numElem); break; 11080 case MVT::i16: widenType = MVT::getVectorVT(MVT::i32, numElem); break; 11081 case MVT::i32: widenType = MVT::getVectorVT(MVT::i64, numElem); break; 11082 default: 11083 llvm_unreachable("Invalid vector element type for padd optimization."); 11084 } 11085 11086 SDValue tmp = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, widenType, Ops); 11087 unsigned ExtOp = VT.bitsGT(tmp.getValueType()) ? ISD::ANY_EXTEND : ISD::TRUNCATE; 11088 return DAG.getNode(ExtOp, dl, VT, tmp); 11089 } 11090 11091 static SDValue findMUL_LOHI(SDValue V) { 11092 if (V->getOpcode() == ISD::UMUL_LOHI || 11093 V->getOpcode() == ISD::SMUL_LOHI) 11094 return V; 11095 return SDValue(); 11096 } 11097 11098 static SDValue AddCombineTo64BitSMLAL16(SDNode *AddcNode, SDNode *AddeNode, 11099 TargetLowering::DAGCombinerInfo &DCI, 11100 const ARMSubtarget *Subtarget) { 11101 if (!Subtarget->hasBaseDSP()) 11102 return SDValue(); 11103 11104 // SMLALBB, SMLALBT, SMLALTB, SMLALTT multiply two 16-bit values and 11105 // accumulates the product into a 64-bit value. The 16-bit values will 11106 // be sign extended somehow or SRA'd into 32-bit values 11107 // (addc (adde (mul 16bit, 16bit), lo), hi) 11108 SDValue Mul = AddcNode->getOperand(0); 11109 SDValue Lo = AddcNode->getOperand(1); 11110 if (Mul.getOpcode() != ISD::MUL) { 11111 Lo = AddcNode->getOperand(0); 11112 Mul = AddcNode->getOperand(1); 11113 if (Mul.getOpcode() != ISD::MUL) 11114 return SDValue(); 11115 } 11116 11117 SDValue SRA = AddeNode->getOperand(0); 11118 SDValue Hi = AddeNode->getOperand(1); 11119 if (SRA.getOpcode() != ISD::SRA) { 11120 SRA = AddeNode->getOperand(1); 11121 Hi = AddeNode->getOperand(0); 11122 if (SRA.getOpcode() != ISD::SRA) 11123 return SDValue(); 11124 } 11125 if (auto Const = dyn_cast<ConstantSDNode>(SRA.getOperand(1))) { 11126 if (Const->getZExtValue() != 31) 11127 return SDValue(); 11128 } else 11129 return SDValue(); 11130 11131 if (SRA.getOperand(0) != Mul) 11132 return SDValue(); 11133 11134 SelectionDAG &DAG = DCI.DAG; 11135 SDLoc dl(AddcNode); 11136 unsigned Opcode = 0; 11137 SDValue Op0; 11138 SDValue Op1; 11139 11140 if (isS16(Mul.getOperand(0), DAG) && isS16(Mul.getOperand(1), DAG)) { 11141 Opcode = ARMISD::SMLALBB; 11142 Op0 = Mul.getOperand(0); 11143 Op1 = Mul.getOperand(1); 11144 } else if (isS16(Mul.getOperand(0), DAG) && isSRA16(Mul.getOperand(1))) { 11145 Opcode = ARMISD::SMLALBT; 11146 Op0 = Mul.getOperand(0); 11147 Op1 = Mul.getOperand(1).getOperand(0); 11148 } else if (isSRA16(Mul.getOperand(0)) && isS16(Mul.getOperand(1), DAG)) { 11149 Opcode = ARMISD::SMLALTB; 11150 Op0 = Mul.getOperand(0).getOperand(0); 11151 Op1 = Mul.getOperand(1); 11152 } else if (isSRA16(Mul.getOperand(0)) && isSRA16(Mul.getOperand(1))) { 11153 Opcode = ARMISD::SMLALTT; 11154 Op0 = Mul->getOperand(0).getOperand(0); 11155 Op1 = Mul->getOperand(1).getOperand(0); 11156 } 11157 11158 if (!Op0 || !Op1) 11159 return SDValue(); 11160 11161 SDValue SMLAL = DAG.getNode(Opcode, dl, DAG.getVTList(MVT::i32, MVT::i32), 11162 Op0, Op1, Lo, Hi); 11163 // Replace the ADDs' nodes uses by the MLA node's values. 11164 SDValue HiMLALResult(SMLAL.getNode(), 1); 11165 SDValue LoMLALResult(SMLAL.getNode(), 0); 11166 11167 DAG.ReplaceAllUsesOfValueWith(SDValue(AddcNode, 0), LoMLALResult); 11168 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeNode, 0), HiMLALResult); 11169 11170 // Return original node to notify the driver to stop replacing. 11171 SDValue resNode(AddcNode, 0); 11172 return resNode; 11173 } 11174 11175 static SDValue AddCombineTo64bitMLAL(SDNode *AddeSubeNode, 11176 TargetLowering::DAGCombinerInfo &DCI, 11177 const ARMSubtarget *Subtarget) { 11178 // Look for multiply add opportunities. 11179 // The pattern is a ISD::UMUL_LOHI followed by two add nodes, where 11180 // each add nodes consumes a value from ISD::UMUL_LOHI and there is 11181 // a glue link from the first add to the second add. 11182 // If we find this pattern, we can replace the U/SMUL_LOHI, ADDC, and ADDE by 11183 // a S/UMLAL instruction. 11184 // UMUL_LOHI 11185 // / :lo \ :hi 11186 // V \ [no multiline comment] 11187 // loAdd -> ADDC | 11188 // \ :carry / 11189 // V V 11190 // ADDE <- hiAdd 11191 // 11192 // In the special case where only the higher part of a signed result is used 11193 // and the add to the low part of the result of ISD::UMUL_LOHI adds or subtracts 11194 // a constant with the exact value of 0x80000000, we recognize we are dealing 11195 // with a "rounded multiply and add" (or subtract) and transform it into 11196 // either a ARMISD::SMMLAR or ARMISD::SMMLSR respectively. 11197 11198 assert((AddeSubeNode->getOpcode() == ARMISD::ADDE || 11199 AddeSubeNode->getOpcode() == ARMISD::SUBE) && 11200 "Expect an ADDE or SUBE"); 11201 11202 assert(AddeSubeNode->getNumOperands() == 3 && 11203 AddeSubeNode->getOperand(2).getValueType() == MVT::i32 && 11204 "ADDE node has the wrong inputs"); 11205 11206 // Check that we are chained to the right ADDC or SUBC node. 11207 SDNode *AddcSubcNode = AddeSubeNode->getOperand(2).getNode(); 11208 if ((AddeSubeNode->getOpcode() == ARMISD::ADDE && 11209 AddcSubcNode->getOpcode() != ARMISD::ADDC) || 11210 (AddeSubeNode->getOpcode() == ARMISD::SUBE && 11211 AddcSubcNode->getOpcode() != ARMISD::SUBC)) 11212 return SDValue(); 11213 11214 SDValue AddcSubcOp0 = AddcSubcNode->getOperand(0); 11215 SDValue AddcSubcOp1 = AddcSubcNode->getOperand(1); 11216 11217 // Check if the two operands are from the same mul_lohi node. 11218 if (AddcSubcOp0.getNode() == AddcSubcOp1.getNode()) 11219 return SDValue(); 11220 11221 assert(AddcSubcNode->getNumValues() == 2 && 11222 AddcSubcNode->getValueType(0) == MVT::i32 && 11223 "Expect ADDC with two result values. First: i32"); 11224 11225 // Check that the ADDC adds the low result of the S/UMUL_LOHI. If not, it 11226 // maybe a SMLAL which multiplies two 16-bit values. 11227 if (AddeSubeNode->getOpcode() == ARMISD::ADDE && 11228 AddcSubcOp0->getOpcode() != ISD::UMUL_LOHI && 11229 AddcSubcOp0->getOpcode() != ISD::SMUL_LOHI && 11230 AddcSubcOp1->getOpcode() != ISD::UMUL_LOHI && 11231 AddcSubcOp1->getOpcode() != ISD::SMUL_LOHI) 11232 return AddCombineTo64BitSMLAL16(AddcSubcNode, AddeSubeNode, DCI, Subtarget); 11233 11234 // Check for the triangle shape. 11235 SDValue AddeSubeOp0 = AddeSubeNode->getOperand(0); 11236 SDValue AddeSubeOp1 = AddeSubeNode->getOperand(1); 11237 11238 // Make sure that the ADDE/SUBE operands are not coming from the same node. 11239 if (AddeSubeOp0.getNode() == AddeSubeOp1.getNode()) 11240 return SDValue(); 11241 11242 // Find the MUL_LOHI node walking up ADDE/SUBE's operands. 11243 bool IsLeftOperandMUL = false; 11244 SDValue MULOp = findMUL_LOHI(AddeSubeOp0); 11245 if (MULOp == SDValue()) 11246 MULOp = findMUL_LOHI(AddeSubeOp1); 11247 else 11248 IsLeftOperandMUL = true; 11249 if (MULOp == SDValue()) 11250 return SDValue(); 11251 11252 // Figure out the right opcode. 11253 unsigned Opc = MULOp->getOpcode(); 11254 unsigned FinalOpc = (Opc == ISD::SMUL_LOHI) ? ARMISD::SMLAL : ARMISD::UMLAL; 11255 11256 // Figure out the high and low input values to the MLAL node. 11257 SDValue *HiAddSub = nullptr; 11258 SDValue *LoMul = nullptr; 11259 SDValue *LowAddSub = nullptr; 11260 11261 // Ensure that ADDE/SUBE is from high result of ISD::xMUL_LOHI. 11262 if ((AddeSubeOp0 != MULOp.getValue(1)) && (AddeSubeOp1 != MULOp.getValue(1))) 11263 return SDValue(); 11264 11265 if (IsLeftOperandMUL) 11266 HiAddSub = &AddeSubeOp1; 11267 else 11268 HiAddSub = &AddeSubeOp0; 11269 11270 // Ensure that LoMul and LowAddSub are taken from correct ISD::SMUL_LOHI node 11271 // whose low result is fed to the ADDC/SUBC we are checking. 11272 11273 if (AddcSubcOp0 == MULOp.getValue(0)) { 11274 LoMul = &AddcSubcOp0; 11275 LowAddSub = &AddcSubcOp1; 11276 } 11277 if (AddcSubcOp1 == MULOp.getValue(0)) { 11278 LoMul = &AddcSubcOp1; 11279 LowAddSub = &AddcSubcOp0; 11280 } 11281 11282 if (!LoMul) 11283 return SDValue(); 11284 11285 // If HiAddSub is the same node as ADDC/SUBC or is a predecessor of ADDC/SUBC 11286 // the replacement below will create a cycle. 11287 if (AddcSubcNode == HiAddSub->getNode() || 11288 AddcSubcNode->isPredecessorOf(HiAddSub->getNode())) 11289 return SDValue(); 11290 11291 // Create the merged node. 11292 SelectionDAG &DAG = DCI.DAG; 11293 11294 // Start building operand list. 11295 SmallVector<SDValue, 8> Ops; 11296 Ops.push_back(LoMul->getOperand(0)); 11297 Ops.push_back(LoMul->getOperand(1)); 11298 11299 // Check whether we can use SMMLAR, SMMLSR or SMMULR instead. For this to be 11300 // the case, we must be doing signed multiplication and only use the higher 11301 // part of the result of the MLAL, furthermore the LowAddSub must be a constant 11302 // addition or subtraction with the value of 0x800000. 11303 if (Subtarget->hasV6Ops() && Subtarget->hasDSP() && Subtarget->useMulOps() && 11304 FinalOpc == ARMISD::SMLAL && !AddeSubeNode->hasAnyUseOfValue(1) && 11305 LowAddSub->getNode()->getOpcode() == ISD::Constant && 11306 static_cast<ConstantSDNode *>(LowAddSub->getNode())->getZExtValue() == 11307 0x80000000) { 11308 Ops.push_back(*HiAddSub); 11309 if (AddcSubcNode->getOpcode() == ARMISD::SUBC) { 11310 FinalOpc = ARMISD::SMMLSR; 11311 } else { 11312 FinalOpc = ARMISD::SMMLAR; 11313 } 11314 SDValue NewNode = DAG.getNode(FinalOpc, SDLoc(AddcSubcNode), MVT::i32, Ops); 11315 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeSubeNode, 0), NewNode); 11316 11317 return SDValue(AddeSubeNode, 0); 11318 } else if (AddcSubcNode->getOpcode() == ARMISD::SUBC) 11319 // SMMLS is generated during instruction selection and the rest of this 11320 // function can not handle the case where AddcSubcNode is a SUBC. 11321 return SDValue(); 11322 11323 // Finish building the operand list for {U/S}MLAL 11324 Ops.push_back(*LowAddSub); 11325 Ops.push_back(*HiAddSub); 11326 11327 SDValue MLALNode = DAG.getNode(FinalOpc, SDLoc(AddcSubcNode), 11328 DAG.getVTList(MVT::i32, MVT::i32), Ops); 11329 11330 // Replace the ADDs' nodes uses by the MLA node's values. 11331 SDValue HiMLALResult(MLALNode.getNode(), 1); 11332 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeSubeNode, 0), HiMLALResult); 11333 11334 SDValue LoMLALResult(MLALNode.getNode(), 0); 11335 DAG.ReplaceAllUsesOfValueWith(SDValue(AddcSubcNode, 0), LoMLALResult); 11336 11337 // Return original node to notify the driver to stop replacing. 11338 return SDValue(AddeSubeNode, 0); 11339 } 11340 11341 static SDValue AddCombineTo64bitUMAAL(SDNode *AddeNode, 11342 TargetLowering::DAGCombinerInfo &DCI, 11343 const ARMSubtarget *Subtarget) { 11344 // UMAAL is similar to UMLAL except that it adds two unsigned values. 11345 // While trying to combine for the other MLAL nodes, first search for the 11346 // chance to use UMAAL. Check if Addc uses a node which has already 11347 // been combined into a UMLAL. The other pattern is UMLAL using Addc/Adde 11348 // as the addend, and it's handled in PerformUMLALCombine. 11349 11350 if (!Subtarget->hasV6Ops() || !Subtarget->hasDSP()) 11351 return AddCombineTo64bitMLAL(AddeNode, DCI, Subtarget); 11352 11353 // Check that we have a glued ADDC node. 11354 SDNode* AddcNode = AddeNode->getOperand(2).getNode(); 11355 if (AddcNode->getOpcode() != ARMISD::ADDC) 11356 return SDValue(); 11357 11358 // Find the converted UMAAL or quit if it doesn't exist. 11359 SDNode *UmlalNode = nullptr; 11360 SDValue AddHi; 11361 if (AddcNode->getOperand(0).getOpcode() == ARMISD::UMLAL) { 11362 UmlalNode = AddcNode->getOperand(0).getNode(); 11363 AddHi = AddcNode->getOperand(1); 11364 } else if (AddcNode->getOperand(1).getOpcode() == ARMISD::UMLAL) { 11365 UmlalNode = AddcNode->getOperand(1).getNode(); 11366 AddHi = AddcNode->getOperand(0); 11367 } else { 11368 return AddCombineTo64bitMLAL(AddeNode, DCI, Subtarget); 11369 } 11370 11371 // The ADDC should be glued to an ADDE node, which uses the same UMLAL as 11372 // the ADDC as well as Zero. 11373 if (!isNullConstant(UmlalNode->getOperand(3))) 11374 return SDValue(); 11375 11376 if ((isNullConstant(AddeNode->getOperand(0)) && 11377 AddeNode->getOperand(1).getNode() == UmlalNode) || 11378 (AddeNode->getOperand(0).getNode() == UmlalNode && 11379 isNullConstant(AddeNode->getOperand(1)))) { 11380 SelectionDAG &DAG = DCI.DAG; 11381 SDValue Ops[] = { UmlalNode->getOperand(0), UmlalNode->getOperand(1), 11382 UmlalNode->getOperand(2), AddHi }; 11383 SDValue UMAAL = DAG.getNode(ARMISD::UMAAL, SDLoc(AddcNode), 11384 DAG.getVTList(MVT::i32, MVT::i32), Ops); 11385 11386 // Replace the ADDs' nodes uses by the UMAAL node's values. 11387 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeNode, 0), SDValue(UMAAL.getNode(), 1)); 11388 DAG.ReplaceAllUsesOfValueWith(SDValue(AddcNode, 0), SDValue(UMAAL.getNode(), 0)); 11389 11390 // Return original node to notify the driver to stop replacing. 11391 return SDValue(AddeNode, 0); 11392 } 11393 return SDValue(); 11394 } 11395 11396 static SDValue PerformUMLALCombine(SDNode *N, SelectionDAG &DAG, 11397 const ARMSubtarget *Subtarget) { 11398 if (!Subtarget->hasV6Ops() || !Subtarget->hasDSP()) 11399 return SDValue(); 11400 11401 // Check that we have a pair of ADDC and ADDE as operands. 11402 // Both addends of the ADDE must be zero. 11403 SDNode* AddcNode = N->getOperand(2).getNode(); 11404 SDNode* AddeNode = N->getOperand(3).getNode(); 11405 if ((AddcNode->getOpcode() == ARMISD::ADDC) && 11406 (AddeNode->getOpcode() == ARMISD::ADDE) && 11407 isNullConstant(AddeNode->getOperand(0)) && 11408 isNullConstant(AddeNode->getOperand(1)) && 11409 (AddeNode->getOperand(2).getNode() == AddcNode)) 11410 return DAG.getNode(ARMISD::UMAAL, SDLoc(N), 11411 DAG.getVTList(MVT::i32, MVT::i32), 11412 {N->getOperand(0), N->getOperand(1), 11413 AddcNode->getOperand(0), AddcNode->getOperand(1)}); 11414 else 11415 return SDValue(); 11416 } 11417 11418 static SDValue PerformAddcSubcCombine(SDNode *N, 11419 TargetLowering::DAGCombinerInfo &DCI, 11420 const ARMSubtarget *Subtarget) { 11421 SelectionDAG &DAG(DCI.DAG); 11422 11423 if (N->getOpcode() == ARMISD::SUBC) { 11424 // (SUBC (ADDE 0, 0, C), 1) -> C 11425 SDValue LHS = N->getOperand(0); 11426 SDValue RHS = N->getOperand(1); 11427 if (LHS->getOpcode() == ARMISD::ADDE && 11428 isNullConstant(LHS->getOperand(0)) && 11429 isNullConstant(LHS->getOperand(1)) && isOneConstant(RHS)) { 11430 return DCI.CombineTo(N, SDValue(N, 0), LHS->getOperand(2)); 11431 } 11432 } 11433 11434 if (Subtarget->isThumb1Only()) { 11435 SDValue RHS = N->getOperand(1); 11436 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(RHS)) { 11437 int32_t imm = C->getSExtValue(); 11438 if (imm < 0 && imm > std::numeric_limits<int>::min()) { 11439 SDLoc DL(N); 11440 RHS = DAG.getConstant(-imm, DL, MVT::i32); 11441 unsigned Opcode = (N->getOpcode() == ARMISD::ADDC) ? ARMISD::SUBC 11442 : ARMISD::ADDC; 11443 return DAG.getNode(Opcode, DL, N->getVTList(), N->getOperand(0), RHS); 11444 } 11445 } 11446 } 11447 11448 return SDValue(); 11449 } 11450 11451 static SDValue PerformAddeSubeCombine(SDNode *N, 11452 TargetLowering::DAGCombinerInfo &DCI, 11453 const ARMSubtarget *Subtarget) { 11454 if (Subtarget->isThumb1Only()) { 11455 SelectionDAG &DAG = DCI.DAG; 11456 SDValue RHS = N->getOperand(1); 11457 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(RHS)) { 11458 int64_t imm = C->getSExtValue(); 11459 if (imm < 0) { 11460 SDLoc DL(N); 11461 11462 // The with-carry-in form matches bitwise not instead of the negation. 11463 // Effectively, the inverse interpretation of the carry flag already 11464 // accounts for part of the negation. 11465 RHS = DAG.getConstant(~imm, DL, MVT::i32); 11466 11467 unsigned Opcode = (N->getOpcode() == ARMISD::ADDE) ? ARMISD::SUBE 11468 : ARMISD::ADDE; 11469 return DAG.getNode(Opcode, DL, N->getVTList(), 11470 N->getOperand(0), RHS, N->getOperand(2)); 11471 } 11472 } 11473 } else if (N->getOperand(1)->getOpcode() == ISD::SMUL_LOHI) { 11474 return AddCombineTo64bitMLAL(N, DCI, Subtarget); 11475 } 11476 return SDValue(); 11477 } 11478 11479 static SDValue PerformABSCombine(SDNode *N, 11480 TargetLowering::DAGCombinerInfo &DCI, 11481 const ARMSubtarget *Subtarget) { 11482 SDValue res; 11483 SelectionDAG &DAG = DCI.DAG; 11484 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 11485 11486 if (TLI.isOperationLegal(N->getOpcode(), N->getValueType(0))) 11487 return SDValue(); 11488 11489 if (!TLI.expandABS(N, res, DAG)) 11490 return SDValue(); 11491 11492 return res; 11493 } 11494 11495 /// PerformADDECombine - Target-specific dag combine transform from 11496 /// ARMISD::ADDC, ARMISD::ADDE, and ISD::MUL_LOHI to MLAL or 11497 /// ARMISD::ADDC, ARMISD::ADDE and ARMISD::UMLAL to ARMISD::UMAAL 11498 static SDValue PerformADDECombine(SDNode *N, 11499 TargetLowering::DAGCombinerInfo &DCI, 11500 const ARMSubtarget *Subtarget) { 11501 // Only ARM and Thumb2 support UMLAL/SMLAL. 11502 if (Subtarget->isThumb1Only()) 11503 return PerformAddeSubeCombine(N, DCI, Subtarget); 11504 11505 // Only perform the checks after legalize when the pattern is available. 11506 if (DCI.isBeforeLegalize()) return SDValue(); 11507 11508 return AddCombineTo64bitUMAAL(N, DCI, Subtarget); 11509 } 11510 11511 /// PerformADDCombineWithOperands - Try DAG combinations for an ADD with 11512 /// operands N0 and N1. This is a helper for PerformADDCombine that is 11513 /// called with the default operands, and if that fails, with commuted 11514 /// operands. 11515 static SDValue PerformADDCombineWithOperands(SDNode *N, SDValue N0, SDValue N1, 11516 TargetLowering::DAGCombinerInfo &DCI, 11517 const ARMSubtarget *Subtarget){ 11518 // Attempt to create vpadd for this add. 11519 if (SDValue Result = AddCombineToVPADD(N, N0, N1, DCI, Subtarget)) 11520 return Result; 11521 11522 // Attempt to create vpaddl for this add. 11523 if (SDValue Result = AddCombineVUZPToVPADDL(N, N0, N1, DCI, Subtarget)) 11524 return Result; 11525 if (SDValue Result = AddCombineBUILD_VECTORToVPADDL(N, N0, N1, DCI, 11526 Subtarget)) 11527 return Result; 11528 11529 // fold (add (select cc, 0, c), x) -> (select cc, x, (add, x, c)) 11530 if (N0.getNode()->hasOneUse()) 11531 if (SDValue Result = combineSelectAndUse(N, N0, N1, DCI)) 11532 return Result; 11533 return SDValue(); 11534 } 11535 11536 bool 11537 ARMTargetLowering::isDesirableToCommuteWithShift(const SDNode *N, 11538 CombineLevel Level) const { 11539 if (Level == BeforeLegalizeTypes) 11540 return true; 11541 11542 if (N->getOpcode() != ISD::SHL) 11543 return true; 11544 11545 if (Subtarget->isThumb1Only()) { 11546 // Avoid making expensive immediates by commuting shifts. (This logic 11547 // only applies to Thumb1 because ARM and Thumb2 immediates can be shifted 11548 // for free.) 11549 if (N->getOpcode() != ISD::SHL) 11550 return true; 11551 SDValue N1 = N->getOperand(0); 11552 if (N1->getOpcode() != ISD::ADD && N1->getOpcode() != ISD::AND && 11553 N1->getOpcode() != ISD::OR && N1->getOpcode() != ISD::XOR) 11554 return true; 11555 if (auto *Const = dyn_cast<ConstantSDNode>(N1->getOperand(1))) { 11556 if (Const->getAPIntValue().ult(256)) 11557 return false; 11558 if (N1->getOpcode() == ISD::ADD && Const->getAPIntValue().slt(0) && 11559 Const->getAPIntValue().sgt(-256)) 11560 return false; 11561 } 11562 return true; 11563 } 11564 11565 // Turn off commute-with-shift transform after legalization, so it doesn't 11566 // conflict with PerformSHLSimplify. (We could try to detect when 11567 // PerformSHLSimplify would trigger more precisely, but it isn't 11568 // really necessary.) 11569 return false; 11570 } 11571 11572 bool ARMTargetLowering::shouldFoldConstantShiftPairToMask( 11573 const SDNode *N, CombineLevel Level) const { 11574 if (!Subtarget->isThumb1Only()) 11575 return true; 11576 11577 if (Level == BeforeLegalizeTypes) 11578 return true; 11579 11580 return false; 11581 } 11582 11583 bool ARMTargetLowering::preferIncOfAddToSubOfNot(EVT VT) const { 11584 if (!Subtarget->hasNEON()) { 11585 if (Subtarget->isThumb1Only()) 11586 return VT.getScalarSizeInBits() <= 32; 11587 return true; 11588 } 11589 return VT.isScalarInteger(); 11590 } 11591 11592 static SDValue PerformSHLSimplify(SDNode *N, 11593 TargetLowering::DAGCombinerInfo &DCI, 11594 const ARMSubtarget *ST) { 11595 // Allow the generic combiner to identify potential bswaps. 11596 if (DCI.isBeforeLegalize()) 11597 return SDValue(); 11598 11599 // DAG combiner will fold: 11600 // (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2) 11601 // (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2 11602 // Other code patterns that can be also be modified have the following form: 11603 // b + ((a << 1) | 510) 11604 // b + ((a << 1) & 510) 11605 // b + ((a << 1) ^ 510) 11606 // b + ((a << 1) + 510) 11607 11608 // Many instructions can perform the shift for free, but it requires both 11609 // the operands to be registers. If c1 << c2 is too large, a mov immediate 11610 // instruction will needed. So, unfold back to the original pattern if: 11611 // - if c1 and c2 are small enough that they don't require mov imms. 11612 // - the user(s) of the node can perform an shl 11613 11614 // No shifted operands for 16-bit instructions. 11615 if (ST->isThumb() && ST->isThumb1Only()) 11616 return SDValue(); 11617 11618 // Check that all the users could perform the shl themselves. 11619 for (auto U : N->uses()) { 11620 switch(U->getOpcode()) { 11621 default: 11622 return SDValue(); 11623 case ISD::SUB: 11624 case ISD::ADD: 11625 case ISD::AND: 11626 case ISD::OR: 11627 case ISD::XOR: 11628 case ISD::SETCC: 11629 case ARMISD::CMP: 11630 // Check that the user isn't already using a constant because there 11631 // aren't any instructions that support an immediate operand and a 11632 // shifted operand. 11633 if (isa<ConstantSDNode>(U->getOperand(0)) || 11634 isa<ConstantSDNode>(U->getOperand(1))) 11635 return SDValue(); 11636 11637 // Check that it's not already using a shift. 11638 if (U->getOperand(0).getOpcode() == ISD::SHL || 11639 U->getOperand(1).getOpcode() == ISD::SHL) 11640 return SDValue(); 11641 break; 11642 } 11643 } 11644 11645 if (N->getOpcode() != ISD::ADD && N->getOpcode() != ISD::OR && 11646 N->getOpcode() != ISD::XOR && N->getOpcode() != ISD::AND) 11647 return SDValue(); 11648 11649 if (N->getOperand(0).getOpcode() != ISD::SHL) 11650 return SDValue(); 11651 11652 SDValue SHL = N->getOperand(0); 11653 11654 auto *C1ShlC2 = dyn_cast<ConstantSDNode>(N->getOperand(1)); 11655 auto *C2 = dyn_cast<ConstantSDNode>(SHL.getOperand(1)); 11656 if (!C1ShlC2 || !C2) 11657 return SDValue(); 11658 11659 APInt C2Int = C2->getAPIntValue(); 11660 APInt C1Int = C1ShlC2->getAPIntValue(); 11661 11662 // Check that performing a lshr will not lose any information. 11663 APInt Mask = APInt::getHighBitsSet(C2Int.getBitWidth(), 11664 C2Int.getBitWidth() - C2->getZExtValue()); 11665 if ((C1Int & Mask) != C1Int) 11666 return SDValue(); 11667 11668 // Shift the first constant. 11669 C1Int.lshrInPlace(C2Int); 11670 11671 // The immediates are encoded as an 8-bit value that can be rotated. 11672 auto LargeImm = [](const APInt &Imm) { 11673 unsigned Zeros = Imm.countLeadingZeros() + Imm.countTrailingZeros(); 11674 return Imm.getBitWidth() - Zeros > 8; 11675 }; 11676 11677 if (LargeImm(C1Int) || LargeImm(C2Int)) 11678 return SDValue(); 11679 11680 SelectionDAG &DAG = DCI.DAG; 11681 SDLoc dl(N); 11682 SDValue X = SHL.getOperand(0); 11683 SDValue BinOp = DAG.getNode(N->getOpcode(), dl, MVT::i32, X, 11684 DAG.getConstant(C1Int, dl, MVT::i32)); 11685 // Shift left to compensate for the lshr of C1Int. 11686 SDValue Res = DAG.getNode(ISD::SHL, dl, MVT::i32, BinOp, SHL.getOperand(1)); 11687 11688 LLVM_DEBUG(dbgs() << "Simplify shl use:\n"; SHL.getOperand(0).dump(); 11689 SHL.dump(); N->dump()); 11690 LLVM_DEBUG(dbgs() << "Into:\n"; X.dump(); BinOp.dump(); Res.dump()); 11691 return Res; 11692 } 11693 11694 11695 /// PerformADDCombine - Target-specific dag combine xforms for ISD::ADD. 11696 /// 11697 static SDValue PerformADDCombine(SDNode *N, 11698 TargetLowering::DAGCombinerInfo &DCI, 11699 const ARMSubtarget *Subtarget) { 11700 SDValue N0 = N->getOperand(0); 11701 SDValue N1 = N->getOperand(1); 11702 11703 // Only works one way, because it needs an immediate operand. 11704 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget)) 11705 return Result; 11706 11707 // First try with the default operand order. 11708 if (SDValue Result = PerformADDCombineWithOperands(N, N0, N1, DCI, Subtarget)) 11709 return Result; 11710 11711 // If that didn't work, try again with the operands commuted. 11712 return PerformADDCombineWithOperands(N, N1, N0, DCI, Subtarget); 11713 } 11714 11715 /// PerformSUBCombine - Target-specific dag combine xforms for ISD::SUB. 11716 /// 11717 static SDValue PerformSUBCombine(SDNode *N, 11718 TargetLowering::DAGCombinerInfo &DCI) { 11719 SDValue N0 = N->getOperand(0); 11720 SDValue N1 = N->getOperand(1); 11721 11722 // fold (sub x, (select cc, 0, c)) -> (select cc, x, (sub, x, c)) 11723 if (N1.getNode()->hasOneUse()) 11724 if (SDValue Result = combineSelectAndUse(N, N1, N0, DCI)) 11725 return Result; 11726 11727 return SDValue(); 11728 } 11729 11730 /// PerformVMULCombine 11731 /// Distribute (A + B) * C to (A * C) + (B * C) to take advantage of the 11732 /// special multiplier accumulator forwarding. 11733 /// vmul d3, d0, d2 11734 /// vmla d3, d1, d2 11735 /// is faster than 11736 /// vadd d3, d0, d1 11737 /// vmul d3, d3, d2 11738 // However, for (A + B) * (A + B), 11739 // vadd d2, d0, d1 11740 // vmul d3, d0, d2 11741 // vmla d3, d1, d2 11742 // is slower than 11743 // vadd d2, d0, d1 11744 // vmul d3, d2, d2 11745 static SDValue PerformVMULCombine(SDNode *N, 11746 TargetLowering::DAGCombinerInfo &DCI, 11747 const ARMSubtarget *Subtarget) { 11748 if (!Subtarget->hasVMLxForwarding()) 11749 return SDValue(); 11750 11751 SelectionDAG &DAG = DCI.DAG; 11752 SDValue N0 = N->getOperand(0); 11753 SDValue N1 = N->getOperand(1); 11754 unsigned Opcode = N0.getOpcode(); 11755 if (Opcode != ISD::ADD && Opcode != ISD::SUB && 11756 Opcode != ISD::FADD && Opcode != ISD::FSUB) { 11757 Opcode = N1.getOpcode(); 11758 if (Opcode != ISD::ADD && Opcode != ISD::SUB && 11759 Opcode != ISD::FADD && Opcode != ISD::FSUB) 11760 return SDValue(); 11761 std::swap(N0, N1); 11762 } 11763 11764 if (N0 == N1) 11765 return SDValue(); 11766 11767 EVT VT = N->getValueType(0); 11768 SDLoc DL(N); 11769 SDValue N00 = N0->getOperand(0); 11770 SDValue N01 = N0->getOperand(1); 11771 return DAG.getNode(Opcode, DL, VT, 11772 DAG.getNode(ISD::MUL, DL, VT, N00, N1), 11773 DAG.getNode(ISD::MUL, DL, VT, N01, N1)); 11774 } 11775 11776 static SDValue PerformMULCombine(SDNode *N, 11777 TargetLowering::DAGCombinerInfo &DCI, 11778 const ARMSubtarget *Subtarget) { 11779 SelectionDAG &DAG = DCI.DAG; 11780 11781 if (Subtarget->isThumb1Only()) 11782 return SDValue(); 11783 11784 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer()) 11785 return SDValue(); 11786 11787 EVT VT = N->getValueType(0); 11788 if (VT.is64BitVector() || VT.is128BitVector()) 11789 return PerformVMULCombine(N, DCI, Subtarget); 11790 if (VT != MVT::i32) 11791 return SDValue(); 11792 11793 ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(1)); 11794 if (!C) 11795 return SDValue(); 11796 11797 int64_t MulAmt = C->getSExtValue(); 11798 unsigned ShiftAmt = countTrailingZeros<uint64_t>(MulAmt); 11799 11800 ShiftAmt = ShiftAmt & (32 - 1); 11801 SDValue V = N->getOperand(0); 11802 SDLoc DL(N); 11803 11804 SDValue Res; 11805 MulAmt >>= ShiftAmt; 11806 11807 if (MulAmt >= 0) { 11808 if (isPowerOf2_32(MulAmt - 1)) { 11809 // (mul x, 2^N + 1) => (add (shl x, N), x) 11810 Res = DAG.getNode(ISD::ADD, DL, VT, 11811 V, 11812 DAG.getNode(ISD::SHL, DL, VT, 11813 V, 11814 DAG.getConstant(Log2_32(MulAmt - 1), DL, 11815 MVT::i32))); 11816 } else if (isPowerOf2_32(MulAmt + 1)) { 11817 // (mul x, 2^N - 1) => (sub (shl x, N), x) 11818 Res = DAG.getNode(ISD::SUB, DL, VT, 11819 DAG.getNode(ISD::SHL, DL, VT, 11820 V, 11821 DAG.getConstant(Log2_32(MulAmt + 1), DL, 11822 MVT::i32)), 11823 V); 11824 } else 11825 return SDValue(); 11826 } else { 11827 uint64_t MulAmtAbs = -MulAmt; 11828 if (isPowerOf2_32(MulAmtAbs + 1)) { 11829 // (mul x, -(2^N - 1)) => (sub x, (shl x, N)) 11830 Res = DAG.getNode(ISD::SUB, DL, VT, 11831 V, 11832 DAG.getNode(ISD::SHL, DL, VT, 11833 V, 11834 DAG.getConstant(Log2_32(MulAmtAbs + 1), DL, 11835 MVT::i32))); 11836 } else if (isPowerOf2_32(MulAmtAbs - 1)) { 11837 // (mul x, -(2^N + 1)) => - (add (shl x, N), x) 11838 Res = DAG.getNode(ISD::ADD, DL, VT, 11839 V, 11840 DAG.getNode(ISD::SHL, DL, VT, 11841 V, 11842 DAG.getConstant(Log2_32(MulAmtAbs - 1), DL, 11843 MVT::i32))); 11844 Res = DAG.getNode(ISD::SUB, DL, VT, 11845 DAG.getConstant(0, DL, MVT::i32), Res); 11846 } else 11847 return SDValue(); 11848 } 11849 11850 if (ShiftAmt != 0) 11851 Res = DAG.getNode(ISD::SHL, DL, VT, 11852 Res, DAG.getConstant(ShiftAmt, DL, MVT::i32)); 11853 11854 // Do not add new nodes to DAG combiner worklist. 11855 DCI.CombineTo(N, Res, false); 11856 return SDValue(); 11857 } 11858 11859 static SDValue CombineANDShift(SDNode *N, 11860 TargetLowering::DAGCombinerInfo &DCI, 11861 const ARMSubtarget *Subtarget) { 11862 // Allow DAGCombine to pattern-match before we touch the canonical form. 11863 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer()) 11864 return SDValue(); 11865 11866 if (N->getValueType(0) != MVT::i32) 11867 return SDValue(); 11868 11869 ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N->getOperand(1)); 11870 if (!N1C) 11871 return SDValue(); 11872 11873 uint32_t C1 = (uint32_t)N1C->getZExtValue(); 11874 // Don't transform uxtb/uxth. 11875 if (C1 == 255 || C1 == 65535) 11876 return SDValue(); 11877 11878 SDNode *N0 = N->getOperand(0).getNode(); 11879 if (!N0->hasOneUse()) 11880 return SDValue(); 11881 11882 if (N0->getOpcode() != ISD::SHL && N0->getOpcode() != ISD::SRL) 11883 return SDValue(); 11884 11885 bool LeftShift = N0->getOpcode() == ISD::SHL; 11886 11887 ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0->getOperand(1)); 11888 if (!N01C) 11889 return SDValue(); 11890 11891 uint32_t C2 = (uint32_t)N01C->getZExtValue(); 11892 if (!C2 || C2 >= 32) 11893 return SDValue(); 11894 11895 // Clear irrelevant bits in the mask. 11896 if (LeftShift) 11897 C1 &= (-1U << C2); 11898 else 11899 C1 &= (-1U >> C2); 11900 11901 SelectionDAG &DAG = DCI.DAG; 11902 SDLoc DL(N); 11903 11904 // We have a pattern of the form "(and (shl x, c2) c1)" or 11905 // "(and (srl x, c2) c1)", where c1 is a shifted mask. Try to 11906 // transform to a pair of shifts, to save materializing c1. 11907 11908 // First pattern: right shift, then mask off leading bits. 11909 // FIXME: Use demanded bits? 11910 if (!LeftShift && isMask_32(C1)) { 11911 uint32_t C3 = countLeadingZeros(C1); 11912 if (C2 < C3) { 11913 SDValue SHL = DAG.getNode(ISD::SHL, DL, MVT::i32, N0->getOperand(0), 11914 DAG.getConstant(C3 - C2, DL, MVT::i32)); 11915 return DAG.getNode(ISD::SRL, DL, MVT::i32, SHL, 11916 DAG.getConstant(C3, DL, MVT::i32)); 11917 } 11918 } 11919 11920 // First pattern, reversed: left shift, then mask off trailing bits. 11921 if (LeftShift && isMask_32(~C1)) { 11922 uint32_t C3 = countTrailingZeros(C1); 11923 if (C2 < C3) { 11924 SDValue SHL = DAG.getNode(ISD::SRL, DL, MVT::i32, N0->getOperand(0), 11925 DAG.getConstant(C3 - C2, DL, MVT::i32)); 11926 return DAG.getNode(ISD::SHL, DL, MVT::i32, SHL, 11927 DAG.getConstant(C3, DL, MVT::i32)); 11928 } 11929 } 11930 11931 // Second pattern: left shift, then mask off leading bits. 11932 // FIXME: Use demanded bits? 11933 if (LeftShift && isShiftedMask_32(C1)) { 11934 uint32_t Trailing = countTrailingZeros(C1); 11935 uint32_t C3 = countLeadingZeros(C1); 11936 if (Trailing == C2 && C2 + C3 < 32) { 11937 SDValue SHL = DAG.getNode(ISD::SHL, DL, MVT::i32, N0->getOperand(0), 11938 DAG.getConstant(C2 + C3, DL, MVT::i32)); 11939 return DAG.getNode(ISD::SRL, DL, MVT::i32, SHL, 11940 DAG.getConstant(C3, DL, MVT::i32)); 11941 } 11942 } 11943 11944 // Second pattern, reversed: right shift, then mask off trailing bits. 11945 // FIXME: Handle other patterns of known/demanded bits. 11946 if (!LeftShift && isShiftedMask_32(C1)) { 11947 uint32_t Leading = countLeadingZeros(C1); 11948 uint32_t C3 = countTrailingZeros(C1); 11949 if (Leading == C2 && C2 + C3 < 32) { 11950 SDValue SHL = DAG.getNode(ISD::SRL, DL, MVT::i32, N0->getOperand(0), 11951 DAG.getConstant(C2 + C3, DL, MVT::i32)); 11952 return DAG.getNode(ISD::SHL, DL, MVT::i32, SHL, 11953 DAG.getConstant(C3, DL, MVT::i32)); 11954 } 11955 } 11956 11957 // FIXME: Transform "(and (shl x, c2) c1)" -> 11958 // "(shl (and x, c1>>c2), c2)" if "c1 >> c2" is a cheaper immediate than 11959 // c1. 11960 return SDValue(); 11961 } 11962 11963 static SDValue PerformANDCombine(SDNode *N, 11964 TargetLowering::DAGCombinerInfo &DCI, 11965 const ARMSubtarget *Subtarget) { 11966 // Attempt to use immediate-form VBIC 11967 BuildVectorSDNode *BVN = dyn_cast<BuildVectorSDNode>(N->getOperand(1)); 11968 SDLoc dl(N); 11969 EVT VT = N->getValueType(0); 11970 SelectionDAG &DAG = DCI.DAG; 11971 11972 if(!DAG.getTargetLoweringInfo().isTypeLegal(VT)) 11973 return SDValue(); 11974 11975 APInt SplatBits, SplatUndef; 11976 unsigned SplatBitSize; 11977 bool HasAnyUndefs; 11978 if (BVN && Subtarget->hasNEON() && 11979 BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs)) { 11980 if (SplatBitSize <= 64) { 11981 EVT VbicVT; 11982 SDValue Val = isVMOVModifiedImm((~SplatBits).getZExtValue(), 11983 SplatUndef.getZExtValue(), SplatBitSize, 11984 DAG, dl, VbicVT, VT.is128BitVector(), 11985 OtherModImm); 11986 if (Val.getNode()) { 11987 SDValue Input = 11988 DAG.getNode(ISD::BITCAST, dl, VbicVT, N->getOperand(0)); 11989 SDValue Vbic = DAG.getNode(ARMISD::VBICIMM, dl, VbicVT, Input, Val); 11990 return DAG.getNode(ISD::BITCAST, dl, VT, Vbic); 11991 } 11992 } 11993 } 11994 11995 if (!Subtarget->isThumb1Only()) { 11996 // fold (and (select cc, -1, c), x) -> (select cc, x, (and, x, c)) 11997 if (SDValue Result = combineSelectAndUseCommutative(N, true, DCI)) 11998 return Result; 11999 12000 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget)) 12001 return Result; 12002 } 12003 12004 if (Subtarget->isThumb1Only()) 12005 if (SDValue Result = CombineANDShift(N, DCI, Subtarget)) 12006 return Result; 12007 12008 return SDValue(); 12009 } 12010 12011 // Try combining OR nodes to SMULWB, SMULWT. 12012 static SDValue PerformORCombineToSMULWBT(SDNode *OR, 12013 TargetLowering::DAGCombinerInfo &DCI, 12014 const ARMSubtarget *Subtarget) { 12015 if (!Subtarget->hasV6Ops() || 12016 (Subtarget->isThumb() && 12017 (!Subtarget->hasThumb2() || !Subtarget->hasDSP()))) 12018 return SDValue(); 12019 12020 SDValue SRL = OR->getOperand(0); 12021 SDValue SHL = OR->getOperand(1); 12022 12023 if (SRL.getOpcode() != ISD::SRL || SHL.getOpcode() != ISD::SHL) { 12024 SRL = OR->getOperand(1); 12025 SHL = OR->getOperand(0); 12026 } 12027 if (!isSRL16(SRL) || !isSHL16(SHL)) 12028 return SDValue(); 12029 12030 // The first operands to the shifts need to be the two results from the 12031 // same smul_lohi node. 12032 if ((SRL.getOperand(0).getNode() != SHL.getOperand(0).getNode()) || 12033 SRL.getOperand(0).getOpcode() != ISD::SMUL_LOHI) 12034 return SDValue(); 12035 12036 SDNode *SMULLOHI = SRL.getOperand(0).getNode(); 12037 if (SRL.getOperand(0) != SDValue(SMULLOHI, 0) || 12038 SHL.getOperand(0) != SDValue(SMULLOHI, 1)) 12039 return SDValue(); 12040 12041 // Now we have: 12042 // (or (srl (smul_lohi ?, ?), 16), (shl (smul_lohi ?, ?), 16))) 12043 // For SMUL[B|T] smul_lohi will take a 32-bit and a 16-bit arguments. 12044 // For SMUWB the 16-bit value will signed extended somehow. 12045 // For SMULWT only the SRA is required. 12046 // Check both sides of SMUL_LOHI 12047 SDValue OpS16 = SMULLOHI->getOperand(0); 12048 SDValue OpS32 = SMULLOHI->getOperand(1); 12049 12050 SelectionDAG &DAG = DCI.DAG; 12051 if (!isS16(OpS16, DAG) && !isSRA16(OpS16)) { 12052 OpS16 = OpS32; 12053 OpS32 = SMULLOHI->getOperand(0); 12054 } 12055 12056 SDLoc dl(OR); 12057 unsigned Opcode = 0; 12058 if (isS16(OpS16, DAG)) 12059 Opcode = ARMISD::SMULWB; 12060 else if (isSRA16(OpS16)) { 12061 Opcode = ARMISD::SMULWT; 12062 OpS16 = OpS16->getOperand(0); 12063 } 12064 else 12065 return SDValue(); 12066 12067 SDValue Res = DAG.getNode(Opcode, dl, MVT::i32, OpS32, OpS16); 12068 DAG.ReplaceAllUsesOfValueWith(SDValue(OR, 0), Res); 12069 return SDValue(OR, 0); 12070 } 12071 12072 static SDValue PerformORCombineToBFI(SDNode *N, 12073 TargetLowering::DAGCombinerInfo &DCI, 12074 const ARMSubtarget *Subtarget) { 12075 // BFI is only available on V6T2+ 12076 if (Subtarget->isThumb1Only() || !Subtarget->hasV6T2Ops()) 12077 return SDValue(); 12078 12079 EVT VT = N->getValueType(0); 12080 SDValue N0 = N->getOperand(0); 12081 SDValue N1 = N->getOperand(1); 12082 SelectionDAG &DAG = DCI.DAG; 12083 SDLoc DL(N); 12084 // 1) or (and A, mask), val => ARMbfi A, val, mask 12085 // iff (val & mask) == val 12086 // 12087 // 2) or (and A, mask), (and B, mask2) => ARMbfi A, (lsr B, amt), mask 12088 // 2a) iff isBitFieldInvertedMask(mask) && isBitFieldInvertedMask(~mask2) 12089 // && mask == ~mask2 12090 // 2b) iff isBitFieldInvertedMask(~mask) && isBitFieldInvertedMask(mask2) 12091 // && ~mask == mask2 12092 // (i.e., copy a bitfield value into another bitfield of the same width) 12093 12094 if (VT != MVT::i32) 12095 return SDValue(); 12096 12097 SDValue N00 = N0.getOperand(0); 12098 12099 // The value and the mask need to be constants so we can verify this is 12100 // actually a bitfield set. If the mask is 0xffff, we can do better 12101 // via a movt instruction, so don't use BFI in that case. 12102 SDValue MaskOp = N0.getOperand(1); 12103 ConstantSDNode *MaskC = dyn_cast<ConstantSDNode>(MaskOp); 12104 if (!MaskC) 12105 return SDValue(); 12106 unsigned Mask = MaskC->getZExtValue(); 12107 if (Mask == 0xffff) 12108 return SDValue(); 12109 SDValue Res; 12110 // Case (1): or (and A, mask), val => ARMbfi A, val, mask 12111 ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1); 12112 if (N1C) { 12113 unsigned Val = N1C->getZExtValue(); 12114 if ((Val & ~Mask) != Val) 12115 return SDValue(); 12116 12117 if (ARM::isBitFieldInvertedMask(Mask)) { 12118 Val >>= countTrailingZeros(~Mask); 12119 12120 Res = DAG.getNode(ARMISD::BFI, DL, VT, N00, 12121 DAG.getConstant(Val, DL, MVT::i32), 12122 DAG.getConstant(Mask, DL, MVT::i32)); 12123 12124 DCI.CombineTo(N, Res, false); 12125 // Return value from the original node to inform the combiner than N is 12126 // now dead. 12127 return SDValue(N, 0); 12128 } 12129 } else if (N1.getOpcode() == ISD::AND) { 12130 // case (2) or (and A, mask), (and B, mask2) => ARMbfi A, (lsr B, amt), mask 12131 ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1)); 12132 if (!N11C) 12133 return SDValue(); 12134 unsigned Mask2 = N11C->getZExtValue(); 12135 12136 // Mask and ~Mask2 (or reverse) must be equivalent for the BFI pattern 12137 // as is to match. 12138 if (ARM::isBitFieldInvertedMask(Mask) && 12139 (Mask == ~Mask2)) { 12140 // The pack halfword instruction works better for masks that fit it, 12141 // so use that when it's available. 12142 if (Subtarget->hasDSP() && 12143 (Mask == 0xffff || Mask == 0xffff0000)) 12144 return SDValue(); 12145 // 2a 12146 unsigned amt = countTrailingZeros(Mask2); 12147 Res = DAG.getNode(ISD::SRL, DL, VT, N1.getOperand(0), 12148 DAG.getConstant(amt, DL, MVT::i32)); 12149 Res = DAG.getNode(ARMISD::BFI, DL, VT, N00, Res, 12150 DAG.getConstant(Mask, DL, MVT::i32)); 12151 DCI.CombineTo(N, Res, false); 12152 // Return value from the original node to inform the combiner than N is 12153 // now dead. 12154 return SDValue(N, 0); 12155 } else if (ARM::isBitFieldInvertedMask(~Mask) && 12156 (~Mask == Mask2)) { 12157 // The pack halfword instruction works better for masks that fit it, 12158 // so use that when it's available. 12159 if (Subtarget->hasDSP() && 12160 (Mask2 == 0xffff || Mask2 == 0xffff0000)) 12161 return SDValue(); 12162 // 2b 12163 unsigned lsb = countTrailingZeros(Mask); 12164 Res = DAG.getNode(ISD::SRL, DL, VT, N00, 12165 DAG.getConstant(lsb, DL, MVT::i32)); 12166 Res = DAG.getNode(ARMISD::BFI, DL, VT, N1.getOperand(0), Res, 12167 DAG.getConstant(Mask2, DL, MVT::i32)); 12168 DCI.CombineTo(N, Res, false); 12169 // Return value from the original node to inform the combiner than N is 12170 // now dead. 12171 return SDValue(N, 0); 12172 } 12173 } 12174 12175 if (DAG.MaskedValueIsZero(N1, MaskC->getAPIntValue()) && 12176 N00.getOpcode() == ISD::SHL && isa<ConstantSDNode>(N00.getOperand(1)) && 12177 ARM::isBitFieldInvertedMask(~Mask)) { 12178 // Case (3): or (and (shl A, #shamt), mask), B => ARMbfi B, A, ~mask 12179 // where lsb(mask) == #shamt and masked bits of B are known zero. 12180 SDValue ShAmt = N00.getOperand(1); 12181 unsigned ShAmtC = cast<ConstantSDNode>(ShAmt)->getZExtValue(); 12182 unsigned LSB = countTrailingZeros(Mask); 12183 if (ShAmtC != LSB) 12184 return SDValue(); 12185 12186 Res = DAG.getNode(ARMISD::BFI, DL, VT, N1, N00.getOperand(0), 12187 DAG.getConstant(~Mask, DL, MVT::i32)); 12188 12189 DCI.CombineTo(N, Res, false); 12190 // Return value from the original node to inform the combiner than N is 12191 // now dead. 12192 return SDValue(N, 0); 12193 } 12194 12195 return SDValue(); 12196 } 12197 12198 static bool isValidMVECond(unsigned CC, bool IsFloat) { 12199 switch (CC) { 12200 case ARMCC::EQ: 12201 case ARMCC::NE: 12202 case ARMCC::LE: 12203 case ARMCC::GT: 12204 case ARMCC::GE: 12205 case ARMCC::LT: 12206 return true; 12207 case ARMCC::HS: 12208 case ARMCC::HI: 12209 return !IsFloat; 12210 default: 12211 return false; 12212 }; 12213 } 12214 12215 static SDValue PerformORCombine_i1(SDNode *N, 12216 TargetLowering::DAGCombinerInfo &DCI, 12217 const ARMSubtarget *Subtarget) { 12218 // Try to invert "or A, B" -> "and ~A, ~B", as the "and" is easier to chain 12219 // together with predicates 12220 EVT VT = N->getValueType(0); 12221 SDValue N0 = N->getOperand(0); 12222 SDValue N1 = N->getOperand(1); 12223 12224 ARMCC::CondCodes CondCode0 = ARMCC::AL; 12225 ARMCC::CondCodes CondCode1 = ARMCC::AL; 12226 if (N0->getOpcode() == ARMISD::VCMP) 12227 CondCode0 = (ARMCC::CondCodes)cast<const ConstantSDNode>(N0->getOperand(2)) 12228 ->getZExtValue(); 12229 else if (N0->getOpcode() == ARMISD::VCMPZ) 12230 CondCode0 = (ARMCC::CondCodes)cast<const ConstantSDNode>(N0->getOperand(1)) 12231 ->getZExtValue(); 12232 if (N1->getOpcode() == ARMISD::VCMP) 12233 CondCode1 = (ARMCC::CondCodes)cast<const ConstantSDNode>(N1->getOperand(2)) 12234 ->getZExtValue(); 12235 else if (N1->getOpcode() == ARMISD::VCMPZ) 12236 CondCode1 = (ARMCC::CondCodes)cast<const ConstantSDNode>(N1->getOperand(1)) 12237 ->getZExtValue(); 12238 12239 if (CondCode0 == ARMCC::AL || CondCode1 == ARMCC::AL) 12240 return SDValue(); 12241 12242 unsigned Opposite0 = ARMCC::getOppositeCondition(CondCode0); 12243 unsigned Opposite1 = ARMCC::getOppositeCondition(CondCode1); 12244 12245 if (!isValidMVECond(Opposite0, 12246 N0->getOperand(0)->getValueType(0).isFloatingPoint()) || 12247 !isValidMVECond(Opposite1, 12248 N1->getOperand(0)->getValueType(0).isFloatingPoint())) 12249 return SDValue(); 12250 12251 SmallVector<SDValue, 4> Ops0; 12252 Ops0.push_back(N0->getOperand(0)); 12253 if (N0->getOpcode() == ARMISD::VCMP) 12254 Ops0.push_back(N0->getOperand(1)); 12255 Ops0.push_back(DCI.DAG.getConstant(Opposite0, SDLoc(N0), MVT::i32)); 12256 SmallVector<SDValue, 4> Ops1; 12257 Ops1.push_back(N1->getOperand(0)); 12258 if (N1->getOpcode() == ARMISD::VCMP) 12259 Ops1.push_back(N1->getOperand(1)); 12260 Ops1.push_back(DCI.DAG.getConstant(Opposite1, SDLoc(N1), MVT::i32)); 12261 12262 SDValue NewN0 = DCI.DAG.getNode(N0->getOpcode(), SDLoc(N0), VT, Ops0); 12263 SDValue NewN1 = DCI.DAG.getNode(N1->getOpcode(), SDLoc(N1), VT, Ops1); 12264 SDValue And = DCI.DAG.getNode(ISD::AND, SDLoc(N), VT, NewN0, NewN1); 12265 return DCI.DAG.getNode(ISD::XOR, SDLoc(N), VT, And, 12266 DCI.DAG.getAllOnesConstant(SDLoc(N), VT)); 12267 } 12268 12269 /// PerformORCombine - Target-specific dag combine xforms for ISD::OR 12270 static SDValue PerformORCombine(SDNode *N, 12271 TargetLowering::DAGCombinerInfo &DCI, 12272 const ARMSubtarget *Subtarget) { 12273 // Attempt to use immediate-form VORR 12274 BuildVectorSDNode *BVN = dyn_cast<BuildVectorSDNode>(N->getOperand(1)); 12275 SDLoc dl(N); 12276 EVT VT = N->getValueType(0); 12277 SelectionDAG &DAG = DCI.DAG; 12278 12279 if(!DAG.getTargetLoweringInfo().isTypeLegal(VT)) 12280 return SDValue(); 12281 12282 APInt SplatBits, SplatUndef; 12283 unsigned SplatBitSize; 12284 bool HasAnyUndefs; 12285 if (BVN && Subtarget->hasNEON() && 12286 BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs)) { 12287 if (SplatBitSize <= 64) { 12288 EVT VorrVT; 12289 SDValue Val = isVMOVModifiedImm(SplatBits.getZExtValue(), 12290 SplatUndef.getZExtValue(), SplatBitSize, 12291 DAG, dl, VorrVT, VT.is128BitVector(), 12292 OtherModImm); 12293 if (Val.getNode()) { 12294 SDValue Input = 12295 DAG.getNode(ISD::BITCAST, dl, VorrVT, N->getOperand(0)); 12296 SDValue Vorr = DAG.getNode(ARMISD::VORRIMM, dl, VorrVT, Input, Val); 12297 return DAG.getNode(ISD::BITCAST, dl, VT, Vorr); 12298 } 12299 } 12300 } 12301 12302 if (!Subtarget->isThumb1Only()) { 12303 // fold (or (select cc, 0, c), x) -> (select cc, x, (or, x, c)) 12304 if (SDValue Result = combineSelectAndUseCommutative(N, false, DCI)) 12305 return Result; 12306 if (SDValue Result = PerformORCombineToSMULWBT(N, DCI, Subtarget)) 12307 return Result; 12308 } 12309 12310 SDValue N0 = N->getOperand(0); 12311 SDValue N1 = N->getOperand(1); 12312 12313 // (or (and B, A), (and C, ~A)) => (VBSL A, B, C) when A is a constant. 12314 if (Subtarget->hasNEON() && N1.getOpcode() == ISD::AND && VT.isVector() && 12315 DAG.getTargetLoweringInfo().isTypeLegal(VT)) { 12316 12317 // The code below optimizes (or (and X, Y), Z). 12318 // The AND operand needs to have a single user to make these optimizations 12319 // profitable. 12320 if (N0.getOpcode() != ISD::AND || !N0.hasOneUse()) 12321 return SDValue(); 12322 12323 APInt SplatUndef; 12324 unsigned SplatBitSize; 12325 bool HasAnyUndefs; 12326 12327 APInt SplatBits0, SplatBits1; 12328 BuildVectorSDNode *BVN0 = dyn_cast<BuildVectorSDNode>(N0->getOperand(1)); 12329 BuildVectorSDNode *BVN1 = dyn_cast<BuildVectorSDNode>(N1->getOperand(1)); 12330 // Ensure that the second operand of both ands are constants 12331 if (BVN0 && BVN0->isConstantSplat(SplatBits0, SplatUndef, SplatBitSize, 12332 HasAnyUndefs) && !HasAnyUndefs) { 12333 if (BVN1 && BVN1->isConstantSplat(SplatBits1, SplatUndef, SplatBitSize, 12334 HasAnyUndefs) && !HasAnyUndefs) { 12335 // Ensure that the bit width of the constants are the same and that 12336 // the splat arguments are logical inverses as per the pattern we 12337 // are trying to simplify. 12338 if (SplatBits0.getBitWidth() == SplatBits1.getBitWidth() && 12339 SplatBits0 == ~SplatBits1) { 12340 // Canonicalize the vector type to make instruction selection 12341 // simpler. 12342 EVT CanonicalVT = VT.is128BitVector() ? MVT::v4i32 : MVT::v2i32; 12343 SDValue Result = DAG.getNode(ARMISD::VBSL, dl, CanonicalVT, 12344 N0->getOperand(1), 12345 N0->getOperand(0), 12346 N1->getOperand(0)); 12347 return DAG.getNode(ISD::BITCAST, dl, VT, Result); 12348 } 12349 } 12350 } 12351 } 12352 12353 if (Subtarget->hasMVEIntegerOps() && 12354 (VT == MVT::v4i1 || VT == MVT::v8i1 || VT == MVT::v16i1)) 12355 return PerformORCombine_i1(N, DCI, Subtarget); 12356 12357 // Try to use the ARM/Thumb2 BFI (bitfield insert) instruction when 12358 // reasonable. 12359 if (N0.getOpcode() == ISD::AND && N0.hasOneUse()) { 12360 if (SDValue Res = PerformORCombineToBFI(N, DCI, Subtarget)) 12361 return Res; 12362 } 12363 12364 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget)) 12365 return Result; 12366 12367 return SDValue(); 12368 } 12369 12370 static SDValue PerformXORCombine(SDNode *N, 12371 TargetLowering::DAGCombinerInfo &DCI, 12372 const ARMSubtarget *Subtarget) { 12373 EVT VT = N->getValueType(0); 12374 SelectionDAG &DAG = DCI.DAG; 12375 12376 if(!DAG.getTargetLoweringInfo().isTypeLegal(VT)) 12377 return SDValue(); 12378 12379 if (!Subtarget->isThumb1Only()) { 12380 // fold (xor (select cc, 0, c), x) -> (select cc, x, (xor, x, c)) 12381 if (SDValue Result = combineSelectAndUseCommutative(N, false, DCI)) 12382 return Result; 12383 12384 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget)) 12385 return Result; 12386 } 12387 12388 return SDValue(); 12389 } 12390 12391 // ParseBFI - given a BFI instruction in N, extract the "from" value (Rn) and return it, 12392 // and fill in FromMask and ToMask with (consecutive) bits in "from" to be extracted and 12393 // their position in "to" (Rd). 12394 static SDValue ParseBFI(SDNode *N, APInt &ToMask, APInt &FromMask) { 12395 assert(N->getOpcode() == ARMISD::BFI); 12396 12397 SDValue From = N->getOperand(1); 12398 ToMask = ~cast<ConstantSDNode>(N->getOperand(2))->getAPIntValue(); 12399 FromMask = APInt::getLowBitsSet(ToMask.getBitWidth(), ToMask.countPopulation()); 12400 12401 // If the Base came from a SHR #C, we can deduce that it is really testing bit 12402 // #C in the base of the SHR. 12403 if (From->getOpcode() == ISD::SRL && 12404 isa<ConstantSDNode>(From->getOperand(1))) { 12405 APInt Shift = cast<ConstantSDNode>(From->getOperand(1))->getAPIntValue(); 12406 assert(Shift.getLimitedValue() < 32 && "Shift too large!"); 12407 FromMask <<= Shift.getLimitedValue(31); 12408 From = From->getOperand(0); 12409 } 12410 12411 return From; 12412 } 12413 12414 // If A and B contain one contiguous set of bits, does A | B == A . B? 12415 // 12416 // Neither A nor B must be zero. 12417 static bool BitsProperlyConcatenate(const APInt &A, const APInt &B) { 12418 unsigned LastActiveBitInA = A.countTrailingZeros(); 12419 unsigned FirstActiveBitInB = B.getBitWidth() - B.countLeadingZeros() - 1; 12420 return LastActiveBitInA - 1 == FirstActiveBitInB; 12421 } 12422 12423 static SDValue FindBFIToCombineWith(SDNode *N) { 12424 // We have a BFI in N. Follow a possible chain of BFIs and find a BFI it can combine with, 12425 // if one exists. 12426 APInt ToMask, FromMask; 12427 SDValue From = ParseBFI(N, ToMask, FromMask); 12428 SDValue To = N->getOperand(0); 12429 12430 // Now check for a compatible BFI to merge with. We can pass through BFIs that 12431 // aren't compatible, but not if they set the same bit in their destination as 12432 // we do (or that of any BFI we're going to combine with). 12433 SDValue V = To; 12434 APInt CombinedToMask = ToMask; 12435 while (V.getOpcode() == ARMISD::BFI) { 12436 APInt NewToMask, NewFromMask; 12437 SDValue NewFrom = ParseBFI(V.getNode(), NewToMask, NewFromMask); 12438 if (NewFrom != From) { 12439 // This BFI has a different base. Keep going. 12440 CombinedToMask |= NewToMask; 12441 V = V.getOperand(0); 12442 continue; 12443 } 12444 12445 // Do the written bits conflict with any we've seen so far? 12446 if ((NewToMask & CombinedToMask).getBoolValue()) 12447 // Conflicting bits - bail out because going further is unsafe. 12448 return SDValue(); 12449 12450 // Are the new bits contiguous when combined with the old bits? 12451 if (BitsProperlyConcatenate(ToMask, NewToMask) && 12452 BitsProperlyConcatenate(FromMask, NewFromMask)) 12453 return V; 12454 if (BitsProperlyConcatenate(NewToMask, ToMask) && 12455 BitsProperlyConcatenate(NewFromMask, FromMask)) 12456 return V; 12457 12458 // We've seen a write to some bits, so track it. 12459 CombinedToMask |= NewToMask; 12460 // Keep going... 12461 V = V.getOperand(0); 12462 } 12463 12464 return SDValue(); 12465 } 12466 12467 static SDValue PerformBFICombine(SDNode *N, 12468 TargetLowering::DAGCombinerInfo &DCI) { 12469 SDValue N1 = N->getOperand(1); 12470 if (N1.getOpcode() == ISD::AND) { 12471 // (bfi A, (and B, Mask1), Mask2) -> (bfi A, B, Mask2) iff 12472 // the bits being cleared by the AND are not demanded by the BFI. 12473 ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1)); 12474 if (!N11C) 12475 return SDValue(); 12476 unsigned InvMask = cast<ConstantSDNode>(N->getOperand(2))->getZExtValue(); 12477 unsigned LSB = countTrailingZeros(~InvMask); 12478 unsigned Width = (32 - countLeadingZeros(~InvMask)) - LSB; 12479 assert(Width < 12480 static_cast<unsigned>(std::numeric_limits<unsigned>::digits) && 12481 "undefined behavior"); 12482 unsigned Mask = (1u << Width) - 1; 12483 unsigned Mask2 = N11C->getZExtValue(); 12484 if ((Mask & (~Mask2)) == 0) 12485 return DCI.DAG.getNode(ARMISD::BFI, SDLoc(N), N->getValueType(0), 12486 N->getOperand(0), N1.getOperand(0), 12487 N->getOperand(2)); 12488 } else if (N->getOperand(0).getOpcode() == ARMISD::BFI) { 12489 // We have a BFI of a BFI. Walk up the BFI chain to see how long it goes. 12490 // Keep track of any consecutive bits set that all come from the same base 12491 // value. We can combine these together into a single BFI. 12492 SDValue CombineBFI = FindBFIToCombineWith(N); 12493 if (CombineBFI == SDValue()) 12494 return SDValue(); 12495 12496 // We've found a BFI. 12497 APInt ToMask1, FromMask1; 12498 SDValue From1 = ParseBFI(N, ToMask1, FromMask1); 12499 12500 APInt ToMask2, FromMask2; 12501 SDValue From2 = ParseBFI(CombineBFI.getNode(), ToMask2, FromMask2); 12502 assert(From1 == From2); 12503 (void)From2; 12504 12505 // First, unlink CombineBFI. 12506 DCI.DAG.ReplaceAllUsesWith(CombineBFI, CombineBFI.getOperand(0)); 12507 // Then create a new BFI, combining the two together. 12508 APInt NewFromMask = FromMask1 | FromMask2; 12509 APInt NewToMask = ToMask1 | ToMask2; 12510 12511 EVT VT = N->getValueType(0); 12512 SDLoc dl(N); 12513 12514 if (NewFromMask[0] == 0) 12515 From1 = DCI.DAG.getNode( 12516 ISD::SRL, dl, VT, From1, 12517 DCI.DAG.getConstant(NewFromMask.countTrailingZeros(), dl, VT)); 12518 return DCI.DAG.getNode(ARMISD::BFI, dl, VT, N->getOperand(0), From1, 12519 DCI.DAG.getConstant(~NewToMask, dl, VT)); 12520 } 12521 return SDValue(); 12522 } 12523 12524 /// PerformVMOVRRDCombine - Target-specific dag combine xforms for 12525 /// ARMISD::VMOVRRD. 12526 static SDValue PerformVMOVRRDCombine(SDNode *N, 12527 TargetLowering::DAGCombinerInfo &DCI, 12528 const ARMSubtarget *Subtarget) { 12529 // vmovrrd(vmovdrr x, y) -> x,y 12530 SDValue InDouble = N->getOperand(0); 12531 if (InDouble.getOpcode() == ARMISD::VMOVDRR && Subtarget->hasFP64()) 12532 return DCI.CombineTo(N, InDouble.getOperand(0), InDouble.getOperand(1)); 12533 12534 // vmovrrd(load f64) -> (load i32), (load i32) 12535 SDNode *InNode = InDouble.getNode(); 12536 if (ISD::isNormalLoad(InNode) && InNode->hasOneUse() && 12537 InNode->getValueType(0) == MVT::f64 && 12538 InNode->getOperand(1).getOpcode() == ISD::FrameIndex && 12539 !cast<LoadSDNode>(InNode)->isVolatile()) { 12540 // TODO: Should this be done for non-FrameIndex operands? 12541 LoadSDNode *LD = cast<LoadSDNode>(InNode); 12542 12543 SelectionDAG &DAG = DCI.DAG; 12544 SDLoc DL(LD); 12545 SDValue BasePtr = LD->getBasePtr(); 12546 SDValue NewLD1 = 12547 DAG.getLoad(MVT::i32, DL, LD->getChain(), BasePtr, LD->getPointerInfo(), 12548 LD->getAlignment(), LD->getMemOperand()->getFlags()); 12549 12550 SDValue OffsetPtr = DAG.getNode(ISD::ADD, DL, MVT::i32, BasePtr, 12551 DAG.getConstant(4, DL, MVT::i32)); 12552 12553 SDValue NewLD2 = DAG.getLoad(MVT::i32, DL, LD->getChain(), OffsetPtr, 12554 LD->getPointerInfo().getWithOffset(4), 12555 std::min(4U, LD->getAlignment()), 12556 LD->getMemOperand()->getFlags()); 12557 12558 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), NewLD2.getValue(1)); 12559 if (DCI.DAG.getDataLayout().isBigEndian()) 12560 std::swap (NewLD1, NewLD2); 12561 SDValue Result = DCI.CombineTo(N, NewLD1, NewLD2); 12562 return Result; 12563 } 12564 12565 return SDValue(); 12566 } 12567 12568 /// PerformVMOVDRRCombine - Target-specific dag combine xforms for 12569 /// ARMISD::VMOVDRR. This is also used for BUILD_VECTORs with 2 operands. 12570 static SDValue PerformVMOVDRRCombine(SDNode *N, SelectionDAG &DAG) { 12571 // N=vmovrrd(X); vmovdrr(N:0, N:1) -> bit_convert(X) 12572 SDValue Op0 = N->getOperand(0); 12573 SDValue Op1 = N->getOperand(1); 12574 if (Op0.getOpcode() == ISD::BITCAST) 12575 Op0 = Op0.getOperand(0); 12576 if (Op1.getOpcode() == ISD::BITCAST) 12577 Op1 = Op1.getOperand(0); 12578 if (Op0.getOpcode() == ARMISD::VMOVRRD && 12579 Op0.getNode() == Op1.getNode() && 12580 Op0.getResNo() == 0 && Op1.getResNo() == 1) 12581 return DAG.getNode(ISD::BITCAST, SDLoc(N), 12582 N->getValueType(0), Op0.getOperand(0)); 12583 return SDValue(); 12584 } 12585 12586 /// hasNormalLoadOperand - Check if any of the operands of a BUILD_VECTOR node 12587 /// are normal, non-volatile loads. If so, it is profitable to bitcast an 12588 /// i64 vector to have f64 elements, since the value can then be loaded 12589 /// directly into a VFP register. 12590 static bool hasNormalLoadOperand(SDNode *N) { 12591 unsigned NumElts = N->getValueType(0).getVectorNumElements(); 12592 for (unsigned i = 0; i < NumElts; ++i) { 12593 SDNode *Elt = N->getOperand(i).getNode(); 12594 if (ISD::isNormalLoad(Elt) && !cast<LoadSDNode>(Elt)->isVolatile()) 12595 return true; 12596 } 12597 return false; 12598 } 12599 12600 /// PerformBUILD_VECTORCombine - Target-specific dag combine xforms for 12601 /// ISD::BUILD_VECTOR. 12602 static SDValue PerformBUILD_VECTORCombine(SDNode *N, 12603 TargetLowering::DAGCombinerInfo &DCI, 12604 const ARMSubtarget *Subtarget) { 12605 // build_vector(N=ARMISD::VMOVRRD(X), N:1) -> bit_convert(X): 12606 // VMOVRRD is introduced when legalizing i64 types. It forces the i64 value 12607 // into a pair of GPRs, which is fine when the value is used as a scalar, 12608 // but if the i64 value is converted to a vector, we need to undo the VMOVRRD. 12609 SelectionDAG &DAG = DCI.DAG; 12610 if (N->getNumOperands() == 2) 12611 if (SDValue RV = PerformVMOVDRRCombine(N, DAG)) 12612 return RV; 12613 12614 // Load i64 elements as f64 values so that type legalization does not split 12615 // them up into i32 values. 12616 EVT VT = N->getValueType(0); 12617 if (VT.getVectorElementType() != MVT::i64 || !hasNormalLoadOperand(N)) 12618 return SDValue(); 12619 SDLoc dl(N); 12620 SmallVector<SDValue, 8> Ops; 12621 unsigned NumElts = VT.getVectorNumElements(); 12622 for (unsigned i = 0; i < NumElts; ++i) { 12623 SDValue V = DAG.getNode(ISD::BITCAST, dl, MVT::f64, N->getOperand(i)); 12624 Ops.push_back(V); 12625 // Make the DAGCombiner fold the bitcast. 12626 DCI.AddToWorklist(V.getNode()); 12627 } 12628 EVT FloatVT = EVT::getVectorVT(*DAG.getContext(), MVT::f64, NumElts); 12629 SDValue BV = DAG.getBuildVector(FloatVT, dl, Ops); 12630 return DAG.getNode(ISD::BITCAST, dl, VT, BV); 12631 } 12632 12633 /// Target-specific dag combine xforms for ARMISD::BUILD_VECTOR. 12634 static SDValue 12635 PerformARMBUILD_VECTORCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI) { 12636 // ARMISD::BUILD_VECTOR is introduced when legalizing ISD::BUILD_VECTOR. 12637 // At that time, we may have inserted bitcasts from integer to float. 12638 // If these bitcasts have survived DAGCombine, change the lowering of this 12639 // BUILD_VECTOR in something more vector friendly, i.e., that does not 12640 // force to use floating point types. 12641 12642 // Make sure we can change the type of the vector. 12643 // This is possible iff: 12644 // 1. The vector is only used in a bitcast to a integer type. I.e., 12645 // 1.1. Vector is used only once. 12646 // 1.2. Use is a bit convert to an integer type. 12647 // 2. The size of its operands are 32-bits (64-bits are not legal). 12648 EVT VT = N->getValueType(0); 12649 EVT EltVT = VT.getVectorElementType(); 12650 12651 // Check 1.1. and 2. 12652 if (EltVT.getSizeInBits() != 32 || !N->hasOneUse()) 12653 return SDValue(); 12654 12655 // By construction, the input type must be float. 12656 assert(EltVT == MVT::f32 && "Unexpected type!"); 12657 12658 // Check 1.2. 12659 SDNode *Use = *N->use_begin(); 12660 if (Use->getOpcode() != ISD::BITCAST || 12661 Use->getValueType(0).isFloatingPoint()) 12662 return SDValue(); 12663 12664 // Check profitability. 12665 // Model is, if more than half of the relevant operands are bitcast from 12666 // i32, turn the build_vector into a sequence of insert_vector_elt. 12667 // Relevant operands are everything that is not statically 12668 // (i.e., at compile time) bitcasted. 12669 unsigned NumOfBitCastedElts = 0; 12670 unsigned NumElts = VT.getVectorNumElements(); 12671 unsigned NumOfRelevantElts = NumElts; 12672 for (unsigned Idx = 0; Idx < NumElts; ++Idx) { 12673 SDValue Elt = N->getOperand(Idx); 12674 if (Elt->getOpcode() == ISD::BITCAST) { 12675 // Assume only bit cast to i32 will go away. 12676 if (Elt->getOperand(0).getValueType() == MVT::i32) 12677 ++NumOfBitCastedElts; 12678 } else if (Elt.isUndef() || isa<ConstantSDNode>(Elt)) 12679 // Constants are statically casted, thus do not count them as 12680 // relevant operands. 12681 --NumOfRelevantElts; 12682 } 12683 12684 // Check if more than half of the elements require a non-free bitcast. 12685 if (NumOfBitCastedElts <= NumOfRelevantElts / 2) 12686 return SDValue(); 12687 12688 SelectionDAG &DAG = DCI.DAG; 12689 // Create the new vector type. 12690 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumElts); 12691 // Check if the type is legal. 12692 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 12693 if (!TLI.isTypeLegal(VecVT)) 12694 return SDValue(); 12695 12696 // Combine: 12697 // ARMISD::BUILD_VECTOR E1, E2, ..., EN. 12698 // => BITCAST INSERT_VECTOR_ELT 12699 // (INSERT_VECTOR_ELT (...), (BITCAST EN-1), N-1), 12700 // (BITCAST EN), N. 12701 SDValue Vec = DAG.getUNDEF(VecVT); 12702 SDLoc dl(N); 12703 for (unsigned Idx = 0 ; Idx < NumElts; ++Idx) { 12704 SDValue V = N->getOperand(Idx); 12705 if (V.isUndef()) 12706 continue; 12707 if (V.getOpcode() == ISD::BITCAST && 12708 V->getOperand(0).getValueType() == MVT::i32) 12709 // Fold obvious case. 12710 V = V.getOperand(0); 12711 else { 12712 V = DAG.getNode(ISD::BITCAST, SDLoc(V), MVT::i32, V); 12713 // Make the DAGCombiner fold the bitcasts. 12714 DCI.AddToWorklist(V.getNode()); 12715 } 12716 SDValue LaneIdx = DAG.getConstant(Idx, dl, MVT::i32); 12717 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VecVT, Vec, V, LaneIdx); 12718 } 12719 Vec = DAG.getNode(ISD::BITCAST, dl, VT, Vec); 12720 // Make the DAGCombiner fold the bitcasts. 12721 DCI.AddToWorklist(Vec.getNode()); 12722 return Vec; 12723 } 12724 12725 static SDValue 12726 PerformPREDICATE_CASTCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI) { 12727 EVT VT = N->getValueType(0); 12728 SDValue Op = N->getOperand(0); 12729 SDLoc dl(N); 12730 12731 // PREDICATE_CAST(PREDICATE_CAST(x)) == PREDICATE_CAST(x) 12732 if (Op->getOpcode() == ARMISD::PREDICATE_CAST) { 12733 // If the valuetypes are the same, we can remove the cast entirely. 12734 if (Op->getOperand(0).getValueType() == VT) 12735 return Op->getOperand(0); 12736 return DCI.DAG.getNode(ARMISD::PREDICATE_CAST, dl, 12737 Op->getOperand(0).getValueType(), Op->getOperand(0)); 12738 } 12739 12740 return SDValue(); 12741 } 12742 12743 /// PerformInsertEltCombine - Target-specific dag combine xforms for 12744 /// ISD::INSERT_VECTOR_ELT. 12745 static SDValue PerformInsertEltCombine(SDNode *N, 12746 TargetLowering::DAGCombinerInfo &DCI) { 12747 // Bitcast an i64 load inserted into a vector to f64. 12748 // Otherwise, the i64 value will be legalized to a pair of i32 values. 12749 EVT VT = N->getValueType(0); 12750 SDNode *Elt = N->getOperand(1).getNode(); 12751 if (VT.getVectorElementType() != MVT::i64 || 12752 !ISD::isNormalLoad(Elt) || cast<LoadSDNode>(Elt)->isVolatile()) 12753 return SDValue(); 12754 12755 SelectionDAG &DAG = DCI.DAG; 12756 SDLoc dl(N); 12757 EVT FloatVT = EVT::getVectorVT(*DAG.getContext(), MVT::f64, 12758 VT.getVectorNumElements()); 12759 SDValue Vec = DAG.getNode(ISD::BITCAST, dl, FloatVT, N->getOperand(0)); 12760 SDValue V = DAG.getNode(ISD::BITCAST, dl, MVT::f64, N->getOperand(1)); 12761 // Make the DAGCombiner fold the bitcasts. 12762 DCI.AddToWorklist(Vec.getNode()); 12763 DCI.AddToWorklist(V.getNode()); 12764 SDValue InsElt = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, FloatVT, 12765 Vec, V, N->getOperand(2)); 12766 return DAG.getNode(ISD::BITCAST, dl, VT, InsElt); 12767 } 12768 12769 /// PerformVECTOR_SHUFFLECombine - Target-specific dag combine xforms for 12770 /// ISD::VECTOR_SHUFFLE. 12771 static SDValue PerformVECTOR_SHUFFLECombine(SDNode *N, SelectionDAG &DAG) { 12772 // The LLVM shufflevector instruction does not require the shuffle mask 12773 // length to match the operand vector length, but ISD::VECTOR_SHUFFLE does 12774 // have that requirement. When translating to ISD::VECTOR_SHUFFLE, if the 12775 // operands do not match the mask length, they are extended by concatenating 12776 // them with undef vectors. That is probably the right thing for other 12777 // targets, but for NEON it is better to concatenate two double-register 12778 // size vector operands into a single quad-register size vector. Do that 12779 // transformation here: 12780 // shuffle(concat(v1, undef), concat(v2, undef)) -> 12781 // shuffle(concat(v1, v2), undef) 12782 SDValue Op0 = N->getOperand(0); 12783 SDValue Op1 = N->getOperand(1); 12784 if (Op0.getOpcode() != ISD::CONCAT_VECTORS || 12785 Op1.getOpcode() != ISD::CONCAT_VECTORS || 12786 Op0.getNumOperands() != 2 || 12787 Op1.getNumOperands() != 2) 12788 return SDValue(); 12789 SDValue Concat0Op1 = Op0.getOperand(1); 12790 SDValue Concat1Op1 = Op1.getOperand(1); 12791 if (!Concat0Op1.isUndef() || !Concat1Op1.isUndef()) 12792 return SDValue(); 12793 // Skip the transformation if any of the types are illegal. 12794 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 12795 EVT VT = N->getValueType(0); 12796 if (!TLI.isTypeLegal(VT) || 12797 !TLI.isTypeLegal(Concat0Op1.getValueType()) || 12798 !TLI.isTypeLegal(Concat1Op1.getValueType())) 12799 return SDValue(); 12800 12801 SDValue NewConcat = DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, 12802 Op0.getOperand(0), Op1.getOperand(0)); 12803 // Translate the shuffle mask. 12804 SmallVector<int, 16> NewMask; 12805 unsigned NumElts = VT.getVectorNumElements(); 12806 unsigned HalfElts = NumElts/2; 12807 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N); 12808 for (unsigned n = 0; n < NumElts; ++n) { 12809 int MaskElt = SVN->getMaskElt(n); 12810 int NewElt = -1; 12811 if (MaskElt < (int)HalfElts) 12812 NewElt = MaskElt; 12813 else if (MaskElt >= (int)NumElts && MaskElt < (int)(NumElts + HalfElts)) 12814 NewElt = HalfElts + MaskElt - NumElts; 12815 NewMask.push_back(NewElt); 12816 } 12817 return DAG.getVectorShuffle(VT, SDLoc(N), NewConcat, 12818 DAG.getUNDEF(VT), NewMask); 12819 } 12820 12821 /// CombineBaseUpdate - Target-specific DAG combine function for VLDDUP, 12822 /// NEON load/store intrinsics, and generic vector load/stores, to merge 12823 /// base address updates. 12824 /// For generic load/stores, the memory type is assumed to be a vector. 12825 /// The caller is assumed to have checked legality. 12826 static SDValue CombineBaseUpdate(SDNode *N, 12827 TargetLowering::DAGCombinerInfo &DCI) { 12828 SelectionDAG &DAG = DCI.DAG; 12829 const bool isIntrinsic = (N->getOpcode() == ISD::INTRINSIC_VOID || 12830 N->getOpcode() == ISD::INTRINSIC_W_CHAIN); 12831 const bool isStore = N->getOpcode() == ISD::STORE; 12832 const unsigned AddrOpIdx = ((isIntrinsic || isStore) ? 2 : 1); 12833 SDValue Addr = N->getOperand(AddrOpIdx); 12834 MemSDNode *MemN = cast<MemSDNode>(N); 12835 SDLoc dl(N); 12836 12837 // Search for a use of the address operand that is an increment. 12838 for (SDNode::use_iterator UI = Addr.getNode()->use_begin(), 12839 UE = Addr.getNode()->use_end(); UI != UE; ++UI) { 12840 SDNode *User = *UI; 12841 if (User->getOpcode() != ISD::ADD || 12842 UI.getUse().getResNo() != Addr.getResNo()) 12843 continue; 12844 12845 // Check that the add is independent of the load/store. Otherwise, folding 12846 // it would create a cycle. We can avoid searching through Addr as it's a 12847 // predecessor to both. 12848 SmallPtrSet<const SDNode *, 32> Visited; 12849 SmallVector<const SDNode *, 16> Worklist; 12850 Visited.insert(Addr.getNode()); 12851 Worklist.push_back(N); 12852 Worklist.push_back(User); 12853 if (SDNode::hasPredecessorHelper(N, Visited, Worklist) || 12854 SDNode::hasPredecessorHelper(User, Visited, Worklist)) 12855 continue; 12856 12857 // Find the new opcode for the updating load/store. 12858 bool isLoadOp = true; 12859 bool isLaneOp = false; 12860 unsigned NewOpc = 0; 12861 unsigned NumVecs = 0; 12862 if (isIntrinsic) { 12863 unsigned IntNo = cast<ConstantSDNode>(N->getOperand(1))->getZExtValue(); 12864 switch (IntNo) { 12865 default: llvm_unreachable("unexpected intrinsic for Neon base update"); 12866 case Intrinsic::arm_neon_vld1: NewOpc = ARMISD::VLD1_UPD; 12867 NumVecs = 1; break; 12868 case Intrinsic::arm_neon_vld2: NewOpc = ARMISD::VLD2_UPD; 12869 NumVecs = 2; break; 12870 case Intrinsic::arm_neon_vld3: NewOpc = ARMISD::VLD3_UPD; 12871 NumVecs = 3; break; 12872 case Intrinsic::arm_neon_vld4: NewOpc = ARMISD::VLD4_UPD; 12873 NumVecs = 4; break; 12874 case Intrinsic::arm_neon_vld2dup: 12875 case Intrinsic::arm_neon_vld3dup: 12876 case Intrinsic::arm_neon_vld4dup: 12877 // TODO: Support updating VLDxDUP nodes. For now, we just skip 12878 // combining base updates for such intrinsics. 12879 continue; 12880 case Intrinsic::arm_neon_vld2lane: NewOpc = ARMISD::VLD2LN_UPD; 12881 NumVecs = 2; isLaneOp = true; break; 12882 case Intrinsic::arm_neon_vld3lane: NewOpc = ARMISD::VLD3LN_UPD; 12883 NumVecs = 3; isLaneOp = true; break; 12884 case Intrinsic::arm_neon_vld4lane: NewOpc = ARMISD::VLD4LN_UPD; 12885 NumVecs = 4; isLaneOp = true; break; 12886 case Intrinsic::arm_neon_vst1: NewOpc = ARMISD::VST1_UPD; 12887 NumVecs = 1; isLoadOp = false; break; 12888 case Intrinsic::arm_neon_vst2: NewOpc = ARMISD::VST2_UPD; 12889 NumVecs = 2; isLoadOp = false; break; 12890 case Intrinsic::arm_neon_vst3: NewOpc = ARMISD::VST3_UPD; 12891 NumVecs = 3; isLoadOp = false; break; 12892 case Intrinsic::arm_neon_vst4: NewOpc = ARMISD::VST4_UPD; 12893 NumVecs = 4; isLoadOp = false; break; 12894 case Intrinsic::arm_neon_vst2lane: NewOpc = ARMISD::VST2LN_UPD; 12895 NumVecs = 2; isLoadOp = false; isLaneOp = true; break; 12896 case Intrinsic::arm_neon_vst3lane: NewOpc = ARMISD::VST3LN_UPD; 12897 NumVecs = 3; isLoadOp = false; isLaneOp = true; break; 12898 case Intrinsic::arm_neon_vst4lane: NewOpc = ARMISD::VST4LN_UPD; 12899 NumVecs = 4; isLoadOp = false; isLaneOp = true; break; 12900 } 12901 } else { 12902 isLaneOp = true; 12903 switch (N->getOpcode()) { 12904 default: llvm_unreachable("unexpected opcode for Neon base update"); 12905 case ARMISD::VLD1DUP: NewOpc = ARMISD::VLD1DUP_UPD; NumVecs = 1; break; 12906 case ARMISD::VLD2DUP: NewOpc = ARMISD::VLD2DUP_UPD; NumVecs = 2; break; 12907 case ARMISD::VLD3DUP: NewOpc = ARMISD::VLD3DUP_UPD; NumVecs = 3; break; 12908 case ARMISD::VLD4DUP: NewOpc = ARMISD::VLD4DUP_UPD; NumVecs = 4; break; 12909 case ISD::LOAD: NewOpc = ARMISD::VLD1_UPD; 12910 NumVecs = 1; isLaneOp = false; break; 12911 case ISD::STORE: NewOpc = ARMISD::VST1_UPD; 12912 NumVecs = 1; isLaneOp = false; isLoadOp = false; break; 12913 } 12914 } 12915 12916 // Find the size of memory referenced by the load/store. 12917 EVT VecTy; 12918 if (isLoadOp) { 12919 VecTy = N->getValueType(0); 12920 } else if (isIntrinsic) { 12921 VecTy = N->getOperand(AddrOpIdx+1).getValueType(); 12922 } else { 12923 assert(isStore && "Node has to be a load, a store, or an intrinsic!"); 12924 VecTy = N->getOperand(1).getValueType(); 12925 } 12926 12927 unsigned NumBytes = NumVecs * VecTy.getSizeInBits() / 8; 12928 if (isLaneOp) 12929 NumBytes /= VecTy.getVectorNumElements(); 12930 12931 // If the increment is a constant, it must match the memory ref size. 12932 SDValue Inc = User->getOperand(User->getOperand(0) == Addr ? 1 : 0); 12933 ConstantSDNode *CInc = dyn_cast<ConstantSDNode>(Inc.getNode()); 12934 if (NumBytes >= 3 * 16 && (!CInc || CInc->getZExtValue() != NumBytes)) { 12935 // VLD3/4 and VST3/4 for 128-bit vectors are implemented with two 12936 // separate instructions that make it harder to use a non-constant update. 12937 continue; 12938 } 12939 12940 // OK, we found an ADD we can fold into the base update. 12941 // Now, create a _UPD node, taking care of not breaking alignment. 12942 12943 EVT AlignedVecTy = VecTy; 12944 unsigned Alignment = MemN->getAlignment(); 12945 12946 // If this is a less-than-standard-aligned load/store, change the type to 12947 // match the standard alignment. 12948 // The alignment is overlooked when selecting _UPD variants; and it's 12949 // easier to introduce bitcasts here than fix that. 12950 // There are 3 ways to get to this base-update combine: 12951 // - intrinsics: they are assumed to be properly aligned (to the standard 12952 // alignment of the memory type), so we don't need to do anything. 12953 // - ARMISD::VLDx nodes: they are only generated from the aforementioned 12954 // intrinsics, so, likewise, there's nothing to do. 12955 // - generic load/store instructions: the alignment is specified as an 12956 // explicit operand, rather than implicitly as the standard alignment 12957 // of the memory type (like the intrisics). We need to change the 12958 // memory type to match the explicit alignment. That way, we don't 12959 // generate non-standard-aligned ARMISD::VLDx nodes. 12960 if (isa<LSBaseSDNode>(N)) { 12961 if (Alignment == 0) 12962 Alignment = 1; 12963 if (Alignment < VecTy.getScalarSizeInBits() / 8) { 12964 MVT EltTy = MVT::getIntegerVT(Alignment * 8); 12965 assert(NumVecs == 1 && "Unexpected multi-element generic load/store."); 12966 assert(!isLaneOp && "Unexpected generic load/store lane."); 12967 unsigned NumElts = NumBytes / (EltTy.getSizeInBits() / 8); 12968 AlignedVecTy = MVT::getVectorVT(EltTy, NumElts); 12969 } 12970 // Don't set an explicit alignment on regular load/stores that we want 12971 // to transform to VLD/VST 1_UPD nodes. 12972 // This matches the behavior of regular load/stores, which only get an 12973 // explicit alignment if the MMO alignment is larger than the standard 12974 // alignment of the memory type. 12975 // Intrinsics, however, always get an explicit alignment, set to the 12976 // alignment of the MMO. 12977 Alignment = 1; 12978 } 12979 12980 // Create the new updating load/store node. 12981 // First, create an SDVTList for the new updating node's results. 12982 EVT Tys[6]; 12983 unsigned NumResultVecs = (isLoadOp ? NumVecs : 0); 12984 unsigned n; 12985 for (n = 0; n < NumResultVecs; ++n) 12986 Tys[n] = AlignedVecTy; 12987 Tys[n++] = MVT::i32; 12988 Tys[n] = MVT::Other; 12989 SDVTList SDTys = DAG.getVTList(makeArrayRef(Tys, NumResultVecs+2)); 12990 12991 // Then, gather the new node's operands. 12992 SmallVector<SDValue, 8> Ops; 12993 Ops.push_back(N->getOperand(0)); // incoming chain 12994 Ops.push_back(N->getOperand(AddrOpIdx)); 12995 Ops.push_back(Inc); 12996 12997 if (StoreSDNode *StN = dyn_cast<StoreSDNode>(N)) { 12998 // Try to match the intrinsic's signature 12999 Ops.push_back(StN->getValue()); 13000 } else { 13001 // Loads (and of course intrinsics) match the intrinsics' signature, 13002 // so just add all but the alignment operand. 13003 for (unsigned i = AddrOpIdx + 1; i < N->getNumOperands() - 1; ++i) 13004 Ops.push_back(N->getOperand(i)); 13005 } 13006 13007 // For all node types, the alignment operand is always the last one. 13008 Ops.push_back(DAG.getConstant(Alignment, dl, MVT::i32)); 13009 13010 // If this is a non-standard-aligned STORE, the penultimate operand is the 13011 // stored value. Bitcast it to the aligned type. 13012 if (AlignedVecTy != VecTy && N->getOpcode() == ISD::STORE) { 13013 SDValue &StVal = Ops[Ops.size()-2]; 13014 StVal = DAG.getNode(ISD::BITCAST, dl, AlignedVecTy, StVal); 13015 } 13016 13017 EVT LoadVT = isLaneOp ? VecTy.getVectorElementType() : AlignedVecTy; 13018 SDValue UpdN = DAG.getMemIntrinsicNode(NewOpc, dl, SDTys, Ops, LoadVT, 13019 MemN->getMemOperand()); 13020 13021 // Update the uses. 13022 SmallVector<SDValue, 5> NewResults; 13023 for (unsigned i = 0; i < NumResultVecs; ++i) 13024 NewResults.push_back(SDValue(UpdN.getNode(), i)); 13025 13026 // If this is an non-standard-aligned LOAD, the first result is the loaded 13027 // value. Bitcast it to the expected result type. 13028 if (AlignedVecTy != VecTy && N->getOpcode() == ISD::LOAD) { 13029 SDValue &LdVal = NewResults[0]; 13030 LdVal = DAG.getNode(ISD::BITCAST, dl, VecTy, LdVal); 13031 } 13032 13033 NewResults.push_back(SDValue(UpdN.getNode(), NumResultVecs+1)); // chain 13034 DCI.CombineTo(N, NewResults); 13035 DCI.CombineTo(User, SDValue(UpdN.getNode(), NumResultVecs)); 13036 13037 break; 13038 } 13039 return SDValue(); 13040 } 13041 13042 static SDValue PerformVLDCombine(SDNode *N, 13043 TargetLowering::DAGCombinerInfo &DCI) { 13044 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer()) 13045 return SDValue(); 13046 13047 return CombineBaseUpdate(N, DCI); 13048 } 13049 13050 /// CombineVLDDUP - For a VDUPLANE node N, check if its source operand is a 13051 /// vldN-lane (N > 1) intrinsic, and if all the other uses of that intrinsic 13052 /// are also VDUPLANEs. If so, combine them to a vldN-dup operation and 13053 /// return true. 13054 static bool CombineVLDDUP(SDNode *N, TargetLowering::DAGCombinerInfo &DCI) { 13055 SelectionDAG &DAG = DCI.DAG; 13056 EVT VT = N->getValueType(0); 13057 // vldN-dup instructions only support 64-bit vectors for N > 1. 13058 if (!VT.is64BitVector()) 13059 return false; 13060 13061 // Check if the VDUPLANE operand is a vldN-dup intrinsic. 13062 SDNode *VLD = N->getOperand(0).getNode(); 13063 if (VLD->getOpcode() != ISD::INTRINSIC_W_CHAIN) 13064 return false; 13065 unsigned NumVecs = 0; 13066 unsigned NewOpc = 0; 13067 unsigned IntNo = cast<ConstantSDNode>(VLD->getOperand(1))->getZExtValue(); 13068 if (IntNo == Intrinsic::arm_neon_vld2lane) { 13069 NumVecs = 2; 13070 NewOpc = ARMISD::VLD2DUP; 13071 } else if (IntNo == Intrinsic::arm_neon_vld3lane) { 13072 NumVecs = 3; 13073 NewOpc = ARMISD::VLD3DUP; 13074 } else if (IntNo == Intrinsic::arm_neon_vld4lane) { 13075 NumVecs = 4; 13076 NewOpc = ARMISD::VLD4DUP; 13077 } else { 13078 return false; 13079 } 13080 13081 // First check that all the vldN-lane uses are VDUPLANEs and that the lane 13082 // numbers match the load. 13083 unsigned VLDLaneNo = 13084 cast<ConstantSDNode>(VLD->getOperand(NumVecs+3))->getZExtValue(); 13085 for (SDNode::use_iterator UI = VLD->use_begin(), UE = VLD->use_end(); 13086 UI != UE; ++UI) { 13087 // Ignore uses of the chain result. 13088 if (UI.getUse().getResNo() == NumVecs) 13089 continue; 13090 SDNode *User = *UI; 13091 if (User->getOpcode() != ARMISD::VDUPLANE || 13092 VLDLaneNo != cast<ConstantSDNode>(User->getOperand(1))->getZExtValue()) 13093 return false; 13094 } 13095 13096 // Create the vldN-dup node. 13097 EVT Tys[5]; 13098 unsigned n; 13099 for (n = 0; n < NumVecs; ++n) 13100 Tys[n] = VT; 13101 Tys[n] = MVT::Other; 13102 SDVTList SDTys = DAG.getVTList(makeArrayRef(Tys, NumVecs+1)); 13103 SDValue Ops[] = { VLD->getOperand(0), VLD->getOperand(2) }; 13104 MemIntrinsicSDNode *VLDMemInt = cast<MemIntrinsicSDNode>(VLD); 13105 SDValue VLDDup = DAG.getMemIntrinsicNode(NewOpc, SDLoc(VLD), SDTys, 13106 Ops, VLDMemInt->getMemoryVT(), 13107 VLDMemInt->getMemOperand()); 13108 13109 // Update the uses. 13110 for (SDNode::use_iterator UI = VLD->use_begin(), UE = VLD->use_end(); 13111 UI != UE; ++UI) { 13112 unsigned ResNo = UI.getUse().getResNo(); 13113 // Ignore uses of the chain result. 13114 if (ResNo == NumVecs) 13115 continue; 13116 SDNode *User = *UI; 13117 DCI.CombineTo(User, SDValue(VLDDup.getNode(), ResNo)); 13118 } 13119 13120 // Now the vldN-lane intrinsic is dead except for its chain result. 13121 // Update uses of the chain. 13122 std::vector<SDValue> VLDDupResults; 13123 for (unsigned n = 0; n < NumVecs; ++n) 13124 VLDDupResults.push_back(SDValue(VLDDup.getNode(), n)); 13125 VLDDupResults.push_back(SDValue(VLDDup.getNode(), NumVecs)); 13126 DCI.CombineTo(VLD, VLDDupResults); 13127 13128 return true; 13129 } 13130 13131 /// PerformVDUPLANECombine - Target-specific dag combine xforms for 13132 /// ARMISD::VDUPLANE. 13133 static SDValue PerformVDUPLANECombine(SDNode *N, 13134 TargetLowering::DAGCombinerInfo &DCI) { 13135 SDValue Op = N->getOperand(0); 13136 13137 // If the source is a vldN-lane (N > 1) intrinsic, and all the other uses 13138 // of that intrinsic are also VDUPLANEs, combine them to a vldN-dup operation. 13139 if (CombineVLDDUP(N, DCI)) 13140 return SDValue(N, 0); 13141 13142 // If the source is already a VMOVIMM or VMVNIMM splat, the VDUPLANE is 13143 // redundant. Ignore bit_converts for now; element sizes are checked below. 13144 while (Op.getOpcode() == ISD::BITCAST) 13145 Op = Op.getOperand(0); 13146 if (Op.getOpcode() != ARMISD::VMOVIMM && Op.getOpcode() != ARMISD::VMVNIMM) 13147 return SDValue(); 13148 13149 // Make sure the VMOV element size is not bigger than the VDUPLANE elements. 13150 unsigned EltSize = Op.getScalarValueSizeInBits(); 13151 // The canonical VMOV for a zero vector uses a 32-bit element size. 13152 unsigned Imm = cast<ConstantSDNode>(Op.getOperand(0))->getZExtValue(); 13153 unsigned EltBits; 13154 if (ARM_AM::decodeVMOVModImm(Imm, EltBits) == 0) 13155 EltSize = 8; 13156 EVT VT = N->getValueType(0); 13157 if (EltSize > VT.getScalarSizeInBits()) 13158 return SDValue(); 13159 13160 return DCI.DAG.getNode(ISD::BITCAST, SDLoc(N), VT, Op); 13161 } 13162 13163 /// PerformVDUPCombine - Target-specific dag combine xforms for ARMISD::VDUP. 13164 static SDValue PerformVDUPCombine(SDNode *N, 13165 TargetLowering::DAGCombinerInfo &DCI, 13166 const ARMSubtarget *Subtarget) { 13167 SelectionDAG &DAG = DCI.DAG; 13168 SDValue Op = N->getOperand(0); 13169 13170 if (!Subtarget->hasNEON()) 13171 return SDValue(); 13172 13173 // Match VDUP(LOAD) -> VLD1DUP. 13174 // We match this pattern here rather than waiting for isel because the 13175 // transform is only legal for unindexed loads. 13176 LoadSDNode *LD = dyn_cast<LoadSDNode>(Op.getNode()); 13177 if (LD && Op.hasOneUse() && LD->isUnindexed() && 13178 LD->getMemoryVT() == N->getValueType(0).getVectorElementType()) { 13179 SDValue Ops[] = { LD->getOperand(0), LD->getOperand(1), 13180 DAG.getConstant(LD->getAlignment(), SDLoc(N), MVT::i32) }; 13181 SDVTList SDTys = DAG.getVTList(N->getValueType(0), MVT::Other); 13182 SDValue VLDDup = DAG.getMemIntrinsicNode(ARMISD::VLD1DUP, SDLoc(N), SDTys, 13183 Ops, LD->getMemoryVT(), 13184 LD->getMemOperand()); 13185 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), VLDDup.getValue(1)); 13186 return VLDDup; 13187 } 13188 13189 return SDValue(); 13190 } 13191 13192 static SDValue PerformLOADCombine(SDNode *N, 13193 TargetLowering::DAGCombinerInfo &DCI) { 13194 EVT VT = N->getValueType(0); 13195 13196 // If this is a legal vector load, try to combine it into a VLD1_UPD. 13197 if (ISD::isNormalLoad(N) && VT.isVector() && 13198 DCI.DAG.getTargetLoweringInfo().isTypeLegal(VT)) 13199 return CombineBaseUpdate(N, DCI); 13200 13201 return SDValue(); 13202 } 13203 13204 // Optimize trunc store (of multiple scalars) to shuffle and store. First, 13205 // pack all of the elements in one place. Next, store to memory in fewer 13206 // chunks. 13207 static SDValue PerformTruncatingStoreCombine(StoreSDNode *St, 13208 SelectionDAG &DAG) { 13209 SDValue StVal = St->getValue(); 13210 EVT VT = StVal.getValueType(); 13211 if (!St->isTruncatingStore() || !VT.isVector()) 13212 return SDValue(); 13213 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 13214 EVT StVT = St->getMemoryVT(); 13215 unsigned NumElems = VT.getVectorNumElements(); 13216 assert(StVT != VT && "Cannot truncate to the same type"); 13217 unsigned FromEltSz = VT.getScalarSizeInBits(); 13218 unsigned ToEltSz = StVT.getScalarSizeInBits(); 13219 13220 // From, To sizes and ElemCount must be pow of two 13221 if (!isPowerOf2_32(NumElems * FromEltSz * ToEltSz)) 13222 return SDValue(); 13223 13224 // We are going to use the original vector elt for storing. 13225 // Accumulated smaller vector elements must be a multiple of the store size. 13226 if (0 != (NumElems * FromEltSz) % ToEltSz) 13227 return SDValue(); 13228 13229 unsigned SizeRatio = FromEltSz / ToEltSz; 13230 assert(SizeRatio * NumElems * ToEltSz == VT.getSizeInBits()); 13231 13232 // Create a type on which we perform the shuffle. 13233 EVT WideVecVT = EVT::getVectorVT(*DAG.getContext(), StVT.getScalarType(), 13234 NumElems * SizeRatio); 13235 assert(WideVecVT.getSizeInBits() == VT.getSizeInBits()); 13236 13237 SDLoc DL(St); 13238 SDValue WideVec = DAG.getNode(ISD::BITCAST, DL, WideVecVT, StVal); 13239 SmallVector<int, 8> ShuffleVec(NumElems * SizeRatio, -1); 13240 for (unsigned i = 0; i < NumElems; ++i) 13241 ShuffleVec[i] = DAG.getDataLayout().isBigEndian() ? (i + 1) * SizeRatio - 1 13242 : i * SizeRatio; 13243 13244 // Can't shuffle using an illegal type. 13245 if (!TLI.isTypeLegal(WideVecVT)) 13246 return SDValue(); 13247 13248 SDValue Shuff = DAG.getVectorShuffle( 13249 WideVecVT, DL, WideVec, DAG.getUNDEF(WideVec.getValueType()), ShuffleVec); 13250 // At this point all of the data is stored at the bottom of the 13251 // register. We now need to save it to mem. 13252 13253 // Find the largest store unit 13254 MVT StoreType = MVT::i8; 13255 for (MVT Tp : MVT::integer_valuetypes()) { 13256 if (TLI.isTypeLegal(Tp) && Tp.getSizeInBits() <= NumElems * ToEltSz) 13257 StoreType = Tp; 13258 } 13259 // Didn't find a legal store type. 13260 if (!TLI.isTypeLegal(StoreType)) 13261 return SDValue(); 13262 13263 // Bitcast the original vector into a vector of store-size units 13264 EVT StoreVecVT = 13265 EVT::getVectorVT(*DAG.getContext(), StoreType, 13266 VT.getSizeInBits() / EVT(StoreType).getSizeInBits()); 13267 assert(StoreVecVT.getSizeInBits() == VT.getSizeInBits()); 13268 SDValue ShuffWide = DAG.getNode(ISD::BITCAST, DL, StoreVecVT, Shuff); 13269 SmallVector<SDValue, 8> Chains; 13270 SDValue Increment = DAG.getConstant(StoreType.getSizeInBits() / 8, DL, 13271 TLI.getPointerTy(DAG.getDataLayout())); 13272 SDValue BasePtr = St->getBasePtr(); 13273 13274 // Perform one or more big stores into memory. 13275 unsigned E = (ToEltSz * NumElems) / StoreType.getSizeInBits(); 13276 for (unsigned I = 0; I < E; I++) { 13277 SDValue SubVec = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, StoreType, 13278 ShuffWide, DAG.getIntPtrConstant(I, DL)); 13279 SDValue Ch = 13280 DAG.getStore(St->getChain(), DL, SubVec, BasePtr, St->getPointerInfo(), 13281 St->getAlignment(), St->getMemOperand()->getFlags()); 13282 BasePtr = 13283 DAG.getNode(ISD::ADD, DL, BasePtr.getValueType(), BasePtr, Increment); 13284 Chains.push_back(Ch); 13285 } 13286 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains); 13287 } 13288 13289 // Try taking a single vector store from an truncate (which would otherwise turn 13290 // into an expensive buildvector) and splitting it into a series of narrowing 13291 // stores. 13292 static SDValue PerformSplittingToNarrowingStores(StoreSDNode *St, 13293 SelectionDAG &DAG) { 13294 if (!St->isSimple() || St->isTruncatingStore() || !St->isUnindexed()) 13295 return SDValue(); 13296 SDValue Trunc = St->getValue(); 13297 if (Trunc->getOpcode() != ISD::TRUNCATE) 13298 return SDValue(); 13299 EVT FromVT = Trunc->getOperand(0).getValueType(); 13300 EVT ToVT = Trunc.getValueType(); 13301 if (!ToVT.isVector()) 13302 return SDValue(); 13303 assert(FromVT.getVectorNumElements() == ToVT.getVectorNumElements()); 13304 EVT ToEltVT = ToVT.getVectorElementType(); 13305 EVT FromEltVT = FromVT.getVectorElementType(); 13306 13307 unsigned NumElements = 0; 13308 if (FromEltVT == MVT::i32 && (ToEltVT == MVT::i16 || ToEltVT == MVT::i8)) 13309 NumElements = 4; 13310 if (FromEltVT == MVT::i16 && ToEltVT == MVT::i8) 13311 NumElements = 8; 13312 if (NumElements == 0 || FromVT.getVectorNumElements() == NumElements || 13313 FromVT.getVectorNumElements() % NumElements != 0) 13314 return SDValue(); 13315 13316 SDLoc DL(St); 13317 // Details about the old store 13318 SDValue Ch = St->getChain(); 13319 SDValue BasePtr = St->getBasePtr(); 13320 unsigned Alignment = St->getOriginalAlignment(); 13321 MachineMemOperand::Flags MMOFlags = St->getMemOperand()->getFlags(); 13322 AAMDNodes AAInfo = St->getAAInfo(); 13323 13324 EVT NewFromVT = EVT::getVectorVT(*DAG.getContext(), FromEltVT, NumElements); 13325 EVT NewToVT = EVT::getVectorVT(*DAG.getContext(), ToEltVT, NumElements); 13326 13327 SmallVector<SDValue, 4> Stores; 13328 for (unsigned i = 0; i < FromVT.getVectorNumElements() / NumElements; i++) { 13329 unsigned NewOffset = i * NumElements * ToEltVT.getSizeInBits() / 8; 13330 SDValue NewPtr = DAG.getObjectPtrOffset(DL, BasePtr, NewOffset); 13331 13332 SDValue Extract = 13333 DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NewFromVT, Trunc.getOperand(0), 13334 DAG.getConstant(i * NumElements, DL, MVT::i32)); 13335 SDValue Store = DAG.getTruncStore( 13336 Ch, DL, Extract, NewPtr, St->getPointerInfo().getWithOffset(NewOffset), 13337 NewToVT, Alignment, MMOFlags, AAInfo); 13338 Stores.push_back(Store); 13339 } 13340 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Stores); 13341 } 13342 13343 /// PerformSTORECombine - Target-specific dag combine xforms for 13344 /// ISD::STORE. 13345 static SDValue PerformSTORECombine(SDNode *N, 13346 TargetLowering::DAGCombinerInfo &DCI, 13347 const ARMSubtarget *Subtarget) { 13348 StoreSDNode *St = cast<StoreSDNode>(N); 13349 if (St->isVolatile()) 13350 return SDValue(); 13351 SDValue StVal = St->getValue(); 13352 EVT VT = StVal.getValueType(); 13353 13354 if (Subtarget->hasNEON()) 13355 if (SDValue Store = PerformTruncatingStoreCombine(St, DCI.DAG)) 13356 return Store; 13357 13358 if (Subtarget->hasMVEIntegerOps()) 13359 if (SDValue NewToken = PerformSplittingToNarrowingStores(St, DCI.DAG)) 13360 return NewToken; 13361 13362 if (!ISD::isNormalStore(St)) 13363 return SDValue(); 13364 13365 // Split a store of a VMOVDRR into two integer stores to avoid mixing NEON and 13366 // ARM stores of arguments in the same cache line. 13367 if (StVal.getNode()->getOpcode() == ARMISD::VMOVDRR && 13368 StVal.getNode()->hasOneUse()) { 13369 SelectionDAG &DAG = DCI.DAG; 13370 bool isBigEndian = DAG.getDataLayout().isBigEndian(); 13371 SDLoc DL(St); 13372 SDValue BasePtr = St->getBasePtr(); 13373 SDValue NewST1 = DAG.getStore( 13374 St->getChain(), DL, StVal.getNode()->getOperand(isBigEndian ? 1 : 0), 13375 BasePtr, St->getPointerInfo(), St->getAlignment(), 13376 St->getMemOperand()->getFlags()); 13377 13378 SDValue OffsetPtr = DAG.getNode(ISD::ADD, DL, MVT::i32, BasePtr, 13379 DAG.getConstant(4, DL, MVT::i32)); 13380 return DAG.getStore(NewST1.getValue(0), DL, 13381 StVal.getNode()->getOperand(isBigEndian ? 0 : 1), 13382 OffsetPtr, St->getPointerInfo(), 13383 std::min(4U, St->getAlignment() / 2), 13384 St->getMemOperand()->getFlags()); 13385 } 13386 13387 if (StVal.getValueType() == MVT::i64 && 13388 StVal.getNode()->getOpcode() == ISD::EXTRACT_VECTOR_ELT) { 13389 13390 // Bitcast an i64 store extracted from a vector to f64. 13391 // Otherwise, the i64 value will be legalized to a pair of i32 values. 13392 SelectionDAG &DAG = DCI.DAG; 13393 SDLoc dl(StVal); 13394 SDValue IntVec = StVal.getOperand(0); 13395 EVT FloatVT = EVT::getVectorVT(*DAG.getContext(), MVT::f64, 13396 IntVec.getValueType().getVectorNumElements()); 13397 SDValue Vec = DAG.getNode(ISD::BITCAST, dl, FloatVT, IntVec); 13398 SDValue ExtElt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, 13399 Vec, StVal.getOperand(1)); 13400 dl = SDLoc(N); 13401 SDValue V = DAG.getNode(ISD::BITCAST, dl, MVT::i64, ExtElt); 13402 // Make the DAGCombiner fold the bitcasts. 13403 DCI.AddToWorklist(Vec.getNode()); 13404 DCI.AddToWorklist(ExtElt.getNode()); 13405 DCI.AddToWorklist(V.getNode()); 13406 return DAG.getStore(St->getChain(), dl, V, St->getBasePtr(), 13407 St->getPointerInfo(), St->getAlignment(), 13408 St->getMemOperand()->getFlags(), St->getAAInfo()); 13409 } 13410 13411 // If this is a legal vector store, try to combine it into a VST1_UPD. 13412 if (Subtarget->hasNEON() && ISD::isNormalStore(N) && VT.isVector() && 13413 DCI.DAG.getTargetLoweringInfo().isTypeLegal(VT)) 13414 return CombineBaseUpdate(N, DCI); 13415 13416 return SDValue(); 13417 } 13418 13419 /// PerformVCVTCombine - VCVT (floating-point to fixed-point, Advanced SIMD) 13420 /// can replace combinations of VMUL and VCVT (floating-point to integer) 13421 /// when the VMUL has a constant operand that is a power of 2. 13422 /// 13423 /// Example (assume d17 = <float 8.000000e+00, float 8.000000e+00>): 13424 /// vmul.f32 d16, d17, d16 13425 /// vcvt.s32.f32 d16, d16 13426 /// becomes: 13427 /// vcvt.s32.f32 d16, d16, #3 13428 static SDValue PerformVCVTCombine(SDNode *N, SelectionDAG &DAG, 13429 const ARMSubtarget *Subtarget) { 13430 if (!Subtarget->hasNEON()) 13431 return SDValue(); 13432 13433 SDValue Op = N->getOperand(0); 13434 if (!Op.getValueType().isVector() || !Op.getValueType().isSimple() || 13435 Op.getOpcode() != ISD::FMUL) 13436 return SDValue(); 13437 13438 SDValue ConstVec = Op->getOperand(1); 13439 if (!isa<BuildVectorSDNode>(ConstVec)) 13440 return SDValue(); 13441 13442 MVT FloatTy = Op.getSimpleValueType().getVectorElementType(); 13443 uint32_t FloatBits = FloatTy.getSizeInBits(); 13444 MVT IntTy = N->getSimpleValueType(0).getVectorElementType(); 13445 uint32_t IntBits = IntTy.getSizeInBits(); 13446 unsigned NumLanes = Op.getValueType().getVectorNumElements(); 13447 if (FloatBits != 32 || IntBits > 32 || (NumLanes != 4 && NumLanes != 2)) { 13448 // These instructions only exist converting from f32 to i32. We can handle 13449 // smaller integers by generating an extra truncate, but larger ones would 13450 // be lossy. We also can't handle anything other than 2 or 4 lanes, since 13451 // these intructions only support v2i32/v4i32 types. 13452 return SDValue(); 13453 } 13454 13455 BitVector UndefElements; 13456 BuildVectorSDNode *BV = cast<BuildVectorSDNode>(ConstVec); 13457 int32_t C = BV->getConstantFPSplatPow2ToLog2Int(&UndefElements, 33); 13458 if (C == -1 || C == 0 || C > 32) 13459 return SDValue(); 13460 13461 SDLoc dl(N); 13462 bool isSigned = N->getOpcode() == ISD::FP_TO_SINT; 13463 unsigned IntrinsicOpcode = isSigned ? Intrinsic::arm_neon_vcvtfp2fxs : 13464 Intrinsic::arm_neon_vcvtfp2fxu; 13465 SDValue FixConv = DAG.getNode( 13466 ISD::INTRINSIC_WO_CHAIN, dl, NumLanes == 2 ? MVT::v2i32 : MVT::v4i32, 13467 DAG.getConstant(IntrinsicOpcode, dl, MVT::i32), Op->getOperand(0), 13468 DAG.getConstant(C, dl, MVT::i32)); 13469 13470 if (IntBits < FloatBits) 13471 FixConv = DAG.getNode(ISD::TRUNCATE, dl, N->getValueType(0), FixConv); 13472 13473 return FixConv; 13474 } 13475 13476 /// PerformVDIVCombine - VCVT (fixed-point to floating-point, Advanced SIMD) 13477 /// can replace combinations of VCVT (integer to floating-point) and VDIV 13478 /// when the VDIV has a constant operand that is a power of 2. 13479 /// 13480 /// Example (assume d17 = <float 8.000000e+00, float 8.000000e+00>): 13481 /// vcvt.f32.s32 d16, d16 13482 /// vdiv.f32 d16, d17, d16 13483 /// becomes: 13484 /// vcvt.f32.s32 d16, d16, #3 13485 static SDValue PerformVDIVCombine(SDNode *N, SelectionDAG &DAG, 13486 const ARMSubtarget *Subtarget) { 13487 if (!Subtarget->hasNEON()) 13488 return SDValue(); 13489 13490 SDValue Op = N->getOperand(0); 13491 unsigned OpOpcode = Op.getNode()->getOpcode(); 13492 if (!N->getValueType(0).isVector() || !N->getValueType(0).isSimple() || 13493 (OpOpcode != ISD::SINT_TO_FP && OpOpcode != ISD::UINT_TO_FP)) 13494 return SDValue(); 13495 13496 SDValue ConstVec = N->getOperand(1); 13497 if (!isa<BuildVectorSDNode>(ConstVec)) 13498 return SDValue(); 13499 13500 MVT FloatTy = N->getSimpleValueType(0).getVectorElementType(); 13501 uint32_t FloatBits = FloatTy.getSizeInBits(); 13502 MVT IntTy = Op.getOperand(0).getSimpleValueType().getVectorElementType(); 13503 uint32_t IntBits = IntTy.getSizeInBits(); 13504 unsigned NumLanes = Op.getValueType().getVectorNumElements(); 13505 if (FloatBits != 32 || IntBits > 32 || (NumLanes != 4 && NumLanes != 2)) { 13506 // These instructions only exist converting from i32 to f32. We can handle 13507 // smaller integers by generating an extra extend, but larger ones would 13508 // be lossy. We also can't handle anything other than 2 or 4 lanes, since 13509 // these intructions only support v2i32/v4i32 types. 13510 return SDValue(); 13511 } 13512 13513 BitVector UndefElements; 13514 BuildVectorSDNode *BV = cast<BuildVectorSDNode>(ConstVec); 13515 int32_t C = BV->getConstantFPSplatPow2ToLog2Int(&UndefElements, 33); 13516 if (C == -1 || C == 0 || C > 32) 13517 return SDValue(); 13518 13519 SDLoc dl(N); 13520 bool isSigned = OpOpcode == ISD::SINT_TO_FP; 13521 SDValue ConvInput = Op.getOperand(0); 13522 if (IntBits < FloatBits) 13523 ConvInput = DAG.getNode(isSigned ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND, 13524 dl, NumLanes == 2 ? MVT::v2i32 : MVT::v4i32, 13525 ConvInput); 13526 13527 unsigned IntrinsicOpcode = isSigned ? Intrinsic::arm_neon_vcvtfxs2fp : 13528 Intrinsic::arm_neon_vcvtfxu2fp; 13529 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, 13530 Op.getValueType(), 13531 DAG.getConstant(IntrinsicOpcode, dl, MVT::i32), 13532 ConvInput, DAG.getConstant(C, dl, MVT::i32)); 13533 } 13534 13535 /// PerformIntrinsicCombine - ARM-specific DAG combining for intrinsics. 13536 static SDValue PerformIntrinsicCombine(SDNode *N, SelectionDAG &DAG) { 13537 unsigned IntNo = cast<ConstantSDNode>(N->getOperand(0))->getZExtValue(); 13538 switch (IntNo) { 13539 default: 13540 // Don't do anything for most intrinsics. 13541 break; 13542 13543 // Vector shifts: check for immediate versions and lower them. 13544 // Note: This is done during DAG combining instead of DAG legalizing because 13545 // the build_vectors for 64-bit vector element shift counts are generally 13546 // not legal, and it is hard to see their values after they get legalized to 13547 // loads from a constant pool. 13548 case Intrinsic::arm_neon_vshifts: 13549 case Intrinsic::arm_neon_vshiftu: 13550 case Intrinsic::arm_neon_vrshifts: 13551 case Intrinsic::arm_neon_vrshiftu: 13552 case Intrinsic::arm_neon_vrshiftn: 13553 case Intrinsic::arm_neon_vqshifts: 13554 case Intrinsic::arm_neon_vqshiftu: 13555 case Intrinsic::arm_neon_vqshiftsu: 13556 case Intrinsic::arm_neon_vqshiftns: 13557 case Intrinsic::arm_neon_vqshiftnu: 13558 case Intrinsic::arm_neon_vqshiftnsu: 13559 case Intrinsic::arm_neon_vqrshiftns: 13560 case Intrinsic::arm_neon_vqrshiftnu: 13561 case Intrinsic::arm_neon_vqrshiftnsu: { 13562 EVT VT = N->getOperand(1).getValueType(); 13563 int64_t Cnt; 13564 unsigned VShiftOpc = 0; 13565 13566 switch (IntNo) { 13567 case Intrinsic::arm_neon_vshifts: 13568 case Intrinsic::arm_neon_vshiftu: 13569 if (isVShiftLImm(N->getOperand(2), VT, false, Cnt)) { 13570 VShiftOpc = ARMISD::VSHLIMM; 13571 break; 13572 } 13573 if (isVShiftRImm(N->getOperand(2), VT, false, true, Cnt)) { 13574 VShiftOpc = (IntNo == Intrinsic::arm_neon_vshifts ? ARMISD::VSHRsIMM 13575 : ARMISD::VSHRuIMM); 13576 break; 13577 } 13578 return SDValue(); 13579 13580 case Intrinsic::arm_neon_vrshifts: 13581 case Intrinsic::arm_neon_vrshiftu: 13582 if (isVShiftRImm(N->getOperand(2), VT, false, true, Cnt)) 13583 break; 13584 return SDValue(); 13585 13586 case Intrinsic::arm_neon_vqshifts: 13587 case Intrinsic::arm_neon_vqshiftu: 13588 if (isVShiftLImm(N->getOperand(2), VT, false, Cnt)) 13589 break; 13590 return SDValue(); 13591 13592 case Intrinsic::arm_neon_vqshiftsu: 13593 if (isVShiftLImm(N->getOperand(2), VT, false, Cnt)) 13594 break; 13595 llvm_unreachable("invalid shift count for vqshlu intrinsic"); 13596 13597 case Intrinsic::arm_neon_vrshiftn: 13598 case Intrinsic::arm_neon_vqshiftns: 13599 case Intrinsic::arm_neon_vqshiftnu: 13600 case Intrinsic::arm_neon_vqshiftnsu: 13601 case Intrinsic::arm_neon_vqrshiftns: 13602 case Intrinsic::arm_neon_vqrshiftnu: 13603 case Intrinsic::arm_neon_vqrshiftnsu: 13604 // Narrowing shifts require an immediate right shift. 13605 if (isVShiftRImm(N->getOperand(2), VT, true, true, Cnt)) 13606 break; 13607 llvm_unreachable("invalid shift count for narrowing vector shift " 13608 "intrinsic"); 13609 13610 default: 13611 llvm_unreachable("unhandled vector shift"); 13612 } 13613 13614 switch (IntNo) { 13615 case Intrinsic::arm_neon_vshifts: 13616 case Intrinsic::arm_neon_vshiftu: 13617 // Opcode already set above. 13618 break; 13619 case Intrinsic::arm_neon_vrshifts: 13620 VShiftOpc = ARMISD::VRSHRsIMM; 13621 break; 13622 case Intrinsic::arm_neon_vrshiftu: 13623 VShiftOpc = ARMISD::VRSHRuIMM; 13624 break; 13625 case Intrinsic::arm_neon_vrshiftn: 13626 VShiftOpc = ARMISD::VRSHRNIMM; 13627 break; 13628 case Intrinsic::arm_neon_vqshifts: 13629 VShiftOpc = ARMISD::VQSHLsIMM; 13630 break; 13631 case Intrinsic::arm_neon_vqshiftu: 13632 VShiftOpc = ARMISD::VQSHLuIMM; 13633 break; 13634 case Intrinsic::arm_neon_vqshiftsu: 13635 VShiftOpc = ARMISD::VQSHLsuIMM; 13636 break; 13637 case Intrinsic::arm_neon_vqshiftns: 13638 VShiftOpc = ARMISD::VQSHRNsIMM; 13639 break; 13640 case Intrinsic::arm_neon_vqshiftnu: 13641 VShiftOpc = ARMISD::VQSHRNuIMM; 13642 break; 13643 case Intrinsic::arm_neon_vqshiftnsu: 13644 VShiftOpc = ARMISD::VQSHRNsuIMM; 13645 break; 13646 case Intrinsic::arm_neon_vqrshiftns: 13647 VShiftOpc = ARMISD::VQRSHRNsIMM; 13648 break; 13649 case Intrinsic::arm_neon_vqrshiftnu: 13650 VShiftOpc = ARMISD::VQRSHRNuIMM; 13651 break; 13652 case Intrinsic::arm_neon_vqrshiftnsu: 13653 VShiftOpc = ARMISD::VQRSHRNsuIMM; 13654 break; 13655 } 13656 13657 SDLoc dl(N); 13658 return DAG.getNode(VShiftOpc, dl, N->getValueType(0), 13659 N->getOperand(1), DAG.getConstant(Cnt, dl, MVT::i32)); 13660 } 13661 13662 case Intrinsic::arm_neon_vshiftins: { 13663 EVT VT = N->getOperand(1).getValueType(); 13664 int64_t Cnt; 13665 unsigned VShiftOpc = 0; 13666 13667 if (isVShiftLImm(N->getOperand(3), VT, false, Cnt)) 13668 VShiftOpc = ARMISD::VSLIIMM; 13669 else if (isVShiftRImm(N->getOperand(3), VT, false, true, Cnt)) 13670 VShiftOpc = ARMISD::VSRIIMM; 13671 else { 13672 llvm_unreachable("invalid shift count for vsli/vsri intrinsic"); 13673 } 13674 13675 SDLoc dl(N); 13676 return DAG.getNode(VShiftOpc, dl, N->getValueType(0), 13677 N->getOperand(1), N->getOperand(2), 13678 DAG.getConstant(Cnt, dl, MVT::i32)); 13679 } 13680 13681 case Intrinsic::arm_neon_vqrshifts: 13682 case Intrinsic::arm_neon_vqrshiftu: 13683 // No immediate versions of these to check for. 13684 break; 13685 } 13686 13687 return SDValue(); 13688 } 13689 13690 /// PerformShiftCombine - Checks for immediate versions of vector shifts and 13691 /// lowers them. As with the vector shift intrinsics, this is done during DAG 13692 /// combining instead of DAG legalizing because the build_vectors for 64-bit 13693 /// vector element shift counts are generally not legal, and it is hard to see 13694 /// their values after they get legalized to loads from a constant pool. 13695 static SDValue PerformShiftCombine(SDNode *N, 13696 TargetLowering::DAGCombinerInfo &DCI, 13697 const ARMSubtarget *ST) { 13698 SelectionDAG &DAG = DCI.DAG; 13699 EVT VT = N->getValueType(0); 13700 if (N->getOpcode() == ISD::SRL && VT == MVT::i32 && ST->hasV6Ops()) { 13701 // Canonicalize (srl (bswap x), 16) to (rotr (bswap x), 16) if the high 13702 // 16-bits of x is zero. This optimizes rev + lsr 16 to rev16. 13703 SDValue N1 = N->getOperand(1); 13704 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(N1)) { 13705 SDValue N0 = N->getOperand(0); 13706 if (C->getZExtValue() == 16 && N0.getOpcode() == ISD::BSWAP && 13707 DAG.MaskedValueIsZero(N0.getOperand(0), 13708 APInt::getHighBitsSet(32, 16))) 13709 return DAG.getNode(ISD::ROTR, SDLoc(N), VT, N0, N1); 13710 } 13711 } 13712 13713 if (ST->isThumb1Only() && N->getOpcode() == ISD::SHL && VT == MVT::i32 && 13714 N->getOperand(0)->getOpcode() == ISD::AND && 13715 N->getOperand(0)->hasOneUse()) { 13716 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer()) 13717 return SDValue(); 13718 // Look for the pattern (shl (and x, AndMask), ShiftAmt). This doesn't 13719 // usually show up because instcombine prefers to canonicalize it to 13720 // (and (shl x, ShiftAmt) (shl AndMask, ShiftAmt)), but the shift can come 13721 // out of GEP lowering in some cases. 13722 SDValue N0 = N->getOperand(0); 13723 ConstantSDNode *ShiftAmtNode = dyn_cast<ConstantSDNode>(N->getOperand(1)); 13724 if (!ShiftAmtNode) 13725 return SDValue(); 13726 uint32_t ShiftAmt = static_cast<uint32_t>(ShiftAmtNode->getZExtValue()); 13727 ConstantSDNode *AndMaskNode = dyn_cast<ConstantSDNode>(N0->getOperand(1)); 13728 if (!AndMaskNode) 13729 return SDValue(); 13730 uint32_t AndMask = static_cast<uint32_t>(AndMaskNode->getZExtValue()); 13731 // Don't transform uxtb/uxth. 13732 if (AndMask == 255 || AndMask == 65535) 13733 return SDValue(); 13734 if (isMask_32(AndMask)) { 13735 uint32_t MaskedBits = countLeadingZeros(AndMask); 13736 if (MaskedBits > ShiftAmt) { 13737 SDLoc DL(N); 13738 SDValue SHL = DAG.getNode(ISD::SHL, DL, MVT::i32, N0->getOperand(0), 13739 DAG.getConstant(MaskedBits, DL, MVT::i32)); 13740 return DAG.getNode( 13741 ISD::SRL, DL, MVT::i32, SHL, 13742 DAG.getConstant(MaskedBits - ShiftAmt, DL, MVT::i32)); 13743 } 13744 } 13745 } 13746 13747 // Nothing to be done for scalar shifts. 13748 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 13749 if (!VT.isVector() || !TLI.isTypeLegal(VT)) 13750 return SDValue(); 13751 if (ST->hasMVEIntegerOps() && VT == MVT::v2i64) 13752 return SDValue(); 13753 13754 int64_t Cnt; 13755 13756 switch (N->getOpcode()) { 13757 default: llvm_unreachable("unexpected shift opcode"); 13758 13759 case ISD::SHL: 13760 if (isVShiftLImm(N->getOperand(1), VT, false, Cnt)) { 13761 SDLoc dl(N); 13762 return DAG.getNode(ARMISD::VSHLIMM, dl, VT, N->getOperand(0), 13763 DAG.getConstant(Cnt, dl, MVT::i32)); 13764 } 13765 break; 13766 13767 case ISD::SRA: 13768 case ISD::SRL: 13769 if (isVShiftRImm(N->getOperand(1), VT, false, false, Cnt)) { 13770 unsigned VShiftOpc = 13771 (N->getOpcode() == ISD::SRA ? ARMISD::VSHRsIMM : ARMISD::VSHRuIMM); 13772 SDLoc dl(N); 13773 return DAG.getNode(VShiftOpc, dl, VT, N->getOperand(0), 13774 DAG.getConstant(Cnt, dl, MVT::i32)); 13775 } 13776 } 13777 return SDValue(); 13778 } 13779 13780 // Look for a sign/zero extend of a larger than legal load. This can be split 13781 // into two extending loads, which are simpler to deal with than an arbitrary 13782 // sign extend. 13783 static SDValue PerformSplittingToWideningLoad(SDNode *N, SelectionDAG &DAG) { 13784 SDValue N0 = N->getOperand(0); 13785 if (N0.getOpcode() != ISD::LOAD) 13786 return SDValue(); 13787 LoadSDNode *LD = cast<LoadSDNode>(N0.getNode()); 13788 if (!LD->isSimple() || !N0.hasOneUse() || LD->isIndexed() || 13789 LD->getExtensionType() != ISD::NON_EXTLOAD) 13790 return SDValue(); 13791 EVT FromVT = LD->getValueType(0); 13792 EVT ToVT = N->getValueType(0); 13793 if (!ToVT.isVector()) 13794 return SDValue(); 13795 assert(FromVT.getVectorNumElements() == ToVT.getVectorNumElements()); 13796 EVT ToEltVT = ToVT.getVectorElementType(); 13797 EVT FromEltVT = FromVT.getVectorElementType(); 13798 13799 unsigned NumElements = 0; 13800 if (ToEltVT == MVT::i32 && (FromEltVT == MVT::i16 || FromEltVT == MVT::i8)) 13801 NumElements = 4; 13802 if (ToEltVT == MVT::i16 && FromEltVT == MVT::i8) 13803 NumElements = 8; 13804 if (NumElements == 0 || 13805 FromVT.getVectorNumElements() == NumElements || 13806 FromVT.getVectorNumElements() % NumElements != 0 || 13807 !isPowerOf2_32(NumElements)) 13808 return SDValue(); 13809 13810 SDLoc DL(LD); 13811 // Details about the old load 13812 SDValue Ch = LD->getChain(); 13813 SDValue BasePtr = LD->getBasePtr(); 13814 unsigned Alignment = LD->getOriginalAlignment(); 13815 MachineMemOperand::Flags MMOFlags = LD->getMemOperand()->getFlags(); 13816 AAMDNodes AAInfo = LD->getAAInfo(); 13817 13818 ISD::LoadExtType NewExtType = 13819 N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SEXTLOAD : ISD::ZEXTLOAD; 13820 SDValue Offset = DAG.getUNDEF(BasePtr.getValueType()); 13821 EVT NewFromVT = FromVT.getHalfNumVectorElementsVT(*DAG.getContext()); 13822 EVT NewToVT = ToVT.getHalfNumVectorElementsVT(*DAG.getContext()); 13823 unsigned NewOffset = NewFromVT.getSizeInBits() / 8; 13824 SDValue NewPtr = DAG.getObjectPtrOffset(DL, BasePtr, NewOffset); 13825 13826 // Split the load in half, each side of which is extended separately. This 13827 // is good enough, as legalisation will take it from there. They are either 13828 // already legal or they will be split further into something that is 13829 // legal. 13830 SDValue NewLoad1 = 13831 DAG.getLoad(ISD::UNINDEXED, NewExtType, NewToVT, DL, Ch, BasePtr, Offset, 13832 LD->getPointerInfo(), NewFromVT, Alignment, MMOFlags, AAInfo); 13833 SDValue NewLoad2 = 13834 DAG.getLoad(ISD::UNINDEXED, NewExtType, NewToVT, DL, Ch, NewPtr, Offset, 13835 LD->getPointerInfo().getWithOffset(NewOffset), NewFromVT, 13836 Alignment, MMOFlags, AAInfo); 13837 13838 SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, 13839 SDValue(NewLoad1.getNode(), 1), 13840 SDValue(NewLoad2.getNode(), 1)); 13841 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), NewChain); 13842 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ToVT, NewLoad1, NewLoad2); 13843 } 13844 13845 /// PerformExtendCombine - Target-specific DAG combining for ISD::SIGN_EXTEND, 13846 /// ISD::ZERO_EXTEND, and ISD::ANY_EXTEND. 13847 static SDValue PerformExtendCombine(SDNode *N, SelectionDAG &DAG, 13848 const ARMSubtarget *ST) { 13849 SDValue N0 = N->getOperand(0); 13850 13851 // Check for sign- and zero-extensions of vector extract operations of 8- 13852 // and 16-bit vector elements. NEON supports these directly. They are 13853 // handled during DAG combining because type legalization will promote them 13854 // to 32-bit types and it is messy to recognize the operations after that. 13855 if (ST->hasNEON() && N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT) { 13856 SDValue Vec = N0.getOperand(0); 13857 SDValue Lane = N0.getOperand(1); 13858 EVT VT = N->getValueType(0); 13859 EVT EltVT = N0.getValueType(); 13860 const TargetLowering &TLI = DAG.getTargetLoweringInfo(); 13861 13862 if (VT == MVT::i32 && 13863 (EltVT == MVT::i8 || EltVT == MVT::i16) && 13864 TLI.isTypeLegal(Vec.getValueType()) && 13865 isa<ConstantSDNode>(Lane)) { 13866 13867 unsigned Opc = 0; 13868 switch (N->getOpcode()) { 13869 default: llvm_unreachable("unexpected opcode"); 13870 case ISD::SIGN_EXTEND: 13871 Opc = ARMISD::VGETLANEs; 13872 break; 13873 case ISD::ZERO_EXTEND: 13874 case ISD::ANY_EXTEND: 13875 Opc = ARMISD::VGETLANEu; 13876 break; 13877 } 13878 return DAG.getNode(Opc, SDLoc(N), VT, Vec, Lane); 13879 } 13880 } 13881 13882 if (ST->hasMVEIntegerOps()) 13883 if (SDValue NewLoad = PerformSplittingToWideningLoad(N, DAG)) 13884 return NewLoad; 13885 13886 return SDValue(); 13887 } 13888 13889 static const APInt *isPowerOf2Constant(SDValue V) { 13890 ConstantSDNode *C = dyn_cast<ConstantSDNode>(V); 13891 if (!C) 13892 return nullptr; 13893 const APInt *CV = &C->getAPIntValue(); 13894 return CV->isPowerOf2() ? CV : nullptr; 13895 } 13896 13897 SDValue ARMTargetLowering::PerformCMOVToBFICombine(SDNode *CMOV, SelectionDAG &DAG) const { 13898 // If we have a CMOV, OR and AND combination such as: 13899 // if (x & CN) 13900 // y |= CM; 13901 // 13902 // And: 13903 // * CN is a single bit; 13904 // * All bits covered by CM are known zero in y 13905 // 13906 // Then we can convert this into a sequence of BFI instructions. This will 13907 // always be a win if CM is a single bit, will always be no worse than the 13908 // TST&OR sequence if CM is two bits, and for thumb will be no worse if CM is 13909 // three bits (due to the extra IT instruction). 13910 13911 SDValue Op0 = CMOV->getOperand(0); 13912 SDValue Op1 = CMOV->getOperand(1); 13913 auto CCNode = cast<ConstantSDNode>(CMOV->getOperand(2)); 13914 auto CC = CCNode->getAPIntValue().getLimitedValue(); 13915 SDValue CmpZ = CMOV->getOperand(4); 13916 13917 // The compare must be against zero. 13918 if (!isNullConstant(CmpZ->getOperand(1))) 13919 return SDValue(); 13920 13921 assert(CmpZ->getOpcode() == ARMISD::CMPZ); 13922 SDValue And = CmpZ->getOperand(0); 13923 if (And->getOpcode() != ISD::AND) 13924 return SDValue(); 13925 const APInt *AndC = isPowerOf2Constant(And->getOperand(1)); 13926 if (!AndC) 13927 return SDValue(); 13928 SDValue X = And->getOperand(0); 13929 13930 if (CC == ARMCC::EQ) { 13931 // We're performing an "equal to zero" compare. Swap the operands so we 13932 // canonicalize on a "not equal to zero" compare. 13933 std::swap(Op0, Op1); 13934 } else { 13935 assert(CC == ARMCC::NE && "How can a CMPZ node not be EQ or NE?"); 13936 } 13937 13938 if (Op1->getOpcode() != ISD::OR) 13939 return SDValue(); 13940 13941 ConstantSDNode *OrC = dyn_cast<ConstantSDNode>(Op1->getOperand(1)); 13942 if (!OrC) 13943 return SDValue(); 13944 SDValue Y = Op1->getOperand(0); 13945 13946 if (Op0 != Y) 13947 return SDValue(); 13948 13949 // Now, is it profitable to continue? 13950 APInt OrCI = OrC->getAPIntValue(); 13951 unsigned Heuristic = Subtarget->isThumb() ? 3 : 2; 13952 if (OrCI.countPopulation() > Heuristic) 13953 return SDValue(); 13954 13955 // Lastly, can we determine that the bits defined by OrCI 13956 // are zero in Y? 13957 KnownBits Known = DAG.computeKnownBits(Y); 13958 if ((OrCI & Known.Zero) != OrCI) 13959 return SDValue(); 13960 13961 // OK, we can do the combine. 13962 SDValue V = Y; 13963 SDLoc dl(X); 13964 EVT VT = X.getValueType(); 13965 unsigned BitInX = AndC->logBase2(); 13966 13967 if (BitInX != 0) { 13968 // We must shift X first. 13969 X = DAG.getNode(ISD::SRL, dl, VT, X, 13970 DAG.getConstant(BitInX, dl, VT)); 13971 } 13972 13973 for (unsigned BitInY = 0, NumActiveBits = OrCI.getActiveBits(); 13974 BitInY < NumActiveBits; ++BitInY) { 13975 if (OrCI[BitInY] == 0) 13976 continue; 13977 APInt Mask(VT.getSizeInBits(), 0); 13978 Mask.setBit(BitInY); 13979 V = DAG.getNode(ARMISD::BFI, dl, VT, V, X, 13980 // Confusingly, the operand is an *inverted* mask. 13981 DAG.getConstant(~Mask, dl, VT)); 13982 } 13983 13984 return V; 13985 } 13986 13987 // Given N, the value controlling the conditional branch, search for the loop 13988 // intrinsic, returning it, along with how the value is used. We need to handle 13989 // patterns such as the following: 13990 // (brcond (xor (setcc (loop.decrement), 0, ne), 1), exit) 13991 // (brcond (setcc (loop.decrement), 0, eq), exit) 13992 // (brcond (setcc (loop.decrement), 0, ne), header) 13993 static SDValue SearchLoopIntrinsic(SDValue N, ISD::CondCode &CC, int &Imm, 13994 bool &Negate) { 13995 switch (N->getOpcode()) { 13996 default: 13997 break; 13998 case ISD::XOR: { 13999 if (!isa<ConstantSDNode>(N.getOperand(1))) 14000 return SDValue(); 14001 if (!cast<ConstantSDNode>(N.getOperand(1))->isOne()) 14002 return SDValue(); 14003 Negate = !Negate; 14004 return SearchLoopIntrinsic(N.getOperand(0), CC, Imm, Negate); 14005 } 14006 case ISD::SETCC: { 14007 auto *Const = dyn_cast<ConstantSDNode>(N.getOperand(1)); 14008 if (!Const) 14009 return SDValue(); 14010 if (Const->isNullValue()) 14011 Imm = 0; 14012 else if (Const->isOne()) 14013 Imm = 1; 14014 else 14015 return SDValue(); 14016 CC = cast<CondCodeSDNode>(N.getOperand(2))->get(); 14017 return SearchLoopIntrinsic(N->getOperand(0), CC, Imm, Negate); 14018 } 14019 case ISD::INTRINSIC_W_CHAIN: { 14020 unsigned IntOp = cast<ConstantSDNode>(N.getOperand(1))->getZExtValue(); 14021 if (IntOp != Intrinsic::test_set_loop_iterations && 14022 IntOp != Intrinsic::loop_decrement_reg) 14023 return SDValue(); 14024 return N; 14025 } 14026 } 14027 return SDValue(); 14028 } 14029 14030 static SDValue PerformHWLoopCombine(SDNode *N, 14031 TargetLowering::DAGCombinerInfo &DCI, 14032 const ARMSubtarget *ST) { 14033 14034 // The hwloop intrinsics that we're interested are used for control-flow, 14035 // either for entering or exiting the loop: 14036 // - test.set.loop.iterations will test whether its operand is zero. If it 14037 // is zero, the proceeding branch should not enter the loop. 14038 // - loop.decrement.reg also tests whether its operand is zero. If it is 14039 // zero, the proceeding branch should not branch back to the beginning of 14040 // the loop. 14041 // So here, we need to check that how the brcond is using the result of each 14042 // of the intrinsics to ensure that we're branching to the right place at the 14043 // right time. 14044 14045 ISD::CondCode CC; 14046 SDValue Cond; 14047 int Imm = 1; 14048 bool Negate = false; 14049 SDValue Chain = N->getOperand(0); 14050 SDValue Dest; 14051 14052 if (N->getOpcode() == ISD::BRCOND) { 14053 CC = ISD::SETEQ; 14054 Cond = N->getOperand(1); 14055 Dest = N->getOperand(2); 14056 } else { 14057 assert(N->getOpcode() == ISD::BR_CC && "Expected BRCOND or BR_CC!"); 14058 CC = cast<CondCodeSDNode>(N->getOperand(1))->get(); 14059 Cond = N->getOperand(2); 14060 Dest = N->getOperand(4); 14061 if (auto *Const = dyn_cast<ConstantSDNode>(N->getOperand(3))) { 14062 if (!Const->isOne() && !Const->isNullValue()) 14063 return SDValue(); 14064 Imm = Const->getZExtValue(); 14065 } else 14066 return SDValue(); 14067 } 14068 14069 SDValue Int = SearchLoopIntrinsic(Cond, CC, Imm, Negate); 14070 if (!Int) 14071 return SDValue(); 14072 14073 if (Negate) 14074 CC = ISD::getSetCCInverse(CC, true); 14075 14076 auto IsTrueIfZero = [](ISD::CondCode CC, int Imm) { 14077 return (CC == ISD::SETEQ && Imm == 0) || 14078 (CC == ISD::SETNE && Imm == 1) || 14079 (CC == ISD::SETLT && Imm == 1) || 14080 (CC == ISD::SETULT && Imm == 1); 14081 }; 14082 14083 auto IsFalseIfZero = [](ISD::CondCode CC, int Imm) { 14084 return (CC == ISD::SETEQ && Imm == 1) || 14085 (CC == ISD::SETNE && Imm == 0) || 14086 (CC == ISD::SETGT && Imm == 0) || 14087 (CC == ISD::SETUGT && Imm == 0) || 14088 (CC == ISD::SETGE && Imm == 1) || 14089 (CC == ISD::SETUGE && Imm == 1); 14090 }; 14091 14092 assert((IsTrueIfZero(CC, Imm) || IsFalseIfZero(CC, Imm)) && 14093 "unsupported condition"); 14094 14095 SDLoc dl(Int); 14096 SelectionDAG &DAG = DCI.DAG; 14097 SDValue Elements = Int.getOperand(2); 14098 unsigned IntOp = cast<ConstantSDNode>(Int->getOperand(1))->getZExtValue(); 14099 assert((N->hasOneUse() && N->use_begin()->getOpcode() == ISD::BR) 14100 && "expected single br user"); 14101 SDNode *Br = *N->use_begin(); 14102 SDValue OtherTarget = Br->getOperand(1); 14103 14104 // Update the unconditional branch to branch to the given Dest. 14105 auto UpdateUncondBr = [](SDNode *Br, SDValue Dest, SelectionDAG &DAG) { 14106 SDValue NewBrOps[] = { Br->getOperand(0), Dest }; 14107 SDValue NewBr = DAG.getNode(ISD::BR, SDLoc(Br), MVT::Other, NewBrOps); 14108 DAG.ReplaceAllUsesOfValueWith(SDValue(Br, 0), NewBr); 14109 }; 14110 14111 if (IntOp == Intrinsic::test_set_loop_iterations) { 14112 SDValue Res; 14113 // We expect this 'instruction' to branch when the counter is zero. 14114 if (IsTrueIfZero(CC, Imm)) { 14115 SDValue Ops[] = { Chain, Elements, Dest }; 14116 Res = DAG.getNode(ARMISD::WLS, dl, MVT::Other, Ops); 14117 } else { 14118 // The logic is the reverse of what we need for WLS, so find the other 14119 // basic block target: the target of the proceeding br. 14120 UpdateUncondBr(Br, Dest, DAG); 14121 14122 SDValue Ops[] = { Chain, Elements, OtherTarget }; 14123 Res = DAG.getNode(ARMISD::WLS, dl, MVT::Other, Ops); 14124 } 14125 DAG.ReplaceAllUsesOfValueWith(Int.getValue(1), Int.getOperand(0)); 14126 return Res; 14127 } else { 14128 SDValue Size = DAG.getTargetConstant( 14129 cast<ConstantSDNode>(Int.getOperand(3))->getZExtValue(), dl, MVT::i32); 14130 SDValue Args[] = { Int.getOperand(0), Elements, Size, }; 14131 SDValue LoopDec = DAG.getNode(ARMISD::LOOP_DEC, dl, 14132 DAG.getVTList(MVT::i32, MVT::Other), Args); 14133 DAG.ReplaceAllUsesWith(Int.getNode(), LoopDec.getNode()); 14134 14135 // We expect this instruction to branch when the count is not zero. 14136 SDValue Target = IsFalseIfZero(CC, Imm) ? Dest : OtherTarget; 14137 14138 // Update the unconditional branch to target the loop preheader if we've 14139 // found the condition has been reversed. 14140 if (Target == OtherTarget) 14141 UpdateUncondBr(Br, Dest, DAG); 14142 14143 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, 14144 SDValue(LoopDec.getNode(), 1), Chain); 14145 14146 SDValue EndArgs[] = { Chain, SDValue(LoopDec.getNode(), 0), Target }; 14147 return DAG.getNode(ARMISD::LE, dl, MVT::Other, EndArgs); 14148 } 14149 return SDValue(); 14150 } 14151 14152 /// PerformBRCONDCombine - Target-specific DAG combining for ARMISD::BRCOND. 14153 SDValue 14154 ARMTargetLowering::PerformBRCONDCombine(SDNode *N, SelectionDAG &DAG) const { 14155 SDValue Cmp = N->getOperand(4); 14156 if (Cmp.getOpcode() != ARMISD::CMPZ) 14157 // Only looking at NE cases. 14158 return SDValue(); 14159 14160 EVT VT = N->getValueType(0); 14161 SDLoc dl(N); 14162 SDValue LHS = Cmp.getOperand(0); 14163 SDValue RHS = Cmp.getOperand(1); 14164 SDValue Chain = N->getOperand(0); 14165 SDValue BB = N->getOperand(1); 14166 SDValue ARMcc = N->getOperand(2); 14167 ARMCC::CondCodes CC = 14168 (ARMCC::CondCodes)cast<ConstantSDNode>(ARMcc)->getZExtValue(); 14169 14170 // (brcond Chain BB ne CPSR (cmpz (and (cmov 0 1 CC CPSR Cmp) 1) 0)) 14171 // -> (brcond Chain BB CC CPSR Cmp) 14172 if (CC == ARMCC::NE && LHS.getOpcode() == ISD::AND && LHS->hasOneUse() && 14173 LHS->getOperand(0)->getOpcode() == ARMISD::CMOV && 14174 LHS->getOperand(0)->hasOneUse()) { 14175 auto *LHS00C = dyn_cast<ConstantSDNode>(LHS->getOperand(0)->getOperand(0)); 14176 auto *LHS01C = dyn_cast<ConstantSDNode>(LHS->getOperand(0)->getOperand(1)); 14177 auto *LHS1C = dyn_cast<ConstantSDNode>(LHS->getOperand(1)); 14178 auto *RHSC = dyn_cast<ConstantSDNode>(RHS); 14179 if ((LHS00C && LHS00C->getZExtValue() == 0) && 14180 (LHS01C && LHS01C->getZExtValue() == 1) && 14181 (LHS1C && LHS1C->getZExtValue() == 1) && 14182 (RHSC && RHSC->getZExtValue() == 0)) { 14183 return DAG.getNode( 14184 ARMISD::BRCOND, dl, VT, Chain, BB, LHS->getOperand(0)->getOperand(2), 14185 LHS->getOperand(0)->getOperand(3), LHS->getOperand(0)->getOperand(4)); 14186 } 14187 } 14188 14189 return SDValue(); 14190 } 14191 14192 /// PerformCMOVCombine - Target-specific DAG combining for ARMISD::CMOV. 14193 SDValue 14194 ARMTargetLowering::PerformCMOVCombine(SDNode *N, SelectionDAG &DAG) const { 14195 SDValue Cmp = N->getOperand(4); 14196 if (Cmp.getOpcode() != ARMISD::CMPZ) 14197 // Only looking at EQ and NE cases. 14198 return SDValue(); 14199 14200 EVT VT = N->getValueType(0); 14201 SDLoc dl(N); 14202 SDValue LHS = Cmp.getOperand(0); 14203 SDValue RHS = Cmp.getOperand(1); 14204 SDValue FalseVal = N->getOperand(0); 14205 SDValue TrueVal = N->getOperand(1); 14206 SDValue ARMcc = N->getOperand(2); 14207 ARMCC::CondCodes CC = 14208 (ARMCC::CondCodes)cast<ConstantSDNode>(ARMcc)->getZExtValue(); 14209 14210 // BFI is only available on V6T2+. 14211 if (!Subtarget->isThumb1Only() && Subtarget->hasV6T2Ops()) { 14212 SDValue R = PerformCMOVToBFICombine(N, DAG); 14213 if (R) 14214 return R; 14215 } 14216 14217 // Simplify 14218 // mov r1, r0 14219 // cmp r1, x 14220 // mov r0, y 14221 // moveq r0, x 14222 // to 14223 // cmp r0, x 14224 // movne r0, y 14225 // 14226 // mov r1, r0 14227 // cmp r1, x 14228 // mov r0, x 14229 // movne r0, y 14230 // to 14231 // cmp r0, x 14232 // movne r0, y 14233 /// FIXME: Turn this into a target neutral optimization? 14234 SDValue Res; 14235 if (CC == ARMCC::NE && FalseVal == RHS && FalseVal != LHS) { 14236 Res = DAG.getNode(ARMISD::CMOV, dl, VT, LHS, TrueVal, ARMcc, 14237 N->getOperand(3), Cmp); 14238 } else if (CC == ARMCC::EQ && TrueVal == RHS) { 14239 SDValue ARMcc; 14240 SDValue NewCmp = getARMCmp(LHS, RHS, ISD::SETNE, ARMcc, DAG, dl); 14241 Res = DAG.getNode(ARMISD::CMOV, dl, VT, LHS, FalseVal, ARMcc, 14242 N->getOperand(3), NewCmp); 14243 } 14244 14245 // (cmov F T ne CPSR (cmpz (cmov 0 1 CC CPSR Cmp) 0)) 14246 // -> (cmov F T CC CPSR Cmp) 14247 if (CC == ARMCC::NE && LHS.getOpcode() == ARMISD::CMOV && LHS->hasOneUse()) { 14248 auto *LHS0C = dyn_cast<ConstantSDNode>(LHS->getOperand(0)); 14249 auto *LHS1C = dyn_cast<ConstantSDNode>(LHS->getOperand(1)); 14250 auto *RHSC = dyn_cast<ConstantSDNode>(RHS); 14251 if ((LHS0C && LHS0C->getZExtValue() == 0) && 14252 (LHS1C && LHS1C->getZExtValue() == 1) && 14253 (RHSC && RHSC->getZExtValue() == 0)) { 14254 return DAG.getNode(ARMISD::CMOV, dl, VT, FalseVal, TrueVal, 14255 LHS->getOperand(2), LHS->getOperand(3), 14256 LHS->getOperand(4)); 14257 } 14258 } 14259 14260 if (!VT.isInteger()) 14261 return SDValue(); 14262 14263 // Materialize a boolean comparison for integers so we can avoid branching. 14264 if (isNullConstant(FalseVal)) { 14265 if (CC == ARMCC::EQ && isOneConstant(TrueVal)) { 14266 if (!Subtarget->isThumb1Only() && Subtarget->hasV5TOps()) { 14267 // If x == y then x - y == 0 and ARM's CLZ will return 32, shifting it 14268 // right 5 bits will make that 32 be 1, otherwise it will be 0. 14269 // CMOV 0, 1, ==, (CMPZ x, y) -> SRL (CTLZ (SUB x, y)), 5 14270 SDValue Sub = DAG.getNode(ISD::SUB, dl, VT, LHS, RHS); 14271 Res = DAG.getNode(ISD::SRL, dl, VT, DAG.getNode(ISD::CTLZ, dl, VT, Sub), 14272 DAG.getConstant(5, dl, MVT::i32)); 14273 } else { 14274 // CMOV 0, 1, ==, (CMPZ x, y) -> 14275 // (ADDCARRY (SUB x, y), t:0, t:1) 14276 // where t = (SUBCARRY 0, (SUB x, y), 0) 14277 // 14278 // The SUBCARRY computes 0 - (x - y) and this will give a borrow when 14279 // x != y. In other words, a carry C == 1 when x == y, C == 0 14280 // otherwise. 14281 // The final ADDCARRY computes 14282 // x - y + (0 - (x - y)) + C == C 14283 SDValue Sub = DAG.getNode(ISD::SUB, dl, VT, LHS, RHS); 14284 SDVTList VTs = DAG.getVTList(VT, MVT::i32); 14285 SDValue Neg = DAG.getNode(ISD::USUBO, dl, VTs, FalseVal, Sub); 14286 // ISD::SUBCARRY returns a borrow but we want the carry here 14287 // actually. 14288 SDValue Carry = 14289 DAG.getNode(ISD::SUB, dl, MVT::i32, 14290 DAG.getConstant(1, dl, MVT::i32), Neg.getValue(1)); 14291 Res = DAG.getNode(ISD::ADDCARRY, dl, VTs, Sub, Neg, Carry); 14292 } 14293 } else if (CC == ARMCC::NE && !isNullConstant(RHS) && 14294 (!Subtarget->isThumb1Only() || isPowerOf2Constant(TrueVal))) { 14295 // This seems pointless but will allow us to combine it further below. 14296 // CMOV 0, z, !=, (CMPZ x, y) -> CMOV (SUBS x, y), z, !=, (SUBS x, y):1 14297 SDValue Sub = 14298 DAG.getNode(ARMISD::SUBS, dl, DAG.getVTList(VT, MVT::i32), LHS, RHS); 14299 SDValue CPSRGlue = DAG.getCopyToReg(DAG.getEntryNode(), dl, ARM::CPSR, 14300 Sub.getValue(1), SDValue()); 14301 Res = DAG.getNode(ARMISD::CMOV, dl, VT, Sub, TrueVal, ARMcc, 14302 N->getOperand(3), CPSRGlue.getValue(1)); 14303 FalseVal = Sub; 14304 } 14305 } else if (isNullConstant(TrueVal)) { 14306 if (CC == ARMCC::EQ && !isNullConstant(RHS) && 14307 (!Subtarget->isThumb1Only() || isPowerOf2Constant(FalseVal))) { 14308 // This seems pointless but will allow us to combine it further below 14309 // Note that we change == for != as this is the dual for the case above. 14310 // CMOV z, 0, ==, (CMPZ x, y) -> CMOV (SUBS x, y), z, !=, (SUBS x, y):1 14311 SDValue Sub = 14312 DAG.getNode(ARMISD::SUBS, dl, DAG.getVTList(VT, MVT::i32), LHS, RHS); 14313 SDValue CPSRGlue = DAG.getCopyToReg(DAG.getEntryNode(), dl, ARM::CPSR, 14314 Sub.getValue(1), SDValue()); 14315 Res = DAG.getNode(ARMISD::CMOV, dl, VT, Sub, FalseVal, 14316 DAG.getConstant(ARMCC::NE, dl, MVT::i32), 14317 N->getOperand(3), CPSRGlue.getValue(1)); 14318 FalseVal = Sub; 14319 } 14320 } 14321 14322 // On Thumb1, the DAG above may be further combined if z is a power of 2 14323 // (z == 2 ^ K). 14324 // CMOV (SUBS x, y), z, !=, (SUBS x, y):1 -> 14325 // t1 = (USUBO (SUB x, y), 1) 14326 // t2 = (SUBCARRY (SUB x, y), t1:0, t1:1) 14327 // Result = if K != 0 then (SHL t2:0, K) else t2:0 14328 // 14329 // This also handles the special case of comparing against zero; it's 14330 // essentially, the same pattern, except there's no SUBS: 14331 // CMOV x, z, !=, (CMPZ x, 0) -> 14332 // t1 = (USUBO x, 1) 14333 // t2 = (SUBCARRY x, t1:0, t1:1) 14334 // Result = if K != 0 then (SHL t2:0, K) else t2:0 14335 const APInt *TrueConst; 14336 if (Subtarget->isThumb1Only() && CC == ARMCC::NE && 14337 ((FalseVal.getOpcode() == ARMISD::SUBS && 14338 FalseVal.getOperand(0) == LHS && FalseVal.getOperand(1) == RHS) || 14339 (FalseVal == LHS && isNullConstant(RHS))) && 14340 (TrueConst = isPowerOf2Constant(TrueVal))) { 14341 SDVTList VTs = DAG.getVTList(VT, MVT::i32); 14342 unsigned ShiftAmount = TrueConst->logBase2(); 14343 if (ShiftAmount) 14344 TrueVal = DAG.getConstant(1, dl, VT); 14345 SDValue Subc = DAG.getNode(ISD::USUBO, dl, VTs, FalseVal, TrueVal); 14346 Res = DAG.getNode(ISD::SUBCARRY, dl, VTs, FalseVal, Subc, Subc.getValue(1)); 14347 14348 if (ShiftAmount) 14349 Res = DAG.getNode(ISD::SHL, dl, VT, Res, 14350 DAG.getConstant(ShiftAmount, dl, MVT::i32)); 14351 } 14352 14353 if (Res.getNode()) { 14354 KnownBits Known = DAG.computeKnownBits(SDValue(N,0)); 14355 // Capture demanded bits information that would be otherwise lost. 14356 if (Known.Zero == 0xfffffffe) 14357 Res = DAG.getNode(ISD::AssertZext, dl, MVT::i32, Res, 14358 DAG.getValueType(MVT::i1)); 14359 else if (Known.Zero == 0xffffff00) 14360 Res = DAG.getNode(ISD::AssertZext, dl, MVT::i32, Res, 14361 DAG.getValueType(MVT::i8)); 14362 else if (Known.Zero == 0xffff0000) 14363 Res = DAG.getNode(ISD::AssertZext, dl, MVT::i32, Res, 14364 DAG.getValueType(MVT::i16)); 14365 } 14366 14367 return Res; 14368 } 14369 14370 SDValue ARMTargetLowering::PerformDAGCombine(SDNode *N, 14371 DAGCombinerInfo &DCI) const { 14372 switch (N->getOpcode()) { 14373 default: break; 14374 case ISD::ABS: return PerformABSCombine(N, DCI, Subtarget); 14375 case ARMISD::ADDE: return PerformADDECombine(N, DCI, Subtarget); 14376 case ARMISD::UMLAL: return PerformUMLALCombine(N, DCI.DAG, Subtarget); 14377 case ISD::ADD: return PerformADDCombine(N, DCI, Subtarget); 14378 case ISD::SUB: return PerformSUBCombine(N, DCI); 14379 case ISD::MUL: return PerformMULCombine(N, DCI, Subtarget); 14380 case ISD::OR: return PerformORCombine(N, DCI, Subtarget); 14381 case ISD::XOR: return PerformXORCombine(N, DCI, Subtarget); 14382 case ISD::AND: return PerformANDCombine(N, DCI, Subtarget); 14383 case ISD::BRCOND: 14384 case ISD::BR_CC: return PerformHWLoopCombine(N, DCI, Subtarget); 14385 case ARMISD::ADDC: 14386 case ARMISD::SUBC: return PerformAddcSubcCombine(N, DCI, Subtarget); 14387 case ARMISD::SUBE: return PerformAddeSubeCombine(N, DCI, Subtarget); 14388 case ARMISD::BFI: return PerformBFICombine(N, DCI); 14389 case ARMISD::VMOVRRD: return PerformVMOVRRDCombine(N, DCI, Subtarget); 14390 case ARMISD::VMOVDRR: return PerformVMOVDRRCombine(N, DCI.DAG); 14391 case ISD::STORE: return PerformSTORECombine(N, DCI, Subtarget); 14392 case ISD::BUILD_VECTOR: return PerformBUILD_VECTORCombine(N, DCI, Subtarget); 14393 case ISD::INSERT_VECTOR_ELT: return PerformInsertEltCombine(N, DCI); 14394 case ISD::VECTOR_SHUFFLE: return PerformVECTOR_SHUFFLECombine(N, DCI.DAG); 14395 case ARMISD::VDUPLANE: return PerformVDUPLANECombine(N, DCI); 14396 case ARMISD::VDUP: return PerformVDUPCombine(N, DCI, Subtarget); 14397 case ISD::FP_TO_SINT: 14398 case ISD::FP_TO_UINT: 14399 return PerformVCVTCombine(N, DCI.DAG, Subtarget); 14400 case ISD::FDIV: 14401 return PerformVDIVCombine(N, DCI.DAG, Subtarget); 14402 case ISD::INTRINSIC_WO_CHAIN: return PerformIntrinsicCombine(N, DCI.DAG); 14403 case ISD::SHL: 14404 case ISD::SRA: 14405 case ISD::SRL: 14406 return PerformShiftCombine(N, DCI, Subtarget); 14407 case ISD::SIGN_EXTEND: 14408 case ISD::ZERO_EXTEND: 14409 case ISD::ANY_EXTEND: return PerformExtendCombine(N, DCI.DAG, Subtarget); 14410 case ARMISD::CMOV: return PerformCMOVCombine(N, DCI.DAG); 14411 case ARMISD::BRCOND: return PerformBRCONDCombine(N, DCI.DAG); 14412 case ISD::LOAD: return PerformLOADCombine(N, DCI); 14413 case ARMISD::VLD1DUP: 14414 case ARMISD::VLD2DUP: 14415 case ARMISD::VLD3DUP: 14416 case ARMISD::VLD4DUP: 14417 return PerformVLDCombine(N, DCI); 14418 case ARMISD::BUILD_VECTOR: 14419 return PerformARMBUILD_VECTORCombine(N, DCI); 14420 case ARMISD::PREDICATE_CAST: 14421 return PerformPREDICATE_CASTCombine(N, DCI); 14422 case ARMISD::SMULWB: { 14423 unsigned BitWidth = N->getValueType(0).getSizeInBits(); 14424 APInt DemandedMask = APInt::getLowBitsSet(BitWidth, 16); 14425 if (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI)) 14426 return SDValue(); 14427 break; 14428 } 14429 case ARMISD::SMULWT: { 14430 unsigned BitWidth = N->getValueType(0).getSizeInBits(); 14431 APInt DemandedMask = APInt::getHighBitsSet(BitWidth, 16); 14432 if (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI)) 14433 return SDValue(); 14434 break; 14435 } 14436 case ARMISD::SMLALBB: 14437 case ARMISD::QADD16b: 14438 case ARMISD::QSUB16b: { 14439 unsigned BitWidth = N->getValueType(0).getSizeInBits(); 14440 APInt DemandedMask = APInt::getLowBitsSet(BitWidth, 16); 14441 if ((SimplifyDemandedBits(N->getOperand(0), DemandedMask, DCI)) || 14442 (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI))) 14443 return SDValue(); 14444 break; 14445 } 14446 case ARMISD::SMLALBT: { 14447 unsigned LowWidth = N->getOperand(0).getValueType().getSizeInBits(); 14448 APInt LowMask = APInt::getLowBitsSet(LowWidth, 16); 14449 unsigned HighWidth = N->getOperand(1).getValueType().getSizeInBits(); 14450 APInt HighMask = APInt::getHighBitsSet(HighWidth, 16); 14451 if ((SimplifyDemandedBits(N->getOperand(0), LowMask, DCI)) || 14452 (SimplifyDemandedBits(N->getOperand(1), HighMask, DCI))) 14453 return SDValue(); 14454 break; 14455 } 14456 case ARMISD::SMLALTB: { 14457 unsigned HighWidth = N->getOperand(0).getValueType().getSizeInBits(); 14458 APInt HighMask = APInt::getHighBitsSet(HighWidth, 16); 14459 unsigned LowWidth = N->getOperand(1).getValueType().getSizeInBits(); 14460 APInt LowMask = APInt::getLowBitsSet(LowWidth, 16); 14461 if ((SimplifyDemandedBits(N->getOperand(0), HighMask, DCI)) || 14462 (SimplifyDemandedBits(N->getOperand(1), LowMask, DCI))) 14463 return SDValue(); 14464 break; 14465 } 14466 case ARMISD::SMLALTT: { 14467 unsigned BitWidth = N->getValueType(0).getSizeInBits(); 14468 APInt DemandedMask = APInt::getHighBitsSet(BitWidth, 16); 14469 if ((SimplifyDemandedBits(N->getOperand(0), DemandedMask, DCI)) || 14470 (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI))) 14471 return SDValue(); 14472 break; 14473 } 14474 case ARMISD::QADD8b: 14475 case ARMISD::QSUB8b: { 14476 unsigned BitWidth = N->getValueType(0).getSizeInBits(); 14477 APInt DemandedMask = APInt::getLowBitsSet(BitWidth, 8); 14478 if ((SimplifyDemandedBits(N->getOperand(0), DemandedMask, DCI)) || 14479 (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI))) 14480 return SDValue(); 14481 break; 14482 } 14483 case ISD::INTRINSIC_VOID: 14484 case ISD::INTRINSIC_W_CHAIN: 14485 switch (cast<ConstantSDNode>(N->getOperand(1))->getZExtValue()) { 14486 case Intrinsic::arm_neon_vld1: 14487 case Intrinsic::arm_neon_vld1x2: 14488 case Intrinsic::arm_neon_vld1x3: 14489 case Intrinsic::arm_neon_vld1x4: 14490 case Intrinsic::arm_neon_vld2: 14491 case Intrinsic::arm_neon_vld3: 14492 case Intrinsic::arm_neon_vld4: 14493 case Intrinsic::arm_neon_vld2lane: 14494 case Intrinsic::arm_neon_vld3lane: 14495 case Intrinsic::arm_neon_vld4lane: 14496 case Intrinsic::arm_neon_vld2dup: 14497 case Intrinsic::arm_neon_vld3dup: 14498 case Intrinsic::arm_neon_vld4dup: 14499 case Intrinsic::arm_neon_vst1: 14500 case Intrinsic::arm_neon_vst1x2: 14501 case Intrinsic::arm_neon_vst1x3: 14502 case Intrinsic::arm_neon_vst1x4: 14503 case Intrinsic::arm_neon_vst2: 14504 case Intrinsic::arm_neon_vst3: 14505 case Intrinsic::arm_neon_vst4: 14506 case Intrinsic::arm_neon_vst2lane: 14507 case Intrinsic::arm_neon_vst3lane: 14508 case Intrinsic::arm_neon_vst4lane: 14509 return PerformVLDCombine(N, DCI); 14510 default: break; 14511 } 14512 break; 14513 } 14514 return SDValue(); 14515 } 14516 14517 bool ARMTargetLowering::isDesirableToTransformToIntegerOp(unsigned Opc, 14518 EVT VT) const { 14519 return (VT == MVT::f32) && (Opc == ISD::LOAD || Opc == ISD::STORE); 14520 } 14521 14522 bool ARMTargetLowering::allowsMisalignedMemoryAccesses(EVT VT, unsigned, 14523 unsigned Alignment, 14524 MachineMemOperand::Flags, 14525 bool *Fast) const { 14526 // Depends what it gets converted into if the type is weird. 14527 if (!VT.isSimple()) 14528 return false; 14529 14530 // The AllowsUnaliged flag models the SCTLR.A setting in ARM cpus 14531 bool AllowsUnaligned = Subtarget->allowsUnalignedMem(); 14532 auto Ty = VT.getSimpleVT().SimpleTy; 14533 14534 if (Ty == MVT::i8 || Ty == MVT::i16 || Ty == MVT::i32) { 14535 // Unaligned access can use (for example) LRDB, LRDH, LDR 14536 if (AllowsUnaligned) { 14537 if (Fast) 14538 *Fast = Subtarget->hasV7Ops(); 14539 return true; 14540 } 14541 } 14542 14543 if (Ty == MVT::f64 || Ty == MVT::v2f64) { 14544 // For any little-endian targets with neon, we can support unaligned ld/st 14545 // of D and Q (e.g. {D0,D1}) registers by using vld1.i8/vst1.i8. 14546 // A big-endian target may also explicitly support unaligned accesses 14547 if (Subtarget->hasNEON() && (AllowsUnaligned || Subtarget->isLittle())) { 14548 if (Fast) 14549 *Fast = true; 14550 return true; 14551 } 14552 } 14553 14554 if (!Subtarget->hasMVEIntegerOps()) 14555 return false; 14556 14557 // These are for predicates 14558 if ((Ty == MVT::v16i1 || Ty == MVT::v8i1 || Ty == MVT::v4i1)) { 14559 if (Fast) 14560 *Fast = true; 14561 return true; 14562 } 14563 14564 // These are for truncated stores/narrowing loads. They are fine so long as 14565 // the alignment is at least the size of the item being loaded 14566 if ((Ty == MVT::v4i8 || Ty == MVT::v8i8 || Ty == MVT::v4i16) && 14567 Alignment >= VT.getScalarSizeInBits() / 8) { 14568 if (Fast) 14569 *Fast = true; 14570 return true; 14571 } 14572 14573 // In little-endian MVE, the store instructions VSTRB.U8, VSTRH.U16 and 14574 // VSTRW.U32 all store the vector register in exactly the same format, and 14575 // differ only in the range of their immediate offset field and the required 14576 // alignment. So there is always a store that can be used, regardless of 14577 // actual type. 14578 // 14579 // For big endian, that is not the case. But can still emit a (VSTRB.U8; 14580 // VREV64.8) pair and get the same effect. This will likely be better than 14581 // aligning the vector through the stack. 14582 if (Ty == MVT::v16i8 || Ty == MVT::v8i16 || Ty == MVT::v8f16 || 14583 Ty == MVT::v4i32 || Ty == MVT::v4f32 || Ty == MVT::v2i64 || 14584 Ty == MVT::v2f64) { 14585 if (Fast) 14586 *Fast = true; 14587 return true; 14588 } 14589 14590 return false; 14591 } 14592 14593 static bool memOpAlign(unsigned DstAlign, unsigned SrcAlign, 14594 unsigned AlignCheck) { 14595 return ((SrcAlign == 0 || SrcAlign % AlignCheck == 0) && 14596 (DstAlign == 0 || DstAlign % AlignCheck == 0)); 14597 } 14598 14599 EVT ARMTargetLowering::getOptimalMemOpType( 14600 uint64_t Size, unsigned DstAlign, unsigned SrcAlign, bool IsMemset, 14601 bool ZeroMemset, bool MemcpyStrSrc, 14602 const AttributeList &FuncAttributes) const { 14603 // See if we can use NEON instructions for this... 14604 if ((!IsMemset || ZeroMemset) && Subtarget->hasNEON() && 14605 !FuncAttributes.hasFnAttribute(Attribute::NoImplicitFloat)) { 14606 bool Fast; 14607 if (Size >= 16 && 14608 (memOpAlign(SrcAlign, DstAlign, 16) || 14609 (allowsMisalignedMemoryAccesses(MVT::v2f64, 0, 1, 14610 MachineMemOperand::MONone, &Fast) && 14611 Fast))) { 14612 return MVT::v2f64; 14613 } else if (Size >= 8 && 14614 (memOpAlign(SrcAlign, DstAlign, 8) || 14615 (allowsMisalignedMemoryAccesses( 14616 MVT::f64, 0, 1, MachineMemOperand::MONone, &Fast) && 14617 Fast))) { 14618 return MVT::f64; 14619 } 14620 } 14621 14622 // Let the target-independent logic figure it out. 14623 return MVT::Other; 14624 } 14625 14626 // 64-bit integers are split into their high and low parts and held in two 14627 // different registers, so the trunc is free since the low register can just 14628 // be used. 14629 bool ARMTargetLowering::isTruncateFree(Type *SrcTy, Type *DstTy) const { 14630 if (!SrcTy->isIntegerTy() || !DstTy->isIntegerTy()) 14631 return false; 14632 unsigned SrcBits = SrcTy->getPrimitiveSizeInBits(); 14633 unsigned DestBits = DstTy->getPrimitiveSizeInBits(); 14634 return (SrcBits == 64 && DestBits == 32); 14635 } 14636 14637 bool ARMTargetLowering::isTruncateFree(EVT SrcVT, EVT DstVT) const { 14638 if (SrcVT.isVector() || DstVT.isVector() || !SrcVT.isInteger() || 14639 !DstVT.isInteger()) 14640 return false; 14641 unsigned SrcBits = SrcVT.getSizeInBits(); 14642 unsigned DestBits = DstVT.getSizeInBits(); 14643 return (SrcBits == 64 && DestBits == 32); 14644 } 14645 14646 bool ARMTargetLowering::isZExtFree(SDValue Val, EVT VT2) const { 14647 if (Val.getOpcode() != ISD::LOAD) 14648 return false; 14649 14650 EVT VT1 = Val.getValueType(); 14651 if (!VT1.isSimple() || !VT1.isInteger() || 14652 !VT2.isSimple() || !VT2.isInteger()) 14653 return false; 14654 14655 switch (VT1.getSimpleVT().SimpleTy) { 14656 default: break; 14657 case MVT::i1: 14658 case MVT::i8: 14659 case MVT::i16: 14660 // 8-bit and 16-bit loads implicitly zero-extend to 32-bits. 14661 return true; 14662 } 14663 14664 return false; 14665 } 14666 14667 bool ARMTargetLowering::isFNegFree(EVT VT) const { 14668 if (!VT.isSimple()) 14669 return false; 14670 14671 // There are quite a few FP16 instructions (e.g. VNMLA, VNMLS, etc.) that 14672 // negate values directly (fneg is free). So, we don't want to let the DAG 14673 // combiner rewrite fneg into xors and some other instructions. For f16 and 14674 // FullFP16 argument passing, some bitcast nodes may be introduced, 14675 // triggering this DAG combine rewrite, so we are avoiding that with this. 14676 switch (VT.getSimpleVT().SimpleTy) { 14677 default: break; 14678 case MVT::f16: 14679 return Subtarget->hasFullFP16(); 14680 } 14681 14682 return false; 14683 } 14684 14685 /// Check if Ext1 and Ext2 are extends of the same type, doubling the bitwidth 14686 /// of the vector elements. 14687 static bool areExtractExts(Value *Ext1, Value *Ext2) { 14688 auto areExtDoubled = [](Instruction *Ext) { 14689 return Ext->getType()->getScalarSizeInBits() == 14690 2 * Ext->getOperand(0)->getType()->getScalarSizeInBits(); 14691 }; 14692 14693 if (!match(Ext1, m_ZExtOrSExt(m_Value())) || 14694 !match(Ext2, m_ZExtOrSExt(m_Value())) || 14695 !areExtDoubled(cast<Instruction>(Ext1)) || 14696 !areExtDoubled(cast<Instruction>(Ext2))) 14697 return false; 14698 14699 return true; 14700 } 14701 14702 /// Check if sinking \p I's operands to I's basic block is profitable, because 14703 /// the operands can be folded into a target instruction, e.g. 14704 /// sext/zext can be folded into vsubl. 14705 bool ARMTargetLowering::shouldSinkOperands(Instruction *I, 14706 SmallVectorImpl<Use *> &Ops) const { 14707 if (!I->getType()->isVectorTy()) 14708 return false; 14709 14710 if (Subtarget->hasNEON()) { 14711 switch (I->getOpcode()) { 14712 case Instruction::Sub: 14713 case Instruction::Add: { 14714 if (!areExtractExts(I->getOperand(0), I->getOperand(1))) 14715 return false; 14716 Ops.push_back(&I->getOperandUse(0)); 14717 Ops.push_back(&I->getOperandUse(1)); 14718 return true; 14719 } 14720 default: 14721 return false; 14722 } 14723 } 14724 14725 if (!Subtarget->hasMVEIntegerOps()) 14726 return false; 14727 14728 auto IsSinker = [](Instruction *I, int Operand) { 14729 switch (I->getOpcode()) { 14730 case Instruction::Add: 14731 case Instruction::Mul: 14732 return true; 14733 case Instruction::Sub: 14734 return Operand == 1; 14735 default: 14736 return false; 14737 } 14738 }; 14739 14740 int Op = 0; 14741 if (!isa<ShuffleVectorInst>(I->getOperand(Op))) 14742 Op = 1; 14743 if (!IsSinker(I, Op)) 14744 return false; 14745 if (!match(I->getOperand(Op), 14746 m_ShuffleVector(m_InsertElement(m_Undef(), m_Value(), m_ZeroInt()), 14747 m_Undef(), m_Zero()))) { 14748 return false; 14749 } 14750 Instruction *Shuffle = cast<Instruction>(I->getOperand(Op)); 14751 // All uses of the shuffle should be sunk to avoid duplicating it across gpr 14752 // and vector registers 14753 for (Use &U : Shuffle->uses()) { 14754 Instruction *Insn = cast<Instruction>(U.getUser()); 14755 if (!IsSinker(Insn, U.getOperandNo())) 14756 return false; 14757 } 14758 Ops.push_back(&Shuffle->getOperandUse(0)); 14759 Ops.push_back(&I->getOperandUse(Op)); 14760 return true; 14761 } 14762 14763 bool ARMTargetLowering::isVectorLoadExtDesirable(SDValue ExtVal) const { 14764 EVT VT = ExtVal.getValueType(); 14765 14766 if (!isTypeLegal(VT)) 14767 return false; 14768 14769 if (auto *Ld = dyn_cast<MaskedLoadSDNode>(ExtVal.getOperand(0))) { 14770 if (Ld->isExpandingLoad()) 14771 return false; 14772 } 14773 14774 // Don't create a loadext if we can fold the extension into a wide/long 14775 // instruction. 14776 // If there's more than one user instruction, the loadext is desirable no 14777 // matter what. There can be two uses by the same instruction. 14778 if (ExtVal->use_empty() || 14779 !ExtVal->use_begin()->isOnlyUserOf(ExtVal.getNode())) 14780 return true; 14781 14782 SDNode *U = *ExtVal->use_begin(); 14783 if ((U->getOpcode() == ISD::ADD || U->getOpcode() == ISD::SUB || 14784 U->getOpcode() == ISD::SHL || U->getOpcode() == ARMISD::VSHLIMM)) 14785 return false; 14786 14787 return true; 14788 } 14789 14790 bool ARMTargetLowering::allowTruncateForTailCall(Type *Ty1, Type *Ty2) const { 14791 if (!Ty1->isIntegerTy() || !Ty2->isIntegerTy()) 14792 return false; 14793 14794 if (!isTypeLegal(EVT::getEVT(Ty1))) 14795 return false; 14796 14797 assert(Ty1->getPrimitiveSizeInBits() <= 64 && "i128 is probably not a noop"); 14798 14799 // Assuming the caller doesn't have a zeroext or signext return parameter, 14800 // truncation all the way down to i1 is valid. 14801 return true; 14802 } 14803 14804 int ARMTargetLowering::getScalingFactorCost(const DataLayout &DL, 14805 const AddrMode &AM, Type *Ty, 14806 unsigned AS) const { 14807 if (isLegalAddressingMode(DL, AM, Ty, AS)) { 14808 if (Subtarget->hasFPAO()) 14809 return AM.Scale < 0 ? 1 : 0; // positive offsets execute faster 14810 return 0; 14811 } 14812 return -1; 14813 } 14814 14815 static bool isLegalT1AddressImmediate(int64_t V, EVT VT) { 14816 if (V < 0) 14817 return false; 14818 14819 unsigned Scale = 1; 14820 switch (VT.getSimpleVT().SimpleTy) { 14821 case MVT::i1: 14822 case MVT::i8: 14823 // Scale == 1; 14824 break; 14825 case MVT::i16: 14826 // Scale == 2; 14827 Scale = 2; 14828 break; 14829 default: 14830 // On thumb1 we load most things (i32, i64, floats, etc) with a LDR 14831 // Scale == 4; 14832 Scale = 4; 14833 break; 14834 } 14835 14836 if ((V & (Scale - 1)) != 0) 14837 return false; 14838 return isUInt<5>(V / Scale); 14839 } 14840 14841 static bool isLegalT2AddressImmediate(int64_t V, EVT VT, 14842 const ARMSubtarget *Subtarget) { 14843 if (!VT.isInteger() && !VT.isFloatingPoint()) 14844 return false; 14845 if (VT.isVector() && Subtarget->hasNEON()) 14846 return false; 14847 if (VT.isVector() && VT.isFloatingPoint() && Subtarget->hasMVEIntegerOps() && 14848 !Subtarget->hasMVEFloatOps()) 14849 return false; 14850 14851 bool IsNeg = false; 14852 if (V < 0) { 14853 IsNeg = true; 14854 V = -V; 14855 } 14856 14857 unsigned NumBytes = std::max(VT.getSizeInBits() / 8, 1U); 14858 14859 // MVE: size * imm7 14860 if (VT.isVector() && Subtarget->hasMVEIntegerOps()) { 14861 switch (VT.getSimpleVT().getVectorElementType().SimpleTy) { 14862 case MVT::i32: 14863 case MVT::f32: 14864 return isShiftedUInt<7,2>(V); 14865 case MVT::i16: 14866 case MVT::f16: 14867 return isShiftedUInt<7,1>(V); 14868 case MVT::i8: 14869 return isUInt<7>(V); 14870 default: 14871 return false; 14872 } 14873 } 14874 14875 // half VLDR: 2 * imm8 14876 if (VT.isFloatingPoint() && NumBytes == 2 && Subtarget->hasFPRegs16()) 14877 return isShiftedUInt<8, 1>(V); 14878 // VLDR and LDRD: 4 * imm8 14879 if ((VT.isFloatingPoint() && Subtarget->hasVFP2Base()) || NumBytes == 8) 14880 return isShiftedUInt<8, 2>(V); 14881 14882 if (NumBytes == 1 || NumBytes == 2 || NumBytes == 4) { 14883 // + imm12 or - imm8 14884 if (IsNeg) 14885 return isUInt<8>(V); 14886 return isUInt<12>(V); 14887 } 14888 14889 return false; 14890 } 14891 14892 /// isLegalAddressImmediate - Return true if the integer value can be used 14893 /// as the offset of the target addressing mode for load / store of the 14894 /// given type. 14895 static bool isLegalAddressImmediate(int64_t V, EVT VT, 14896 const ARMSubtarget *Subtarget) { 14897 if (V == 0) 14898 return true; 14899 14900 if (!VT.isSimple()) 14901 return false; 14902 14903 if (Subtarget->isThumb1Only()) 14904 return isLegalT1AddressImmediate(V, VT); 14905 else if (Subtarget->isThumb2()) 14906 return isLegalT2AddressImmediate(V, VT, Subtarget); 14907 14908 // ARM mode. 14909 if (V < 0) 14910 V = - V; 14911 switch (VT.getSimpleVT().SimpleTy) { 14912 default: return false; 14913 case MVT::i1: 14914 case MVT::i8: 14915 case MVT::i32: 14916 // +- imm12 14917 return isUInt<12>(V); 14918 case MVT::i16: 14919 // +- imm8 14920 return isUInt<8>(V); 14921 case MVT::f32: 14922 case MVT::f64: 14923 if (!Subtarget->hasVFP2Base()) // FIXME: NEON? 14924 return false; 14925 return isShiftedUInt<8, 2>(V); 14926 } 14927 } 14928 14929 bool ARMTargetLowering::isLegalT2ScaledAddressingMode(const AddrMode &AM, 14930 EVT VT) const { 14931 int Scale = AM.Scale; 14932 if (Scale < 0) 14933 return false; 14934 14935 switch (VT.getSimpleVT().SimpleTy) { 14936 default: return false; 14937 case MVT::i1: 14938 case MVT::i8: 14939 case MVT::i16: 14940 case MVT::i32: 14941 if (Scale == 1) 14942 return true; 14943 // r + r << imm 14944 Scale = Scale & ~1; 14945 return Scale == 2 || Scale == 4 || Scale == 8; 14946 case MVT::i64: 14947 // FIXME: What are we trying to model here? ldrd doesn't have an r + r 14948 // version in Thumb mode. 14949 // r + r 14950 if (Scale == 1) 14951 return true; 14952 // r * 2 (this can be lowered to r + r). 14953 if (!AM.HasBaseReg && Scale == 2) 14954 return true; 14955 return false; 14956 case MVT::isVoid: 14957 // Note, we allow "void" uses (basically, uses that aren't loads or 14958 // stores), because arm allows folding a scale into many arithmetic 14959 // operations. This should be made more precise and revisited later. 14960 14961 // Allow r << imm, but the imm has to be a multiple of two. 14962 if (Scale & 1) return false; 14963 return isPowerOf2_32(Scale); 14964 } 14965 } 14966 14967 bool ARMTargetLowering::isLegalT1ScaledAddressingMode(const AddrMode &AM, 14968 EVT VT) const { 14969 const int Scale = AM.Scale; 14970 14971 // Negative scales are not supported in Thumb1. 14972 if (Scale < 0) 14973 return false; 14974 14975 // Thumb1 addressing modes do not support register scaling excepting the 14976 // following cases: 14977 // 1. Scale == 1 means no scaling. 14978 // 2. Scale == 2 this can be lowered to r + r if there is no base register. 14979 return (Scale == 1) || (!AM.HasBaseReg && Scale == 2); 14980 } 14981 14982 /// isLegalAddressingMode - Return true if the addressing mode represented 14983 /// by AM is legal for this target, for a load/store of the specified type. 14984 bool ARMTargetLowering::isLegalAddressingMode(const DataLayout &DL, 14985 const AddrMode &AM, Type *Ty, 14986 unsigned AS, Instruction *I) const { 14987 EVT VT = getValueType(DL, Ty, true); 14988 if (!isLegalAddressImmediate(AM.BaseOffs, VT, Subtarget)) 14989 return false; 14990 14991 // Can never fold addr of global into load/store. 14992 if (AM.BaseGV) 14993 return false; 14994 14995 switch (AM.Scale) { 14996 case 0: // no scale reg, must be "r+i" or "r", or "i". 14997 break; 14998 default: 14999 // ARM doesn't support any R+R*scale+imm addr modes. 15000 if (AM.BaseOffs) 15001 return false; 15002 15003 if (!VT.isSimple()) 15004 return false; 15005 15006 if (Subtarget->isThumb1Only()) 15007 return isLegalT1ScaledAddressingMode(AM, VT); 15008 15009 if (Subtarget->isThumb2()) 15010 return isLegalT2ScaledAddressingMode(AM, VT); 15011 15012 int Scale = AM.Scale; 15013 switch (VT.getSimpleVT().SimpleTy) { 15014 default: return false; 15015 case MVT::i1: 15016 case MVT::i8: 15017 case MVT::i32: 15018 if (Scale < 0) Scale = -Scale; 15019 if (Scale == 1) 15020 return true; 15021 // r + r << imm 15022 return isPowerOf2_32(Scale & ~1); 15023 case MVT::i16: 15024 case MVT::i64: 15025 // r +/- r 15026 if (Scale == 1 || (AM.HasBaseReg && Scale == -1)) 15027 return true; 15028 // r * 2 (this can be lowered to r + r). 15029 if (!AM.HasBaseReg && Scale == 2) 15030 return true; 15031 return false; 15032 15033 case MVT::isVoid: 15034 // Note, we allow "void" uses (basically, uses that aren't loads or 15035 // stores), because arm allows folding a scale into many arithmetic 15036 // operations. This should be made more precise and revisited later. 15037 15038 // Allow r << imm, but the imm has to be a multiple of two. 15039 if (Scale & 1) return false; 15040 return isPowerOf2_32(Scale); 15041 } 15042 } 15043 return true; 15044 } 15045 15046 /// isLegalICmpImmediate - Return true if the specified immediate is legal 15047 /// icmp immediate, that is the target has icmp instructions which can compare 15048 /// a register against the immediate without having to materialize the 15049 /// immediate into a register. 15050 bool ARMTargetLowering::isLegalICmpImmediate(int64_t Imm) const { 15051 // Thumb2 and ARM modes can use cmn for negative immediates. 15052 if (!Subtarget->isThumb()) 15053 return ARM_AM::getSOImmVal((uint32_t)Imm) != -1 || 15054 ARM_AM::getSOImmVal(-(uint32_t)Imm) != -1; 15055 if (Subtarget->isThumb2()) 15056 return ARM_AM::getT2SOImmVal((uint32_t)Imm) != -1 || 15057 ARM_AM::getT2SOImmVal(-(uint32_t)Imm) != -1; 15058 // Thumb1 doesn't have cmn, and only 8-bit immediates. 15059 return Imm >= 0 && Imm <= 255; 15060 } 15061 15062 /// isLegalAddImmediate - Return true if the specified immediate is a legal add 15063 /// *or sub* immediate, that is the target has add or sub instructions which can 15064 /// add a register with the immediate without having to materialize the 15065 /// immediate into a register. 15066 bool ARMTargetLowering::isLegalAddImmediate(int64_t Imm) const { 15067 // Same encoding for add/sub, just flip the sign. 15068 int64_t AbsImm = std::abs(Imm); 15069 if (!Subtarget->isThumb()) 15070 return ARM_AM::getSOImmVal(AbsImm) != -1; 15071 if (Subtarget->isThumb2()) 15072 return ARM_AM::getT2SOImmVal(AbsImm) != -1; 15073 // Thumb1 only has 8-bit unsigned immediate. 15074 return AbsImm >= 0 && AbsImm <= 255; 15075 } 15076 15077 static bool getARMIndexedAddressParts(SDNode *Ptr, EVT VT, 15078 bool isSEXTLoad, SDValue &Base, 15079 SDValue &Offset, bool &isInc, 15080 SelectionDAG &DAG) { 15081 if (Ptr->getOpcode() != ISD::ADD && Ptr->getOpcode() != ISD::SUB) 15082 return false; 15083 15084 if (VT == MVT::i16 || ((VT == MVT::i8 || VT == MVT::i1) && isSEXTLoad)) { 15085 // AddressingMode 3 15086 Base = Ptr->getOperand(0); 15087 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(Ptr->getOperand(1))) { 15088 int RHSC = (int)RHS->getZExtValue(); 15089 if (RHSC < 0 && RHSC > -256) { 15090 assert(Ptr->getOpcode() == ISD::ADD); 15091 isInc = false; 15092 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0)); 15093 return true; 15094 } 15095 } 15096 isInc = (Ptr->getOpcode() == ISD::ADD); 15097 Offset = Ptr->getOperand(1); 15098 return true; 15099 } else if (VT == MVT::i32 || VT == MVT::i8 || VT == MVT::i1) { 15100 // AddressingMode 2 15101 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(Ptr->getOperand(1))) { 15102 int RHSC = (int)RHS->getZExtValue(); 15103 if (RHSC < 0 && RHSC > -0x1000) { 15104 assert(Ptr->getOpcode() == ISD::ADD); 15105 isInc = false; 15106 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0)); 15107 Base = Ptr->getOperand(0); 15108 return true; 15109 } 15110 } 15111 15112 if (Ptr->getOpcode() == ISD::ADD) { 15113 isInc = true; 15114 ARM_AM::ShiftOpc ShOpcVal= 15115 ARM_AM::getShiftOpcForNode(Ptr->getOperand(0).getOpcode()); 15116 if (ShOpcVal != ARM_AM::no_shift) { 15117 Base = Ptr->getOperand(1); 15118 Offset = Ptr->getOperand(0); 15119 } else { 15120 Base = Ptr->getOperand(0); 15121 Offset = Ptr->getOperand(1); 15122 } 15123 return true; 15124 } 15125 15126 isInc = (Ptr->getOpcode() == ISD::ADD); 15127 Base = Ptr->getOperand(0); 15128 Offset = Ptr->getOperand(1); 15129 return true; 15130 } 15131 15132 // FIXME: Use VLDM / VSTM to emulate indexed FP load / store. 15133 return false; 15134 } 15135 15136 static bool getT2IndexedAddressParts(SDNode *Ptr, EVT VT, 15137 bool isSEXTLoad, SDValue &Base, 15138 SDValue &Offset, bool &isInc, 15139 SelectionDAG &DAG) { 15140 if (Ptr->getOpcode() != ISD::ADD && Ptr->getOpcode() != ISD::SUB) 15141 return false; 15142 15143 Base = Ptr->getOperand(0); 15144 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(Ptr->getOperand(1))) { 15145 int RHSC = (int)RHS->getZExtValue(); 15146 if (RHSC < 0 && RHSC > -0x100) { // 8 bits. 15147 assert(Ptr->getOpcode() == ISD::ADD); 15148 isInc = false; 15149 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0)); 15150 return true; 15151 } else if (RHSC > 0 && RHSC < 0x100) { // 8 bit, no zero. 15152 isInc = Ptr->getOpcode() == ISD::ADD; 15153 Offset = DAG.getConstant(RHSC, SDLoc(Ptr), RHS->getValueType(0)); 15154 return true; 15155 } 15156 } 15157 15158 return false; 15159 } 15160 15161 static bool getMVEIndexedAddressParts(SDNode *Ptr, EVT VT, unsigned Align, 15162 bool isSEXTLoad, bool isLE, SDValue &Base, 15163 SDValue &Offset, bool &isInc, 15164 SelectionDAG &DAG) { 15165 if (Ptr->getOpcode() != ISD::ADD && Ptr->getOpcode() != ISD::SUB) 15166 return false; 15167 if (!isa<ConstantSDNode>(Ptr->getOperand(1))) 15168 return false; 15169 15170 ConstantSDNode *RHS = cast<ConstantSDNode>(Ptr->getOperand(1)); 15171 int RHSC = (int)RHS->getZExtValue(); 15172 15173 auto IsInRange = [&](int RHSC, int Limit, int Scale) { 15174 if (RHSC < 0 && RHSC > -Limit * Scale && RHSC % Scale == 0) { 15175 assert(Ptr->getOpcode() == ISD::ADD); 15176 isInc = false; 15177 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0)); 15178 return true; 15179 } else if (RHSC > 0 && RHSC < Limit * Scale && RHSC % Scale == 0) { 15180 isInc = Ptr->getOpcode() == ISD::ADD; 15181 Offset = DAG.getConstant(RHSC, SDLoc(Ptr), RHS->getValueType(0)); 15182 return true; 15183 } 15184 return false; 15185 }; 15186 15187 // Try to find a matching instruction based on s/zext, Alignment, Offset and 15188 // (in BE) type. 15189 Base = Ptr->getOperand(0); 15190 if (VT == MVT::v4i16) { 15191 if (Align >= 2 && IsInRange(RHSC, 0x80, 2)) 15192 return true; 15193 } else if (VT == MVT::v4i8 || VT == MVT::v8i8) { 15194 if (IsInRange(RHSC, 0x80, 1)) 15195 return true; 15196 } else if (Align >= 4 && (isLE || VT == MVT::v4i32 || VT == MVT::v4f32) && 15197 IsInRange(RHSC, 0x80, 4)) 15198 return true; 15199 else if (Align >= 2 && (isLE || VT == MVT::v8i16 || VT == MVT::v8f16) && 15200 IsInRange(RHSC, 0x80, 2)) 15201 return true; 15202 else if ((isLE || VT == MVT::v16i8) && IsInRange(RHSC, 0x80, 1)) 15203 return true; 15204 return false; 15205 } 15206 15207 /// getPreIndexedAddressParts - returns true by value, base pointer and 15208 /// offset pointer and addressing mode by reference if the node's address 15209 /// can be legally represented as pre-indexed load / store address. 15210 bool 15211 ARMTargetLowering::getPreIndexedAddressParts(SDNode *N, SDValue &Base, 15212 SDValue &Offset, 15213 ISD::MemIndexedMode &AM, 15214 SelectionDAG &DAG) const { 15215 if (Subtarget->isThumb1Only()) 15216 return false; 15217 15218 EVT VT; 15219 SDValue Ptr; 15220 unsigned Align; 15221 bool isSEXTLoad = false; 15222 if (LoadSDNode *LD = dyn_cast<LoadSDNode>(N)) { 15223 Ptr = LD->getBasePtr(); 15224 VT = LD->getMemoryVT(); 15225 Align = LD->getAlignment(); 15226 isSEXTLoad = LD->getExtensionType() == ISD::SEXTLOAD; 15227 } else if (StoreSDNode *ST = dyn_cast<StoreSDNode>(N)) { 15228 Ptr = ST->getBasePtr(); 15229 VT = ST->getMemoryVT(); 15230 Align = ST->getAlignment(); 15231 } else 15232 return false; 15233 15234 bool isInc; 15235 bool isLegal = false; 15236 if (VT.isVector()) 15237 isLegal = Subtarget->hasMVEIntegerOps() && 15238 getMVEIndexedAddressParts(Ptr.getNode(), VT, Align, isSEXTLoad, 15239 Subtarget->isLittle(), Base, Offset, 15240 isInc, DAG); 15241 else { 15242 if (Subtarget->isThumb2()) 15243 isLegal = getT2IndexedAddressParts(Ptr.getNode(), VT, isSEXTLoad, Base, 15244 Offset, isInc, DAG); 15245 else 15246 isLegal = getARMIndexedAddressParts(Ptr.getNode(), VT, isSEXTLoad, Base, 15247 Offset, isInc, DAG); 15248 } 15249 if (!isLegal) 15250 return false; 15251 15252 AM = isInc ? ISD::PRE_INC : ISD::PRE_DEC; 15253 return true; 15254 } 15255 15256 /// getPostIndexedAddressParts - returns true by value, base pointer and 15257 /// offset pointer and addressing mode by reference if this node can be 15258 /// combined with a load / store to form a post-indexed load / store. 15259 bool ARMTargetLowering::getPostIndexedAddressParts(SDNode *N, SDNode *Op, 15260 SDValue &Base, 15261 SDValue &Offset, 15262 ISD::MemIndexedMode &AM, 15263 SelectionDAG &DAG) const { 15264 EVT VT; 15265 SDValue Ptr; 15266 unsigned Align; 15267 bool isSEXTLoad = false, isNonExt; 15268 if (LoadSDNode *LD = dyn_cast<LoadSDNode>(N)) { 15269 VT = LD->getMemoryVT(); 15270 Ptr = LD->getBasePtr(); 15271 Align = LD->getAlignment(); 15272 isSEXTLoad = LD->getExtensionType() == ISD::SEXTLOAD; 15273 isNonExt = LD->getExtensionType() == ISD::NON_EXTLOAD; 15274 } else if (StoreSDNode *ST = dyn_cast<StoreSDNode>(N)) { 15275 VT = ST->getMemoryVT(); 15276 Ptr = ST->getBasePtr(); 15277 Align = ST->getAlignment(); 15278 isNonExt = !ST->isTruncatingStore(); 15279 } else 15280 return false; 15281 15282 if (Subtarget->isThumb1Only()) { 15283 // Thumb-1 can do a limited post-inc load or store as an updating LDM. It 15284 // must be non-extending/truncating, i32, with an offset of 4. 15285 assert(Op->getValueType(0) == MVT::i32 && "Non-i32 post-inc op?!"); 15286 if (Op->getOpcode() != ISD::ADD || !isNonExt) 15287 return false; 15288 auto *RHS = dyn_cast<ConstantSDNode>(Op->getOperand(1)); 15289 if (!RHS || RHS->getZExtValue() != 4) 15290 return false; 15291 15292 Offset = Op->getOperand(1); 15293 Base = Op->getOperand(0); 15294 AM = ISD::POST_INC; 15295 return true; 15296 } 15297 15298 bool isInc; 15299 bool isLegal = false; 15300 if (VT.isVector()) 15301 isLegal = Subtarget->hasMVEIntegerOps() && 15302 getMVEIndexedAddressParts(Op, VT, Align, isSEXTLoad, 15303 Subtarget->isLittle(), Base, Offset, 15304 isInc, DAG); 15305 else { 15306 if (Subtarget->isThumb2()) 15307 isLegal = getT2IndexedAddressParts(Op, VT, isSEXTLoad, Base, Offset, 15308 isInc, DAG); 15309 else 15310 isLegal = getARMIndexedAddressParts(Op, VT, isSEXTLoad, Base, Offset, 15311 isInc, DAG); 15312 } 15313 if (!isLegal) 15314 return false; 15315 15316 if (Ptr != Base) { 15317 // Swap base ptr and offset to catch more post-index load / store when 15318 // it's legal. In Thumb2 mode, offset must be an immediate. 15319 if (Ptr == Offset && Op->getOpcode() == ISD::ADD && 15320 !Subtarget->isThumb2()) 15321 std::swap(Base, Offset); 15322 15323 // Post-indexed load / store update the base pointer. 15324 if (Ptr != Base) 15325 return false; 15326 } 15327 15328 AM = isInc ? ISD::POST_INC : ISD::POST_DEC; 15329 return true; 15330 } 15331 15332 void ARMTargetLowering::computeKnownBitsForTargetNode(const SDValue Op, 15333 KnownBits &Known, 15334 const APInt &DemandedElts, 15335 const SelectionDAG &DAG, 15336 unsigned Depth) const { 15337 unsigned BitWidth = Known.getBitWidth(); 15338 Known.resetAll(); 15339 switch (Op.getOpcode()) { 15340 default: break; 15341 case ARMISD::ADDC: 15342 case ARMISD::ADDE: 15343 case ARMISD::SUBC: 15344 case ARMISD::SUBE: 15345 // Special cases when we convert a carry to a boolean. 15346 if (Op.getResNo() == 0) { 15347 SDValue LHS = Op.getOperand(0); 15348 SDValue RHS = Op.getOperand(1); 15349 // (ADDE 0, 0, C) will give us a single bit. 15350 if (Op->getOpcode() == ARMISD::ADDE && isNullConstant(LHS) && 15351 isNullConstant(RHS)) { 15352 Known.Zero |= APInt::getHighBitsSet(BitWidth, BitWidth - 1); 15353 return; 15354 } 15355 } 15356 break; 15357 case ARMISD::CMOV: { 15358 // Bits are known zero/one if known on the LHS and RHS. 15359 Known = DAG.computeKnownBits(Op.getOperand(0), Depth+1); 15360 if (Known.isUnknown()) 15361 return; 15362 15363 KnownBits KnownRHS = DAG.computeKnownBits(Op.getOperand(1), Depth+1); 15364 Known.Zero &= KnownRHS.Zero; 15365 Known.One &= KnownRHS.One; 15366 return; 15367 } 15368 case ISD::INTRINSIC_W_CHAIN: { 15369 ConstantSDNode *CN = cast<ConstantSDNode>(Op->getOperand(1)); 15370 Intrinsic::ID IntID = static_cast<Intrinsic::ID>(CN->getZExtValue()); 15371 switch (IntID) { 15372 default: return; 15373 case Intrinsic::arm_ldaex: 15374 case Intrinsic::arm_ldrex: { 15375 EVT VT = cast<MemIntrinsicSDNode>(Op)->getMemoryVT(); 15376 unsigned MemBits = VT.getScalarSizeInBits(); 15377 Known.Zero |= APInt::getHighBitsSet(BitWidth, BitWidth - MemBits); 15378 return; 15379 } 15380 } 15381 } 15382 case ARMISD::BFI: { 15383 // Conservatively, we can recurse down the first operand 15384 // and just mask out all affected bits. 15385 Known = DAG.computeKnownBits(Op.getOperand(0), Depth + 1); 15386 15387 // The operand to BFI is already a mask suitable for removing the bits it 15388 // sets. 15389 ConstantSDNode *CI = cast<ConstantSDNode>(Op.getOperand(2)); 15390 const APInt &Mask = CI->getAPIntValue(); 15391 Known.Zero &= Mask; 15392 Known.One &= Mask; 15393 return; 15394 } 15395 case ARMISD::VGETLANEs: 15396 case ARMISD::VGETLANEu: { 15397 const SDValue &SrcSV = Op.getOperand(0); 15398 EVT VecVT = SrcSV.getValueType(); 15399 assert(VecVT.isVector() && "VGETLANE expected a vector type"); 15400 const unsigned NumSrcElts = VecVT.getVectorNumElements(); 15401 ConstantSDNode *Pos = cast<ConstantSDNode>(Op.getOperand(1).getNode()); 15402 assert(Pos->getAPIntValue().ult(NumSrcElts) && 15403 "VGETLANE index out of bounds"); 15404 unsigned Idx = Pos->getZExtValue(); 15405 APInt DemandedElt = APInt::getOneBitSet(NumSrcElts, Idx); 15406 Known = DAG.computeKnownBits(SrcSV, DemandedElt, Depth + 1); 15407 15408 EVT VT = Op.getValueType(); 15409 const unsigned DstSz = VT.getScalarSizeInBits(); 15410 const unsigned SrcSz = VecVT.getVectorElementType().getSizeInBits(); 15411 (void)SrcSz; 15412 assert(SrcSz == Known.getBitWidth()); 15413 assert(DstSz > SrcSz); 15414 if (Op.getOpcode() == ARMISD::VGETLANEs) 15415 Known = Known.sext(DstSz); 15416 else { 15417 Known = Known.zext(DstSz, true /* extended bits are known zero */); 15418 } 15419 assert(DstSz == Known.getBitWidth()); 15420 break; 15421 } 15422 } 15423 } 15424 15425 bool 15426 ARMTargetLowering::targetShrinkDemandedConstant(SDValue Op, 15427 const APInt &DemandedAPInt, 15428 TargetLoweringOpt &TLO) const { 15429 // Delay optimization, so we don't have to deal with illegal types, or block 15430 // optimizations. 15431 if (!TLO.LegalOps) 15432 return false; 15433 15434 // Only optimize AND for now. 15435 if (Op.getOpcode() != ISD::AND) 15436 return false; 15437 15438 EVT VT = Op.getValueType(); 15439 15440 // Ignore vectors. 15441 if (VT.isVector()) 15442 return false; 15443 15444 assert(VT == MVT::i32 && "Unexpected integer type"); 15445 15446 // Make sure the RHS really is a constant. 15447 ConstantSDNode *C = dyn_cast<ConstantSDNode>(Op.getOperand(1)); 15448 if (!C) 15449 return false; 15450 15451 unsigned Mask = C->getZExtValue(); 15452 15453 unsigned Demanded = DemandedAPInt.getZExtValue(); 15454 unsigned ShrunkMask = Mask & Demanded; 15455 unsigned ExpandedMask = Mask | ~Demanded; 15456 15457 // If the mask is all zeros, let the target-independent code replace the 15458 // result with zero. 15459 if (ShrunkMask == 0) 15460 return false; 15461 15462 // If the mask is all ones, erase the AND. (Currently, the target-independent 15463 // code won't do this, so we have to do it explicitly to avoid an infinite 15464 // loop in obscure cases.) 15465 if (ExpandedMask == ~0U) 15466 return TLO.CombineTo(Op, Op.getOperand(0)); 15467 15468 auto IsLegalMask = [ShrunkMask, ExpandedMask](unsigned Mask) -> bool { 15469 return (ShrunkMask & Mask) == ShrunkMask && (~ExpandedMask & Mask) == 0; 15470 }; 15471 auto UseMask = [Mask, Op, VT, &TLO](unsigned NewMask) -> bool { 15472 if (NewMask == Mask) 15473 return true; 15474 SDLoc DL(Op); 15475 SDValue NewC = TLO.DAG.getConstant(NewMask, DL, VT); 15476 SDValue NewOp = TLO.DAG.getNode(ISD::AND, DL, VT, Op.getOperand(0), NewC); 15477 return TLO.CombineTo(Op, NewOp); 15478 }; 15479 15480 // Prefer uxtb mask. 15481 if (IsLegalMask(0xFF)) 15482 return UseMask(0xFF); 15483 15484 // Prefer uxth mask. 15485 if (IsLegalMask(0xFFFF)) 15486 return UseMask(0xFFFF); 15487 15488 // [1, 255] is Thumb1 movs+ands, legal immediate for ARM/Thumb2. 15489 // FIXME: Prefer a contiguous sequence of bits for other optimizations. 15490 if (ShrunkMask < 256) 15491 return UseMask(ShrunkMask); 15492 15493 // [-256, -2] is Thumb1 movs+bics, legal immediate for ARM/Thumb2. 15494 // FIXME: Prefer a contiguous sequence of bits for other optimizations. 15495 if ((int)ExpandedMask <= -2 && (int)ExpandedMask >= -256) 15496 return UseMask(ExpandedMask); 15497 15498 // Potential improvements: 15499 // 15500 // We could try to recognize lsls+lsrs or lsrs+lsls pairs here. 15501 // We could try to prefer Thumb1 immediates which can be lowered to a 15502 // two-instruction sequence. 15503 // We could try to recognize more legal ARM/Thumb2 immediates here. 15504 15505 return false; 15506 } 15507 15508 15509 //===----------------------------------------------------------------------===// 15510 // ARM Inline Assembly Support 15511 //===----------------------------------------------------------------------===// 15512 15513 bool ARMTargetLowering::ExpandInlineAsm(CallInst *CI) const { 15514 // Looking for "rev" which is V6+. 15515 if (!Subtarget->hasV6Ops()) 15516 return false; 15517 15518 InlineAsm *IA = cast<InlineAsm>(CI->getCalledValue()); 15519 std::string AsmStr = IA->getAsmString(); 15520 SmallVector<StringRef, 4> AsmPieces; 15521 SplitString(AsmStr, AsmPieces, ";\n"); 15522 15523 switch (AsmPieces.size()) { 15524 default: return false; 15525 case 1: 15526 AsmStr = AsmPieces[0]; 15527 AsmPieces.clear(); 15528 SplitString(AsmStr, AsmPieces, " \t,"); 15529 15530 // rev $0, $1 15531 if (AsmPieces.size() == 3 && 15532 AsmPieces[0] == "rev" && AsmPieces[1] == "$0" && AsmPieces[2] == "$1" && 15533 IA->getConstraintString().compare(0, 4, "=l,l") == 0) { 15534 IntegerType *Ty = dyn_cast<IntegerType>(CI->getType()); 15535 if (Ty && Ty->getBitWidth() == 32) 15536 return IntrinsicLowering::LowerToByteSwap(CI); 15537 } 15538 break; 15539 } 15540 15541 return false; 15542 } 15543 15544 const char *ARMTargetLowering::LowerXConstraint(EVT ConstraintVT) const { 15545 // At this point, we have to lower this constraint to something else, so we 15546 // lower it to an "r" or "w". However, by doing this we will force the result 15547 // to be in register, while the X constraint is much more permissive. 15548 // 15549 // Although we are correct (we are free to emit anything, without 15550 // constraints), we might break use cases that would expect us to be more 15551 // efficient and emit something else. 15552 if (!Subtarget->hasVFP2Base()) 15553 return "r"; 15554 if (ConstraintVT.isFloatingPoint()) 15555 return "w"; 15556 if (ConstraintVT.isVector() && Subtarget->hasNEON() && 15557 (ConstraintVT.getSizeInBits() == 64 || 15558 ConstraintVT.getSizeInBits() == 128)) 15559 return "w"; 15560 15561 return "r"; 15562 } 15563 15564 /// getConstraintType - Given a constraint letter, return the type of 15565 /// constraint it is for this target. 15566 ARMTargetLowering::ConstraintType 15567 ARMTargetLowering::getConstraintType(StringRef Constraint) const { 15568 unsigned S = Constraint.size(); 15569 if (S == 1) { 15570 switch (Constraint[0]) { 15571 default: break; 15572 case 'l': return C_RegisterClass; 15573 case 'w': return C_RegisterClass; 15574 case 'h': return C_RegisterClass; 15575 case 'x': return C_RegisterClass; 15576 case 't': return C_RegisterClass; 15577 case 'j': return C_Immediate; // Constant for movw. 15578 // An address with a single base register. Due to the way we 15579 // currently handle addresses it is the same as an 'r' memory constraint. 15580 case 'Q': return C_Memory; 15581 } 15582 } else if (S == 2) { 15583 switch (Constraint[0]) { 15584 default: break; 15585 case 'T': return C_RegisterClass; 15586 // All 'U+' constraints are addresses. 15587 case 'U': return C_Memory; 15588 } 15589 } 15590 return TargetLowering::getConstraintType(Constraint); 15591 } 15592 15593 /// Examine constraint type and operand type and determine a weight value. 15594 /// This object must already have been set up with the operand type 15595 /// and the current alternative constraint selected. 15596 TargetLowering::ConstraintWeight 15597 ARMTargetLowering::getSingleConstraintMatchWeight( 15598 AsmOperandInfo &info, const char *constraint) const { 15599 ConstraintWeight weight = CW_Invalid; 15600 Value *CallOperandVal = info.CallOperandVal; 15601 // If we don't have a value, we can't do a match, 15602 // but allow it at the lowest weight. 15603 if (!CallOperandVal) 15604 return CW_Default; 15605 Type *type = CallOperandVal->getType(); 15606 // Look at the constraint type. 15607 switch (*constraint) { 15608 default: 15609 weight = TargetLowering::getSingleConstraintMatchWeight(info, constraint); 15610 break; 15611 case 'l': 15612 if (type->isIntegerTy()) { 15613 if (Subtarget->isThumb()) 15614 weight = CW_SpecificReg; 15615 else 15616 weight = CW_Register; 15617 } 15618 break; 15619 case 'w': 15620 if (type->isFloatingPointTy()) 15621 weight = CW_Register; 15622 break; 15623 } 15624 return weight; 15625 } 15626 15627 using RCPair = std::pair<unsigned, const TargetRegisterClass *>; 15628 15629 RCPair ARMTargetLowering::getRegForInlineAsmConstraint( 15630 const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const { 15631 switch (Constraint.size()) { 15632 case 1: 15633 // GCC ARM Constraint Letters 15634 switch (Constraint[0]) { 15635 case 'l': // Low regs or general regs. 15636 if (Subtarget->isThumb()) 15637 return RCPair(0U, &ARM::tGPRRegClass); 15638 return RCPair(0U, &ARM::GPRRegClass); 15639 case 'h': // High regs or no regs. 15640 if (Subtarget->isThumb()) 15641 return RCPair(0U, &ARM::hGPRRegClass); 15642 break; 15643 case 'r': 15644 if (Subtarget->isThumb1Only()) 15645 return RCPair(0U, &ARM::tGPRRegClass); 15646 return RCPair(0U, &ARM::GPRRegClass); 15647 case 'w': 15648 if (VT == MVT::Other) 15649 break; 15650 if (VT == MVT::f32) 15651 return RCPair(0U, &ARM::SPRRegClass); 15652 if (VT.getSizeInBits() == 64) 15653 return RCPair(0U, &ARM::DPRRegClass); 15654 if (VT.getSizeInBits() == 128) 15655 return RCPair(0U, &ARM::QPRRegClass); 15656 break; 15657 case 'x': 15658 if (VT == MVT::Other) 15659 break; 15660 if (VT == MVT::f32) 15661 return RCPair(0U, &ARM::SPR_8RegClass); 15662 if (VT.getSizeInBits() == 64) 15663 return RCPair(0U, &ARM::DPR_8RegClass); 15664 if (VT.getSizeInBits() == 128) 15665 return RCPair(0U, &ARM::QPR_8RegClass); 15666 break; 15667 case 't': 15668 if (VT == MVT::Other) 15669 break; 15670 if (VT == MVT::f32 || VT == MVT::i32) 15671 return RCPair(0U, &ARM::SPRRegClass); 15672 if (VT.getSizeInBits() == 64) 15673 return RCPair(0U, &ARM::DPR_VFP2RegClass); 15674 if (VT.getSizeInBits() == 128) 15675 return RCPair(0U, &ARM::QPR_VFP2RegClass); 15676 break; 15677 } 15678 break; 15679 15680 case 2: 15681 if (Constraint[0] == 'T') { 15682 switch (Constraint[1]) { 15683 default: 15684 break; 15685 case 'e': 15686 return RCPair(0U, &ARM::tGPREvenRegClass); 15687 case 'o': 15688 return RCPair(0U, &ARM::tGPROddRegClass); 15689 } 15690 } 15691 break; 15692 15693 default: 15694 break; 15695 } 15696 15697 if (StringRef("{cc}").equals_lower(Constraint)) 15698 return std::make_pair(unsigned(ARM::CPSR), &ARM::CCRRegClass); 15699 15700 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT); 15701 } 15702 15703 /// LowerAsmOperandForConstraint - Lower the specified operand into the Ops 15704 /// vector. If it is invalid, don't add anything to Ops. 15705 void ARMTargetLowering::LowerAsmOperandForConstraint(SDValue Op, 15706 std::string &Constraint, 15707 std::vector<SDValue>&Ops, 15708 SelectionDAG &DAG) const { 15709 SDValue Result; 15710 15711 // Currently only support length 1 constraints. 15712 if (Constraint.length() != 1) return; 15713 15714 char ConstraintLetter = Constraint[0]; 15715 switch (ConstraintLetter) { 15716 default: break; 15717 case 'j': 15718 case 'I': case 'J': case 'K': case 'L': 15719 case 'M': case 'N': case 'O': 15720 ConstantSDNode *C = dyn_cast<ConstantSDNode>(Op); 15721 if (!C) 15722 return; 15723 15724 int64_t CVal64 = C->getSExtValue(); 15725 int CVal = (int) CVal64; 15726 // None of these constraints allow values larger than 32 bits. Check 15727 // that the value fits in an int. 15728 if (CVal != CVal64) 15729 return; 15730 15731 switch (ConstraintLetter) { 15732 case 'j': 15733 // Constant suitable for movw, must be between 0 and 15734 // 65535. 15735 if (Subtarget->hasV6T2Ops() || (Subtarget->hasV8MBaselineOps())) 15736 if (CVal >= 0 && CVal <= 65535) 15737 break; 15738 return; 15739 case 'I': 15740 if (Subtarget->isThumb1Only()) { 15741 // This must be a constant between 0 and 255, for ADD 15742 // immediates. 15743 if (CVal >= 0 && CVal <= 255) 15744 break; 15745 } else if (Subtarget->isThumb2()) { 15746 // A constant that can be used as an immediate value in a 15747 // data-processing instruction. 15748 if (ARM_AM::getT2SOImmVal(CVal) != -1) 15749 break; 15750 } else { 15751 // A constant that can be used as an immediate value in a 15752 // data-processing instruction. 15753 if (ARM_AM::getSOImmVal(CVal) != -1) 15754 break; 15755 } 15756 return; 15757 15758 case 'J': 15759 if (Subtarget->isThumb1Only()) { 15760 // This must be a constant between -255 and -1, for negated ADD 15761 // immediates. This can be used in GCC with an "n" modifier that 15762 // prints the negated value, for use with SUB instructions. It is 15763 // not useful otherwise but is implemented for compatibility. 15764 if (CVal >= -255 && CVal <= -1) 15765 break; 15766 } else { 15767 // This must be a constant between -4095 and 4095. It is not clear 15768 // what this constraint is intended for. Implemented for 15769 // compatibility with GCC. 15770 if (CVal >= -4095 && CVal <= 4095) 15771 break; 15772 } 15773 return; 15774 15775 case 'K': 15776 if (Subtarget->isThumb1Only()) { 15777 // A 32-bit value where only one byte has a nonzero value. Exclude 15778 // zero to match GCC. This constraint is used by GCC internally for 15779 // constants that can be loaded with a move/shift combination. 15780 // It is not useful otherwise but is implemented for compatibility. 15781 if (CVal != 0 && ARM_AM::isThumbImmShiftedVal(CVal)) 15782 break; 15783 } else if (Subtarget->isThumb2()) { 15784 // A constant whose bitwise inverse can be used as an immediate 15785 // value in a data-processing instruction. This can be used in GCC 15786 // with a "B" modifier that prints the inverted value, for use with 15787 // BIC and MVN instructions. It is not useful otherwise but is 15788 // implemented for compatibility. 15789 if (ARM_AM::getT2SOImmVal(~CVal) != -1) 15790 break; 15791 } else { 15792 // A constant whose bitwise inverse can be used as an immediate 15793 // value in a data-processing instruction. This can be used in GCC 15794 // with a "B" modifier that prints the inverted value, for use with 15795 // BIC and MVN instructions. It is not useful otherwise but is 15796 // implemented for compatibility. 15797 if (ARM_AM::getSOImmVal(~CVal) != -1) 15798 break; 15799 } 15800 return; 15801 15802 case 'L': 15803 if (Subtarget->isThumb1Only()) { 15804 // This must be a constant between -7 and 7, 15805 // for 3-operand ADD/SUB immediate instructions. 15806 if (CVal >= -7 && CVal < 7) 15807 break; 15808 } else if (Subtarget->isThumb2()) { 15809 // A constant whose negation can be used as an immediate value in a 15810 // data-processing instruction. This can be used in GCC with an "n" 15811 // modifier that prints the negated value, for use with SUB 15812 // instructions. It is not useful otherwise but is implemented for 15813 // compatibility. 15814 if (ARM_AM::getT2SOImmVal(-CVal) != -1) 15815 break; 15816 } else { 15817 // A constant whose negation can be used as an immediate value in a 15818 // data-processing instruction. This can be used in GCC with an "n" 15819 // modifier that prints the negated value, for use with SUB 15820 // instructions. It is not useful otherwise but is implemented for 15821 // compatibility. 15822 if (ARM_AM::getSOImmVal(-CVal) != -1) 15823 break; 15824 } 15825 return; 15826 15827 case 'M': 15828 if (Subtarget->isThumb1Only()) { 15829 // This must be a multiple of 4 between 0 and 1020, for 15830 // ADD sp + immediate. 15831 if ((CVal >= 0 && CVal <= 1020) && ((CVal & 3) == 0)) 15832 break; 15833 } else { 15834 // A power of two or a constant between 0 and 32. This is used in 15835 // GCC for the shift amount on shifted register operands, but it is 15836 // useful in general for any shift amounts. 15837 if ((CVal >= 0 && CVal <= 32) || ((CVal & (CVal - 1)) == 0)) 15838 break; 15839 } 15840 return; 15841 15842 case 'N': 15843 if (Subtarget->isThumb1Only()) { 15844 // This must be a constant between 0 and 31, for shift amounts. 15845 if (CVal >= 0 && CVal <= 31) 15846 break; 15847 } 15848 return; 15849 15850 case 'O': 15851 if (Subtarget->isThumb1Only()) { 15852 // This must be a multiple of 4 between -508 and 508, for 15853 // ADD/SUB sp = sp + immediate. 15854 if ((CVal >= -508 && CVal <= 508) && ((CVal & 3) == 0)) 15855 break; 15856 } 15857 return; 15858 } 15859 Result = DAG.getTargetConstant(CVal, SDLoc(Op), Op.getValueType()); 15860 break; 15861 } 15862 15863 if (Result.getNode()) { 15864 Ops.push_back(Result); 15865 return; 15866 } 15867 return TargetLowering::LowerAsmOperandForConstraint(Op, Constraint, Ops, DAG); 15868 } 15869 15870 static RTLIB::Libcall getDivRemLibcall( 15871 const SDNode *N, MVT::SimpleValueType SVT) { 15872 assert((N->getOpcode() == ISD::SDIVREM || N->getOpcode() == ISD::UDIVREM || 15873 N->getOpcode() == ISD::SREM || N->getOpcode() == ISD::UREM) && 15874 "Unhandled Opcode in getDivRemLibcall"); 15875 bool isSigned = N->getOpcode() == ISD::SDIVREM || 15876 N->getOpcode() == ISD::SREM; 15877 RTLIB::Libcall LC; 15878 switch (SVT) { 15879 default: llvm_unreachable("Unexpected request for libcall!"); 15880 case MVT::i8: LC = isSigned ? RTLIB::SDIVREM_I8 : RTLIB::UDIVREM_I8; break; 15881 case MVT::i16: LC = isSigned ? RTLIB::SDIVREM_I16 : RTLIB::UDIVREM_I16; break; 15882 case MVT::i32: LC = isSigned ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32; break; 15883 case MVT::i64: LC = isSigned ? RTLIB::SDIVREM_I64 : RTLIB::UDIVREM_I64; break; 15884 } 15885 return LC; 15886 } 15887 15888 static TargetLowering::ArgListTy getDivRemArgList( 15889 const SDNode *N, LLVMContext *Context, const ARMSubtarget *Subtarget) { 15890 assert((N->getOpcode() == ISD::SDIVREM || N->getOpcode() == ISD::UDIVREM || 15891 N->getOpcode() == ISD::SREM || N->getOpcode() == ISD::UREM) && 15892 "Unhandled Opcode in getDivRemArgList"); 15893 bool isSigned = N->getOpcode() == ISD::SDIVREM || 15894 N->getOpcode() == ISD::SREM; 15895 TargetLowering::ArgListTy Args; 15896 TargetLowering::ArgListEntry Entry; 15897 for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) { 15898 EVT ArgVT = N->getOperand(i).getValueType(); 15899 Type *ArgTy = ArgVT.getTypeForEVT(*Context); 15900 Entry.Node = N->getOperand(i); 15901 Entry.Ty = ArgTy; 15902 Entry.IsSExt = isSigned; 15903 Entry.IsZExt = !isSigned; 15904 Args.push_back(Entry); 15905 } 15906 if (Subtarget->isTargetWindows() && Args.size() >= 2) 15907 std::swap(Args[0], Args[1]); 15908 return Args; 15909 } 15910 15911 SDValue ARMTargetLowering::LowerDivRem(SDValue Op, SelectionDAG &DAG) const { 15912 assert((Subtarget->isTargetAEABI() || Subtarget->isTargetAndroid() || 15913 Subtarget->isTargetGNUAEABI() || Subtarget->isTargetMuslAEABI() || 15914 Subtarget->isTargetWindows()) && 15915 "Register-based DivRem lowering only"); 15916 unsigned Opcode = Op->getOpcode(); 15917 assert((Opcode == ISD::SDIVREM || Opcode == ISD::UDIVREM) && 15918 "Invalid opcode for Div/Rem lowering"); 15919 bool isSigned = (Opcode == ISD::SDIVREM); 15920 EVT VT = Op->getValueType(0); 15921 Type *Ty = VT.getTypeForEVT(*DAG.getContext()); 15922 SDLoc dl(Op); 15923 15924 // If the target has hardware divide, use divide + multiply + subtract: 15925 // div = a / b 15926 // rem = a - b * div 15927 // return {div, rem} 15928 // This should be lowered into UDIV/SDIV + MLS later on. 15929 bool hasDivide = Subtarget->isThumb() ? Subtarget->hasDivideInThumbMode() 15930 : Subtarget->hasDivideInARMMode(); 15931 if (hasDivide && Op->getValueType(0).isSimple() && 15932 Op->getSimpleValueType(0) == MVT::i32) { 15933 unsigned DivOpcode = isSigned ? ISD::SDIV : ISD::UDIV; 15934 const SDValue Dividend = Op->getOperand(0); 15935 const SDValue Divisor = Op->getOperand(1); 15936 SDValue Div = DAG.getNode(DivOpcode, dl, VT, Dividend, Divisor); 15937 SDValue Mul = DAG.getNode(ISD::MUL, dl, VT, Div, Divisor); 15938 SDValue Rem = DAG.getNode(ISD::SUB, dl, VT, Dividend, Mul); 15939 15940 SDValue Values[2] = {Div, Rem}; 15941 return DAG.getNode(ISD::MERGE_VALUES, dl, DAG.getVTList(VT, VT), Values); 15942 } 15943 15944 RTLIB::Libcall LC = getDivRemLibcall(Op.getNode(), 15945 VT.getSimpleVT().SimpleTy); 15946 SDValue InChain = DAG.getEntryNode(); 15947 15948 TargetLowering::ArgListTy Args = getDivRemArgList(Op.getNode(), 15949 DAG.getContext(), 15950 Subtarget); 15951 15952 SDValue Callee = DAG.getExternalSymbol(getLibcallName(LC), 15953 getPointerTy(DAG.getDataLayout())); 15954 15955 Type *RetTy = StructType::get(Ty, Ty); 15956 15957 if (Subtarget->isTargetWindows()) 15958 InChain = WinDBZCheckDenominator(DAG, Op.getNode(), InChain); 15959 15960 TargetLowering::CallLoweringInfo CLI(DAG); 15961 CLI.setDebugLoc(dl).setChain(InChain) 15962 .setCallee(getLibcallCallingConv(LC), RetTy, Callee, std::move(Args)) 15963 .setInRegister().setSExtResult(isSigned).setZExtResult(!isSigned); 15964 15965 std::pair<SDValue, SDValue> CallInfo = LowerCallTo(CLI); 15966 return CallInfo.first; 15967 } 15968 15969 // Lowers REM using divmod helpers 15970 // see RTABI section 4.2/4.3 15971 SDValue ARMTargetLowering::LowerREM(SDNode *N, SelectionDAG &DAG) const { 15972 // Build return types (div and rem) 15973 std::vector<Type*> RetTyParams; 15974 Type *RetTyElement; 15975 15976 switch (N->getValueType(0).getSimpleVT().SimpleTy) { 15977 default: llvm_unreachable("Unexpected request for libcall!"); 15978 case MVT::i8: RetTyElement = Type::getInt8Ty(*DAG.getContext()); break; 15979 case MVT::i16: RetTyElement = Type::getInt16Ty(*DAG.getContext()); break; 15980 case MVT::i32: RetTyElement = Type::getInt32Ty(*DAG.getContext()); break; 15981 case MVT::i64: RetTyElement = Type::getInt64Ty(*DAG.getContext()); break; 15982 } 15983 15984 RetTyParams.push_back(RetTyElement); 15985 RetTyParams.push_back(RetTyElement); 15986 ArrayRef<Type*> ret = ArrayRef<Type*>(RetTyParams); 15987 Type *RetTy = StructType::get(*DAG.getContext(), ret); 15988 15989 RTLIB::Libcall LC = getDivRemLibcall(N, N->getValueType(0).getSimpleVT(). 15990 SimpleTy); 15991 SDValue InChain = DAG.getEntryNode(); 15992 TargetLowering::ArgListTy Args = getDivRemArgList(N, DAG.getContext(), 15993 Subtarget); 15994 bool isSigned = N->getOpcode() == ISD::SREM; 15995 SDValue Callee = DAG.getExternalSymbol(getLibcallName(LC), 15996 getPointerTy(DAG.getDataLayout())); 15997 15998 if (Subtarget->isTargetWindows()) 15999 InChain = WinDBZCheckDenominator(DAG, N, InChain); 16000 16001 // Lower call 16002 CallLoweringInfo CLI(DAG); 16003 CLI.setChain(InChain) 16004 .setCallee(CallingConv::ARM_AAPCS, RetTy, Callee, std::move(Args)) 16005 .setSExtResult(isSigned).setZExtResult(!isSigned).setDebugLoc(SDLoc(N)); 16006 std::pair<SDValue, SDValue> CallResult = LowerCallTo(CLI); 16007 16008 // Return second (rem) result operand (first contains div) 16009 SDNode *ResNode = CallResult.first.getNode(); 16010 assert(ResNode->getNumOperands() == 2 && "divmod should return two operands"); 16011 return ResNode->getOperand(1); 16012 } 16013 16014 SDValue 16015 ARMTargetLowering::LowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const { 16016 assert(Subtarget->isTargetWindows() && "unsupported target platform"); 16017 SDLoc DL(Op); 16018 16019 // Get the inputs. 16020 SDValue Chain = Op.getOperand(0); 16021 SDValue Size = Op.getOperand(1); 16022 16023 if (DAG.getMachineFunction().getFunction().hasFnAttribute( 16024 "no-stack-arg-probe")) { 16025 unsigned Align = cast<ConstantSDNode>(Op.getOperand(2))->getZExtValue(); 16026 SDValue SP = DAG.getCopyFromReg(Chain, DL, ARM::SP, MVT::i32); 16027 Chain = SP.getValue(1); 16028 SP = DAG.getNode(ISD::SUB, DL, MVT::i32, SP, Size); 16029 if (Align) 16030 SP = DAG.getNode(ISD::AND, DL, MVT::i32, SP.getValue(0), 16031 DAG.getConstant(-(uint64_t)Align, DL, MVT::i32)); 16032 Chain = DAG.getCopyToReg(Chain, DL, ARM::SP, SP); 16033 SDValue Ops[2] = { SP, Chain }; 16034 return DAG.getMergeValues(Ops, DL); 16035 } 16036 16037 SDValue Words = DAG.getNode(ISD::SRL, DL, MVT::i32, Size, 16038 DAG.getConstant(2, DL, MVT::i32)); 16039 16040 SDValue Flag; 16041 Chain = DAG.getCopyToReg(Chain, DL, ARM::R4, Words, Flag); 16042 Flag = Chain.getValue(1); 16043 16044 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue); 16045 Chain = DAG.getNode(ARMISD::WIN__CHKSTK, DL, NodeTys, Chain, Flag); 16046 16047 SDValue NewSP = DAG.getCopyFromReg(Chain, DL, ARM::SP, MVT::i32); 16048 Chain = NewSP.getValue(1); 16049 16050 SDValue Ops[2] = { NewSP, Chain }; 16051 return DAG.getMergeValues(Ops, DL); 16052 } 16053 16054 SDValue ARMTargetLowering::LowerFP_EXTEND(SDValue Op, SelectionDAG &DAG) const { 16055 SDValue SrcVal = Op.getOperand(0); 16056 const unsigned DstSz = Op.getValueType().getSizeInBits(); 16057 const unsigned SrcSz = SrcVal.getValueType().getSizeInBits(); 16058 assert(DstSz > SrcSz && DstSz <= 64 && SrcSz >= 16 && 16059 "Unexpected type for custom-lowering FP_EXTEND"); 16060 16061 assert((!Subtarget->hasFP64() || !Subtarget->hasFPARMv8Base()) && 16062 "With both FP DP and 16, any FP conversion is legal!"); 16063 16064 assert(!(DstSz == 32 && Subtarget->hasFP16()) && 16065 "With FP16, 16 to 32 conversion is legal!"); 16066 16067 // Either we are converting from 16 -> 64, without FP16 and/or 16068 // FP.double-precision or without Armv8-fp. So we must do it in two 16069 // steps. 16070 // Or we are converting from 32 -> 64 without fp.double-precision or 16 -> 32 16071 // without FP16. So we must do a function call. 16072 SDLoc Loc(Op); 16073 RTLIB::Libcall LC; 16074 MakeLibCallOptions CallOptions; 16075 if (SrcSz == 16) { 16076 // Instruction from 16 -> 32 16077 if (Subtarget->hasFP16()) 16078 SrcVal = DAG.getNode(ISD::FP_EXTEND, Loc, MVT::f32, SrcVal); 16079 // Lib call from 16 -> 32 16080 else { 16081 LC = RTLIB::getFPEXT(MVT::f16, MVT::f32); 16082 assert(LC != RTLIB::UNKNOWN_LIBCALL && 16083 "Unexpected type for custom-lowering FP_EXTEND"); 16084 SrcVal = 16085 makeLibCall(DAG, LC, MVT::f32, SrcVal, CallOptions, Loc).first; 16086 } 16087 } 16088 16089 if (DstSz != 64) 16090 return SrcVal; 16091 // For sure now SrcVal is 32 bits 16092 if (Subtarget->hasFP64()) // Instruction from 32 -> 64 16093 return DAG.getNode(ISD::FP_EXTEND, Loc, MVT::f64, SrcVal); 16094 16095 LC = RTLIB::getFPEXT(MVT::f32, MVT::f64); 16096 assert(LC != RTLIB::UNKNOWN_LIBCALL && 16097 "Unexpected type for custom-lowering FP_EXTEND"); 16098 return makeLibCall(DAG, LC, MVT::f64, SrcVal, CallOptions, Loc).first; 16099 } 16100 16101 SDValue ARMTargetLowering::LowerFP_ROUND(SDValue Op, SelectionDAG &DAG) const { 16102 SDValue SrcVal = Op.getOperand(0); 16103 EVT SrcVT = SrcVal.getValueType(); 16104 EVT DstVT = Op.getValueType(); 16105 const unsigned DstSz = Op.getValueType().getSizeInBits(); 16106 const unsigned SrcSz = SrcVT.getSizeInBits(); 16107 (void)DstSz; 16108 assert(DstSz < SrcSz && SrcSz <= 64 && DstSz >= 16 && 16109 "Unexpected type for custom-lowering FP_ROUND"); 16110 16111 assert((!Subtarget->hasFP64() || !Subtarget->hasFPARMv8Base()) && 16112 "With both FP DP and 16, any FP conversion is legal!"); 16113 16114 SDLoc Loc(Op); 16115 16116 // Instruction from 32 -> 16 if hasFP16 is valid 16117 if (SrcSz == 32 && Subtarget->hasFP16()) 16118 return Op; 16119 16120 // Lib call from 32 -> 16 / 64 -> [32, 16] 16121 RTLIB::Libcall LC = RTLIB::getFPROUND(SrcVT, DstVT); 16122 assert(LC != RTLIB::UNKNOWN_LIBCALL && 16123 "Unexpected type for custom-lowering FP_ROUND"); 16124 MakeLibCallOptions CallOptions; 16125 return makeLibCall(DAG, LC, DstVT, SrcVal, CallOptions, Loc).first; 16126 } 16127 16128 void ARMTargetLowering::lowerABS(SDNode *N, SmallVectorImpl<SDValue> &Results, 16129 SelectionDAG &DAG) const { 16130 assert(N->getValueType(0) == MVT::i64 && "Unexpected type (!= i64) on ABS."); 16131 MVT HalfT = MVT::i32; 16132 SDLoc dl(N); 16133 SDValue Hi, Lo, Tmp; 16134 16135 if (!isOperationLegalOrCustom(ISD::ADDCARRY, HalfT) || 16136 !isOperationLegalOrCustom(ISD::UADDO, HalfT)) 16137 return ; 16138 16139 unsigned OpTypeBits = HalfT.getScalarSizeInBits(); 16140 SDVTList VTList = DAG.getVTList(HalfT, MVT::i1); 16141 16142 Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, HalfT, N->getOperand(0), 16143 DAG.getConstant(0, dl, HalfT)); 16144 Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, dl, HalfT, N->getOperand(0), 16145 DAG.getConstant(1, dl, HalfT)); 16146 16147 Tmp = DAG.getNode(ISD::SRA, dl, HalfT, Hi, 16148 DAG.getConstant(OpTypeBits - 1, dl, 16149 getShiftAmountTy(HalfT, DAG.getDataLayout()))); 16150 Lo = DAG.getNode(ISD::UADDO, dl, VTList, Tmp, Lo); 16151 Hi = DAG.getNode(ISD::ADDCARRY, dl, VTList, Tmp, Hi, 16152 SDValue(Lo.getNode(), 1)); 16153 Hi = DAG.getNode(ISD::XOR, dl, HalfT, Tmp, Hi); 16154 Lo = DAG.getNode(ISD::XOR, dl, HalfT, Tmp, Lo); 16155 16156 Results.push_back(Lo); 16157 Results.push_back(Hi); 16158 } 16159 16160 bool 16161 ARMTargetLowering::isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const { 16162 // The ARM target isn't yet aware of offsets. 16163 return false; 16164 } 16165 16166 bool ARM::isBitFieldInvertedMask(unsigned v) { 16167 if (v == 0xffffffff) 16168 return false; 16169 16170 // there can be 1's on either or both "outsides", all the "inside" 16171 // bits must be 0's 16172 return isShiftedMask_32(~v); 16173 } 16174 16175 /// isFPImmLegal - Returns true if the target can instruction select the 16176 /// specified FP immediate natively. If false, the legalizer will 16177 /// materialize the FP immediate as a load from a constant pool. 16178 bool ARMTargetLowering::isFPImmLegal(const APFloat &Imm, EVT VT, 16179 bool ForCodeSize) const { 16180 if (!Subtarget->hasVFP3Base()) 16181 return false; 16182 if (VT == MVT::f16 && Subtarget->hasFullFP16()) 16183 return ARM_AM::getFP16Imm(Imm) != -1; 16184 if (VT == MVT::f32) 16185 return ARM_AM::getFP32Imm(Imm) != -1; 16186 if (VT == MVT::f64 && Subtarget->hasFP64()) 16187 return ARM_AM::getFP64Imm(Imm) != -1; 16188 return false; 16189 } 16190 16191 /// getTgtMemIntrinsic - Represent NEON load and store intrinsics as 16192 /// MemIntrinsicNodes. The associated MachineMemOperands record the alignment 16193 /// specified in the intrinsic calls. 16194 bool ARMTargetLowering::getTgtMemIntrinsic(IntrinsicInfo &Info, 16195 const CallInst &I, 16196 MachineFunction &MF, 16197 unsigned Intrinsic) const { 16198 switch (Intrinsic) { 16199 case Intrinsic::arm_neon_vld1: 16200 case Intrinsic::arm_neon_vld2: 16201 case Intrinsic::arm_neon_vld3: 16202 case Intrinsic::arm_neon_vld4: 16203 case Intrinsic::arm_neon_vld2lane: 16204 case Intrinsic::arm_neon_vld3lane: 16205 case Intrinsic::arm_neon_vld4lane: 16206 case Intrinsic::arm_neon_vld2dup: 16207 case Intrinsic::arm_neon_vld3dup: 16208 case Intrinsic::arm_neon_vld4dup: { 16209 Info.opc = ISD::INTRINSIC_W_CHAIN; 16210 // Conservatively set memVT to the entire set of vectors loaded. 16211 auto &DL = I.getCalledFunction()->getParent()->getDataLayout(); 16212 uint64_t NumElts = DL.getTypeSizeInBits(I.getType()) / 64; 16213 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts); 16214 Info.ptrVal = I.getArgOperand(0); 16215 Info.offset = 0; 16216 Value *AlignArg = I.getArgOperand(I.getNumArgOperands() - 1); 16217 Info.align = MaybeAlign(cast<ConstantInt>(AlignArg)->getZExtValue()); 16218 // volatile loads with NEON intrinsics not supported 16219 Info.flags = MachineMemOperand::MOLoad; 16220 return true; 16221 } 16222 case Intrinsic::arm_neon_vld1x2: 16223 case Intrinsic::arm_neon_vld1x3: 16224 case Intrinsic::arm_neon_vld1x4: { 16225 Info.opc = ISD::INTRINSIC_W_CHAIN; 16226 // Conservatively set memVT to the entire set of vectors loaded. 16227 auto &DL = I.getCalledFunction()->getParent()->getDataLayout(); 16228 uint64_t NumElts = DL.getTypeSizeInBits(I.getType()) / 64; 16229 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts); 16230 Info.ptrVal = I.getArgOperand(I.getNumArgOperands() - 1); 16231 Info.offset = 0; 16232 Info.align.reset(); 16233 // volatile loads with NEON intrinsics not supported 16234 Info.flags = MachineMemOperand::MOLoad; 16235 return true; 16236 } 16237 case Intrinsic::arm_neon_vst1: 16238 case Intrinsic::arm_neon_vst2: 16239 case Intrinsic::arm_neon_vst3: 16240 case Intrinsic::arm_neon_vst4: 16241 case Intrinsic::arm_neon_vst2lane: 16242 case Intrinsic::arm_neon_vst3lane: 16243 case Intrinsic::arm_neon_vst4lane: { 16244 Info.opc = ISD::INTRINSIC_VOID; 16245 // Conservatively set memVT to the entire set of vectors stored. 16246 auto &DL = I.getCalledFunction()->getParent()->getDataLayout(); 16247 unsigned NumElts = 0; 16248 for (unsigned ArgI = 1, ArgE = I.getNumArgOperands(); ArgI < ArgE; ++ArgI) { 16249 Type *ArgTy = I.getArgOperand(ArgI)->getType(); 16250 if (!ArgTy->isVectorTy()) 16251 break; 16252 NumElts += DL.getTypeSizeInBits(ArgTy) / 64; 16253 } 16254 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts); 16255 Info.ptrVal = I.getArgOperand(0); 16256 Info.offset = 0; 16257 Value *AlignArg = I.getArgOperand(I.getNumArgOperands() - 1); 16258 Info.align = MaybeAlign(cast<ConstantInt>(AlignArg)->getZExtValue()); 16259 // volatile stores with NEON intrinsics not supported 16260 Info.flags = MachineMemOperand::MOStore; 16261 return true; 16262 } 16263 case Intrinsic::arm_neon_vst1x2: 16264 case Intrinsic::arm_neon_vst1x3: 16265 case Intrinsic::arm_neon_vst1x4: { 16266 Info.opc = ISD::INTRINSIC_VOID; 16267 // Conservatively set memVT to the entire set of vectors stored. 16268 auto &DL = I.getCalledFunction()->getParent()->getDataLayout(); 16269 unsigned NumElts = 0; 16270 for (unsigned ArgI = 1, ArgE = I.getNumArgOperands(); ArgI < ArgE; ++ArgI) { 16271 Type *ArgTy = I.getArgOperand(ArgI)->getType(); 16272 if (!ArgTy->isVectorTy()) 16273 break; 16274 NumElts += DL.getTypeSizeInBits(ArgTy) / 64; 16275 } 16276 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts); 16277 Info.ptrVal = I.getArgOperand(0); 16278 Info.offset = 0; 16279 Info.align.reset(); 16280 // volatile stores with NEON intrinsics not supported 16281 Info.flags = MachineMemOperand::MOStore; 16282 return true; 16283 } 16284 case Intrinsic::arm_ldaex: 16285 case Intrinsic::arm_ldrex: { 16286 auto &DL = I.getCalledFunction()->getParent()->getDataLayout(); 16287 PointerType *PtrTy = cast<PointerType>(I.getArgOperand(0)->getType()); 16288 Info.opc = ISD::INTRINSIC_W_CHAIN; 16289 Info.memVT = MVT::getVT(PtrTy->getElementType()); 16290 Info.ptrVal = I.getArgOperand(0); 16291 Info.offset = 0; 16292 Info.align = MaybeAlign(DL.getABITypeAlignment(PtrTy->getElementType())); 16293 Info.flags = MachineMemOperand::MOLoad | MachineMemOperand::MOVolatile; 16294 return true; 16295 } 16296 case Intrinsic::arm_stlex: 16297 case Intrinsic::arm_strex: { 16298 auto &DL = I.getCalledFunction()->getParent()->getDataLayout(); 16299 PointerType *PtrTy = cast<PointerType>(I.getArgOperand(1)->getType()); 16300 Info.opc = ISD::INTRINSIC_W_CHAIN; 16301 Info.memVT = MVT::getVT(PtrTy->getElementType()); 16302 Info.ptrVal = I.getArgOperand(1); 16303 Info.offset = 0; 16304 Info.align = MaybeAlign(DL.getABITypeAlignment(PtrTy->getElementType())); 16305 Info.flags = MachineMemOperand::MOStore | MachineMemOperand::MOVolatile; 16306 return true; 16307 } 16308 case Intrinsic::arm_stlexd: 16309 case Intrinsic::arm_strexd: 16310 Info.opc = ISD::INTRINSIC_W_CHAIN; 16311 Info.memVT = MVT::i64; 16312 Info.ptrVal = I.getArgOperand(2); 16313 Info.offset = 0; 16314 Info.align = Align(8); 16315 Info.flags = MachineMemOperand::MOStore | MachineMemOperand::MOVolatile; 16316 return true; 16317 16318 case Intrinsic::arm_ldaexd: 16319 case Intrinsic::arm_ldrexd: 16320 Info.opc = ISD::INTRINSIC_W_CHAIN; 16321 Info.memVT = MVT::i64; 16322 Info.ptrVal = I.getArgOperand(0); 16323 Info.offset = 0; 16324 Info.align = Align(8); 16325 Info.flags = MachineMemOperand::MOLoad | MachineMemOperand::MOVolatile; 16326 return true; 16327 16328 default: 16329 break; 16330 } 16331 16332 return false; 16333 } 16334 16335 /// Returns true if it is beneficial to convert a load of a constant 16336 /// to just the constant itself. 16337 bool ARMTargetLowering::shouldConvertConstantLoadToIntImm(const APInt &Imm, 16338 Type *Ty) const { 16339 assert(Ty->isIntegerTy()); 16340 16341 unsigned Bits = Ty->getPrimitiveSizeInBits(); 16342 if (Bits == 0 || Bits > 32) 16343 return false; 16344 return true; 16345 } 16346 16347 bool ARMTargetLowering::isExtractSubvectorCheap(EVT ResVT, EVT SrcVT, 16348 unsigned Index) const { 16349 if (!isOperationLegalOrCustom(ISD::EXTRACT_SUBVECTOR, ResVT)) 16350 return false; 16351 16352 return (Index == 0 || Index == ResVT.getVectorNumElements()); 16353 } 16354 16355 Instruction* ARMTargetLowering::makeDMB(IRBuilder<> &Builder, 16356 ARM_MB::MemBOpt Domain) const { 16357 Module *M = Builder.GetInsertBlock()->getParent()->getParent(); 16358 16359 // First, if the target has no DMB, see what fallback we can use. 16360 if (!Subtarget->hasDataBarrier()) { 16361 // Some ARMv6 cpus can support data barriers with an mcr instruction. 16362 // Thumb1 and pre-v6 ARM mode use a libcall instead and should never get 16363 // here. 16364 if (Subtarget->hasV6Ops() && !Subtarget->isThumb()) { 16365 Function *MCR = Intrinsic::getDeclaration(M, Intrinsic::arm_mcr); 16366 Value* args[6] = {Builder.getInt32(15), Builder.getInt32(0), 16367 Builder.getInt32(0), Builder.getInt32(7), 16368 Builder.getInt32(10), Builder.getInt32(5)}; 16369 return Builder.CreateCall(MCR, args); 16370 } else { 16371 // Instead of using barriers, atomic accesses on these subtargets use 16372 // libcalls. 16373 llvm_unreachable("makeDMB on a target so old that it has no barriers"); 16374 } 16375 } else { 16376 Function *DMB = Intrinsic::getDeclaration(M, Intrinsic::arm_dmb); 16377 // Only a full system barrier exists in the M-class architectures. 16378 Domain = Subtarget->isMClass() ? ARM_MB::SY : Domain; 16379 Constant *CDomain = Builder.getInt32(Domain); 16380 return Builder.CreateCall(DMB, CDomain); 16381 } 16382 } 16383 16384 // Based on http://www.cl.cam.ac.uk/~pes20/cpp/cpp0xmappings.html 16385 Instruction *ARMTargetLowering::emitLeadingFence(IRBuilder<> &Builder, 16386 Instruction *Inst, 16387 AtomicOrdering Ord) const { 16388 switch (Ord) { 16389 case AtomicOrdering::NotAtomic: 16390 case AtomicOrdering::Unordered: 16391 llvm_unreachable("Invalid fence: unordered/non-atomic"); 16392 case AtomicOrdering::Monotonic: 16393 case AtomicOrdering::Acquire: 16394 return nullptr; // Nothing to do 16395 case AtomicOrdering::SequentiallyConsistent: 16396 if (!Inst->hasAtomicStore()) 16397 return nullptr; // Nothing to do 16398 LLVM_FALLTHROUGH; 16399 case AtomicOrdering::Release: 16400 case AtomicOrdering::AcquireRelease: 16401 if (Subtarget->preferISHSTBarriers()) 16402 return makeDMB(Builder, ARM_MB::ISHST); 16403 // FIXME: add a comment with a link to documentation justifying this. 16404 else 16405 return makeDMB(Builder, ARM_MB::ISH); 16406 } 16407 llvm_unreachable("Unknown fence ordering in emitLeadingFence"); 16408 } 16409 16410 Instruction *ARMTargetLowering::emitTrailingFence(IRBuilder<> &Builder, 16411 Instruction *Inst, 16412 AtomicOrdering Ord) const { 16413 switch (Ord) { 16414 case AtomicOrdering::NotAtomic: 16415 case AtomicOrdering::Unordered: 16416 llvm_unreachable("Invalid fence: unordered/not-atomic"); 16417 case AtomicOrdering::Monotonic: 16418 case AtomicOrdering::Release: 16419 return nullptr; // Nothing to do 16420 case AtomicOrdering::Acquire: 16421 case AtomicOrdering::AcquireRelease: 16422 case AtomicOrdering::SequentiallyConsistent: 16423 return makeDMB(Builder, ARM_MB::ISH); 16424 } 16425 llvm_unreachable("Unknown fence ordering in emitTrailingFence"); 16426 } 16427 16428 // Loads and stores less than 64-bits are already atomic; ones above that 16429 // are doomed anyway, so defer to the default libcall and blame the OS when 16430 // things go wrong. Cortex M doesn't have ldrexd/strexd though, so don't emit 16431 // anything for those. 16432 bool ARMTargetLowering::shouldExpandAtomicStoreInIR(StoreInst *SI) const { 16433 unsigned Size = SI->getValueOperand()->getType()->getPrimitiveSizeInBits(); 16434 return (Size == 64) && !Subtarget->isMClass(); 16435 } 16436 16437 // Loads and stores less than 64-bits are already atomic; ones above that 16438 // are doomed anyway, so defer to the default libcall and blame the OS when 16439 // things go wrong. Cortex M doesn't have ldrexd/strexd though, so don't emit 16440 // anything for those. 16441 // FIXME: ldrd and strd are atomic if the CPU has LPAE (e.g. A15 has that 16442 // guarantee, see DDI0406C ARM architecture reference manual, 16443 // sections A8.8.72-74 LDRD) 16444 TargetLowering::AtomicExpansionKind 16445 ARMTargetLowering::shouldExpandAtomicLoadInIR(LoadInst *LI) const { 16446 unsigned Size = LI->getType()->getPrimitiveSizeInBits(); 16447 return ((Size == 64) && !Subtarget->isMClass()) ? AtomicExpansionKind::LLOnly 16448 : AtomicExpansionKind::None; 16449 } 16450 16451 // For the real atomic operations, we have ldrex/strex up to 32 bits, 16452 // and up to 64 bits on the non-M profiles 16453 TargetLowering::AtomicExpansionKind 16454 ARMTargetLowering::shouldExpandAtomicRMWInIR(AtomicRMWInst *AI) const { 16455 if (AI->isFloatingPointOperation()) 16456 return AtomicExpansionKind::CmpXChg; 16457 16458 unsigned Size = AI->getType()->getPrimitiveSizeInBits(); 16459 bool hasAtomicRMW = !Subtarget->isThumb() || Subtarget->hasV8MBaselineOps(); 16460 return (Size <= (Subtarget->isMClass() ? 32U : 64U) && hasAtomicRMW) 16461 ? AtomicExpansionKind::LLSC 16462 : AtomicExpansionKind::None; 16463 } 16464 16465 TargetLowering::AtomicExpansionKind 16466 ARMTargetLowering::shouldExpandAtomicCmpXchgInIR(AtomicCmpXchgInst *AI) const { 16467 // At -O0, fast-regalloc cannot cope with the live vregs necessary to 16468 // implement cmpxchg without spilling. If the address being exchanged is also 16469 // on the stack and close enough to the spill slot, this can lead to a 16470 // situation where the monitor always gets cleared and the atomic operation 16471 // can never succeed. So at -O0 we need a late-expanded pseudo-inst instead. 16472 bool HasAtomicCmpXchg = 16473 !Subtarget->isThumb() || Subtarget->hasV8MBaselineOps(); 16474 if (getTargetMachine().getOptLevel() != 0 && HasAtomicCmpXchg) 16475 return AtomicExpansionKind::LLSC; 16476 return AtomicExpansionKind::None; 16477 } 16478 16479 bool ARMTargetLowering::shouldInsertFencesForAtomic( 16480 const Instruction *I) const { 16481 return InsertFencesForAtomic; 16482 } 16483 16484 // This has so far only been implemented for MachO. 16485 bool ARMTargetLowering::useLoadStackGuardNode() const { 16486 return Subtarget->isTargetMachO(); 16487 } 16488 16489 void ARMTargetLowering::insertSSPDeclarations(Module &M) const { 16490 if (!Subtarget->getTargetTriple().isWindowsMSVCEnvironment()) 16491 return TargetLowering::insertSSPDeclarations(M); 16492 16493 // MSVC CRT has a global variable holding security cookie. 16494 M.getOrInsertGlobal("__security_cookie", 16495 Type::getInt8PtrTy(M.getContext())); 16496 16497 // MSVC CRT has a function to validate security cookie. 16498 FunctionCallee SecurityCheckCookie = M.getOrInsertFunction( 16499 "__security_check_cookie", Type::getVoidTy(M.getContext()), 16500 Type::getInt8PtrTy(M.getContext())); 16501 if (Function *F = dyn_cast<Function>(SecurityCheckCookie.getCallee())) 16502 F->addAttribute(1, Attribute::AttrKind::InReg); 16503 } 16504 16505 Value *ARMTargetLowering::getSDagStackGuard(const Module &M) const { 16506 // MSVC CRT has a global variable holding security cookie. 16507 if (Subtarget->getTargetTriple().isWindowsMSVCEnvironment()) 16508 return M.getGlobalVariable("__security_cookie"); 16509 return TargetLowering::getSDagStackGuard(M); 16510 } 16511 16512 Function *ARMTargetLowering::getSSPStackGuardCheck(const Module &M) const { 16513 // MSVC CRT has a function to validate security cookie. 16514 if (Subtarget->getTargetTriple().isWindowsMSVCEnvironment()) 16515 return M.getFunction("__security_check_cookie"); 16516 return TargetLowering::getSSPStackGuardCheck(M); 16517 } 16518 16519 bool ARMTargetLowering::canCombineStoreAndExtract(Type *VectorTy, Value *Idx, 16520 unsigned &Cost) const { 16521 // If we do not have NEON, vector types are not natively supported. 16522 if (!Subtarget->hasNEON()) 16523 return false; 16524 16525 // Floating point values and vector values map to the same register file. 16526 // Therefore, although we could do a store extract of a vector type, this is 16527 // better to leave at float as we have more freedom in the addressing mode for 16528 // those. 16529 if (VectorTy->isFPOrFPVectorTy()) 16530 return false; 16531 16532 // If the index is unknown at compile time, this is very expensive to lower 16533 // and it is not possible to combine the store with the extract. 16534 if (!isa<ConstantInt>(Idx)) 16535 return false; 16536 16537 assert(VectorTy->isVectorTy() && "VectorTy is not a vector type"); 16538 unsigned BitWidth = cast<VectorType>(VectorTy)->getBitWidth(); 16539 // We can do a store + vector extract on any vector that fits perfectly in a D 16540 // or Q register. 16541 if (BitWidth == 64 || BitWidth == 128) { 16542 Cost = 0; 16543 return true; 16544 } 16545 return false; 16546 } 16547 16548 bool ARMTargetLowering::isCheapToSpeculateCttz() const { 16549 return Subtarget->hasV6T2Ops(); 16550 } 16551 16552 bool ARMTargetLowering::isCheapToSpeculateCtlz() const { 16553 return Subtarget->hasV6T2Ops(); 16554 } 16555 16556 bool ARMTargetLowering::shouldExpandShift(SelectionDAG &DAG, SDNode *N) const { 16557 return !Subtarget->hasMinSize(); 16558 } 16559 16560 Value *ARMTargetLowering::emitLoadLinked(IRBuilder<> &Builder, Value *Addr, 16561 AtomicOrdering Ord) const { 16562 Module *M = Builder.GetInsertBlock()->getParent()->getParent(); 16563 Type *ValTy = cast<PointerType>(Addr->getType())->getElementType(); 16564 bool IsAcquire = isAcquireOrStronger(Ord); 16565 16566 // Since i64 isn't legal and intrinsics don't get type-lowered, the ldrexd 16567 // intrinsic must return {i32, i32} and we have to recombine them into a 16568 // single i64 here. 16569 if (ValTy->getPrimitiveSizeInBits() == 64) { 16570 Intrinsic::ID Int = 16571 IsAcquire ? Intrinsic::arm_ldaexd : Intrinsic::arm_ldrexd; 16572 Function *Ldrex = Intrinsic::getDeclaration(M, Int); 16573 16574 Addr = Builder.CreateBitCast(Addr, Type::getInt8PtrTy(M->getContext())); 16575 Value *LoHi = Builder.CreateCall(Ldrex, Addr, "lohi"); 16576 16577 Value *Lo = Builder.CreateExtractValue(LoHi, 0, "lo"); 16578 Value *Hi = Builder.CreateExtractValue(LoHi, 1, "hi"); 16579 if (!Subtarget->isLittle()) 16580 std::swap (Lo, Hi); 16581 Lo = Builder.CreateZExt(Lo, ValTy, "lo64"); 16582 Hi = Builder.CreateZExt(Hi, ValTy, "hi64"); 16583 return Builder.CreateOr( 16584 Lo, Builder.CreateShl(Hi, ConstantInt::get(ValTy, 32)), "val64"); 16585 } 16586 16587 Type *Tys[] = { Addr->getType() }; 16588 Intrinsic::ID Int = IsAcquire ? Intrinsic::arm_ldaex : Intrinsic::arm_ldrex; 16589 Function *Ldrex = Intrinsic::getDeclaration(M, Int, Tys); 16590 16591 return Builder.CreateTruncOrBitCast( 16592 Builder.CreateCall(Ldrex, Addr), 16593 cast<PointerType>(Addr->getType())->getElementType()); 16594 } 16595 16596 void ARMTargetLowering::emitAtomicCmpXchgNoStoreLLBalance( 16597 IRBuilder<> &Builder) const { 16598 if (!Subtarget->hasV7Ops()) 16599 return; 16600 Module *M = Builder.GetInsertBlock()->getParent()->getParent(); 16601 Builder.CreateCall(Intrinsic::getDeclaration(M, Intrinsic::arm_clrex)); 16602 } 16603 16604 Value *ARMTargetLowering::emitStoreConditional(IRBuilder<> &Builder, Value *Val, 16605 Value *Addr, 16606 AtomicOrdering Ord) const { 16607 Module *M = Builder.GetInsertBlock()->getParent()->getParent(); 16608 bool IsRelease = isReleaseOrStronger(Ord); 16609 16610 // Since the intrinsics must have legal type, the i64 intrinsics take two 16611 // parameters: "i32, i32". We must marshal Val into the appropriate form 16612 // before the call. 16613 if (Val->getType()->getPrimitiveSizeInBits() == 64) { 16614 Intrinsic::ID Int = 16615 IsRelease ? Intrinsic::arm_stlexd : Intrinsic::arm_strexd; 16616 Function *Strex = Intrinsic::getDeclaration(M, Int); 16617 Type *Int32Ty = Type::getInt32Ty(M->getContext()); 16618 16619 Value *Lo = Builder.CreateTrunc(Val, Int32Ty, "lo"); 16620 Value *Hi = Builder.CreateTrunc(Builder.CreateLShr(Val, 32), Int32Ty, "hi"); 16621 if (!Subtarget->isLittle()) 16622 std::swap(Lo, Hi); 16623 Addr = Builder.CreateBitCast(Addr, Type::getInt8PtrTy(M->getContext())); 16624 return Builder.CreateCall(Strex, {Lo, Hi, Addr}); 16625 } 16626 16627 Intrinsic::ID Int = IsRelease ? Intrinsic::arm_stlex : Intrinsic::arm_strex; 16628 Type *Tys[] = { Addr->getType() }; 16629 Function *Strex = Intrinsic::getDeclaration(M, Int, Tys); 16630 16631 return Builder.CreateCall( 16632 Strex, {Builder.CreateZExtOrBitCast( 16633 Val, Strex->getFunctionType()->getParamType(0)), 16634 Addr}); 16635 } 16636 16637 16638 bool ARMTargetLowering::alignLoopsWithOptSize() const { 16639 return Subtarget->isMClass(); 16640 } 16641 16642 /// A helper function for determining the number of interleaved accesses we 16643 /// will generate when lowering accesses of the given type. 16644 unsigned 16645 ARMTargetLowering::getNumInterleavedAccesses(VectorType *VecTy, 16646 const DataLayout &DL) const { 16647 return (DL.getTypeSizeInBits(VecTy) + 127) / 128; 16648 } 16649 16650 bool ARMTargetLowering::isLegalInterleavedAccessType( 16651 VectorType *VecTy, const DataLayout &DL) const { 16652 16653 unsigned VecSize = DL.getTypeSizeInBits(VecTy); 16654 unsigned ElSize = DL.getTypeSizeInBits(VecTy->getElementType()); 16655 16656 // Ensure the vector doesn't have f16 elements. Even though we could do an 16657 // i16 vldN, we can't hold the f16 vectors and will end up converting via 16658 // f32. 16659 if (VecTy->getElementType()->isHalfTy()) 16660 return false; 16661 16662 // Ensure the number of vector elements is greater than 1. 16663 if (VecTy->getNumElements() < 2) 16664 return false; 16665 16666 // Ensure the element type is legal. 16667 if (ElSize != 8 && ElSize != 16 && ElSize != 32) 16668 return false; 16669 16670 // Ensure the total vector size is 64 or a multiple of 128. Types larger than 16671 // 128 will be split into multiple interleaved accesses. 16672 return VecSize == 64 || VecSize % 128 == 0; 16673 } 16674 16675 unsigned ARMTargetLowering::getMaxSupportedInterleaveFactor() const { 16676 if (Subtarget->hasNEON()) 16677 return 4; 16678 return TargetLoweringBase::getMaxSupportedInterleaveFactor(); 16679 } 16680 16681 /// Lower an interleaved load into a vldN intrinsic. 16682 /// 16683 /// E.g. Lower an interleaved load (Factor = 2): 16684 /// %wide.vec = load <8 x i32>, <8 x i32>* %ptr, align 4 16685 /// %v0 = shuffle %wide.vec, undef, <0, 2, 4, 6> ; Extract even elements 16686 /// %v1 = shuffle %wide.vec, undef, <1, 3, 5, 7> ; Extract odd elements 16687 /// 16688 /// Into: 16689 /// %vld2 = { <4 x i32>, <4 x i32> } call llvm.arm.neon.vld2(%ptr, 4) 16690 /// %vec0 = extractelement { <4 x i32>, <4 x i32> } %vld2, i32 0 16691 /// %vec1 = extractelement { <4 x i32>, <4 x i32> } %vld2, i32 1 16692 bool ARMTargetLowering::lowerInterleavedLoad( 16693 LoadInst *LI, ArrayRef<ShuffleVectorInst *> Shuffles, 16694 ArrayRef<unsigned> Indices, unsigned Factor) const { 16695 assert(Factor >= 2 && Factor <= getMaxSupportedInterleaveFactor() && 16696 "Invalid interleave factor"); 16697 assert(!Shuffles.empty() && "Empty shufflevector input"); 16698 assert(Shuffles.size() == Indices.size() && 16699 "Unmatched number of shufflevectors and indices"); 16700 16701 VectorType *VecTy = Shuffles[0]->getType(); 16702 Type *EltTy = VecTy->getVectorElementType(); 16703 16704 const DataLayout &DL = LI->getModule()->getDataLayout(); 16705 16706 // Skip if we do not have NEON and skip illegal vector types. We can 16707 // "legalize" wide vector types into multiple interleaved accesses as long as 16708 // the vector types are divisible by 128. 16709 if (!Subtarget->hasNEON() || !isLegalInterleavedAccessType(VecTy, DL)) 16710 return false; 16711 16712 unsigned NumLoads = getNumInterleavedAccesses(VecTy, DL); 16713 16714 // A pointer vector can not be the return type of the ldN intrinsics. Need to 16715 // load integer vectors first and then convert to pointer vectors. 16716 if (EltTy->isPointerTy()) 16717 VecTy = 16718 VectorType::get(DL.getIntPtrType(EltTy), VecTy->getVectorNumElements()); 16719 16720 IRBuilder<> Builder(LI); 16721 16722 // The base address of the load. 16723 Value *BaseAddr = LI->getPointerOperand(); 16724 16725 if (NumLoads > 1) { 16726 // If we're going to generate more than one load, reset the sub-vector type 16727 // to something legal. 16728 VecTy = VectorType::get(VecTy->getVectorElementType(), 16729 VecTy->getVectorNumElements() / NumLoads); 16730 16731 // We will compute the pointer operand of each load from the original base 16732 // address using GEPs. Cast the base address to a pointer to the scalar 16733 // element type. 16734 BaseAddr = Builder.CreateBitCast( 16735 BaseAddr, VecTy->getVectorElementType()->getPointerTo( 16736 LI->getPointerAddressSpace())); 16737 } 16738 16739 assert(isTypeLegal(EVT::getEVT(VecTy)) && "Illegal vldN vector type!"); 16740 16741 Type *Int8Ptr = Builder.getInt8PtrTy(LI->getPointerAddressSpace()); 16742 Type *Tys[] = {VecTy, Int8Ptr}; 16743 static const Intrinsic::ID LoadInts[3] = {Intrinsic::arm_neon_vld2, 16744 Intrinsic::arm_neon_vld3, 16745 Intrinsic::arm_neon_vld4}; 16746 Function *VldnFunc = 16747 Intrinsic::getDeclaration(LI->getModule(), LoadInts[Factor - 2], Tys); 16748 16749 // Holds sub-vectors extracted from the load intrinsic return values. The 16750 // sub-vectors are associated with the shufflevector instructions they will 16751 // replace. 16752 DenseMap<ShuffleVectorInst *, SmallVector<Value *, 4>> SubVecs; 16753 16754 for (unsigned LoadCount = 0; LoadCount < NumLoads; ++LoadCount) { 16755 // If we're generating more than one load, compute the base address of 16756 // subsequent loads as an offset from the previous. 16757 if (LoadCount > 0) 16758 BaseAddr = 16759 Builder.CreateConstGEP1_32(VecTy->getVectorElementType(), BaseAddr, 16760 VecTy->getVectorNumElements() * Factor); 16761 16762 SmallVector<Value *, 2> Ops; 16763 Ops.push_back(Builder.CreateBitCast(BaseAddr, Int8Ptr)); 16764 Ops.push_back(Builder.getInt32(LI->getAlignment())); 16765 16766 CallInst *VldN = Builder.CreateCall(VldnFunc, Ops, "vldN"); 16767 16768 // Replace uses of each shufflevector with the corresponding vector loaded 16769 // by ldN. 16770 for (unsigned i = 0; i < Shuffles.size(); i++) { 16771 ShuffleVectorInst *SV = Shuffles[i]; 16772 unsigned Index = Indices[i]; 16773 16774 Value *SubVec = Builder.CreateExtractValue(VldN, Index); 16775 16776 // Convert the integer vector to pointer vector if the element is pointer. 16777 if (EltTy->isPointerTy()) 16778 SubVec = Builder.CreateIntToPtr( 16779 SubVec, VectorType::get(SV->getType()->getVectorElementType(), 16780 VecTy->getVectorNumElements())); 16781 16782 SubVecs[SV].push_back(SubVec); 16783 } 16784 } 16785 16786 // Replace uses of the shufflevector instructions with the sub-vectors 16787 // returned by the load intrinsic. If a shufflevector instruction is 16788 // associated with more than one sub-vector, those sub-vectors will be 16789 // concatenated into a single wide vector. 16790 for (ShuffleVectorInst *SVI : Shuffles) { 16791 auto &SubVec = SubVecs[SVI]; 16792 auto *WideVec = 16793 SubVec.size() > 1 ? concatenateVectors(Builder, SubVec) : SubVec[0]; 16794 SVI->replaceAllUsesWith(WideVec); 16795 } 16796 16797 return true; 16798 } 16799 16800 /// Lower an interleaved store into a vstN intrinsic. 16801 /// 16802 /// E.g. Lower an interleaved store (Factor = 3): 16803 /// %i.vec = shuffle <8 x i32> %v0, <8 x i32> %v1, 16804 /// <0, 4, 8, 1, 5, 9, 2, 6, 10, 3, 7, 11> 16805 /// store <12 x i32> %i.vec, <12 x i32>* %ptr, align 4 16806 /// 16807 /// Into: 16808 /// %sub.v0 = shuffle <8 x i32> %v0, <8 x i32> v1, <0, 1, 2, 3> 16809 /// %sub.v1 = shuffle <8 x i32> %v0, <8 x i32> v1, <4, 5, 6, 7> 16810 /// %sub.v2 = shuffle <8 x i32> %v0, <8 x i32> v1, <8, 9, 10, 11> 16811 /// call void llvm.arm.neon.vst3(%ptr, %sub.v0, %sub.v1, %sub.v2, 4) 16812 /// 16813 /// Note that the new shufflevectors will be removed and we'll only generate one 16814 /// vst3 instruction in CodeGen. 16815 /// 16816 /// Example for a more general valid mask (Factor 3). Lower: 16817 /// %i.vec = shuffle <32 x i32> %v0, <32 x i32> %v1, 16818 /// <4, 32, 16, 5, 33, 17, 6, 34, 18, 7, 35, 19> 16819 /// store <12 x i32> %i.vec, <12 x i32>* %ptr 16820 /// 16821 /// Into: 16822 /// %sub.v0 = shuffle <32 x i32> %v0, <32 x i32> v1, <4, 5, 6, 7> 16823 /// %sub.v1 = shuffle <32 x i32> %v0, <32 x i32> v1, <32, 33, 34, 35> 16824 /// %sub.v2 = shuffle <32 x i32> %v0, <32 x i32> v1, <16, 17, 18, 19> 16825 /// call void llvm.arm.neon.vst3(%ptr, %sub.v0, %sub.v1, %sub.v2, 4) 16826 bool ARMTargetLowering::lowerInterleavedStore(StoreInst *SI, 16827 ShuffleVectorInst *SVI, 16828 unsigned Factor) const { 16829 assert(Factor >= 2 && Factor <= getMaxSupportedInterleaveFactor() && 16830 "Invalid interleave factor"); 16831 16832 VectorType *VecTy = SVI->getType(); 16833 assert(VecTy->getVectorNumElements() % Factor == 0 && 16834 "Invalid interleaved store"); 16835 16836 unsigned LaneLen = VecTy->getVectorNumElements() / Factor; 16837 Type *EltTy = VecTy->getVectorElementType(); 16838 VectorType *SubVecTy = VectorType::get(EltTy, LaneLen); 16839 16840 const DataLayout &DL = SI->getModule()->getDataLayout(); 16841 16842 // Skip if we do not have NEON and skip illegal vector types. We can 16843 // "legalize" wide vector types into multiple interleaved accesses as long as 16844 // the vector types are divisible by 128. 16845 if (!Subtarget->hasNEON() || !isLegalInterleavedAccessType(SubVecTy, DL)) 16846 return false; 16847 16848 unsigned NumStores = getNumInterleavedAccesses(SubVecTy, DL); 16849 16850 Value *Op0 = SVI->getOperand(0); 16851 Value *Op1 = SVI->getOperand(1); 16852 IRBuilder<> Builder(SI); 16853 16854 // StN intrinsics don't support pointer vectors as arguments. Convert pointer 16855 // vectors to integer vectors. 16856 if (EltTy->isPointerTy()) { 16857 Type *IntTy = DL.getIntPtrType(EltTy); 16858 16859 // Convert to the corresponding integer vector. 16860 Type *IntVecTy = 16861 VectorType::get(IntTy, Op0->getType()->getVectorNumElements()); 16862 Op0 = Builder.CreatePtrToInt(Op0, IntVecTy); 16863 Op1 = Builder.CreatePtrToInt(Op1, IntVecTy); 16864 16865 SubVecTy = VectorType::get(IntTy, LaneLen); 16866 } 16867 16868 // The base address of the store. 16869 Value *BaseAddr = SI->getPointerOperand(); 16870 16871 if (NumStores > 1) { 16872 // If we're going to generate more than one store, reset the lane length 16873 // and sub-vector type to something legal. 16874 LaneLen /= NumStores; 16875 SubVecTy = VectorType::get(SubVecTy->getVectorElementType(), LaneLen); 16876 16877 // We will compute the pointer operand of each store from the original base 16878 // address using GEPs. Cast the base address to a pointer to the scalar 16879 // element type. 16880 BaseAddr = Builder.CreateBitCast( 16881 BaseAddr, SubVecTy->getVectorElementType()->getPointerTo( 16882 SI->getPointerAddressSpace())); 16883 } 16884 16885 assert(isTypeLegal(EVT::getEVT(SubVecTy)) && "Illegal vstN vector type!"); 16886 16887 auto Mask = SVI->getShuffleMask(); 16888 16889 Type *Int8Ptr = Builder.getInt8PtrTy(SI->getPointerAddressSpace()); 16890 Type *Tys[] = {Int8Ptr, SubVecTy}; 16891 static const Intrinsic::ID StoreInts[3] = {Intrinsic::arm_neon_vst2, 16892 Intrinsic::arm_neon_vst3, 16893 Intrinsic::arm_neon_vst4}; 16894 16895 for (unsigned StoreCount = 0; StoreCount < NumStores; ++StoreCount) { 16896 // If we generating more than one store, we compute the base address of 16897 // subsequent stores as an offset from the previous. 16898 if (StoreCount > 0) 16899 BaseAddr = Builder.CreateConstGEP1_32(SubVecTy->getVectorElementType(), 16900 BaseAddr, LaneLen * Factor); 16901 16902 SmallVector<Value *, 6> Ops; 16903 Ops.push_back(Builder.CreateBitCast(BaseAddr, Int8Ptr)); 16904 16905 Function *VstNFunc = 16906 Intrinsic::getDeclaration(SI->getModule(), StoreInts[Factor - 2], Tys); 16907 16908 // Split the shufflevector operands into sub vectors for the new vstN call. 16909 for (unsigned i = 0; i < Factor; i++) { 16910 unsigned IdxI = StoreCount * LaneLen * Factor + i; 16911 if (Mask[IdxI] >= 0) { 16912 Ops.push_back(Builder.CreateShuffleVector( 16913 Op0, Op1, createSequentialMask(Builder, Mask[IdxI], LaneLen, 0))); 16914 } else { 16915 unsigned StartMask = 0; 16916 for (unsigned j = 1; j < LaneLen; j++) { 16917 unsigned IdxJ = StoreCount * LaneLen * Factor + j; 16918 if (Mask[IdxJ * Factor + IdxI] >= 0) { 16919 StartMask = Mask[IdxJ * Factor + IdxI] - IdxJ; 16920 break; 16921 } 16922 } 16923 // Note: If all elements in a chunk are undefs, StartMask=0! 16924 // Note: Filling undef gaps with random elements is ok, since 16925 // those elements were being written anyway (with undefs). 16926 // In the case of all undefs we're defaulting to using elems from 0 16927 // Note: StartMask cannot be negative, it's checked in 16928 // isReInterleaveMask 16929 Ops.push_back(Builder.CreateShuffleVector( 16930 Op0, Op1, createSequentialMask(Builder, StartMask, LaneLen, 0))); 16931 } 16932 } 16933 16934 Ops.push_back(Builder.getInt32(SI->getAlignment())); 16935 Builder.CreateCall(VstNFunc, Ops); 16936 } 16937 return true; 16938 } 16939 16940 enum HABaseType { 16941 HA_UNKNOWN = 0, 16942 HA_FLOAT, 16943 HA_DOUBLE, 16944 HA_VECT64, 16945 HA_VECT128 16946 }; 16947 16948 static bool isHomogeneousAggregate(Type *Ty, HABaseType &Base, 16949 uint64_t &Members) { 16950 if (auto *ST = dyn_cast<StructType>(Ty)) { 16951 for (unsigned i = 0; i < ST->getNumElements(); ++i) { 16952 uint64_t SubMembers = 0; 16953 if (!isHomogeneousAggregate(ST->getElementType(i), Base, SubMembers)) 16954 return false; 16955 Members += SubMembers; 16956 } 16957 } else if (auto *AT = dyn_cast<ArrayType>(Ty)) { 16958 uint64_t SubMembers = 0; 16959 if (!isHomogeneousAggregate(AT->getElementType(), Base, SubMembers)) 16960 return false; 16961 Members += SubMembers * AT->getNumElements(); 16962 } else if (Ty->isFloatTy()) { 16963 if (Base != HA_UNKNOWN && Base != HA_FLOAT) 16964 return false; 16965 Members = 1; 16966 Base = HA_FLOAT; 16967 } else if (Ty->isDoubleTy()) { 16968 if (Base != HA_UNKNOWN && Base != HA_DOUBLE) 16969 return false; 16970 Members = 1; 16971 Base = HA_DOUBLE; 16972 } else if (auto *VT = dyn_cast<VectorType>(Ty)) { 16973 Members = 1; 16974 switch (Base) { 16975 case HA_FLOAT: 16976 case HA_DOUBLE: 16977 return false; 16978 case HA_VECT64: 16979 return VT->getBitWidth() == 64; 16980 case HA_VECT128: 16981 return VT->getBitWidth() == 128; 16982 case HA_UNKNOWN: 16983 switch (VT->getBitWidth()) { 16984 case 64: 16985 Base = HA_VECT64; 16986 return true; 16987 case 128: 16988 Base = HA_VECT128; 16989 return true; 16990 default: 16991 return false; 16992 } 16993 } 16994 } 16995 16996 return (Members > 0 && Members <= 4); 16997 } 16998 16999 /// Return the correct alignment for the current calling convention. 17000 Align ARMTargetLowering::getABIAlignmentForCallingConv(Type *ArgTy, 17001 DataLayout DL) const { 17002 const Align ABITypeAlign(DL.getABITypeAlignment(ArgTy)); 17003 if (!ArgTy->isVectorTy()) 17004 return ABITypeAlign; 17005 17006 // Avoid over-aligning vector parameters. It would require realigning the 17007 // stack and waste space for no real benefit. 17008 return std::min(ABITypeAlign, DL.getStackAlignment()); 17009 } 17010 17011 /// Return true if a type is an AAPCS-VFP homogeneous aggregate or one of 17012 /// [N x i32] or [N x i64]. This allows front-ends to skip emitting padding when 17013 /// passing according to AAPCS rules. 17014 bool ARMTargetLowering::functionArgumentNeedsConsecutiveRegisters( 17015 Type *Ty, CallingConv::ID CallConv, bool isVarArg) const { 17016 if (getEffectiveCallingConv(CallConv, isVarArg) != 17017 CallingConv::ARM_AAPCS_VFP) 17018 return false; 17019 17020 HABaseType Base = HA_UNKNOWN; 17021 uint64_t Members = 0; 17022 bool IsHA = isHomogeneousAggregate(Ty, Base, Members); 17023 LLVM_DEBUG(dbgs() << "isHA: " << IsHA << " "; Ty->dump()); 17024 17025 bool IsIntArray = Ty->isArrayTy() && Ty->getArrayElementType()->isIntegerTy(); 17026 return IsHA || IsIntArray; 17027 } 17028 17029 unsigned ARMTargetLowering::getExceptionPointerRegister( 17030 const Constant *PersonalityFn) const { 17031 // Platforms which do not use SjLj EH may return values in these registers 17032 // via the personality function. 17033 return Subtarget->useSjLjEH() ? ARM::NoRegister : ARM::R0; 17034 } 17035 17036 unsigned ARMTargetLowering::getExceptionSelectorRegister( 17037 const Constant *PersonalityFn) const { 17038 // Platforms which do not use SjLj EH may return values in these registers 17039 // via the personality function. 17040 return Subtarget->useSjLjEH() ? ARM::NoRegister : ARM::R1; 17041 } 17042 17043 void ARMTargetLowering::initializeSplitCSR(MachineBasicBlock *Entry) const { 17044 // Update IsSplitCSR in ARMFunctionInfo. 17045 ARMFunctionInfo *AFI = Entry->getParent()->getInfo<ARMFunctionInfo>(); 17046 AFI->setIsSplitCSR(true); 17047 } 17048 17049 void ARMTargetLowering::insertCopiesSplitCSR( 17050 MachineBasicBlock *Entry, 17051 const SmallVectorImpl<MachineBasicBlock *> &Exits) const { 17052 const ARMBaseRegisterInfo *TRI = Subtarget->getRegisterInfo(); 17053 const MCPhysReg *IStart = TRI->getCalleeSavedRegsViaCopy(Entry->getParent()); 17054 if (!IStart) 17055 return; 17056 17057 const TargetInstrInfo *TII = Subtarget->getInstrInfo(); 17058 MachineRegisterInfo *MRI = &Entry->getParent()->getRegInfo(); 17059 MachineBasicBlock::iterator MBBI = Entry->begin(); 17060 for (const MCPhysReg *I = IStart; *I; ++I) { 17061 const TargetRegisterClass *RC = nullptr; 17062 if (ARM::GPRRegClass.contains(*I)) 17063 RC = &ARM::GPRRegClass; 17064 else if (ARM::DPRRegClass.contains(*I)) 17065 RC = &ARM::DPRRegClass; 17066 else 17067 llvm_unreachable("Unexpected register class in CSRsViaCopy!"); 17068 17069 Register NewVR = MRI->createVirtualRegister(RC); 17070 // Create copy from CSR to a virtual register. 17071 // FIXME: this currently does not emit CFI pseudo-instructions, it works 17072 // fine for CXX_FAST_TLS since the C++-style TLS access functions should be 17073 // nounwind. If we want to generalize this later, we may need to emit 17074 // CFI pseudo-instructions. 17075 assert(Entry->getParent()->getFunction().hasFnAttribute( 17076 Attribute::NoUnwind) && 17077 "Function should be nounwind in insertCopiesSplitCSR!"); 17078 Entry->addLiveIn(*I); 17079 BuildMI(*Entry, MBBI, DebugLoc(), TII->get(TargetOpcode::COPY), NewVR) 17080 .addReg(*I); 17081 17082 // Insert the copy-back instructions right before the terminator. 17083 for (auto *Exit : Exits) 17084 BuildMI(*Exit, Exit->getFirstTerminator(), DebugLoc(), 17085 TII->get(TargetOpcode::COPY), *I) 17086 .addReg(NewVR); 17087 } 17088 } 17089 17090 void ARMTargetLowering::finalizeLowering(MachineFunction &MF) const { 17091 MF.getFrameInfo().computeMaxCallFrameSize(MF); 17092 TargetLoweringBase::finalizeLowering(MF); 17093 } 17094