1//===-- AMDGPUInstructions.td - Common instruction defs ---*- tablegen -*-===// 2// 3// The LLVM Compiler Infrastructure 4// 5// This file is distributed under the University of Illinois Open Source 6// License. See LICENSE.TXT for details. 7// 8//===----------------------------------------------------------------------===// 9// 10// This file contains instruction defs that are common to all hw codegen 11// targets. 12// 13//===----------------------------------------------------------------------===// 14 15class AMDGPUInst <dag outs, dag ins, string asm = "", 16 list<dag> pattern = []> : Instruction { 17 field bit isRegisterLoad = 0; 18 field bit isRegisterStore = 0; 19 20 let Namespace = "AMDGPU"; 21 let OutOperandList = outs; 22 let InOperandList = ins; 23 let AsmString = asm; 24 let Pattern = pattern; 25 let Itinerary = NullALU; 26 27 // SoftFail is a field the disassembler can use to provide a way for 28 // instructions to not match without killing the whole decode process. It is 29 // mainly used for ARM, but Tablegen expects this field to exist or it fails 30 // to build the decode table. 31 field bits<64> SoftFail = 0; 32 33 let DecoderNamespace = Namespace; 34 35 let TSFlags{63} = isRegisterLoad; 36 let TSFlags{62} = isRegisterStore; 37} 38 39class AMDGPUShaderInst <dag outs, dag ins, string asm = "", 40 list<dag> pattern = []> : AMDGPUInst<outs, ins, asm, pattern> { 41 42 field bits<32> Inst = 0xffffffff; 43} 44 45def FP16Denormals : Predicate<"Subtarget->hasFP16Denormals()">; 46def FP32Denormals : Predicate<"Subtarget->hasFP32Denormals()">; 47def FP64Denormals : Predicate<"Subtarget->hasFP64Denormals()">; 48def NoFP16Denormals : Predicate<"!Subtarget->hasFP16Denormals()">; 49def NoFP32Denormals : Predicate<"!Subtarget->hasFP32Denormals()">; 50def NoFP64Denormals : Predicate<"!Subtarget->hasFP64Denormals()">; 51def UnsafeFPMath : Predicate<"TM.Options.UnsafeFPMath">; 52def FMA : Predicate<"Subtarget->hasFMA()">; 53 54def InstFlag : OperandWithDefaultOps <i32, (ops (i32 0))>; 55def ADDRIndirect : ComplexPattern<iPTR, 2, "SelectADDRIndirect", [], []>; 56 57def u16ImmTarget : AsmOperandClass { 58 let Name = "U16Imm"; 59 let RenderMethod = "addImmOperands"; 60} 61 62def s16ImmTarget : AsmOperandClass { 63 let Name = "S16Imm"; 64 let RenderMethod = "addImmOperands"; 65} 66 67let OperandType = "OPERAND_IMMEDIATE" in { 68 69def u32imm : Operand<i32> { 70 let PrintMethod = "printU32ImmOperand"; 71} 72 73def u16imm : Operand<i16> { 74 let PrintMethod = "printU16ImmOperand"; 75 let ParserMatchClass = u16ImmTarget; 76} 77 78def s16imm : Operand<i16> { 79 let PrintMethod = "printU16ImmOperand"; 80 let ParserMatchClass = s16ImmTarget; 81} 82 83def u8imm : Operand<i8> { 84 let PrintMethod = "printU8ImmOperand"; 85} 86 87} // End OperandType = "OPERAND_IMMEDIATE" 88 89//===--------------------------------------------------------------------===// 90// Custom Operands 91//===--------------------------------------------------------------------===// 92def brtarget : Operand<OtherVT>; 93 94//===----------------------------------------------------------------------===// 95// Misc. PatFrags 96//===----------------------------------------------------------------------===// 97 98class HasOneUseUnaryOp<SDPatternOperator op> : PatFrag< 99 (ops node:$src0), 100 (op $src0), 101 [{ return N->hasOneUse(); }] 102>; 103 104class HasOneUseBinOp<SDPatternOperator op> : PatFrag< 105 (ops node:$src0, node:$src1), 106 (op $src0, $src1), 107 [{ return N->hasOneUse(); }] 108>; 109 110class HasOneUseTernaryOp<SDPatternOperator op> : PatFrag< 111 (ops node:$src0, node:$src1, node:$src2), 112 (op $src0, $src1, $src2), 113 [{ return N->hasOneUse(); }] 114>; 115 116def trunc_oneuse : HasOneUseUnaryOp<trunc>; 117 118let Properties = [SDNPCommutative, SDNPAssociative] in { 119def smax_oneuse : HasOneUseBinOp<smax>; 120def smin_oneuse : HasOneUseBinOp<smin>; 121def umax_oneuse : HasOneUseBinOp<umax>; 122def umin_oneuse : HasOneUseBinOp<umin>; 123def fminnum_oneuse : HasOneUseBinOp<fminnum>; 124def fmaxnum_oneuse : HasOneUseBinOp<fmaxnum>; 125def and_oneuse : HasOneUseBinOp<and>; 126def or_oneuse : HasOneUseBinOp<or>; 127def xor_oneuse : HasOneUseBinOp<xor>; 128} // Properties = [SDNPCommutative, SDNPAssociative] 129 130def sub_oneuse : HasOneUseBinOp<sub>; 131 132def srl_oneuse : HasOneUseBinOp<srl>; 133def shl_oneuse : HasOneUseBinOp<shl>; 134 135def select_oneuse : HasOneUseTernaryOp<select>; 136 137def srl_16 : PatFrag< 138 (ops node:$src0), (srl_oneuse node:$src0, (i32 16)) 139>; 140 141 142def hi_i16_elt : PatFrag< 143 (ops node:$src0), (i16 (trunc (i32 (srl_16 node:$src0)))) 144>; 145 146 147def hi_f16_elt : PatLeaf< 148 (vt), [{ 149 if (N->getOpcode() != ISD::BITCAST) 150 return false; 151 SDValue Tmp = N->getOperand(0); 152 153 if (Tmp.getOpcode() != ISD::SRL) 154 return false; 155 if (const auto *RHS = dyn_cast<ConstantSDNode>(Tmp.getOperand(1)) 156 return RHS->getZExtValue() == 16; 157 return false; 158}]>; 159 160//===----------------------------------------------------------------------===// 161// PatLeafs for floating-point comparisons 162//===----------------------------------------------------------------------===// 163 164def COND_OEQ : PatLeaf < 165 (cond), 166 [{return N->get() == ISD::SETOEQ || N->get() == ISD::SETEQ;}] 167>; 168 169def COND_ONE : PatLeaf < 170 (cond), 171 [{return N->get() == ISD::SETONE || N->get() == ISD::SETNE;}] 172>; 173 174def COND_OGT : PatLeaf < 175 (cond), 176 [{return N->get() == ISD::SETOGT || N->get() == ISD::SETGT;}] 177>; 178 179def COND_OGE : PatLeaf < 180 (cond), 181 [{return N->get() == ISD::SETOGE || N->get() == ISD::SETGE;}] 182>; 183 184def COND_OLT : PatLeaf < 185 (cond), 186 [{return N->get() == ISD::SETOLT || N->get() == ISD::SETLT;}] 187>; 188 189def COND_OLE : PatLeaf < 190 (cond), 191 [{return N->get() == ISD::SETOLE || N->get() == ISD::SETLE;}] 192>; 193 194def COND_O : PatLeaf <(cond), [{return N->get() == ISD::SETO;}]>; 195def COND_UO : PatLeaf <(cond), [{return N->get() == ISD::SETUO;}]>; 196 197//===----------------------------------------------------------------------===// 198// PatLeafs for unsigned / unordered comparisons 199//===----------------------------------------------------------------------===// 200 201def COND_UEQ : PatLeaf <(cond), [{return N->get() == ISD::SETUEQ;}]>; 202def COND_UNE : PatLeaf <(cond), [{return N->get() == ISD::SETUNE;}]>; 203def COND_UGT : PatLeaf <(cond), [{return N->get() == ISD::SETUGT;}]>; 204def COND_UGE : PatLeaf <(cond), [{return N->get() == ISD::SETUGE;}]>; 205def COND_ULT : PatLeaf <(cond), [{return N->get() == ISD::SETULT;}]>; 206def COND_ULE : PatLeaf <(cond), [{return N->get() == ISD::SETULE;}]>; 207 208// XXX - For some reason R600 version is preferring to use unordered 209// for setne? 210def COND_UNE_NE : PatLeaf < 211 (cond), 212 [{return N->get() == ISD::SETUNE || N->get() == ISD::SETNE;}] 213>; 214 215//===----------------------------------------------------------------------===// 216// PatLeafs for signed comparisons 217//===----------------------------------------------------------------------===// 218 219def COND_SGT : PatLeaf <(cond), [{return N->get() == ISD::SETGT;}]>; 220def COND_SGE : PatLeaf <(cond), [{return N->get() == ISD::SETGE;}]>; 221def COND_SLT : PatLeaf <(cond), [{return N->get() == ISD::SETLT;}]>; 222def COND_SLE : PatLeaf <(cond), [{return N->get() == ISD::SETLE;}]>; 223 224//===----------------------------------------------------------------------===// 225// PatLeafs for integer equality 226//===----------------------------------------------------------------------===// 227 228def COND_EQ : PatLeaf < 229 (cond), 230 [{return N->get() == ISD::SETEQ || N->get() == ISD::SETUEQ;}] 231>; 232 233def COND_NE : PatLeaf < 234 (cond), 235 [{return N->get() == ISD::SETNE || N->get() == ISD::SETUNE;}] 236>; 237 238def COND_NULL : PatLeaf < 239 (cond), 240 [{(void)N; return false;}] 241>; 242 243 244//===----------------------------------------------------------------------===// 245// Load/Store Pattern Fragments 246//===----------------------------------------------------------------------===// 247 248class Aligned8Bytes <dag ops, dag frag> : PatFrag <ops, frag, [{ 249 return cast<MemSDNode>(N)->getAlignment() % 8 == 0; 250}]>; 251 252class LoadFrag <SDPatternOperator op> : PatFrag<(ops node:$ptr), (op node:$ptr)>; 253 254class StoreFrag<SDPatternOperator op> : PatFrag < 255 (ops node:$value, node:$ptr), (op node:$value, node:$ptr) 256>; 257 258class StoreHi16<SDPatternOperator op> : PatFrag < 259 (ops node:$value, node:$ptr), (op (srl node:$value, (i32 16)), node:$ptr) 260>; 261 262class PrivateAddress : CodePatPred<[{ 263 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.PRIVATE_ADDRESS; 264}]>; 265 266class ConstantAddress : CodePatPred<[{ 267 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.CONSTANT_ADDRESS; 268}]>; 269 270class LocalAddress : CodePatPred<[{ 271 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.LOCAL_ADDRESS; 272}]>; 273 274class GlobalAddress : CodePatPred<[{ 275 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS; 276}]>; 277 278class GlobalLoadAddress : CodePatPred<[{ 279 auto AS = cast<MemSDNode>(N)->getAddressSpace(); 280 return AS == AMDGPUASI.GLOBAL_ADDRESS || AS == AMDGPUASI.CONSTANT_ADDRESS; 281}]>; 282 283class FlatLoadAddress : CodePatPred<[{ 284 const auto AS = cast<MemSDNode>(N)->getAddressSpace(); 285 return AS == AMDGPUASI.FLAT_ADDRESS || 286 AS == AMDGPUASI.GLOBAL_ADDRESS || 287 AS == AMDGPUASI.CONSTANT_ADDRESS; 288}]>; 289 290class FlatStoreAddress : CodePatPred<[{ 291 const auto AS = cast<MemSDNode>(N)->getAddressSpace(); 292 return AS == AMDGPUASI.FLAT_ADDRESS || 293 AS == AMDGPUASI.GLOBAL_ADDRESS; 294}]>; 295 296class AZExtLoadBase <SDPatternOperator ld_node>: PatFrag<(ops node:$ptr), 297 (ld_node node:$ptr), [{ 298 LoadSDNode *L = cast<LoadSDNode>(N); 299 return L->getExtensionType() == ISD::ZEXTLOAD || 300 L->getExtensionType() == ISD::EXTLOAD; 301}]>; 302 303def az_extload : AZExtLoadBase <unindexedload>; 304 305def az_extloadi8 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 306 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i8; 307}]>; 308 309def az_extloadi16 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 310 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i16; 311}]>; 312 313def az_extloadi32 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 314 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i32; 315}]>; 316 317class PrivateLoad <SDPatternOperator op> : LoadFrag <op>, PrivateAddress; 318class PrivateStore <SDPatternOperator op> : StoreFrag <op>, PrivateAddress; 319 320class LocalLoad <SDPatternOperator op> : LoadFrag <op>, LocalAddress; 321class LocalStore <SDPatternOperator op> : StoreFrag <op>, LocalAddress; 322 323class GlobalLoad <SDPatternOperator op> : LoadFrag<op>, GlobalLoadAddress; 324class GlobalStore <SDPatternOperator op> : StoreFrag<op>, GlobalAddress; 325 326class FlatLoad <SDPatternOperator op> : LoadFrag <op>, FlatLoadAddress; 327class FlatStore <SDPatternOperator op> : StoreFrag <op>, FlatStoreAddress; 328 329class ConstantLoad <SDPatternOperator op> : LoadFrag <op>, ConstantAddress; 330 331 332def load_private : PrivateLoad <load>; 333def az_extloadi8_private : PrivateLoad <az_extloadi8>; 334def sextloadi8_private : PrivateLoad <sextloadi8>; 335def az_extloadi16_private : PrivateLoad <az_extloadi16>; 336def sextloadi16_private : PrivateLoad <sextloadi16>; 337 338def store_private : PrivateStore <store>; 339def truncstorei8_private : PrivateStore<truncstorei8>; 340def truncstorei16_private : PrivateStore <truncstorei16>; 341def store_hi16_private : StoreHi16 <truncstorei16>, PrivateAddress; 342def truncstorei8_hi16_private : StoreHi16<truncstorei8>, PrivateAddress; 343 344 345def load_global : GlobalLoad <load>; 346def sextloadi8_global : GlobalLoad <sextloadi8>; 347def az_extloadi8_global : GlobalLoad <az_extloadi8>; 348def sextloadi16_global : GlobalLoad <sextloadi16>; 349def az_extloadi16_global : GlobalLoad <az_extloadi16>; 350def atomic_load_global : GlobalLoad<atomic_load>; 351 352def store_global : GlobalStore <store>; 353def truncstorei8_global : GlobalStore <truncstorei8>; 354def truncstorei16_global : GlobalStore <truncstorei16>; 355def store_atomic_global : GlobalStore<atomic_store>; 356def truncstorei8_hi16_global : StoreHi16 <truncstorei8>, GlobalAddress; 357def truncstorei16_hi16_global : StoreHi16 <truncstorei16>, GlobalAddress; 358 359def load_local : LocalLoad <load>; 360def az_extloadi8_local : LocalLoad <az_extloadi8>; 361def sextloadi8_local : LocalLoad <sextloadi8>; 362def az_extloadi16_local : LocalLoad <az_extloadi16>; 363def sextloadi16_local : LocalLoad <sextloadi16>; 364 365def store_local : LocalStore <store>; 366def truncstorei8_local : LocalStore <truncstorei8>; 367def truncstorei16_local : LocalStore <truncstorei16>; 368def store_local_hi16 : StoreHi16 <truncstorei16>, LocalAddress; 369def truncstorei8_local_hi16 : StoreHi16<truncstorei8>, LocalAddress; 370 371def load_align8_local : Aligned8Bytes < 372 (ops node:$ptr), (load_local node:$ptr) 373>; 374 375def store_align8_local : Aligned8Bytes < 376 (ops node:$val, node:$ptr), (store_local node:$val, node:$ptr) 377>; 378 379 380def load_flat : FlatLoad <load>; 381def az_extloadi8_flat : FlatLoad <az_extloadi8>; 382def sextloadi8_flat : FlatLoad <sextloadi8>; 383def az_extloadi16_flat : FlatLoad <az_extloadi16>; 384def sextloadi16_flat : FlatLoad <sextloadi16>; 385def atomic_load_flat : FlatLoad<atomic_load>; 386 387def store_flat : FlatStore <store>; 388def truncstorei8_flat : FlatStore <truncstorei8>; 389def truncstorei16_flat : FlatStore <truncstorei16>; 390def atomic_store_flat : FlatStore <atomic_store>; 391def truncstorei8_hi16_flat : StoreHi16<truncstorei8>, FlatStoreAddress; 392def truncstorei16_hi16_flat : StoreHi16<truncstorei16>, FlatStoreAddress; 393 394 395def constant_load : ConstantLoad<load>; 396def sextloadi8_constant : ConstantLoad <sextloadi8>; 397def az_extloadi8_constant : ConstantLoad <az_extloadi8>; 398def sextloadi16_constant : ConstantLoad <sextloadi16>; 399def az_extloadi16_constant : ConstantLoad <az_extloadi16>; 400 401 402class local_binary_atomic_op<SDNode atomic_op> : 403 PatFrag<(ops node:$ptr, node:$value), 404 (atomic_op node:$ptr, node:$value), [{ 405 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.LOCAL_ADDRESS; 406}]>; 407 408def atomic_swap_local : local_binary_atomic_op<atomic_swap>; 409def atomic_load_add_local : local_binary_atomic_op<atomic_load_add>; 410def atomic_load_sub_local : local_binary_atomic_op<atomic_load_sub>; 411def atomic_load_and_local : local_binary_atomic_op<atomic_load_and>; 412def atomic_load_or_local : local_binary_atomic_op<atomic_load_or>; 413def atomic_load_xor_local : local_binary_atomic_op<atomic_load_xor>; 414def atomic_load_nand_local : local_binary_atomic_op<atomic_load_nand>; 415def atomic_load_min_local : local_binary_atomic_op<atomic_load_min>; 416def atomic_load_max_local : local_binary_atomic_op<atomic_load_max>; 417def atomic_load_umin_local : local_binary_atomic_op<atomic_load_umin>; 418def atomic_load_umax_local : local_binary_atomic_op<atomic_load_umax>; 419 420def mskor_global : PatFrag<(ops node:$val, node:$ptr), 421 (AMDGPUstore_mskor node:$val, node:$ptr), [{ 422 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS; 423}]>; 424 425class AtomicCmpSwapLocal <SDNode cmp_swap_node> : PatFrag< 426 (ops node:$ptr, node:$cmp, node:$swap), 427 (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{ 428 AtomicSDNode *AN = cast<AtomicSDNode>(N); 429 return AN->getAddressSpace() == AMDGPUASI.LOCAL_ADDRESS; 430}]>; 431 432def atomic_cmp_swap_local : AtomicCmpSwapLocal <atomic_cmp_swap>; 433 434multiclass global_binary_atomic_op<SDNode atomic_op> { 435 def "" : PatFrag< 436 (ops node:$ptr, node:$value), 437 (atomic_op node:$ptr, node:$value), 438 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS;}]>; 439 440 def _noret : PatFrag< 441 (ops node:$ptr, node:$value), 442 (atomic_op node:$ptr, node:$value), 443 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>; 444 445 def _ret : PatFrag< 446 (ops node:$ptr, node:$value), 447 (atomic_op node:$ptr, node:$value), 448 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>; 449} 450 451defm atomic_swap_global : global_binary_atomic_op<atomic_swap>; 452defm atomic_add_global : global_binary_atomic_op<atomic_load_add>; 453defm atomic_and_global : global_binary_atomic_op<atomic_load_and>; 454defm atomic_max_global : global_binary_atomic_op<atomic_load_max>; 455defm atomic_min_global : global_binary_atomic_op<atomic_load_min>; 456defm atomic_or_global : global_binary_atomic_op<atomic_load_or>; 457defm atomic_sub_global : global_binary_atomic_op<atomic_load_sub>; 458defm atomic_umax_global : global_binary_atomic_op<atomic_load_umax>; 459defm atomic_umin_global : global_binary_atomic_op<atomic_load_umin>; 460defm atomic_xor_global : global_binary_atomic_op<atomic_load_xor>; 461 462// Legacy. 463def AMDGPUatomic_cmp_swap_global : PatFrag< 464 (ops node:$ptr, node:$value), 465 (AMDGPUatomic_cmp_swap node:$ptr, node:$value)>, GlobalAddress; 466 467def atomic_cmp_swap_global : PatFrag< 468 (ops node:$ptr, node:$cmp, node:$value), 469 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value)>, GlobalAddress; 470 471 472def atomic_cmp_swap_global_noret : PatFrag< 473 (ops node:$ptr, node:$cmp, node:$value), 474 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value), 475 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>; 476 477def atomic_cmp_swap_global_ret : PatFrag< 478 (ops node:$ptr, node:$cmp, node:$value), 479 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value), 480 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>; 481 482//===----------------------------------------------------------------------===// 483// Misc Pattern Fragments 484//===----------------------------------------------------------------------===// 485 486class Constants { 487int TWO_PI = 0x40c90fdb; 488int PI = 0x40490fdb; 489int TWO_PI_INV = 0x3e22f983; 490int FP_UINT_MAX_PLUS_1 = 0x4f800000; // 1 << 32 in floating point encoding 491int FP16_ONE = 0x3C00; 492int V2FP16_ONE = 0x3C003C00; 493int FP32_ONE = 0x3f800000; 494int FP32_NEG_ONE = 0xbf800000; 495int FP64_ONE = 0x3ff0000000000000; 496int FP64_NEG_ONE = 0xbff0000000000000; 497} 498def CONST : Constants; 499 500def FP_ZERO : PatLeaf < 501 (fpimm), 502 [{return N->getValueAPF().isZero();}] 503>; 504 505def FP_ONE : PatLeaf < 506 (fpimm), 507 [{return N->isExactlyValue(1.0);}] 508>; 509 510def FP_HALF : PatLeaf < 511 (fpimm), 512 [{return N->isExactlyValue(0.5);}] 513>; 514 515/* Generic helper patterns for intrinsics */ 516/* -------------------------------------- */ 517 518class POW_Common <AMDGPUInst log_ieee, AMDGPUInst exp_ieee, AMDGPUInst mul> 519 : AMDGPUPat < 520 (fpow f32:$src0, f32:$src1), 521 (exp_ieee (mul f32:$src1, (log_ieee f32:$src0))) 522>; 523 524/* Other helper patterns */ 525/* --------------------- */ 526 527/* Extract element pattern */ 528class Extract_Element <ValueType sub_type, ValueType vec_type, int sub_idx, 529 SubRegIndex sub_reg> 530 : AMDGPUPat< 531 (sub_type (extractelt vec_type:$src, sub_idx)), 532 (EXTRACT_SUBREG $src, sub_reg) 533> { 534 let SubtargetPredicate = TruePredicate; 535} 536 537/* Insert element pattern */ 538class Insert_Element <ValueType elem_type, ValueType vec_type, 539 int sub_idx, SubRegIndex sub_reg> 540 : AMDGPUPat < 541 (insertelt vec_type:$vec, elem_type:$elem, sub_idx), 542 (INSERT_SUBREG $vec, $elem, sub_reg) 543> { 544 let SubtargetPredicate = TruePredicate; 545} 546 547// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 548// can handle COPY instructions. 549// bitconvert pattern 550class BitConvert <ValueType dt, ValueType st, RegisterClass rc> : AMDGPUPat < 551 (dt (bitconvert (st rc:$src0))), 552 (dt rc:$src0) 553>; 554 555// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 556// can handle COPY instructions. 557class DwordAddrPat<ValueType vt, RegisterClass rc> : AMDGPUPat < 558 (vt (AMDGPUdwordaddr (vt rc:$addr))), 559 (vt rc:$addr) 560>; 561 562// BFI_INT patterns 563 564multiclass BFIPatterns <Instruction BFI_INT, 565 Instruction LoadImm32, 566 RegisterClass RC64> { 567 // Definition from ISA doc: 568 // (y & x) | (z & ~x) 569 def : AMDGPUPat < 570 (or (and i32:$y, i32:$x), (and i32:$z, (not i32:$x))), 571 (BFI_INT $x, $y, $z) 572 >; 573 574 // SHA-256 Ch function 575 // z ^ (x & (y ^ z)) 576 def : AMDGPUPat < 577 (xor i32:$z, (and i32:$x, (xor i32:$y, i32:$z))), 578 (BFI_INT $x, $y, $z) 579 >; 580 581 def : AMDGPUPat < 582 (fcopysign f32:$src0, f32:$src1), 583 (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0, $src1) 584 >; 585 586 def : AMDGPUPat < 587 (f32 (fcopysign f32:$src0, f64:$src1)), 588 (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0, 589 (i32 (EXTRACT_SUBREG $src1, sub1))) 590 >; 591 592 def : AMDGPUPat < 593 (f64 (fcopysign f64:$src0, f64:$src1)), 594 (REG_SEQUENCE RC64, 595 (i32 (EXTRACT_SUBREG $src0, sub0)), sub0, 596 (BFI_INT (LoadImm32 (i32 0x7fffffff)), 597 (i32 (EXTRACT_SUBREG $src0, sub1)), 598 (i32 (EXTRACT_SUBREG $src1, sub1))), sub1) 599 >; 600 601 def : AMDGPUPat < 602 (f64 (fcopysign f64:$src0, f32:$src1)), 603 (REG_SEQUENCE RC64, 604 (i32 (EXTRACT_SUBREG $src0, sub0)), sub0, 605 (BFI_INT (LoadImm32 (i32 0x7fffffff)), 606 (i32 (EXTRACT_SUBREG $src0, sub1)), 607 $src1), sub1) 608 >; 609} 610 611// SHA-256 Ma patterns 612 613// ((x & z) | (y & (x | z))) -> BFI_INT (XOR x, y), z, y 614class SHA256MaPattern <Instruction BFI_INT, Instruction XOR> : AMDGPUPat < 615 (or (and i32:$x, i32:$z), (and i32:$y, (or i32:$x, i32:$z))), 616 (BFI_INT (XOR i32:$x, i32:$y), i32:$z, i32:$y) 617>; 618 619// Bitfield extract patterns 620 621def IMMZeroBasedBitfieldMask : PatLeaf <(imm), [{ 622 return isMask_32(N->getZExtValue()); 623}]>; 624 625def IMMPopCount : SDNodeXForm<imm, [{ 626 return CurDAG->getTargetConstant(countPopulation(N->getZExtValue()), SDLoc(N), 627 MVT::i32); 628}]>; 629 630multiclass BFEPattern <Instruction UBFE, Instruction SBFE, Instruction MOV> { 631 def : AMDGPUPat < 632 (i32 (and (i32 (srl i32:$src, i32:$rshift)), IMMZeroBasedBitfieldMask:$mask)), 633 (UBFE $src, $rshift, (MOV (i32 (IMMPopCount $mask)))) 634 >; 635 636 def : AMDGPUPat < 637 (srl (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)), 638 (UBFE $src, (i32 0), $width) 639 >; 640 641 def : AMDGPUPat < 642 (sra (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)), 643 (SBFE $src, (i32 0), $width) 644 >; 645} 646 647// rotr pattern 648class ROTRPattern <Instruction BIT_ALIGN> : AMDGPUPat < 649 (rotr i32:$src0, i32:$src1), 650 (BIT_ALIGN $src0, $src0, $src1) 651>; 652 653// This matches 16 permutations of 654// max(min(x, y), min(max(x, y), z)) 655class IntMed3Pat<Instruction med3Inst, 656 SDPatternOperator max, 657 SDPatternOperator max_oneuse, 658 SDPatternOperator min_oneuse, 659 ValueType vt = i32> : AMDGPUPat< 660 (max (min_oneuse vt:$src0, vt:$src1), 661 (min_oneuse (max_oneuse vt:$src0, vt:$src1), vt:$src2)), 662 (med3Inst $src0, $src1, $src2) 663>; 664 665// Special conversion patterns 666 667def cvt_rpi_i32_f32 : PatFrag < 668 (ops node:$src), 669 (fp_to_sint (ffloor (fadd $src, FP_HALF))), 670 [{ (void) N; return TM.Options.NoNaNsFPMath; }] 671>; 672 673def cvt_flr_i32_f32 : PatFrag < 674 (ops node:$src), 675 (fp_to_sint (ffloor $src)), 676 [{ (void)N; return TM.Options.NoNaNsFPMath; }] 677>; 678 679class IMad24Pat<Instruction Inst, bit HasClamp = 0> : AMDGPUPat < 680 (add (AMDGPUmul_i24 i32:$src0, i32:$src1), i32:$src2), 681 !if(HasClamp, (Inst $src0, $src1, $src2, (i1 0)), 682 (Inst $src0, $src1, $src2)) 683>; 684 685class UMad24Pat<Instruction Inst, bit HasClamp = 0> : AMDGPUPat < 686 (add (AMDGPUmul_u24 i32:$src0, i32:$src1), i32:$src2), 687 !if(HasClamp, (Inst $src0, $src1, $src2, (i1 0)), 688 (Inst $src0, $src1, $src2)) 689>; 690 691class RcpPat<Instruction RcpInst, ValueType vt> : AMDGPUPat < 692 (fdiv FP_ONE, vt:$src), 693 (RcpInst $src) 694>; 695 696class RsqPat<Instruction RsqInst, ValueType vt> : AMDGPUPat < 697 (AMDGPUrcp (fsqrt vt:$src)), 698 (RsqInst $src) 699>; 700 701include "R600Instructions.td" 702include "R700Instructions.td" 703include "EvergreenInstructions.td" 704include "CaymanInstructions.td" 705 706include "SIInstrInfo.td" 707 708