1//===-- AMDGPUInstructions.td - Common instruction defs ---*- tablegen -*-===// 2// 3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4// See https://llvm.org/LICENSE.txt for license information. 5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6// 7//===----------------------------------------------------------------------===// 8// 9// This file contains instruction defs that are common to all hw codegen 10// targets. 11// 12//===----------------------------------------------------------------------===// 13 14class AMDGPUInst <dag outs, dag ins, string asm = "", 15 list<dag> pattern = []> : Instruction { 16 field bit isRegisterLoad = 0; 17 field bit isRegisterStore = 0; 18 19 let Namespace = "AMDGPU"; 20 let OutOperandList = outs; 21 let InOperandList = ins; 22 let AsmString = asm; 23 let Pattern = pattern; 24 let Itinerary = NullALU; 25 26 // SoftFail is a field the disassembler can use to provide a way for 27 // instructions to not match without killing the whole decode process. It is 28 // mainly used for ARM, but Tablegen expects this field to exist or it fails 29 // to build the decode table. 30 field bits<64> SoftFail = 0; 31 32 let DecoderNamespace = Namespace; 33 34 let TSFlags{63} = isRegisterLoad; 35 let TSFlags{62} = isRegisterStore; 36} 37 38class AMDGPUShaderInst <dag outs, dag ins, string asm = "", 39 list<dag> pattern = []> : AMDGPUInst<outs, ins, asm, pattern> { 40 41 field bits<32> Inst = 0xffffffff; 42} 43 44//===---------------------------------------------------------------------===// 45// Return instruction 46//===---------------------------------------------------------------------===// 47 48class ILFormat<dag outs, dag ins, string asmstr, list<dag> pattern> 49: Instruction { 50 51 let Namespace = "AMDGPU"; 52 dag OutOperandList = outs; 53 dag InOperandList = ins; 54 let Pattern = pattern; 55 let AsmString = !strconcat(asmstr, "\n"); 56 let isPseudo = 1; 57 let Itinerary = NullALU; 58 bit hasIEEEFlag = 0; 59 bit hasZeroOpFlag = 0; 60 let mayLoad = 0; 61 let mayStore = 0; 62 let hasSideEffects = 0; 63 let isCodeGenOnly = 1; 64} 65 66def TruePredicate : Predicate<"true">; 67 68class PredicateControl { 69 Predicate SubtargetPredicate = TruePredicate; 70 list<Predicate> AssemblerPredicates = []; 71 Predicate AssemblerPredicate = TruePredicate; 72 Predicate WaveSizePredicate = TruePredicate; 73 list<Predicate> OtherPredicates = []; 74 list<Predicate> Predicates = !listconcat([SubtargetPredicate, 75 AssemblerPredicate, 76 WaveSizePredicate], 77 AssemblerPredicates, 78 OtherPredicates); 79} 80class AMDGPUPat<dag pattern, dag result> : Pat<pattern, result>, 81 PredicateControl; 82 83def FP16Denormals : Predicate<"Subtarget->hasFP16Denormals()">; 84def FP32Denormals : Predicate<"Subtarget->hasFP32Denormals()">; 85def FP64Denormals : Predicate<"Subtarget->hasFP64Denormals()">; 86def NoFP16Denormals : Predicate<"!Subtarget->hasFP16Denormals()">; 87def NoFP32Denormals : Predicate<"!Subtarget->hasFP32Denormals()">; 88def NoFP64Denormals : Predicate<"!Subtarget->hasFP64Denormals()">; 89def UnsafeFPMath : Predicate<"TM.Options.UnsafeFPMath">; 90def FMA : Predicate<"Subtarget->hasFMA()">; 91 92def InstFlag : OperandWithDefaultOps <i32, (ops (i32 0))>; 93 94def u16ImmTarget : AsmOperandClass { 95 let Name = "U16Imm"; 96 let RenderMethod = "addImmOperands"; 97} 98 99def s16ImmTarget : AsmOperandClass { 100 let Name = "S16Imm"; 101 let RenderMethod = "addImmOperands"; 102} 103 104let OperandType = "OPERAND_IMMEDIATE" in { 105 106def u32imm : Operand<i32> { 107 let PrintMethod = "printU32ImmOperand"; 108} 109 110def u16imm : Operand<i16> { 111 let PrintMethod = "printU16ImmOperand"; 112 let ParserMatchClass = u16ImmTarget; 113} 114 115def s16imm : Operand<i16> { 116 let PrintMethod = "printU16ImmOperand"; 117 let ParserMatchClass = s16ImmTarget; 118} 119 120def u8imm : Operand<i8> { 121 let PrintMethod = "printU8ImmOperand"; 122} 123 124} // End OperandType = "OPERAND_IMMEDIATE" 125 126//===--------------------------------------------------------------------===// 127// Custom Operands 128//===--------------------------------------------------------------------===// 129def brtarget : Operand<OtherVT>; 130 131//===----------------------------------------------------------------------===// 132// Misc. PatFrags 133//===----------------------------------------------------------------------===// 134 135class HasOneUseUnaryOp<SDPatternOperator op> : PatFrag< 136 (ops node:$src0), 137 (op $src0), 138 [{ return N->hasOneUse(); }] 139>; 140 141class HasOneUseBinOp<SDPatternOperator op> : PatFrag< 142 (ops node:$src0, node:$src1), 143 (op $src0, $src1), 144 [{ return N->hasOneUse(); }] 145>; 146 147class HasOneUseTernaryOp<SDPatternOperator op> : PatFrag< 148 (ops node:$src0, node:$src1, node:$src2), 149 (op $src0, $src1, $src2), 150 [{ return N->hasOneUse(); }] 151>; 152 153let Properties = [SDNPCommutative, SDNPAssociative] in { 154def smax_oneuse : HasOneUseBinOp<smax>; 155def smin_oneuse : HasOneUseBinOp<smin>; 156def umax_oneuse : HasOneUseBinOp<umax>; 157def umin_oneuse : HasOneUseBinOp<umin>; 158 159def fminnum_oneuse : HasOneUseBinOp<fminnum>; 160def fmaxnum_oneuse : HasOneUseBinOp<fmaxnum>; 161 162def fminnum_ieee_oneuse : HasOneUseBinOp<fminnum_ieee>; 163def fmaxnum_ieee_oneuse : HasOneUseBinOp<fmaxnum_ieee>; 164 165 166def and_oneuse : HasOneUseBinOp<and>; 167def or_oneuse : HasOneUseBinOp<or>; 168def xor_oneuse : HasOneUseBinOp<xor>; 169} // Properties = [SDNPCommutative, SDNPAssociative] 170 171def not_oneuse : HasOneUseUnaryOp<not>; 172 173def add_oneuse : HasOneUseBinOp<add>; 174def sub_oneuse : HasOneUseBinOp<sub>; 175 176def srl_oneuse : HasOneUseBinOp<srl>; 177def shl_oneuse : HasOneUseBinOp<shl>; 178 179def select_oneuse : HasOneUseTernaryOp<select>; 180 181def AMDGPUmul_u24_oneuse : HasOneUseBinOp<AMDGPUmul_u24>; 182def AMDGPUmul_i24_oneuse : HasOneUseBinOp<AMDGPUmul_i24>; 183 184def srl_16 : PatFrag< 185 (ops node:$src0), (srl_oneuse node:$src0, (i32 16)) 186>; 187 188 189def hi_i16_elt : PatFrag< 190 (ops node:$src0), (i16 (trunc (i32 (srl_16 node:$src0)))) 191>; 192 193 194def hi_f16_elt : PatLeaf< 195 (vt), [{ 196 if (N->getOpcode() != ISD::BITCAST) 197 return false; 198 SDValue Tmp = N->getOperand(0); 199 200 if (Tmp.getOpcode() != ISD::SRL) 201 return false; 202 if (const auto *RHS = dyn_cast<ConstantSDNode>(Tmp.getOperand(1)) 203 return RHS->getZExtValue() == 16; 204 return false; 205}]>; 206 207//===----------------------------------------------------------------------===// 208// PatLeafs for floating-point comparisons 209//===----------------------------------------------------------------------===// 210 211def COND_OEQ : PatLeaf < 212 (cond), 213 [{return N->get() == ISD::SETOEQ || N->get() == ISD::SETEQ;}] 214>; 215 216def COND_ONE : PatLeaf < 217 (cond), 218 [{return N->get() == ISD::SETONE || N->get() == ISD::SETNE;}] 219>; 220 221def COND_OGT : PatLeaf < 222 (cond), 223 [{return N->get() == ISD::SETOGT || N->get() == ISD::SETGT;}] 224>; 225 226def COND_OGE : PatLeaf < 227 (cond), 228 [{return N->get() == ISD::SETOGE || N->get() == ISD::SETGE;}] 229>; 230 231def COND_OLT : PatLeaf < 232 (cond), 233 [{return N->get() == ISD::SETOLT || N->get() == ISD::SETLT;}] 234>; 235 236def COND_OLE : PatLeaf < 237 (cond), 238 [{return N->get() == ISD::SETOLE || N->get() == ISD::SETLE;}] 239>; 240 241def COND_O : PatLeaf <(cond), [{return N->get() == ISD::SETO;}]>; 242def COND_UO : PatLeaf <(cond), [{return N->get() == ISD::SETUO;}]>; 243 244//===----------------------------------------------------------------------===// 245// PatLeafs for unsigned / unordered comparisons 246//===----------------------------------------------------------------------===// 247 248def COND_UEQ : PatLeaf <(cond), [{return N->get() == ISD::SETUEQ;}]>; 249def COND_UNE : PatLeaf <(cond), [{return N->get() == ISD::SETUNE;}]>; 250def COND_UGT : PatLeaf <(cond), [{return N->get() == ISD::SETUGT;}]>; 251def COND_UGE : PatLeaf <(cond), [{return N->get() == ISD::SETUGE;}]>; 252def COND_ULT : PatLeaf <(cond), [{return N->get() == ISD::SETULT;}]>; 253def COND_ULE : PatLeaf <(cond), [{return N->get() == ISD::SETULE;}]>; 254 255// XXX - For some reason R600 version is preferring to use unordered 256// for setne? 257def COND_UNE_NE : PatLeaf < 258 (cond), 259 [{return N->get() == ISD::SETUNE || N->get() == ISD::SETNE;}] 260>; 261 262//===----------------------------------------------------------------------===// 263// PatLeafs for signed comparisons 264//===----------------------------------------------------------------------===// 265 266def COND_SGT : PatLeaf <(cond), [{return N->get() == ISD::SETGT;}]>; 267def COND_SGE : PatLeaf <(cond), [{return N->get() == ISD::SETGE;}]>; 268def COND_SLT : PatLeaf <(cond), [{return N->get() == ISD::SETLT;}]>; 269def COND_SLE : PatLeaf <(cond), [{return N->get() == ISD::SETLE;}]>; 270 271//===----------------------------------------------------------------------===// 272// PatLeafs for integer equality 273//===----------------------------------------------------------------------===// 274 275def COND_EQ : PatLeaf < 276 (cond), 277 [{return N->get() == ISD::SETEQ || N->get() == ISD::SETUEQ;}] 278>; 279 280def COND_NE : PatLeaf < 281 (cond), 282 [{return N->get() == ISD::SETNE || N->get() == ISD::SETUNE;}] 283>; 284 285def COND_NULL : PatLeaf < 286 (cond), 287 [{(void)N; return false;}] 288>; 289 290//===----------------------------------------------------------------------===// 291// PatLeafs for Texture Constants 292//===----------------------------------------------------------------------===// 293 294def TEX_ARRAY : PatLeaf< 295 (imm), 296 [{uint32_t TType = (uint32_t)N->getZExtValue(); 297 return TType == 9 || TType == 10 || TType == 16; 298 }] 299>; 300 301def TEX_RECT : PatLeaf< 302 (imm), 303 [{uint32_t TType = (uint32_t)N->getZExtValue(); 304 return TType == 5; 305 }] 306>; 307 308def TEX_SHADOW : PatLeaf< 309 (imm), 310 [{uint32_t TType = (uint32_t)N->getZExtValue(); 311 return (TType >= 6 && TType <= 8) || TType == 13; 312 }] 313>; 314 315def TEX_SHADOW_ARRAY : PatLeaf< 316 (imm), 317 [{uint32_t TType = (uint32_t)N->getZExtValue(); 318 return TType == 11 || TType == 12 || TType == 17; 319 }] 320>; 321 322//===----------------------------------------------------------------------===// 323// Load/Store Pattern Fragments 324//===----------------------------------------------------------------------===// 325 326class Aligned8Bytes <dag ops, dag frag> : PatFrag <ops, frag, [{ 327 return cast<MemSDNode>(N)->getAlignment() % 8 == 0; 328}]>; 329 330class Aligned16Bytes <dag ops, dag frag> : PatFrag <ops, frag, [{ 331 return cast<MemSDNode>(N)->getAlignment() >= 16; 332}]>; 333 334class LoadFrag <SDPatternOperator op> : PatFrag<(ops node:$ptr), (op node:$ptr)>; 335 336class StoreFrag<SDPatternOperator op> : PatFrag < 337 (ops node:$value, node:$ptr), (op node:$value, node:$ptr) 338>; 339 340class StoreHi16<SDPatternOperator op> : PatFrag < 341 (ops node:$value, node:$ptr), (op (srl node:$value, (i32 16)), node:$ptr) 342>; 343 344class PrivateAddress : CodePatPred<[{ 345 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS; 346}]>; 347 348class ConstantAddress : CodePatPred<[{ 349 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::CONSTANT_ADDRESS; 350}]>; 351 352class LocalAddress : CodePatPred<[{ 353 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 354}]>; 355 356class GlobalAddress : CodePatPred<[{ 357 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS; 358}]>; 359 360class GlobalLoadAddress : CodePatPred<[{ 361 auto AS = cast<MemSDNode>(N)->getAddressSpace(); 362 return AS == AMDGPUAS::GLOBAL_ADDRESS || AS == AMDGPUAS::CONSTANT_ADDRESS; 363}]>; 364 365class FlatLoadAddress : CodePatPred<[{ 366 const auto AS = cast<MemSDNode>(N)->getAddressSpace(); 367 return AS == AMDGPUAS::FLAT_ADDRESS || 368 AS == AMDGPUAS::GLOBAL_ADDRESS || 369 AS == AMDGPUAS::CONSTANT_ADDRESS; 370}]>; 371 372class FlatStoreAddress : CodePatPred<[{ 373 const auto AS = cast<MemSDNode>(N)->getAddressSpace(); 374 return AS == AMDGPUAS::FLAT_ADDRESS || 375 AS == AMDGPUAS::GLOBAL_ADDRESS; 376}]>; 377 378class AZExtLoadBase <SDPatternOperator ld_node>: PatFrag<(ops node:$ptr), 379 (ld_node node:$ptr), [{ 380 LoadSDNode *L = cast<LoadSDNode>(N); 381 return L->getExtensionType() == ISD::ZEXTLOAD || 382 L->getExtensionType() == ISD::EXTLOAD; 383}]>; 384 385def az_extload : AZExtLoadBase <unindexedload>; 386 387def az_extloadi8 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 388 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i8; 389}]>; 390 391def az_extloadi16 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 392 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i16; 393}]>; 394 395def az_extloadi32 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 396 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i32; 397}]>; 398 399class PrivateLoad <SDPatternOperator op> : LoadFrag <op>, PrivateAddress; 400class PrivateStore <SDPatternOperator op> : StoreFrag <op>, PrivateAddress; 401 402class LocalLoad <SDPatternOperator op> : LoadFrag <op>, LocalAddress; 403class LocalStore <SDPatternOperator op> : StoreFrag <op>, LocalAddress; 404 405class GlobalLoad <SDPatternOperator op> : LoadFrag<op>, GlobalLoadAddress; 406class GlobalStore <SDPatternOperator op> : StoreFrag<op>, GlobalAddress; 407 408class FlatLoad <SDPatternOperator op> : LoadFrag <op>, FlatLoadAddress; 409class FlatStore <SDPatternOperator op> : StoreFrag <op>, FlatStoreAddress; 410 411class ConstantLoad <SDPatternOperator op> : LoadFrag <op>, ConstantAddress; 412 413 414def load_private : PrivateLoad <load>; 415def az_extloadi8_private : PrivateLoad <az_extloadi8>; 416def sextloadi8_private : PrivateLoad <sextloadi8>; 417def az_extloadi16_private : PrivateLoad <az_extloadi16>; 418def sextloadi16_private : PrivateLoad <sextloadi16>; 419 420def store_private : PrivateStore <store>; 421def truncstorei8_private : PrivateStore<truncstorei8>; 422def truncstorei16_private : PrivateStore <truncstorei16>; 423def store_hi16_private : StoreHi16 <truncstorei16>, PrivateAddress; 424def truncstorei8_hi16_private : StoreHi16<truncstorei8>, PrivateAddress; 425 426 427def load_global : GlobalLoad <load>; 428def sextloadi8_global : GlobalLoad <sextloadi8>; 429def az_extloadi8_global : GlobalLoad <az_extloadi8>; 430def sextloadi16_global : GlobalLoad <sextloadi16>; 431def az_extloadi16_global : GlobalLoad <az_extloadi16>; 432def atomic_load_global : GlobalLoad<atomic_load>; 433 434def store_global : GlobalStore <store>; 435def truncstorei8_global : GlobalStore <truncstorei8>; 436def truncstorei16_global : GlobalStore <truncstorei16>; 437def store_atomic_global : GlobalStore<atomic_store>; 438def truncstorei8_hi16_global : StoreHi16 <truncstorei8>, GlobalAddress; 439def truncstorei16_hi16_global : StoreHi16 <truncstorei16>, GlobalAddress; 440 441def load_local : LocalLoad <load>; 442def az_extloadi8_local : LocalLoad <az_extloadi8>; 443def sextloadi8_local : LocalLoad <sextloadi8>; 444def az_extloadi16_local : LocalLoad <az_extloadi16>; 445def sextloadi16_local : LocalLoad <sextloadi16>; 446def atomic_load_32_local : LocalLoad<atomic_load_32>; 447def atomic_load_64_local : LocalLoad<atomic_load_64>; 448 449def store_local : LocalStore <store>; 450def truncstorei8_local : LocalStore <truncstorei8>; 451def truncstorei16_local : LocalStore <truncstorei16>; 452def store_local_hi16 : StoreHi16 <truncstorei16>, LocalAddress; 453def truncstorei8_local_hi16 : StoreHi16<truncstorei8>, LocalAddress; 454def atomic_store_local : LocalStore <atomic_store>; 455 456def load_align8_local : Aligned8Bytes < 457 (ops node:$ptr), (load_local node:$ptr) 458>; 459 460def load_align16_local : Aligned16Bytes < 461 (ops node:$ptr), (load_local node:$ptr) 462>; 463 464def store_align8_local : Aligned8Bytes < 465 (ops node:$val, node:$ptr), (store_local node:$val, node:$ptr) 466>; 467 468def store_align16_local : Aligned16Bytes < 469 (ops node:$val, node:$ptr), (store_local node:$val, node:$ptr) 470>; 471 472def load_flat : FlatLoad <load>; 473def az_extloadi8_flat : FlatLoad <az_extloadi8>; 474def sextloadi8_flat : FlatLoad <sextloadi8>; 475def az_extloadi16_flat : FlatLoad <az_extloadi16>; 476def sextloadi16_flat : FlatLoad <sextloadi16>; 477def atomic_load_flat : FlatLoad<atomic_load>; 478 479def store_flat : FlatStore <store>; 480def truncstorei8_flat : FlatStore <truncstorei8>; 481def truncstorei16_flat : FlatStore <truncstorei16>; 482def atomic_store_flat : FlatStore <atomic_store>; 483def truncstorei8_hi16_flat : StoreHi16<truncstorei8>, FlatStoreAddress; 484def truncstorei16_hi16_flat : StoreHi16<truncstorei16>, FlatStoreAddress; 485 486 487def constant_load : ConstantLoad<load>; 488def sextloadi8_constant : ConstantLoad <sextloadi8>; 489def az_extloadi8_constant : ConstantLoad <az_extloadi8>; 490def sextloadi16_constant : ConstantLoad <sextloadi16>; 491def az_extloadi16_constant : ConstantLoad <az_extloadi16>; 492 493 494class local_binary_atomic_op<SDNode atomic_op> : 495 PatFrag<(ops node:$ptr, node:$value), 496 (atomic_op node:$ptr, node:$value), [{ 497 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 498}]>; 499 500def atomic_swap_local : local_binary_atomic_op<atomic_swap>; 501def atomic_load_add_local : local_binary_atomic_op<atomic_load_add>; 502def atomic_load_sub_local : local_binary_atomic_op<atomic_load_sub>; 503def atomic_load_and_local : local_binary_atomic_op<atomic_load_and>; 504def atomic_load_or_local : local_binary_atomic_op<atomic_load_or>; 505def atomic_load_xor_local : local_binary_atomic_op<atomic_load_xor>; 506def atomic_load_nand_local : local_binary_atomic_op<atomic_load_nand>; 507def atomic_load_min_local : local_binary_atomic_op<atomic_load_min>; 508def atomic_load_max_local : local_binary_atomic_op<atomic_load_max>; 509def atomic_load_umin_local : local_binary_atomic_op<atomic_load_umin>; 510def atomic_load_umax_local : local_binary_atomic_op<atomic_load_umax>; 511 512def mskor_global : PatFrag<(ops node:$val, node:$ptr), 513 (AMDGPUstore_mskor node:$val, node:$ptr), [{ 514 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS; 515}]>; 516 517class AtomicCmpSwapLocal <SDNode cmp_swap_node> : PatFrag< 518 (ops node:$ptr, node:$cmp, node:$swap), 519 (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{ 520 AtomicSDNode *AN = cast<AtomicSDNode>(N); 521 return AN->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 522}]>; 523 524def atomic_cmp_swap_local : AtomicCmpSwapLocal <atomic_cmp_swap>; 525 526multiclass global_binary_atomic_op<SDNode atomic_op> { 527 def "" : PatFrag< 528 (ops node:$ptr, node:$value), 529 (atomic_op node:$ptr, node:$value), 530 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}]>; 531 532 def _noret : PatFrag< 533 (ops node:$ptr, node:$value), 534 (atomic_op node:$ptr, node:$value), 535 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>; 536 537 def _ret : PatFrag< 538 (ops node:$ptr, node:$value), 539 (atomic_op node:$ptr, node:$value), 540 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>; 541} 542 543defm atomic_swap_global : global_binary_atomic_op<atomic_swap>; 544defm atomic_add_global : global_binary_atomic_op<atomic_load_add>; 545defm atomic_and_global : global_binary_atomic_op<atomic_load_and>; 546defm atomic_max_global : global_binary_atomic_op<atomic_load_max>; 547defm atomic_min_global : global_binary_atomic_op<atomic_load_min>; 548defm atomic_or_global : global_binary_atomic_op<atomic_load_or>; 549defm atomic_sub_global : global_binary_atomic_op<atomic_load_sub>; 550defm atomic_umax_global : global_binary_atomic_op<atomic_load_umax>; 551defm atomic_umin_global : global_binary_atomic_op<atomic_load_umin>; 552defm atomic_xor_global : global_binary_atomic_op<atomic_load_xor>; 553 554// Legacy. 555def AMDGPUatomic_cmp_swap_global : PatFrag< 556 (ops node:$ptr, node:$value), 557 (AMDGPUatomic_cmp_swap node:$ptr, node:$value)>, GlobalAddress; 558 559def atomic_cmp_swap_global : PatFrag< 560 (ops node:$ptr, node:$cmp, node:$value), 561 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value)>, GlobalAddress; 562 563 564def atomic_cmp_swap_global_noret : PatFrag< 565 (ops node:$ptr, node:$cmp, node:$value), 566 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value), 567 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>; 568 569def atomic_cmp_swap_global_ret : PatFrag< 570 (ops node:$ptr, node:$cmp, node:$value), 571 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value), 572 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>; 573 574//===----------------------------------------------------------------------===// 575// Misc Pattern Fragments 576//===----------------------------------------------------------------------===// 577 578class Constants { 579int TWO_PI = 0x40c90fdb; 580int PI = 0x40490fdb; 581int TWO_PI_INV = 0x3e22f983; 582int FP_UINT_MAX_PLUS_1 = 0x4f800000; // 1 << 32 in floating point encoding 583int FP16_ONE = 0x3C00; 584int FP16_NEG_ONE = 0xBC00; 585int FP32_ONE = 0x3f800000; 586int FP32_NEG_ONE = 0xbf800000; 587int FP64_ONE = 0x3ff0000000000000; 588int FP64_NEG_ONE = 0xbff0000000000000; 589} 590def CONST : Constants; 591 592def FP_ZERO : PatLeaf < 593 (fpimm), 594 [{return N->getValueAPF().isZero();}] 595>; 596 597def FP_ONE : PatLeaf < 598 (fpimm), 599 [{return N->isExactlyValue(1.0);}] 600>; 601 602def FP_HALF : PatLeaf < 603 (fpimm), 604 [{return N->isExactlyValue(0.5);}] 605>; 606 607/* Generic helper patterns for intrinsics */ 608/* -------------------------------------- */ 609 610class POW_Common <AMDGPUInst log_ieee, AMDGPUInst exp_ieee, AMDGPUInst mul> 611 : AMDGPUPat < 612 (fpow f32:$src0, f32:$src1), 613 (exp_ieee (mul f32:$src1, (log_ieee f32:$src0))) 614>; 615 616/* Other helper patterns */ 617/* --------------------- */ 618 619/* Extract element pattern */ 620class Extract_Element <ValueType sub_type, ValueType vec_type, int sub_idx, 621 SubRegIndex sub_reg> 622 : AMDGPUPat< 623 (sub_type (extractelt vec_type:$src, sub_idx)), 624 (EXTRACT_SUBREG $src, sub_reg) 625>; 626 627/* Insert element pattern */ 628class Insert_Element <ValueType elem_type, ValueType vec_type, 629 int sub_idx, SubRegIndex sub_reg> 630 : AMDGPUPat < 631 (insertelt vec_type:$vec, elem_type:$elem, sub_idx), 632 (INSERT_SUBREG $vec, $elem, sub_reg) 633>; 634 635// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 636// can handle COPY instructions. 637// bitconvert pattern 638class BitConvert <ValueType dt, ValueType st, RegisterClass rc> : AMDGPUPat < 639 (dt (bitconvert (st rc:$src0))), 640 (dt rc:$src0) 641>; 642 643// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 644// can handle COPY instructions. 645class DwordAddrPat<ValueType vt, RegisterClass rc> : AMDGPUPat < 646 (vt (AMDGPUdwordaddr (vt rc:$addr))), 647 (vt rc:$addr) 648>; 649 650// BFI_INT patterns 651 652multiclass BFIPatterns <Instruction BFI_INT, 653 Instruction LoadImm32, 654 RegisterClass RC64> { 655 // Definition from ISA doc: 656 // (y & x) | (z & ~x) 657 def : AMDGPUPat < 658 (or (and i32:$y, i32:$x), (and i32:$z, (not i32:$x))), 659 (BFI_INT $x, $y, $z) 660 >; 661 662 // 64-bit version 663 def : AMDGPUPat < 664 (or (and i64:$y, i64:$x), (and i64:$z, (not i64:$x))), 665 (REG_SEQUENCE RC64, 666 (BFI_INT (i32 (EXTRACT_SUBREG $x, sub0)), 667 (i32 (EXTRACT_SUBREG $y, sub0)), 668 (i32 (EXTRACT_SUBREG $z, sub0))), sub0, 669 (BFI_INT (i32 (EXTRACT_SUBREG $x, sub1)), 670 (i32 (EXTRACT_SUBREG $y, sub1)), 671 (i32 (EXTRACT_SUBREG $z, sub1))), sub1) 672 >; 673 674 // SHA-256 Ch function 675 // z ^ (x & (y ^ z)) 676 def : AMDGPUPat < 677 (xor i32:$z, (and i32:$x, (xor i32:$y, i32:$z))), 678 (BFI_INT $x, $y, $z) 679 >; 680 681 // 64-bit version 682 def : AMDGPUPat < 683 (xor i64:$z, (and i64:$x, (xor i64:$y, i64:$z))), 684 (REG_SEQUENCE RC64, 685 (BFI_INT (i32 (EXTRACT_SUBREG $x, sub0)), 686 (i32 (EXTRACT_SUBREG $y, sub0)), 687 (i32 (EXTRACT_SUBREG $z, sub0))), sub0, 688 (BFI_INT (i32 (EXTRACT_SUBREG $x, sub1)), 689 (i32 (EXTRACT_SUBREG $y, sub1)), 690 (i32 (EXTRACT_SUBREG $z, sub1))), sub1) 691 >; 692 693 def : AMDGPUPat < 694 (fcopysign f32:$src0, f32:$src1), 695 (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0, $src1) 696 >; 697 698 def : AMDGPUPat < 699 (f32 (fcopysign f32:$src0, f64:$src1)), 700 (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0, 701 (i32 (EXTRACT_SUBREG $src1, sub1))) 702 >; 703 704 def : AMDGPUPat < 705 (f64 (fcopysign f64:$src0, f64:$src1)), 706 (REG_SEQUENCE RC64, 707 (i32 (EXTRACT_SUBREG $src0, sub0)), sub0, 708 (BFI_INT (LoadImm32 (i32 0x7fffffff)), 709 (i32 (EXTRACT_SUBREG $src0, sub1)), 710 (i32 (EXTRACT_SUBREG $src1, sub1))), sub1) 711 >; 712 713 def : AMDGPUPat < 714 (f64 (fcopysign f64:$src0, f32:$src1)), 715 (REG_SEQUENCE RC64, 716 (i32 (EXTRACT_SUBREG $src0, sub0)), sub0, 717 (BFI_INT (LoadImm32 (i32 0x7fffffff)), 718 (i32 (EXTRACT_SUBREG $src0, sub1)), 719 $src1), sub1) 720 >; 721} 722 723// SHA-256 Ma patterns 724 725// ((x & z) | (y & (x | z))) -> BFI_INT (XOR x, y), z, y 726multiclass SHA256MaPattern <Instruction BFI_INT, Instruction XOR, RegisterClass RC64> { 727 def : AMDGPUPat < 728 (or (and i32:$x, i32:$z), (and i32:$y, (or i32:$x, i32:$z))), 729 (BFI_INT (XOR i32:$x, i32:$y), i32:$z, i32:$y) 730 >; 731 732 def : AMDGPUPat < 733 (or (and i64:$x, i64:$z), (and i64:$y, (or i64:$x, i64:$z))), 734 (REG_SEQUENCE RC64, 735 (BFI_INT (XOR (i32 (EXTRACT_SUBREG $x, sub0)), 736 (i32 (EXTRACT_SUBREG $y, sub0))), 737 (i32 (EXTRACT_SUBREG $z, sub0)), 738 (i32 (EXTRACT_SUBREG $y, sub0))), sub0, 739 (BFI_INT (XOR (i32 (EXTRACT_SUBREG $x, sub1)), 740 (i32 (EXTRACT_SUBREG $y, sub1))), 741 (i32 (EXTRACT_SUBREG $z, sub1)), 742 (i32 (EXTRACT_SUBREG $y, sub1))), sub1) 743 >; 744} 745 746// Bitfield extract patterns 747 748def IMMZeroBasedBitfieldMask : PatLeaf <(imm), [{ 749 return isMask_32(N->getZExtValue()); 750}]>; 751 752def IMMPopCount : SDNodeXForm<imm, [{ 753 return CurDAG->getTargetConstant(countPopulation(N->getZExtValue()), SDLoc(N), 754 MVT::i32); 755}]>; 756 757multiclass BFEPattern <Instruction UBFE, Instruction SBFE, Instruction MOV> { 758 def : AMDGPUPat < 759 (i32 (and (i32 (srl i32:$src, i32:$rshift)), IMMZeroBasedBitfieldMask:$mask)), 760 (UBFE $src, $rshift, (MOV (i32 (IMMPopCount $mask)))) 761 >; 762 763 // x & ((1 << y) - 1) 764 def : AMDGPUPat < 765 (and i32:$src, (add_oneuse (shl_oneuse 1, i32:$width), -1)), 766 (UBFE $src, (MOV (i32 0)), $width) 767 >; 768 769 // x & ~(-1 << y) 770 def : AMDGPUPat < 771 (and i32:$src, (xor_oneuse (shl_oneuse -1, i32:$width), -1)), 772 (UBFE $src, (MOV (i32 0)), $width) 773 >; 774 775 // x & (-1 >> (bitwidth - y)) 776 def : AMDGPUPat < 777 (and i32:$src, (srl_oneuse -1, (sub 32, i32:$width))), 778 (UBFE $src, (MOV (i32 0)), $width) 779 >; 780 781 // x << (bitwidth - y) >> (bitwidth - y) 782 def : AMDGPUPat < 783 (srl (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)), 784 (UBFE $src, (MOV (i32 0)), $width) 785 >; 786 787 def : AMDGPUPat < 788 (sra (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)), 789 (SBFE $src, (MOV (i32 0)), $width) 790 >; 791} 792 793// rotr pattern 794class ROTRPattern <Instruction BIT_ALIGN> : AMDGPUPat < 795 (rotr i32:$src0, i32:$src1), 796 (BIT_ALIGN $src0, $src0, $src1) 797>; 798 799multiclass IntMed3Pat<Instruction med3Inst, 800 SDPatternOperator min, 801 SDPatternOperator max, 802 SDPatternOperator min_oneuse, 803 SDPatternOperator max_oneuse, 804 ValueType vt = i32> { 805 806 // This matches 16 permutations of 807 // min(max(a, b), max(min(a, b), c)) 808 def : AMDGPUPat < 809 (min (max_oneuse vt:$src0, vt:$src1), 810 (max_oneuse (min_oneuse vt:$src0, vt:$src1), vt:$src2)), 811 (med3Inst vt:$src0, vt:$src1, vt:$src2) 812>; 813 814 // This matches 16 permutations of 815 // max(min(x, y), min(max(x, y), z)) 816 def : AMDGPUPat < 817 (max (min_oneuse vt:$src0, vt:$src1), 818 (min_oneuse (max_oneuse vt:$src0, vt:$src1), vt:$src2)), 819 (med3Inst $src0, $src1, $src2) 820>; 821} 822 823// Special conversion patterns 824 825def cvt_rpi_i32_f32 : PatFrag < 826 (ops node:$src), 827 (fp_to_sint (ffloor (fadd $src, FP_HALF))), 828 [{ (void) N; return TM.Options.NoNaNsFPMath; }] 829>; 830 831def cvt_flr_i32_f32 : PatFrag < 832 (ops node:$src), 833 (fp_to_sint (ffloor $src)), 834 [{ (void)N; return TM.Options.NoNaNsFPMath; }] 835>; 836 837let AddedComplexity = 2 in { 838class IMad24Pat<Instruction Inst, bit HasClamp = 0> : AMDGPUPat < 839 (add (AMDGPUmul_i24 i32:$src0, i32:$src1), i32:$src2), 840 !if(HasClamp, (Inst $src0, $src1, $src2, (i1 0)), 841 (Inst $src0, $src1, $src2)) 842>; 843 844class UMad24Pat<Instruction Inst, bit HasClamp = 0> : AMDGPUPat < 845 (add (AMDGPUmul_u24 i32:$src0, i32:$src1), i32:$src2), 846 !if(HasClamp, (Inst $src0, $src1, $src2, (i1 0)), 847 (Inst $src0, $src1, $src2)) 848>; 849} // AddedComplexity. 850 851class RcpPat<Instruction RcpInst, ValueType vt> : AMDGPUPat < 852 (fdiv FP_ONE, vt:$src), 853 (RcpInst $src) 854>; 855 856class RsqPat<Instruction RsqInst, ValueType vt> : AMDGPUPat < 857 (AMDGPUrcp (fsqrt vt:$src)), 858 (RsqInst $src) 859>; 860 861// Instructions which select to the same v_min_f* 862def fminnum_like : PatFrags<(ops node:$src0, node:$src1), 863 [(fminnum_ieee node:$src0, node:$src1), 864 (fminnum node:$src0, node:$src1)] 865>; 866 867// Instructions which select to the same v_max_f* 868def fmaxnum_like : PatFrags<(ops node:$src0, node:$src1), 869 [(fmaxnum_ieee node:$src0, node:$src1), 870 (fmaxnum node:$src0, node:$src1)] 871>; 872 873def fminnum_like_oneuse : PatFrags<(ops node:$src0, node:$src1), 874 [(fminnum_ieee_oneuse node:$src0, node:$src1), 875 (fminnum_oneuse node:$src0, node:$src1)] 876>; 877 878def fmaxnum_like_oneuse : PatFrags<(ops node:$src0, node:$src1), 879 [(fmaxnum_ieee_oneuse node:$src0, node:$src1), 880 (fmaxnum_oneuse node:$src0, node:$src1)] 881>; 882