1//===-- AMDGPUInstructions.td - Common instruction defs ---*- tablegen -*-===// 2// 3// The LLVM Compiler Infrastructure 4// 5// This file is distributed under the University of Illinois Open Source 6// License. See LICENSE.TXT for details. 7// 8//===----------------------------------------------------------------------===// 9// 10// This file contains instruction defs that are common to all hw codegen 11// targets. 12// 13//===----------------------------------------------------------------------===// 14 15class AMDGPUInst <dag outs, dag ins, string asm, list<dag> pattern> : Instruction { 16 field bit isRegisterLoad = 0; 17 field bit isRegisterStore = 0; 18 19 let Namespace = "AMDGPU"; 20 let OutOperandList = outs; 21 let InOperandList = ins; 22 let AsmString = asm; 23 let Pattern = pattern; 24 let Itinerary = NullALU; 25 26 // SoftFail is a field the disassembler can use to provide a way for 27 // instructions to not match without killing the whole decode process. It is 28 // mainly used for ARM, but Tablegen expects this field to exist or it fails 29 // to build the decode table. 30 field bits<64> SoftFail = 0; 31 32 let DecoderNamespace = Namespace; 33 34 let TSFlags{63} = isRegisterLoad; 35 let TSFlags{62} = isRegisterStore; 36} 37 38class AMDGPUShaderInst <dag outs, dag ins, string asm, list<dag> pattern> 39 : AMDGPUInst<outs, ins, asm, pattern> { 40 41 field bits<32> Inst = 0xffffffff; 42 43} 44 45def FP32Denormals : Predicate<"Subtarget.hasFP32Denormals()">; 46def FP64Denormals : Predicate<"Subtarget.hasFP64Denormals()">; 47def UnsafeFPMath : Predicate<"TM.Options.UnsafeFPMath">; 48 49def InstFlag : OperandWithDefaultOps <i32, (ops (i32 0))>; 50def ADDRIndirect : ComplexPattern<iPTR, 2, "SelectADDRIndirect", [], []>; 51 52let OperandType = "OPERAND_IMMEDIATE" in { 53 54def u32imm : Operand<i32> { 55 let PrintMethod = "printU32ImmOperand"; 56} 57 58def u16imm : Operand<i16> { 59 let PrintMethod = "printU16ImmOperand"; 60} 61 62def u8imm : Operand<i8> { 63 let PrintMethod = "printU8ImmOperand"; 64} 65 66} // End OperandType = "OPERAND_IMMEDIATE" 67 68//===--------------------------------------------------------------------===// 69// Custom Operands 70//===--------------------------------------------------------------------===// 71def brtarget : Operand<OtherVT>; 72 73//===----------------------------------------------------------------------===// 74// PatLeafs for floating-point comparisons 75//===----------------------------------------------------------------------===// 76 77def COND_OEQ : PatLeaf < 78 (cond), 79 [{return N->get() == ISD::SETOEQ || N->get() == ISD::SETEQ;}] 80>; 81 82def COND_ONE : PatLeaf < 83 (cond), 84 [{return N->get() == ISD::SETONE || N->get() == ISD::SETNE;}] 85>; 86 87def COND_OGT : PatLeaf < 88 (cond), 89 [{return N->get() == ISD::SETOGT || N->get() == ISD::SETGT;}] 90>; 91 92def COND_OGE : PatLeaf < 93 (cond), 94 [{return N->get() == ISD::SETOGE || N->get() == ISD::SETGE;}] 95>; 96 97def COND_OLT : PatLeaf < 98 (cond), 99 [{return N->get() == ISD::SETOLT || N->get() == ISD::SETLT;}] 100>; 101 102def COND_OLE : PatLeaf < 103 (cond), 104 [{return N->get() == ISD::SETOLE || N->get() == ISD::SETLE;}] 105>; 106 107 108def COND_O : PatLeaf <(cond), [{return N->get() == ISD::SETO;}]>; 109def COND_UO : PatLeaf <(cond), [{return N->get() == ISD::SETUO;}]>; 110 111//===----------------------------------------------------------------------===// 112// PatLeafs for unsigned / unordered comparisons 113//===----------------------------------------------------------------------===// 114 115def COND_UEQ : PatLeaf <(cond), [{return N->get() == ISD::SETUEQ;}]>; 116def COND_UNE : PatLeaf <(cond), [{return N->get() == ISD::SETUNE;}]>; 117def COND_UGT : PatLeaf <(cond), [{return N->get() == ISD::SETUGT;}]>; 118def COND_UGE : PatLeaf <(cond), [{return N->get() == ISD::SETUGE;}]>; 119def COND_ULT : PatLeaf <(cond), [{return N->get() == ISD::SETULT;}]>; 120def COND_ULE : PatLeaf <(cond), [{return N->get() == ISD::SETULE;}]>; 121 122// XXX - For some reason R600 version is preferring to use unordered 123// for setne? 124def COND_UNE_NE : PatLeaf < 125 (cond), 126 [{return N->get() == ISD::SETUNE || N->get() == ISD::SETNE;}] 127>; 128 129//===----------------------------------------------------------------------===// 130// PatLeafs for signed comparisons 131//===----------------------------------------------------------------------===// 132 133def COND_SGT : PatLeaf <(cond), [{return N->get() == ISD::SETGT;}]>; 134def COND_SGE : PatLeaf <(cond), [{return N->get() == ISD::SETGE;}]>; 135def COND_SLT : PatLeaf <(cond), [{return N->get() == ISD::SETLT;}]>; 136def COND_SLE : PatLeaf <(cond), [{return N->get() == ISD::SETLE;}]>; 137 138//===----------------------------------------------------------------------===// 139// PatLeafs for integer equality 140//===----------------------------------------------------------------------===// 141 142def COND_EQ : PatLeaf < 143 (cond), 144 [{return N->get() == ISD::SETEQ || N->get() == ISD::SETUEQ;}] 145>; 146 147def COND_NE : PatLeaf < 148 (cond), 149 [{return N->get() == ISD::SETNE || N->get() == ISD::SETUNE;}] 150>; 151 152def COND_NULL : PatLeaf < 153 (cond), 154 [{(void)N; return false;}] 155>; 156 157 158//===----------------------------------------------------------------------===// 159// Misc. PatFrags 160//===----------------------------------------------------------------------===// 161 162class HasOneUseBinOp<SDPatternOperator op> : PatFrag< 163 (ops node:$src0, node:$src1), 164 (op $src0, $src1), 165 [{ return N->hasOneUse(); }] 166>; 167 168//===----------------------------------------------------------------------===// 169// Load/Store Pattern Fragments 170//===----------------------------------------------------------------------===// 171 172class PrivateMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{ 173 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS; 174}]>; 175 176class PrivateLoad <SDPatternOperator op> : PrivateMemOp < 177 (ops node:$ptr), (op node:$ptr) 178>; 179 180class PrivateStore <SDPatternOperator op> : PrivateMemOp < 181 (ops node:$value, node:$ptr), (op node:$value, node:$ptr) 182>; 183 184def load_private : PrivateLoad <load>; 185 186def truncstorei8_private : PrivateStore <truncstorei8>; 187def truncstorei16_private : PrivateStore <truncstorei16>; 188def store_private : PrivateStore <store>; 189 190def global_store : PatFrag<(ops node:$val, node:$ptr), 191 (store node:$val, node:$ptr), [{ 192 return isGlobalStore(dyn_cast<StoreSDNode>(N)); 193}]>; 194 195def global_store_atomic : PatFrag<(ops node:$val, node:$ptr), 196 (atomic_store node:$val, node:$ptr), [{ 197 return isGlobalStore(dyn_cast<MemSDNode>(N)); 198}]>; 199 200// Global address space loads 201def global_load : PatFrag<(ops node:$ptr), (load node:$ptr), [{ 202 return isGlobalLoad(dyn_cast<LoadSDNode>(N)); 203}]>; 204 205// Constant address space loads 206def constant_load : PatFrag<(ops node:$ptr), (load node:$ptr), [{ 207 return isConstantLoad(dyn_cast<LoadSDNode>(N), -1); 208}]>; 209 210class AZExtLoadBase <SDPatternOperator ld_node>: PatFrag<(ops node:$ptr), 211 (ld_node node:$ptr), [{ 212 LoadSDNode *L = cast<LoadSDNode>(N); 213 return L->getExtensionType() == ISD::ZEXTLOAD || 214 L->getExtensionType() == ISD::EXTLOAD; 215}]>; 216 217def az_extload : AZExtLoadBase <unindexedload>; 218 219def az_extloadi8 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 220 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i8; 221}]>; 222 223def az_extloadi8_global : PatFrag<(ops node:$ptr), (az_extloadi8 node:$ptr), [{ 224 return isGlobalLoad(dyn_cast<LoadSDNode>(N)); 225}]>; 226 227def sextloadi8_global : PatFrag<(ops node:$ptr), (sextloadi8 node:$ptr), [{ 228 return isGlobalLoad(dyn_cast<LoadSDNode>(N)); 229}]>; 230 231def az_extloadi8_constant : PatFrag<(ops node:$ptr), (az_extloadi8 node:$ptr), [{ 232 return isConstantLoad(dyn_cast<LoadSDNode>(N), -1); 233}]>; 234 235def sextloadi8_constant : PatFrag<(ops node:$ptr), (sextloadi8 node:$ptr), [{ 236 return isConstantLoad(dyn_cast<LoadSDNode>(N), -1); 237}]>; 238 239def az_extloadi8_local : PatFrag<(ops node:$ptr), (az_extloadi8 node:$ptr), [{ 240 return isLocalLoad(dyn_cast<LoadSDNode>(N)); 241}]>; 242 243def sextloadi8_local : PatFrag<(ops node:$ptr), (sextloadi8 node:$ptr), [{ 244 return isLocalLoad(dyn_cast<LoadSDNode>(N)); 245}]>; 246 247def extloadi8_private : PrivateLoad <az_extloadi8>; 248def sextloadi8_private : PrivateLoad <sextloadi8>; 249 250def az_extloadi16 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 251 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i16; 252}]>; 253 254def az_extloadi16_global : PatFrag<(ops node:$ptr), (az_extloadi16 node:$ptr), [{ 255 return isGlobalLoad(dyn_cast<LoadSDNode>(N)); 256}]>; 257 258def sextloadi16_global : PatFrag<(ops node:$ptr), (sextloadi16 node:$ptr), [{ 259 return isGlobalLoad(dyn_cast<LoadSDNode>(N)); 260}]>; 261 262def az_extloadi16_constant : PatFrag<(ops node:$ptr), (az_extloadi16 node:$ptr), [{ 263 return isConstantLoad(dyn_cast<LoadSDNode>(N), -1); 264}]>; 265 266def sextloadi16_constant : PatFrag<(ops node:$ptr), (sextloadi16 node:$ptr), [{ 267 return isConstantLoad(dyn_cast<LoadSDNode>(N), -1); 268}]>; 269 270def az_extloadi16_local : PatFrag<(ops node:$ptr), (az_extloadi16 node:$ptr), [{ 271 return isLocalLoad(dyn_cast<LoadSDNode>(N)); 272}]>; 273 274def sextloadi16_local : PatFrag<(ops node:$ptr), (sextloadi16 node:$ptr), [{ 275 return isLocalLoad(dyn_cast<LoadSDNode>(N)); 276}]>; 277 278def extloadi16_private : PrivateLoad <az_extloadi16>; 279def sextloadi16_private : PrivateLoad <sextloadi16>; 280 281def az_extloadi32 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 282 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i32; 283}]>; 284 285def az_extloadi32_global : PatFrag<(ops node:$ptr), 286 (az_extloadi32 node:$ptr), [{ 287 return isGlobalLoad(dyn_cast<LoadSDNode>(N)); 288}]>; 289 290def az_extloadi32_flat : PatFrag<(ops node:$ptr), 291 (az_extloadi32 node:$ptr), [{ 292 return isFlatLoad(dyn_cast<LoadSDNode>(N)); 293}]>; 294 295def az_extloadi32_constant : PatFrag<(ops node:$ptr), 296 (az_extloadi32 node:$ptr), [{ 297 return isConstantLoad(dyn_cast<LoadSDNode>(N), -1); 298}]>; 299 300def truncstorei8_global : PatFrag<(ops node:$val, node:$ptr), 301 (truncstorei8 node:$val, node:$ptr), [{ 302 return isGlobalStore(dyn_cast<StoreSDNode>(N)); 303}]>; 304 305def truncstorei16_global : PatFrag<(ops node:$val, node:$ptr), 306 (truncstorei16 node:$val, node:$ptr), [{ 307 return isGlobalStore(dyn_cast<StoreSDNode>(N)); 308}]>; 309 310def local_store : PatFrag<(ops node:$val, node:$ptr), 311 (store node:$val, node:$ptr), [{ 312 return isLocalStore(dyn_cast<StoreSDNode>(N)); 313}]>; 314 315def truncstorei8_local : PatFrag<(ops node:$val, node:$ptr), 316 (truncstorei8 node:$val, node:$ptr), [{ 317 return isLocalStore(dyn_cast<StoreSDNode>(N)); 318}]>; 319 320def truncstorei16_local : PatFrag<(ops node:$val, node:$ptr), 321 (truncstorei16 node:$val, node:$ptr), [{ 322 return isLocalStore(dyn_cast<StoreSDNode>(N)); 323}]>; 324 325def local_load : PatFrag<(ops node:$ptr), (load node:$ptr), [{ 326 return isLocalLoad(dyn_cast<LoadSDNode>(N)); 327}]>; 328 329class Aligned8Bytes <dag ops, dag frag> : PatFrag <ops, frag, [{ 330 return cast<MemSDNode>(N)->getAlignment() % 8 == 0; 331}]>; 332 333def local_load_aligned8bytes : Aligned8Bytes < 334 (ops node:$ptr), (local_load node:$ptr) 335>; 336 337def local_store_aligned8bytes : Aligned8Bytes < 338 (ops node:$val, node:$ptr), (local_store node:$val, node:$ptr) 339>; 340 341class local_binary_atomic_op<SDNode atomic_op> : 342 PatFrag<(ops node:$ptr, node:$value), 343 (atomic_op node:$ptr, node:$value), [{ 344 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 345}]>; 346 347 348def atomic_swap_local : local_binary_atomic_op<atomic_swap>; 349def atomic_load_add_local : local_binary_atomic_op<atomic_load_add>; 350def atomic_load_sub_local : local_binary_atomic_op<atomic_load_sub>; 351def atomic_load_and_local : local_binary_atomic_op<atomic_load_and>; 352def atomic_load_or_local : local_binary_atomic_op<atomic_load_or>; 353def atomic_load_xor_local : local_binary_atomic_op<atomic_load_xor>; 354def atomic_load_nand_local : local_binary_atomic_op<atomic_load_nand>; 355def atomic_load_min_local : local_binary_atomic_op<atomic_load_min>; 356def atomic_load_max_local : local_binary_atomic_op<atomic_load_max>; 357def atomic_load_umin_local : local_binary_atomic_op<atomic_load_umin>; 358def atomic_load_umax_local : local_binary_atomic_op<atomic_load_umax>; 359 360def mskor_global : PatFrag<(ops node:$val, node:$ptr), 361 (AMDGPUstore_mskor node:$val, node:$ptr), [{ 362 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS; 363}]>; 364 365multiclass AtomicCmpSwapLocal <SDNode cmp_swap_node> { 366 367 def _32_local : PatFrag < 368 (ops node:$ptr, node:$cmp, node:$swap), 369 (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{ 370 AtomicSDNode *AN = cast<AtomicSDNode>(N); 371 return AN->getMemoryVT() == MVT::i32 && 372 AN->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 373 }]>; 374 375 def _64_local : PatFrag< 376 (ops node:$ptr, node:$cmp, node:$swap), 377 (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{ 378 AtomicSDNode *AN = cast<AtomicSDNode>(N); 379 return AN->getMemoryVT() == MVT::i64 && 380 AN->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 381 }]>; 382} 383 384defm atomic_cmp_swap : AtomicCmpSwapLocal <atomic_cmp_swap>; 385 386def mskor_flat : PatFrag<(ops node:$val, node:$ptr), 387 (AMDGPUstore_mskor node:$val, node:$ptr), [{ 388 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::FLAT_ADDRESS; 389}]>; 390 391class global_binary_atomic_op<SDNode atomic_op> : PatFrag< 392 (ops node:$ptr, node:$value), 393 (atomic_op node:$ptr, node:$value), 394 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}] 395>; 396 397def atomic_swap_global : global_binary_atomic_op<atomic_swap>; 398def atomic_add_global : global_binary_atomic_op<atomic_load_add>; 399def atomic_and_global : global_binary_atomic_op<atomic_load_and>; 400def atomic_max_global : global_binary_atomic_op<atomic_load_max>; 401def atomic_min_global : global_binary_atomic_op<atomic_load_min>; 402def atomic_or_global : global_binary_atomic_op<atomic_load_or>; 403def atomic_sub_global : global_binary_atomic_op<atomic_load_sub>; 404def atomic_umax_global : global_binary_atomic_op<atomic_load_umax>; 405def atomic_umin_global : global_binary_atomic_op<atomic_load_umin>; 406def atomic_xor_global : global_binary_atomic_op<atomic_load_xor>; 407 408def atomic_cmp_swap_global : global_binary_atomic_op<AMDGPUatomic_cmp_swap>; 409def atomic_cmp_swap_global_nortn : PatFrag< 410 (ops node:$ptr, node:$value), 411 (atomic_cmp_swap_global node:$ptr, node:$value), 412 [{ return SDValue(N, 0).use_empty(); }] 413>; 414 415//===----------------------------------------------------------------------===// 416// Misc Pattern Fragments 417//===----------------------------------------------------------------------===// 418 419class Constants { 420int TWO_PI = 0x40c90fdb; 421int PI = 0x40490fdb; 422int TWO_PI_INV = 0x3e22f983; 423int FP_UINT_MAX_PLUS_1 = 0x4f800000; // 1 << 32 in floating point encoding 424int FP32_NEG_ONE = 0xbf800000; 425int FP32_ONE = 0x3f800000; 426int FP64_ONE = 0x3ff0000000000000; 427} 428def CONST : Constants; 429 430def FP_ZERO : PatLeaf < 431 (fpimm), 432 [{return N->getValueAPF().isZero();}] 433>; 434 435def FP_ONE : PatLeaf < 436 (fpimm), 437 [{return N->isExactlyValue(1.0);}] 438>; 439 440def FP_HALF : PatLeaf < 441 (fpimm), 442 [{return N->isExactlyValue(0.5);}] 443>; 444 445let isCodeGenOnly = 1, isPseudo = 1 in { 446 447let usesCustomInserter = 1 in { 448 449class CLAMP <RegisterClass rc> : AMDGPUShaderInst < 450 (outs rc:$dst), 451 (ins rc:$src0), 452 "CLAMP $dst, $src0", 453 [(set f32:$dst, (AMDGPUclamp f32:$src0, (f32 FP_ZERO), (f32 FP_ONE)))] 454>; 455 456class FABS <RegisterClass rc> : AMDGPUShaderInst < 457 (outs rc:$dst), 458 (ins rc:$src0), 459 "FABS $dst, $src0", 460 [(set f32:$dst, (fabs f32:$src0))] 461>; 462 463class FNEG <RegisterClass rc> : AMDGPUShaderInst < 464 (outs rc:$dst), 465 (ins rc:$src0), 466 "FNEG $dst, $src0", 467 [(set f32:$dst, (fneg f32:$src0))] 468>; 469 470} // usesCustomInserter = 1 471 472multiclass RegisterLoadStore <RegisterClass dstClass, Operand addrClass, 473 ComplexPattern addrPat> { 474let UseNamedOperandTable = 1 in { 475 476 def RegisterLoad : AMDGPUShaderInst < 477 (outs dstClass:$dst), 478 (ins addrClass:$addr, i32imm:$chan), 479 "RegisterLoad $dst, $addr", 480 [(set i32:$dst, (AMDGPUregister_load addrPat:$addr, (i32 timm:$chan)))] 481 > { 482 let isRegisterLoad = 1; 483 } 484 485 def RegisterStore : AMDGPUShaderInst < 486 (outs), 487 (ins dstClass:$val, addrClass:$addr, i32imm:$chan), 488 "RegisterStore $val, $addr", 489 [(AMDGPUregister_store i32:$val, addrPat:$addr, (i32 timm:$chan))] 490 > { 491 let isRegisterStore = 1; 492 } 493} 494} 495 496} // End isCodeGenOnly = 1, isPseudo = 1 497 498/* Generic helper patterns for intrinsics */ 499/* -------------------------------------- */ 500 501class POW_Common <AMDGPUInst log_ieee, AMDGPUInst exp_ieee, AMDGPUInst mul> 502 : Pat < 503 (fpow f32:$src0, f32:$src1), 504 (exp_ieee (mul f32:$src1, (log_ieee f32:$src0))) 505>; 506 507/* Other helper patterns */ 508/* --------------------- */ 509 510/* Extract element pattern */ 511class Extract_Element <ValueType sub_type, ValueType vec_type, int sub_idx, 512 SubRegIndex sub_reg> 513 : Pat< 514 (sub_type (extractelt vec_type:$src, sub_idx)), 515 (EXTRACT_SUBREG $src, sub_reg) 516>; 517 518/* Insert element pattern */ 519class Insert_Element <ValueType elem_type, ValueType vec_type, 520 int sub_idx, SubRegIndex sub_reg> 521 : Pat < 522 (insertelt vec_type:$vec, elem_type:$elem, sub_idx), 523 (INSERT_SUBREG $vec, $elem, sub_reg) 524>; 525 526// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 527// can handle COPY instructions. 528// bitconvert pattern 529class BitConvert <ValueType dt, ValueType st, RegisterClass rc> : Pat < 530 (dt (bitconvert (st rc:$src0))), 531 (dt rc:$src0) 532>; 533 534// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 535// can handle COPY instructions. 536class DwordAddrPat<ValueType vt, RegisterClass rc> : Pat < 537 (vt (AMDGPUdwordaddr (vt rc:$addr))), 538 (vt rc:$addr) 539>; 540 541// BFI_INT patterns 542 543multiclass BFIPatterns <Instruction BFI_INT, 544 Instruction LoadImm32, 545 RegisterClass RC64> { 546 // Definition from ISA doc: 547 // (y & x) | (z & ~x) 548 def : Pat < 549 (or (and i32:$y, i32:$x), (and i32:$z, (not i32:$x))), 550 (BFI_INT $x, $y, $z) 551 >; 552 553 // SHA-256 Ch function 554 // z ^ (x & (y ^ z)) 555 def : Pat < 556 (xor i32:$z, (and i32:$x, (xor i32:$y, i32:$z))), 557 (BFI_INT $x, $y, $z) 558 >; 559 560 def : Pat < 561 (fcopysign f32:$src0, f32:$src1), 562 (BFI_INT (LoadImm32 0x7fffffff), $src0, $src1) 563 >; 564 565 def : Pat < 566 (f64 (fcopysign f64:$src0, f64:$src1)), 567 (REG_SEQUENCE RC64, 568 (i32 (EXTRACT_SUBREG $src0, sub0)), sub0, 569 (BFI_INT (LoadImm32 0x7fffffff), 570 (i32 (EXTRACT_SUBREG $src0, sub1)), 571 (i32 (EXTRACT_SUBREG $src1, sub1))), sub1) 572 >; 573} 574 575// SHA-256 Ma patterns 576 577// ((x & z) | (y & (x | z))) -> BFI_INT (XOR x, y), z, y 578class SHA256MaPattern <Instruction BFI_INT, Instruction XOR> : Pat < 579 (or (and i32:$x, i32:$z), (and i32:$y, (or i32:$x, i32:$z))), 580 (BFI_INT (XOR i32:$x, i32:$y), i32:$z, i32:$y) 581>; 582 583// Bitfield extract patterns 584 585def IMMZeroBasedBitfieldMask : PatLeaf <(imm), [{ 586 return isMask_32(N->getZExtValue()); 587}]>; 588 589def IMMPopCount : SDNodeXForm<imm, [{ 590 return CurDAG->getTargetConstant(countPopulation(N->getZExtValue()), SDLoc(N), 591 MVT::i32); 592}]>; 593 594class BFEPattern <Instruction BFE, Instruction MOV> : Pat < 595 (i32 (and (i32 (srl i32:$src, i32:$rshift)), IMMZeroBasedBitfieldMask:$mask)), 596 (BFE $src, $rshift, (MOV (i32 (IMMPopCount $mask)))) 597>; 598 599// rotr pattern 600class ROTRPattern <Instruction BIT_ALIGN> : Pat < 601 (rotr i32:$src0, i32:$src1), 602 (BIT_ALIGN $src0, $src0, $src1) 603>; 604 605// This matches 16 permutations of 606// max(min(x, y), min(max(x, y), z)) 607class IntMed3Pat<Instruction med3Inst, 608 SDPatternOperator max, 609 SDPatternOperator max_oneuse, 610 SDPatternOperator min_oneuse> : Pat< 611 (max (min_oneuse i32:$src0, i32:$src1), 612 (min_oneuse (max_oneuse i32:$src0, i32:$src1), i32:$src2)), 613 (med3Inst $src0, $src1, $src2) 614>; 615 616let Properties = [SDNPCommutative, SDNPAssociative] in { 617def smax_oneuse : HasOneUseBinOp<smax>; 618def smin_oneuse : HasOneUseBinOp<smin>; 619def umax_oneuse : HasOneUseBinOp<umax>; 620def umin_oneuse : HasOneUseBinOp<umin>; 621} // Properties = [SDNPCommutative, SDNPAssociative] 622 623 624// 24-bit arithmetic patterns 625def umul24 : PatFrag <(ops node:$x, node:$y), (mul node:$x, node:$y)>; 626 627// Special conversion patterns 628 629def cvt_rpi_i32_f32 : PatFrag < 630 (ops node:$src), 631 (fp_to_sint (ffloor (fadd $src, FP_HALF))), 632 [{ (void) N; return TM.Options.NoNaNsFPMath; }] 633>; 634 635def cvt_flr_i32_f32 : PatFrag < 636 (ops node:$src), 637 (fp_to_sint (ffloor $src)), 638 [{ (void)N; return TM.Options.NoNaNsFPMath; }] 639>; 640 641class IMad24Pat<Instruction Inst> : Pat < 642 (add (AMDGPUmul_i24 i32:$src0, i32:$src1), i32:$src2), 643 (Inst $src0, $src1, $src2) 644>; 645 646class UMad24Pat<Instruction Inst> : Pat < 647 (add (AMDGPUmul_u24 i32:$src0, i32:$src1), i32:$src2), 648 (Inst $src0, $src1, $src2) 649>; 650 651class RcpPat<Instruction RcpInst, ValueType vt> : Pat < 652 (fdiv FP_ONE, vt:$src), 653 (RcpInst $src) 654>; 655 656class RsqPat<Instruction RsqInst, ValueType vt> : Pat < 657 (AMDGPUrcp (fsqrt vt:$src)), 658 (RsqInst $src) 659>; 660 661include "R600Instructions.td" 662include "R700Instructions.td" 663include "EvergreenInstructions.td" 664include "CaymanInstructions.td" 665 666include "SIInstrInfo.td" 667 668