1//===-- AMDGPUInstructions.td - Common instruction defs ---*- tablegen -*-===// 2// 3// The LLVM Compiler Infrastructure 4// 5// This file is distributed under the University of Illinois Open Source 6// License. See LICENSE.TXT for details. 7// 8//===----------------------------------------------------------------------===// 9// 10// This file contains instruction defs that are common to all hw codegen 11// targets. 12// 13//===----------------------------------------------------------------------===// 14 15class AMDGPUInst <dag outs, dag ins, string asm = "", 16 list<dag> pattern = []> : Instruction { 17 field bit isRegisterLoad = 0; 18 field bit isRegisterStore = 0; 19 20 let Namespace = "AMDGPU"; 21 let OutOperandList = outs; 22 let InOperandList = ins; 23 let AsmString = asm; 24 let Pattern = pattern; 25 let Itinerary = NullALU; 26 27 // SoftFail is a field the disassembler can use to provide a way for 28 // instructions to not match without killing the whole decode process. It is 29 // mainly used for ARM, but Tablegen expects this field to exist or it fails 30 // to build the decode table. 31 field bits<64> SoftFail = 0; 32 33 let DecoderNamespace = Namespace; 34 35 let TSFlags{63} = isRegisterLoad; 36 let TSFlags{62} = isRegisterStore; 37} 38 39class AMDGPUShaderInst <dag outs, dag ins, string asm = "", 40 list<dag> pattern = []> : AMDGPUInst<outs, ins, asm, pattern> { 41 42 field bits<32> Inst = 0xffffffff; 43} 44 45def FP16Denormals : Predicate<"Subtarget.hasFP16Denormals()">; 46def FP32Denormals : Predicate<"Subtarget.hasFP32Denormals()">; 47def FP64Denormals : Predicate<"Subtarget.hasFP64Denormals()">; 48def UnsafeFPMath : Predicate<"TM.Options.UnsafeFPMath">; 49 50def InstFlag : OperandWithDefaultOps <i32, (ops (i32 0))>; 51def ADDRIndirect : ComplexPattern<iPTR, 2, "SelectADDRIndirect", [], []>; 52 53let OperandType = "OPERAND_IMMEDIATE" in { 54 55def u32imm : Operand<i32> { 56 let PrintMethod = "printU32ImmOperand"; 57} 58 59def u16imm : Operand<i16> { 60 let PrintMethod = "printU16ImmOperand"; 61} 62 63def u8imm : Operand<i8> { 64 let PrintMethod = "printU8ImmOperand"; 65} 66 67} // End OperandType = "OPERAND_IMMEDIATE" 68 69//===--------------------------------------------------------------------===// 70// Custom Operands 71//===--------------------------------------------------------------------===// 72def brtarget : Operand<OtherVT>; 73 74//===----------------------------------------------------------------------===// 75// Misc. PatFrags 76//===----------------------------------------------------------------------===// 77 78class HasOneUseBinOp<SDPatternOperator op> : PatFrag< 79 (ops node:$src0, node:$src1), 80 (op $src0, $src1), 81 [{ return N->hasOneUse(); }] 82>; 83 84class HasOneUseTernaryOp<SDPatternOperator op> : PatFrag< 85 (ops node:$src0, node:$src1, node:$src2), 86 (op $src0, $src1, $src2), 87 [{ return N->hasOneUse(); }] 88>; 89 90 91let Properties = [SDNPCommutative, SDNPAssociative] in { 92def smax_oneuse : HasOneUseBinOp<smax>; 93def smin_oneuse : HasOneUseBinOp<smin>; 94def umax_oneuse : HasOneUseBinOp<umax>; 95def umin_oneuse : HasOneUseBinOp<umin>; 96def fminnum_oneuse : HasOneUseBinOp<fminnum>; 97def fmaxnum_oneuse : HasOneUseBinOp<fmaxnum>; 98def and_oneuse : HasOneUseBinOp<and>; 99def or_oneuse : HasOneUseBinOp<or>; 100def xor_oneuse : HasOneUseBinOp<xor>; 101} // Properties = [SDNPCommutative, SDNPAssociative] 102 103def sub_oneuse : HasOneUseBinOp<sub>; 104def shl_oneuse : HasOneUseBinOp<shl>; 105 106def select_oneuse : HasOneUseTernaryOp<select>; 107 108//===----------------------------------------------------------------------===// 109// PatLeafs for floating-point comparisons 110//===----------------------------------------------------------------------===// 111 112def COND_OEQ : PatLeaf < 113 (cond), 114 [{return N->get() == ISD::SETOEQ || N->get() == ISD::SETEQ;}] 115>; 116 117def COND_ONE : PatLeaf < 118 (cond), 119 [{return N->get() == ISD::SETONE || N->get() == ISD::SETNE;}] 120>; 121 122def COND_OGT : PatLeaf < 123 (cond), 124 [{return N->get() == ISD::SETOGT || N->get() == ISD::SETGT;}] 125>; 126 127def COND_OGE : PatLeaf < 128 (cond), 129 [{return N->get() == ISD::SETOGE || N->get() == ISD::SETGE;}] 130>; 131 132def COND_OLT : PatLeaf < 133 (cond), 134 [{return N->get() == ISD::SETOLT || N->get() == ISD::SETLT;}] 135>; 136 137def COND_OLE : PatLeaf < 138 (cond), 139 [{return N->get() == ISD::SETOLE || N->get() == ISD::SETLE;}] 140>; 141 142 143def COND_O : PatLeaf <(cond), [{return N->get() == ISD::SETO;}]>; 144def COND_UO : PatLeaf <(cond), [{return N->get() == ISD::SETUO;}]>; 145 146//===----------------------------------------------------------------------===// 147// PatLeafs for unsigned / unordered comparisons 148//===----------------------------------------------------------------------===// 149 150def COND_UEQ : PatLeaf <(cond), [{return N->get() == ISD::SETUEQ;}]>; 151def COND_UNE : PatLeaf <(cond), [{return N->get() == ISD::SETUNE;}]>; 152def COND_UGT : PatLeaf <(cond), [{return N->get() == ISD::SETUGT;}]>; 153def COND_UGE : PatLeaf <(cond), [{return N->get() == ISD::SETUGE;}]>; 154def COND_ULT : PatLeaf <(cond), [{return N->get() == ISD::SETULT;}]>; 155def COND_ULE : PatLeaf <(cond), [{return N->get() == ISD::SETULE;}]>; 156 157// XXX - For some reason R600 version is preferring to use unordered 158// for setne? 159def COND_UNE_NE : PatLeaf < 160 (cond), 161 [{return N->get() == ISD::SETUNE || N->get() == ISD::SETNE;}] 162>; 163 164//===----------------------------------------------------------------------===// 165// PatLeafs for signed comparisons 166//===----------------------------------------------------------------------===// 167 168def COND_SGT : PatLeaf <(cond), [{return N->get() == ISD::SETGT;}]>; 169def COND_SGE : PatLeaf <(cond), [{return N->get() == ISD::SETGE;}]>; 170def COND_SLT : PatLeaf <(cond), [{return N->get() == ISD::SETLT;}]>; 171def COND_SLE : PatLeaf <(cond), [{return N->get() == ISD::SETLE;}]>; 172 173//===----------------------------------------------------------------------===// 174// PatLeafs for integer equality 175//===----------------------------------------------------------------------===// 176 177def COND_EQ : PatLeaf < 178 (cond), 179 [{return N->get() == ISD::SETEQ || N->get() == ISD::SETUEQ;}] 180>; 181 182def COND_NE : PatLeaf < 183 (cond), 184 [{return N->get() == ISD::SETNE || N->get() == ISD::SETUNE;}] 185>; 186 187def COND_NULL : PatLeaf < 188 (cond), 189 [{(void)N; return false;}] 190>; 191 192 193//===----------------------------------------------------------------------===// 194// Load/Store Pattern Fragments 195//===----------------------------------------------------------------------===// 196 197class PrivateMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{ 198 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS; 199}]>; 200 201class PrivateLoad <SDPatternOperator op> : PrivateMemOp < 202 (ops node:$ptr), (op node:$ptr) 203>; 204 205class PrivateStore <SDPatternOperator op> : PrivateMemOp < 206 (ops node:$value, node:$ptr), (op node:$value, node:$ptr) 207>; 208 209def load_private : PrivateLoad <load>; 210 211def truncstorei8_private : PrivateStore <truncstorei8>; 212def truncstorei16_private : PrivateStore <truncstorei16>; 213def store_private : PrivateStore <store>; 214 215class GlobalMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{ 216 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS; 217}]>; 218 219// Global address space loads 220class GlobalLoad <SDPatternOperator op> : GlobalMemOp < 221 (ops node:$ptr), (op node:$ptr) 222>; 223 224def global_load : GlobalLoad <load>; 225 226// Global address space stores 227class GlobalStore <SDPatternOperator op> : GlobalMemOp < 228 (ops node:$value, node:$ptr), (op node:$value, node:$ptr) 229>; 230 231def global_store : GlobalStore <store>; 232def global_store_atomic : GlobalStore<atomic_store>; 233 234 235class ConstantMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{ 236 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::CONSTANT_ADDRESS; 237}]>; 238 239// Constant address space loads 240class ConstantLoad <SDPatternOperator op> : ConstantMemOp < 241 (ops node:$ptr), (op node:$ptr) 242>; 243 244def constant_load : ConstantLoad<load>; 245 246class LocalMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{ 247 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 248}]>; 249 250// Local address space loads 251class LocalLoad <SDPatternOperator op> : LocalMemOp < 252 (ops node:$ptr), (op node:$ptr) 253>; 254 255class LocalStore <SDPatternOperator op> : LocalMemOp < 256 (ops node:$value, node:$ptr), (op node:$value, node:$ptr) 257>; 258 259class FlatMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{ 260 return cast<MemSDNode>(N)->getAddressSPace() == AMDGPUAS::FLAT_ADDRESS; 261}]>; 262 263class FlatLoad <SDPatternOperator op> : FlatMemOp < 264 (ops node:$ptr), (op node:$ptr) 265>; 266 267class AZExtLoadBase <SDPatternOperator ld_node>: PatFrag<(ops node:$ptr), 268 (ld_node node:$ptr), [{ 269 LoadSDNode *L = cast<LoadSDNode>(N); 270 return L->getExtensionType() == ISD::ZEXTLOAD || 271 L->getExtensionType() == ISD::EXTLOAD; 272}]>; 273 274def az_extload : AZExtLoadBase <unindexedload>; 275 276def az_extloadi8 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 277 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i8; 278}]>; 279 280def az_extloadi8_global : GlobalLoad <az_extloadi8>; 281def sextloadi8_global : GlobalLoad <sextloadi8>; 282 283def az_extloadi8_constant : ConstantLoad <az_extloadi8>; 284def sextloadi8_constant : ConstantLoad <sextloadi8>; 285 286def az_extloadi8_local : LocalLoad <az_extloadi8>; 287def sextloadi8_local : LocalLoad <sextloadi8>; 288 289def extloadi8_private : PrivateLoad <az_extloadi8>; 290def sextloadi8_private : PrivateLoad <sextloadi8>; 291 292def az_extloadi16 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 293 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i16; 294}]>; 295 296def az_extloadi16_global : GlobalLoad <az_extloadi16>; 297def sextloadi16_global : GlobalLoad <sextloadi16>; 298 299def az_extloadi16_constant : ConstantLoad <az_extloadi16>; 300def sextloadi16_constant : ConstantLoad <sextloadi16>; 301 302def az_extloadi16_local : LocalLoad <az_extloadi16>; 303def sextloadi16_local : LocalLoad <sextloadi16>; 304 305def extloadi16_private : PrivateLoad <az_extloadi16>; 306def sextloadi16_private : PrivateLoad <sextloadi16>; 307 308def az_extloadi32 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{ 309 return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i32; 310}]>; 311 312def az_extloadi32_global : GlobalLoad <az_extloadi32>; 313 314def az_extloadi32_flat : FlatLoad <az_extloadi32>; 315 316def az_extloadi32_constant : ConstantLoad <az_extloadi32>; 317 318def truncstorei8_global : GlobalStore <truncstorei8>; 319def truncstorei16_global : GlobalStore <truncstorei16>; 320 321def local_store : LocalStore <store>; 322def truncstorei8_local : LocalStore <truncstorei8>; 323def truncstorei16_local : LocalStore <truncstorei16>; 324 325def local_load : LocalLoad <load>; 326 327class Aligned8Bytes <dag ops, dag frag> : PatFrag <ops, frag, [{ 328 return cast<MemSDNode>(N)->getAlignment() % 8 == 0; 329}]>; 330 331def local_load_aligned8bytes : Aligned8Bytes < 332 (ops node:$ptr), (local_load node:$ptr) 333>; 334 335def local_store_aligned8bytes : Aligned8Bytes < 336 (ops node:$val, node:$ptr), (local_store node:$val, node:$ptr) 337>; 338 339class local_binary_atomic_op<SDNode atomic_op> : 340 PatFrag<(ops node:$ptr, node:$value), 341 (atomic_op node:$ptr, node:$value), [{ 342 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 343}]>; 344 345 346def atomic_swap_local : local_binary_atomic_op<atomic_swap>; 347def atomic_load_add_local : local_binary_atomic_op<atomic_load_add>; 348def atomic_load_sub_local : local_binary_atomic_op<atomic_load_sub>; 349def atomic_load_and_local : local_binary_atomic_op<atomic_load_and>; 350def atomic_load_or_local : local_binary_atomic_op<atomic_load_or>; 351def atomic_load_xor_local : local_binary_atomic_op<atomic_load_xor>; 352def atomic_load_nand_local : local_binary_atomic_op<atomic_load_nand>; 353def atomic_load_min_local : local_binary_atomic_op<atomic_load_min>; 354def atomic_load_max_local : local_binary_atomic_op<atomic_load_max>; 355def atomic_load_umin_local : local_binary_atomic_op<atomic_load_umin>; 356def atomic_load_umax_local : local_binary_atomic_op<atomic_load_umax>; 357 358def mskor_global : PatFrag<(ops node:$val, node:$ptr), 359 (AMDGPUstore_mskor node:$val, node:$ptr), [{ 360 return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS; 361}]>; 362 363multiclass AtomicCmpSwapLocal <SDNode cmp_swap_node> { 364 365 def _32_local : PatFrag < 366 (ops node:$ptr, node:$cmp, node:$swap), 367 (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{ 368 AtomicSDNode *AN = cast<AtomicSDNode>(N); 369 return AN->getMemoryVT() == MVT::i32 && 370 AN->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 371 }]>; 372 373 def _64_local : PatFrag< 374 (ops node:$ptr, node:$cmp, node:$swap), 375 (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{ 376 AtomicSDNode *AN = cast<AtomicSDNode>(N); 377 return AN->getMemoryVT() == MVT::i64 && 378 AN->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS; 379 }]>; 380} 381 382defm atomic_cmp_swap : AtomicCmpSwapLocal <atomic_cmp_swap>; 383 384multiclass global_binary_atomic_op<SDNode atomic_op> { 385 def "" : PatFrag< 386 (ops node:$ptr, node:$value), 387 (atomic_op node:$ptr, node:$value), 388 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}]>; 389 390 def _noret : PatFrag< 391 (ops node:$ptr, node:$value), 392 (atomic_op node:$ptr, node:$value), 393 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>; 394 395 def _ret : PatFrag< 396 (ops node:$ptr, node:$value), 397 (atomic_op node:$ptr, node:$value), 398 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>; 399} 400 401defm atomic_swap_global : global_binary_atomic_op<atomic_swap>; 402defm atomic_add_global : global_binary_atomic_op<atomic_load_add>; 403defm atomic_and_global : global_binary_atomic_op<atomic_load_and>; 404defm atomic_max_global : global_binary_atomic_op<atomic_load_max>; 405defm atomic_min_global : global_binary_atomic_op<atomic_load_min>; 406defm atomic_or_global : global_binary_atomic_op<atomic_load_or>; 407defm atomic_sub_global : global_binary_atomic_op<atomic_load_sub>; 408defm atomic_umax_global : global_binary_atomic_op<atomic_load_umax>; 409defm atomic_umin_global : global_binary_atomic_op<atomic_load_umin>; 410defm atomic_xor_global : global_binary_atomic_op<atomic_load_xor>; 411 412//legacy 413def AMDGPUatomic_cmp_swap_global : PatFrag< 414 (ops node:$ptr, node:$value), 415 (AMDGPUatomic_cmp_swap node:$ptr, node:$value), 416 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}]>; 417 418def atomic_cmp_swap_global : PatFrag< 419 (ops node:$ptr, node:$cmp, node:$value), 420 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value), 421 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}]>; 422 423def atomic_cmp_swap_global_noret : PatFrag< 424 (ops node:$ptr, node:$cmp, node:$value), 425 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value), 426 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>; 427 428def atomic_cmp_swap_global_ret : PatFrag< 429 (ops node:$ptr, node:$cmp, node:$value), 430 (atomic_cmp_swap node:$ptr, node:$cmp, node:$value), 431 [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>; 432 433//===----------------------------------------------------------------------===// 434// Misc Pattern Fragments 435//===----------------------------------------------------------------------===// 436 437class Constants { 438int TWO_PI = 0x40c90fdb; 439int PI = 0x40490fdb; 440int TWO_PI_INV = 0x3e22f983; 441int FP_UINT_MAX_PLUS_1 = 0x4f800000; // 1 << 32 in floating point encoding 442int FP16_ONE = 0x3C00; 443int FP32_ONE = 0x3f800000; 444int FP32_NEG_ONE = 0xbf800000; 445int FP64_ONE = 0x3ff0000000000000; 446int FP64_NEG_ONE = 0xbff0000000000000; 447} 448def CONST : Constants; 449 450def FP_ZERO : PatLeaf < 451 (fpimm), 452 [{return N->getValueAPF().isZero();}] 453>; 454 455def FP_ONE : PatLeaf < 456 (fpimm), 457 [{return N->isExactlyValue(1.0);}] 458>; 459 460def FP_HALF : PatLeaf < 461 (fpimm), 462 [{return N->isExactlyValue(0.5);}] 463>; 464 465let isCodeGenOnly = 1, isPseudo = 1 in { 466 467let usesCustomInserter = 1 in { 468 469class CLAMP <RegisterClass rc> : AMDGPUShaderInst < 470 (outs rc:$dst), 471 (ins rc:$src0), 472 "CLAMP $dst, $src0", 473 [(set f32:$dst, (AMDGPUclamp f32:$src0))] 474>; 475 476class FABS <RegisterClass rc> : AMDGPUShaderInst < 477 (outs rc:$dst), 478 (ins rc:$src0), 479 "FABS $dst, $src0", 480 [(set f32:$dst, (fabs f32:$src0))] 481>; 482 483class FNEG <RegisterClass rc> : AMDGPUShaderInst < 484 (outs rc:$dst), 485 (ins rc:$src0), 486 "FNEG $dst, $src0", 487 [(set f32:$dst, (fneg f32:$src0))] 488>; 489 490} // usesCustomInserter = 1 491 492multiclass RegisterLoadStore <RegisterClass dstClass, Operand addrClass, 493 ComplexPattern addrPat> { 494let UseNamedOperandTable = 1 in { 495 496 def RegisterLoad : AMDGPUShaderInst < 497 (outs dstClass:$dst), 498 (ins addrClass:$addr, i32imm:$chan), 499 "RegisterLoad $dst, $addr", 500 [(set i32:$dst, (AMDGPUregister_load addrPat:$addr, (i32 timm:$chan)))] 501 > { 502 let isRegisterLoad = 1; 503 } 504 505 def RegisterStore : AMDGPUShaderInst < 506 (outs), 507 (ins dstClass:$val, addrClass:$addr, i32imm:$chan), 508 "RegisterStore $val, $addr", 509 [(AMDGPUregister_store i32:$val, addrPat:$addr, (i32 timm:$chan))] 510 > { 511 let isRegisterStore = 1; 512 } 513} 514} 515 516} // End isCodeGenOnly = 1, isPseudo = 1 517 518/* Generic helper patterns for intrinsics */ 519/* -------------------------------------- */ 520 521class POW_Common <AMDGPUInst log_ieee, AMDGPUInst exp_ieee, AMDGPUInst mul> 522 : Pat < 523 (fpow f32:$src0, f32:$src1), 524 (exp_ieee (mul f32:$src1, (log_ieee f32:$src0))) 525>; 526 527/* Other helper patterns */ 528/* --------------------- */ 529 530/* Extract element pattern */ 531class Extract_Element <ValueType sub_type, ValueType vec_type, int sub_idx, 532 SubRegIndex sub_reg> 533 : Pat< 534 (sub_type (extractelt vec_type:$src, sub_idx)), 535 (EXTRACT_SUBREG $src, sub_reg) 536>; 537 538/* Insert element pattern */ 539class Insert_Element <ValueType elem_type, ValueType vec_type, 540 int sub_idx, SubRegIndex sub_reg> 541 : Pat < 542 (insertelt vec_type:$vec, elem_type:$elem, sub_idx), 543 (INSERT_SUBREG $vec, $elem, sub_reg) 544>; 545 546// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 547// can handle COPY instructions. 548// bitconvert pattern 549class BitConvert <ValueType dt, ValueType st, RegisterClass rc> : Pat < 550 (dt (bitconvert (st rc:$src0))), 551 (dt rc:$src0) 552>; 553 554// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer 555// can handle COPY instructions. 556class DwordAddrPat<ValueType vt, RegisterClass rc> : Pat < 557 (vt (AMDGPUdwordaddr (vt rc:$addr))), 558 (vt rc:$addr) 559>; 560 561// BFI_INT patterns 562 563multiclass BFIPatterns <Instruction BFI_INT, 564 Instruction LoadImm32, 565 RegisterClass RC64> { 566 // Definition from ISA doc: 567 // (y & x) | (z & ~x) 568 def : Pat < 569 (or (and i32:$y, i32:$x), (and i32:$z, (not i32:$x))), 570 (BFI_INT $x, $y, $z) 571 >; 572 573 // SHA-256 Ch function 574 // z ^ (x & (y ^ z)) 575 def : Pat < 576 (xor i32:$z, (and i32:$x, (xor i32:$y, i32:$z))), 577 (BFI_INT $x, $y, $z) 578 >; 579 580 def : Pat < 581 (fcopysign f32:$src0, f32:$src1), 582 (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0, $src1) 583 >; 584 585 def : Pat < 586 (f32 (fcopysign f32:$src0, f64:$src1)), 587 (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0, 588 (i32 (EXTRACT_SUBREG $src1, sub1))) 589 >; 590 591 def : Pat < 592 (f64 (fcopysign f64:$src0, f64:$src1)), 593 (REG_SEQUENCE RC64, 594 (i32 (EXTRACT_SUBREG $src0, sub0)), sub0, 595 (BFI_INT (LoadImm32 (i32 0x7fffffff)), 596 (i32 (EXTRACT_SUBREG $src0, sub1)), 597 (i32 (EXTRACT_SUBREG $src1, sub1))), sub1) 598 >; 599 600 def : Pat < 601 (f64 (fcopysign f64:$src0, f32:$src1)), 602 (REG_SEQUENCE RC64, 603 (i32 (EXTRACT_SUBREG $src0, sub0)), sub0, 604 (BFI_INT (LoadImm32 (i32 0x7fffffff)), 605 (i32 (EXTRACT_SUBREG $src0, sub1)), 606 $src1), sub1) 607 >; 608} 609 610// SHA-256 Ma patterns 611 612// ((x & z) | (y & (x | z))) -> BFI_INT (XOR x, y), z, y 613class SHA256MaPattern <Instruction BFI_INT, Instruction XOR> : Pat < 614 (or (and i32:$x, i32:$z), (and i32:$y, (or i32:$x, i32:$z))), 615 (BFI_INT (XOR i32:$x, i32:$y), i32:$z, i32:$y) 616>; 617 618// Bitfield extract patterns 619 620def IMMZeroBasedBitfieldMask : PatLeaf <(imm), [{ 621 return isMask_32(N->getZExtValue()); 622}]>; 623 624def IMMPopCount : SDNodeXForm<imm, [{ 625 return CurDAG->getTargetConstant(countPopulation(N->getZExtValue()), SDLoc(N), 626 MVT::i32); 627}]>; 628 629multiclass BFEPattern <Instruction UBFE, Instruction SBFE, Instruction MOV> { 630 def : Pat < 631 (i32 (and (i32 (srl i32:$src, i32:$rshift)), IMMZeroBasedBitfieldMask:$mask)), 632 (UBFE $src, $rshift, (MOV (i32 (IMMPopCount $mask)))) 633 >; 634 635 def : Pat < 636 (srl (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)), 637 (UBFE $src, (i32 0), $width) 638 >; 639 640 def : Pat < 641 (sra (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)), 642 (SBFE $src, (i32 0), $width) 643 >; 644} 645 646// rotr pattern 647class ROTRPattern <Instruction BIT_ALIGN> : Pat < 648 (rotr i32:$src0, i32:$src1), 649 (BIT_ALIGN $src0, $src0, $src1) 650>; 651 652// This matches 16 permutations of 653// max(min(x, y), min(max(x, y), z)) 654class IntMed3Pat<Instruction med3Inst, 655 SDPatternOperator max, 656 SDPatternOperator max_oneuse, 657 SDPatternOperator min_oneuse> : Pat< 658 (max (min_oneuse i32:$src0, i32:$src1), 659 (min_oneuse (max_oneuse i32:$src0, i32:$src1), i32:$src2)), 660 (med3Inst $src0, $src1, $src2) 661>; 662 663// Special conversion patterns 664 665def cvt_rpi_i32_f32 : PatFrag < 666 (ops node:$src), 667 (fp_to_sint (ffloor (fadd $src, FP_HALF))), 668 [{ (void) N; return TM.Options.NoNaNsFPMath; }] 669>; 670 671def cvt_flr_i32_f32 : PatFrag < 672 (ops node:$src), 673 (fp_to_sint (ffloor $src)), 674 [{ (void)N; return TM.Options.NoNaNsFPMath; }] 675>; 676 677class IMad24Pat<Instruction Inst> : Pat < 678 (add (AMDGPUmul_i24 i32:$src0, i32:$src1), i32:$src2), 679 (Inst $src0, $src1, $src2) 680>; 681 682class UMad24Pat<Instruction Inst> : Pat < 683 (add (AMDGPUmul_u24 i32:$src0, i32:$src1), i32:$src2), 684 (Inst $src0, $src1, $src2) 685>; 686 687class RcpPat<Instruction RcpInst, ValueType vt> : Pat < 688 (fdiv FP_ONE, vt:$src), 689 (RcpInst $src) 690>; 691 692class RsqPat<Instruction RsqInst, ValueType vt> : Pat < 693 (AMDGPUrcp (fsqrt vt:$src)), 694 (RsqInst $src) 695>; 696 697include "R600Instructions.td" 698include "R700Instructions.td" 699include "EvergreenInstructions.td" 700include "CaymanInstructions.td" 701 702include "SIInstrInfo.td" 703 704