1//===-- AMDGPUInstrInfo.td - AMDGPU DAG nodes --------------*- tablegen -*-===// 2// 3// The LLVM Compiler Infrastructure 4// 5// This file is distributed under the University of Illinois Open Source 6// License. See LICENSE.TXT for details. 7// 8//===----------------------------------------------------------------------===// 9// 10// This file contains DAG node defintions for the AMDGPU target. 11// 12//===----------------------------------------------------------------------===// 13 14//===----------------------------------------------------------------------===// 15// AMDGPU DAG Profiles 16//===----------------------------------------------------------------------===// 17 18def AMDGPUDTIntTernaryOp : SDTypeProfile<1, 3, [ 19 SDTCisSameAs<0, 1>, SDTCisSameAs<0, 2>, SDTCisInt<0>, SDTCisInt<3> 20]>; 21 22def AMDGPUTrigPreOp : SDTypeProfile<1, 2, 23 [SDTCisSameAs<0, 1>, SDTCisFP<0>, SDTCisInt<2>] 24>; 25 26def AMDGPULdExpOp : SDTypeProfile<1, 2, 27 [SDTCisSameAs<0, 1>, SDTCisFP<0>, SDTCisInt<2>] 28>; 29 30def AMDGPUFPClassOp : SDTypeProfile<1, 2, 31 [SDTCisInt<0>, SDTCisFP<1>, SDTCisInt<2>] 32>; 33 34def AMDGPUFPPackOp : SDTypeProfile<1, 2, 35 [SDTCisFP<1>, SDTCisSameAs<1, 2>] 36>; 37 38def AMDGPUDivScaleOp : SDTypeProfile<2, 3, 39 [SDTCisFP<0>, SDTCisInt<1>, SDTCisSameAs<0, 2>, SDTCisSameAs<0, 3>, SDTCisSameAs<0, 4>] 40>; 41 42// float, float, float, vcc 43def AMDGPUFmasOp : SDTypeProfile<1, 4, 44 [SDTCisFP<0>, SDTCisSameAs<0, 1>, SDTCisSameAs<0, 2>, SDTCisSameAs<0, 3>, SDTCisInt<4>] 45>; 46 47def AMDGPUKillSDT : SDTypeProfile<0, 1, [SDTCisInt<0>]>; 48 49//===----------------------------------------------------------------------===// 50// AMDGPU DAG Nodes 51// 52 53def AMDGPUconstdata_ptr : SDNode< 54 "AMDGPUISD::CONST_DATA_PTR", SDTypeProfile <1, 1, [SDTCisVT<0, iPTR>, 55 SDTCisVT<0, iPTR>]> 56>; 57 58// This argument to this node is a dword address. 59def AMDGPUdwordaddr : SDNode<"AMDGPUISD::DWORDADDR", SDTIntUnaryOp>; 60 61// Force dependencies for vector trunc stores 62def R600dummy_chain : SDNode<"AMDGPUISD::DUMMY_CHAIN", SDTNone, [SDNPHasChain]>; 63 64def AMDGPUcos : SDNode<"AMDGPUISD::COS_HW", SDTFPUnaryOp>; 65def AMDGPUsin : SDNode<"AMDGPUISD::SIN_HW", SDTFPUnaryOp>; 66 67// out = a - floor(a) 68def AMDGPUfract : SDNode<"AMDGPUISD::FRACT", SDTFPUnaryOp>; 69 70// out = 1.0 / a 71def AMDGPUrcp : SDNode<"AMDGPUISD::RCP", SDTFPUnaryOp>; 72 73// out = 1.0 / sqrt(a) 74def AMDGPUrsq : SDNode<"AMDGPUISD::RSQ", SDTFPUnaryOp>; 75 76// out = 1.0 / sqrt(a) 77def AMDGPUrcp_legacy : SDNode<"AMDGPUISD::RCP_LEGACY", SDTFPUnaryOp>; 78def AMDGPUrsq_legacy : SDNode<"AMDGPUISD::RSQ_LEGACY", SDTFPUnaryOp>; 79 80// out = 1.0 / sqrt(a) result clamped to +/- max_float. 81def AMDGPUrsq_clamp : SDNode<"AMDGPUISD::RSQ_CLAMP", SDTFPUnaryOp>; 82 83def AMDGPUldexp : SDNode<"AMDGPUISD::LDEXP", AMDGPULdExpOp>; 84 85def AMDGPUpkrtz_f16_f32 : SDNode<"AMDGPUISD::CVT_PKRTZ_F16_F32", AMDGPUFPPackOp>; 86 87def AMDGPUfp_class : SDNode<"AMDGPUISD::FP_CLASS", AMDGPUFPClassOp>; 88 89// out = max(a, b) a and b are floats, where a nan comparison fails. 90// This is not commutative because this gives the second operand: 91// x < nan ? x : nan -> nan 92// nan < x ? nan : x -> x 93def AMDGPUfmax_legacy : SDNode<"AMDGPUISD::FMAX_LEGACY", SDTFPBinOp, 94 [] 95>; 96 97def AMDGPUfmul_legacy : SDNode<"AMDGPUISD::FMUL_LEGACY", SDTFPBinOp, 98 [SDNPCommutative, SDNPAssociative] 99>; 100 101def AMDGPUclamp : SDNode<"AMDGPUISD::CLAMP", SDTFPUnaryOp>; 102 103// out = min(a, b) a and b are floats, where a nan comparison fails. 104def AMDGPUfmin_legacy : SDNode<"AMDGPUISD::FMIN_LEGACY", SDTFPBinOp, 105 [] 106>; 107 108// FIXME: TableGen doesn't like commutative instructions with more 109// than 2 operands. 110// out = max(a, b, c) a, b and c are floats 111def AMDGPUfmax3 : SDNode<"AMDGPUISD::FMAX3", SDTFPTernaryOp, 112 [/*SDNPCommutative, SDNPAssociative*/] 113>; 114 115// out = max(a, b, c) a, b, and c are signed ints 116def AMDGPUsmax3 : SDNode<"AMDGPUISD::SMAX3", AMDGPUDTIntTernaryOp, 117 [/*SDNPCommutative, SDNPAssociative*/] 118>; 119 120// out = max(a, b, c) a, b and c are unsigned ints 121def AMDGPUumax3 : SDNode<"AMDGPUISD::UMAX3", AMDGPUDTIntTernaryOp, 122 [/*SDNPCommutative, SDNPAssociative*/] 123>; 124 125// out = min(a, b, c) a, b and c are floats 126def AMDGPUfmin3 : SDNode<"AMDGPUISD::FMIN3", SDTFPTernaryOp, 127 [/*SDNPCommutative, SDNPAssociative*/] 128>; 129 130// out = min(a, b, c) a, b and c are signed ints 131def AMDGPUsmin3 : SDNode<"AMDGPUISD::SMIN3", AMDGPUDTIntTernaryOp, 132 [/*SDNPCommutative, SDNPAssociative*/] 133>; 134 135// out = min(a, b) a and b are unsigned ints 136def AMDGPUumin3 : SDNode<"AMDGPUISD::UMIN3", AMDGPUDTIntTernaryOp, 137 [/*SDNPCommutative, SDNPAssociative*/] 138>; 139 140// out = (src0 + src1 > 0xFFFFFFFF) ? 1 : 0 141def AMDGPUcarry : SDNode<"AMDGPUISD::CARRY", SDTIntBinOp, []>; 142 143// out = (src1 > src0) ? 1 : 0 144def AMDGPUborrow : SDNode<"AMDGPUISD::BORROW", SDTIntBinOp, []>; 145 146def AMDGPUSetCCOp : SDTypeProfile<1, 3, [ // setcc 147 SDTCisVT<0, i64>, SDTCisSameAs<1, 2>, SDTCisVT<3, OtherVT> 148]>; 149 150def AMDGPUsetcc : SDNode<"AMDGPUISD::SETCC", AMDGPUSetCCOp>; 151 152def AMDGPUSetRegOp : SDTypeProfile<0, 2, [ 153 SDTCisInt<0>, SDTCisInt<1> 154]>; 155 156def AMDGPUsetreg : SDNode<"AMDGPUISD::SETREG", AMDGPUSetRegOp, [ 157 SDNPHasChain, SDNPSideEffect, SDNPOptInGlue, SDNPOutGlue]>; 158 159def AMDGPUfma : SDNode<"AMDGPUISD::FMA_W_CHAIN", SDTFPTernaryOp, [ 160 SDNPHasChain, SDNPOptInGlue, SDNPOutGlue]>; 161 162def AMDGPUmul : SDNode<"AMDGPUISD::FMUL_W_CHAIN", SDTFPBinOp, [ 163 SDNPHasChain, SDNPOptInGlue, SDNPOutGlue]>; 164 165def AMDGPUcvt_f32_ubyte0 : SDNode<"AMDGPUISD::CVT_F32_UBYTE0", 166 SDTIntToFPOp, []>; 167def AMDGPUcvt_f32_ubyte1 : SDNode<"AMDGPUISD::CVT_F32_UBYTE1", 168 SDTIntToFPOp, []>; 169def AMDGPUcvt_f32_ubyte2 : SDNode<"AMDGPUISD::CVT_F32_UBYTE2", 170 SDTIntToFPOp, []>; 171def AMDGPUcvt_f32_ubyte3 : SDNode<"AMDGPUISD::CVT_F32_UBYTE3", 172 SDTIntToFPOp, []>; 173 174 175// urecip - This operation is a helper for integer division, it returns the 176// result of 1 / a as a fractional unsigned integer. 177// out = (2^32 / a) + e 178// e is rounding error 179def AMDGPUurecip : SDNode<"AMDGPUISD::URECIP", SDTIntUnaryOp>; 180 181// Special case divide preop and flags. 182def AMDGPUdiv_scale : SDNode<"AMDGPUISD::DIV_SCALE", AMDGPUDivScaleOp>; 183 184// Special case divide FMA with scale and flags (src0 = Quotient, 185// src1 = Denominator, src2 = Numerator). 186def AMDGPUdiv_fmas : SDNode<"AMDGPUISD::DIV_FMAS", AMDGPUFmasOp>; 187 188// Single or double precision division fixup. 189// Special case divide fixup and flags(src0 = Quotient, src1 = 190// Denominator, src2 = Numerator). 191def AMDGPUdiv_fixup : SDNode<"AMDGPUISD::DIV_FIXUP", SDTFPTernaryOp>; 192 193// Look Up 2.0 / pi src0 with segment select src1[4:0] 194def AMDGPUtrig_preop : SDNode<"AMDGPUISD::TRIG_PREOP", AMDGPUTrigPreOp>; 195 196def AMDGPUregister_load : SDNode<"AMDGPUISD::REGISTER_LOAD", 197 SDTypeProfile<1, 2, [SDTCisPtrTy<1>, SDTCisInt<2>]>, 198 [SDNPHasChain, SDNPMayLoad]>; 199 200def AMDGPUregister_store : SDNode<"AMDGPUISD::REGISTER_STORE", 201 SDTypeProfile<0, 3, [SDTCisPtrTy<1>, SDTCisInt<2>]>, 202 [SDNPHasChain, SDNPMayStore]>; 203 204// MSKOR instructions are atomic memory instructions used mainly for storing 205// 8-bit and 16-bit values. The definition is: 206// 207// MSKOR(dst, mask, src) MEM[dst] = ((MEM[dst] & ~mask) | src) 208// 209// src0: vec4(src, 0, 0, mask) 210// src1: dst - rat offset (aka pointer) in dwords 211def AMDGPUstore_mskor : SDNode<"AMDGPUISD::STORE_MSKOR", 212 SDTypeProfile<0, 2, []>, 213 [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>; 214 215def AMDGPUatomic_cmp_swap : SDNode<"AMDGPUISD::ATOMIC_CMP_SWAP", 216 SDTypeProfile<1, 2, [SDTCisPtrTy<1>, SDTCisVec<2>]>, 217 [SDNPHasChain, SDNPMayStore, SDNPMayLoad, 218 SDNPMemOperand]>; 219 220def AMDGPUround : SDNode<"ISD::FROUND", 221 SDTypeProfile<1, 1, [SDTCisFP<0>, SDTCisSameAs<0,1>]>>; 222 223def AMDGPUbfe_u32 : SDNode<"AMDGPUISD::BFE_U32", AMDGPUDTIntTernaryOp>; 224def AMDGPUbfe_i32 : SDNode<"AMDGPUISD::BFE_I32", AMDGPUDTIntTernaryOp>; 225def AMDGPUbfi : SDNode<"AMDGPUISD::BFI", AMDGPUDTIntTernaryOp>; 226def AMDGPUbfm : SDNode<"AMDGPUISD::BFM", SDTIntBinOp>; 227 228def AMDGPUffbh_u32 : SDNode<"AMDGPUISD::FFBH_U32", SDTIntUnaryOp>; 229def AMDGPUffbh_i32 : SDNode<"AMDGPUISD::FFBH_I32", SDTIntUnaryOp>; 230 231// Signed and unsigned 24-bit multiply. The highest 8-bits are ignore 232// when performing the mulitply. The result is a 32-bit value. 233def AMDGPUmul_u24 : SDNode<"AMDGPUISD::MUL_U24", SDTIntBinOp, 234 [SDNPCommutative, SDNPAssociative] 235>; 236def AMDGPUmul_i24 : SDNode<"AMDGPUISD::MUL_I24", SDTIntBinOp, 237 [SDNPCommutative, SDNPAssociative] 238>; 239 240def AMDGPUmulhi_u24 : SDNode<"AMDGPUISD::MULHI_U24", SDTIntBinOp, 241 [SDNPCommutative, SDNPAssociative] 242>; 243def AMDGPUmulhi_i24 : SDNode<"AMDGPUISD::MULHI_I24", SDTIntBinOp, 244 [SDNPCommutative, SDNPAssociative] 245>; 246 247def AMDGPUmad_u24 : SDNode<"AMDGPUISD::MAD_U24", AMDGPUDTIntTernaryOp, 248 [] 249>; 250def AMDGPUmad_i24 : SDNode<"AMDGPUISD::MAD_I24", AMDGPUDTIntTernaryOp, 251 [] 252>; 253 254def AMDGPUsmed3 : SDNode<"AMDGPUISD::SMED3", AMDGPUDTIntTernaryOp, 255 [] 256>; 257 258def AMDGPUumed3 : SDNode<"AMDGPUISD::UMED3", AMDGPUDTIntTernaryOp, 259 [] 260>; 261 262def AMDGPUfmed3 : SDNode<"AMDGPUISD::FMED3", SDTFPTernaryOp, []>; 263 264def AMDGPUsendmsg : SDNode<"AMDGPUISD::SENDMSG", 265 SDTypeProfile<0, 1, [SDTCisInt<0>]>, 266 [SDNPHasChain, SDNPInGlue]>; 267 268def AMDGPUsendmsghalt : SDNode<"AMDGPUISD::SENDMSGHALT", 269 SDTypeProfile<0, 1, [SDTCisInt<0>]>, 270 [SDNPHasChain, SDNPInGlue]>; 271 272def AMDGPUinterp_mov : SDNode<"AMDGPUISD::INTERP_MOV", 273 SDTypeProfile<1, 3, [SDTCisFP<0>]>, 274 [SDNPInGlue]>; 275 276def AMDGPUinterp_p1 : SDNode<"AMDGPUISD::INTERP_P1", 277 SDTypeProfile<1, 3, [SDTCisFP<0>]>, 278 [SDNPInGlue, SDNPOutGlue]>; 279 280def AMDGPUinterp_p2 : SDNode<"AMDGPUISD::INTERP_P2", 281 SDTypeProfile<1, 4, [SDTCisFP<0>]>, 282 [SDNPInGlue]>; 283 284 285def AMDGPUkill : SDNode<"AMDGPUISD::KILL", AMDGPUKillSDT, 286 [SDNPHasChain, SDNPSideEffect]>; 287 288// SI+ export 289def AMDGPUExportOp : SDTypeProfile<0, 8, [ 290 SDTCisInt<0>, // i8 tgt 291 SDTCisInt<1>, // i8 en 292 // i32 or f32 src0 293 SDTCisSameAs<3, 2>, // f32 src1 294 SDTCisSameAs<4, 2>, // f32 src2 295 SDTCisSameAs<5, 2>, // f32 src3 296 SDTCisInt<6>, // i1 compr 297 // skip done 298 SDTCisInt<1> // i1 vm 299 300]>; 301 302def AMDGPUexport: SDNode<"AMDGPUISD::EXPORT", AMDGPUExportOp, 303 [SDNPHasChain, SDNPMayStore]>; 304 305def AMDGPUexport_done: SDNode<"AMDGPUISD::EXPORT_DONE", AMDGPUExportOp, 306 [SDNPHasChain, SDNPMayLoad, SDNPMayStore]>; 307 308 309def R600ExportOp : SDTypeProfile<0, 7, [SDTCisFP<0>, SDTCisInt<1>]>; 310 311def R600_EXPORT: SDNode<"AMDGPUISD::R600_EXPORT", R600ExportOp, 312 [SDNPHasChain, SDNPSideEffect]>; 313 314//===----------------------------------------------------------------------===// 315// Flow Control Profile Types 316//===----------------------------------------------------------------------===// 317// Branch instruction where second and third are basic blocks 318def SDTIL_BRCond : SDTypeProfile<0, 2, [ 319 SDTCisVT<0, OtherVT> 320 ]>; 321 322//===----------------------------------------------------------------------===// 323// Flow Control DAG Nodes 324//===----------------------------------------------------------------------===// 325def IL_brcond : SDNode<"AMDGPUISD::BRANCH_COND", SDTIL_BRCond, [SDNPHasChain]>; 326 327//===----------------------------------------------------------------------===// 328// Call/Return DAG Nodes 329//===----------------------------------------------------------------------===// 330def AMDGPUendpgm : SDNode<"AMDGPUISD::ENDPGM", SDTNone, 331 [SDNPHasChain, SDNPOptInGlue]>; 332 333def AMDGPUreturn : SDNode<"AMDGPUISD::RETURN", SDTNone, 334 [SDNPHasChain, SDNPOptInGlue, SDNPVariadic]>; 335