1//===-- AMDGPU.td - AMDGPU Tablegen files --------*- tablegen -*-===// 2// 3// The LLVM Compiler Infrastructure 4// 5// This file is distributed under the University of Illinois Open Source 6// License. See LICENSE.TXT for details. 7// 8//===------------------------------------------------------------===// 9 10include "llvm/Target/Target.td" 11 12//===------------------------------------------------------------===// 13// Subtarget Features (device properties) 14//===------------------------------------------------------------===// 15 16def FeatureFP64 : SubtargetFeature<"fp64", 17 "FP64", 18 "true", 19 "Enable double precision operations" 20>; 21 22def FeatureFastFMAF32 : SubtargetFeature<"fast-fmaf", 23 "FastFMAF32", 24 "true", 25 "Assuming f32 fma is at least as fast as mul + add" 26>; 27 28def HalfRate64Ops : SubtargetFeature<"half-rate-64-ops", 29 "HalfRate64Ops", 30 "true", 31 "Most fp64 instructions are half rate instead of quarter" 32>; 33 34def FeatureR600ALUInst : SubtargetFeature<"R600ALUInst", 35 "R600ALUInst", 36 "false", 37 "Older version of ALU instructions encoding" 38>; 39 40def FeatureVertexCache : SubtargetFeature<"HasVertexCache", 41 "HasVertexCache", 42 "true", 43 "Specify use of dedicated vertex cache" 44>; 45 46def FeatureCaymanISA : SubtargetFeature<"caymanISA", 47 "CaymanISA", 48 "true", 49 "Use Cayman ISA" 50>; 51 52def FeatureCFALUBug : SubtargetFeature<"cfalubug", 53 "CFALUBug", 54 "true", 55 "GPU has CF_ALU bug" 56>; 57 58def FeatureFlatAddressSpace : SubtargetFeature<"flat-address-space", 59 "FlatAddressSpace", 60 "true", 61 "Support flat address space" 62>; 63 64def FeatureUnalignedBufferAccess : SubtargetFeature<"unaligned-buffer-access", 65 "UnalignedBufferAccess", 66 "true", 67 "Support unaligned global loads and stores" 68>; 69 70def FeatureXNACK : SubtargetFeature<"xnack", 71 "EnableXNACK", 72 "true", 73 "Enable XNACK support" 74>; 75 76def FeatureSGPRInitBug : SubtargetFeature<"sgpr-init-bug", 77 "SGPRInitBug", 78 "true", 79 "VI SGPR initilization bug requiring a fixed SGPR allocation size" 80>; 81 82class SubtargetFeatureFetchLimit <string Value> : 83 SubtargetFeature <"fetch"#Value, 84 "TexVTXClauseSize", 85 Value, 86 "Limit the maximum number of fetches in a clause to "#Value 87>; 88 89def FeatureFetchLimit8 : SubtargetFeatureFetchLimit <"8">; 90def FeatureFetchLimit16 : SubtargetFeatureFetchLimit <"16">; 91 92class SubtargetFeatureWavefrontSize <int Value> : SubtargetFeature< 93 "wavefrontsize"#Value, 94 "WavefrontSize", 95 !cast<string>(Value), 96 "The number of threads per wavefront" 97>; 98 99def FeatureWavefrontSize16 : SubtargetFeatureWavefrontSize<16>; 100def FeatureWavefrontSize32 : SubtargetFeatureWavefrontSize<32>; 101def FeatureWavefrontSize64 : SubtargetFeatureWavefrontSize<64>; 102 103class SubtargetFeatureLDSBankCount <int Value> : SubtargetFeature < 104 "ldsbankcount"#Value, 105 "LDSBankCount", 106 !cast<string>(Value), 107 "The number of LDS banks per compute unit." 108>; 109 110def FeatureLDSBankCount16 : SubtargetFeatureLDSBankCount<16>; 111def FeatureLDSBankCount32 : SubtargetFeatureLDSBankCount<32>; 112 113class SubtargetFeatureISAVersion <int Major, int Minor, int Stepping> 114 : SubtargetFeature < 115 "isaver"#Major#"."#Minor#"."#Stepping, 116 "IsaVersion", 117 "ISAVersion"#Major#"_"#Minor#"_"#Stepping, 118 "Instruction set version number" 119>; 120 121def FeatureISAVersion7_0_0 : SubtargetFeatureISAVersion <7,0,0>; 122def FeatureISAVersion7_0_1 : SubtargetFeatureISAVersion <7,0,1>; 123def FeatureISAVersion8_0_0 : SubtargetFeatureISAVersion <8,0,0>; 124def FeatureISAVersion8_0_1 : SubtargetFeatureISAVersion <8,0,1>; 125def FeatureISAVersion8_0_2 : SubtargetFeatureISAVersion <8,0,2>; 126def FeatureISAVersion8_0_3 : SubtargetFeatureISAVersion <8,0,3>; 127 128class SubtargetFeatureLocalMemorySize <int Value> : SubtargetFeature< 129 "localmemorysize"#Value, 130 "LocalMemorySize", 131 !cast<string>(Value), 132 "The size of local memory in bytes" 133>; 134 135def FeatureGCN : SubtargetFeature<"gcn", 136 "IsGCN", 137 "true", 138 "GCN or newer GPU" 139>; 140 141def FeatureGCN1Encoding : SubtargetFeature<"gcn1-encoding", 142 "GCN1Encoding", 143 "true", 144 "Encoding format for SI and CI" 145>; 146 147def FeatureGCN3Encoding : SubtargetFeature<"gcn3-encoding", 148 "GCN3Encoding", 149 "true", 150 "Encoding format for VI" 151>; 152 153def FeatureCIInsts : SubtargetFeature<"ci-insts", 154 "CIInsts", 155 "true", 156 "Additional intstructions for CI+" 157>; 158 159def FeatureSMemRealTime : SubtargetFeature<"s-memrealtime", 160 "HasSMemRealTime", 161 "true", 162 "Has s_memrealtime instruction" 163>; 164 165def Feature16BitInsts : SubtargetFeature<"16-bit-insts", 166 "Has16BitInsts", 167 "true", 168 "Has i16/f16 instructions" 169>; 170 171//===------------------------------------------------------------===// 172// Subtarget Features (options and debugging) 173//===------------------------------------------------------------===// 174 175// Some instructions do not support denormals despite this flag. Using 176// fp32 denormals also causes instructions to run at the double 177// precision rate for the device. 178def FeatureFP32Denormals : SubtargetFeature<"fp32-denormals", 179 "FP32Denormals", 180 "true", 181 "Enable single precision denormal handling" 182>; 183 184def FeatureFP64Denormals : SubtargetFeature<"fp64-denormals", 185 "FP64Denormals", 186 "true", 187 "Enable double precision denormal handling", 188 [FeatureFP64] 189>; 190 191def FeatureFPExceptions : SubtargetFeature<"fp-exceptions", 192 "FPExceptions", 193 "true", 194 "Enable floating point exceptions" 195>; 196 197class FeatureMaxPrivateElementSize<int size> : SubtargetFeature< 198 "max-private-element-size-"#size, 199 "MaxPrivateElementSize", 200 !cast<string>(size), 201 "Maximum private access size may be "#size 202>; 203 204def FeatureMaxPrivateElementSize4 : FeatureMaxPrivateElementSize<4>; 205def FeatureMaxPrivateElementSize8 : FeatureMaxPrivateElementSize<8>; 206def FeatureMaxPrivateElementSize16 : FeatureMaxPrivateElementSize<16>; 207 208def FeatureVGPRSpilling : SubtargetFeature<"vgpr-spilling", 209 "EnableVGPRSpilling", 210 "true", 211 "Enable spilling of VGPRs to scratch memory" 212>; 213 214def FeatureDumpCode : SubtargetFeature <"DumpCode", 215 "DumpCode", 216 "true", 217 "Dump MachineInstrs in the CodeEmitter" 218>; 219 220def FeatureDumpCodeLower : SubtargetFeature <"dumpcode", 221 "DumpCode", 222 "true", 223 "Dump MachineInstrs in the CodeEmitter" 224>; 225 226def FeaturePromoteAlloca : SubtargetFeature <"promote-alloca", 227 "EnablePromoteAlloca", 228 "true", 229 "Enable promote alloca pass" 230>; 231 232// XXX - This should probably be removed once enabled by default 233def FeatureEnableLoadStoreOpt : SubtargetFeature <"load-store-opt", 234 "EnableLoadStoreOpt", 235 "true", 236 "Enable SI load/store optimizer pass" 237>; 238 239// Performance debugging feature. Allow using DS instruction immediate 240// offsets even if the base pointer can't be proven to be base. On SI, 241// base pointer values that won't give the same result as a 16-bit add 242// are not safe to fold, but this will override the conservative test 243// for the base pointer. 244def FeatureEnableUnsafeDSOffsetFolding : SubtargetFeature < 245 "unsafe-ds-offset-folding", 246 "EnableUnsafeDSOffsetFolding", 247 "true", 248 "Force using DS instruction immediate offsets on SI" 249>; 250 251def FeatureEnableSIScheduler : SubtargetFeature<"si-scheduler", 252 "EnableSIScheduler", 253 "true", 254 "Enable SI Machine Scheduler" 255>; 256 257def FeatureFlatForGlobal : SubtargetFeature<"flat-for-global", 258 "FlatForGlobal", 259 "true", 260 "Force to generate flat instruction for global" 261>; 262 263// Dummy feature used to disable assembler instructions. 264def FeatureDisable : SubtargetFeature<"", 265 "FeatureDisable","true", 266 "Dummy feature to disable assembler instructions" 267>; 268 269class SubtargetFeatureGeneration <string Value, 270 list<SubtargetFeature> Implies> : 271 SubtargetFeature <Value, "Gen", "AMDGPUSubtarget::"#Value, 272 Value#" GPU generation", Implies>; 273 274def FeatureLocalMemorySize0 : SubtargetFeatureLocalMemorySize<0>; 275def FeatureLocalMemorySize32768 : SubtargetFeatureLocalMemorySize<32768>; 276def FeatureLocalMemorySize65536 : SubtargetFeatureLocalMemorySize<65536>; 277 278def FeatureR600 : SubtargetFeatureGeneration<"R600", 279 [FeatureR600ALUInst, FeatureFetchLimit8, FeatureLocalMemorySize0] 280>; 281 282def FeatureR700 : SubtargetFeatureGeneration<"R700", 283 [FeatureFetchLimit16, FeatureLocalMemorySize0] 284>; 285 286def FeatureEvergreen : SubtargetFeatureGeneration<"EVERGREEN", 287 [FeatureFetchLimit16, FeatureLocalMemorySize32768] 288>; 289 290def FeatureNorthernIslands : SubtargetFeatureGeneration<"NORTHERN_ISLANDS", 291 [FeatureFetchLimit16, FeatureWavefrontSize64, 292 FeatureLocalMemorySize32768] 293>; 294 295def FeatureSouthernIslands : SubtargetFeatureGeneration<"SOUTHERN_ISLANDS", 296 [FeatureFP64, FeatureLocalMemorySize32768, 297 FeatureWavefrontSize64, FeatureGCN, FeatureGCN1Encoding, 298 FeatureLDSBankCount32] 299>; 300 301def FeatureSeaIslands : SubtargetFeatureGeneration<"SEA_ISLANDS", 302 [FeatureFP64, FeatureLocalMemorySize65536, 303 FeatureWavefrontSize64, FeatureGCN, FeatureFlatAddressSpace, 304 FeatureGCN1Encoding, FeatureCIInsts] 305>; 306 307def FeatureVolcanicIslands : SubtargetFeatureGeneration<"VOLCANIC_ISLANDS", 308 [FeatureFP64, FeatureLocalMemorySize65536, 309 FeatureWavefrontSize64, FeatureFlatAddressSpace, FeatureGCN, 310 FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts, 311 FeatureSMemRealTime 312 ] 313>; 314 315//===----------------------------------------------------------------------===// 316// Debugger related subtarget features. 317//===----------------------------------------------------------------------===// 318 319def FeatureDebuggerInsertNops : SubtargetFeature< 320 "amdgpu-debugger-insert-nops", 321 "DebuggerInsertNops", 322 "true", 323 "Insert one nop instruction for each high level source statement" 324>; 325 326def FeatureDebuggerReserveRegs : SubtargetFeature< 327 "amdgpu-debugger-reserve-regs", 328 "DebuggerReserveRegs", 329 "true", 330 "Reserve registers for debugger usage" 331>; 332 333def FeatureDebuggerEmitPrologue : SubtargetFeature< 334 "amdgpu-debugger-emit-prologue", 335 "DebuggerEmitPrologue", 336 "true", 337 "Emit debugger prologue" 338>; 339 340//===----------------------------------------------------------------------===// 341 342def AMDGPUInstrInfo : InstrInfo { 343 let guessInstructionProperties = 1; 344 let noNamedPositionallyEncodedOperands = 1; 345} 346 347def AMDGPUAsmParser : AsmParser { 348 // Some of the R600 registers have the same name, so this crashes. 349 // For example T0_XYZW and T0_XY both have the asm name T0. 350 let ShouldEmitMatchRegisterName = 0; 351} 352 353def AMDGPUAsmWriter : AsmWriter { 354 int PassSubtarget = 1; 355} 356 357def AMDGPUAsmVariants { 358 string Default = "Default"; 359 int Default_ID = 0; 360 string VOP3 = "VOP3"; 361 int VOP3_ID = 1; 362 string SDWA = "SDWA"; 363 int SDWA_ID = 2; 364 string DPP = "DPP"; 365 int DPP_ID = 3; 366 string Disable = "Disable"; 367 int Disable_ID = 4; 368} 369 370def DefaultAMDGPUAsmParserVariant : AsmParserVariant { 371 let Variant = AMDGPUAsmVariants.Default_ID; 372 let Name = AMDGPUAsmVariants.Default; 373} 374 375def VOP3AsmParserVariant : AsmParserVariant { 376 let Variant = AMDGPUAsmVariants.VOP3_ID; 377 let Name = AMDGPUAsmVariants.VOP3; 378} 379 380def SDWAAsmParserVariant : AsmParserVariant { 381 let Variant = AMDGPUAsmVariants.SDWA_ID; 382 let Name = AMDGPUAsmVariants.SDWA; 383} 384 385def DPPAsmParserVariant : AsmParserVariant { 386 let Variant = AMDGPUAsmVariants.DPP_ID; 387 let Name = AMDGPUAsmVariants.DPP; 388} 389 390def AMDGPU : Target { 391 // Pull in Instruction Info: 392 let InstructionSet = AMDGPUInstrInfo; 393 let AssemblyParsers = [AMDGPUAsmParser]; 394 let AssemblyParserVariants = [DefaultAMDGPUAsmParserVariant, 395 VOP3AsmParserVariant, 396 SDWAAsmParserVariant, 397 DPPAsmParserVariant]; 398 let AssemblyWriters = [AMDGPUAsmWriter]; 399} 400 401// Dummy Instruction itineraries for pseudo instructions 402def ALU_NULL : FuncUnit; 403def NullALU : InstrItinClass; 404 405//===----------------------------------------------------------------------===// 406// Predicate helper class 407//===----------------------------------------------------------------------===// 408 409def TruePredicate : Predicate<"true">; 410 411def isSICI : Predicate< 412 "Subtarget->getGeneration() == AMDGPUSubtarget::SOUTHERN_ISLANDS ||" 413 "Subtarget->getGeneration() == AMDGPUSubtarget::SEA_ISLANDS" 414>, AssemblerPredicate<"FeatureGCN1Encoding">; 415 416def isVI : Predicate < 417 "Subtarget->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS">, 418 AssemblerPredicate<"FeatureGCN3Encoding">; 419 420def isCIVI : Predicate < 421 "Subtarget->getGeneration() == AMDGPUSubtarget::SEA_ISLANDS || " 422 "Subtarget->getGeneration() == AMDGPUSubtarget::VOLCANIC_ISLANDS" 423>, AssemblerPredicate<"FeatureCIInsts">; 424 425def HasFlatAddressSpace : Predicate<"Subtarget->hasFlatAddressSpace()">; 426 427class PredicateControl { 428 Predicate SubtargetPredicate; 429 Predicate SIAssemblerPredicate = isSICI; 430 Predicate VIAssemblerPredicate = isVI; 431 list<Predicate> AssemblerPredicates = []; 432 Predicate AssemblerPredicate = TruePredicate; 433 list<Predicate> OtherPredicates = []; 434 list<Predicate> Predicates = !listconcat([SubtargetPredicate, AssemblerPredicate], 435 AssemblerPredicates, 436 OtherPredicates); 437} 438 439// Include AMDGPU TD files 440include "R600Schedule.td" 441include "SISchedule.td" 442include "Processors.td" 443include "AMDGPUInstrInfo.td" 444include "AMDGPUIntrinsics.td" 445include "AMDGPURegisterInfo.td" 446include "AMDGPUInstructions.td" 447include "AMDGPUCallingConv.td" 448