1//===-- AMDGPU.td - AMDGPU Tablegen files --------*- tablegen -*-===// 2// 3// The LLVM Compiler Infrastructure 4// 5// This file is distributed under the University of Illinois Open Source 6// License. See LICENSE.TXT for details. 7// 8//===------------------------------------------------------------===// 9 10include "llvm/Target/Target.td" 11 12//===------------------------------------------------------------===// 13// Subtarget Features (device properties) 14//===------------------------------------------------------------===// 15 16def FeatureFP64 : SubtargetFeature<"fp64", 17 "FP64", 18 "true", 19 "Enable double precision operations" 20>; 21 22def FeatureFastFMAF32 : SubtargetFeature<"fast-fmaf", 23 "FastFMAF32", 24 "true", 25 "Assuming f32 fma is at least as fast as mul + add" 26>; 27 28def HalfRate64Ops : SubtargetFeature<"half-rate-64-ops", 29 "HalfRate64Ops", 30 "true", 31 "Most fp64 instructions are half rate instead of quarter" 32>; 33 34def FeatureR600ALUInst : SubtargetFeature<"R600ALUInst", 35 "R600ALUInst", 36 "false", 37 "Older version of ALU instructions encoding" 38>; 39 40def FeatureVertexCache : SubtargetFeature<"HasVertexCache", 41 "HasVertexCache", 42 "true", 43 "Specify use of dedicated vertex cache" 44>; 45 46def FeatureCaymanISA : SubtargetFeature<"caymanISA", 47 "CaymanISA", 48 "true", 49 "Use Cayman ISA" 50>; 51 52def FeatureCFALUBug : SubtargetFeature<"cfalubug", 53 "CFALUBug", 54 "true", 55 "GPU has CF_ALU bug" 56>; 57 58def FeatureFlatAddressSpace : SubtargetFeature<"flat-address-space", 59 "FlatAddressSpace", 60 "true", 61 "Support flat address space" 62>; 63 64def FeatureUnalignedBufferAccess : SubtargetFeature<"unaligned-buffer-access", 65 "UnalignedBufferAccess", 66 "true", 67 "Support unaligned global loads and stores" 68>; 69 70def FeatureTrapHandler: SubtargetFeature<"trap-handler", 71 "TrapHandler", 72 "true", 73 "Trap handler support" 74>; 75 76def FeatureUnalignedScratchAccess : SubtargetFeature<"unaligned-scratch-access", 77 "UnalignedScratchAccess", 78 "true", 79 "Support unaligned scratch loads and stores" 80>; 81 82def FeatureApertureRegs : SubtargetFeature<"aperture-regs", 83 "HasApertureRegs", 84 "true", 85 "Has Memory Aperture Base and Size Registers" 86>; 87 88// XNACK is disabled if SH_MEM_CONFIG.ADDRESS_MODE = GPUVM on chips that support 89// XNACK. The current default kernel driver setting is: 90// - graphics ring: XNACK disabled 91// - compute ring: XNACK enabled 92// 93// If XNACK is enabled, the VMEM latency can be worse. 94// If XNACK is disabled, the 2 SGPRs can be used for general purposes. 95def FeatureXNACK : SubtargetFeature<"xnack", 96 "EnableXNACK", 97 "true", 98 "Enable XNACK support" 99>; 100 101def FeatureSGPRInitBug : SubtargetFeature<"sgpr-init-bug", 102 "SGPRInitBug", 103 "true", 104 "VI SGPR initilization bug requiring a fixed SGPR allocation size" 105>; 106 107class SubtargetFeatureFetchLimit <string Value> : 108 SubtargetFeature <"fetch"#Value, 109 "TexVTXClauseSize", 110 Value, 111 "Limit the maximum number of fetches in a clause to "#Value 112>; 113 114def FeatureFetchLimit8 : SubtargetFeatureFetchLimit <"8">; 115def FeatureFetchLimit16 : SubtargetFeatureFetchLimit <"16">; 116 117class SubtargetFeatureWavefrontSize <int Value> : SubtargetFeature< 118 "wavefrontsize"#Value, 119 "WavefrontSize", 120 !cast<string>(Value), 121 "The number of threads per wavefront" 122>; 123 124def FeatureWavefrontSize16 : SubtargetFeatureWavefrontSize<16>; 125def FeatureWavefrontSize32 : SubtargetFeatureWavefrontSize<32>; 126def FeatureWavefrontSize64 : SubtargetFeatureWavefrontSize<64>; 127 128class SubtargetFeatureLDSBankCount <int Value> : SubtargetFeature < 129 "ldsbankcount"#Value, 130 "LDSBankCount", 131 !cast<string>(Value), 132 "The number of LDS banks per compute unit." 133>; 134 135def FeatureLDSBankCount16 : SubtargetFeatureLDSBankCount<16>; 136def FeatureLDSBankCount32 : SubtargetFeatureLDSBankCount<32>; 137 138class SubtargetFeatureLocalMemorySize <int Value> : SubtargetFeature< 139 "localmemorysize"#Value, 140 "LocalMemorySize", 141 !cast<string>(Value), 142 "The size of local memory in bytes" 143>; 144 145def FeatureGCN : SubtargetFeature<"gcn", 146 "IsGCN", 147 "true", 148 "GCN or newer GPU" 149>; 150 151def FeatureGCN1Encoding : SubtargetFeature<"gcn1-encoding", 152 "GCN1Encoding", 153 "true", 154 "Encoding format for SI and CI" 155>; 156 157def FeatureGCN3Encoding : SubtargetFeature<"gcn3-encoding", 158 "GCN3Encoding", 159 "true", 160 "Encoding format for VI" 161>; 162 163def FeatureCIInsts : SubtargetFeature<"ci-insts", 164 "CIInsts", 165 "true", 166 "Additional intstructions for CI+" 167>; 168 169def FeatureGFX9Insts : SubtargetFeature<"gfx9-insts", 170 "GFX9Insts", 171 "true", 172 "Additional intstructions for GFX9+" 173>; 174 175def FeatureSMemRealTime : SubtargetFeature<"s-memrealtime", 176 "HasSMemRealTime", 177 "true", 178 "Has s_memrealtime instruction" 179>; 180 181def FeatureInv2PiInlineImm : SubtargetFeature<"inv-2pi-inline-imm", 182 "HasInv2PiInlineImm", 183 "true", 184 "Has 1 / (2 * pi) as inline immediate" 185>; 186 187def Feature16BitInsts : SubtargetFeature<"16-bit-insts", 188 "Has16BitInsts", 189 "true", 190 "Has i16/f16 instructions" 191>; 192 193def FeatureMovrel : SubtargetFeature<"movrel", 194 "HasMovrel", 195 "true", 196 "Has v_movrel*_b32 instructions" 197>; 198 199def FeatureVGPRIndexMode : SubtargetFeature<"vgpr-index-mode", 200 "HasVGPRIndexMode", 201 "true", 202 "Has VGPR mode register indexing" 203>; 204 205def FeatureScalarStores : SubtargetFeature<"scalar-stores", 206 "HasScalarStores", 207 "true", 208 "Has store scalar memory instructions" 209>; 210 211def FeatureSDWA : SubtargetFeature<"sdwa", 212 "HasSDWA", 213 "true", 214 "Support SDWA (Sub-DWORD Addressing) extension" 215>; 216 217def FeatureDPP : SubtargetFeature<"dpp", 218 "HasDPP", 219 "true", 220 "Support DPP (Data Parallel Primitives) extension" 221>; 222 223//===------------------------------------------------------------===// 224// Subtarget Features (options and debugging) 225//===------------------------------------------------------------===// 226 227// Some instructions do not support denormals despite this flag. Using 228// fp32 denormals also causes instructions to run at the double 229// precision rate for the device. 230def FeatureFP32Denormals : SubtargetFeature<"fp32-denormals", 231 "FP32Denormals", 232 "true", 233 "Enable single precision denormal handling" 234>; 235 236// Denormal handling for fp64 and fp16 is controlled by the same 237// config register when fp16 supported. 238// TODO: Do we need a separate f16 setting when not legal? 239def FeatureFP64FP16Denormals : SubtargetFeature<"fp64-fp16-denormals", 240 "FP64FP16Denormals", 241 "true", 242 "Enable double and half precision denormal handling", 243 [FeatureFP64] 244>; 245 246def FeatureFP64Denormals : SubtargetFeature<"fp64-denormals", 247 "FP64FP16Denormals", 248 "true", 249 "Enable double and half precision denormal handling", 250 [FeatureFP64, FeatureFP64FP16Denormals] 251>; 252 253def FeatureFP16Denormals : SubtargetFeature<"fp16-denormals", 254 "FP64FP16Denormals", 255 "true", 256 "Enable half precision denormal handling", 257 [FeatureFP64FP16Denormals] 258>; 259 260def FeatureDX10Clamp : SubtargetFeature<"dx10-clamp", 261 "DX10Clamp", 262 "true", 263 "clamp modifier clamps NaNs to 0.0" 264>; 265 266def FeatureFPExceptions : SubtargetFeature<"fp-exceptions", 267 "FPExceptions", 268 "true", 269 "Enable floating point exceptions" 270>; 271 272class FeatureMaxPrivateElementSize<int size> : SubtargetFeature< 273 "max-private-element-size-"#size, 274 "MaxPrivateElementSize", 275 !cast<string>(size), 276 "Maximum private access size may be "#size 277>; 278 279def FeatureMaxPrivateElementSize4 : FeatureMaxPrivateElementSize<4>; 280def FeatureMaxPrivateElementSize8 : FeatureMaxPrivateElementSize<8>; 281def FeatureMaxPrivateElementSize16 : FeatureMaxPrivateElementSize<16>; 282 283def FeatureVGPRSpilling : SubtargetFeature<"vgpr-spilling", 284 "EnableVGPRSpilling", 285 "true", 286 "Enable spilling of VGPRs to scratch memory" 287>; 288 289def FeatureDumpCode : SubtargetFeature <"DumpCode", 290 "DumpCode", 291 "true", 292 "Dump MachineInstrs in the CodeEmitter" 293>; 294 295def FeatureDumpCodeLower : SubtargetFeature <"dumpcode", 296 "DumpCode", 297 "true", 298 "Dump MachineInstrs in the CodeEmitter" 299>; 300 301def FeaturePromoteAlloca : SubtargetFeature <"promote-alloca", 302 "EnablePromoteAlloca", 303 "true", 304 "Enable promote alloca pass" 305>; 306 307// XXX - This should probably be removed once enabled by default 308def FeatureEnableLoadStoreOpt : SubtargetFeature <"load-store-opt", 309 "EnableLoadStoreOpt", 310 "true", 311 "Enable SI load/store optimizer pass" 312>; 313 314// Performance debugging feature. Allow using DS instruction immediate 315// offsets even if the base pointer can't be proven to be base. On SI, 316// base pointer values that won't give the same result as a 16-bit add 317// are not safe to fold, but this will override the conservative test 318// for the base pointer. 319def FeatureEnableUnsafeDSOffsetFolding : SubtargetFeature < 320 "unsafe-ds-offset-folding", 321 "EnableUnsafeDSOffsetFolding", 322 "true", 323 "Force using DS instruction immediate offsets on SI" 324>; 325 326def FeatureEnableSIScheduler : SubtargetFeature<"si-scheduler", 327 "EnableSIScheduler", 328 "true", 329 "Enable SI Machine Scheduler" 330>; 331 332// Unless +-flat-for-global is specified, turn on FlatForGlobal for 333// all OS-es on VI and newer hardware to avoid assertion failures due 334// to missing ADDR64 variants of MUBUF instructions. 335// FIXME: moveToVALU should be able to handle converting addr64 MUBUF 336// instructions. 337 338def FeatureFlatForGlobal : SubtargetFeature<"flat-for-global", 339 "FlatForGlobal", 340 "true", 341 "Force to generate flat instruction for global" 342>; 343 344// Dummy feature used to disable assembler instructions. 345def FeatureDisable : SubtargetFeature<"", 346 "FeatureDisable","true", 347 "Dummy feature to disable assembler instructions" 348>; 349 350class SubtargetFeatureGeneration <string Value, 351 list<SubtargetFeature> Implies> : 352 SubtargetFeature <Value, "Gen", "AMDGPUSubtarget::"#Value, 353 Value#" GPU generation", Implies>; 354 355def FeatureLocalMemorySize0 : SubtargetFeatureLocalMemorySize<0>; 356def FeatureLocalMemorySize32768 : SubtargetFeatureLocalMemorySize<32768>; 357def FeatureLocalMemorySize65536 : SubtargetFeatureLocalMemorySize<65536>; 358 359def FeatureR600 : SubtargetFeatureGeneration<"R600", 360 [FeatureR600ALUInst, FeatureFetchLimit8, FeatureLocalMemorySize0] 361>; 362 363def FeatureR700 : SubtargetFeatureGeneration<"R700", 364 [FeatureFetchLimit16, FeatureLocalMemorySize0] 365>; 366 367def FeatureEvergreen : SubtargetFeatureGeneration<"EVERGREEN", 368 [FeatureFetchLimit16, FeatureLocalMemorySize32768] 369>; 370 371def FeatureNorthernIslands : SubtargetFeatureGeneration<"NORTHERN_ISLANDS", 372 [FeatureFetchLimit16, FeatureWavefrontSize64, 373 FeatureLocalMemorySize32768] 374>; 375 376def FeatureSouthernIslands : SubtargetFeatureGeneration<"SOUTHERN_ISLANDS", 377 [FeatureFP64, FeatureLocalMemorySize32768, 378 FeatureWavefrontSize64, FeatureGCN, FeatureGCN1Encoding, 379 FeatureLDSBankCount32, FeatureMovrel] 380>; 381 382def FeatureSeaIslands : SubtargetFeatureGeneration<"SEA_ISLANDS", 383 [FeatureFP64, FeatureLocalMemorySize65536, 384 FeatureWavefrontSize64, FeatureGCN, FeatureFlatAddressSpace, 385 FeatureGCN1Encoding, FeatureCIInsts, FeatureMovrel] 386>; 387 388def FeatureVolcanicIslands : SubtargetFeatureGeneration<"VOLCANIC_ISLANDS", 389 [FeatureFP64, FeatureLocalMemorySize65536, 390 FeatureWavefrontSize64, FeatureFlatAddressSpace, FeatureGCN, 391 FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts, 392 FeatureSMemRealTime, FeatureVGPRIndexMode, FeatureMovrel, 393 FeatureScalarStores, FeatureInv2PiInlineImm, FeatureSDWA, 394 FeatureDPP 395 ] 396>; 397 398def FeatureGFX9 : SubtargetFeatureGeneration<"GFX9", 399 [FeatureFP64, FeatureLocalMemorySize65536, 400 FeatureWavefrontSize64, FeatureFlatAddressSpace, FeatureGCN, 401 FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts, 402 FeatureSMemRealTime, FeatureScalarStores, FeatureInv2PiInlineImm, 403 FeatureApertureRegs, FeatureGFX9Insts 404 ] 405>; 406 407class SubtargetFeatureISAVersion <int Major, int Minor, int Stepping, 408 list<SubtargetFeature> Implies> 409 : SubtargetFeature < 410 "isaver"#Major#"."#Minor#"."#Stepping, 411 "IsaVersion", 412 "ISAVersion"#Major#"_"#Minor#"_"#Stepping, 413 "Instruction set version number", 414 Implies 415>; 416 417def FeatureISAVersion7_0_0 : SubtargetFeatureISAVersion <7,0,0, 418 [FeatureSeaIslands, 419 FeatureLDSBankCount32]>; 420 421def FeatureISAVersion7_0_1 : SubtargetFeatureISAVersion <7,0,1, 422 [FeatureSeaIslands, 423 HalfRate64Ops, 424 FeatureLDSBankCount32, 425 FeatureFastFMAF32]>; 426 427def FeatureISAVersion7_0_2 : SubtargetFeatureISAVersion <7,0,2, 428 [FeatureSeaIslands, 429 FeatureLDSBankCount16]>; 430 431def FeatureISAVersion8_0_0 : SubtargetFeatureISAVersion <8,0,0, 432 [FeatureVolcanicIslands, 433 FeatureLDSBankCount32, 434 FeatureSGPRInitBug]>; 435 436def FeatureISAVersion8_0_1 : SubtargetFeatureISAVersion <8,0,1, 437 [FeatureVolcanicIslands, 438 FeatureLDSBankCount32, 439 FeatureXNACK]>; 440 441def FeatureISAVersion8_0_2 : SubtargetFeatureISAVersion <8,0,2, 442 [FeatureVolcanicIslands, 443 FeatureLDSBankCount32, 444 FeatureSGPRInitBug]>; 445 446def FeatureISAVersion8_0_3 : SubtargetFeatureISAVersion <8,0,3, 447 [FeatureVolcanicIslands, 448 FeatureLDSBankCount32]>; 449 450def FeatureISAVersion8_0_4 : SubtargetFeatureISAVersion <8,0,4, 451 [FeatureVolcanicIslands, 452 FeatureLDSBankCount32]>; 453 454def FeatureISAVersion8_1_0 : SubtargetFeatureISAVersion <8,1,0, 455 [FeatureVolcanicIslands, 456 FeatureLDSBankCount16, 457 FeatureXNACK]>; 458 459def FeatureISAVersion9_0_0 : SubtargetFeatureISAVersion <9,0,0,[]>; 460def FeatureISAVersion9_0_1 : SubtargetFeatureISAVersion <9,0,1,[]>; 461 462//===----------------------------------------------------------------------===// 463// Debugger related subtarget features. 464//===----------------------------------------------------------------------===// 465 466def FeatureDebuggerInsertNops : SubtargetFeature< 467 "amdgpu-debugger-insert-nops", 468 "DebuggerInsertNops", 469 "true", 470 "Insert one nop instruction for each high level source statement" 471>; 472 473def FeatureDebuggerReserveRegs : SubtargetFeature< 474 "amdgpu-debugger-reserve-regs", 475 "DebuggerReserveRegs", 476 "true", 477 "Reserve registers for debugger usage" 478>; 479 480def FeatureDebuggerEmitPrologue : SubtargetFeature< 481 "amdgpu-debugger-emit-prologue", 482 "DebuggerEmitPrologue", 483 "true", 484 "Emit debugger prologue" 485>; 486 487//===----------------------------------------------------------------------===// 488 489def AMDGPUInstrInfo : InstrInfo { 490 let guessInstructionProperties = 1; 491 let noNamedPositionallyEncodedOperands = 1; 492} 493 494def AMDGPUAsmParser : AsmParser { 495 // Some of the R600 registers have the same name, so this crashes. 496 // For example T0_XYZW and T0_XY both have the asm name T0. 497 let ShouldEmitMatchRegisterName = 0; 498} 499 500def AMDGPUAsmWriter : AsmWriter { 501 int PassSubtarget = 1; 502} 503 504def AMDGPUAsmVariants { 505 string Default = "Default"; 506 int Default_ID = 0; 507 string VOP3 = "VOP3"; 508 int VOP3_ID = 1; 509 string SDWA = "SDWA"; 510 int SDWA_ID = 2; 511 string DPP = "DPP"; 512 int DPP_ID = 3; 513 string Disable = "Disable"; 514 int Disable_ID = 4; 515} 516 517def DefaultAMDGPUAsmParserVariant : AsmParserVariant { 518 let Variant = AMDGPUAsmVariants.Default_ID; 519 let Name = AMDGPUAsmVariants.Default; 520} 521 522def VOP3AsmParserVariant : AsmParserVariant { 523 let Variant = AMDGPUAsmVariants.VOP3_ID; 524 let Name = AMDGPUAsmVariants.VOP3; 525} 526 527def SDWAAsmParserVariant : AsmParserVariant { 528 let Variant = AMDGPUAsmVariants.SDWA_ID; 529 let Name = AMDGPUAsmVariants.SDWA; 530} 531 532def DPPAsmParserVariant : AsmParserVariant { 533 let Variant = AMDGPUAsmVariants.DPP_ID; 534 let Name = AMDGPUAsmVariants.DPP; 535} 536 537def AMDGPU : Target { 538 // Pull in Instruction Info: 539 let InstructionSet = AMDGPUInstrInfo; 540 let AssemblyParsers = [AMDGPUAsmParser]; 541 let AssemblyParserVariants = [DefaultAMDGPUAsmParserVariant, 542 VOP3AsmParserVariant, 543 SDWAAsmParserVariant, 544 DPPAsmParserVariant]; 545 let AssemblyWriters = [AMDGPUAsmWriter]; 546} 547 548// Dummy Instruction itineraries for pseudo instructions 549def ALU_NULL : FuncUnit; 550def NullALU : InstrItinClass; 551 552//===----------------------------------------------------------------------===// 553// Predicate helper class 554//===----------------------------------------------------------------------===// 555 556def TruePredicate : Predicate<"true">; 557 558def isSICI : Predicate< 559 "Subtarget->getGeneration() == AMDGPUSubtarget::SOUTHERN_ISLANDS ||" 560 "Subtarget->getGeneration() == AMDGPUSubtarget::SEA_ISLANDS" 561>, AssemblerPredicate<"FeatureGCN1Encoding">; 562 563def isVI : Predicate < 564 "Subtarget->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS">, 565 AssemblerPredicate<"FeatureGCN3Encoding">; 566 567def isGFX9 : Predicate < 568 "Subtarget->getGeneration() >= AMDGPUSubtarget::GFX9">, 569 AssemblerPredicate<"FeatureGFX9Insts">; 570 571// TODO: Either the name to be changed or we simply use IsCI! 572def isCIVI : Predicate < 573 "Subtarget->getGeneration() >= AMDGPUSubtarget::SEA_ISLANDS">, 574 AssemblerPredicate<"FeatureCIInsts">; 575 576def HasFlatAddressSpace : Predicate<"Subtarget->hasFlatAddressSpace()">; 577 578def Has16BitInsts : Predicate<"Subtarget->has16BitInsts()">; 579 580def HasSDWA : Predicate<"Subtarget->hasSDWA()">, 581 AssemblerPredicate<"FeatureSDWA">; 582 583def HasDPP : Predicate<"Subtarget->hasDPP()">, 584 AssemblerPredicate<"FeatureDPP">; 585 586class PredicateControl { 587 Predicate SubtargetPredicate; 588 Predicate SIAssemblerPredicate = isSICI; 589 Predicate VIAssemblerPredicate = isVI; 590 list<Predicate> AssemblerPredicates = []; 591 Predicate AssemblerPredicate = TruePredicate; 592 list<Predicate> OtherPredicates = []; 593 list<Predicate> Predicates = !listconcat([SubtargetPredicate, AssemblerPredicate], 594 AssemblerPredicates, 595 OtherPredicates); 596} 597 598// Include AMDGPU TD files 599include "R600Schedule.td" 600include "SISchedule.td" 601include "Processors.td" 602include "AMDGPUInstrInfo.td" 603include "AMDGPUIntrinsics.td" 604include "AMDGPURegisterInfo.td" 605include "AMDGPURegisterBanks.td" 606include "AMDGPUInstructions.td" 607include "AMDGPUCallingConv.td" 608