1//===-- AMDGPU.td - AMDGPU Tablegen files --------*- tablegen -*-===//
2//
3//                     The LLVM Compiler Infrastructure
4//
5// This file is distributed under the University of Illinois Open Source
6// License. See LICENSE.TXT for details.
7//
8//===------------------------------------------------------------===//
9
10include "llvm/Target/Target.td"
11
12//===------------------------------------------------------------===//
13// Subtarget Features (device properties)
14//===------------------------------------------------------------===//
15
16def FeatureFP64 : SubtargetFeature<"fp64",
17  "FP64",
18  "true",
19  "Enable double precision operations"
20>;
21
22def FeatureFastFMAF32 : SubtargetFeature<"fast-fmaf",
23  "FastFMAF32",
24  "true",
25  "Assuming f32 fma is at least as fast as mul + add"
26>;
27
28def HalfRate64Ops : SubtargetFeature<"half-rate-64-ops",
29  "HalfRate64Ops",
30  "true",
31  "Most fp64 instructions are half rate instead of quarter"
32>;
33
34def FeatureR600ALUInst : SubtargetFeature<"R600ALUInst",
35  "R600ALUInst",
36  "false",
37  "Older version of ALU instructions encoding"
38>;
39
40def FeatureVertexCache : SubtargetFeature<"HasVertexCache",
41  "HasVertexCache",
42  "true",
43  "Specify use of dedicated vertex cache"
44>;
45
46def FeatureCaymanISA : SubtargetFeature<"caymanISA",
47  "CaymanISA",
48  "true",
49  "Use Cayman ISA"
50>;
51
52def FeatureCFALUBug : SubtargetFeature<"cfalubug",
53  "CFALUBug",
54  "true",
55  "GPU has CF_ALU bug"
56>;
57
58def FeatureFlatAddressSpace : SubtargetFeature<"flat-address-space",
59  "FlatAddressSpace",
60  "true",
61  "Support flat address space"
62>;
63
64def FeatureUnalignedBufferAccess : SubtargetFeature<"unaligned-buffer-access",
65  "UnalignedBufferAccess",
66  "true",
67  "Support unaligned global loads and stores"
68>;
69
70def FeatureTrapHandler: SubtargetFeature<"trap-handler",
71  "TrapHandler",
72  "true",
73  "Trap handler support"
74>;
75
76def FeatureUnalignedScratchAccess : SubtargetFeature<"unaligned-scratch-access",
77  "UnalignedScratchAccess",
78  "true",
79  "Support unaligned scratch loads and stores"
80>;
81
82def FeatureApertureRegs : SubtargetFeature<"aperture-regs",
83  "HasApertureRegs",
84  "true",
85  "Has Memory Aperture Base and Size Registers"
86>;
87
88// XNACK is disabled if SH_MEM_CONFIG.ADDRESS_MODE = GPUVM on chips that support
89// XNACK. The current default kernel driver setting is:
90// - graphics ring: XNACK disabled
91// - compute ring: XNACK enabled
92//
93// If XNACK is enabled, the VMEM latency can be worse.
94// If XNACK is disabled, the 2 SGPRs can be used for general purposes.
95def FeatureXNACK : SubtargetFeature<"xnack",
96  "EnableXNACK",
97  "true",
98  "Enable XNACK support"
99>;
100
101def FeatureSGPRInitBug : SubtargetFeature<"sgpr-init-bug",
102  "SGPRInitBug",
103  "true",
104  "VI SGPR initilization bug requiring a fixed SGPR allocation size"
105>;
106
107class SubtargetFeatureFetchLimit <string Value> :
108                          SubtargetFeature <"fetch"#Value,
109  "TexVTXClauseSize",
110  Value,
111  "Limit the maximum number of fetches in a clause to "#Value
112>;
113
114def FeatureFetchLimit8 : SubtargetFeatureFetchLimit <"8">;
115def FeatureFetchLimit16 : SubtargetFeatureFetchLimit <"16">;
116
117class SubtargetFeatureWavefrontSize <int Value> : SubtargetFeature<
118  "wavefrontsize"#Value,
119  "WavefrontSize",
120  !cast<string>(Value),
121  "The number of threads per wavefront"
122>;
123
124def FeatureWavefrontSize16 : SubtargetFeatureWavefrontSize<16>;
125def FeatureWavefrontSize32 : SubtargetFeatureWavefrontSize<32>;
126def FeatureWavefrontSize64 : SubtargetFeatureWavefrontSize<64>;
127
128class SubtargetFeatureLDSBankCount <int Value> : SubtargetFeature <
129  "ldsbankcount"#Value,
130  "LDSBankCount",
131  !cast<string>(Value),
132  "The number of LDS banks per compute unit."
133>;
134
135def FeatureLDSBankCount16 : SubtargetFeatureLDSBankCount<16>;
136def FeatureLDSBankCount32 : SubtargetFeatureLDSBankCount<32>;
137
138class SubtargetFeatureLocalMemorySize <int Value> : SubtargetFeature<
139  "localmemorysize"#Value,
140  "LocalMemorySize",
141  !cast<string>(Value),
142  "The size of local memory in bytes"
143>;
144
145def FeatureGCN : SubtargetFeature<"gcn",
146  "IsGCN",
147  "true",
148  "GCN or newer GPU"
149>;
150
151def FeatureGCN1Encoding : SubtargetFeature<"gcn1-encoding",
152  "GCN1Encoding",
153  "true",
154  "Encoding format for SI and CI"
155>;
156
157def FeatureGCN3Encoding : SubtargetFeature<"gcn3-encoding",
158  "GCN3Encoding",
159  "true",
160  "Encoding format for VI"
161>;
162
163def FeatureCIInsts : SubtargetFeature<"ci-insts",
164  "CIInsts",
165  "true",
166  "Additional intstructions for CI+"
167>;
168
169def FeatureGFX9Insts : SubtargetFeature<"gfx9-insts",
170  "GFX9Insts",
171  "true",
172  "Additional intstructions for GFX9+"
173>;
174
175def FeatureSMemRealTime : SubtargetFeature<"s-memrealtime",
176  "HasSMemRealTime",
177  "true",
178  "Has s_memrealtime instruction"
179>;
180
181def FeatureInv2PiInlineImm : SubtargetFeature<"inv-2pi-inline-imm",
182  "HasInv2PiInlineImm",
183  "true",
184  "Has 1 / (2 * pi) as inline immediate"
185>;
186
187def Feature16BitInsts : SubtargetFeature<"16-bit-insts",
188  "Has16BitInsts",
189  "true",
190  "Has i16/f16 instructions"
191>;
192
193def FeatureMovrel : SubtargetFeature<"movrel",
194  "HasMovrel",
195  "true",
196  "Has v_movrel*_b32 instructions"
197>;
198
199def FeatureVGPRIndexMode : SubtargetFeature<"vgpr-index-mode",
200  "HasVGPRIndexMode",
201  "true",
202  "Has VGPR mode register indexing"
203>;
204
205def FeatureScalarStores : SubtargetFeature<"scalar-stores",
206  "HasScalarStores",
207  "true",
208  "Has store scalar memory instructions"
209>;
210
211def FeatureSDWA : SubtargetFeature<"sdwa",
212  "HasSDWA",
213  "true",
214  "Support SDWA (Sub-DWORD Addressing) extension"
215>;
216
217def FeatureDPP : SubtargetFeature<"dpp",
218  "HasDPP",
219  "true",
220  "Support DPP (Data Parallel Primitives) extension"
221>;
222
223//===------------------------------------------------------------===//
224// Subtarget Features (options and debugging)
225//===------------------------------------------------------------===//
226
227// Some instructions do not support denormals despite this flag. Using
228// fp32 denormals also causes instructions to run at the double
229// precision rate for the device.
230def FeatureFP32Denormals : SubtargetFeature<"fp32-denormals",
231  "FP32Denormals",
232  "true",
233  "Enable single precision denormal handling"
234>;
235
236// Denormal handling for fp64 and fp16 is controlled by the same
237// config register when fp16 supported.
238// TODO: Do we need a separate f16 setting when not legal?
239def FeatureFP64FP16Denormals : SubtargetFeature<"fp64-fp16-denormals",
240  "FP64FP16Denormals",
241  "true",
242  "Enable double and half precision denormal handling",
243  [FeatureFP64]
244>;
245
246def FeatureFP64Denormals : SubtargetFeature<"fp64-denormals",
247  "FP64FP16Denormals",
248  "true",
249  "Enable double and half precision denormal handling",
250  [FeatureFP64, FeatureFP64FP16Denormals]
251>;
252
253def FeatureFP16Denormals : SubtargetFeature<"fp16-denormals",
254  "FP64FP16Denormals",
255  "true",
256  "Enable half precision denormal handling",
257  [FeatureFP64FP16Denormals]
258>;
259
260def FeatureDX10Clamp : SubtargetFeature<"dx10-clamp",
261  "DX10Clamp",
262  "true",
263  "clamp modifier clamps NaNs to 0.0"
264>;
265
266def FeatureFPExceptions : SubtargetFeature<"fp-exceptions",
267  "FPExceptions",
268  "true",
269  "Enable floating point exceptions"
270>;
271
272class FeatureMaxPrivateElementSize<int size> : SubtargetFeature<
273  "max-private-element-size-"#size,
274  "MaxPrivateElementSize",
275  !cast<string>(size),
276  "Maximum private access size may be "#size
277>;
278
279def FeatureMaxPrivateElementSize4 : FeatureMaxPrivateElementSize<4>;
280def FeatureMaxPrivateElementSize8 : FeatureMaxPrivateElementSize<8>;
281def FeatureMaxPrivateElementSize16 : FeatureMaxPrivateElementSize<16>;
282
283def FeatureVGPRSpilling : SubtargetFeature<"vgpr-spilling",
284  "EnableVGPRSpilling",
285  "true",
286  "Enable spilling of VGPRs to scratch memory"
287>;
288
289def FeatureDumpCode : SubtargetFeature <"DumpCode",
290  "DumpCode",
291  "true",
292  "Dump MachineInstrs in the CodeEmitter"
293>;
294
295def FeatureDumpCodeLower : SubtargetFeature <"dumpcode",
296  "DumpCode",
297  "true",
298  "Dump MachineInstrs in the CodeEmitter"
299>;
300
301def FeaturePromoteAlloca : SubtargetFeature <"promote-alloca",
302  "EnablePromoteAlloca",
303  "true",
304  "Enable promote alloca pass"
305>;
306
307// XXX - This should probably be removed once enabled by default
308def FeatureEnableLoadStoreOpt : SubtargetFeature <"load-store-opt",
309  "EnableLoadStoreOpt",
310  "true",
311  "Enable SI load/store optimizer pass"
312>;
313
314// Performance debugging feature. Allow using DS instruction immediate
315// offsets even if the base pointer can't be proven to be base. On SI,
316// base pointer values that won't give the same result as a 16-bit add
317// are not safe to fold, but this will override the conservative test
318// for the base pointer.
319def FeatureEnableUnsafeDSOffsetFolding : SubtargetFeature <
320  "unsafe-ds-offset-folding",
321  "EnableUnsafeDSOffsetFolding",
322  "true",
323  "Force using DS instruction immediate offsets on SI"
324>;
325
326def FeatureEnableSIScheduler : SubtargetFeature<"si-scheduler",
327  "EnableSIScheduler",
328  "true",
329  "Enable SI Machine Scheduler"
330>;
331
332// Unless +-flat-for-global is specified, turn on FlatForGlobal for
333// all OS-es on VI and newer hardware to avoid assertion failures due
334// to missing ADDR64 variants of MUBUF instructions.
335// FIXME: moveToVALU should be able to handle converting addr64 MUBUF
336// instructions.
337
338def FeatureFlatForGlobal : SubtargetFeature<"flat-for-global",
339  "FlatForGlobal",
340  "true",
341  "Force to generate flat instruction for global"
342>;
343
344// Dummy feature used to disable assembler instructions.
345def FeatureDisable : SubtargetFeature<"",
346  "FeatureDisable","true",
347  "Dummy feature to disable assembler instructions"
348>;
349
350class SubtargetFeatureGeneration <string Value,
351                                  list<SubtargetFeature> Implies> :
352        SubtargetFeature <Value, "Gen", "AMDGPUSubtarget::"#Value,
353                          Value#" GPU generation", Implies>;
354
355def FeatureLocalMemorySize0 : SubtargetFeatureLocalMemorySize<0>;
356def FeatureLocalMemorySize32768 : SubtargetFeatureLocalMemorySize<32768>;
357def FeatureLocalMemorySize65536 : SubtargetFeatureLocalMemorySize<65536>;
358
359def FeatureR600 : SubtargetFeatureGeneration<"R600",
360  [FeatureR600ALUInst, FeatureFetchLimit8, FeatureLocalMemorySize0]
361>;
362
363def FeatureR700 : SubtargetFeatureGeneration<"R700",
364  [FeatureFetchLimit16, FeatureLocalMemorySize0]
365>;
366
367def FeatureEvergreen : SubtargetFeatureGeneration<"EVERGREEN",
368  [FeatureFetchLimit16, FeatureLocalMemorySize32768]
369>;
370
371def FeatureNorthernIslands : SubtargetFeatureGeneration<"NORTHERN_ISLANDS",
372  [FeatureFetchLimit16, FeatureWavefrontSize64,
373   FeatureLocalMemorySize32768]
374>;
375
376def FeatureSouthernIslands : SubtargetFeatureGeneration<"SOUTHERN_ISLANDS",
377  [FeatureFP64, FeatureLocalMemorySize32768,
378  FeatureWavefrontSize64, FeatureGCN, FeatureGCN1Encoding,
379  FeatureLDSBankCount32, FeatureMovrel]
380>;
381
382def FeatureSeaIslands : SubtargetFeatureGeneration<"SEA_ISLANDS",
383  [FeatureFP64, FeatureLocalMemorySize65536,
384  FeatureWavefrontSize64, FeatureGCN, FeatureFlatAddressSpace,
385  FeatureGCN1Encoding, FeatureCIInsts, FeatureMovrel]
386>;
387
388def FeatureVolcanicIslands : SubtargetFeatureGeneration<"VOLCANIC_ISLANDS",
389  [FeatureFP64, FeatureLocalMemorySize65536,
390   FeatureWavefrontSize64, FeatureFlatAddressSpace, FeatureGCN,
391   FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts,
392   FeatureSMemRealTime, FeatureVGPRIndexMode, FeatureMovrel,
393   FeatureScalarStores, FeatureInv2PiInlineImm, FeatureSDWA,
394   FeatureDPP
395  ]
396>;
397
398def FeatureGFX9 : SubtargetFeatureGeneration<"GFX9",
399  [FeatureFP64, FeatureLocalMemorySize65536,
400   FeatureWavefrontSize64, FeatureFlatAddressSpace, FeatureGCN,
401   FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts,
402   FeatureSMemRealTime, FeatureScalarStores, FeatureInv2PiInlineImm,
403   FeatureApertureRegs, FeatureGFX9Insts
404  ]
405>;
406
407class SubtargetFeatureISAVersion <int Major, int Minor, int Stepping,
408                                  list<SubtargetFeature> Implies>
409                                 : SubtargetFeature <
410  "isaver"#Major#"."#Minor#"."#Stepping,
411  "IsaVersion",
412  "ISAVersion"#Major#"_"#Minor#"_"#Stepping,
413  "Instruction set version number",
414  Implies
415>;
416
417def FeatureISAVersion7_0_0 : SubtargetFeatureISAVersion <7,0,0,
418  [FeatureSeaIslands,
419   FeatureLDSBankCount32]>;
420
421def FeatureISAVersion7_0_1 : SubtargetFeatureISAVersion <7,0,1,
422  [FeatureSeaIslands,
423   HalfRate64Ops,
424   FeatureLDSBankCount32,
425   FeatureFastFMAF32]>;
426
427def FeatureISAVersion7_0_2 : SubtargetFeatureISAVersion <7,0,2,
428  [FeatureSeaIslands,
429   FeatureLDSBankCount16]>;
430
431def FeatureISAVersion8_0_0 : SubtargetFeatureISAVersion <8,0,0,
432  [FeatureVolcanicIslands,
433   FeatureLDSBankCount32,
434   FeatureSGPRInitBug]>;
435
436def FeatureISAVersion8_0_1 : SubtargetFeatureISAVersion <8,0,1,
437  [FeatureVolcanicIslands,
438   FeatureLDSBankCount32,
439   FeatureXNACK]>;
440
441def FeatureISAVersion8_0_2 : SubtargetFeatureISAVersion <8,0,2,
442  [FeatureVolcanicIslands,
443   FeatureLDSBankCount32,
444   FeatureSGPRInitBug]>;
445
446def FeatureISAVersion8_0_3 : SubtargetFeatureISAVersion <8,0,3,
447  [FeatureVolcanicIslands,
448   FeatureLDSBankCount32]>;
449
450def FeatureISAVersion8_0_4 : SubtargetFeatureISAVersion <8,0,4,
451  [FeatureVolcanicIslands,
452   FeatureLDSBankCount32]>;
453
454def FeatureISAVersion8_1_0 : SubtargetFeatureISAVersion <8,1,0,
455  [FeatureVolcanicIslands,
456   FeatureLDSBankCount16,
457   FeatureXNACK]>;
458
459def FeatureISAVersion9_0_0 : SubtargetFeatureISAVersion <9,0,0,[]>;
460def FeatureISAVersion9_0_1 : SubtargetFeatureISAVersion <9,0,1,[]>;
461
462//===----------------------------------------------------------------------===//
463// Debugger related subtarget features.
464//===----------------------------------------------------------------------===//
465
466def FeatureDebuggerInsertNops : SubtargetFeature<
467  "amdgpu-debugger-insert-nops",
468  "DebuggerInsertNops",
469  "true",
470  "Insert one nop instruction for each high level source statement"
471>;
472
473def FeatureDebuggerReserveRegs : SubtargetFeature<
474  "amdgpu-debugger-reserve-regs",
475  "DebuggerReserveRegs",
476  "true",
477  "Reserve registers for debugger usage"
478>;
479
480def FeatureDebuggerEmitPrologue : SubtargetFeature<
481  "amdgpu-debugger-emit-prologue",
482  "DebuggerEmitPrologue",
483  "true",
484  "Emit debugger prologue"
485>;
486
487//===----------------------------------------------------------------------===//
488
489def AMDGPUInstrInfo : InstrInfo {
490  let guessInstructionProperties = 1;
491  let noNamedPositionallyEncodedOperands = 1;
492}
493
494def AMDGPUAsmParser : AsmParser {
495  // Some of the R600 registers have the same name, so this crashes.
496  // For example T0_XYZW and T0_XY both have the asm name T0.
497  let ShouldEmitMatchRegisterName = 0;
498}
499
500def AMDGPUAsmWriter : AsmWriter {
501  int PassSubtarget = 1;
502}
503
504def AMDGPUAsmVariants {
505  string Default = "Default";
506  int Default_ID = 0;
507  string VOP3 = "VOP3";
508  int VOP3_ID = 1;
509  string SDWA = "SDWA";
510  int SDWA_ID = 2;
511  string DPP = "DPP";
512  int DPP_ID = 3;
513  string Disable = "Disable";
514  int Disable_ID = 4;
515}
516
517def DefaultAMDGPUAsmParserVariant : AsmParserVariant {
518  let Variant = AMDGPUAsmVariants.Default_ID;
519  let Name = AMDGPUAsmVariants.Default;
520}
521
522def VOP3AsmParserVariant : AsmParserVariant {
523  let Variant = AMDGPUAsmVariants.VOP3_ID;
524  let Name = AMDGPUAsmVariants.VOP3;
525}
526
527def SDWAAsmParserVariant : AsmParserVariant {
528  let Variant = AMDGPUAsmVariants.SDWA_ID;
529  let Name = AMDGPUAsmVariants.SDWA;
530}
531
532def DPPAsmParserVariant : AsmParserVariant {
533  let Variant = AMDGPUAsmVariants.DPP_ID;
534  let Name = AMDGPUAsmVariants.DPP;
535}
536
537def AMDGPU : Target {
538  // Pull in Instruction Info:
539  let InstructionSet = AMDGPUInstrInfo;
540  let AssemblyParsers = [AMDGPUAsmParser];
541  let AssemblyParserVariants = [DefaultAMDGPUAsmParserVariant,
542                                VOP3AsmParserVariant,
543                                SDWAAsmParserVariant,
544                                DPPAsmParserVariant];
545  let AssemblyWriters = [AMDGPUAsmWriter];
546}
547
548// Dummy Instruction itineraries for pseudo instructions
549def ALU_NULL : FuncUnit;
550def NullALU : InstrItinClass;
551
552//===----------------------------------------------------------------------===//
553// Predicate helper class
554//===----------------------------------------------------------------------===//
555
556def TruePredicate : Predicate<"true">;
557
558def isSICI : Predicate<
559  "Subtarget->getGeneration() == AMDGPUSubtarget::SOUTHERN_ISLANDS ||"
560  "Subtarget->getGeneration() == AMDGPUSubtarget::SEA_ISLANDS"
561>, AssemblerPredicate<"FeatureGCN1Encoding">;
562
563def isVI : Predicate <
564  "Subtarget->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS">,
565  AssemblerPredicate<"FeatureGCN3Encoding">;
566
567def isGFX9 : Predicate <
568  "Subtarget->getGeneration() >= AMDGPUSubtarget::GFX9">,
569  AssemblerPredicate<"FeatureGFX9Insts">;
570
571// TODO: Either the name to be changed or we simply use IsCI!
572def isCIVI : Predicate <
573  "Subtarget->getGeneration() >= AMDGPUSubtarget::SEA_ISLANDS">,
574  AssemblerPredicate<"FeatureCIInsts">;
575
576def HasFlatAddressSpace : Predicate<"Subtarget->hasFlatAddressSpace()">;
577
578def Has16BitInsts : Predicate<"Subtarget->has16BitInsts()">;
579
580def HasSDWA : Predicate<"Subtarget->hasSDWA()">,
581  AssemblerPredicate<"FeatureSDWA">;
582
583def HasDPP : Predicate<"Subtarget->hasDPP()">,
584  AssemblerPredicate<"FeatureDPP">;
585
586class PredicateControl {
587  Predicate SubtargetPredicate;
588  Predicate SIAssemblerPredicate = isSICI;
589  Predicate VIAssemblerPredicate = isVI;
590  list<Predicate> AssemblerPredicates = [];
591  Predicate AssemblerPredicate = TruePredicate;
592  list<Predicate> OtherPredicates = [];
593  list<Predicate> Predicates = !listconcat([SubtargetPredicate, AssemblerPredicate],
594                                            AssemblerPredicates,
595                                            OtherPredicates);
596}
597
598// Include AMDGPU TD files
599include "R600Schedule.td"
600include "SISchedule.td"
601include "Processors.td"
602include "AMDGPUInstrInfo.td"
603include "AMDGPUIntrinsics.td"
604include "AMDGPURegisterInfo.td"
605include "AMDGPURegisterBanks.td"
606include "AMDGPUInstructions.td"
607include "AMDGPUCallingConv.td"
608