1 //=====-- AMDGPUSubtarget.h - Define Subtarget for AMDGPU ------*- C++ -*-====//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //==-----------------------------------------------------------------------===//
9 //
10 /// \file
11 /// \brief AMDGPU specific subclass of TargetSubtarget.
12 //
13 //===----------------------------------------------------------------------===//
14 
15 #ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H
16 #define LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H
17 
18 #include "AMDGPU.h"
19 #include "R600InstrInfo.h"
20 #include "R600ISelLowering.h"
21 #include "R600FrameLowering.h"
22 #include "SIInstrInfo.h"
23 #include "SIISelLowering.h"
24 #include "SIFrameLowering.h"
25 #include "Utils/AMDGPUBaseInfo.h"
26 #include "llvm/CodeGen/GlobalISel/GISelAccessor.h"
27 #include "llvm/CodeGen/SelectionDAGTargetInfo.h"
28 #include "llvm/Target/TargetSubtargetInfo.h"
29 
30 #define GET_SUBTARGETINFO_HEADER
31 #include "AMDGPUGenSubtargetInfo.inc"
32 
33 namespace llvm {
34 
35 class SIMachineFunctionInfo;
36 class StringRef;
37 
38 class AMDGPUSubtarget : public AMDGPUGenSubtargetInfo {
39 public:
40   enum Generation {
41     R600 = 0,
42     R700,
43     EVERGREEN,
44     NORTHERN_ISLANDS,
45     SOUTHERN_ISLANDS,
46     SEA_ISLANDS,
47     VOLCANIC_ISLANDS,
48   };
49 
50   enum {
51     ISAVersion0_0_0,
52     ISAVersion7_0_0,
53     ISAVersion7_0_1,
54     ISAVersion8_0_0,
55     ISAVersion8_0_1,
56     ISAVersion8_0_2,
57     ISAVersion8_0_3
58   };
59 
60 protected:
61   // Basic subtarget description.
62   Triple TargetTriple;
63   Generation Gen;
64   unsigned IsaVersion;
65   unsigned WavefrontSize;
66   int LocalMemorySize;
67   int LDSBankCount;
68   unsigned MaxPrivateElementSize;
69 
70   // Possibly statically set by tablegen, but may want to be overridden.
71   bool FastFMAF32;
72   bool HalfRate64Ops;
73 
74   // Dynamially set bits that enable features.
75   bool FP32Denormals;
76   bool FP64Denormals;
77   bool FPExceptions;
78   bool FlatForGlobal;
79   bool UnalignedBufferAccess;
80   bool EnableXNACK;
81   bool DebuggerInsertNops;
82   bool DebuggerReserveRegs;
83   bool DebuggerEmitPrologue;
84 
85   // Used as options.
86   bool EnableVGPRSpilling;
87   bool EnablePromoteAlloca;
88   bool EnableLoadStoreOpt;
89   bool EnableUnsafeDSOffsetFolding;
90   bool EnableSIScheduler;
91   bool DumpCode;
92 
93   // Subtarget statically properties set by tablegen
94   bool FP64;
95   bool IsGCN;
96   bool GCN1Encoding;
97   bool GCN3Encoding;
98   bool CIInsts;
99   bool SGPRInitBug;
100   bool HasSMemRealTime;
101   bool Has16BitInsts;
102   bool FlatAddressSpace;
103   bool R600ALUInst;
104   bool CaymanISA;
105   bool CFALUBug;
106   bool HasVertexCache;
107   short TexVTXClauseSize;
108 
109   // Dummy feature to use for assembler in tablegen.
110   bool FeatureDisable;
111 
112   InstrItineraryData InstrItins;
113   SelectionDAGTargetInfo TSInfo;
114 
115 public:
116   AMDGPUSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
117                   const TargetMachine &TM);
118   virtual ~AMDGPUSubtarget();
119   AMDGPUSubtarget &initializeSubtargetDependencies(const Triple &TT,
120                                                    StringRef GPU, StringRef FS);
121 
122   const AMDGPUInstrInfo *getInstrInfo() const override = 0;
123   const AMDGPUFrameLowering *getFrameLowering() const override = 0;
124   const AMDGPUTargetLowering *getTargetLowering() const override = 0;
125   const AMDGPURegisterInfo *getRegisterInfo() const override = 0;
126 
127   const InstrItineraryData *getInstrItineraryData() const override {
128     return &InstrItins;
129   }
130 
131   // Nothing implemented, just prevent crashes on use.
132   const SelectionDAGTargetInfo *getSelectionDAGInfo() const override {
133     return &TSInfo;
134   }
135 
136   void ParseSubtargetFeatures(StringRef CPU, StringRef FS);
137 
138   bool isAmdHsaOS() const {
139     return TargetTriple.getOS() == Triple::AMDHSA;
140   }
141 
142   bool isMesa3DOS() const {
143     return TargetTriple.getOS() == Triple::Mesa3D;
144   }
145 
146   bool isOpenCLEnv() const {
147     return TargetTriple.getEnvironment() == Triple::OpenCL;
148   }
149 
150   Generation getGeneration() const {
151     return Gen;
152   }
153 
154   unsigned getWavefrontSize() const {
155     return WavefrontSize;
156   }
157 
158   int getLocalMemorySize() const {
159     return LocalMemorySize;
160   }
161 
162   int getLDSBankCount() const {
163     return LDSBankCount;
164   }
165 
166   unsigned getMaxPrivateElementSize() const {
167     return MaxPrivateElementSize;
168   }
169 
170   bool hasHWFP64() const {
171     return FP64;
172   }
173 
174   bool hasFastFMAF32() const {
175     return FastFMAF32;
176   }
177 
178   bool hasHalfRate64Ops() const {
179     return HalfRate64Ops;
180   }
181 
182   bool hasAddr64() const {
183     return (getGeneration() < VOLCANIC_ISLANDS);
184   }
185 
186   bool hasBFE() const {
187     return (getGeneration() >= EVERGREEN);
188   }
189 
190   bool hasBFI() const {
191     return (getGeneration() >= EVERGREEN);
192   }
193 
194   bool hasBFM() const {
195     return hasBFE();
196   }
197 
198   bool hasBCNT(unsigned Size) const {
199     if (Size == 32)
200       return (getGeneration() >= EVERGREEN);
201 
202     if (Size == 64)
203       return (getGeneration() >= SOUTHERN_ISLANDS);
204 
205     return false;
206   }
207 
208   bool hasMulU24() const {
209     return (getGeneration() >= EVERGREEN);
210   }
211 
212   bool hasMulI24() const {
213     return (getGeneration() >= SOUTHERN_ISLANDS ||
214             hasCaymanISA());
215   }
216 
217   bool hasFFBL() const {
218     return (getGeneration() >= EVERGREEN);
219   }
220 
221   bool hasFFBH() const {
222     return (getGeneration() >= EVERGREEN);
223   }
224 
225   bool hasCARRY() const {
226     return (getGeneration() >= EVERGREEN);
227   }
228 
229   bool hasBORROW() const {
230     return (getGeneration() >= EVERGREEN);
231   }
232 
233   bool hasCaymanISA() const {
234     return CaymanISA;
235   }
236 
237   bool isPromoteAllocaEnabled() const {
238     return EnablePromoteAlloca;
239   }
240 
241   bool unsafeDSOffsetFoldingEnabled() const {
242     return EnableUnsafeDSOffsetFolding;
243   }
244 
245   bool dumpCode() const {
246     return DumpCode;
247   }
248 
249   /// Return the amount of LDS that can be used that will not restrict the
250   /// occupancy lower than WaveCount.
251   unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount) const;
252 
253   /// Inverse of getMaxLocalMemWithWaveCount. Return the maximum wavecount if
254   /// the given LDS memory size is the only constraint.
255   unsigned getOccupancyWithLocalMemSize(uint32_t Bytes) const;
256 
257 
258   bool hasFP32Denormals() const {
259     return FP32Denormals;
260   }
261 
262   bool hasFP64Denormals() const {
263     return FP64Denormals;
264   }
265 
266   bool hasFPExceptions() const {
267     return FPExceptions;
268   }
269 
270   bool useFlatForGlobal() const {
271     return FlatForGlobal;
272   }
273 
274   bool hasUnalignedBufferAccess() const {
275     return UnalignedBufferAccess;
276   }
277 
278   bool isXNACKEnabled() const {
279     return EnableXNACK;
280   }
281 
282   bool isAmdCodeObjectV2() const {
283     return isAmdHsaOS() || isMesa3DOS();
284   }
285 
286   /// \brief Returns the offset in bytes from the start of the input buffer
287   ///        of the first explicit kernel argument.
288   unsigned getExplicitKernelArgOffset() const {
289     return isAmdCodeObjectV2() ? 0 : 36;
290   }
291 
292   unsigned getAlignmentForImplicitArgPtr() const {
293     return isAmdHsaOS() ? 8 : 4;
294   }
295 
296   unsigned getImplicitArgNumBytes() const {
297     if (isMesa3DOS())
298       return 16;
299     if (isAmdHsaOS() && isOpenCLEnv())
300       return 32;
301     return 0;
302   }
303 
304   unsigned getStackAlignment() const {
305     // Scratch is allocated in 256 dword per wave blocks.
306     return 4 * 256 / getWavefrontSize();
307   }
308 
309   bool enableMachineScheduler() const override {
310     return true;
311   }
312 
313   bool enableSubRegLiveness() const override {
314     return true;
315   }
316 
317   /// \returns Number of execution units per compute unit supported by the
318   /// subtarget.
319   unsigned getEUsPerCU() const {
320     return 4;
321   }
322 
323   /// \returns Maximum number of work groups per compute unit supported by the
324   /// subtarget and limited by given flat work group size.
325   unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const {
326     if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS)
327       return 8;
328     return getWavesPerWorkGroup(FlatWorkGroupSize) == 1 ? 40 : 16;
329   }
330 
331   /// \returns Maximum number of waves per compute unit supported by the
332   /// subtarget without any kind of limitation.
333   unsigned getMaxWavesPerCU() const {
334     return getMaxWavesPerEU() * getEUsPerCU();
335   }
336 
337   /// \returns Maximum number of waves per compute unit supported by the
338   /// subtarget and limited by given flat work group size.
339   unsigned getMaxWavesPerCU(unsigned FlatWorkGroupSize) const {
340     return getWavesPerWorkGroup(FlatWorkGroupSize);
341   }
342 
343   /// \returns Minimum number of waves per execution unit supported by the
344   /// subtarget.
345   unsigned getMinWavesPerEU() const {
346     return 1;
347   }
348 
349   /// \returns Maximum number of waves per execution unit supported by the
350   /// subtarget without any kind of limitation.
351   unsigned getMaxWavesPerEU() const {
352     if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS)
353       return 8;
354     // FIXME: Need to take scratch memory into account.
355     return 10;
356   }
357 
358   /// \returns Maximum number of waves per execution unit supported by the
359   /// subtarget and limited by given flat work group size.
360   unsigned getMaxWavesPerEU(unsigned FlatWorkGroupSize) const {
361     return alignTo(getMaxWavesPerCU(FlatWorkGroupSize), getEUsPerCU()) /
362       getEUsPerCU();
363   }
364 
365   /// \returns Minimum flat work group size supported by the subtarget.
366   unsigned getMinFlatWorkGroupSize() const {
367     return 1;
368   }
369 
370   /// \returns Maximum flat work group size supported by the subtarget.
371   unsigned getMaxFlatWorkGroupSize() const {
372     return 2048;
373   }
374 
375   /// \returns Number of waves per work group given the flat work group size.
376   unsigned getWavesPerWorkGroup(unsigned FlatWorkGroupSize) const {
377     return alignTo(FlatWorkGroupSize, getWavefrontSize()) / getWavefrontSize();
378   }
379 
380   /// \returns Subtarget's default pair of minimum/maximum flat work group sizes
381   /// for function \p F, or minimum/maximum flat work group sizes explicitly
382   /// requested using "amdgpu-flat-work-group-size" attribute attached to
383   /// function \p F.
384   ///
385   /// \returns Subtarget's default values if explicitly requested values cannot
386   /// be converted to integer, or violate subtarget's specifications.
387   std::pair<unsigned, unsigned> getFlatWorkGroupSizes(const Function &F) const;
388 
389   /// \returns Subtarget's default pair of minimum/maximum number of waves per
390   /// execution unit for function \p F, or minimum/maximum number of waves per
391   /// execution unit explicitly requested using "amdgpu-waves-per-eu" attribute
392   /// attached to function \p F.
393   ///
394   /// \returns Subtarget's default values if explicitly requested values cannot
395   /// be converted to integer, violate subtarget's specifications, or are not
396   /// compatible with minimum/maximum number of waves limited by flat work group
397   /// size, register usage, and/or lds usage.
398   std::pair<unsigned, unsigned> getWavesPerEU(const Function &F) const;
399 };
400 
401 class R600Subtarget final : public AMDGPUSubtarget {
402 private:
403   R600InstrInfo InstrInfo;
404   R600FrameLowering FrameLowering;
405   R600TargetLowering TLInfo;
406 
407 public:
408   R600Subtarget(const Triple &TT, StringRef CPU, StringRef FS,
409                 const TargetMachine &TM);
410 
411   const R600InstrInfo *getInstrInfo() const override {
412     return &InstrInfo;
413   }
414 
415   const R600FrameLowering *getFrameLowering() const override {
416     return &FrameLowering;
417   }
418 
419   const R600TargetLowering *getTargetLowering() const override {
420     return &TLInfo;
421   }
422 
423   const R600RegisterInfo *getRegisterInfo() const override {
424     return &InstrInfo.getRegisterInfo();
425   }
426 
427   bool hasCFAluBug() const {
428     return CFALUBug;
429   }
430 
431   bool hasVertexCache() const {
432     return HasVertexCache;
433   }
434 
435   short getTexVTXClauseSize() const {
436     return TexVTXClauseSize;
437   }
438 };
439 
440 class SISubtarget final : public AMDGPUSubtarget {
441 public:
442   enum {
443     // The closed Vulkan driver sets 96, which limits the wave count to 8 but
444     // doesn't spill SGPRs as much as when 80 is set.
445     FIXED_SGPR_COUNT_FOR_INIT_BUG = 96
446   };
447 
448 private:
449   SIInstrInfo InstrInfo;
450   SIFrameLowering FrameLowering;
451   SITargetLowering TLInfo;
452   std::unique_ptr<GISelAccessor> GISel;
453 
454 public:
455   SISubtarget(const Triple &TT, StringRef CPU, StringRef FS,
456               const TargetMachine &TM);
457 
458   const SIInstrInfo *getInstrInfo() const override {
459     return &InstrInfo;
460   }
461 
462   const SIFrameLowering *getFrameLowering() const override {
463     return &FrameLowering;
464   }
465 
466   const SITargetLowering *getTargetLowering() const override {
467     return &TLInfo;
468   }
469 
470   const CallLowering *getCallLowering() const override {
471     assert(GISel && "Access to GlobalISel APIs not set");
472     return GISel->getCallLowering();
473   }
474 
475   const SIRegisterInfo *getRegisterInfo() const override {
476     return &InstrInfo.getRegisterInfo();
477   }
478 
479   void setGISelAccessor(GISelAccessor &GISel) {
480     this->GISel.reset(&GISel);
481   }
482 
483   void overrideSchedPolicy(MachineSchedPolicy &Policy,
484                            unsigned NumRegionInstrs) const override;
485 
486   bool isVGPRSpillingEnabled(const Function& F) const;
487 
488   unsigned getMaxNumUserSGPRs() const {
489     return 16;
490   }
491 
492   bool hasFlatAddressSpace() const {
493     return FlatAddressSpace;
494   }
495 
496   bool hasSMemRealTime() const {
497     return HasSMemRealTime;
498   }
499 
500   bool has16BitInsts() const {
501     return Has16BitInsts;
502   }
503 
504   bool hasScalarCompareEq64() const {
505     return getGeneration() >= VOLCANIC_ISLANDS;
506   }
507 
508   bool enableSIScheduler() const {
509     return EnableSIScheduler;
510   }
511 
512   bool debuggerSupported() const {
513     return debuggerInsertNops() && debuggerReserveRegs() &&
514       debuggerEmitPrologue();
515   }
516 
517   bool debuggerInsertNops() const {
518     return DebuggerInsertNops;
519   }
520 
521   bool debuggerReserveRegs() const {
522     return DebuggerReserveRegs;
523   }
524 
525   bool debuggerEmitPrologue() const {
526     return DebuggerEmitPrologue;
527   }
528 
529   bool loadStoreOptEnabled() const {
530     return EnableLoadStoreOpt;
531   }
532 
533   bool hasSGPRInitBug() const {
534     return SGPRInitBug;
535   }
536 
537   unsigned getKernArgSegmentSize(unsigned ExplictArgBytes) const;
538 
539   /// Return the maximum number of waves per SIMD for kernels using \p SGPRs SGPRs
540   unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const;
541 
542   /// Return the maximum number of waves per SIMD for kernels using \p VGPRs VGPRs
543   unsigned getOccupancyWithNumVGPRs(unsigned VGPRs) const;
544 
545   /// \returns True if waitcnt instruction is needed before barrier instruction,
546   /// false otherwise.
547   bool needWaitcntBeforeBarrier() const {
548     return true;
549   }
550 };
551 
552 } // End namespace llvm
553 
554 #endif
555