1 //=====-- AMDGPUSubtarget.h - Define Subtarget for AMDGPU ------*- C++ -*-====//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //==-----------------------------------------------------------------------===//
9 //
10 /// \file
11 /// \brief AMDGPU specific subclass of TargetSubtarget.
12 //
13 //===----------------------------------------------------------------------===//
14 
15 #ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H
16 #define LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H
17 
18 #include "AMDGPU.h"
19 #include "R600InstrInfo.h"
20 #include "R600ISelLowering.h"
21 #include "R600FrameLowering.h"
22 #include "SIInstrInfo.h"
23 #include "SIISelLowering.h"
24 #include "SIFrameLowering.h"
25 #include "Utils/AMDGPUBaseInfo.h"
26 #include "llvm/CodeGen/GlobalISel/GISelAccessor.h"
27 #include "llvm/CodeGen/SelectionDAGTargetInfo.h"
28 #include "llvm/Target/TargetSubtargetInfo.h"
29 
30 #define GET_SUBTARGETINFO_HEADER
31 #include "AMDGPUGenSubtargetInfo.inc"
32 
33 namespace llvm {
34 
35 class SIMachineFunctionInfo;
36 class StringRef;
37 
38 class AMDGPUSubtarget : public AMDGPUGenSubtargetInfo {
39 public:
40   enum Generation {
41     R600 = 0,
42     R700,
43     EVERGREEN,
44     NORTHERN_ISLANDS,
45     SOUTHERN_ISLANDS,
46     SEA_ISLANDS,
47     VOLCANIC_ISLANDS,
48   };
49 
50   enum {
51     ISAVersion0_0_0,
52     ISAVersion7_0_0,
53     ISAVersion7_0_1,
54     ISAVersion8_0_0,
55     ISAVersion8_0_1,
56     ISAVersion8_0_2,
57     ISAVersion8_0_3
58   };
59 
60 protected:
61   // Basic subtarget description.
62   Triple TargetTriple;
63   Generation Gen;
64   unsigned IsaVersion;
65   unsigned WavefrontSize;
66   int LocalMemorySize;
67   int LDSBankCount;
68   unsigned MaxPrivateElementSize;
69 
70   // Possibly statically set by tablegen, but may want to be overridden.
71   bool FastFMAF32;
72   bool HalfRate64Ops;
73 
74   // Dynamially set bits that enable features.
75   bool FP32Denormals;
76   bool FP64Denormals;
77   bool FPExceptions;
78   bool FlatForGlobal;
79   bool UnalignedScratchAccess;
80   bool UnalignedBufferAccess;
81   bool EnableXNACK;
82   bool DebuggerInsertNops;
83   bool DebuggerReserveRegs;
84   bool DebuggerEmitPrologue;
85 
86   // Used as options.
87   bool EnableVGPRSpilling;
88   bool EnablePromoteAlloca;
89   bool EnableLoadStoreOpt;
90   bool EnableUnsafeDSOffsetFolding;
91   bool EnableSIScheduler;
92   bool DumpCode;
93 
94   // Subtarget statically properties set by tablegen
95   bool FP64;
96   bool IsGCN;
97   bool GCN1Encoding;
98   bool GCN3Encoding;
99   bool CIInsts;
100   bool SGPRInitBug;
101   bool HasSMemRealTime;
102   bool Has16BitInsts;
103   bool HasMovrel;
104   bool HasVGPRIndexMode;
105   bool FlatAddressSpace;
106   bool R600ALUInst;
107   bool CaymanISA;
108   bool CFALUBug;
109   bool HasVertexCache;
110   short TexVTXClauseSize;
111 
112   // Dummy feature to use for assembler in tablegen.
113   bool FeatureDisable;
114 
115   InstrItineraryData InstrItins;
116   SelectionDAGTargetInfo TSInfo;
117 
118 public:
119   AMDGPUSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
120                   const TargetMachine &TM);
121   virtual ~AMDGPUSubtarget();
122   AMDGPUSubtarget &initializeSubtargetDependencies(const Triple &TT,
123                                                    StringRef GPU, StringRef FS);
124 
125   const AMDGPUInstrInfo *getInstrInfo() const override = 0;
126   const AMDGPUFrameLowering *getFrameLowering() const override = 0;
127   const AMDGPUTargetLowering *getTargetLowering() const override = 0;
128   const AMDGPURegisterInfo *getRegisterInfo() const override = 0;
129 
130   const InstrItineraryData *getInstrItineraryData() const override {
131     return &InstrItins;
132   }
133 
134   // Nothing implemented, just prevent crashes on use.
135   const SelectionDAGTargetInfo *getSelectionDAGInfo() const override {
136     return &TSInfo;
137   }
138 
139   void ParseSubtargetFeatures(StringRef CPU, StringRef FS);
140 
141   bool isAmdHsaOS() const {
142     return TargetTriple.getOS() == Triple::AMDHSA;
143   }
144 
145   bool isMesa3DOS() const {
146     return TargetTriple.getOS() == Triple::Mesa3D;
147   }
148 
149   bool isOpenCLEnv() const {
150     return TargetTriple.getEnvironment() == Triple::OpenCL;
151   }
152 
153   Generation getGeneration() const {
154     return Gen;
155   }
156 
157   unsigned getWavefrontSize() const {
158     return WavefrontSize;
159   }
160 
161   int getLocalMemorySize() const {
162     return LocalMemorySize;
163   }
164 
165   int getLDSBankCount() const {
166     return LDSBankCount;
167   }
168 
169   unsigned getMaxPrivateElementSize() const {
170     return MaxPrivateElementSize;
171   }
172 
173   bool hasHWFP64() const {
174     return FP64;
175   }
176 
177   bool hasFastFMAF32() const {
178     return FastFMAF32;
179   }
180 
181   bool hasHalfRate64Ops() const {
182     return HalfRate64Ops;
183   }
184 
185   bool hasAddr64() const {
186     return (getGeneration() < VOLCANIC_ISLANDS);
187   }
188 
189   bool hasBFE() const {
190     return (getGeneration() >= EVERGREEN);
191   }
192 
193   bool hasBFI() const {
194     return (getGeneration() >= EVERGREEN);
195   }
196 
197   bool hasBFM() const {
198     return hasBFE();
199   }
200 
201   bool hasBCNT(unsigned Size) const {
202     if (Size == 32)
203       return (getGeneration() >= EVERGREEN);
204 
205     if (Size == 64)
206       return (getGeneration() >= SOUTHERN_ISLANDS);
207 
208     return false;
209   }
210 
211   bool hasMulU24() const {
212     return (getGeneration() >= EVERGREEN);
213   }
214 
215   bool hasMulI24() const {
216     return (getGeneration() >= SOUTHERN_ISLANDS ||
217             hasCaymanISA());
218   }
219 
220   bool hasFFBL() const {
221     return (getGeneration() >= EVERGREEN);
222   }
223 
224   bool hasFFBH() const {
225     return (getGeneration() >= EVERGREEN);
226   }
227 
228   bool hasCARRY() const {
229     return (getGeneration() >= EVERGREEN);
230   }
231 
232   bool hasBORROW() const {
233     return (getGeneration() >= EVERGREEN);
234   }
235 
236   bool hasCaymanISA() const {
237     return CaymanISA;
238   }
239 
240   bool isPromoteAllocaEnabled() const {
241     return EnablePromoteAlloca;
242   }
243 
244   bool unsafeDSOffsetFoldingEnabled() const {
245     return EnableUnsafeDSOffsetFolding;
246   }
247 
248   bool dumpCode() const {
249     return DumpCode;
250   }
251 
252   /// Return the amount of LDS that can be used that will not restrict the
253   /// occupancy lower than WaveCount.
254   unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount) const;
255 
256   /// Inverse of getMaxLocalMemWithWaveCount. Return the maximum wavecount if
257   /// the given LDS memory size is the only constraint.
258   unsigned getOccupancyWithLocalMemSize(uint32_t Bytes) const;
259 
260 
261   bool hasFP32Denormals() const {
262     return FP32Denormals;
263   }
264 
265   bool hasFP64Denormals() const {
266     return FP64Denormals;
267   }
268 
269   bool hasFPExceptions() const {
270     return FPExceptions;
271   }
272 
273   bool useFlatForGlobal() const {
274     return FlatForGlobal;
275   }
276 
277   bool hasUnalignedBufferAccess() const {
278     return UnalignedBufferAccess;
279   }
280 
281   bool hasUnalignedScratchAccess() const {
282     return UnalignedScratchAccess;
283   }
284 
285   bool isXNACKEnabled() const {
286     return EnableXNACK;
287   }
288 
289   bool isAmdCodeObjectV2() const {
290     return isAmdHsaOS() || isMesa3DOS();
291   }
292 
293   /// \brief Returns the offset in bytes from the start of the input buffer
294   ///        of the first explicit kernel argument.
295   unsigned getExplicitKernelArgOffset() const {
296     return isAmdCodeObjectV2() ? 0 : 36;
297   }
298 
299   unsigned getAlignmentForImplicitArgPtr() const {
300     return isAmdHsaOS() ? 8 : 4;
301   }
302 
303   unsigned getImplicitArgNumBytes() const {
304     if (isMesa3DOS())
305       return 16;
306     if (isAmdHsaOS() && isOpenCLEnv())
307       return 32;
308     return 0;
309   }
310 
311   unsigned getStackAlignment() const {
312     // Scratch is allocated in 256 dword per wave blocks.
313     return 4 * 256 / getWavefrontSize();
314   }
315 
316   bool enableMachineScheduler() const override {
317     return true;
318   }
319 
320   bool enableSubRegLiveness() const override {
321     return true;
322   }
323 
324   /// \returns Number of execution units per compute unit supported by the
325   /// subtarget.
326   unsigned getEUsPerCU() const {
327     return 4;
328   }
329 
330   /// \returns Maximum number of work groups per compute unit supported by the
331   /// subtarget and limited by given flat work group size.
332   unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const {
333     if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS)
334       return 8;
335     return getWavesPerWorkGroup(FlatWorkGroupSize) == 1 ? 40 : 16;
336   }
337 
338   /// \returns Maximum number of waves per compute unit supported by the
339   /// subtarget without any kind of limitation.
340   unsigned getMaxWavesPerCU() const {
341     return getMaxWavesPerEU() * getEUsPerCU();
342   }
343 
344   /// \returns Maximum number of waves per compute unit supported by the
345   /// subtarget and limited by given flat work group size.
346   unsigned getMaxWavesPerCU(unsigned FlatWorkGroupSize) const {
347     return getWavesPerWorkGroup(FlatWorkGroupSize);
348   }
349 
350   /// \returns Minimum number of waves per execution unit supported by the
351   /// subtarget.
352   unsigned getMinWavesPerEU() const {
353     return 1;
354   }
355 
356   /// \returns Maximum number of waves per execution unit supported by the
357   /// subtarget without any kind of limitation.
358   unsigned getMaxWavesPerEU() const {
359     if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS)
360       return 8;
361     // FIXME: Need to take scratch memory into account.
362     return 10;
363   }
364 
365   /// \returns Maximum number of waves per execution unit supported by the
366   /// subtarget and limited by given flat work group size.
367   unsigned getMaxWavesPerEU(unsigned FlatWorkGroupSize) const {
368     return alignTo(getMaxWavesPerCU(FlatWorkGroupSize), getEUsPerCU()) /
369       getEUsPerCU();
370   }
371 
372   /// \returns Minimum flat work group size supported by the subtarget.
373   unsigned getMinFlatWorkGroupSize() const {
374     return 1;
375   }
376 
377   /// \returns Maximum flat work group size supported by the subtarget.
378   unsigned getMaxFlatWorkGroupSize() const {
379     return 2048;
380   }
381 
382   /// \returns Number of waves per work group given the flat work group size.
383   unsigned getWavesPerWorkGroup(unsigned FlatWorkGroupSize) const {
384     return alignTo(FlatWorkGroupSize, getWavefrontSize()) / getWavefrontSize();
385   }
386 
387   /// \returns Subtarget's default pair of minimum/maximum flat work group sizes
388   /// for function \p F, or minimum/maximum flat work group sizes explicitly
389   /// requested using "amdgpu-flat-work-group-size" attribute attached to
390   /// function \p F.
391   ///
392   /// \returns Subtarget's default values if explicitly requested values cannot
393   /// be converted to integer, or violate subtarget's specifications.
394   std::pair<unsigned, unsigned> getFlatWorkGroupSizes(const Function &F) const;
395 
396   /// \returns Subtarget's default pair of minimum/maximum number of waves per
397   /// execution unit for function \p F, or minimum/maximum number of waves per
398   /// execution unit explicitly requested using "amdgpu-waves-per-eu" attribute
399   /// attached to function \p F.
400   ///
401   /// \returns Subtarget's default values if explicitly requested values cannot
402   /// be converted to integer, violate subtarget's specifications, or are not
403   /// compatible with minimum/maximum number of waves limited by flat work group
404   /// size, register usage, and/or lds usage.
405   std::pair<unsigned, unsigned> getWavesPerEU(const Function &F) const;
406 };
407 
408 class R600Subtarget final : public AMDGPUSubtarget {
409 private:
410   R600InstrInfo InstrInfo;
411   R600FrameLowering FrameLowering;
412   R600TargetLowering TLInfo;
413 
414 public:
415   R600Subtarget(const Triple &TT, StringRef CPU, StringRef FS,
416                 const TargetMachine &TM);
417 
418   const R600InstrInfo *getInstrInfo() const override {
419     return &InstrInfo;
420   }
421 
422   const R600FrameLowering *getFrameLowering() const override {
423     return &FrameLowering;
424   }
425 
426   const R600TargetLowering *getTargetLowering() const override {
427     return &TLInfo;
428   }
429 
430   const R600RegisterInfo *getRegisterInfo() const override {
431     return &InstrInfo.getRegisterInfo();
432   }
433 
434   bool hasCFAluBug() const {
435     return CFALUBug;
436   }
437 
438   bool hasVertexCache() const {
439     return HasVertexCache;
440   }
441 
442   short getTexVTXClauseSize() const {
443     return TexVTXClauseSize;
444   }
445 };
446 
447 class SISubtarget final : public AMDGPUSubtarget {
448 public:
449   enum {
450     // The closed Vulkan driver sets 96, which limits the wave count to 8 but
451     // doesn't spill SGPRs as much as when 80 is set.
452     FIXED_SGPR_COUNT_FOR_INIT_BUG = 96
453   };
454 
455 private:
456   SIInstrInfo InstrInfo;
457   SIFrameLowering FrameLowering;
458   SITargetLowering TLInfo;
459   std::unique_ptr<GISelAccessor> GISel;
460 
461 public:
462   SISubtarget(const Triple &TT, StringRef CPU, StringRef FS,
463               const TargetMachine &TM);
464 
465   const SIInstrInfo *getInstrInfo() const override {
466     return &InstrInfo;
467   }
468 
469   const SIFrameLowering *getFrameLowering() const override {
470     return &FrameLowering;
471   }
472 
473   const SITargetLowering *getTargetLowering() const override {
474     return &TLInfo;
475   }
476 
477   const CallLowering *getCallLowering() const override {
478     assert(GISel && "Access to GlobalISel APIs not set");
479     return GISel->getCallLowering();
480   }
481 
482   const SIRegisterInfo *getRegisterInfo() const override {
483     return &InstrInfo.getRegisterInfo();
484   }
485 
486   void setGISelAccessor(GISelAccessor &GISel) {
487     this->GISel.reset(&GISel);
488   }
489 
490   void overrideSchedPolicy(MachineSchedPolicy &Policy,
491                            unsigned NumRegionInstrs) const override;
492 
493   bool isVGPRSpillingEnabled(const Function& F) const;
494 
495   unsigned getMaxNumUserSGPRs() const {
496     return 16;
497   }
498 
499   bool hasFlatAddressSpace() const {
500     return FlatAddressSpace;
501   }
502 
503   bool hasSMemRealTime() const {
504     return HasSMemRealTime;
505   }
506 
507   bool has16BitInsts() const {
508     return Has16BitInsts;
509   }
510 
511   bool hasMovrel() const {
512     return HasMovrel;
513   }
514 
515   bool hasVGPRIndexMode() const {
516     return HasVGPRIndexMode;
517   }
518 
519   bool hasScalarCompareEq64() const {
520     return getGeneration() >= VOLCANIC_ISLANDS;
521   }
522 
523   bool enableSIScheduler() const {
524     return EnableSIScheduler;
525   }
526 
527   bool debuggerSupported() const {
528     return debuggerInsertNops() && debuggerReserveRegs() &&
529       debuggerEmitPrologue();
530   }
531 
532   bool debuggerInsertNops() const {
533     return DebuggerInsertNops;
534   }
535 
536   bool debuggerReserveRegs() const {
537     return DebuggerReserveRegs;
538   }
539 
540   bool debuggerEmitPrologue() const {
541     return DebuggerEmitPrologue;
542   }
543 
544   bool loadStoreOptEnabled() const {
545     return EnableLoadStoreOpt;
546   }
547 
548   bool hasSGPRInitBug() const {
549     return SGPRInitBug;
550   }
551 
552   unsigned getKernArgSegmentSize(unsigned ExplictArgBytes) const;
553 
554   /// Return the maximum number of waves per SIMD for kernels using \p SGPRs SGPRs
555   unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const;
556 
557   /// Return the maximum number of waves per SIMD for kernels using \p VGPRs VGPRs
558   unsigned getOccupancyWithNumVGPRs(unsigned VGPRs) const;
559 
560   /// \returns True if waitcnt instruction is needed before barrier instruction,
561   /// false otherwise.
562   bool needWaitcntBeforeBarrier() const {
563     return true;
564   }
565 };
566 
567 } // End namespace llvm
568 
569 #endif
570