1 //===-- AMDGPUSubtarget.cpp - AMDGPU Subtarget Information ----------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 /// \file
11 /// \brief Implements the AMDGPU specific subclass of TargetSubtarget.
12 //
13 //===----------------------------------------------------------------------===//
14 
15 #include "AMDGPUSubtarget.h"
16 #include "SIMachineFunctionInfo.h"
17 #include "llvm/ADT/SmallString.h"
18 #include "llvm/CodeGen/MachineScheduler.h"
19 #include "llvm/Target/TargetFrameLowering.h"
20 #include <algorithm>
21 
22 using namespace llvm;
23 
24 #define DEBUG_TYPE "amdgpu-subtarget"
25 
26 #define GET_SUBTARGETINFO_ENUM
27 #define GET_SUBTARGETINFO_TARGET_DESC
28 #define GET_SUBTARGETINFO_CTOR
29 #include "AMDGPUGenSubtargetInfo.inc"
30 
31 AMDGPUSubtarget::~AMDGPUSubtarget() = default;
32 
33 AMDGPUSubtarget &
34 AMDGPUSubtarget::initializeSubtargetDependencies(const Triple &TT,
35                                                  StringRef GPU, StringRef FS) {
36   // Determine default and user-specified characteristics
37   // On SI+, we want FP64 denormals to be on by default. FP32 denormals can be
38   // enabled, but some instructions do not respect them and they run at the
39   // double precision rate, so don't enable by default.
40   //
41   // We want to be able to turn these off, but making this a subtarget feature
42   // for SI has the unhelpful behavior that it unsets everything else if you
43   // disable it.
44 
45   SmallString<256> FullFS("+promote-alloca,+fp64-fp16-denormals,+dx10-clamp,+load-store-opt,");
46   if (isAmdHsaOS()) // Turn on FlatForGlobal for HSA.
47     FullFS += "+flat-for-global,+unaligned-buffer-access,+trap-handler,";
48 
49   FullFS += FS;
50 
51   ParseSubtargetFeatures(GPU, FullFS);
52 
53   // Unless +-flat-for-global is specified, turn on FlatForGlobal for all OS-es
54   // on VI and newer hardware to avoid assertion failures due to missing ADDR64
55   // variants of MUBUF instructions.
56   if (!hasAddr64() && !FS.contains("flat-for-global")) {
57     FlatForGlobal = true;
58   }
59 
60   // FIXME: I don't think think Evergreen has any useful support for
61   // denormals, but should be checked. Should we issue a warning somewhere
62   // if someone tries to enable these?
63   if (getGeneration() <= AMDGPUSubtarget::NORTHERN_ISLANDS) {
64     FP64FP16Denormals = false;
65     FP32Denormals = false;
66   }
67 
68   // Set defaults if needed.
69   if (MaxPrivateElementSize == 0)
70     MaxPrivateElementSize = 4;
71 
72   return *this;
73 }
74 
75 AMDGPUSubtarget::AMDGPUSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
76                                  const TargetMachine &TM)
77   : AMDGPUGenSubtargetInfo(TT, GPU, FS),
78     TargetTriple(TT),
79     Gen(TT.getArch() == Triple::amdgcn ? SOUTHERN_ISLANDS : R600),
80     IsaVersion(ISAVersion0_0_0),
81     WavefrontSize(64),
82     LocalMemorySize(0),
83     LDSBankCount(0),
84     MaxPrivateElementSize(0),
85 
86     FastFMAF32(false),
87     HalfRate64Ops(false),
88 
89     FP32Denormals(false),
90     FP64FP16Denormals(false),
91     FPExceptions(false),
92     DX10Clamp(false),
93     FlatForGlobal(false),
94     UnalignedScratchAccess(false),
95     UnalignedBufferAccess(false),
96 
97     HasApertureRegs(false),
98     EnableXNACK(false),
99     TrapHandler(false),
100     DebuggerInsertNops(false),
101     DebuggerReserveRegs(false),
102     DebuggerEmitPrologue(false),
103 
104     EnableVGPRSpilling(false),
105     EnablePromoteAlloca(false),
106     EnableLoadStoreOpt(false),
107     EnableUnsafeDSOffsetFolding(false),
108     EnableSIScheduler(false),
109     DumpCode(false),
110 
111     FP64(false),
112     IsGCN(false),
113     GCN1Encoding(false),
114     GCN3Encoding(false),
115     CIInsts(false),
116     GFX9Insts(false),
117     SGPRInitBug(false),
118     HasSMemRealTime(false),
119     Has16BitInsts(false),
120     HasMovrel(false),
121     HasVGPRIndexMode(false),
122     HasScalarStores(false),
123     HasInv2PiInlineImm(false),
124     HasSDWA(false),
125     HasDPP(false),
126     FlatAddressSpace(false),
127 
128     R600ALUInst(false),
129     CaymanISA(false),
130     CFALUBug(false),
131     HasVertexCache(false),
132     TexVTXClauseSize(0),
133     ScalarizeGlobal(false),
134 
135     FeatureDisable(false),
136     InstrItins(getInstrItineraryForCPU(GPU)) {
137   initializeSubtargetDependencies(TT, GPU, FS);
138 }
139 
140 unsigned AMDGPUSubtarget::getMaxLocalMemSizeWithWaveCount(unsigned NWaves,
141   const Function &F) const {
142   if (NWaves == 1)
143     return getLocalMemorySize();
144   unsigned WorkGroupSize = getFlatWorkGroupSizes(F).second;
145   unsigned WorkGroupsPerCu = getMaxWorkGroupsPerCU(WorkGroupSize);
146   unsigned MaxWaves = getMaxWavesPerEU();
147   return getLocalMemorySize() * MaxWaves / WorkGroupsPerCu / NWaves;
148 }
149 
150 unsigned AMDGPUSubtarget::getOccupancyWithLocalMemSize(uint32_t Bytes,
151   const Function &F) const {
152   unsigned WorkGroupSize = getFlatWorkGroupSizes(F).second;
153   unsigned WorkGroupsPerCu = getMaxWorkGroupsPerCU(WorkGroupSize);
154   unsigned MaxWaves = getMaxWavesPerEU();
155   unsigned Limit = getLocalMemorySize() * MaxWaves / WorkGroupsPerCu;
156   unsigned NumWaves = Limit / (Bytes ? Bytes : 1u);
157   NumWaves = std::min(NumWaves, MaxWaves);
158   NumWaves = std::max(NumWaves, 1u);
159   return NumWaves;
160 }
161 
162 std::pair<unsigned, unsigned> AMDGPUSubtarget::getFlatWorkGroupSizes(
163   const Function &F) const {
164   // Default minimum/maximum flat work group sizes.
165   std::pair<unsigned, unsigned> Default =
166     AMDGPU::isCompute(F.getCallingConv()) ?
167       std::pair<unsigned, unsigned>(getWavefrontSize() * 2,
168                                     getWavefrontSize() * 4) :
169       std::pair<unsigned, unsigned>(1, getWavefrontSize());
170 
171   // TODO: Do not process "amdgpu-max-work-group-size" attribute once mesa
172   // starts using "amdgpu-flat-work-group-size" attribute.
173   Default.second = AMDGPU::getIntegerAttribute(
174     F, "amdgpu-max-work-group-size", Default.second);
175   Default.first = std::min(Default.first, Default.second);
176 
177   // Requested minimum/maximum flat work group sizes.
178   std::pair<unsigned, unsigned> Requested = AMDGPU::getIntegerPairAttribute(
179     F, "amdgpu-flat-work-group-size", Default);
180 
181   // Make sure requested minimum is less than requested maximum.
182   if (Requested.first > Requested.second)
183     return Default;
184 
185   // Make sure requested values do not violate subtarget's specifications.
186   if (Requested.first < getMinFlatWorkGroupSize())
187     return Default;
188   if (Requested.second > getMaxFlatWorkGroupSize())
189     return Default;
190 
191   return Requested;
192 }
193 
194 std::pair<unsigned, unsigned> AMDGPUSubtarget::getWavesPerEU(
195   const Function &F) const {
196   // Default minimum/maximum number of waves per execution unit.
197   std::pair<unsigned, unsigned> Default(1, getMaxWavesPerEU());
198 
199   // Default/requested minimum/maximum flat work group sizes.
200   std::pair<unsigned, unsigned> FlatWorkGroupSizes = getFlatWorkGroupSizes(F);
201 
202   // If minimum/maximum flat work group sizes were explicitly requested using
203   // "amdgpu-flat-work-group-size" attribute, then set default minimum/maximum
204   // number of waves per execution unit to values implied by requested
205   // minimum/maximum flat work group sizes.
206   unsigned MinImpliedByFlatWorkGroupSize =
207     getMaxWavesPerEU(FlatWorkGroupSizes.second);
208   bool RequestedFlatWorkGroupSize = false;
209 
210   // TODO: Do not process "amdgpu-max-work-group-size" attribute once mesa
211   // starts using "amdgpu-flat-work-group-size" attribute.
212   if (F.hasFnAttribute("amdgpu-max-work-group-size") ||
213       F.hasFnAttribute("amdgpu-flat-work-group-size")) {
214     Default.first = MinImpliedByFlatWorkGroupSize;
215     RequestedFlatWorkGroupSize = true;
216   }
217 
218   // Requested minimum/maximum number of waves per execution unit.
219   std::pair<unsigned, unsigned> Requested = AMDGPU::getIntegerPairAttribute(
220     F, "amdgpu-waves-per-eu", Default, true);
221 
222   // Make sure requested minimum is less than requested maximum.
223   if (Requested.second && Requested.first > Requested.second)
224     return Default;
225 
226   // Make sure requested values do not violate subtarget's specifications.
227   if (Requested.first < getMinWavesPerEU() ||
228       Requested.first > getMaxWavesPerEU())
229     return Default;
230   if (Requested.second > getMaxWavesPerEU())
231     return Default;
232 
233   // Make sure requested values are compatible with values implied by requested
234   // minimum/maximum flat work group sizes.
235   if (RequestedFlatWorkGroupSize &&
236       Requested.first > MinImpliedByFlatWorkGroupSize)
237     return Default;
238 
239   return Requested;
240 }
241 
242 R600Subtarget::R600Subtarget(const Triple &TT, StringRef GPU, StringRef FS,
243                              const TargetMachine &TM) :
244   AMDGPUSubtarget(TT, GPU, FS, TM),
245   InstrInfo(*this),
246   FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0),
247   TLInfo(TM, *this) {}
248 
249 SISubtarget::SISubtarget(const Triple &TT, StringRef GPU, StringRef FS,
250                          const TargetMachine &TM) :
251   AMDGPUSubtarget(TT, GPU, FS, TM),
252   InstrInfo(*this),
253   FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0),
254   TLInfo(TM, *this) {}
255 
256 void SISubtarget::overrideSchedPolicy(MachineSchedPolicy &Policy,
257                                       unsigned NumRegionInstrs) const {
258   // Track register pressure so the scheduler can try to decrease
259   // pressure once register usage is above the threshold defined by
260   // SIRegisterInfo::getRegPressureSetLimit()
261   Policy.ShouldTrackPressure = true;
262 
263   // Enabling both top down and bottom up scheduling seems to give us less
264   // register spills than just using one of these approaches on its own.
265   Policy.OnlyTopDown = false;
266   Policy.OnlyBottomUp = false;
267 
268   // Enabling ShouldTrackLaneMasks crashes the SI Machine Scheduler.
269   if (!enableSIScheduler())
270     Policy.ShouldTrackLaneMasks = true;
271 }
272 
273 bool SISubtarget::isVGPRSpillingEnabled(const Function& F) const {
274   return EnableVGPRSpilling || !AMDGPU::isShader(F.getCallingConv());
275 }
276 
277 unsigned SISubtarget::getKernArgSegmentSize(const MachineFunction &MF,
278                                             unsigned ExplicitArgBytes) const {
279   unsigned ImplicitBytes = getImplicitArgNumBytes(MF);
280   if (ImplicitBytes == 0)
281     return ExplicitArgBytes;
282 
283   unsigned Alignment = getAlignmentForImplicitArgPtr();
284   return alignTo(ExplicitArgBytes, Alignment) + ImplicitBytes;
285 }
286 
287 unsigned SISubtarget::getOccupancyWithNumSGPRs(unsigned SGPRs) const {
288   if (getGeneration() >= SISubtarget::VOLCANIC_ISLANDS) {
289     if (SGPRs <= 80)
290       return 10;
291     if (SGPRs <= 88)
292       return 9;
293     if (SGPRs <= 100)
294       return 8;
295     return 7;
296   }
297   if (SGPRs <= 48)
298     return 10;
299   if (SGPRs <= 56)
300     return 9;
301   if (SGPRs <= 64)
302     return 8;
303   if (SGPRs <= 72)
304     return 7;
305   if (SGPRs <= 80)
306     return 6;
307   return 5;
308 }
309 
310 unsigned SISubtarget::getOccupancyWithNumVGPRs(unsigned VGPRs) const {
311   if (VGPRs <= 24)
312     return 10;
313   if (VGPRs <= 28)
314     return 9;
315   if (VGPRs <= 32)
316     return 8;
317   if (VGPRs <= 36)
318     return 7;
319   if (VGPRs <= 40)
320     return 6;
321   if (VGPRs <= 48)
322     return 5;
323   if (VGPRs <= 64)
324     return 4;
325   if (VGPRs <= 84)
326     return 3;
327   if (VGPRs <= 128)
328     return 2;
329   return 1;
330 }
331 
332 unsigned SISubtarget::getReservedNumSGPRs(const MachineFunction &MF) const {
333   const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>();
334   if (MFI.hasFlatScratchInit()) {
335     if (getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
336       return 6; // FLAT_SCRATCH, XNACK, VCC (in that order).
337     if (getGeneration() == AMDGPUSubtarget::SEA_ISLANDS)
338       return 4; // FLAT_SCRATCH, VCC (in that order).
339   }
340 
341   if (isXNACKEnabled())
342     return 4; // XNACK, VCC (in that order).
343   return 2; // VCC.
344 }
345 
346 unsigned SISubtarget::getMaxNumSGPRs(const MachineFunction &MF) const {
347   const Function &F = *MF.getFunction();
348   const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>();
349 
350   // Compute maximum number of SGPRs function can use using default/requested
351   // minimum number of waves per execution unit.
352   std::pair<unsigned, unsigned> WavesPerEU = MFI.getWavesPerEU();
353   unsigned MaxNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, false);
354   unsigned MaxAddressableNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, true);
355 
356   // Check if maximum number of SGPRs was explicitly requested using
357   // "amdgpu-num-sgpr" attribute.
358   if (F.hasFnAttribute("amdgpu-num-sgpr")) {
359     unsigned Requested = AMDGPU::getIntegerAttribute(
360       F, "amdgpu-num-sgpr", MaxNumSGPRs);
361 
362     // Make sure requested value does not violate subtarget's specifications.
363     if (Requested && (Requested <= getReservedNumSGPRs(MF)))
364       Requested = 0;
365 
366     // If more SGPRs are required to support the input user/system SGPRs,
367     // increase to accommodate them.
368     //
369     // FIXME: This really ends up using the requested number of SGPRs + number
370     // of reserved special registers in total. Theoretically you could re-use
371     // the last input registers for these special registers, but this would
372     // require a lot of complexity to deal with the weird aliasing.
373     unsigned InputNumSGPRs = MFI.getNumPreloadedSGPRs();
374     if (Requested && Requested < InputNumSGPRs)
375       Requested = InputNumSGPRs;
376 
377     // Make sure requested value is compatible with values implied by
378     // default/requested minimum/maximum number of waves per execution unit.
379     if (Requested && Requested > getMaxNumSGPRs(WavesPerEU.first, false))
380       Requested = 0;
381     if (WavesPerEU.second &&
382         Requested && Requested < getMinNumSGPRs(WavesPerEU.second))
383       Requested = 0;
384 
385     if (Requested)
386       MaxNumSGPRs = Requested;
387   }
388 
389   if (hasSGPRInitBug())
390     MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG;
391 
392   return std::min(MaxNumSGPRs - getReservedNumSGPRs(MF),
393                   MaxAddressableNumSGPRs);
394 }
395 
396 unsigned SISubtarget::getMaxNumVGPRs(const MachineFunction &MF) const {
397   const Function &F = *MF.getFunction();
398   const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>();
399 
400   // Compute maximum number of VGPRs function can use using default/requested
401   // minimum number of waves per execution unit.
402   std::pair<unsigned, unsigned> WavesPerEU = MFI.getWavesPerEU();
403   unsigned MaxNumVGPRs = getMaxNumVGPRs(WavesPerEU.first);
404 
405   // Check if maximum number of VGPRs was explicitly requested using
406   // "amdgpu-num-vgpr" attribute.
407   if (F.hasFnAttribute("amdgpu-num-vgpr")) {
408     unsigned Requested = AMDGPU::getIntegerAttribute(
409       F, "amdgpu-num-vgpr", MaxNumVGPRs);
410 
411     // Make sure requested value does not violate subtarget's specifications.
412     if (Requested && Requested <= getReservedNumVGPRs(MF))
413       Requested = 0;
414 
415     // Make sure requested value is compatible with values implied by
416     // default/requested minimum/maximum number of waves per execution unit.
417     if (Requested && Requested > getMaxNumVGPRs(WavesPerEU.first))
418       Requested = 0;
419     if (WavesPerEU.second &&
420         Requested && Requested < getMinNumVGPRs(WavesPerEU.second))
421       Requested = 0;
422 
423     if (Requested)
424       MaxNumVGPRs = Requested;
425   }
426 
427   return MaxNumVGPRs - getReservedNumVGPRs(MF);
428 }
429