1 //===-- AMDGPUSubtarget.cpp - AMDGPU Subtarget Information ----------------===// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //===----------------------------------------------------------------------===// 9 // 10 /// \file 11 /// \brief Implements the AMDGPU specific subclass of TargetSubtarget. 12 // 13 //===----------------------------------------------------------------------===// 14 15 #include "AMDGPUSubtarget.h" 16 #include "SIMachineFunctionInfo.h" 17 #include "llvm/ADT/SmallString.h" 18 #include "llvm/CodeGen/MachineScheduler.h" 19 #include "llvm/Target/TargetFrameLowering.h" 20 #include <algorithm> 21 22 using namespace llvm; 23 24 #define DEBUG_TYPE "amdgpu-subtarget" 25 26 #define GET_SUBTARGETINFO_ENUM 27 #define GET_SUBTARGETINFO_TARGET_DESC 28 #define GET_SUBTARGETINFO_CTOR 29 #include "AMDGPUGenSubtargetInfo.inc" 30 31 AMDGPUSubtarget::~AMDGPUSubtarget() = default; 32 33 AMDGPUSubtarget & 34 AMDGPUSubtarget::initializeSubtargetDependencies(const Triple &TT, 35 StringRef GPU, StringRef FS) { 36 // Determine default and user-specified characteristics 37 // On SI+, we want FP64 denormals to be on by default. FP32 denormals can be 38 // enabled, but some instructions do not respect them and they run at the 39 // double precision rate, so don't enable by default. 40 // 41 // We want to be able to turn these off, but making this a subtarget feature 42 // for SI has the unhelpful behavior that it unsets everything else if you 43 // disable it. 44 45 SmallString<256> FullFS("+promote-alloca,+fp64-fp16-denormals,+dx10-clamp,+load-store-opt,"); 46 if (isAmdHsaOS()) // Turn on FlatForGlobal for HSA. 47 FullFS += "+flat-for-global,+unaligned-buffer-access,+trap-handler,"; 48 49 FullFS += FS; 50 51 ParseSubtargetFeatures(GPU, FullFS); 52 53 // Unless +-flat-for-global is specified, turn on FlatForGlobal for all OS-es 54 // on VI and newer hardware to avoid assertion failures due to missing ADDR64 55 // variants of MUBUF instructions. 56 if (!hasAddr64() && !FS.contains("flat-for-global")) { 57 FlatForGlobal = true; 58 } 59 60 // FIXME: I don't think think Evergreen has any useful support for 61 // denormals, but should be checked. Should we issue a warning somewhere 62 // if someone tries to enable these? 63 if (getGeneration() <= AMDGPUSubtarget::NORTHERN_ISLANDS) { 64 FP64FP16Denormals = false; 65 FP32Denormals = false; 66 } 67 68 // Set defaults if needed. 69 if (MaxPrivateElementSize == 0) 70 MaxPrivateElementSize = 4; 71 72 return *this; 73 } 74 75 AMDGPUSubtarget::AMDGPUSubtarget(const Triple &TT, StringRef GPU, StringRef FS, 76 const TargetMachine &TM) 77 : AMDGPUGenSubtargetInfo(TT, GPU, FS), 78 TargetTriple(TT), 79 Gen(TT.getArch() == Triple::amdgcn ? SOUTHERN_ISLANDS : R600), 80 IsaVersion(ISAVersion0_0_0), 81 WavefrontSize(64), 82 LocalMemorySize(0), 83 LDSBankCount(0), 84 MaxPrivateElementSize(0), 85 86 FastFMAF32(false), 87 HalfRate64Ops(false), 88 89 FP32Denormals(false), 90 FP64FP16Denormals(false), 91 FPExceptions(false), 92 DX10Clamp(false), 93 FlatForGlobal(false), 94 UnalignedScratchAccess(false), 95 UnalignedBufferAccess(false), 96 97 HasApertureRegs(false), 98 EnableXNACK(false), 99 TrapHandler(false), 100 DebuggerInsertNops(false), 101 DebuggerReserveRegs(false), 102 DebuggerEmitPrologue(false), 103 104 EnableVGPRSpilling(false), 105 EnablePromoteAlloca(false), 106 EnableLoadStoreOpt(false), 107 EnableUnsafeDSOffsetFolding(false), 108 EnableSIScheduler(false), 109 DumpCode(false), 110 111 FP64(false), 112 IsGCN(false), 113 GCN1Encoding(false), 114 GCN3Encoding(false), 115 CIInsts(false), 116 GFX9Insts(false), 117 SGPRInitBug(false), 118 HasSMemRealTime(false), 119 Has16BitInsts(false), 120 HasMovrel(false), 121 HasVGPRIndexMode(false), 122 HasScalarStores(false), 123 HasInv2PiInlineImm(false), 124 HasSDWA(false), 125 HasDPP(false), 126 FlatAddressSpace(false), 127 128 R600ALUInst(false), 129 CaymanISA(false), 130 CFALUBug(false), 131 HasVertexCache(false), 132 TexVTXClauseSize(0), 133 ScalarizeGlobal(false), 134 135 FeatureDisable(false), 136 InstrItins(getInstrItineraryForCPU(GPU)) { 137 initializeSubtargetDependencies(TT, GPU, FS); 138 } 139 140 unsigned AMDGPUSubtarget::getMaxLocalMemSizeWithWaveCount(unsigned NWaves, 141 const Function &F) const { 142 if (NWaves == 1) 143 return getLocalMemorySize(); 144 unsigned WorkGroupSize = getFlatWorkGroupSizes(F).second; 145 unsigned WorkGroupsPerCu = getMaxWorkGroupsPerCU(WorkGroupSize); 146 unsigned MaxWaves = getMaxWavesPerEU(); 147 return getLocalMemorySize() * MaxWaves / WorkGroupsPerCu / NWaves; 148 } 149 150 unsigned AMDGPUSubtarget::getOccupancyWithLocalMemSize(uint32_t Bytes, 151 const Function &F) const { 152 unsigned WorkGroupSize = getFlatWorkGroupSizes(F).second; 153 unsigned WorkGroupsPerCu = getMaxWorkGroupsPerCU(WorkGroupSize); 154 unsigned MaxWaves = getMaxWavesPerEU(); 155 unsigned Limit = getLocalMemorySize() * MaxWaves / WorkGroupsPerCu; 156 unsigned NumWaves = Limit / (Bytes ? Bytes : 1u); 157 NumWaves = std::min(NumWaves, MaxWaves); 158 NumWaves = std::max(NumWaves, 1u); 159 return NumWaves; 160 } 161 162 std::pair<unsigned, unsigned> AMDGPUSubtarget::getFlatWorkGroupSizes( 163 const Function &F) const { 164 // Default minimum/maximum flat work group sizes. 165 std::pair<unsigned, unsigned> Default = 166 AMDGPU::isCompute(F.getCallingConv()) ? 167 std::pair<unsigned, unsigned>(getWavefrontSize() * 2, 168 getWavefrontSize() * 4) : 169 std::pair<unsigned, unsigned>(1, getWavefrontSize()); 170 171 // TODO: Do not process "amdgpu-max-work-group-size" attribute once mesa 172 // starts using "amdgpu-flat-work-group-size" attribute. 173 Default.second = AMDGPU::getIntegerAttribute( 174 F, "amdgpu-max-work-group-size", Default.second); 175 Default.first = std::min(Default.first, Default.second); 176 177 // Requested minimum/maximum flat work group sizes. 178 std::pair<unsigned, unsigned> Requested = AMDGPU::getIntegerPairAttribute( 179 F, "amdgpu-flat-work-group-size", Default); 180 181 // Make sure requested minimum is less than requested maximum. 182 if (Requested.first > Requested.second) 183 return Default; 184 185 // Make sure requested values do not violate subtarget's specifications. 186 if (Requested.first < getMinFlatWorkGroupSize()) 187 return Default; 188 if (Requested.second > getMaxFlatWorkGroupSize()) 189 return Default; 190 191 return Requested; 192 } 193 194 std::pair<unsigned, unsigned> AMDGPUSubtarget::getWavesPerEU( 195 const Function &F) const { 196 // Default minimum/maximum number of waves per execution unit. 197 std::pair<unsigned, unsigned> Default(1, getMaxWavesPerEU()); 198 199 // Default/requested minimum/maximum flat work group sizes. 200 std::pair<unsigned, unsigned> FlatWorkGroupSizes = getFlatWorkGroupSizes(F); 201 202 // If minimum/maximum flat work group sizes were explicitly requested using 203 // "amdgpu-flat-work-group-size" attribute, then set default minimum/maximum 204 // number of waves per execution unit to values implied by requested 205 // minimum/maximum flat work group sizes. 206 unsigned MinImpliedByFlatWorkGroupSize = 207 getMaxWavesPerEU(FlatWorkGroupSizes.second); 208 bool RequestedFlatWorkGroupSize = false; 209 210 // TODO: Do not process "amdgpu-max-work-group-size" attribute once mesa 211 // starts using "amdgpu-flat-work-group-size" attribute. 212 if (F.hasFnAttribute("amdgpu-max-work-group-size") || 213 F.hasFnAttribute("amdgpu-flat-work-group-size")) { 214 Default.first = MinImpliedByFlatWorkGroupSize; 215 RequestedFlatWorkGroupSize = true; 216 } 217 218 // Requested minimum/maximum number of waves per execution unit. 219 std::pair<unsigned, unsigned> Requested = AMDGPU::getIntegerPairAttribute( 220 F, "amdgpu-waves-per-eu", Default, true); 221 222 // Make sure requested minimum is less than requested maximum. 223 if (Requested.second && Requested.first > Requested.second) 224 return Default; 225 226 // Make sure requested values do not violate subtarget's specifications. 227 if (Requested.first < getMinWavesPerEU() || 228 Requested.first > getMaxWavesPerEU()) 229 return Default; 230 if (Requested.second > getMaxWavesPerEU()) 231 return Default; 232 233 // Make sure requested values are compatible with values implied by requested 234 // minimum/maximum flat work group sizes. 235 if (RequestedFlatWorkGroupSize && 236 Requested.first > MinImpliedByFlatWorkGroupSize) 237 return Default; 238 239 return Requested; 240 } 241 242 R600Subtarget::R600Subtarget(const Triple &TT, StringRef GPU, StringRef FS, 243 const TargetMachine &TM) : 244 AMDGPUSubtarget(TT, GPU, FS, TM), 245 InstrInfo(*this), 246 FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0), 247 TLInfo(TM, *this) {} 248 249 SISubtarget::SISubtarget(const Triple &TT, StringRef GPU, StringRef FS, 250 const TargetMachine &TM) : 251 AMDGPUSubtarget(TT, GPU, FS, TM), 252 InstrInfo(*this), 253 FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0), 254 TLInfo(TM, *this) {} 255 256 void SISubtarget::overrideSchedPolicy(MachineSchedPolicy &Policy, 257 unsigned NumRegionInstrs) const { 258 // Track register pressure so the scheduler can try to decrease 259 // pressure once register usage is above the threshold defined by 260 // SIRegisterInfo::getRegPressureSetLimit() 261 Policy.ShouldTrackPressure = true; 262 263 // Enabling both top down and bottom up scheduling seems to give us less 264 // register spills than just using one of these approaches on its own. 265 Policy.OnlyTopDown = false; 266 Policy.OnlyBottomUp = false; 267 268 // Enabling ShouldTrackLaneMasks crashes the SI Machine Scheduler. 269 if (!enableSIScheduler()) 270 Policy.ShouldTrackLaneMasks = true; 271 } 272 273 bool SISubtarget::isVGPRSpillingEnabled(const Function& F) const { 274 return EnableVGPRSpilling || !AMDGPU::isShader(F.getCallingConv()); 275 } 276 277 unsigned SISubtarget::getKernArgSegmentSize(const MachineFunction &MF, 278 unsigned ExplicitArgBytes) const { 279 unsigned ImplicitBytes = getImplicitArgNumBytes(MF); 280 if (ImplicitBytes == 0) 281 return ExplicitArgBytes; 282 283 unsigned Alignment = getAlignmentForImplicitArgPtr(); 284 return alignTo(ExplicitArgBytes, Alignment) + ImplicitBytes; 285 } 286 287 unsigned SISubtarget::getOccupancyWithNumSGPRs(unsigned SGPRs) const { 288 if (getGeneration() >= SISubtarget::VOLCANIC_ISLANDS) { 289 if (SGPRs <= 80) 290 return 10; 291 if (SGPRs <= 88) 292 return 9; 293 if (SGPRs <= 100) 294 return 8; 295 return 7; 296 } 297 if (SGPRs <= 48) 298 return 10; 299 if (SGPRs <= 56) 300 return 9; 301 if (SGPRs <= 64) 302 return 8; 303 if (SGPRs <= 72) 304 return 7; 305 if (SGPRs <= 80) 306 return 6; 307 return 5; 308 } 309 310 unsigned SISubtarget::getOccupancyWithNumVGPRs(unsigned VGPRs) const { 311 if (VGPRs <= 24) 312 return 10; 313 if (VGPRs <= 28) 314 return 9; 315 if (VGPRs <= 32) 316 return 8; 317 if (VGPRs <= 36) 318 return 7; 319 if (VGPRs <= 40) 320 return 6; 321 if (VGPRs <= 48) 322 return 5; 323 if (VGPRs <= 64) 324 return 4; 325 if (VGPRs <= 84) 326 return 3; 327 if (VGPRs <= 128) 328 return 2; 329 return 1; 330 } 331 332 unsigned SISubtarget::getReservedNumSGPRs(const MachineFunction &MF) const { 333 const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>(); 334 if (MFI.hasFlatScratchInit()) { 335 if (getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS) 336 return 6; // FLAT_SCRATCH, XNACK, VCC (in that order). 337 if (getGeneration() == AMDGPUSubtarget::SEA_ISLANDS) 338 return 4; // FLAT_SCRATCH, VCC (in that order). 339 } 340 341 if (isXNACKEnabled()) 342 return 4; // XNACK, VCC (in that order). 343 return 2; // VCC. 344 } 345 346 unsigned SISubtarget::getMaxNumSGPRs(const MachineFunction &MF) const { 347 const Function &F = *MF.getFunction(); 348 const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>(); 349 350 // Compute maximum number of SGPRs function can use using default/requested 351 // minimum number of waves per execution unit. 352 std::pair<unsigned, unsigned> WavesPerEU = MFI.getWavesPerEU(); 353 unsigned MaxNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, false); 354 unsigned MaxAddressableNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, true); 355 356 // Check if maximum number of SGPRs was explicitly requested using 357 // "amdgpu-num-sgpr" attribute. 358 if (F.hasFnAttribute("amdgpu-num-sgpr")) { 359 unsigned Requested = AMDGPU::getIntegerAttribute( 360 F, "amdgpu-num-sgpr", MaxNumSGPRs); 361 362 // Make sure requested value does not violate subtarget's specifications. 363 if (Requested && (Requested <= getReservedNumSGPRs(MF))) 364 Requested = 0; 365 366 // If more SGPRs are required to support the input user/system SGPRs, 367 // increase to accommodate them. 368 // 369 // FIXME: This really ends up using the requested number of SGPRs + number 370 // of reserved special registers in total. Theoretically you could re-use 371 // the last input registers for these special registers, but this would 372 // require a lot of complexity to deal with the weird aliasing. 373 unsigned InputNumSGPRs = MFI.getNumPreloadedSGPRs(); 374 if (Requested && Requested < InputNumSGPRs) 375 Requested = InputNumSGPRs; 376 377 // Make sure requested value is compatible with values implied by 378 // default/requested minimum/maximum number of waves per execution unit. 379 if (Requested && Requested > getMaxNumSGPRs(WavesPerEU.first, false)) 380 Requested = 0; 381 if (WavesPerEU.second && 382 Requested && Requested < getMinNumSGPRs(WavesPerEU.second)) 383 Requested = 0; 384 385 if (Requested) 386 MaxNumSGPRs = Requested; 387 } 388 389 if (hasSGPRInitBug()) 390 MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG; 391 392 return std::min(MaxNumSGPRs - getReservedNumSGPRs(MF), 393 MaxAddressableNumSGPRs); 394 } 395 396 unsigned SISubtarget::getMaxNumVGPRs(const MachineFunction &MF) const { 397 const Function &F = *MF.getFunction(); 398 const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>(); 399 400 // Compute maximum number of VGPRs function can use using default/requested 401 // minimum number of waves per execution unit. 402 std::pair<unsigned, unsigned> WavesPerEU = MFI.getWavesPerEU(); 403 unsigned MaxNumVGPRs = getMaxNumVGPRs(WavesPerEU.first); 404 405 // Check if maximum number of VGPRs was explicitly requested using 406 // "amdgpu-num-vgpr" attribute. 407 if (F.hasFnAttribute("amdgpu-num-vgpr")) { 408 unsigned Requested = AMDGPU::getIntegerAttribute( 409 F, "amdgpu-num-vgpr", MaxNumVGPRs); 410 411 // Make sure requested value does not violate subtarget's specifications. 412 if (Requested && Requested <= getReservedNumVGPRs(MF)) 413 Requested = 0; 414 415 // Make sure requested value is compatible with values implied by 416 // default/requested minimum/maximum number of waves per execution unit. 417 if (Requested && Requested > getMaxNumVGPRs(WavesPerEU.first)) 418 Requested = 0; 419 if (WavesPerEU.second && 420 Requested && Requested < getMinNumVGPRs(WavesPerEU.second)) 421 Requested = 0; 422 423 if (Requested) 424 MaxNumVGPRs = Requested; 425 } 426 427 return MaxNumVGPRs - getReservedNumVGPRs(MF); 428 } 429