1 //=====-- AMDGPUSubtarget.h - Define Subtarget for AMDGPU ------*- C++ -*-====// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //==-----------------------------------------------------------------------===// 9 // 10 /// \file 11 /// \brief AMDGPU specific subclass of TargetSubtarget. 12 // 13 //===----------------------------------------------------------------------===// 14 15 #ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H 16 #define LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H 17 18 #include "AMDGPU.h" 19 #include "R600InstrInfo.h" 20 #include "R600ISelLowering.h" 21 #include "R600FrameLowering.h" 22 #include "SIInstrInfo.h" 23 #include "SIISelLowering.h" 24 #include "SIFrameLowering.h" 25 #include "Utils/AMDGPUBaseInfo.h" 26 #include "llvm/CodeGen/GlobalISel/GISelAccessor.h" 27 #include "llvm/CodeGen/SelectionDAGTargetInfo.h" 28 #include "llvm/Target/TargetSubtargetInfo.h" 29 30 #define GET_SUBTARGETINFO_HEADER 31 #include "AMDGPUGenSubtargetInfo.inc" 32 33 namespace llvm { 34 35 class SIMachineFunctionInfo; 36 class StringRef; 37 38 class AMDGPUSubtarget : public AMDGPUGenSubtargetInfo { 39 public: 40 enum Generation { 41 R600 = 0, 42 R700, 43 EVERGREEN, 44 NORTHERN_ISLANDS, 45 SOUTHERN_ISLANDS, 46 SEA_ISLANDS, 47 VOLCANIC_ISLANDS, 48 }; 49 50 enum { 51 ISAVersion0_0_0, 52 ISAVersion7_0_0, 53 ISAVersion7_0_1, 54 ISAVersion8_0_0, 55 ISAVersion8_0_1, 56 ISAVersion8_0_2, 57 ISAVersion8_0_3 58 }; 59 60 protected: 61 // Basic subtarget description. 62 Triple TargetTriple; 63 Generation Gen; 64 unsigned IsaVersion; 65 unsigned WavefrontSize; 66 int LocalMemorySize; 67 int LDSBankCount; 68 unsigned MaxPrivateElementSize; 69 70 // Possibly statically set by tablegen, but may want to be overridden. 71 bool FastFMAF32; 72 bool HalfRate64Ops; 73 74 // Dynamially set bits that enable features. 75 bool FP32Denormals; 76 bool FP64Denormals; 77 bool FPExceptions; 78 bool FlatForGlobal; 79 bool UnalignedBufferAccess; 80 bool EnableXNACK; 81 bool DebuggerInsertNops; 82 bool DebuggerReserveRegs; 83 bool DebuggerEmitPrologue; 84 85 // Used as options. 86 bool EnableVGPRSpilling; 87 bool EnablePromoteAlloca; 88 bool EnableLoadStoreOpt; 89 bool EnableUnsafeDSOffsetFolding; 90 bool EnableSIScheduler; 91 bool DumpCode; 92 93 // Subtarget statically properties set by tablegen 94 bool FP64; 95 bool IsGCN; 96 bool GCN1Encoding; 97 bool GCN3Encoding; 98 bool CIInsts; 99 bool SGPRInitBug; 100 bool HasSMemRealTime; 101 bool Has16BitInsts; 102 bool FlatAddressSpace; 103 bool R600ALUInst; 104 bool CaymanISA; 105 bool CFALUBug; 106 bool HasVertexCache; 107 short TexVTXClauseSize; 108 109 // Dummy feature to use for assembler in tablegen. 110 bool FeatureDisable; 111 112 InstrItineraryData InstrItins; 113 SelectionDAGTargetInfo TSInfo; 114 115 public: 116 AMDGPUSubtarget(const Triple &TT, StringRef GPU, StringRef FS, 117 const TargetMachine &TM); 118 virtual ~AMDGPUSubtarget(); 119 AMDGPUSubtarget &initializeSubtargetDependencies(const Triple &TT, 120 StringRef GPU, StringRef FS); 121 122 const AMDGPUInstrInfo *getInstrInfo() const override = 0; 123 const AMDGPUFrameLowering *getFrameLowering() const override = 0; 124 const AMDGPUTargetLowering *getTargetLowering() const override = 0; 125 const AMDGPURegisterInfo *getRegisterInfo() const override = 0; 126 127 const InstrItineraryData *getInstrItineraryData() const override { 128 return &InstrItins; 129 } 130 131 // Nothing implemented, just prevent crashes on use. 132 const SelectionDAGTargetInfo *getSelectionDAGInfo() const override { 133 return &TSInfo; 134 } 135 136 void ParseSubtargetFeatures(StringRef CPU, StringRef FS); 137 138 bool isAmdHsaOS() const { 139 return TargetTriple.getOS() == Triple::AMDHSA; 140 } 141 142 bool isMesa3DOS() const { 143 return TargetTriple.getOS() == Triple::Mesa3D; 144 } 145 146 bool isOpenCLEnv() const { 147 return TargetTriple.getEnvironment() == Triple::OpenCL; 148 } 149 150 Generation getGeneration() const { 151 return Gen; 152 } 153 154 unsigned getWavefrontSize() const { 155 return WavefrontSize; 156 } 157 158 int getLocalMemorySize() const { 159 return LocalMemorySize; 160 } 161 162 int getLDSBankCount() const { 163 return LDSBankCount; 164 } 165 166 unsigned getMaxPrivateElementSize() const { 167 return MaxPrivateElementSize; 168 } 169 170 bool hasHWFP64() const { 171 return FP64; 172 } 173 174 bool hasFastFMAF32() const { 175 return FastFMAF32; 176 } 177 178 bool hasHalfRate64Ops() const { 179 return HalfRate64Ops; 180 } 181 182 bool hasAddr64() const { 183 return (getGeneration() < VOLCANIC_ISLANDS); 184 } 185 186 bool hasBFE() const { 187 return (getGeneration() >= EVERGREEN); 188 } 189 190 bool hasBFI() const { 191 return (getGeneration() >= EVERGREEN); 192 } 193 194 bool hasBFM() const { 195 return hasBFE(); 196 } 197 198 bool hasBCNT(unsigned Size) const { 199 if (Size == 32) 200 return (getGeneration() >= EVERGREEN); 201 202 if (Size == 64) 203 return (getGeneration() >= SOUTHERN_ISLANDS); 204 205 return false; 206 } 207 208 bool hasMulU24() const { 209 return (getGeneration() >= EVERGREEN); 210 } 211 212 bool hasMulI24() const { 213 return (getGeneration() >= SOUTHERN_ISLANDS || 214 hasCaymanISA()); 215 } 216 217 bool hasFFBL() const { 218 return (getGeneration() >= EVERGREEN); 219 } 220 221 bool hasFFBH() const { 222 return (getGeneration() >= EVERGREEN); 223 } 224 225 bool hasCARRY() const { 226 return (getGeneration() >= EVERGREEN); 227 } 228 229 bool hasBORROW() const { 230 return (getGeneration() >= EVERGREEN); 231 } 232 233 bool hasCaymanISA() const { 234 return CaymanISA; 235 } 236 237 bool isPromoteAllocaEnabled() const { 238 return EnablePromoteAlloca; 239 } 240 241 bool unsafeDSOffsetFoldingEnabled() const { 242 return EnableUnsafeDSOffsetFolding; 243 } 244 245 bool dumpCode() const { 246 return DumpCode; 247 } 248 249 /// Return the amount of LDS that can be used that will not restrict the 250 /// occupancy lower than WaveCount. 251 unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount) const; 252 253 /// Inverse of getMaxLocalMemWithWaveCount. Return the maximum wavecount if 254 /// the given LDS memory size is the only constraint. 255 unsigned getOccupancyWithLocalMemSize(uint32_t Bytes) const; 256 257 258 bool hasFP32Denormals() const { 259 return FP32Denormals; 260 } 261 262 bool hasFP64Denormals() const { 263 return FP64Denormals; 264 } 265 266 bool hasFPExceptions() const { 267 return FPExceptions; 268 } 269 270 bool useFlatForGlobal() const { 271 return FlatForGlobal; 272 } 273 274 bool hasUnalignedBufferAccess() const { 275 return UnalignedBufferAccess; 276 } 277 278 bool isXNACKEnabled() const { 279 return EnableXNACK; 280 } 281 282 bool isAmdCodeObjectV2() const { 283 return isAmdHsaOS() || isMesa3DOS(); 284 } 285 286 /// \brief Returns the offset in bytes from the start of the input buffer 287 /// of the first explicit kernel argument. 288 unsigned getExplicitKernelArgOffset() const { 289 return isAmdCodeObjectV2() ? 0 : 36; 290 } 291 292 unsigned getAlignmentForImplicitArgPtr() const { 293 return isAmdHsaOS() ? 8 : 4; 294 } 295 296 unsigned getImplicitArgNumBytes() const { 297 if (isMesa3DOS()) 298 return 16; 299 if (isAmdHsaOS() && isOpenCLEnv()) 300 return 32; 301 return 0; 302 } 303 304 unsigned getStackAlignment() const { 305 // Scratch is allocated in 256 dword per wave blocks. 306 return 4 * 256 / getWavefrontSize(); 307 } 308 309 bool enableMachineScheduler() const override { 310 return true; 311 } 312 313 bool enableSubRegLiveness() const override { 314 return true; 315 } 316 317 /// \returns Number of execution units per compute unit supported by the 318 /// subtarget. 319 unsigned getEUsPerCU() const { 320 return 4; 321 } 322 323 /// \returns Maximum number of work groups per compute unit supported by the 324 /// subtarget and limited by given flat work group size. 325 unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const { 326 if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS) 327 return 8; 328 return getWavesPerWorkGroup(FlatWorkGroupSize) == 1 ? 40 : 16; 329 } 330 331 /// \returns Maximum number of waves per compute unit supported by the 332 /// subtarget without any kind of limitation. 333 unsigned getMaxWavesPerCU() const { 334 return getMaxWavesPerEU() * getEUsPerCU(); 335 } 336 337 /// \returns Maximum number of waves per compute unit supported by the 338 /// subtarget and limited by given flat work group size. 339 unsigned getMaxWavesPerCU(unsigned FlatWorkGroupSize) const { 340 return getWavesPerWorkGroup(FlatWorkGroupSize); 341 } 342 343 /// \returns Minimum number of waves per execution unit supported by the 344 /// subtarget. 345 unsigned getMinWavesPerEU() const { 346 return 1; 347 } 348 349 /// \returns Maximum number of waves per execution unit supported by the 350 /// subtarget without any kind of limitation. 351 unsigned getMaxWavesPerEU() const { 352 if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS) 353 return 8; 354 // FIXME: Need to take scratch memory into account. 355 return 10; 356 } 357 358 /// \returns Maximum number of waves per execution unit supported by the 359 /// subtarget and limited by given flat work group size. 360 unsigned getMaxWavesPerEU(unsigned FlatWorkGroupSize) const { 361 return alignTo(getMaxWavesPerCU(FlatWorkGroupSize), getEUsPerCU()) / 362 getEUsPerCU(); 363 } 364 365 /// \returns Minimum flat work group size supported by the subtarget. 366 unsigned getMinFlatWorkGroupSize() const { 367 return 1; 368 } 369 370 /// \returns Maximum flat work group size supported by the subtarget. 371 unsigned getMaxFlatWorkGroupSize() const { 372 return 2048; 373 } 374 375 /// \returns Number of waves per work group given the flat work group size. 376 unsigned getWavesPerWorkGroup(unsigned FlatWorkGroupSize) const { 377 return alignTo(FlatWorkGroupSize, getWavefrontSize()) / getWavefrontSize(); 378 } 379 380 /// \returns Subtarget's default pair of minimum/maximum flat work group sizes 381 /// for function \p F, or minimum/maximum flat work group sizes explicitly 382 /// requested using "amdgpu-flat-work-group-size" attribute attached to 383 /// function \p F. 384 /// 385 /// \returns Subtarget's default values if explicitly requested values cannot 386 /// be converted to integer, or violate subtarget's specifications. 387 std::pair<unsigned, unsigned> getFlatWorkGroupSizes(const Function &F) const; 388 389 /// \returns Subtarget's default pair of minimum/maximum number of waves per 390 /// execution unit for function \p F, or minimum/maximum number of waves per 391 /// execution unit explicitly requested using "amdgpu-waves-per-eu" attribute 392 /// attached to function \p F. 393 /// 394 /// \returns Subtarget's default values if explicitly requested values cannot 395 /// be converted to integer, violate subtarget's specifications, or are not 396 /// compatible with minimum/maximum number of waves limited by flat work group 397 /// size, register usage, and/or lds usage. 398 std::pair<unsigned, unsigned> getWavesPerEU(const Function &F) const; 399 }; 400 401 class R600Subtarget final : public AMDGPUSubtarget { 402 private: 403 R600InstrInfo InstrInfo; 404 R600FrameLowering FrameLowering; 405 R600TargetLowering TLInfo; 406 407 public: 408 R600Subtarget(const Triple &TT, StringRef CPU, StringRef FS, 409 const TargetMachine &TM); 410 411 const R600InstrInfo *getInstrInfo() const override { 412 return &InstrInfo; 413 } 414 415 const R600FrameLowering *getFrameLowering() const override { 416 return &FrameLowering; 417 } 418 419 const R600TargetLowering *getTargetLowering() const override { 420 return &TLInfo; 421 } 422 423 const R600RegisterInfo *getRegisterInfo() const override { 424 return &InstrInfo.getRegisterInfo(); 425 } 426 427 bool hasCFAluBug() const { 428 return CFALUBug; 429 } 430 431 bool hasVertexCache() const { 432 return HasVertexCache; 433 } 434 435 short getTexVTXClauseSize() const { 436 return TexVTXClauseSize; 437 } 438 }; 439 440 class SISubtarget final : public AMDGPUSubtarget { 441 public: 442 enum { 443 // The closed Vulkan driver sets 96, which limits the wave count to 8 but 444 // doesn't spill SGPRs as much as when 80 is set. 445 FIXED_SGPR_COUNT_FOR_INIT_BUG = 96 446 }; 447 448 private: 449 SIInstrInfo InstrInfo; 450 SIFrameLowering FrameLowering; 451 SITargetLowering TLInfo; 452 std::unique_ptr<GISelAccessor> GISel; 453 454 public: 455 SISubtarget(const Triple &TT, StringRef CPU, StringRef FS, 456 const TargetMachine &TM); 457 458 const SIInstrInfo *getInstrInfo() const override { 459 return &InstrInfo; 460 } 461 462 const SIFrameLowering *getFrameLowering() const override { 463 return &FrameLowering; 464 } 465 466 const SITargetLowering *getTargetLowering() const override { 467 return &TLInfo; 468 } 469 470 const CallLowering *getCallLowering() const override { 471 assert(GISel && "Access to GlobalISel APIs not set"); 472 return GISel->getCallLowering(); 473 } 474 475 const SIRegisterInfo *getRegisterInfo() const override { 476 return &InstrInfo.getRegisterInfo(); 477 } 478 479 void setGISelAccessor(GISelAccessor &GISel) { 480 this->GISel.reset(&GISel); 481 } 482 483 void overrideSchedPolicy(MachineSchedPolicy &Policy, 484 unsigned NumRegionInstrs) const override; 485 486 bool isVGPRSpillingEnabled(const Function& F) const; 487 488 unsigned getMaxNumUserSGPRs() const { 489 return 16; 490 } 491 492 bool hasFlatAddressSpace() const { 493 return FlatAddressSpace; 494 } 495 496 bool hasSMemRealTime() const { 497 return HasSMemRealTime; 498 } 499 500 bool has16BitInsts() const { 501 return Has16BitInsts; 502 } 503 504 bool hasScalarCompareEq64() const { 505 return getGeneration() >= VOLCANIC_ISLANDS; 506 } 507 508 bool enableSIScheduler() const { 509 return EnableSIScheduler; 510 } 511 512 bool debuggerSupported() const { 513 return debuggerInsertNops() && debuggerReserveRegs() && 514 debuggerEmitPrologue(); 515 } 516 517 bool debuggerInsertNops() const { 518 return DebuggerInsertNops; 519 } 520 521 bool debuggerReserveRegs() const { 522 return DebuggerReserveRegs; 523 } 524 525 bool debuggerEmitPrologue() const { 526 return DebuggerEmitPrologue; 527 } 528 529 bool loadStoreOptEnabled() const { 530 return EnableLoadStoreOpt; 531 } 532 533 bool hasSGPRInitBug() const { 534 return SGPRInitBug; 535 } 536 537 unsigned getKernArgSegmentSize(unsigned ExplictArgBytes) const; 538 539 /// Return the maximum number of waves per SIMD for kernels using \p SGPRs SGPRs 540 unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const; 541 542 /// Return the maximum number of waves per SIMD for kernels using \p VGPRs VGPRs 543 unsigned getOccupancyWithNumVGPRs(unsigned VGPRs) const; 544 545 /// \returns True if waitcnt instruction is needed before barrier instruction, 546 /// false otherwise. 547 bool needWaitcntBeforeBarrier() const { 548 return true; 549 } 550 }; 551 552 } // End namespace llvm 553 554 #endif 555