1 //=====-- AMDGPUSubtarget.h - Define Subtarget for AMDGPU ------*- C++ -*-====// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //==-----------------------------------------------------------------------===// 9 // 10 /// \file 11 /// \brief AMDGPU specific subclass of TargetSubtarget. 12 // 13 //===----------------------------------------------------------------------===// 14 15 #ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H 16 #define LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H 17 18 #include "AMDGPU.h" 19 #include "R600InstrInfo.h" 20 #include "R600ISelLowering.h" 21 #include "R600FrameLowering.h" 22 #include "SIInstrInfo.h" 23 #include "SIISelLowering.h" 24 #include "SIFrameLowering.h" 25 #include "Utils/AMDGPUBaseInfo.h" 26 #include "llvm/CodeGen/GlobalISel/GISelAccessor.h" 27 #include "llvm/CodeGen/SelectionDAGTargetInfo.h" 28 #include "llvm/Target/TargetSubtargetInfo.h" 29 30 #define GET_SUBTARGETINFO_HEADER 31 #include "AMDGPUGenSubtargetInfo.inc" 32 33 namespace llvm { 34 35 class SIMachineFunctionInfo; 36 class StringRef; 37 38 class AMDGPUSubtarget : public AMDGPUGenSubtargetInfo { 39 public: 40 enum Generation { 41 R600 = 0, 42 R700, 43 EVERGREEN, 44 NORTHERN_ISLANDS, 45 SOUTHERN_ISLANDS, 46 SEA_ISLANDS, 47 VOLCANIC_ISLANDS, 48 }; 49 50 enum { 51 ISAVersion0_0_0, 52 ISAVersion7_0_0, 53 ISAVersion7_0_1, 54 ISAVersion8_0_0, 55 ISAVersion8_0_1, 56 ISAVersion8_0_2, 57 ISAVersion8_0_3 58 }; 59 60 protected: 61 // Basic subtarget description. 62 Triple TargetTriple; 63 Generation Gen; 64 unsigned IsaVersion; 65 unsigned WavefrontSize; 66 int LocalMemorySize; 67 int LDSBankCount; 68 unsigned MaxPrivateElementSize; 69 70 // Possibly statically set by tablegen, but may want to be overridden. 71 bool FastFMAF32; 72 bool HalfRate64Ops; 73 74 // Dynamially set bits that enable features. 75 bool FP32Denormals; 76 bool FP64Denormals; 77 bool FPExceptions; 78 bool FlatForGlobal; 79 bool UnalignedScratchAccess; 80 bool UnalignedBufferAccess; 81 bool EnableXNACK; 82 bool DebuggerInsertNops; 83 bool DebuggerReserveRegs; 84 bool DebuggerEmitPrologue; 85 86 // Used as options. 87 bool EnableVGPRSpilling; 88 bool EnablePromoteAlloca; 89 bool EnableLoadStoreOpt; 90 bool EnableUnsafeDSOffsetFolding; 91 bool EnableSIScheduler; 92 bool DumpCode; 93 94 // Subtarget statically properties set by tablegen 95 bool FP64; 96 bool IsGCN; 97 bool GCN1Encoding; 98 bool GCN3Encoding; 99 bool CIInsts; 100 bool SGPRInitBug; 101 bool HasSMemRealTime; 102 bool Has16BitInsts; 103 bool HasMovrel; 104 bool HasVGPRIndexMode; 105 bool FlatAddressSpace; 106 bool R600ALUInst; 107 bool CaymanISA; 108 bool CFALUBug; 109 bool HasVertexCache; 110 short TexVTXClauseSize; 111 112 // Dummy feature to use for assembler in tablegen. 113 bool FeatureDisable; 114 115 InstrItineraryData InstrItins; 116 SelectionDAGTargetInfo TSInfo; 117 118 public: 119 AMDGPUSubtarget(const Triple &TT, StringRef GPU, StringRef FS, 120 const TargetMachine &TM); 121 virtual ~AMDGPUSubtarget(); 122 AMDGPUSubtarget &initializeSubtargetDependencies(const Triple &TT, 123 StringRef GPU, StringRef FS); 124 125 const AMDGPUInstrInfo *getInstrInfo() const override = 0; 126 const AMDGPUFrameLowering *getFrameLowering() const override = 0; 127 const AMDGPUTargetLowering *getTargetLowering() const override = 0; 128 const AMDGPURegisterInfo *getRegisterInfo() const override = 0; 129 130 const InstrItineraryData *getInstrItineraryData() const override { 131 return &InstrItins; 132 } 133 134 // Nothing implemented, just prevent crashes on use. 135 const SelectionDAGTargetInfo *getSelectionDAGInfo() const override { 136 return &TSInfo; 137 } 138 139 void ParseSubtargetFeatures(StringRef CPU, StringRef FS); 140 141 bool isAmdHsaOS() const { 142 return TargetTriple.getOS() == Triple::AMDHSA; 143 } 144 145 bool isMesa3DOS() const { 146 return TargetTriple.getOS() == Triple::Mesa3D; 147 } 148 149 bool isOpenCLEnv() const { 150 return TargetTriple.getEnvironment() == Triple::OpenCL; 151 } 152 153 Generation getGeneration() const { 154 return Gen; 155 } 156 157 unsigned getWavefrontSize() const { 158 return WavefrontSize; 159 } 160 161 int getLocalMemorySize() const { 162 return LocalMemorySize; 163 } 164 165 int getLDSBankCount() const { 166 return LDSBankCount; 167 } 168 169 unsigned getMaxPrivateElementSize() const { 170 return MaxPrivateElementSize; 171 } 172 173 bool hasHWFP64() const { 174 return FP64; 175 } 176 177 bool hasFastFMAF32() const { 178 return FastFMAF32; 179 } 180 181 bool hasHalfRate64Ops() const { 182 return HalfRate64Ops; 183 } 184 185 bool hasAddr64() const { 186 return (getGeneration() < VOLCANIC_ISLANDS); 187 } 188 189 bool hasBFE() const { 190 return (getGeneration() >= EVERGREEN); 191 } 192 193 bool hasBFI() const { 194 return (getGeneration() >= EVERGREEN); 195 } 196 197 bool hasBFM() const { 198 return hasBFE(); 199 } 200 201 bool hasBCNT(unsigned Size) const { 202 if (Size == 32) 203 return (getGeneration() >= EVERGREEN); 204 205 if (Size == 64) 206 return (getGeneration() >= SOUTHERN_ISLANDS); 207 208 return false; 209 } 210 211 bool hasMulU24() const { 212 return (getGeneration() >= EVERGREEN); 213 } 214 215 bool hasMulI24() const { 216 return (getGeneration() >= SOUTHERN_ISLANDS || 217 hasCaymanISA()); 218 } 219 220 bool hasFFBL() const { 221 return (getGeneration() >= EVERGREEN); 222 } 223 224 bool hasFFBH() const { 225 return (getGeneration() >= EVERGREEN); 226 } 227 228 bool hasCARRY() const { 229 return (getGeneration() >= EVERGREEN); 230 } 231 232 bool hasBORROW() const { 233 return (getGeneration() >= EVERGREEN); 234 } 235 236 bool hasCaymanISA() const { 237 return CaymanISA; 238 } 239 240 bool isPromoteAllocaEnabled() const { 241 return EnablePromoteAlloca; 242 } 243 244 bool unsafeDSOffsetFoldingEnabled() const { 245 return EnableUnsafeDSOffsetFolding; 246 } 247 248 bool dumpCode() const { 249 return DumpCode; 250 } 251 252 /// Return the amount of LDS that can be used that will not restrict the 253 /// occupancy lower than WaveCount. 254 unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount) const; 255 256 /// Inverse of getMaxLocalMemWithWaveCount. Return the maximum wavecount if 257 /// the given LDS memory size is the only constraint. 258 unsigned getOccupancyWithLocalMemSize(uint32_t Bytes) const; 259 260 261 bool hasFP32Denormals() const { 262 return FP32Denormals; 263 } 264 265 bool hasFP64Denormals() const { 266 return FP64Denormals; 267 } 268 269 bool hasFPExceptions() const { 270 return FPExceptions; 271 } 272 273 bool useFlatForGlobal() const { 274 return FlatForGlobal; 275 } 276 277 bool hasUnalignedBufferAccess() const { 278 return UnalignedBufferAccess; 279 } 280 281 bool hasUnalignedScratchAccess() const { 282 return UnalignedScratchAccess; 283 } 284 285 bool isXNACKEnabled() const { 286 return EnableXNACK; 287 } 288 289 bool isAmdCodeObjectV2() const { 290 return isAmdHsaOS() || isMesa3DOS(); 291 } 292 293 /// \brief Returns the offset in bytes from the start of the input buffer 294 /// of the first explicit kernel argument. 295 unsigned getExplicitKernelArgOffset() const { 296 return isAmdCodeObjectV2() ? 0 : 36; 297 } 298 299 unsigned getAlignmentForImplicitArgPtr() const { 300 return isAmdHsaOS() ? 8 : 4; 301 } 302 303 unsigned getImplicitArgNumBytes() const { 304 if (isMesa3DOS()) 305 return 16; 306 if (isAmdHsaOS() && isOpenCLEnv()) 307 return 32; 308 return 0; 309 } 310 311 unsigned getStackAlignment() const { 312 // Scratch is allocated in 256 dword per wave blocks. 313 return 4 * 256 / getWavefrontSize(); 314 } 315 316 bool enableMachineScheduler() const override { 317 return true; 318 } 319 320 bool enableSubRegLiveness() const override { 321 return true; 322 } 323 324 /// \returns Number of execution units per compute unit supported by the 325 /// subtarget. 326 unsigned getEUsPerCU() const { 327 return 4; 328 } 329 330 /// \returns Maximum number of work groups per compute unit supported by the 331 /// subtarget and limited by given flat work group size. 332 unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const { 333 if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS) 334 return 8; 335 return getWavesPerWorkGroup(FlatWorkGroupSize) == 1 ? 40 : 16; 336 } 337 338 /// \returns Maximum number of waves per compute unit supported by the 339 /// subtarget without any kind of limitation. 340 unsigned getMaxWavesPerCU() const { 341 return getMaxWavesPerEU() * getEUsPerCU(); 342 } 343 344 /// \returns Maximum number of waves per compute unit supported by the 345 /// subtarget and limited by given flat work group size. 346 unsigned getMaxWavesPerCU(unsigned FlatWorkGroupSize) const { 347 return getWavesPerWorkGroup(FlatWorkGroupSize); 348 } 349 350 /// \returns Minimum number of waves per execution unit supported by the 351 /// subtarget. 352 unsigned getMinWavesPerEU() const { 353 return 1; 354 } 355 356 /// \returns Maximum number of waves per execution unit supported by the 357 /// subtarget without any kind of limitation. 358 unsigned getMaxWavesPerEU() const { 359 if (getGeneration() < AMDGPUSubtarget::SOUTHERN_ISLANDS) 360 return 8; 361 // FIXME: Need to take scratch memory into account. 362 return 10; 363 } 364 365 /// \returns Maximum number of waves per execution unit supported by the 366 /// subtarget and limited by given flat work group size. 367 unsigned getMaxWavesPerEU(unsigned FlatWorkGroupSize) const { 368 return alignTo(getMaxWavesPerCU(FlatWorkGroupSize), getEUsPerCU()) / 369 getEUsPerCU(); 370 } 371 372 /// \returns Minimum flat work group size supported by the subtarget. 373 unsigned getMinFlatWorkGroupSize() const { 374 return 1; 375 } 376 377 /// \returns Maximum flat work group size supported by the subtarget. 378 unsigned getMaxFlatWorkGroupSize() const { 379 return 2048; 380 } 381 382 /// \returns Number of waves per work group given the flat work group size. 383 unsigned getWavesPerWorkGroup(unsigned FlatWorkGroupSize) const { 384 return alignTo(FlatWorkGroupSize, getWavefrontSize()) / getWavefrontSize(); 385 } 386 387 /// \returns Subtarget's default pair of minimum/maximum flat work group sizes 388 /// for function \p F, or minimum/maximum flat work group sizes explicitly 389 /// requested using "amdgpu-flat-work-group-size" attribute attached to 390 /// function \p F. 391 /// 392 /// \returns Subtarget's default values if explicitly requested values cannot 393 /// be converted to integer, or violate subtarget's specifications. 394 std::pair<unsigned, unsigned> getFlatWorkGroupSizes(const Function &F) const; 395 396 /// \returns Subtarget's default pair of minimum/maximum number of waves per 397 /// execution unit for function \p F, or minimum/maximum number of waves per 398 /// execution unit explicitly requested using "amdgpu-waves-per-eu" attribute 399 /// attached to function \p F. 400 /// 401 /// \returns Subtarget's default values if explicitly requested values cannot 402 /// be converted to integer, violate subtarget's specifications, or are not 403 /// compatible with minimum/maximum number of waves limited by flat work group 404 /// size, register usage, and/or lds usage. 405 std::pair<unsigned, unsigned> getWavesPerEU(const Function &F) const; 406 }; 407 408 class R600Subtarget final : public AMDGPUSubtarget { 409 private: 410 R600InstrInfo InstrInfo; 411 R600FrameLowering FrameLowering; 412 R600TargetLowering TLInfo; 413 414 public: 415 R600Subtarget(const Triple &TT, StringRef CPU, StringRef FS, 416 const TargetMachine &TM); 417 418 const R600InstrInfo *getInstrInfo() const override { 419 return &InstrInfo; 420 } 421 422 const R600FrameLowering *getFrameLowering() const override { 423 return &FrameLowering; 424 } 425 426 const R600TargetLowering *getTargetLowering() const override { 427 return &TLInfo; 428 } 429 430 const R600RegisterInfo *getRegisterInfo() const override { 431 return &InstrInfo.getRegisterInfo(); 432 } 433 434 bool hasCFAluBug() const { 435 return CFALUBug; 436 } 437 438 bool hasVertexCache() const { 439 return HasVertexCache; 440 } 441 442 short getTexVTXClauseSize() const { 443 return TexVTXClauseSize; 444 } 445 }; 446 447 class SISubtarget final : public AMDGPUSubtarget { 448 public: 449 enum { 450 // The closed Vulkan driver sets 96, which limits the wave count to 8 but 451 // doesn't spill SGPRs as much as when 80 is set. 452 FIXED_SGPR_COUNT_FOR_INIT_BUG = 96 453 }; 454 455 private: 456 SIInstrInfo InstrInfo; 457 SIFrameLowering FrameLowering; 458 SITargetLowering TLInfo; 459 std::unique_ptr<GISelAccessor> GISel; 460 461 public: 462 SISubtarget(const Triple &TT, StringRef CPU, StringRef FS, 463 const TargetMachine &TM); 464 465 const SIInstrInfo *getInstrInfo() const override { 466 return &InstrInfo; 467 } 468 469 const SIFrameLowering *getFrameLowering() const override { 470 return &FrameLowering; 471 } 472 473 const SITargetLowering *getTargetLowering() const override { 474 return &TLInfo; 475 } 476 477 const CallLowering *getCallLowering() const override { 478 assert(GISel && "Access to GlobalISel APIs not set"); 479 return GISel->getCallLowering(); 480 } 481 482 const SIRegisterInfo *getRegisterInfo() const override { 483 return &InstrInfo.getRegisterInfo(); 484 } 485 486 void setGISelAccessor(GISelAccessor &GISel) { 487 this->GISel.reset(&GISel); 488 } 489 490 void overrideSchedPolicy(MachineSchedPolicy &Policy, 491 unsigned NumRegionInstrs) const override; 492 493 bool isVGPRSpillingEnabled(const Function& F) const; 494 495 unsigned getMaxNumUserSGPRs() const { 496 return 16; 497 } 498 499 bool hasFlatAddressSpace() const { 500 return FlatAddressSpace; 501 } 502 503 bool hasSMemRealTime() const { 504 return HasSMemRealTime; 505 } 506 507 bool has16BitInsts() const { 508 return Has16BitInsts; 509 } 510 511 bool hasMovrel() const { 512 return HasMovrel; 513 } 514 515 bool hasVGPRIndexMode() const { 516 return HasVGPRIndexMode; 517 } 518 519 bool hasScalarCompareEq64() const { 520 return getGeneration() >= VOLCANIC_ISLANDS; 521 } 522 523 bool enableSIScheduler() const { 524 return EnableSIScheduler; 525 } 526 527 bool debuggerSupported() const { 528 return debuggerInsertNops() && debuggerReserveRegs() && 529 debuggerEmitPrologue(); 530 } 531 532 bool debuggerInsertNops() const { 533 return DebuggerInsertNops; 534 } 535 536 bool debuggerReserveRegs() const { 537 return DebuggerReserveRegs; 538 } 539 540 bool debuggerEmitPrologue() const { 541 return DebuggerEmitPrologue; 542 } 543 544 bool loadStoreOptEnabled() const { 545 return EnableLoadStoreOpt; 546 } 547 548 bool hasSGPRInitBug() const { 549 return SGPRInitBug; 550 } 551 552 unsigned getKernArgSegmentSize(unsigned ExplictArgBytes) const; 553 554 /// Return the maximum number of waves per SIMD for kernels using \p SGPRs SGPRs 555 unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const; 556 557 /// Return the maximum number of waves per SIMD for kernels using \p VGPRs VGPRs 558 unsigned getOccupancyWithNumVGPRs(unsigned VGPRs) const; 559 560 /// \returns True if waitcnt instruction is needed before barrier instruction, 561 /// false otherwise. 562 bool needWaitcntBeforeBarrier() const { 563 return true; 564 } 565 }; 566 567 } // End namespace llvm 568 569 #endif 570