1 //=====-- AMDGPUSubtarget.h - Define Subtarget for AMDGPU ------*- C++ -*-====// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //==-----------------------------------------------------------------------===// 8 // 9 /// \file 10 /// AMDGPU specific subclass of TargetSubtarget. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H 15 #define LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H 16 17 #include "AMDGPU.h" 18 #include "AMDGPUCallLowering.h" 19 #include "MCTargetDesc/AMDGPUMCTargetDesc.h" 20 #include "R600FrameLowering.h" 21 #include "R600ISelLowering.h" 22 #include "R600InstrInfo.h" 23 #include "SIFrameLowering.h" 24 #include "SIISelLowering.h" 25 #include "SIInstrInfo.h" 26 #include "Utils/AMDGPUBaseInfo.h" 27 #include "llvm/ADT/Triple.h" 28 #include "llvm/CodeGen/GlobalISel/InlineAsmLowering.h" 29 #include "llvm/CodeGen/GlobalISel/InstructionSelector.h" 30 #include "llvm/CodeGen/GlobalISel/LegalizerInfo.h" 31 #include "llvm/CodeGen/GlobalISel/RegisterBankInfo.h" 32 #include "llvm/CodeGen/MachineFunction.h" 33 #include "llvm/CodeGen/SelectionDAGTargetInfo.h" 34 #include "llvm/MC/MCInstrItineraries.h" 35 #include "llvm/Support/MathExtras.h" 36 #include <cassert> 37 #include <cstdint> 38 #include <memory> 39 #include <utility> 40 41 #define GET_SUBTARGETINFO_HEADER 42 #include "AMDGPUGenSubtargetInfo.inc" 43 #define GET_SUBTARGETINFO_HEADER 44 #include "R600GenSubtargetInfo.inc" 45 46 namespace llvm { 47 48 class StringRef; 49 50 class AMDGPUSubtarget { 51 public: 52 enum Generation { 53 R600 = 0, 54 R700 = 1, 55 EVERGREEN = 2, 56 NORTHERN_ISLANDS = 3, 57 SOUTHERN_ISLANDS = 4, 58 SEA_ISLANDS = 5, 59 VOLCANIC_ISLANDS = 6, 60 GFX9 = 7, 61 GFX10 = 8 62 }; 63 64 private: 65 Triple TargetTriple; 66 67 protected: 68 bool Has16BitInsts; 69 bool HasMadMixInsts; 70 bool HasMadMacF32Insts; 71 bool HasDsSrc2Insts; 72 bool HasSDWA; 73 bool HasVOP3PInsts; 74 bool HasMulI24; 75 bool HasMulU24; 76 bool HasInv2PiInlineImm; 77 bool HasFminFmaxLegacy; 78 bool EnablePromoteAlloca; 79 bool HasTrigReducedRange; 80 unsigned MaxWavesPerEU; 81 unsigned LocalMemorySize; 82 char WavefrontSizeLog2; 83 84 public: 85 AMDGPUSubtarget(const Triple &TT); 86 87 static const AMDGPUSubtarget &get(const MachineFunction &MF); 88 static const AMDGPUSubtarget &get(const TargetMachine &TM, 89 const Function &F); 90 91 /// \returns Default range flat work group size for a calling convention. 92 std::pair<unsigned, unsigned> getDefaultFlatWorkGroupSize(CallingConv::ID CC) const; 93 94 /// \returns Subtarget's default pair of minimum/maximum flat work group sizes 95 /// for function \p F, or minimum/maximum flat work group sizes explicitly 96 /// requested using "amdgpu-flat-work-group-size" attribute attached to 97 /// function \p F. 98 /// 99 /// \returns Subtarget's default values if explicitly requested values cannot 100 /// be converted to integer, or violate subtarget's specifications. 101 std::pair<unsigned, unsigned> getFlatWorkGroupSizes(const Function &F) const; 102 103 /// \returns Subtarget's default pair of minimum/maximum number of waves per 104 /// execution unit for function \p F, or minimum/maximum number of waves per 105 /// execution unit explicitly requested using "amdgpu-waves-per-eu" attribute 106 /// attached to function \p F. 107 /// 108 /// \returns Subtarget's default values if explicitly requested values cannot 109 /// be converted to integer, violate subtarget's specifications, or are not 110 /// compatible with minimum/maximum number of waves limited by flat work group 111 /// size, register usage, and/or lds usage. 112 std::pair<unsigned, unsigned> getWavesPerEU(const Function &F) const; 113 114 /// Return the amount of LDS that can be used that will not restrict the 115 /// occupancy lower than WaveCount. 116 unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount, 117 const Function &) const; 118 119 /// Inverse of getMaxLocalMemWithWaveCount. Return the maximum wavecount if 120 /// the given LDS memory size is the only constraint. 121 unsigned getOccupancyWithLocalMemSize(uint32_t Bytes, const Function &) const; 122 123 unsigned getOccupancyWithLocalMemSize(const MachineFunction &MF) const; 124 125 bool isAmdHsaOS() const { 126 return TargetTriple.getOS() == Triple::AMDHSA; 127 } 128 129 bool isAmdPalOS() const { 130 return TargetTriple.getOS() == Triple::AMDPAL; 131 } 132 133 bool isMesa3DOS() const { 134 return TargetTriple.getOS() == Triple::Mesa3D; 135 } 136 137 bool isMesaKernel(const Function &F) const { 138 return isMesa3DOS() && !AMDGPU::isShader(F.getCallingConv()); 139 } 140 141 bool isAmdHsaOrMesa(const Function &F) const { 142 return isAmdHsaOS() || isMesaKernel(F); 143 } 144 145 bool isGCN() const { 146 return TargetTriple.getArch() == Triple::amdgcn; 147 } 148 149 bool has16BitInsts() const { 150 return Has16BitInsts; 151 } 152 153 bool hasMadMixInsts() const { 154 return HasMadMixInsts; 155 } 156 157 bool hasMadMacF32Insts() const { 158 return HasMadMacF32Insts || !isGCN(); 159 } 160 161 bool hasDsSrc2Insts() const { 162 return HasDsSrc2Insts; 163 } 164 165 bool hasSDWA() const { 166 return HasSDWA; 167 } 168 169 bool hasVOP3PInsts() const { 170 return HasVOP3PInsts; 171 } 172 173 bool hasMulI24() const { 174 return HasMulI24; 175 } 176 177 bool hasMulU24() const { 178 return HasMulU24; 179 } 180 181 bool hasInv2PiInlineImm() const { 182 return HasInv2PiInlineImm; 183 } 184 185 bool hasFminFmaxLegacy() const { 186 return HasFminFmaxLegacy; 187 } 188 189 bool hasTrigReducedRange() const { 190 return HasTrigReducedRange; 191 } 192 193 bool isPromoteAllocaEnabled() const { 194 return EnablePromoteAlloca; 195 } 196 197 unsigned getWavefrontSize() const { 198 return 1 << WavefrontSizeLog2; 199 } 200 201 unsigned getWavefrontSizeLog2() const { 202 return WavefrontSizeLog2; 203 } 204 205 unsigned getLocalMemorySize() const { 206 return LocalMemorySize; 207 } 208 209 Align getAlignmentForImplicitArgPtr() const { 210 return isAmdHsaOS() ? Align(8) : Align(4); 211 } 212 213 /// Returns the offset in bytes from the start of the input buffer 214 /// of the first explicit kernel argument. 215 unsigned getExplicitKernelArgOffset(const Function &F) const { 216 return isAmdHsaOrMesa(F) ? 0 : 36; 217 } 218 219 /// \returns Maximum number of work groups per compute unit supported by the 220 /// subtarget and limited by given \p FlatWorkGroupSize. 221 virtual unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const = 0; 222 223 /// \returns Minimum flat work group size supported by the subtarget. 224 virtual unsigned getMinFlatWorkGroupSize() const = 0; 225 226 /// \returns Maximum flat work group size supported by the subtarget. 227 virtual unsigned getMaxFlatWorkGroupSize() const = 0; 228 229 /// \returns Number of waves per execution unit required to support the given 230 /// \p FlatWorkGroupSize. 231 virtual unsigned 232 getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const = 0; 233 234 /// \returns Minimum number of waves per execution unit supported by the 235 /// subtarget. 236 virtual unsigned getMinWavesPerEU() const = 0; 237 238 /// \returns Maximum number of waves per execution unit supported by the 239 /// subtarget without any kind of limitation. 240 unsigned getMaxWavesPerEU() const { return MaxWavesPerEU; } 241 242 /// Return the maximum workitem ID value in the function, for the given (0, 1, 243 /// 2) dimension. 244 unsigned getMaxWorkitemID(const Function &Kernel, unsigned Dimension) const; 245 246 /// Creates value range metadata on an workitemid.* intrinsic call or load. 247 bool makeLIDRangeMetadata(Instruction *I) const; 248 249 /// \returns Number of bytes of arguments that are passed to a shader or 250 /// kernel in addition to the explicit ones declared for the function. 251 unsigned getImplicitArgNumBytes(const Function &F) const { 252 if (isMesaKernel(F)) 253 return 16; 254 return AMDGPU::getIntegerAttribute(F, "amdgpu-implicitarg-num-bytes", 0); 255 } 256 uint64_t getExplicitKernArgSize(const Function &F, Align &MaxAlign) const; 257 unsigned getKernArgSegmentSize(const Function &F, Align &MaxAlign) const; 258 259 /// \returns Corresponsing DWARF register number mapping flavour for the 260 /// \p WavefrontSize. 261 AMDGPUDwarfFlavour getAMDGPUDwarfFlavour() const { 262 return getWavefrontSize() == 32 ? AMDGPUDwarfFlavour::Wave32 263 : AMDGPUDwarfFlavour::Wave64; 264 } 265 266 virtual ~AMDGPUSubtarget() {} 267 }; 268 269 class GCNSubtarget : public AMDGPUGenSubtargetInfo, 270 public AMDGPUSubtarget { 271 272 using AMDGPUSubtarget::getMaxWavesPerEU; 273 274 public: 275 enum TrapHandlerAbi { 276 TrapHandlerAbiNone = 0, 277 TrapHandlerAbiHsa = 1 278 }; 279 280 enum TrapID { 281 TrapIDHardwareReserved = 0, 282 TrapIDHSADebugTrap = 1, 283 TrapIDLLVMTrap = 2, 284 TrapIDLLVMDebugTrap = 3, 285 TrapIDDebugBreakpoint = 7, 286 TrapIDDebugReserved8 = 8, 287 TrapIDDebugReservedFE = 0xfe, 288 TrapIDDebugReservedFF = 0xff 289 }; 290 291 enum TrapRegValues { 292 LLVMTrapHandlerRegValue = 1 293 }; 294 295 private: 296 /// GlobalISel related APIs. 297 std::unique_ptr<AMDGPUCallLowering> CallLoweringInfo; 298 std::unique_ptr<InlineAsmLowering> InlineAsmLoweringInfo; 299 std::unique_ptr<InstructionSelector> InstSelector; 300 std::unique_ptr<LegalizerInfo> Legalizer; 301 std::unique_ptr<RegisterBankInfo> RegBankInfo; 302 303 protected: 304 // Basic subtarget description. 305 Triple TargetTriple; 306 unsigned Gen; 307 InstrItineraryData InstrItins; 308 int LDSBankCount; 309 unsigned MaxPrivateElementSize; 310 311 // Possibly statically set by tablegen, but may want to be overridden. 312 bool FastFMAF32; 313 bool FastDenormalF32; 314 bool HalfRate64Ops; 315 316 // Dynamially set bits that enable features. 317 bool FlatForGlobal; 318 bool AutoWaitcntBeforeBarrier; 319 bool UnalignedScratchAccess; 320 bool UnalignedBufferAccess; 321 bool UnalignedAccessMode; 322 bool HasApertureRegs; 323 bool EnableXNACK; 324 bool DoesNotSupportXNACK; 325 bool EnableCuMode; 326 bool TrapHandler; 327 328 // Used as options. 329 bool EnableLoadStoreOpt; 330 bool EnableUnsafeDSOffsetFolding; 331 bool EnableSIScheduler; 332 bool EnableDS128; 333 bool EnablePRTStrictNull; 334 bool DumpCode; 335 336 // Subtarget statically properties set by tablegen 337 bool FP64; 338 bool FMA; 339 bool MIMG_R128; 340 bool IsGCN; 341 bool GCN3Encoding; 342 bool CIInsts; 343 bool GFX8Insts; 344 bool GFX9Insts; 345 bool GFX10Insts; 346 bool GFX10_3Insts; 347 bool GFX7GFX8GFX9Insts; 348 bool SGPRInitBug; 349 bool HasSMemRealTime; 350 bool HasIntClamp; 351 bool HasFmaMixInsts; 352 bool HasMovrel; 353 bool HasVGPRIndexMode; 354 bool HasScalarStores; 355 bool HasScalarAtomics; 356 bool HasSDWAOmod; 357 bool HasSDWAScalar; 358 bool HasSDWASdst; 359 bool HasSDWAMac; 360 bool HasSDWAOutModsVOPC; 361 bool HasDPP; 362 bool HasDPP8; 363 bool HasR128A16; 364 bool HasGFX10A16; 365 bool HasG16; 366 bool HasNSAEncoding; 367 bool GFX10_BEncoding; 368 bool HasDLInsts; 369 bool HasDot1Insts; 370 bool HasDot2Insts; 371 bool HasDot3Insts; 372 bool HasDot4Insts; 373 bool HasDot5Insts; 374 bool HasDot6Insts; 375 bool HasMAIInsts; 376 bool HasPkFmacF16Inst; 377 bool HasAtomicFaddInsts; 378 bool EnableSRAMECC; 379 bool DoesNotSupportSRAMECC; 380 bool HasNoSdstCMPX; 381 bool HasVscnt; 382 bool HasGetWaveIdInst; 383 bool HasSMemTimeInst; 384 bool HasRegisterBanking; 385 bool HasVOP3Literal; 386 bool HasNoDataDepHazard; 387 bool FlatAddressSpace; 388 bool FlatInstOffsets; 389 bool FlatGlobalInsts; 390 bool FlatScratchInsts; 391 bool ScalarFlatScratchInsts; 392 bool AddNoCarryInsts; 393 bool HasUnpackedD16VMem; 394 bool R600ALUInst; 395 bool CaymanISA; 396 bool CFALUBug; 397 bool LDSMisalignedBug; 398 bool HasMFMAInlineLiteralBug; 399 bool HasVertexCache; 400 short TexVTXClauseSize; 401 bool UnalignedDSAccess; 402 bool ScalarizeGlobal; 403 404 bool HasVcmpxPermlaneHazard; 405 bool HasVMEMtoScalarWriteHazard; 406 bool HasSMEMtoVectorWriteHazard; 407 bool HasInstFwdPrefetchBug; 408 bool HasVcmpxExecWARHazard; 409 bool HasLdsBranchVmemWARHazard; 410 bool HasNSAtoVMEMBug; 411 bool HasOffset3fBug; 412 bool HasFlatSegmentOffsetBug; 413 bool HasImageStoreD16Bug; 414 bool HasImageGather4D16Bug; 415 416 // Dummy feature to use for assembler in tablegen. 417 bool FeatureDisable; 418 419 SelectionDAGTargetInfo TSInfo; 420 private: 421 SIInstrInfo InstrInfo; 422 SITargetLowering TLInfo; 423 SIFrameLowering FrameLowering; 424 425 // See COMPUTE_TMPRING_SIZE.WAVESIZE, 13-bit field in units of 256-dword. 426 static const unsigned MaxWaveScratchSize = (256 * 4) * ((1 << 13) - 1); 427 428 public: 429 GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS, 430 const GCNTargetMachine &TM); 431 ~GCNSubtarget() override; 432 433 GCNSubtarget &initializeSubtargetDependencies(const Triple &TT, 434 StringRef GPU, StringRef FS); 435 436 const SIInstrInfo *getInstrInfo() const override { 437 return &InstrInfo; 438 } 439 440 const SIFrameLowering *getFrameLowering() const override { 441 return &FrameLowering; 442 } 443 444 const SITargetLowering *getTargetLowering() const override { 445 return &TLInfo; 446 } 447 448 const SIRegisterInfo *getRegisterInfo() const override { 449 return &InstrInfo.getRegisterInfo(); 450 } 451 452 const CallLowering *getCallLowering() const override { 453 return CallLoweringInfo.get(); 454 } 455 456 const InlineAsmLowering *getInlineAsmLowering() const override { 457 return InlineAsmLoweringInfo.get(); 458 } 459 460 InstructionSelector *getInstructionSelector() const override { 461 return InstSelector.get(); 462 } 463 464 const LegalizerInfo *getLegalizerInfo() const override { 465 return Legalizer.get(); 466 } 467 468 const RegisterBankInfo *getRegBankInfo() const override { 469 return RegBankInfo.get(); 470 } 471 472 // Nothing implemented, just prevent crashes on use. 473 const SelectionDAGTargetInfo *getSelectionDAGInfo() const override { 474 return &TSInfo; 475 } 476 477 const InstrItineraryData *getInstrItineraryData() const override { 478 return &InstrItins; 479 } 480 481 void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS); 482 483 Generation getGeneration() const { 484 return (Generation)Gen; 485 } 486 487 /// Return the number of high bits known to be zero fror a frame index. 488 unsigned getKnownHighZeroBitsForFrameIndex() const { 489 return countLeadingZeros(MaxWaveScratchSize) + getWavefrontSizeLog2(); 490 } 491 492 int getLDSBankCount() const { 493 return LDSBankCount; 494 } 495 496 unsigned getMaxPrivateElementSize() const { 497 return MaxPrivateElementSize; 498 } 499 500 unsigned getConstantBusLimit(unsigned Opcode) const; 501 502 bool hasIntClamp() const { 503 return HasIntClamp; 504 } 505 506 bool hasFP64() const { 507 return FP64; 508 } 509 510 bool hasMIMG_R128() const { 511 return MIMG_R128; 512 } 513 514 bool hasHWFP64() const { 515 return FP64; 516 } 517 518 bool hasFastFMAF32() const { 519 return FastFMAF32; 520 } 521 522 bool hasHalfRate64Ops() const { 523 return HalfRate64Ops; 524 } 525 526 bool hasAddr64() const { 527 return (getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS); 528 } 529 530 // Return true if the target only has the reverse operand versions of VALU 531 // shift instructions (e.g. v_lshrrev_b32, and no v_lshr_b32). 532 bool hasOnlyRevVALUShifts() const { 533 return getGeneration() >= VOLCANIC_ISLANDS; 534 } 535 536 bool hasFractBug() const { 537 return getGeneration() == SOUTHERN_ISLANDS; 538 } 539 540 bool hasBFE() const { 541 return true; 542 } 543 544 bool hasBFI() const { 545 return true; 546 } 547 548 bool hasBFM() const { 549 return hasBFE(); 550 } 551 552 bool hasBCNT(unsigned Size) const { 553 return true; 554 } 555 556 bool hasFFBL() const { 557 return true; 558 } 559 560 bool hasFFBH() const { 561 return true; 562 } 563 564 bool hasMed3_16() const { 565 return getGeneration() >= AMDGPUSubtarget::GFX9; 566 } 567 568 bool hasMin3Max3_16() const { 569 return getGeneration() >= AMDGPUSubtarget::GFX9; 570 } 571 572 bool hasFmaMixInsts() const { 573 return HasFmaMixInsts; 574 } 575 576 bool hasCARRY() const { 577 return true; 578 } 579 580 bool hasFMA() const { 581 return FMA; 582 } 583 584 bool hasSwap() const { 585 return GFX9Insts; 586 } 587 588 bool hasScalarPackInsts() const { 589 return GFX9Insts; 590 } 591 592 bool hasScalarMulHiInsts() const { 593 return GFX9Insts; 594 } 595 596 TrapHandlerAbi getTrapHandlerAbi() const { 597 return isAmdHsaOS() ? TrapHandlerAbiHsa : TrapHandlerAbiNone; 598 } 599 600 /// True if the offset field of DS instructions works as expected. On SI, the 601 /// offset uses a 16-bit adder and does not always wrap properly. 602 bool hasUsableDSOffset() const { 603 return getGeneration() >= SEA_ISLANDS; 604 } 605 606 bool unsafeDSOffsetFoldingEnabled() const { 607 return EnableUnsafeDSOffsetFolding; 608 } 609 610 /// Condition output from div_scale is usable. 611 bool hasUsableDivScaleConditionOutput() const { 612 return getGeneration() != SOUTHERN_ISLANDS; 613 } 614 615 /// Extra wait hazard is needed in some cases before 616 /// s_cbranch_vccnz/s_cbranch_vccz. 617 bool hasReadVCCZBug() const { 618 return getGeneration() <= SEA_ISLANDS; 619 } 620 621 /// Writes to VCC_LO/VCC_HI update the VCCZ flag. 622 bool partialVCCWritesUpdateVCCZ() const { 623 return getGeneration() >= GFX10; 624 } 625 626 /// A read of an SGPR by SMRD instruction requires 4 wait states when the SGPR 627 /// was written by a VALU instruction. 628 bool hasSMRDReadVALUDefHazard() const { 629 return getGeneration() == SOUTHERN_ISLANDS; 630 } 631 632 /// A read of an SGPR by a VMEM instruction requires 5 wait states when the 633 /// SGPR was written by a VALU Instruction. 634 bool hasVMEMReadSGPRVALUDefHazard() const { 635 return getGeneration() >= VOLCANIC_ISLANDS; 636 } 637 638 bool hasRFEHazards() const { 639 return getGeneration() >= VOLCANIC_ISLANDS; 640 } 641 642 /// Number of hazard wait states for s_setreg_b32/s_setreg_imm32_b32. 643 unsigned getSetRegWaitStates() const { 644 return getGeneration() <= SEA_ISLANDS ? 1 : 2; 645 } 646 647 bool dumpCode() const { 648 return DumpCode; 649 } 650 651 /// Return the amount of LDS that can be used that will not restrict the 652 /// occupancy lower than WaveCount. 653 unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount, 654 const Function &) const; 655 656 bool supportsMinMaxDenormModes() const { 657 return getGeneration() >= AMDGPUSubtarget::GFX9; 658 } 659 660 /// \returns If target supports S_DENORM_MODE. 661 bool hasDenormModeInst() const { 662 return getGeneration() >= AMDGPUSubtarget::GFX10; 663 } 664 665 bool useFlatForGlobal() const { 666 return FlatForGlobal; 667 } 668 669 /// \returns If target supports ds_read/write_b128 and user enables generation 670 /// of ds_read/write_b128. 671 bool useDS128() const { 672 return CIInsts && EnableDS128; 673 } 674 675 /// \return If target supports ds_read/write_b96/128. 676 bool hasDS96AndDS128() const { 677 return CIInsts; 678 } 679 680 /// Have v_trunc_f64, v_ceil_f64, v_rndne_f64 681 bool haveRoundOpsF64() const { 682 return CIInsts; 683 } 684 685 /// \returns If MUBUF instructions always perform range checking, even for 686 /// buffer resources used for private memory access. 687 bool privateMemoryResourceIsRangeChecked() const { 688 return getGeneration() < AMDGPUSubtarget::GFX9; 689 } 690 691 /// \returns If target requires PRT Struct NULL support (zero result registers 692 /// for sparse texture support). 693 bool usePRTStrictNull() const { 694 return EnablePRTStrictNull; 695 } 696 697 bool hasAutoWaitcntBeforeBarrier() const { 698 return AutoWaitcntBeforeBarrier; 699 } 700 701 bool hasUnalignedBufferAccess() const { 702 return UnalignedBufferAccess; 703 } 704 705 bool hasUnalignedScratchAccess() const { 706 return UnalignedScratchAccess; 707 } 708 709 bool hasUnalignedAccessMode() const { 710 return UnalignedAccessMode; 711 } 712 713 bool hasUnalignedDSAccess() const { 714 return UnalignedDSAccess; 715 } 716 717 bool hasApertureRegs() const { 718 return HasApertureRegs; 719 } 720 721 bool isTrapHandlerEnabled() const { 722 return TrapHandler; 723 } 724 725 bool isXNACKEnabled() const { 726 return EnableXNACK; 727 } 728 729 bool isCuModeEnabled() const { 730 return EnableCuMode; 731 } 732 733 bool hasFlatAddressSpace() const { 734 return FlatAddressSpace; 735 } 736 737 bool hasFlatScrRegister() const { 738 return hasFlatAddressSpace(); 739 } 740 741 bool hasFlatInstOffsets() const { 742 return FlatInstOffsets; 743 } 744 745 bool hasFlatGlobalInsts() const { 746 return FlatGlobalInsts; 747 } 748 749 bool hasFlatScratchInsts() const { 750 return FlatScratchInsts; 751 } 752 753 bool hasScalarFlatScratchInsts() const { 754 return ScalarFlatScratchInsts; 755 } 756 757 bool hasGlobalAddTidInsts() const { 758 return GFX10_BEncoding; 759 } 760 761 bool hasAtomicCSub() const { 762 return GFX10_BEncoding; 763 } 764 765 bool hasMultiDwordFlatScratchAddressing() const { 766 return getGeneration() >= GFX9; 767 } 768 769 bool hasFlatSegmentOffsetBug() const { 770 return HasFlatSegmentOffsetBug; 771 } 772 773 bool hasFlatLgkmVMemCountInOrder() const { 774 return getGeneration() > GFX9; 775 } 776 777 bool hasD16LoadStore() const { 778 return getGeneration() >= GFX9; 779 } 780 781 bool d16PreservesUnusedBits() const { 782 return hasD16LoadStore() && !isSRAMECCEnabled(); 783 } 784 785 bool hasD16Images() const { 786 return getGeneration() >= VOLCANIC_ISLANDS; 787 } 788 789 /// Return if most LDS instructions have an m0 use that require m0 to be 790 /// iniitalized. 791 bool ldsRequiresM0Init() const { 792 return getGeneration() < GFX9; 793 } 794 795 // True if the hardware rewinds and replays GWS operations if a wave is 796 // preempted. 797 // 798 // If this is false, a GWS operation requires testing if a nack set the 799 // MEM_VIOL bit, and repeating if so. 800 bool hasGWSAutoReplay() const { 801 return getGeneration() >= GFX9; 802 } 803 804 /// \returns if target has ds_gws_sema_release_all instruction. 805 bool hasGWSSemaReleaseAll() const { 806 return CIInsts; 807 } 808 809 /// \returns true if the target has integer add/sub instructions that do not 810 /// produce a carry-out. This includes v_add_[iu]32, v_sub_[iu]32, 811 /// v_add_[iu]16, and v_sub_[iu]16, all of which support the clamp modifier 812 /// for saturation. 813 bool hasAddNoCarry() const { 814 return AddNoCarryInsts; 815 } 816 817 bool hasUnpackedD16VMem() const { 818 return HasUnpackedD16VMem; 819 } 820 821 // Covers VS/PS/CS graphics shaders 822 bool isMesaGfxShader(const Function &F) const { 823 return isMesa3DOS() && AMDGPU::isShader(F.getCallingConv()); 824 } 825 826 bool hasMad64_32() const { 827 return getGeneration() >= SEA_ISLANDS; 828 } 829 830 bool hasSDWAOmod() const { 831 return HasSDWAOmod; 832 } 833 834 bool hasSDWAScalar() const { 835 return HasSDWAScalar; 836 } 837 838 bool hasSDWASdst() const { 839 return HasSDWASdst; 840 } 841 842 bool hasSDWAMac() const { 843 return HasSDWAMac; 844 } 845 846 bool hasSDWAOutModsVOPC() const { 847 return HasSDWAOutModsVOPC; 848 } 849 850 bool hasDLInsts() const { 851 return HasDLInsts; 852 } 853 854 bool hasDot1Insts() const { 855 return HasDot1Insts; 856 } 857 858 bool hasDot2Insts() const { 859 return HasDot2Insts; 860 } 861 862 bool hasDot3Insts() const { 863 return HasDot3Insts; 864 } 865 866 bool hasDot4Insts() const { 867 return HasDot4Insts; 868 } 869 870 bool hasDot5Insts() const { 871 return HasDot5Insts; 872 } 873 874 bool hasDot6Insts() const { 875 return HasDot6Insts; 876 } 877 878 bool hasMAIInsts() const { 879 return HasMAIInsts; 880 } 881 882 bool hasPkFmacF16Inst() const { 883 return HasPkFmacF16Inst; 884 } 885 886 bool hasAtomicFaddInsts() const { 887 return HasAtomicFaddInsts; 888 } 889 890 bool isSRAMECCEnabled() const { 891 return EnableSRAMECC; 892 } 893 894 bool hasNoSdstCMPX() const { 895 return HasNoSdstCMPX; 896 } 897 898 bool hasVscnt() const { 899 return HasVscnt; 900 } 901 902 bool hasGetWaveIdInst() const { 903 return HasGetWaveIdInst; 904 } 905 906 bool hasSMemTimeInst() const { 907 return HasSMemTimeInst; 908 } 909 910 bool hasRegisterBanking() const { 911 return HasRegisterBanking; 912 } 913 914 bool hasVOP3Literal() const { 915 return HasVOP3Literal; 916 } 917 918 bool hasNoDataDepHazard() const { 919 return HasNoDataDepHazard; 920 } 921 922 bool vmemWriteNeedsExpWaitcnt() const { 923 return getGeneration() < SEA_ISLANDS; 924 } 925 926 // Scratch is allocated in 256 dword per wave blocks for the entire 927 // wavefront. When viewed from the perspecive of an arbitrary workitem, this 928 // is 4-byte aligned. 929 // 930 // Only 4-byte alignment is really needed to access anything. Transformations 931 // on the pointer value itself may rely on the alignment / known low bits of 932 // the pointer. Set this to something above the minimum to avoid needing 933 // dynamic realignment in common cases. 934 Align getStackAlignment() const { return Align(16); } 935 936 bool enableMachineScheduler() const override { 937 return true; 938 } 939 940 bool enableSubRegLiveness() const override { 941 return true; 942 } 943 944 void setScalarizeGlobalBehavior(bool b) { ScalarizeGlobal = b; } 945 bool getScalarizeGlobalBehavior() const { return ScalarizeGlobal; } 946 947 // static wrappers 948 static bool hasHalfRate64Ops(const TargetSubtargetInfo &STI); 949 950 // XXX - Why is this here if it isn't in the default pass set? 951 bool enableEarlyIfConversion() const override { 952 return true; 953 } 954 955 void overrideSchedPolicy(MachineSchedPolicy &Policy, 956 unsigned NumRegionInstrs) const override; 957 958 unsigned getMaxNumUserSGPRs() const { 959 return 16; 960 } 961 962 bool hasSMemRealTime() const { 963 return HasSMemRealTime; 964 } 965 966 bool hasMovrel() const { 967 return HasMovrel; 968 } 969 970 bool hasVGPRIndexMode() const { 971 return HasVGPRIndexMode; 972 } 973 974 bool useVGPRIndexMode() const; 975 976 bool hasScalarCompareEq64() const { 977 return getGeneration() >= VOLCANIC_ISLANDS; 978 } 979 980 bool hasScalarStores() const { 981 return HasScalarStores; 982 } 983 984 bool hasScalarAtomics() const { 985 return HasScalarAtomics; 986 } 987 988 bool hasLDSFPAtomics() const { 989 return GFX8Insts; 990 } 991 992 bool hasDPP() const { 993 return HasDPP; 994 } 995 996 bool hasDPPBroadcasts() const { 997 return HasDPP && getGeneration() < GFX10; 998 } 999 1000 bool hasDPPWavefrontShifts() const { 1001 return HasDPP && getGeneration() < GFX10; 1002 } 1003 1004 bool hasDPP8() const { 1005 return HasDPP8; 1006 } 1007 1008 bool hasR128A16() const { 1009 return HasR128A16; 1010 } 1011 1012 bool hasGFX10A16() const { 1013 return HasGFX10A16; 1014 } 1015 1016 bool hasA16() const { return hasR128A16() || hasGFX10A16(); } 1017 1018 bool hasG16() const { return HasG16; } 1019 1020 bool hasOffset3fBug() const { 1021 return HasOffset3fBug; 1022 } 1023 1024 bool hasImageStoreD16Bug() const { return HasImageStoreD16Bug; } 1025 1026 bool hasImageGather4D16Bug() const { return HasImageGather4D16Bug; } 1027 1028 bool hasNSAEncoding() const { return HasNSAEncoding; } 1029 1030 bool hasGFX10_BEncoding() const { 1031 return GFX10_BEncoding; 1032 } 1033 1034 bool hasGFX10_3Insts() const { 1035 return GFX10_3Insts; 1036 } 1037 1038 bool hasMadF16() const; 1039 1040 bool enableSIScheduler() const { 1041 return EnableSIScheduler; 1042 } 1043 1044 bool loadStoreOptEnabled() const { 1045 return EnableLoadStoreOpt; 1046 } 1047 1048 bool hasSGPRInitBug() const { 1049 return SGPRInitBug; 1050 } 1051 1052 bool hasMFMAInlineLiteralBug() const { 1053 return HasMFMAInlineLiteralBug; 1054 } 1055 1056 bool has12DWordStoreHazard() const { 1057 return getGeneration() != AMDGPUSubtarget::SOUTHERN_ISLANDS; 1058 } 1059 1060 // \returns true if the subtarget supports DWORDX3 load/store instructions. 1061 bool hasDwordx3LoadStores() const { 1062 return CIInsts; 1063 } 1064 1065 bool hasReadM0MovRelInterpHazard() const { 1066 return getGeneration() == AMDGPUSubtarget::GFX9; 1067 } 1068 1069 bool hasReadM0SendMsgHazard() const { 1070 return getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS && 1071 getGeneration() <= AMDGPUSubtarget::GFX9; 1072 } 1073 1074 bool hasVcmpxPermlaneHazard() const { 1075 return HasVcmpxPermlaneHazard; 1076 } 1077 1078 bool hasVMEMtoScalarWriteHazard() const { 1079 return HasVMEMtoScalarWriteHazard; 1080 } 1081 1082 bool hasSMEMtoVectorWriteHazard() const { 1083 return HasSMEMtoVectorWriteHazard; 1084 } 1085 1086 bool hasLDSMisalignedBug() const { 1087 return LDSMisalignedBug && !EnableCuMode; 1088 } 1089 1090 bool hasInstFwdPrefetchBug() const { 1091 return HasInstFwdPrefetchBug; 1092 } 1093 1094 bool hasVcmpxExecWARHazard() const { 1095 return HasVcmpxExecWARHazard; 1096 } 1097 1098 bool hasLdsBranchVmemWARHazard() const { 1099 return HasLdsBranchVmemWARHazard; 1100 } 1101 1102 bool hasNSAtoVMEMBug() const { 1103 return HasNSAtoVMEMBug; 1104 } 1105 1106 bool hasHardClauses() const { return getGeneration() >= GFX10; } 1107 1108 /// Return the maximum number of waves per SIMD for kernels using \p SGPRs 1109 /// SGPRs 1110 unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const; 1111 1112 /// Return the maximum number of waves per SIMD for kernels using \p VGPRs 1113 /// VGPRs 1114 unsigned getOccupancyWithNumVGPRs(unsigned VGPRs) const; 1115 1116 /// Return occupancy for the given function. Used LDS and a number of 1117 /// registers if provided. 1118 /// Note, occupancy can be affected by the scratch allocation as well, but 1119 /// we do not have enough information to compute it. 1120 unsigned computeOccupancy(const Function &F, unsigned LDSSize = 0, 1121 unsigned NumSGPRs = 0, unsigned NumVGPRs = 0) const; 1122 1123 /// \returns true if the flat_scratch register should be initialized with the 1124 /// pointer to the wave's scratch memory rather than a size and offset. 1125 bool flatScratchIsPointer() const { 1126 return getGeneration() >= AMDGPUSubtarget::GFX9; 1127 } 1128 1129 /// \returns true if the machine has merged shaders in which s0-s7 are 1130 /// reserved by the hardware and user SGPRs start at s8 1131 bool hasMergedShaders() const { 1132 return getGeneration() >= GFX9; 1133 } 1134 1135 /// \returns SGPR allocation granularity supported by the subtarget. 1136 unsigned getSGPRAllocGranule() const { 1137 return AMDGPU::IsaInfo::getSGPRAllocGranule(this); 1138 } 1139 1140 /// \returns SGPR encoding granularity supported by the subtarget. 1141 unsigned getSGPREncodingGranule() const { 1142 return AMDGPU::IsaInfo::getSGPREncodingGranule(this); 1143 } 1144 1145 /// \returns Total number of SGPRs supported by the subtarget. 1146 unsigned getTotalNumSGPRs() const { 1147 return AMDGPU::IsaInfo::getTotalNumSGPRs(this); 1148 } 1149 1150 /// \returns Addressable number of SGPRs supported by the subtarget. 1151 unsigned getAddressableNumSGPRs() const { 1152 return AMDGPU::IsaInfo::getAddressableNumSGPRs(this); 1153 } 1154 1155 /// \returns Minimum number of SGPRs that meets the given number of waves per 1156 /// execution unit requirement supported by the subtarget. 1157 unsigned getMinNumSGPRs(unsigned WavesPerEU) const { 1158 return AMDGPU::IsaInfo::getMinNumSGPRs(this, WavesPerEU); 1159 } 1160 1161 /// \returns Maximum number of SGPRs that meets the given number of waves per 1162 /// execution unit requirement supported by the subtarget. 1163 unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const { 1164 return AMDGPU::IsaInfo::getMaxNumSGPRs(this, WavesPerEU, Addressable); 1165 } 1166 1167 /// \returns Reserved number of SGPRs for given function \p MF. 1168 unsigned getReservedNumSGPRs(const MachineFunction &MF) const; 1169 1170 /// \returns Maximum number of SGPRs that meets number of waves per execution 1171 /// unit requirement for function \p MF, or number of SGPRs explicitly 1172 /// requested using "amdgpu-num-sgpr" attribute attached to function \p MF. 1173 /// 1174 /// \returns Value that meets number of waves per execution unit requirement 1175 /// if explicitly requested value cannot be converted to integer, violates 1176 /// subtarget's specifications, or does not meet number of waves per execution 1177 /// unit requirement. 1178 unsigned getMaxNumSGPRs(const MachineFunction &MF) const; 1179 1180 /// \returns VGPR allocation granularity supported by the subtarget. 1181 unsigned getVGPRAllocGranule() const { 1182 return AMDGPU::IsaInfo::getVGPRAllocGranule(this); 1183 } 1184 1185 /// \returns VGPR encoding granularity supported by the subtarget. 1186 unsigned getVGPREncodingGranule() const { 1187 return AMDGPU::IsaInfo::getVGPREncodingGranule(this); 1188 } 1189 1190 /// \returns Total number of VGPRs supported by the subtarget. 1191 unsigned getTotalNumVGPRs() const { 1192 return AMDGPU::IsaInfo::getTotalNumVGPRs(this); 1193 } 1194 1195 /// \returns Addressable number of VGPRs supported by the subtarget. 1196 unsigned getAddressableNumVGPRs() const { 1197 return AMDGPU::IsaInfo::getAddressableNumVGPRs(this); 1198 } 1199 1200 /// \returns Minimum number of VGPRs that meets given number of waves per 1201 /// execution unit requirement supported by the subtarget. 1202 unsigned getMinNumVGPRs(unsigned WavesPerEU) const { 1203 return AMDGPU::IsaInfo::getMinNumVGPRs(this, WavesPerEU); 1204 } 1205 1206 /// \returns Maximum number of VGPRs that meets given number of waves per 1207 /// execution unit requirement supported by the subtarget. 1208 unsigned getMaxNumVGPRs(unsigned WavesPerEU) const { 1209 return AMDGPU::IsaInfo::getMaxNumVGPRs(this, WavesPerEU); 1210 } 1211 1212 /// \returns Maximum number of VGPRs that meets number of waves per execution 1213 /// unit requirement for function \p MF, or number of VGPRs explicitly 1214 /// requested using "amdgpu-num-vgpr" attribute attached to function \p MF. 1215 /// 1216 /// \returns Value that meets number of waves per execution unit requirement 1217 /// if explicitly requested value cannot be converted to integer, violates 1218 /// subtarget's specifications, or does not meet number of waves per execution 1219 /// unit requirement. 1220 unsigned getMaxNumVGPRs(const MachineFunction &MF) const; 1221 1222 void getPostRAMutations( 1223 std::vector<std::unique_ptr<ScheduleDAGMutation>> &Mutations) 1224 const override; 1225 1226 bool isWave32() const { 1227 return getWavefrontSize() == 32; 1228 } 1229 1230 bool isWave64() const { 1231 return getWavefrontSize() == 64; 1232 } 1233 1234 const TargetRegisterClass *getBoolRC() const { 1235 return getRegisterInfo()->getBoolRC(); 1236 } 1237 1238 /// \returns Maximum number of work groups per compute unit supported by the 1239 /// subtarget and limited by given \p FlatWorkGroupSize. 1240 unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const override { 1241 return AMDGPU::IsaInfo::getMaxWorkGroupsPerCU(this, FlatWorkGroupSize); 1242 } 1243 1244 /// \returns Minimum flat work group size supported by the subtarget. 1245 unsigned getMinFlatWorkGroupSize() const override { 1246 return AMDGPU::IsaInfo::getMinFlatWorkGroupSize(this); 1247 } 1248 1249 /// \returns Maximum flat work group size supported by the subtarget. 1250 unsigned getMaxFlatWorkGroupSize() const override { 1251 return AMDGPU::IsaInfo::getMaxFlatWorkGroupSize(this); 1252 } 1253 1254 /// \returns Number of waves per execution unit required to support the given 1255 /// \p FlatWorkGroupSize. 1256 unsigned 1257 getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const override { 1258 return AMDGPU::IsaInfo::getWavesPerEUForWorkGroup(this, FlatWorkGroupSize); 1259 } 1260 1261 /// \returns Minimum number of waves per execution unit supported by the 1262 /// subtarget. 1263 unsigned getMinWavesPerEU() const override { 1264 return AMDGPU::IsaInfo::getMinWavesPerEU(this); 1265 } 1266 1267 void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, 1268 SDep &Dep) const override; 1269 }; 1270 1271 class R600Subtarget final : public R600GenSubtargetInfo, 1272 public AMDGPUSubtarget { 1273 private: 1274 R600InstrInfo InstrInfo; 1275 R600FrameLowering FrameLowering; 1276 bool FMA; 1277 bool CaymanISA; 1278 bool CFALUBug; 1279 bool HasVertexCache; 1280 bool R600ALUInst; 1281 bool FP64; 1282 short TexVTXClauseSize; 1283 Generation Gen; 1284 R600TargetLowering TLInfo; 1285 InstrItineraryData InstrItins; 1286 SelectionDAGTargetInfo TSInfo; 1287 1288 public: 1289 R600Subtarget(const Triple &TT, StringRef CPU, StringRef FS, 1290 const TargetMachine &TM); 1291 1292 const R600InstrInfo *getInstrInfo() const override { return &InstrInfo; } 1293 1294 const R600FrameLowering *getFrameLowering() const override { 1295 return &FrameLowering; 1296 } 1297 1298 const R600TargetLowering *getTargetLowering() const override { 1299 return &TLInfo; 1300 } 1301 1302 const R600RegisterInfo *getRegisterInfo() const override { 1303 return &InstrInfo.getRegisterInfo(); 1304 } 1305 1306 const InstrItineraryData *getInstrItineraryData() const override { 1307 return &InstrItins; 1308 } 1309 1310 // Nothing implemented, just prevent crashes on use. 1311 const SelectionDAGTargetInfo *getSelectionDAGInfo() const override { 1312 return &TSInfo; 1313 } 1314 1315 void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS); 1316 1317 Generation getGeneration() const { 1318 return Gen; 1319 } 1320 1321 Align getStackAlignment() const { return Align(4); } 1322 1323 R600Subtarget &initializeSubtargetDependencies(const Triple &TT, 1324 StringRef GPU, StringRef FS); 1325 1326 bool hasBFE() const { 1327 return (getGeneration() >= EVERGREEN); 1328 } 1329 1330 bool hasBFI() const { 1331 return (getGeneration() >= EVERGREEN); 1332 } 1333 1334 bool hasBCNT(unsigned Size) const { 1335 if (Size == 32) 1336 return (getGeneration() >= EVERGREEN); 1337 1338 return false; 1339 } 1340 1341 bool hasBORROW() const { 1342 return (getGeneration() >= EVERGREEN); 1343 } 1344 1345 bool hasCARRY() const { 1346 return (getGeneration() >= EVERGREEN); 1347 } 1348 1349 bool hasCaymanISA() const { 1350 return CaymanISA; 1351 } 1352 1353 bool hasFFBL() const { 1354 return (getGeneration() >= EVERGREEN); 1355 } 1356 1357 bool hasFFBH() const { 1358 return (getGeneration() >= EVERGREEN); 1359 } 1360 1361 bool hasFMA() const { return FMA; } 1362 1363 bool hasCFAluBug() const { return CFALUBug; } 1364 1365 bool hasVertexCache() const { return HasVertexCache; } 1366 1367 short getTexVTXClauseSize() const { return TexVTXClauseSize; } 1368 1369 bool enableMachineScheduler() const override { 1370 return true; 1371 } 1372 1373 bool enableSubRegLiveness() const override { 1374 return true; 1375 } 1376 1377 /// \returns Maximum number of work groups per compute unit supported by the 1378 /// subtarget and limited by given \p FlatWorkGroupSize. 1379 unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const override { 1380 return AMDGPU::IsaInfo::getMaxWorkGroupsPerCU(this, FlatWorkGroupSize); 1381 } 1382 1383 /// \returns Minimum flat work group size supported by the subtarget. 1384 unsigned getMinFlatWorkGroupSize() const override { 1385 return AMDGPU::IsaInfo::getMinFlatWorkGroupSize(this); 1386 } 1387 1388 /// \returns Maximum flat work group size supported by the subtarget. 1389 unsigned getMaxFlatWorkGroupSize() const override { 1390 return AMDGPU::IsaInfo::getMaxFlatWorkGroupSize(this); 1391 } 1392 1393 /// \returns Number of waves per execution unit required to support the given 1394 /// \p FlatWorkGroupSize. 1395 unsigned 1396 getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const override { 1397 return AMDGPU::IsaInfo::getWavesPerEUForWorkGroup(this, FlatWorkGroupSize); 1398 } 1399 1400 /// \returns Minimum number of waves per execution unit supported by the 1401 /// subtarget. 1402 unsigned getMinWavesPerEU() const override { 1403 return AMDGPU::IsaInfo::getMinWavesPerEU(this); 1404 } 1405 }; 1406 1407 } // end namespace llvm 1408 1409 #endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H 1410