1 //===- AMDGPUTargetTransformInfo.h - AMDGPU specific TTI --------*- C++ -*-===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 /// \file 10 /// This file a TargetTransformInfo::Concept conforming object specific to the 11 /// AMDGPU target machine. It uses the target's detailed information to 12 /// provide more precise answers to certain TTI queries, while letting the 13 /// target independent and default TTI implementations handle the rest. 14 // 15 //===----------------------------------------------------------------------===// 16 17 #ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUTARGETTRANSFORMINFO_H 18 #define LLVM_LIB_TARGET_AMDGPU_AMDGPUTARGETTRANSFORMINFO_H 19 20 #include "AMDGPU.h" 21 #include "AMDGPUSubtarget.h" 22 #include "AMDGPUTargetMachine.h" 23 #include "MCTargetDesc/AMDGPUMCTargetDesc.h" 24 #include "Utils/AMDGPUBaseInfo.h" 25 #include "llvm/ADT/ArrayRef.h" 26 #include "llvm/Analysis/TargetTransformInfo.h" 27 #include "llvm/CodeGen/BasicTTIImpl.h" 28 #include "llvm/IR/Function.h" 29 #include "llvm/MC/SubtargetFeature.h" 30 #include "llvm/Support/MathExtras.h" 31 #include <cassert> 32 33 namespace llvm { 34 35 class AMDGPUTargetLowering; 36 class InstCombiner; 37 class Loop; 38 class ScalarEvolution; 39 class Type; 40 class Value; 41 42 class AMDGPUTTIImpl final : public BasicTTIImplBase<AMDGPUTTIImpl> { 43 using BaseT = BasicTTIImplBase<AMDGPUTTIImpl>; 44 using TTI = TargetTransformInfo; 45 46 friend BaseT; 47 48 Triple TargetTriple; 49 50 const GCNSubtarget *ST; 51 const TargetLoweringBase *TLI; 52 53 const TargetSubtargetInfo *getST() const { return ST; } 54 const TargetLoweringBase *getTLI() const { return TLI; } 55 56 public: 57 explicit AMDGPUTTIImpl(const AMDGPUTargetMachine *TM, const Function &F) 58 : BaseT(TM, F.getParent()->getDataLayout()), 59 TargetTriple(TM->getTargetTriple()), 60 ST(static_cast<const GCNSubtarget *>(TM->getSubtargetImpl(F))), 61 TLI(ST->getTargetLowering()) {} 62 63 void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, 64 TTI::UnrollingPreferences &UP); 65 66 void getPeelingPreferences(Loop *L, ScalarEvolution &SE, 67 TTI::PeelingPreferences &PP); 68 }; 69 70 class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> { 71 using BaseT = BasicTTIImplBase<GCNTTIImpl>; 72 using TTI = TargetTransformInfo; 73 74 friend BaseT; 75 76 const GCNSubtarget *ST; 77 const SITargetLowering *TLI; 78 AMDGPUTTIImpl CommonTTI; 79 bool IsGraphicsShader; 80 bool HasFP32Denormals; 81 bool HasFP64FP16Denormals; 82 unsigned MaxVGPRs; 83 84 const FeatureBitset InlineFeatureIgnoreList = { 85 // Codegen control options which don't matter. 86 AMDGPU::FeatureEnableLoadStoreOpt, 87 AMDGPU::FeatureEnableSIScheduler, 88 AMDGPU::FeatureEnableUnsafeDSOffsetFolding, 89 AMDGPU::FeatureFlatForGlobal, 90 AMDGPU::FeaturePromoteAlloca, 91 AMDGPU::FeatureUnalignedBufferAccess, 92 AMDGPU::FeatureUnalignedScratchAccess, 93 AMDGPU::FeatureUnalignedAccessMode, 94 95 AMDGPU::FeatureAutoWaitcntBeforeBarrier, 96 97 // Property of the kernel/environment which can't actually differ. 98 AMDGPU::FeatureSGPRInitBug, 99 AMDGPU::FeatureXNACK, 100 AMDGPU::FeatureTrapHandler, 101 102 // The default assumption needs to be ecc is enabled, but no directly 103 // exposed operations depend on it, so it can be safely inlined. 104 AMDGPU::FeatureSRAMECC, 105 106 // Perf-tuning features 107 AMDGPU::FeatureFastFMAF32, 108 AMDGPU::HalfRate64Ops 109 }; 110 111 const GCNSubtarget *getST() const { return ST; } 112 const AMDGPUTargetLowering *getTLI() const { return TLI; } 113 114 static inline int getFullRateInstrCost() { 115 return TargetTransformInfo::TCC_Basic; 116 } 117 118 static inline int getHalfRateInstrCost() { 119 return 2 * TargetTransformInfo::TCC_Basic; 120 } 121 122 // TODO: The size is usually 8 bytes, but takes 4x as many cycles. Maybe 123 // should be 2 or 4. 124 static inline int getQuarterRateInstrCost() { 125 return 3 * TargetTransformInfo::TCC_Basic; 126 } 127 128 // On some parts, normal fp64 operations are half rate, and others 129 // quarter. This also applies to some integer operations. 130 inline int get64BitInstrCost() const { 131 return ST->hasHalfRate64Ops() ? 132 getHalfRateInstrCost() : getQuarterRateInstrCost(); 133 } 134 135 public: 136 explicit GCNTTIImpl(const AMDGPUTargetMachine *TM, const Function &F) 137 : BaseT(TM, F.getParent()->getDataLayout()), 138 ST(static_cast<const GCNSubtarget *>(TM->getSubtargetImpl(F))), 139 TLI(ST->getTargetLowering()), CommonTTI(TM, F), 140 IsGraphicsShader(AMDGPU::isShader(F.getCallingConv())), 141 MaxVGPRs(ST->getMaxNumVGPRs( 142 std::max(ST->getWavesPerEU(F).first, 143 ST->getWavesPerEUForWorkGroup( 144 ST->getFlatWorkGroupSizes(F).second)))) { 145 AMDGPU::SIModeRegisterDefaults Mode(F); 146 HasFP32Denormals = Mode.allFP32Denormals(); 147 HasFP64FP16Denormals = Mode.allFP64FP16Denormals(); 148 } 149 150 bool hasBranchDivergence() { return true; } 151 bool useGPUDivergenceAnalysis() const; 152 153 void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, 154 TTI::UnrollingPreferences &UP); 155 156 void getPeelingPreferences(Loop *L, ScalarEvolution &SE, 157 TTI::PeelingPreferences &PP); 158 159 TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) { 160 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2"); 161 return TTI::PSK_FastHardware; 162 } 163 164 unsigned getHardwareNumberOfRegisters(bool Vector) const; 165 unsigned getNumberOfRegisters(bool Vector) const; 166 unsigned getNumberOfRegisters(unsigned RCID) const; 167 unsigned getRegisterBitWidth(bool Vector) const; 168 unsigned getMinVectorRegisterBitWidth() const; 169 unsigned getLoadVectorFactor(unsigned VF, unsigned LoadSize, 170 unsigned ChainSizeInBytes, 171 VectorType *VecTy) const; 172 unsigned getStoreVectorFactor(unsigned VF, unsigned StoreSize, 173 unsigned ChainSizeInBytes, 174 VectorType *VecTy) const; 175 unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const; 176 177 bool isLegalToVectorizeMemChain(unsigned ChainSizeInBytes, Align Alignment, 178 unsigned AddrSpace) const; 179 bool isLegalToVectorizeLoadChain(unsigned ChainSizeInBytes, Align Alignment, 180 unsigned AddrSpace) const; 181 bool isLegalToVectorizeStoreChain(unsigned ChainSizeInBytes, Align Alignment, 182 unsigned AddrSpace) const; 183 Type *getMemcpyLoopLoweringType(LLVMContext &Context, Value *Length, 184 unsigned SrcAddrSpace, unsigned DestAddrSpace, 185 unsigned SrcAlign, unsigned DestAlign) const; 186 187 void getMemcpyLoopResidualLoweringType(SmallVectorImpl<Type *> &OpsOut, 188 LLVMContext &Context, 189 unsigned RemainingBytes, 190 unsigned SrcAddrSpace, 191 unsigned DestAddrSpace, 192 unsigned SrcAlign, 193 unsigned DestAlign) const; 194 unsigned getMaxInterleaveFactor(unsigned VF); 195 196 bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const; 197 198 int getArithmeticInstrCost( 199 unsigned Opcode, Type *Ty, 200 TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput, 201 TTI::OperandValueKind Opd1Info = TTI::OK_AnyValue, 202 TTI::OperandValueKind Opd2Info = TTI::OK_AnyValue, 203 TTI::OperandValueProperties Opd1PropInfo = TTI::OP_None, 204 TTI::OperandValueProperties Opd2PropInfo = TTI::OP_None, 205 ArrayRef<const Value *> Args = ArrayRef<const Value *>(), 206 const Instruction *CxtI = nullptr); 207 208 unsigned getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind); 209 210 bool isInlineAsmSourceOfDivergence(const CallInst *CI, 211 ArrayRef<unsigned> Indices = {}) const; 212 213 int getVectorInstrCost(unsigned Opcode, Type *ValTy, unsigned Index); 214 bool isSourceOfDivergence(const Value *V) const; 215 bool isAlwaysUniform(const Value *V) const; 216 217 unsigned getFlatAddressSpace() const { 218 // Don't bother running InferAddressSpaces pass on graphics shaders which 219 // don't use flat addressing. 220 if (IsGraphicsShader) 221 return -1; 222 return AMDGPUAS::FLAT_ADDRESS; 223 } 224 225 bool collectFlatAddressOperands(SmallVectorImpl<int> &OpIndexes, 226 Intrinsic::ID IID) const; 227 Value *rewriteIntrinsicWithAddressSpace(IntrinsicInst *II, Value *OldV, 228 Value *NewV) const; 229 230 Optional<Instruction *> instCombineIntrinsic(InstCombiner &IC, 231 IntrinsicInst &II) const; 232 Optional<Value *> simplifyDemandedVectorEltsIntrinsic( 233 InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, 234 APInt &UndefElts2, APInt &UndefElts3, 235 std::function<void(Instruction *, unsigned, APInt, APInt &)> 236 SimplifyAndSetOp) const; 237 238 unsigned getVectorSplitCost() { return 0; } 239 240 unsigned getShuffleCost(TTI::ShuffleKind Kind, VectorType *Tp, int Index, 241 VectorType *SubTp); 242 243 bool areInlineCompatible(const Function *Caller, 244 const Function *Callee) const; 245 246 unsigned getInliningThresholdMultiplier() { return 11; } 247 248 int getInlinerVectorBonusPercent() { return 0; } 249 250 int getArithmeticReductionCost( 251 unsigned Opcode, 252 VectorType *Ty, 253 bool IsPairwise, 254 TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput); 255 256 int getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, 257 TTI::TargetCostKind CostKind); 258 int getMinMaxReductionCost( 259 VectorType *Ty, VectorType *CondTy, bool IsPairwiseForm, bool IsUnsigned, 260 TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput); 261 }; 262 263 class R600TTIImpl final : public BasicTTIImplBase<R600TTIImpl> { 264 using BaseT = BasicTTIImplBase<R600TTIImpl>; 265 using TTI = TargetTransformInfo; 266 267 friend BaseT; 268 269 const R600Subtarget *ST; 270 const AMDGPUTargetLowering *TLI; 271 AMDGPUTTIImpl CommonTTI; 272 273 public: 274 explicit R600TTIImpl(const AMDGPUTargetMachine *TM, const Function &F) 275 : BaseT(TM, F.getParent()->getDataLayout()), 276 ST(static_cast<const R600Subtarget*>(TM->getSubtargetImpl(F))), 277 TLI(ST->getTargetLowering()), 278 CommonTTI(TM, F) {} 279 280 const R600Subtarget *getST() const { return ST; } 281 const AMDGPUTargetLowering *getTLI() const { return TLI; } 282 283 void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, 284 TTI::UnrollingPreferences &UP); 285 void getPeelingPreferences(Loop *L, ScalarEvolution &SE, 286 TTI::PeelingPreferences &PP); 287 unsigned getHardwareNumberOfRegisters(bool Vec) const; 288 unsigned getNumberOfRegisters(bool Vec) const; 289 unsigned getRegisterBitWidth(bool Vector) const; 290 unsigned getMinVectorRegisterBitWidth() const; 291 unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const; 292 bool isLegalToVectorizeMemChain(unsigned ChainSizeInBytes, Align Alignment, 293 unsigned AddrSpace) const; 294 bool isLegalToVectorizeLoadChain(unsigned ChainSizeInBytes, Align Alignment, 295 unsigned AddrSpace) const; 296 bool isLegalToVectorizeStoreChain(unsigned ChainSizeInBytes, Align Alignment, 297 unsigned AddrSpace) const; 298 unsigned getMaxInterleaveFactor(unsigned VF); 299 unsigned getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind); 300 int getVectorInstrCost(unsigned Opcode, Type *ValTy, unsigned Index); 301 }; 302 303 } // end namespace llvm 304 305 #endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUTARGETTRANSFORMINFO_H 306