1 //===-- R600EmitClauseMarkers.cpp - Emit CF_ALU ---------------------------===// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //===----------------------------------------------------------------------===// 9 // 10 /// \file 11 /// Add CF_ALU. R600 Alu instructions are grouped in clause which can hold 12 /// 128 Alu instructions ; these instructions can access up to 4 prefetched 13 /// 4 lines of 16 registers from constant buffers. Such ALU clauses are 14 /// initiated by CF_ALU instructions. 15 //===----------------------------------------------------------------------===// 16 17 #include "AMDGPU.h" 18 #include "R600Defines.h" 19 #include "R600InstrInfo.h" 20 #include "R600RegisterInfo.h" 21 #include "AMDGPUSubtarget.h" 22 #include "llvm/ADT/SmallVector.h" 23 #include "llvm/ADT/StringRef.h" 24 #include "llvm/CodeGen/MachineBasicBlock.h" 25 #include "llvm/CodeGen/MachineFunction.h" 26 #include "llvm/CodeGen/MachineFunctionPass.h" 27 #include "llvm/CodeGen/MachineInstr.h" 28 #include "llvm/CodeGen/MachineInstrBuilder.h" 29 #include "llvm/CodeGen/MachineOperand.h" 30 #include "llvm/Pass.h" 31 #include "llvm/Support/ErrorHandling.h" 32 #include <cassert> 33 #include <cstdint> 34 #include <utility> 35 #include <vector> 36 37 using namespace llvm; 38 39 namespace llvm { 40 41 void initializeR600EmitClauseMarkersPass(PassRegistry&); 42 43 } // end namespace llvm 44 45 namespace { 46 47 class R600EmitClauseMarkers : public MachineFunctionPass { 48 private: 49 const R600InstrInfo *TII = nullptr; 50 int Address = 0; 51 52 unsigned OccupiedDwords(MachineInstr &MI) const { 53 switch (MI.getOpcode()) { 54 case AMDGPU::INTERP_PAIR_XY: 55 case AMDGPU::INTERP_PAIR_ZW: 56 case AMDGPU::INTERP_VEC_LOAD: 57 case AMDGPU::DOT_4: 58 return 4; 59 case AMDGPU::KILL: 60 return 0; 61 default: 62 break; 63 } 64 65 // These will be expanded to two ALU instructions in the 66 // ExpandSpecialInstructions pass. 67 if (TII->isLDSRetInstr(MI.getOpcode())) 68 return 2; 69 70 if (TII->isVector(MI) || TII->isCubeOp(MI.getOpcode()) || 71 TII->isReductionOp(MI.getOpcode())) 72 return 4; 73 74 unsigned NumLiteral = 0; 75 for (MachineInstr::mop_iterator It = MI.operands_begin(), 76 E = MI.operands_end(); 77 It != E; ++It) { 78 MachineOperand &MO = *It; 79 if (MO.isReg() && MO.getReg() == AMDGPU::ALU_LITERAL_X) 80 ++NumLiteral; 81 } 82 return 1 + NumLiteral; 83 } 84 85 bool isALU(const MachineInstr &MI) const { 86 if (TII->isALUInstr(MI.getOpcode())) 87 return true; 88 if (TII->isVector(MI) || TII->isCubeOp(MI.getOpcode())) 89 return true; 90 switch (MI.getOpcode()) { 91 case AMDGPU::PRED_X: 92 case AMDGPU::INTERP_PAIR_XY: 93 case AMDGPU::INTERP_PAIR_ZW: 94 case AMDGPU::INTERP_VEC_LOAD: 95 case AMDGPU::COPY: 96 case AMDGPU::DOT_4: 97 return true; 98 default: 99 return false; 100 } 101 } 102 103 bool IsTrivialInst(MachineInstr &MI) const { 104 switch (MI.getOpcode()) { 105 case AMDGPU::KILL: 106 case AMDGPU::RETURN: 107 case AMDGPU::IMPLICIT_DEF: 108 return true; 109 default: 110 return false; 111 } 112 } 113 114 std::pair<unsigned, unsigned> getAccessedBankLine(unsigned Sel) const { 115 // Sel is (512 + (kc_bank << 12) + ConstIndex) << 2 116 // (See also R600ISelLowering.cpp) 117 // ConstIndex value is in [0, 4095]; 118 return std::pair<unsigned, unsigned>( 119 ((Sel >> 2) - 512) >> 12, // KC_BANK 120 // Line Number of ConstIndex 121 // A line contains 16 constant registers however KCX bank can lock 122 // two line at the same time ; thus we want to get an even line number. 123 // Line number can be retrieved with (>>4), using (>>5) <<1 generates 124 // an even number. 125 ((((Sel >> 2) - 512) & 4095) >> 5) << 1); 126 } 127 128 bool 129 SubstituteKCacheBank(MachineInstr &MI, 130 std::vector<std::pair<unsigned, unsigned>> &CachedConsts, 131 bool UpdateInstr = true) const { 132 std::vector<std::pair<unsigned, unsigned>> UsedKCache; 133 134 if (!TII->isALUInstr(MI.getOpcode()) && MI.getOpcode() != AMDGPU::DOT_4) 135 return true; 136 137 const SmallVectorImpl<std::pair<MachineOperand *, int64_t>> &Consts = 138 TII->getSrcs(MI); 139 assert( 140 (TII->isALUInstr(MI.getOpcode()) || MI.getOpcode() == AMDGPU::DOT_4) && 141 "Can't assign Const"); 142 for (unsigned i = 0, n = Consts.size(); i < n; ++i) { 143 if (Consts[i].first->getReg() != AMDGPU::ALU_CONST) 144 continue; 145 unsigned Sel = Consts[i].second; 146 unsigned Chan = Sel & 3, Index = ((Sel >> 2) - 512) & 31; 147 unsigned KCacheIndex = Index * 4 + Chan; 148 const std::pair<unsigned, unsigned> &BankLine = getAccessedBankLine(Sel); 149 if (CachedConsts.empty()) { 150 CachedConsts.push_back(BankLine); 151 UsedKCache.push_back(std::pair<unsigned, unsigned>(0, KCacheIndex)); 152 continue; 153 } 154 if (CachedConsts[0] == BankLine) { 155 UsedKCache.push_back(std::pair<unsigned, unsigned>(0, KCacheIndex)); 156 continue; 157 } 158 if (CachedConsts.size() == 1) { 159 CachedConsts.push_back(BankLine); 160 UsedKCache.push_back(std::pair<unsigned, unsigned>(1, KCacheIndex)); 161 continue; 162 } 163 if (CachedConsts[1] == BankLine) { 164 UsedKCache.push_back(std::pair<unsigned, unsigned>(1, KCacheIndex)); 165 continue; 166 } 167 return false; 168 } 169 170 if (!UpdateInstr) 171 return true; 172 173 for (unsigned i = 0, j = 0, n = Consts.size(); i < n; ++i) { 174 if (Consts[i].first->getReg() != AMDGPU::ALU_CONST) 175 continue; 176 switch(UsedKCache[j].first) { 177 case 0: 178 Consts[i].first->setReg( 179 AMDGPU::R600_KC0RegClass.getRegister(UsedKCache[j].second)); 180 break; 181 case 1: 182 Consts[i].first->setReg( 183 AMDGPU::R600_KC1RegClass.getRegister(UsedKCache[j].second)); 184 break; 185 default: 186 llvm_unreachable("Wrong Cache Line"); 187 } 188 j++; 189 } 190 return true; 191 } 192 193 bool canClauseLocalKillFitInClause( 194 unsigned AluInstCount, 195 std::vector<std::pair<unsigned, unsigned>> KCacheBanks, 196 MachineBasicBlock::iterator Def, 197 MachineBasicBlock::iterator BBEnd) { 198 const R600RegisterInfo &TRI = TII->getRegisterInfo(); 199 for (MachineInstr::const_mop_iterator 200 MOI = Def->operands_begin(), 201 MOE = Def->operands_end(); MOI != MOE; ++MOI) { 202 if (!MOI->isReg() || !MOI->isDef() || 203 TRI.isPhysRegLiveAcrossClauses(MOI->getReg())) 204 continue; 205 206 // Def defines a clause local register, so check that its use will fit 207 // in the clause. 208 unsigned LastUseCount = 0; 209 for (MachineBasicBlock::iterator UseI = Def; UseI != BBEnd; ++UseI) { 210 AluInstCount += OccupiedDwords(*UseI); 211 // Make sure we won't need to end the clause due to KCache limitations. 212 if (!SubstituteKCacheBank(*UseI, KCacheBanks, false)) 213 return false; 214 215 // We have reached the maximum instruction limit before finding the 216 // use that kills this register, so we cannot use this def in the 217 // current clause. 218 if (AluInstCount >= TII->getMaxAlusPerClause()) 219 return false; 220 221 // Register kill flags have been cleared by the time we get to this 222 // pass, but it is safe to assume that all uses of this register 223 // occur in the same basic block as its definition, because 224 // it is illegal for the scheduler to schedule them in 225 // different blocks. 226 if (UseI->findRegisterUseOperandIdx(MOI->getReg())) 227 LastUseCount = AluInstCount; 228 229 if (UseI != Def && UseI->findRegisterDefOperandIdx(MOI->getReg()) != -1) 230 break; 231 } 232 if (LastUseCount) 233 return LastUseCount <= TII->getMaxAlusPerClause(); 234 llvm_unreachable("Clause local register live at end of clause."); 235 } 236 return true; 237 } 238 239 MachineBasicBlock::iterator 240 MakeALUClause(MachineBasicBlock &MBB, MachineBasicBlock::iterator I) { 241 MachineBasicBlock::iterator ClauseHead = I; 242 std::vector<std::pair<unsigned, unsigned>> KCacheBanks; 243 bool PushBeforeModifier = false; 244 unsigned AluInstCount = 0; 245 for (MachineBasicBlock::iterator E = MBB.end(); I != E; ++I) { 246 if (IsTrivialInst(*I)) 247 continue; 248 if (!isALU(*I)) 249 break; 250 if (AluInstCount > TII->getMaxAlusPerClause()) 251 break; 252 if (I->getOpcode() == AMDGPU::PRED_X) { 253 // We put PRED_X in its own clause to ensure that ifcvt won't create 254 // clauses with more than 128 insts. 255 // IfCvt is indeed checking that "then" and "else" branches of an if 256 // statement have less than ~60 insts thus converted clauses can't be 257 // bigger than ~121 insts (predicate setter needs to be in the same 258 // clause as predicated alus). 259 if (AluInstCount > 0) 260 break; 261 if (TII->getFlagOp(*I).getImm() & MO_FLAG_PUSH) 262 PushBeforeModifier = true; 263 AluInstCount ++; 264 continue; 265 } 266 // XXX: GROUP_BARRIER instructions cannot be in the same ALU clause as: 267 // 268 // * KILL or INTERP instructions 269 // * Any instruction that sets UPDATE_EXEC_MASK or UPDATE_PRED bits 270 // * Uses waterfalling (i.e. INDEX_MODE = AR.X) 271 // 272 // XXX: These checks have not been implemented yet. 273 if (TII->mustBeLastInClause(I->getOpcode())) { 274 I++; 275 break; 276 } 277 278 // If this instruction defines a clause local register, make sure 279 // its use can fit in this clause. 280 if (!canClauseLocalKillFitInClause(AluInstCount, KCacheBanks, I, E)) 281 break; 282 283 if (!SubstituteKCacheBank(*I, KCacheBanks)) 284 break; 285 AluInstCount += OccupiedDwords(*I); 286 } 287 unsigned Opcode = PushBeforeModifier ? 288 AMDGPU::CF_ALU_PUSH_BEFORE : AMDGPU::CF_ALU; 289 BuildMI(MBB, ClauseHead, MBB.findDebugLoc(ClauseHead), TII->get(Opcode)) 290 // We don't use the ADDR field until R600ControlFlowFinalizer pass, where 291 // it is safe to assume it is 0. However if we always put 0 here, the ifcvt 292 // pass may assume that identical ALU clause starter at the beginning of a 293 // true and false branch can be factorized which is not the case. 294 .addImm(Address++) // ADDR 295 .addImm(KCacheBanks.empty()?0:KCacheBanks[0].first) // KB0 296 .addImm((KCacheBanks.size() < 2)?0:KCacheBanks[1].first) // KB1 297 .addImm(KCacheBanks.empty()?0:2) // KM0 298 .addImm((KCacheBanks.size() < 2)?0:2) // KM1 299 .addImm(KCacheBanks.empty()?0:KCacheBanks[0].second) // KLINE0 300 .addImm((KCacheBanks.size() < 2)?0:KCacheBanks[1].second) // KLINE1 301 .addImm(AluInstCount) // COUNT 302 .addImm(1); // Enabled 303 return I; 304 } 305 306 public: 307 static char ID; 308 309 R600EmitClauseMarkers() : MachineFunctionPass(ID) { 310 initializeR600EmitClauseMarkersPass(*PassRegistry::getPassRegistry()); 311 } 312 313 bool runOnMachineFunction(MachineFunction &MF) override { 314 const R600Subtarget &ST = MF.getSubtarget<R600Subtarget>(); 315 TII = ST.getInstrInfo(); 316 317 for (MachineFunction::iterator BB = MF.begin(), BB_E = MF.end(); 318 BB != BB_E; ++BB) { 319 MachineBasicBlock &MBB = *BB; 320 MachineBasicBlock::iterator I = MBB.begin(); 321 if (I != MBB.end() && I->getOpcode() == AMDGPU::CF_ALU) 322 continue; // BB was already parsed 323 for (MachineBasicBlock::iterator E = MBB.end(); I != E;) { 324 if (isALU(*I)) { 325 auto next = MakeALUClause(MBB, I); 326 assert(next != I); 327 I = next; 328 } else 329 ++I; 330 } 331 } 332 return false; 333 } 334 335 StringRef getPassName() const override { 336 return "R600 Emit Clause Markers Pass"; 337 } 338 }; 339 340 char R600EmitClauseMarkers::ID = 0; 341 342 } // end anonymous namespace 343 344 INITIALIZE_PASS_BEGIN(R600EmitClauseMarkers, "emitclausemarkers", 345 "R600 Emit Clause Markters", false, false) 346 INITIALIZE_PASS_END(R600EmitClauseMarkers, "emitclausemarkers", 347 "R600 Emit Clause Markters", false, false) 348 349 FunctionPass *llvm::createR600EmitClauseMarkers() { 350 return new R600EmitClauseMarkers(); 351 } 352