1 //===-- R600EmitClauseMarkers.cpp - Emit CF_ALU ---------------------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 /// \file
11 /// Add CF_ALU. R600 Alu instructions are grouped in clause which can hold
12 /// 128 Alu instructions ; these instructions can access up to 4 prefetched
13 /// 4 lines of 16 registers from constant buffers. Such ALU clauses are
14 /// initiated by CF_ALU instructions.
15 //===----------------------------------------------------------------------===//
16 
17 #include "AMDGPU.h"
18 #include "R600Defines.h"
19 #include "R600InstrInfo.h"
20 #include "R600RegisterInfo.h"
21 #include "AMDGPUSubtarget.h"
22 #include "llvm/ADT/SmallVector.h"
23 #include "llvm/ADT/StringRef.h"
24 #include "llvm/CodeGen/MachineBasicBlock.h"
25 #include "llvm/CodeGen/MachineFunction.h"
26 #include "llvm/CodeGen/MachineFunctionPass.h"
27 #include "llvm/CodeGen/MachineInstr.h"
28 #include "llvm/CodeGen/MachineInstrBuilder.h"
29 #include "llvm/CodeGen/MachineOperand.h"
30 #include "llvm/Pass.h"
31 #include "llvm/Support/ErrorHandling.h"
32 #include <cassert>
33 #include <cstdint>
34 #include <utility>
35 #include <vector>
36 
37 using namespace llvm;
38 
39 namespace llvm {
40 
41   void initializeR600EmitClauseMarkersPass(PassRegistry&);
42 
43 } // end namespace llvm
44 
45 namespace {
46 
47 class R600EmitClauseMarkers : public MachineFunctionPass {
48 private:
49   const R600InstrInfo *TII = nullptr;
50   int Address = 0;
51 
52   unsigned OccupiedDwords(MachineInstr &MI) const {
53     switch (MI.getOpcode()) {
54     case AMDGPU::INTERP_PAIR_XY:
55     case AMDGPU::INTERP_PAIR_ZW:
56     case AMDGPU::INTERP_VEC_LOAD:
57     case AMDGPU::DOT_4:
58       return 4;
59     case AMDGPU::KILL:
60       return 0;
61     default:
62       break;
63     }
64 
65     // These will be expanded to two ALU instructions in the
66     // ExpandSpecialInstructions pass.
67     if (TII->isLDSRetInstr(MI.getOpcode()))
68       return 2;
69 
70     if (TII->isVector(MI) || TII->isCubeOp(MI.getOpcode()) ||
71         TII->isReductionOp(MI.getOpcode()))
72       return 4;
73 
74     unsigned NumLiteral = 0;
75     for (MachineInstr::mop_iterator It = MI.operands_begin(),
76                                     E = MI.operands_end();
77          It != E; ++It) {
78       MachineOperand &MO = *It;
79       if (MO.isReg() && MO.getReg() == AMDGPU::ALU_LITERAL_X)
80         ++NumLiteral;
81     }
82     return 1 + NumLiteral;
83   }
84 
85   bool isALU(const MachineInstr &MI) const {
86     if (TII->isALUInstr(MI.getOpcode()))
87       return true;
88     if (TII->isVector(MI) || TII->isCubeOp(MI.getOpcode()))
89       return true;
90     switch (MI.getOpcode()) {
91     case AMDGPU::PRED_X:
92     case AMDGPU::INTERP_PAIR_XY:
93     case AMDGPU::INTERP_PAIR_ZW:
94     case AMDGPU::INTERP_VEC_LOAD:
95     case AMDGPU::COPY:
96     case AMDGPU::DOT_4:
97       return true;
98     default:
99       return false;
100     }
101   }
102 
103   bool IsTrivialInst(MachineInstr &MI) const {
104     switch (MI.getOpcode()) {
105     case AMDGPU::KILL:
106     case AMDGPU::RETURN:
107     case AMDGPU::IMPLICIT_DEF:
108       return true;
109     default:
110       return false;
111     }
112   }
113 
114   std::pair<unsigned, unsigned> getAccessedBankLine(unsigned Sel) const {
115     // Sel is (512 + (kc_bank << 12) + ConstIndex) << 2
116     // (See also R600ISelLowering.cpp)
117     // ConstIndex value is in [0, 4095];
118     return std::pair<unsigned, unsigned>(
119         ((Sel >> 2) - 512) >> 12, // KC_BANK
120         // Line Number of ConstIndex
121         // A line contains 16 constant registers however KCX bank can lock
122         // two line at the same time ; thus we want to get an even line number.
123         // Line number can be retrieved with (>>4), using (>>5) <<1 generates
124         // an even number.
125         ((((Sel >> 2) - 512) & 4095) >> 5) << 1);
126   }
127 
128   bool
129   SubstituteKCacheBank(MachineInstr &MI,
130                        std::vector<std::pair<unsigned, unsigned>> &CachedConsts,
131                        bool UpdateInstr = true) const {
132     std::vector<std::pair<unsigned, unsigned>> UsedKCache;
133 
134     if (!TII->isALUInstr(MI.getOpcode()) && MI.getOpcode() != AMDGPU::DOT_4)
135       return true;
136 
137     const SmallVectorImpl<std::pair<MachineOperand *, int64_t>> &Consts =
138         TII->getSrcs(MI);
139     assert(
140         (TII->isALUInstr(MI.getOpcode()) || MI.getOpcode() == AMDGPU::DOT_4) &&
141         "Can't assign Const");
142     for (unsigned i = 0, n = Consts.size(); i < n; ++i) {
143       if (Consts[i].first->getReg() != AMDGPU::ALU_CONST)
144         continue;
145       unsigned Sel = Consts[i].second;
146       unsigned Chan = Sel & 3, Index = ((Sel >> 2) - 512) & 31;
147       unsigned KCacheIndex = Index * 4 + Chan;
148       const std::pair<unsigned, unsigned> &BankLine = getAccessedBankLine(Sel);
149       if (CachedConsts.empty()) {
150         CachedConsts.push_back(BankLine);
151         UsedKCache.push_back(std::pair<unsigned, unsigned>(0, KCacheIndex));
152         continue;
153       }
154       if (CachedConsts[0] == BankLine) {
155         UsedKCache.push_back(std::pair<unsigned, unsigned>(0, KCacheIndex));
156         continue;
157       }
158       if (CachedConsts.size() == 1) {
159         CachedConsts.push_back(BankLine);
160         UsedKCache.push_back(std::pair<unsigned, unsigned>(1, KCacheIndex));
161         continue;
162       }
163       if (CachedConsts[1] == BankLine) {
164         UsedKCache.push_back(std::pair<unsigned, unsigned>(1, KCacheIndex));
165         continue;
166       }
167       return false;
168     }
169 
170     if (!UpdateInstr)
171       return true;
172 
173     for (unsigned i = 0, j = 0, n = Consts.size(); i < n; ++i) {
174       if (Consts[i].first->getReg() != AMDGPU::ALU_CONST)
175         continue;
176       switch(UsedKCache[j].first) {
177       case 0:
178         Consts[i].first->setReg(
179             AMDGPU::R600_KC0RegClass.getRegister(UsedKCache[j].second));
180         break;
181       case 1:
182         Consts[i].first->setReg(
183             AMDGPU::R600_KC1RegClass.getRegister(UsedKCache[j].second));
184         break;
185       default:
186         llvm_unreachable("Wrong Cache Line");
187       }
188       j++;
189     }
190     return true;
191   }
192 
193   bool canClauseLocalKillFitInClause(
194                         unsigned AluInstCount,
195                         std::vector<std::pair<unsigned, unsigned>> KCacheBanks,
196                         MachineBasicBlock::iterator Def,
197                         MachineBasicBlock::iterator BBEnd) {
198     const R600RegisterInfo &TRI = TII->getRegisterInfo();
199     for (MachineInstr::const_mop_iterator
200            MOI = Def->operands_begin(),
201            MOE = Def->operands_end(); MOI != MOE; ++MOI) {
202       if (!MOI->isReg() || !MOI->isDef() ||
203           TRI.isPhysRegLiveAcrossClauses(MOI->getReg()))
204         continue;
205 
206       // Def defines a clause local register, so check that its use will fit
207       // in the clause.
208       unsigned LastUseCount = 0;
209       for (MachineBasicBlock::iterator UseI = Def; UseI != BBEnd; ++UseI) {
210         AluInstCount += OccupiedDwords(*UseI);
211         // Make sure we won't need to end the clause due to KCache limitations.
212         if (!SubstituteKCacheBank(*UseI, KCacheBanks, false))
213           return false;
214 
215         // We have reached the maximum instruction limit before finding the
216         // use that kills this register, so we cannot use this def in the
217         // current clause.
218         if (AluInstCount >= TII->getMaxAlusPerClause())
219           return false;
220 
221         // Register kill flags have been cleared by the time we get to this
222         // pass, but it is safe to assume that all uses of this register
223         // occur in the same basic block as its definition, because
224         // it is illegal for the scheduler to schedule them in
225         // different blocks.
226         if (UseI->findRegisterUseOperandIdx(MOI->getReg()))
227           LastUseCount = AluInstCount;
228 
229         if (UseI != Def && UseI->findRegisterDefOperandIdx(MOI->getReg()) != -1)
230           break;
231       }
232       if (LastUseCount)
233         return LastUseCount <= TII->getMaxAlusPerClause();
234       llvm_unreachable("Clause local register live at end of clause.");
235     }
236     return true;
237   }
238 
239   MachineBasicBlock::iterator
240   MakeALUClause(MachineBasicBlock &MBB, MachineBasicBlock::iterator I) {
241     MachineBasicBlock::iterator ClauseHead = I;
242     std::vector<std::pair<unsigned, unsigned>> KCacheBanks;
243     bool PushBeforeModifier = false;
244     unsigned AluInstCount = 0;
245     for (MachineBasicBlock::iterator E = MBB.end(); I != E; ++I) {
246       if (IsTrivialInst(*I))
247         continue;
248       if (!isALU(*I))
249         break;
250       if (AluInstCount > TII->getMaxAlusPerClause())
251         break;
252       if (I->getOpcode() == AMDGPU::PRED_X) {
253         // We put PRED_X in its own clause to ensure that ifcvt won't create
254         // clauses with more than 128 insts.
255         // IfCvt is indeed checking that "then" and "else" branches of an if
256         // statement have less than ~60 insts thus converted clauses can't be
257         // bigger than ~121 insts (predicate setter needs to be in the same
258         // clause as predicated alus).
259         if (AluInstCount > 0)
260           break;
261         if (TII->getFlagOp(*I).getImm() & MO_FLAG_PUSH)
262           PushBeforeModifier = true;
263         AluInstCount ++;
264         continue;
265       }
266       // XXX: GROUP_BARRIER instructions cannot be in the same ALU clause as:
267       //
268       // * KILL or INTERP instructions
269       // * Any instruction that sets UPDATE_EXEC_MASK or UPDATE_PRED bits
270       // * Uses waterfalling (i.e. INDEX_MODE = AR.X)
271       //
272       // XXX: These checks have not been implemented yet.
273       if (TII->mustBeLastInClause(I->getOpcode())) {
274         I++;
275         break;
276       }
277 
278       // If this instruction defines a clause local register, make sure
279       // its use can fit in this clause.
280       if (!canClauseLocalKillFitInClause(AluInstCount, KCacheBanks, I, E))
281         break;
282 
283       if (!SubstituteKCacheBank(*I, KCacheBanks))
284         break;
285       AluInstCount += OccupiedDwords(*I);
286     }
287     unsigned Opcode = PushBeforeModifier ?
288         AMDGPU::CF_ALU_PUSH_BEFORE : AMDGPU::CF_ALU;
289     BuildMI(MBB, ClauseHead, MBB.findDebugLoc(ClauseHead), TII->get(Opcode))
290     // We don't use the ADDR field until R600ControlFlowFinalizer pass, where
291     // it is safe to assume it is 0. However if we always put 0 here, the ifcvt
292     // pass may assume that identical ALU clause starter at the beginning of a
293     // true and false branch can be factorized which is not the case.
294         .addImm(Address++) // ADDR
295         .addImm(KCacheBanks.empty()?0:KCacheBanks[0].first) // KB0
296         .addImm((KCacheBanks.size() < 2)?0:KCacheBanks[1].first) // KB1
297         .addImm(KCacheBanks.empty()?0:2) // KM0
298         .addImm((KCacheBanks.size() < 2)?0:2) // KM1
299         .addImm(KCacheBanks.empty()?0:KCacheBanks[0].second) // KLINE0
300         .addImm((KCacheBanks.size() < 2)?0:KCacheBanks[1].second) // KLINE1
301         .addImm(AluInstCount) // COUNT
302         .addImm(1); // Enabled
303     return I;
304   }
305 
306 public:
307   static char ID;
308 
309   R600EmitClauseMarkers() : MachineFunctionPass(ID) {
310     initializeR600EmitClauseMarkersPass(*PassRegistry::getPassRegistry());
311   }
312 
313   bool runOnMachineFunction(MachineFunction &MF) override {
314     const R600Subtarget &ST = MF.getSubtarget<R600Subtarget>();
315     TII = ST.getInstrInfo();
316 
317     for (MachineFunction::iterator BB = MF.begin(), BB_E = MF.end();
318                                                     BB != BB_E; ++BB) {
319       MachineBasicBlock &MBB = *BB;
320       MachineBasicBlock::iterator I = MBB.begin();
321       if (I != MBB.end() && I->getOpcode() == AMDGPU::CF_ALU)
322         continue; // BB was already parsed
323       for (MachineBasicBlock::iterator E = MBB.end(); I != E;) {
324         if (isALU(*I)) {
325           auto next = MakeALUClause(MBB, I);
326           assert(next != I);
327           I = next;
328         } else
329           ++I;
330       }
331     }
332     return false;
333   }
334 
335   StringRef getPassName() const override {
336     return "R600 Emit Clause Markers Pass";
337   }
338 };
339 
340 char R600EmitClauseMarkers::ID = 0;
341 
342 } // end anonymous namespace
343 
344 INITIALIZE_PASS_BEGIN(R600EmitClauseMarkers, "emitclausemarkers",
345                       "R600 Emit Clause Markters", false, false)
346 INITIALIZE_PASS_END(R600EmitClauseMarkers, "emitclausemarkers",
347                       "R600 Emit Clause Markters", false, false)
348 
349 FunctionPass *llvm::createR600EmitClauseMarkers() {
350   return new R600EmitClauseMarkers();
351 }
352