1 //===- RISCVInsertVSETVLI.cpp - Insert VSETVLI instructions ---------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This file implements a function pass that inserts VSETVLI instructions where
10 // needed and expands the vl outputs of VLEFF/VLSEGFF to PseudoReadVL
11 // instructions.
12 //
13 // This pass consists of 3 phases:
14 //
15 // Phase 1 collects how each basic block affects VL/VTYPE.
16 //
17 // Phase 2 uses the information from phase 1 to do a data flow analysis to
18 // propagate the VL/VTYPE changes through the function. This gives us the
19 // VL/VTYPE at the start of each basic block.
20 //
21 // Phase 3 inserts VSETVLI instructions in each basic block. Information from
22 // phase 2 is used to prevent inserting a VSETVLI before the first vector
23 // instruction in the block if possible.
24 //
25 //===----------------------------------------------------------------------===//
26 
27 #include "RISCV.h"
28 #include "RISCVSubtarget.h"
29 #include "llvm/CodeGen/LiveIntervals.h"
30 #include "llvm/CodeGen/MachineFunctionPass.h"
31 #include <queue>
32 using namespace llvm;
33 
34 #define DEBUG_TYPE "riscv-insert-vsetvli"
35 #define RISCV_INSERT_VSETVLI_NAME "RISCV Insert VSETVLI pass"
36 
37 static cl::opt<bool> DisableInsertVSETVLPHIOpt(
38     "riscv-disable-insert-vsetvl-phi-opt", cl::init(false), cl::Hidden,
39     cl::desc("Disable looking through phis when inserting vsetvlis."));
40 
41 static cl::opt<bool> UseStrictAsserts(
42     "riscv-insert-vsetvl-strict-asserts", cl::init(true), cl::Hidden,
43     cl::desc("Enable strict assertion checking for the dataflow algorithm"));
44 
45 namespace {
46 
47 static unsigned getVLOpNum(const MachineInstr &MI) {
48   return RISCVII::getVLOpNum(MI.getDesc());
49 }
50 
51 static unsigned getSEWOpNum(const MachineInstr &MI) {
52   return RISCVII::getSEWOpNum(MI.getDesc());
53 }
54 
55 static bool isScalarMoveInstr(const MachineInstr &MI) {
56   switch (MI.getOpcode()) {
57   default:
58     return false;
59   case RISCV::PseudoVMV_S_X_M1:
60   case RISCV::PseudoVMV_S_X_M2:
61   case RISCV::PseudoVMV_S_X_M4:
62   case RISCV::PseudoVMV_S_X_M8:
63   case RISCV::PseudoVMV_S_X_MF2:
64   case RISCV::PseudoVMV_S_X_MF4:
65   case RISCV::PseudoVMV_S_X_MF8:
66   case RISCV::PseudoVFMV_S_F16_M1:
67   case RISCV::PseudoVFMV_S_F16_M2:
68   case RISCV::PseudoVFMV_S_F16_M4:
69   case RISCV::PseudoVFMV_S_F16_M8:
70   case RISCV::PseudoVFMV_S_F16_MF2:
71   case RISCV::PseudoVFMV_S_F16_MF4:
72   case RISCV::PseudoVFMV_S_F32_M1:
73   case RISCV::PseudoVFMV_S_F32_M2:
74   case RISCV::PseudoVFMV_S_F32_M4:
75   case RISCV::PseudoVFMV_S_F32_M8:
76   case RISCV::PseudoVFMV_S_F32_MF2:
77   case RISCV::PseudoVFMV_S_F64_M1:
78   case RISCV::PseudoVFMV_S_F64_M2:
79   case RISCV::PseudoVFMV_S_F64_M4:
80   case RISCV::PseudoVFMV_S_F64_M8:
81     return true;
82   }
83 }
84 
85 static bool isSplatMoveInstr(const MachineInstr &MI) {
86   switch (MI.getOpcode()) {
87   default:
88     return false;
89   case RISCV::PseudoVMV_V_X_M1:
90   case RISCV::PseudoVMV_V_X_M2:
91   case RISCV::PseudoVMV_V_X_M4:
92   case RISCV::PseudoVMV_V_X_M8:
93   case RISCV::PseudoVMV_V_X_MF2:
94   case RISCV::PseudoVMV_V_X_MF4:
95   case RISCV::PseudoVMV_V_X_MF8:
96   case RISCV::PseudoVMV_V_I_M1:
97   case RISCV::PseudoVMV_V_I_M2:
98   case RISCV::PseudoVMV_V_I_M4:
99   case RISCV::PseudoVMV_V_I_M8:
100   case RISCV::PseudoVMV_V_I_MF2:
101   case RISCV::PseudoVMV_V_I_MF4:
102   case RISCV::PseudoVMV_V_I_MF8:
103     return true;
104   }
105 }
106 
107 static bool isSplatOfZeroOrMinusOne(const MachineInstr &MI) {
108   if (!isSplatMoveInstr(MI))
109     return false;
110 
111   const MachineOperand &SrcMO = MI.getOperand(1);
112   if (SrcMO.isImm())
113     return SrcMO.getImm() == 0 || SrcMO.getImm() == -1;
114   return SrcMO.isReg() && SrcMO.getReg() == RISCV::X0;
115 }
116 
117 /// Get the EEW for a load or store instruction.  Return None if MI is not
118 /// a load or store which ignores SEW.
119 static Optional<unsigned> getEEWForLoadStore(const MachineInstr &MI) {
120   switch (MI.getOpcode()) {
121   default:
122     return None;
123   case RISCV::PseudoVLE8_V_M1:
124   case RISCV::PseudoVLE8_V_M1_MASK:
125   case RISCV::PseudoVLE8_V_M2:
126   case RISCV::PseudoVLE8_V_M2_MASK:
127   case RISCV::PseudoVLE8_V_M4:
128   case RISCV::PseudoVLE8_V_M4_MASK:
129   case RISCV::PseudoVLE8_V_M8:
130   case RISCV::PseudoVLE8_V_M8_MASK:
131   case RISCV::PseudoVLE8_V_MF2:
132   case RISCV::PseudoVLE8_V_MF2_MASK:
133   case RISCV::PseudoVLE8_V_MF4:
134   case RISCV::PseudoVLE8_V_MF4_MASK:
135   case RISCV::PseudoVLE8_V_MF8:
136   case RISCV::PseudoVLE8_V_MF8_MASK:
137   case RISCV::PseudoVLSE8_V_M1:
138   case RISCV::PseudoVLSE8_V_M1_MASK:
139   case RISCV::PseudoVLSE8_V_M2:
140   case RISCV::PseudoVLSE8_V_M2_MASK:
141   case RISCV::PseudoVLSE8_V_M4:
142   case RISCV::PseudoVLSE8_V_M4_MASK:
143   case RISCV::PseudoVLSE8_V_M8:
144   case RISCV::PseudoVLSE8_V_M8_MASK:
145   case RISCV::PseudoVLSE8_V_MF2:
146   case RISCV::PseudoVLSE8_V_MF2_MASK:
147   case RISCV::PseudoVLSE8_V_MF4:
148   case RISCV::PseudoVLSE8_V_MF4_MASK:
149   case RISCV::PseudoVLSE8_V_MF8:
150   case RISCV::PseudoVLSE8_V_MF8_MASK:
151   case RISCV::PseudoVSE8_V_M1:
152   case RISCV::PseudoVSE8_V_M1_MASK:
153   case RISCV::PseudoVSE8_V_M2:
154   case RISCV::PseudoVSE8_V_M2_MASK:
155   case RISCV::PseudoVSE8_V_M4:
156   case RISCV::PseudoVSE8_V_M4_MASK:
157   case RISCV::PseudoVSE8_V_M8:
158   case RISCV::PseudoVSE8_V_M8_MASK:
159   case RISCV::PseudoVSE8_V_MF2:
160   case RISCV::PseudoVSE8_V_MF2_MASK:
161   case RISCV::PseudoVSE8_V_MF4:
162   case RISCV::PseudoVSE8_V_MF4_MASK:
163   case RISCV::PseudoVSE8_V_MF8:
164   case RISCV::PseudoVSE8_V_MF8_MASK:
165   case RISCV::PseudoVSSE8_V_M1:
166   case RISCV::PseudoVSSE8_V_M1_MASK:
167   case RISCV::PseudoVSSE8_V_M2:
168   case RISCV::PseudoVSSE8_V_M2_MASK:
169   case RISCV::PseudoVSSE8_V_M4:
170   case RISCV::PseudoVSSE8_V_M4_MASK:
171   case RISCV::PseudoVSSE8_V_M8:
172   case RISCV::PseudoVSSE8_V_M8_MASK:
173   case RISCV::PseudoVSSE8_V_MF2:
174   case RISCV::PseudoVSSE8_V_MF2_MASK:
175   case RISCV::PseudoVSSE8_V_MF4:
176   case RISCV::PseudoVSSE8_V_MF4_MASK:
177   case RISCV::PseudoVSSE8_V_MF8:
178   case RISCV::PseudoVSSE8_V_MF8_MASK:
179     return 8;
180   case RISCV::PseudoVLE16_V_M1:
181   case RISCV::PseudoVLE16_V_M1_MASK:
182   case RISCV::PseudoVLE16_V_M2:
183   case RISCV::PseudoVLE16_V_M2_MASK:
184   case RISCV::PseudoVLE16_V_M4:
185   case RISCV::PseudoVLE16_V_M4_MASK:
186   case RISCV::PseudoVLE16_V_M8:
187   case RISCV::PseudoVLE16_V_M8_MASK:
188   case RISCV::PseudoVLE16_V_MF2:
189   case RISCV::PseudoVLE16_V_MF2_MASK:
190   case RISCV::PseudoVLE16_V_MF4:
191   case RISCV::PseudoVLE16_V_MF4_MASK:
192   case RISCV::PseudoVLSE16_V_M1:
193   case RISCV::PseudoVLSE16_V_M1_MASK:
194   case RISCV::PseudoVLSE16_V_M2:
195   case RISCV::PseudoVLSE16_V_M2_MASK:
196   case RISCV::PseudoVLSE16_V_M4:
197   case RISCV::PseudoVLSE16_V_M4_MASK:
198   case RISCV::PseudoVLSE16_V_M8:
199   case RISCV::PseudoVLSE16_V_M8_MASK:
200   case RISCV::PseudoVLSE16_V_MF2:
201   case RISCV::PseudoVLSE16_V_MF2_MASK:
202   case RISCV::PseudoVLSE16_V_MF4:
203   case RISCV::PseudoVLSE16_V_MF4_MASK:
204   case RISCV::PseudoVSE16_V_M1:
205   case RISCV::PseudoVSE16_V_M1_MASK:
206   case RISCV::PseudoVSE16_V_M2:
207   case RISCV::PseudoVSE16_V_M2_MASK:
208   case RISCV::PseudoVSE16_V_M4:
209   case RISCV::PseudoVSE16_V_M4_MASK:
210   case RISCV::PseudoVSE16_V_M8:
211   case RISCV::PseudoVSE16_V_M8_MASK:
212   case RISCV::PseudoVSE16_V_MF2:
213   case RISCV::PseudoVSE16_V_MF2_MASK:
214   case RISCV::PseudoVSE16_V_MF4:
215   case RISCV::PseudoVSE16_V_MF4_MASK:
216   case RISCV::PseudoVSSE16_V_M1:
217   case RISCV::PseudoVSSE16_V_M1_MASK:
218   case RISCV::PseudoVSSE16_V_M2:
219   case RISCV::PseudoVSSE16_V_M2_MASK:
220   case RISCV::PseudoVSSE16_V_M4:
221   case RISCV::PseudoVSSE16_V_M4_MASK:
222   case RISCV::PseudoVSSE16_V_M8:
223   case RISCV::PseudoVSSE16_V_M8_MASK:
224   case RISCV::PseudoVSSE16_V_MF2:
225   case RISCV::PseudoVSSE16_V_MF2_MASK:
226   case RISCV::PseudoVSSE16_V_MF4:
227   case RISCV::PseudoVSSE16_V_MF4_MASK:
228     return 16;
229   case RISCV::PseudoVLE32_V_M1:
230   case RISCV::PseudoVLE32_V_M1_MASK:
231   case RISCV::PseudoVLE32_V_M2:
232   case RISCV::PseudoVLE32_V_M2_MASK:
233   case RISCV::PseudoVLE32_V_M4:
234   case RISCV::PseudoVLE32_V_M4_MASK:
235   case RISCV::PseudoVLE32_V_M8:
236   case RISCV::PseudoVLE32_V_M8_MASK:
237   case RISCV::PseudoVLE32_V_MF2:
238   case RISCV::PseudoVLE32_V_MF2_MASK:
239   case RISCV::PseudoVLSE32_V_M1:
240   case RISCV::PseudoVLSE32_V_M1_MASK:
241   case RISCV::PseudoVLSE32_V_M2:
242   case RISCV::PseudoVLSE32_V_M2_MASK:
243   case RISCV::PseudoVLSE32_V_M4:
244   case RISCV::PseudoVLSE32_V_M4_MASK:
245   case RISCV::PseudoVLSE32_V_M8:
246   case RISCV::PseudoVLSE32_V_M8_MASK:
247   case RISCV::PseudoVLSE32_V_MF2:
248   case RISCV::PseudoVLSE32_V_MF2_MASK:
249   case RISCV::PseudoVSE32_V_M1:
250   case RISCV::PseudoVSE32_V_M1_MASK:
251   case RISCV::PseudoVSE32_V_M2:
252   case RISCV::PseudoVSE32_V_M2_MASK:
253   case RISCV::PseudoVSE32_V_M4:
254   case RISCV::PseudoVSE32_V_M4_MASK:
255   case RISCV::PseudoVSE32_V_M8:
256   case RISCV::PseudoVSE32_V_M8_MASK:
257   case RISCV::PseudoVSE32_V_MF2:
258   case RISCV::PseudoVSE32_V_MF2_MASK:
259   case RISCV::PseudoVSSE32_V_M1:
260   case RISCV::PseudoVSSE32_V_M1_MASK:
261   case RISCV::PseudoVSSE32_V_M2:
262   case RISCV::PseudoVSSE32_V_M2_MASK:
263   case RISCV::PseudoVSSE32_V_M4:
264   case RISCV::PseudoVSSE32_V_M4_MASK:
265   case RISCV::PseudoVSSE32_V_M8:
266   case RISCV::PseudoVSSE32_V_M8_MASK:
267   case RISCV::PseudoVSSE32_V_MF2:
268   case RISCV::PseudoVSSE32_V_MF2_MASK:
269     return 32;
270   case RISCV::PseudoVLE64_V_M1:
271   case RISCV::PseudoVLE64_V_M1_MASK:
272   case RISCV::PseudoVLE64_V_M2:
273   case RISCV::PseudoVLE64_V_M2_MASK:
274   case RISCV::PseudoVLE64_V_M4:
275   case RISCV::PseudoVLE64_V_M4_MASK:
276   case RISCV::PseudoVLE64_V_M8:
277   case RISCV::PseudoVLE64_V_M8_MASK:
278   case RISCV::PseudoVLSE64_V_M1:
279   case RISCV::PseudoVLSE64_V_M1_MASK:
280   case RISCV::PseudoVLSE64_V_M2:
281   case RISCV::PseudoVLSE64_V_M2_MASK:
282   case RISCV::PseudoVLSE64_V_M4:
283   case RISCV::PseudoVLSE64_V_M4_MASK:
284   case RISCV::PseudoVLSE64_V_M8:
285   case RISCV::PseudoVLSE64_V_M8_MASK:
286   case RISCV::PseudoVSE64_V_M1:
287   case RISCV::PseudoVSE64_V_M1_MASK:
288   case RISCV::PseudoVSE64_V_M2:
289   case RISCV::PseudoVSE64_V_M2_MASK:
290   case RISCV::PseudoVSE64_V_M4:
291   case RISCV::PseudoVSE64_V_M4_MASK:
292   case RISCV::PseudoVSE64_V_M8:
293   case RISCV::PseudoVSE64_V_M8_MASK:
294   case RISCV::PseudoVSSE64_V_M1:
295   case RISCV::PseudoVSSE64_V_M1_MASK:
296   case RISCV::PseudoVSSE64_V_M2:
297   case RISCV::PseudoVSSE64_V_M2_MASK:
298   case RISCV::PseudoVSSE64_V_M4:
299   case RISCV::PseudoVSSE64_V_M4_MASK:
300   case RISCV::PseudoVSSE64_V_M8:
301   case RISCV::PseudoVSSE64_V_M8_MASK:
302     return 64;
303   }
304 }
305 
306 /// Return true if this is an operation on mask registers.  Note that
307 /// this includes both arithmetic/logical ops and load/store (vlm/vsm).
308 static bool isMaskRegOp(const MachineInstr &MI) {
309   if (RISCVII::hasSEWOp(MI.getDesc().TSFlags)) {
310     const unsigned Log2SEW = MI.getOperand(getSEWOpNum(MI)).getImm();
311     // A Log2SEW of 0 is an operation on mask registers only.
312     return Log2SEW == 0;
313   }
314   return false;
315 }
316 
317 static unsigned getSEWLMULRatio(unsigned SEW, RISCVII::VLMUL VLMul) {
318   unsigned LMul;
319   bool Fractional;
320   std::tie(LMul, Fractional) = RISCVVType::decodeVLMUL(VLMul);
321 
322   // Convert LMul to a fixed point value with 3 fractional bits.
323   LMul = Fractional ? (8 / LMul) : (LMul * 8);
324 
325   assert(SEW >= 8 && "Unexpected SEW value");
326   return (SEW * 8) / LMul;
327 }
328 
329 /// Which subfields of VL or VTYPE have values we need to preserve?
330 struct DemandedFields {
331   bool VL = false;
332   bool SEW = false;
333   bool LMUL = false;
334   bool SEWLMULRatio = false;
335   bool TailPolicy = false;
336   bool MaskPolicy = false;
337 
338   // Return true if any part of VTYPE was used
339   bool usedVTYPE() {
340     return SEW || LMUL || SEWLMULRatio || TailPolicy || MaskPolicy;
341   }
342 
343   // Mark all VTYPE subfields and properties as demanded
344   void demandVTYPE() {
345     SEW = true;
346     LMUL = true;
347     SEWLMULRatio = true;
348     TailPolicy = true;
349     MaskPolicy = true;
350   }
351 };
352 
353 /// Return true if the two values of the VTYPE register provided are
354 /// indistinguishable from the perspective of an instruction (or set of
355 /// instructions) which use only the Used subfields and properties.
356 static bool areCompatibleVTYPEs(uint64_t VType1,
357                                 uint64_t VType2,
358                                 const DemandedFields &Used) {
359   if (Used.SEW &&
360       RISCVVType::getSEW(VType1) != RISCVVType::getSEW(VType2))
361     return false;
362 
363   if (Used.LMUL &&
364       RISCVVType::getVLMUL(VType1) != RISCVVType::getVLMUL(VType2))
365     return false;
366 
367   if (Used.SEWLMULRatio) {
368     auto Ratio1 = getSEWLMULRatio(RISCVVType::getSEW(VType1),
369                                   RISCVVType::getVLMUL(VType1));
370     auto Ratio2 = getSEWLMULRatio(RISCVVType::getSEW(VType2),
371                                   RISCVVType::getVLMUL(VType2));
372     if (Ratio1 != Ratio2)
373       return false;
374   }
375 
376   if (Used.TailPolicy &&
377       RISCVVType::isTailAgnostic(VType1) != RISCVVType::isTailAgnostic(VType2))
378     return false;
379   if (Used.MaskPolicy &&
380       RISCVVType::isMaskAgnostic(VType1) != RISCVVType::isMaskAgnostic(VType2))
381     return false;
382   return true;
383 }
384 
385 /// Return the fields and properties demanded by the provided instruction.
386 static DemandedFields getDemanded(const MachineInstr &MI) {
387   // Warning: This function has to work on both the lowered (i.e. post
388   // emitVSETVLIs) and pre-lowering forms.  The main implication of this is
389   // that it can't use the value of a SEW, VL, or Policy operand as they might
390   // be stale after lowering.
391 
392   // Most instructions don't use any of these subfeilds.
393   DemandedFields Res;
394   // Start conservative if registers are used
395   if (MI.isCall() || MI.isInlineAsm() || MI.readsRegister(RISCV::VL))
396     Res.VL = true;
397   if (MI.isCall() || MI.isInlineAsm() || MI.readsRegister(RISCV::VTYPE))
398     Res.demandVTYPE();
399   // Start conservative on the unlowered form too
400   uint64_t TSFlags = MI.getDesc().TSFlags;
401   if (RISCVII::hasSEWOp(TSFlags)) {
402     Res.demandVTYPE();
403     if (RISCVII::hasVLOp(TSFlags))
404       Res.VL = true;
405   }
406 
407   // Loads and stores with implicit EEW do not demand SEW or LMUL directly.
408   // They instead demand the ratio of the two which is used in computing
409   // EMUL, but which allows us the flexibility to change SEW and LMUL
410   // provided we don't change the ratio.
411   if (getEEWForLoadStore(MI)) {
412     Res.SEW = false;
413     Res.LMUL = false;
414   }
415 
416   // Store instructions don't use the policy fields.
417   if (RISCVII::hasSEWOp(TSFlags) && MI.getNumExplicitDefs() == 0) {
418     Res.TailPolicy = false;
419     Res.MaskPolicy = false;
420   }
421 
422   // A splat of 0/-1 is always a splat of 0/-1, regardless of etype.
423   // TODO: We're currently demanding VL + SEWLMULRatio which is sufficient
424   // but not neccessary.  What we really need is VLInBytes.
425   if (isSplatOfZeroOrMinusOne(MI)) {
426     Res.SEW = false;
427     Res.LMUL = false;
428   }
429 
430   // If this is a mask reg operation, it only cares about VLMAX.
431   // TODO: Possible extensions to this logic
432   // * Probably ok if available VLMax is larger than demanded
433   // * The policy bits can probably be ignored..
434   if (isMaskRegOp(MI)) {
435     Res.SEW = false;
436     Res.LMUL = false;
437   }
438 
439   return Res;
440 }
441 
442 /// Defines the abstract state with which the forward dataflow models the
443 /// values of the VL and VTYPE registers after insertion.
444 class VSETVLIInfo {
445   union {
446     Register AVLReg;
447     unsigned AVLImm;
448   };
449 
450   enum : uint8_t {
451     Uninitialized,
452     AVLIsReg,
453     AVLIsImm,
454     Unknown,
455   } State = Uninitialized;
456 
457   // Fields from VTYPE.
458   RISCVII::VLMUL VLMul = RISCVII::LMUL_1;
459   uint8_t SEW = 0;
460   uint8_t TailAgnostic : 1;
461   uint8_t MaskAgnostic : 1;
462   uint8_t SEWLMULRatioOnly : 1;
463 
464 public:
465   VSETVLIInfo()
466       : AVLImm(0), TailAgnostic(false), MaskAgnostic(false),
467         SEWLMULRatioOnly(false) {}
468 
469   static VSETVLIInfo getUnknown() {
470     VSETVLIInfo Info;
471     Info.setUnknown();
472     return Info;
473   }
474 
475   bool isValid() const { return State != Uninitialized; }
476   void setUnknown() { State = Unknown; }
477   bool isUnknown() const { return State == Unknown; }
478 
479   void setAVLReg(Register Reg) {
480     AVLReg = Reg;
481     State = AVLIsReg;
482   }
483 
484   void setAVLImm(unsigned Imm) {
485     AVLImm = Imm;
486     State = AVLIsImm;
487   }
488 
489   bool hasAVLImm() const { return State == AVLIsImm; }
490   bool hasAVLReg() const { return State == AVLIsReg; }
491   Register getAVLReg() const {
492     assert(hasAVLReg());
493     return AVLReg;
494   }
495   unsigned getAVLImm() const {
496     assert(hasAVLImm());
497     return AVLImm;
498   }
499 
500   unsigned getSEW() const { return SEW; }
501   RISCVII::VLMUL getVLMUL() const { return VLMul; }
502 
503   bool hasZeroAVL() const {
504     if (hasAVLImm())
505       return getAVLImm() == 0;
506     return false;
507   }
508   bool hasNonZeroAVL() const {
509     if (hasAVLImm())
510       return getAVLImm() > 0;
511     if (hasAVLReg())
512       return getAVLReg() == RISCV::X0;
513     return false;
514   }
515 
516   bool hasSameAVL(const VSETVLIInfo &Other) const {
517     assert(isValid() && Other.isValid() &&
518            "Can't compare invalid VSETVLIInfos");
519     assert(!isUnknown() && !Other.isUnknown() &&
520            "Can't compare AVL in unknown state");
521     if (hasAVLReg() && Other.hasAVLReg())
522       return getAVLReg() == Other.getAVLReg();
523 
524     if (hasAVLImm() && Other.hasAVLImm())
525       return getAVLImm() == Other.getAVLImm();
526 
527     return false;
528   }
529 
530   void setVTYPE(unsigned VType) {
531     assert(isValid() && !isUnknown() &&
532            "Can't set VTYPE for uninitialized or unknown");
533     VLMul = RISCVVType::getVLMUL(VType);
534     SEW = RISCVVType::getSEW(VType);
535     TailAgnostic = RISCVVType::isTailAgnostic(VType);
536     MaskAgnostic = RISCVVType::isMaskAgnostic(VType);
537   }
538   void setVTYPE(RISCVII::VLMUL L, unsigned S, bool TA, bool MA) {
539     assert(isValid() && !isUnknown() &&
540            "Can't set VTYPE for uninitialized or unknown");
541     VLMul = L;
542     SEW = S;
543     TailAgnostic = TA;
544     MaskAgnostic = MA;
545   }
546 
547   unsigned encodeVTYPE() const {
548     assert(isValid() && !isUnknown() && !SEWLMULRatioOnly &&
549            "Can't encode VTYPE for uninitialized or unknown");
550     return RISCVVType::encodeVTYPE(VLMul, SEW, TailAgnostic, MaskAgnostic);
551   }
552 
553   bool hasSEWLMULRatioOnly() const { return SEWLMULRatioOnly; }
554 
555   bool hasSameSEW(const VSETVLIInfo &Other) const {
556     assert(isValid() && Other.isValid() &&
557            "Can't compare invalid VSETVLIInfos");
558     assert(!isUnknown() && !Other.isUnknown() &&
559            "Can't compare VTYPE in unknown state");
560     assert(!SEWLMULRatioOnly && !Other.SEWLMULRatioOnly &&
561            "Can't compare when only LMUL/SEW ratio is valid.");
562     return SEW == Other.SEW;
563   }
564 
565   bool hasSameVTYPE(const VSETVLIInfo &Other) const {
566     assert(isValid() && Other.isValid() &&
567            "Can't compare invalid VSETVLIInfos");
568     assert(!isUnknown() && !Other.isUnknown() &&
569            "Can't compare VTYPE in unknown state");
570     assert(!SEWLMULRatioOnly && !Other.SEWLMULRatioOnly &&
571            "Can't compare when only LMUL/SEW ratio is valid.");
572     return std::tie(VLMul, SEW, TailAgnostic, MaskAgnostic) ==
573            std::tie(Other.VLMul, Other.SEW, Other.TailAgnostic,
574                     Other.MaskAgnostic);
575   }
576 
577   unsigned getSEWLMULRatio() const {
578     assert(isValid() && !isUnknown() &&
579            "Can't use VTYPE for uninitialized or unknown");
580     return ::getSEWLMULRatio(SEW, VLMul);
581   }
582 
583   // Check if the VTYPE for these two VSETVLIInfos produce the same VLMAX.
584   // Note that having the same VLMAX ensures that both share the same
585   // function from AVL to VL; that is, they must produce the same VL value
586   // for any given AVL value.
587   bool hasSameVLMAX(const VSETVLIInfo &Other) const {
588     assert(isValid() && Other.isValid() &&
589            "Can't compare invalid VSETVLIInfos");
590     assert(!isUnknown() && !Other.isUnknown() &&
591            "Can't compare VTYPE in unknown state");
592     return getSEWLMULRatio() == Other.getSEWLMULRatio();
593   }
594 
595   bool hasSamePolicy(const VSETVLIInfo &Other) const {
596     assert(isValid() && Other.isValid() &&
597            "Can't compare invalid VSETVLIInfos");
598     assert(!isUnknown() && !Other.isUnknown() &&
599            "Can't compare VTYPE in unknown state");
600     return TailAgnostic == Other.TailAgnostic &&
601            MaskAgnostic == Other.MaskAgnostic;
602   }
603 
604   bool hasCompatibleVTYPE(const MachineInstr &MI,
605                           const VSETVLIInfo &Require) const {
606     const DemandedFields Used = getDemanded(MI);
607     return areCompatibleVTYPEs(encodeVTYPE(), Require.encodeVTYPE(), Used);
608   }
609 
610   // Determine whether the vector instructions requirements represented by
611   // Require are compatible with the previous vsetvli instruction represented
612   // by this.  MI is the instruction whose requirements we're considering.
613   bool isCompatible(const MachineInstr &MI, const VSETVLIInfo &Require) const {
614     assert(isValid() && Require.isValid() &&
615            "Can't compare invalid VSETVLIInfos");
616     assert(!Require.SEWLMULRatioOnly &&
617            "Expected a valid VTYPE for instruction!");
618     // Nothing is compatible with Unknown.
619     if (isUnknown() || Require.isUnknown())
620       return false;
621 
622     // If only our VLMAX ratio is valid, then this isn't compatible.
623     if (SEWLMULRatioOnly)
624       return false;
625 
626     // If the instruction doesn't need an AVLReg and the SEW matches, consider
627     // it compatible.
628     if (Require.hasAVLReg() && Require.AVLReg == RISCV::NoRegister)
629       if (SEW == Require.SEW)
630         return true;
631 
632     return hasSameAVL(Require) && hasCompatibleVTYPE(MI, Require);
633   }
634 
635   bool operator==(const VSETVLIInfo &Other) const {
636     // Uninitialized is only equal to another Uninitialized.
637     if (!isValid())
638       return !Other.isValid();
639     if (!Other.isValid())
640       return !isValid();
641 
642     // Unknown is only equal to another Unknown.
643     if (isUnknown())
644       return Other.isUnknown();
645     if (Other.isUnknown())
646       return isUnknown();
647 
648     if (!hasSameAVL(Other))
649       return false;
650 
651     // If the SEWLMULRatioOnly bits are different, then they aren't equal.
652     if (SEWLMULRatioOnly != Other.SEWLMULRatioOnly)
653       return false;
654 
655     // If only the VLMAX is valid, check that it is the same.
656     if (SEWLMULRatioOnly)
657       return hasSameVLMAX(Other);
658 
659     // If the full VTYPE is valid, check that it is the same.
660     return hasSameVTYPE(Other);
661   }
662 
663   bool operator!=(const VSETVLIInfo &Other) const {
664     return !(*this == Other);
665   }
666 
667   // Calculate the VSETVLIInfo visible to a block assuming this and Other are
668   // both predecessors.
669   VSETVLIInfo intersect(const VSETVLIInfo &Other) const {
670     // If the new value isn't valid, ignore it.
671     if (!Other.isValid())
672       return *this;
673 
674     // If this value isn't valid, this must be the first predecessor, use it.
675     if (!isValid())
676       return Other;
677 
678     // If either is unknown, the result is unknown.
679     if (isUnknown() || Other.isUnknown())
680       return VSETVLIInfo::getUnknown();
681 
682     // If we have an exact, match return this.
683     if (*this == Other)
684       return *this;
685 
686     // Not an exact match, but maybe the AVL and VLMAX are the same. If so,
687     // return an SEW/LMUL ratio only value.
688     if (hasSameAVL(Other) && hasSameVLMAX(Other)) {
689       VSETVLIInfo MergeInfo = *this;
690       MergeInfo.SEWLMULRatioOnly = true;
691       return MergeInfo;
692     }
693 
694     // Otherwise the result is unknown.
695     return VSETVLIInfo::getUnknown();
696   }
697 
698 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
699   /// Support for debugging, callable in GDB: V->dump()
700   LLVM_DUMP_METHOD void dump() const {
701     print(dbgs());
702     dbgs() << "\n";
703   }
704 
705   /// Implement operator<<.
706   /// @{
707   void print(raw_ostream &OS) const {
708     OS << "{";
709     if (!isValid())
710       OS << "Uninitialized";
711     if (isUnknown())
712       OS << "unknown";;
713     if (hasAVLReg())
714       OS << "AVLReg=" << (unsigned)AVLReg;
715     if (hasAVLImm())
716       OS << "AVLImm=" << (unsigned)AVLImm;
717     OS << ", "
718        << "VLMul=" << (unsigned)VLMul << ", "
719        << "SEW=" << (unsigned)SEW << ", "
720        << "TailAgnostic=" << (bool)TailAgnostic << ", "
721        << "MaskAgnostic=" << (bool)MaskAgnostic << ", "
722        << "SEWLMULRatioOnly=" << (bool)SEWLMULRatioOnly << "}";
723   }
724 #endif
725 };
726 
727 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
728 LLVM_ATTRIBUTE_USED
729 inline raw_ostream &operator<<(raw_ostream &OS, const VSETVLIInfo &V) {
730   V.print(OS);
731   return OS;
732 }
733 #endif
734 
735 struct BlockData {
736   // The VSETVLIInfo that represents the net changes to the VL/VTYPE registers
737   // made by this block. Calculated in Phase 1.
738   VSETVLIInfo Change;
739 
740   // The VSETVLIInfo that represents the VL/VTYPE settings on exit from this
741   // block. Calculated in Phase 2.
742   VSETVLIInfo Exit;
743 
744   // The VSETVLIInfo that represents the VL/VTYPE settings from all predecessor
745   // blocks. Calculated in Phase 2, and used by Phase 3.
746   VSETVLIInfo Pred;
747 
748   // Keeps track of whether the block is already in the queue.
749   bool InQueue = false;
750 
751   BlockData() = default;
752 };
753 
754 class RISCVInsertVSETVLI : public MachineFunctionPass {
755   const TargetInstrInfo *TII;
756   MachineRegisterInfo *MRI;
757 
758   std::vector<BlockData> BlockInfo;
759   std::queue<const MachineBasicBlock *> WorkList;
760 
761 public:
762   static char ID;
763 
764   RISCVInsertVSETVLI() : MachineFunctionPass(ID) {
765     initializeRISCVInsertVSETVLIPass(*PassRegistry::getPassRegistry());
766   }
767   bool runOnMachineFunction(MachineFunction &MF) override;
768 
769   void getAnalysisUsage(AnalysisUsage &AU) const override {
770     AU.setPreservesCFG();
771     MachineFunctionPass::getAnalysisUsage(AU);
772   }
773 
774   StringRef getPassName() const override { return RISCV_INSERT_VSETVLI_NAME; }
775 
776 private:
777   bool needVSETVLI(const MachineInstr &MI, const VSETVLIInfo &Require,
778                    const VSETVLIInfo &CurInfo) const;
779   bool needVSETVLIPHI(const VSETVLIInfo &Require,
780                       const MachineBasicBlock &MBB) const;
781   void insertVSETVLI(MachineBasicBlock &MBB, MachineInstr &MI,
782                      const VSETVLIInfo &Info, const VSETVLIInfo &PrevInfo);
783   void insertVSETVLI(MachineBasicBlock &MBB,
784                      MachineBasicBlock::iterator InsertPt, DebugLoc DL,
785                      const VSETVLIInfo &Info, const VSETVLIInfo &PrevInfo);
786 
787   void transferBefore(VSETVLIInfo &Info, const MachineInstr &MI);
788   void transferAfter(VSETVLIInfo &Info, const MachineInstr &MI);
789   bool computeVLVTYPEChanges(const MachineBasicBlock &MBB);
790   void computeIncomingVLVTYPE(const MachineBasicBlock &MBB);
791   void emitVSETVLIs(MachineBasicBlock &MBB);
792   void doLocalPrepass(MachineBasicBlock &MBB);
793   void doLocalPostpass(MachineBasicBlock &MBB);
794   void doPRE(MachineBasicBlock &MBB);
795   void insertReadVL(MachineBasicBlock &MBB);
796 };
797 
798 } // end anonymous namespace
799 
800 char RISCVInsertVSETVLI::ID = 0;
801 
802 INITIALIZE_PASS(RISCVInsertVSETVLI, DEBUG_TYPE, RISCV_INSERT_VSETVLI_NAME,
803                 false, false)
804 
805 static bool isVectorConfigInstr(const MachineInstr &MI) {
806   return MI.getOpcode() == RISCV::PseudoVSETVLI ||
807          MI.getOpcode() == RISCV::PseudoVSETVLIX0 ||
808          MI.getOpcode() == RISCV::PseudoVSETIVLI;
809 }
810 
811 /// Return true if this is 'vsetvli x0, x0, vtype' which preserves
812 /// VL and only sets VTYPE.
813 static bool isVLPreservingConfig(const MachineInstr &MI) {
814   if (MI.getOpcode() != RISCV::PseudoVSETVLIX0)
815     return false;
816   assert(RISCV::X0 == MI.getOperand(1).getReg());
817   return RISCV::X0 == MI.getOperand(0).getReg();
818 }
819 
820 static VSETVLIInfo computeInfoForInstr(const MachineInstr &MI, uint64_t TSFlags,
821                                        const MachineRegisterInfo *MRI) {
822   VSETVLIInfo InstrInfo;
823 
824   // If the instruction has policy argument, use the argument.
825   // If there is no policy argument, default to tail agnostic unless the
826   // destination is tied to a source. Unless the source is undef. In that case
827   // the user would have some control over the policy values.
828   bool TailAgnostic = true;
829   bool UsesMaskPolicy = RISCVII::usesMaskPolicy(TSFlags);
830   // FIXME: Could we look at the above or below instructions to choose the
831   // matched mask policy to reduce vsetvli instructions? Default mask policy is
832   // agnostic if instructions use mask policy, otherwise is undisturbed. Because
833   // most mask operations are mask undisturbed, so we could possibly reduce the
834   // vsetvli between mask and nomasked instruction sequence.
835   bool MaskAgnostic = UsesMaskPolicy;
836   unsigned UseOpIdx;
837   if (RISCVII::hasVecPolicyOp(TSFlags)) {
838     const MachineOperand &Op = MI.getOperand(MI.getNumExplicitOperands() - 1);
839     uint64_t Policy = Op.getImm();
840     assert(Policy <= (RISCVII::TAIL_AGNOSTIC | RISCVII::MASK_AGNOSTIC) &&
841            "Invalid Policy Value");
842     // Although in some cases, mismatched passthru/maskedoff with policy value
843     // does not make sense (ex. tied operand is IMPLICIT_DEF with non-TAMA
844     // policy, or tied operand is not IMPLICIT_DEF with TAMA policy), but users
845     // have set the policy value explicitly, so compiler would not fix it.
846     TailAgnostic = Policy & RISCVII::TAIL_AGNOSTIC;
847     MaskAgnostic = Policy & RISCVII::MASK_AGNOSTIC;
848   } else if (MI.isRegTiedToUseOperand(0, &UseOpIdx)) {
849     TailAgnostic = false;
850     if (UsesMaskPolicy)
851       MaskAgnostic = false;
852     // If the tied operand is an IMPLICIT_DEF we can keep TailAgnostic.
853     const MachineOperand &UseMO = MI.getOperand(UseOpIdx);
854     MachineInstr *UseMI = MRI->getVRegDef(UseMO.getReg());
855     if (UseMI && UseMI->isImplicitDef()) {
856       TailAgnostic = true;
857       if (UsesMaskPolicy)
858         MaskAgnostic = true;
859     }
860     // Some pseudo instructions force a tail agnostic policy despite having a
861     // tied def.
862     if (RISCVII::doesForceTailAgnostic(TSFlags))
863       TailAgnostic = true;
864   }
865 
866   RISCVII::VLMUL VLMul = RISCVII::getLMul(TSFlags);
867 
868   unsigned Log2SEW = MI.getOperand(getSEWOpNum(MI)).getImm();
869   // A Log2SEW of 0 is an operation on mask registers only.
870   unsigned SEW = Log2SEW ? 1 << Log2SEW : 8;
871   assert(RISCVVType::isValidSEW(SEW) && "Unexpected SEW");
872 
873   if (RISCVII::hasVLOp(TSFlags)) {
874     const MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI));
875     if (VLOp.isImm()) {
876       int64_t Imm = VLOp.getImm();
877       // Conver the VLMax sentintel to X0 register.
878       if (Imm == RISCV::VLMaxSentinel)
879         InstrInfo.setAVLReg(RISCV::X0);
880       else
881         InstrInfo.setAVLImm(Imm);
882     } else {
883       InstrInfo.setAVLReg(VLOp.getReg());
884     }
885   } else {
886     InstrInfo.setAVLReg(RISCV::NoRegister);
887   }
888   InstrInfo.setVTYPE(VLMul, SEW, TailAgnostic, MaskAgnostic);
889 
890   return InstrInfo;
891 }
892 
893 void RISCVInsertVSETVLI::insertVSETVLI(MachineBasicBlock &MBB, MachineInstr &MI,
894                                        const VSETVLIInfo &Info,
895                                        const VSETVLIInfo &PrevInfo) {
896   DebugLoc DL = MI.getDebugLoc();
897   insertVSETVLI(MBB, MachineBasicBlock::iterator(&MI), DL, Info, PrevInfo);
898 }
899 
900 void RISCVInsertVSETVLI::insertVSETVLI(MachineBasicBlock &MBB,
901                      MachineBasicBlock::iterator InsertPt, DebugLoc DL,
902                      const VSETVLIInfo &Info, const VSETVLIInfo &PrevInfo) {
903 
904   // Use X0, X0 form if the AVL is the same and the SEW+LMUL gives the same
905   // VLMAX.
906   if (PrevInfo.isValid() && !PrevInfo.isUnknown() &&
907       Info.hasSameAVL(PrevInfo) && Info.hasSameVLMAX(PrevInfo)) {
908     BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETVLIX0))
909         .addReg(RISCV::X0, RegState::Define | RegState::Dead)
910         .addReg(RISCV::X0, RegState::Kill)
911         .addImm(Info.encodeVTYPE())
912         .addReg(RISCV::VL, RegState::Implicit);
913     return;
914   }
915 
916   if (Info.hasAVLImm()) {
917     BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETIVLI))
918         .addReg(RISCV::X0, RegState::Define | RegState::Dead)
919         .addImm(Info.getAVLImm())
920         .addImm(Info.encodeVTYPE());
921     return;
922   }
923 
924   Register AVLReg = Info.getAVLReg();
925   if (AVLReg == RISCV::NoRegister) {
926     // We can only use x0, x0 if there's no chance of the vtype change causing
927     // the previous vl to become invalid.
928     if (PrevInfo.isValid() && !PrevInfo.isUnknown() &&
929         Info.hasSameVLMAX(PrevInfo)) {
930       BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETVLIX0))
931           .addReg(RISCV::X0, RegState::Define | RegState::Dead)
932           .addReg(RISCV::X0, RegState::Kill)
933           .addImm(Info.encodeVTYPE())
934           .addReg(RISCV::VL, RegState::Implicit);
935       return;
936     }
937     // Otherwise use an AVL of 0 to avoid depending on previous vl.
938     BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETIVLI))
939         .addReg(RISCV::X0, RegState::Define | RegState::Dead)
940         .addImm(0)
941         .addImm(Info.encodeVTYPE());
942     return;
943   }
944 
945   if (AVLReg.isVirtual())
946     MRI->constrainRegClass(AVLReg, &RISCV::GPRNoX0RegClass);
947 
948   // Use X0 as the DestReg unless AVLReg is X0. We also need to change the
949   // opcode if the AVLReg is X0 as they have different register classes for
950   // the AVL operand.
951   Register DestReg = RISCV::X0;
952   unsigned Opcode = RISCV::PseudoVSETVLI;
953   if (AVLReg == RISCV::X0) {
954     DestReg = MRI->createVirtualRegister(&RISCV::GPRRegClass);
955     Opcode = RISCV::PseudoVSETVLIX0;
956   }
957   BuildMI(MBB, InsertPt, DL, TII->get(Opcode))
958       .addReg(DestReg, RegState::Define | RegState::Dead)
959       .addReg(AVLReg)
960       .addImm(Info.encodeVTYPE());
961 }
962 
963 // Return a VSETVLIInfo representing the changes made by this VSETVLI or
964 // VSETIVLI instruction.
965 static VSETVLIInfo getInfoForVSETVLI(const MachineInstr &MI) {
966   VSETVLIInfo NewInfo;
967   if (MI.getOpcode() == RISCV::PseudoVSETIVLI) {
968     NewInfo.setAVLImm(MI.getOperand(1).getImm());
969   } else {
970     assert(MI.getOpcode() == RISCV::PseudoVSETVLI ||
971            MI.getOpcode() == RISCV::PseudoVSETVLIX0);
972     Register AVLReg = MI.getOperand(1).getReg();
973     assert((AVLReg != RISCV::X0 || MI.getOperand(0).getReg() != RISCV::X0) &&
974            "Can't handle X0, X0 vsetvli yet");
975     NewInfo.setAVLReg(AVLReg);
976   }
977   NewInfo.setVTYPE(MI.getOperand(2).getImm());
978 
979   return NewInfo;
980 }
981 
982 /// Return true if a VSETVLI is required to transition from CurInfo to Require
983 /// before MI.
984 bool RISCVInsertVSETVLI::needVSETVLI(const MachineInstr &MI,
985                                      const VSETVLIInfo &Require,
986                                      const VSETVLIInfo &CurInfo) const {
987   assert(Require == computeInfoForInstr(MI, MI.getDesc().TSFlags, MRI));
988 
989   if (CurInfo.isCompatible(MI, Require))
990     return false;
991 
992   // For vmv.s.x and vfmv.s.f, there is only two behaviors, VL = 0 and VL > 0.
993   // So it's compatible when we could make sure that both VL be the same
994   // situation.  Additionally, if writing to an implicit_def operand, we
995   // don't need to preserve any other bits and are thus compatible with any
996   // larger etype, and can disregard policy bits.
997   if (isScalarMoveInstr(MI) &&
998       ((CurInfo.hasNonZeroAVL() && Require.hasNonZeroAVL()) ||
999        (CurInfo.hasZeroAVL() && Require.hasZeroAVL()))) {
1000     auto *VRegDef = MRI->getVRegDef(MI.getOperand(1).getReg());
1001     if (VRegDef && VRegDef->isImplicitDef() &&
1002         CurInfo.getSEW() >= Require.getSEW())
1003       return false;
1004     if (CurInfo.hasSameSEW(Require) && CurInfo.hasSamePolicy(Require))
1005       return false;
1006   }
1007 
1008   // We didn't find a compatible value. If our AVL is a virtual register,
1009   // it might be defined by a VSET(I)VLI. If it has the same VLMAX we need
1010   // and the last VL/VTYPE we observed is the same, we don't need a
1011   // VSETVLI here.
1012   if (!CurInfo.isUnknown() && Require.hasAVLReg() &&
1013       Require.getAVLReg().isVirtual() && !CurInfo.hasSEWLMULRatioOnly() &&
1014       CurInfo.hasCompatibleVTYPE(MI, Require)) {
1015     if (MachineInstr *DefMI = MRI->getVRegDef(Require.getAVLReg())) {
1016       if (isVectorConfigInstr(*DefMI)) {
1017         VSETVLIInfo DefInfo = getInfoForVSETVLI(*DefMI);
1018         if (DefInfo.hasSameAVL(CurInfo) && DefInfo.hasSameVLMAX(CurInfo))
1019           return false;
1020       }
1021     }
1022   }
1023 
1024   return true;
1025 }
1026 
1027 // Given an incoming state reaching MI, modifies that state so that it is minimally
1028 // compatible with MI.  The resulting state is guaranteed to be semantically legal
1029 // for MI, but may not be the state requested by MI.
1030 void RISCVInsertVSETVLI::transferBefore(VSETVLIInfo &Info, const MachineInstr &MI) {
1031   uint64_t TSFlags = MI.getDesc().TSFlags;
1032   if (!RISCVII::hasSEWOp(TSFlags))
1033     return;
1034   VSETVLIInfo NewInfo = computeInfoForInstr(MI, TSFlags, MRI);
1035 
1036   if (!Info.isValid()) {
1037     Info = NewInfo;
1038   } else {
1039     // If this instruction isn't compatible with the previous VL/VTYPE
1040     // we need to insert a VSETVLI.
1041     // NOTE: We only do this if the vtype we're comparing against was
1042     // created in this block. We need the first and third phase to treat
1043     // the store the same way.
1044     if (needVSETVLI(MI, NewInfo, Info))
1045       Info = NewInfo;
1046   }
1047 }
1048 
1049 // Given a state with which we evaluated MI (see transferBefore above for why
1050 // this might be different that the state MI requested), modify the state to
1051 // reflect the changes MI might make.
1052 void RISCVInsertVSETVLI::transferAfter(VSETVLIInfo &Info, const MachineInstr &MI) {
1053   if (isVectorConfigInstr(MI)) {
1054     Info = getInfoForVSETVLI(MI);
1055     return;
1056   }
1057 
1058   if (RISCV::isFaultFirstLoad(MI)) {
1059     // Update AVL to vl-output of the fault first load.
1060     Info.setAVLReg(MI.getOperand(1).getReg());
1061     return;
1062   }
1063 
1064   // If this is something that updates VL/VTYPE that we don't know about, set
1065   // the state to unknown.
1066   if (MI.isCall() || MI.isInlineAsm() || MI.modifiesRegister(RISCV::VL) ||
1067       MI.modifiesRegister(RISCV::VTYPE))
1068     Info = VSETVLIInfo::getUnknown();
1069 }
1070 
1071 bool RISCVInsertVSETVLI::computeVLVTYPEChanges(const MachineBasicBlock &MBB) {
1072   bool HadVectorOp = false;
1073 
1074   BlockData &BBInfo = BlockInfo[MBB.getNumber()];
1075   BBInfo.Change = BBInfo.Pred;
1076   for (const MachineInstr &MI : MBB) {
1077     transferBefore(BBInfo.Change, MI);
1078 
1079     if (isVectorConfigInstr(MI) || RISCVII::hasSEWOp(MI.getDesc().TSFlags))
1080       HadVectorOp = true;
1081 
1082     transferAfter(BBInfo.Change, MI);
1083   }
1084 
1085   return HadVectorOp;
1086 }
1087 
1088 void RISCVInsertVSETVLI::computeIncomingVLVTYPE(const MachineBasicBlock &MBB) {
1089 
1090   BlockData &BBInfo = BlockInfo[MBB.getNumber()];
1091 
1092   BBInfo.InQueue = false;
1093 
1094   VSETVLIInfo InInfo;
1095   if (MBB.pred_empty()) {
1096     // There are no predecessors, so use the default starting status.
1097     InInfo.setUnknown();
1098   } else {
1099     for (MachineBasicBlock *P : MBB.predecessors())
1100       InInfo = InInfo.intersect(BlockInfo[P->getNumber()].Exit);
1101   }
1102 
1103   // If we don't have any valid predecessor value, wait until we do.
1104   if (!InInfo.isValid())
1105     return;
1106 
1107   // If no change, no need to rerun block
1108   if (InInfo == BBInfo.Pred)
1109     return;
1110 
1111   BBInfo.Pred = InInfo;
1112   LLVM_DEBUG(dbgs() << "Entry state of " << printMBBReference(MBB)
1113                     << " changed to " << BBInfo.Pred << "\n");
1114 
1115   // Note: It's tempting to cache the state changes here, but due to the
1116   // compatibility checks performed a blocks output state can change based on
1117   // the input state.  To cache, we'd have to add logic for finding
1118   // never-compatible state changes.
1119   computeVLVTYPEChanges(MBB);
1120   VSETVLIInfo TmpStatus = BBInfo.Change;
1121 
1122   // If the new exit value matches the old exit value, we don't need to revisit
1123   // any blocks.
1124   if (BBInfo.Exit == TmpStatus)
1125     return;
1126 
1127   BBInfo.Exit = TmpStatus;
1128   LLVM_DEBUG(dbgs() << "Exit state of " << printMBBReference(MBB)
1129                     << " changed to " << BBInfo.Exit << "\n");
1130 
1131   // Add the successors to the work list so we can propagate the changed exit
1132   // status.
1133   for (MachineBasicBlock *S : MBB.successors())
1134     if (!BlockInfo[S->getNumber()].InQueue)
1135       WorkList.push(S);
1136 }
1137 
1138 // If we weren't able to prove a vsetvli was directly unneeded, it might still
1139 // be unneeded if the AVL is a phi node where all incoming values are VL
1140 // outputs from the last VSETVLI in their respective basic blocks.
1141 bool RISCVInsertVSETVLI::needVSETVLIPHI(const VSETVLIInfo &Require,
1142                                         const MachineBasicBlock &MBB) const {
1143   if (DisableInsertVSETVLPHIOpt)
1144     return true;
1145 
1146   if (!Require.hasAVLReg())
1147     return true;
1148 
1149   Register AVLReg = Require.getAVLReg();
1150   if (!AVLReg.isVirtual())
1151     return true;
1152 
1153   // We need the AVL to be produce by a PHI node in this basic block.
1154   MachineInstr *PHI = MRI->getVRegDef(AVLReg);
1155   if (!PHI || PHI->getOpcode() != RISCV::PHI || PHI->getParent() != &MBB)
1156     return true;
1157 
1158   for (unsigned PHIOp = 1, NumOps = PHI->getNumOperands(); PHIOp != NumOps;
1159        PHIOp += 2) {
1160     Register InReg = PHI->getOperand(PHIOp).getReg();
1161     MachineBasicBlock *PBB = PHI->getOperand(PHIOp + 1).getMBB();
1162     const BlockData &PBBInfo = BlockInfo[PBB->getNumber()];
1163     // If the exit from the predecessor has the VTYPE we are looking for
1164     // we might be able to avoid a VSETVLI.
1165     if (PBBInfo.Exit.isUnknown() || !PBBInfo.Exit.hasSameVTYPE(Require))
1166       return true;
1167 
1168     // We need the PHI input to the be the output of a VSET(I)VLI.
1169     MachineInstr *DefMI = MRI->getVRegDef(InReg);
1170     if (!DefMI || !isVectorConfigInstr(*DefMI))
1171       return true;
1172 
1173     // We found a VSET(I)VLI make sure it matches the output of the
1174     // predecessor block.
1175     VSETVLIInfo DefInfo = getInfoForVSETVLI(*DefMI);
1176     if (!DefInfo.hasSameAVL(PBBInfo.Exit) ||
1177         !DefInfo.hasSameVTYPE(PBBInfo.Exit))
1178       return true;
1179   }
1180 
1181   // If all the incoming values to the PHI checked out, we don't need
1182   // to insert a VSETVLI.
1183   return false;
1184 }
1185 
1186 void RISCVInsertVSETVLI::emitVSETVLIs(MachineBasicBlock &MBB) {
1187   VSETVLIInfo CurInfo = BlockInfo[MBB.getNumber()].Pred;
1188   // Track whether the prefix of the block we've scanned is transparent
1189   // (meaning has not yet changed the abstract state).
1190   bool PrefixTransparent = true;
1191   for (MachineInstr &MI : MBB) {
1192     const VSETVLIInfo PrevInfo = CurInfo;
1193     transferBefore(CurInfo, MI);
1194 
1195     // If this is an explicit VSETVLI or VSETIVLI, update our state.
1196     if (isVectorConfigInstr(MI)) {
1197       // Conservatively, mark the VL and VTYPE as live.
1198       assert(MI.getOperand(3).getReg() == RISCV::VL &&
1199              MI.getOperand(4).getReg() == RISCV::VTYPE &&
1200              "Unexpected operands where VL and VTYPE should be");
1201       MI.getOperand(3).setIsDead(false);
1202       MI.getOperand(4).setIsDead(false);
1203       PrefixTransparent = false;
1204     }
1205 
1206     uint64_t TSFlags = MI.getDesc().TSFlags;
1207     if (RISCVII::hasSEWOp(TSFlags)) {
1208       if (PrevInfo != CurInfo) {
1209         // If this is the first implicit state change, and the state change
1210         // requested can be proven to produce the same register contents, we
1211         // can skip emitting the actual state change and continue as if we
1212         // had since we know the GPR result of the implicit state change
1213         // wouldn't be used and VL/VTYPE registers are correct.  Note that
1214         // we *do* need to model the state as if it changed as while the
1215         // register contents are unchanged, the abstract model can change.
1216         if (!PrefixTransparent || needVSETVLIPHI(CurInfo, MBB))
1217           insertVSETVLI(MBB, MI, CurInfo, PrevInfo);
1218         PrefixTransparent = false;
1219       }
1220 
1221       if (RISCVII::hasVLOp(TSFlags)) {
1222         MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI));
1223         if (VLOp.isReg()) {
1224           // Erase the AVL operand from the instruction.
1225           VLOp.setReg(RISCV::NoRegister);
1226           VLOp.setIsKill(false);
1227         }
1228         MI.addOperand(MachineOperand::CreateReg(RISCV::VL, /*isDef*/ false,
1229                                                 /*isImp*/ true));
1230       }
1231       MI.addOperand(MachineOperand::CreateReg(RISCV::VTYPE, /*isDef*/ false,
1232                                               /*isImp*/ true));
1233     }
1234 
1235     if (MI.isCall() || MI.isInlineAsm() || MI.modifiesRegister(RISCV::VL) ||
1236         MI.modifiesRegister(RISCV::VTYPE))
1237       PrefixTransparent = false;
1238 
1239     transferAfter(CurInfo, MI);
1240   }
1241 
1242   // If we reach the end of the block and our current info doesn't match the
1243   // expected info, insert a vsetvli to correct.
1244   if (!UseStrictAsserts) {
1245     const VSETVLIInfo &ExitInfo = BlockInfo[MBB.getNumber()].Exit;
1246     if (CurInfo.isValid() && ExitInfo.isValid() && !ExitInfo.isUnknown() &&
1247         CurInfo != ExitInfo) {
1248       // Note there's an implicit assumption here that terminators never use
1249       // or modify VL or VTYPE.  Also, fallthrough will return end().
1250       auto InsertPt = MBB.getFirstInstrTerminator();
1251       insertVSETVLI(MBB, InsertPt, MBB.findDebugLoc(InsertPt), ExitInfo,
1252                     CurInfo);
1253       CurInfo = ExitInfo;
1254     }
1255   }
1256 
1257   if (UseStrictAsserts && CurInfo.isValid()) {
1258     const auto &Info = BlockInfo[MBB.getNumber()];
1259     if (CurInfo != Info.Exit) {
1260       LLVM_DEBUG(dbgs() << "in block " << printMBBReference(MBB) << "\n");
1261       LLVM_DEBUG(dbgs() << "  begin        state: " << Info.Pred << "\n");
1262       LLVM_DEBUG(dbgs() << "  expected end state: " << Info.Exit << "\n");
1263       LLVM_DEBUG(dbgs() << "  actual   end state: " << CurInfo << "\n");
1264     }
1265     assert(CurInfo == Info.Exit &&
1266            "InsertVSETVLI dataflow invariant violated");
1267   }
1268 }
1269 
1270 void RISCVInsertVSETVLI::doLocalPrepass(MachineBasicBlock &MBB) {
1271   VSETVLIInfo CurInfo = VSETVLIInfo::getUnknown();
1272   for (MachineInstr &MI : MBB) {
1273     // If this is an explicit VSETVLI or VSETIVLI, update our state.
1274     if (isVectorConfigInstr(MI)) {
1275       CurInfo = getInfoForVSETVLI(MI);
1276       continue;
1277     }
1278 
1279     const uint64_t TSFlags = MI.getDesc().TSFlags;
1280     if (isScalarMoveInstr(MI)) {
1281       assert(RISCVII::hasSEWOp(TSFlags) && RISCVII::hasVLOp(TSFlags));
1282       const VSETVLIInfo NewInfo = computeInfoForInstr(MI, TSFlags, MRI);
1283 
1284       // For vmv.s.x and vfmv.s.f, there are only two behaviors, VL = 0 and
1285       // VL > 0. We can discard the user requested AVL and just use the last
1286       // one if we can prove it equally zero.  This removes a vsetvli entirely
1287       // if the types match or allows use of cheaper avl preserving variant
1288       // if VLMAX doesn't change.  If VLMAX might change, we couldn't use
1289       // the 'vsetvli x0, x0, vtype" variant, so we avoid the transform to
1290       // prevent extending live range of an avl register operand.
1291       // TODO: We can probably relax this for immediates.
1292       if (((CurInfo.hasNonZeroAVL() && NewInfo.hasNonZeroAVL()) ||
1293            (CurInfo.hasZeroAVL() && NewInfo.hasZeroAVL())) &&
1294           NewInfo.hasSameVLMAX(CurInfo)) {
1295         MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI));
1296         if (CurInfo.hasAVLImm())
1297           VLOp.ChangeToImmediate(CurInfo.getAVLImm());
1298         else
1299           VLOp.ChangeToRegister(CurInfo.getAVLReg(), /*IsDef*/ false);
1300         CurInfo = computeInfoForInstr(MI, TSFlags, MRI);
1301         continue;
1302       }
1303     }
1304 
1305     if (RISCVII::hasSEWOp(TSFlags)) {
1306       if (RISCVII::hasVLOp(TSFlags)) {
1307         const auto Require = computeInfoForInstr(MI, TSFlags, MRI);
1308         // Two cases involving an AVL resulting from a previous vsetvli.
1309         // 1) If the AVL is the result of a previous vsetvli which has the
1310         //    same AVL and VLMAX as our current state, we can reuse the AVL
1311         //    from the current state for the new one.  This allows us to
1312         //    generate 'vsetvli x0, x0, vtype" or possible skip the transition
1313         //    entirely.
1314         // 2) If AVL is defined by a vsetvli with the same VLMAX, we can
1315         //    replace the AVL operand with the AVL of the defining vsetvli.
1316         //    We avoid general register AVLs to avoid extending live ranges
1317         //    without being sure we can kill the original source reg entirely.
1318         if (Require.hasAVLReg() && Require.getAVLReg().isVirtual()) {
1319           if (MachineInstr *DefMI = MRI->getVRegDef(Require.getAVLReg())) {
1320             if (isVectorConfigInstr(*DefMI)) {
1321               VSETVLIInfo DefInfo = getInfoForVSETVLI(*DefMI);
1322               // case 1
1323               if (!CurInfo.isUnknown() && DefInfo.hasSameAVL(CurInfo) &&
1324                   DefInfo.hasSameVLMAX(CurInfo)) {
1325                 MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI));
1326                 if (CurInfo.hasAVLImm())
1327                   VLOp.ChangeToImmediate(CurInfo.getAVLImm());
1328                 else {
1329                   MRI->clearKillFlags(CurInfo.getAVLReg());
1330                   VLOp.ChangeToRegister(CurInfo.getAVLReg(), /*IsDef*/ false);
1331                 }
1332                 CurInfo = computeInfoForInstr(MI, TSFlags, MRI);
1333                 continue;
1334               }
1335               // case 2
1336               if (DefInfo.hasSameVLMAX(Require) &&
1337                   (DefInfo.hasAVLImm() || DefInfo.getAVLReg() == RISCV::X0)) {
1338                 MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI));
1339                 if (DefInfo.hasAVLImm())
1340                   VLOp.ChangeToImmediate(DefInfo.getAVLImm());
1341                 else
1342                   VLOp.ChangeToRegister(DefInfo.getAVLReg(), /*IsDef*/ false);
1343                 CurInfo = computeInfoForInstr(MI, TSFlags, MRI);
1344                 continue;
1345               }
1346             }
1347           }
1348         }
1349       }
1350       CurInfo = computeInfoForInstr(MI, TSFlags, MRI);
1351       continue;
1352     }
1353 
1354     transferAfter(CurInfo, MI);
1355   }
1356 }
1357 
1358 /// Return true if the VL value configured must be equal to the requested one.
1359 static bool hasFixedResult(const VSETVLIInfo &Info, const RISCVSubtarget &ST) {
1360   if (!Info.hasAVLImm())
1361     // VLMAX is always the same value.
1362     // TODO: Could extend to other registers by looking at the associated vreg
1363     // def placement.
1364     return RISCV::X0 == Info.getAVLReg();
1365 
1366   unsigned AVL = Info.getAVLImm();
1367   unsigned SEW = Info.getSEW();
1368   unsigned AVLInBits = AVL * SEW;
1369 
1370   unsigned LMul;
1371   bool Fractional;
1372   std::tie(LMul, Fractional) = RISCVVType::decodeVLMUL(Info.getVLMUL());
1373 
1374   if (Fractional)
1375     return ST.getRealMinVLen() / LMul >= AVLInBits;
1376   return ST.getRealMinVLen() * LMul >= AVLInBits;
1377 }
1378 
1379 /// Perform simple partial redundancy elimination of the VSETVLI instructions
1380 /// we're about to insert by looking for cases where we can PRE from the
1381 /// beginning of one block to the end of one of its predecessors.  Specifically,
1382 /// this is geared to catch the common case of a fixed length vsetvl in a single
1383 /// block loop when it could execute once in the preheader instead.
1384 void RISCVInsertVSETVLI::doPRE(MachineBasicBlock &MBB) {
1385   const MachineFunction &MF = *MBB.getParent();
1386   const RISCVSubtarget &ST = MF.getSubtarget<RISCVSubtarget>();
1387 
1388   if (!BlockInfo[MBB.getNumber()].Pred.isUnknown())
1389     return;
1390 
1391   MachineBasicBlock *UnavailablePred = nullptr;
1392   VSETVLIInfo AvailableInfo;
1393   for (MachineBasicBlock *P : MBB.predecessors()) {
1394     const VSETVLIInfo &PredInfo = BlockInfo[P->getNumber()].Exit;
1395     if (PredInfo.isUnknown()) {
1396       if (UnavailablePred)
1397         return;
1398       UnavailablePred = P;
1399     } else if (!AvailableInfo.isValid()) {
1400       AvailableInfo = PredInfo;
1401     } else if (AvailableInfo != PredInfo) {
1402       return;
1403     }
1404   }
1405 
1406   // Unreachable, single pred, or full redundancy. Note that FRE is handled by
1407   // phase 3.
1408   if (!UnavailablePred || !AvailableInfo.isValid())
1409     return;
1410 
1411   // Critical edge - TODO: consider splitting?
1412   if (UnavailablePred->succ_size() != 1)
1413     return;
1414 
1415   // If VL can be less than AVL, then we can't reduce the frequency of exec.
1416   if (!hasFixedResult(AvailableInfo, ST))
1417     return;
1418 
1419   // Does it actually let us remove an implicit transition in MBB?
1420   bool Found = false;
1421   for (auto &MI : MBB) {
1422     if (isVectorConfigInstr(MI))
1423       return;
1424 
1425     const uint64_t TSFlags = MI.getDesc().TSFlags;
1426     if (RISCVII::hasSEWOp(TSFlags)) {
1427       if (AvailableInfo != computeInfoForInstr(MI, TSFlags, MRI))
1428         return;
1429       Found = true;
1430       break;
1431     }
1432   }
1433   if (!Found)
1434     return;
1435 
1436   // Finally, update both data flow state and insert the actual vsetvli.
1437   // Doing both keeps the code in sync with the dataflow results, which
1438   // is critical for correctness of phase 3.
1439   auto OldInfo = BlockInfo[UnavailablePred->getNumber()].Exit;
1440   LLVM_DEBUG(dbgs() << "PRE VSETVLI from " << MBB.getName() << " to "
1441                     << UnavailablePred->getName() << " with state "
1442                     << AvailableInfo << "\n");
1443   BlockInfo[UnavailablePred->getNumber()].Exit = AvailableInfo;
1444   BlockInfo[MBB.getNumber()].Pred = AvailableInfo;
1445 
1446   // Note there's an implicit assumption here that terminators never use
1447   // or modify VL or VTYPE.  Also, fallthrough will return end().
1448   auto InsertPt = UnavailablePred->getFirstInstrTerminator();
1449   insertVSETVLI(*UnavailablePred, InsertPt,
1450                 UnavailablePred->findDebugLoc(InsertPt),
1451                 AvailableInfo, OldInfo);
1452 }
1453 
1454 static void doUnion(DemandedFields &A, DemandedFields B) {
1455   A.VL |= B.VL;
1456   A.SEW |= B.SEW;
1457   A.LMUL |= B.LMUL;
1458   A.SEWLMULRatio |= B.SEWLMULRatio;
1459   A.TailPolicy |= B.TailPolicy;
1460   A.MaskPolicy |= B.MaskPolicy;
1461 }
1462 
1463 // Return true if we can mutate PrevMI's VTYPE to match MI's
1464 // without changing any the fields which have been used.
1465 // TODO: Restructure code to allow code reuse between this and isCompatible
1466 // above.
1467 static bool canMutatePriorConfig(const MachineInstr &PrevMI,
1468                                  const MachineInstr &MI,
1469                                  const DemandedFields &Used) {
1470   // TODO: Extend this to handle cases where VL does change, but VL
1471   // has not been used.  (e.g. over a vmv.x.s)
1472   if (!isVLPreservingConfig(MI))
1473     // Note: `vsetvli x0, x0, vtype' is the canonical instruction
1474     // for this case.  If you find yourself wanting to add other forms
1475     // to this "unused VTYPE" case, we're probably missing a
1476     // canonicalization earlier.
1477     return false;
1478 
1479   if (!PrevMI.getOperand(2).isImm() || !MI.getOperand(2).isImm())
1480     return false;
1481 
1482   auto PriorVType = PrevMI.getOperand(2).getImm();
1483   auto VType = MI.getOperand(2).getImm();
1484   return areCompatibleVTYPEs(PriorVType, VType, Used);
1485 }
1486 
1487 void RISCVInsertVSETVLI::doLocalPostpass(MachineBasicBlock &MBB) {
1488   MachineInstr *PrevMI = nullptr;
1489   DemandedFields Used;
1490   SmallVector<MachineInstr*> ToDelete;
1491   for (MachineInstr &MI : MBB) {
1492     // Note: Must be *before* vsetvli handling to account for config cases
1493     // which only change some subfields.
1494     doUnion(Used, getDemanded(MI));
1495 
1496     if (!isVectorConfigInstr(MI))
1497       continue;
1498 
1499     if (PrevMI) {
1500       if (!Used.VL && !Used.usedVTYPE()) {
1501         ToDelete.push_back(PrevMI);
1502         // fallthrough
1503       } else if (canMutatePriorConfig(*PrevMI, MI, Used)) {
1504         PrevMI->getOperand(2).setImm(MI.getOperand(2).getImm());
1505         ToDelete.push_back(&MI);
1506         // Leave PrevMI unchanged
1507         continue;
1508       }
1509     }
1510     PrevMI = &MI;
1511     Used = getDemanded(MI);
1512     Register VRegDef = MI.getOperand(0).getReg();
1513     if (VRegDef != RISCV::X0 &&
1514         !(VRegDef.isVirtual() && MRI->use_nodbg_empty(VRegDef)))
1515       Used.VL = true;
1516   }
1517 
1518   for (auto *MI : ToDelete)
1519     MI->eraseFromParent();
1520 }
1521 
1522 void RISCVInsertVSETVLI::insertReadVL(MachineBasicBlock &MBB) {
1523   for (auto I = MBB.begin(), E = MBB.end(); I != E;) {
1524     MachineInstr &MI = *I++;
1525     if (RISCV::isFaultFirstLoad(MI)) {
1526       Register VLOutput = MI.getOperand(1).getReg();
1527       if (!MRI->use_nodbg_empty(VLOutput))
1528         BuildMI(MBB, I, MI.getDebugLoc(), TII->get(RISCV::PseudoReadVL),
1529                 VLOutput);
1530       // We don't use the vl output of the VLEFF/VLSEGFF anymore.
1531       MI.getOperand(1).setReg(RISCV::X0);
1532     }
1533   }
1534 }
1535 
1536 bool RISCVInsertVSETVLI::runOnMachineFunction(MachineFunction &MF) {
1537   // Skip if the vector extension is not enabled.
1538   const RISCVSubtarget &ST = MF.getSubtarget<RISCVSubtarget>();
1539   if (!ST.hasVInstructions())
1540     return false;
1541 
1542   LLVM_DEBUG(dbgs() << "Entering InsertVSETVLI for " << MF.getName() << "\n");
1543 
1544   TII = ST.getInstrInfo();
1545   MRI = &MF.getRegInfo();
1546 
1547   assert(BlockInfo.empty() && "Expect empty block infos");
1548   BlockInfo.resize(MF.getNumBlockIDs());
1549 
1550   // Scan the block locally for cases where we can mutate the operands
1551   // of the instructions to reduce state transitions.  Critically, this
1552   // must be done before we start propagating data flow states as these
1553   // transforms are allowed to change the contents of VTYPE and VL so
1554   // long as the semantics of the program stays the same.
1555   for (MachineBasicBlock &MBB : MF)
1556     doLocalPrepass(MBB);
1557 
1558   bool HaveVectorOp = false;
1559 
1560   // Phase 1 - determine how VL/VTYPE are affected by the each block.
1561   for (const MachineBasicBlock &MBB : MF) {
1562     HaveVectorOp |= computeVLVTYPEChanges(MBB);
1563     // Initial exit state is whatever change we found in the block.
1564     BlockData &BBInfo = BlockInfo[MBB.getNumber()];
1565     BBInfo.Exit = BBInfo.Change;
1566     LLVM_DEBUG(dbgs() << "Initial exit state of " << printMBBReference(MBB)
1567                       << " is " << BBInfo.Exit << "\n");
1568 
1569   }
1570 
1571   // If we didn't find any instructions that need VSETVLI, we're done.
1572   if (!HaveVectorOp) {
1573     BlockInfo.clear();
1574     return false;
1575   }
1576 
1577   // Phase 2 - determine the exit VL/VTYPE from each block. We add all
1578   // blocks to the list here, but will also add any that need to be revisited
1579   // during Phase 2 processing.
1580   for (const MachineBasicBlock &MBB : MF) {
1581     WorkList.push(&MBB);
1582     BlockInfo[MBB.getNumber()].InQueue = true;
1583   }
1584   while (!WorkList.empty()) {
1585     const MachineBasicBlock &MBB = *WorkList.front();
1586     WorkList.pop();
1587     computeIncomingVLVTYPE(MBB);
1588   }
1589 
1590   // Perform partial redundancy elimination of vsetvli transitions.
1591   for (MachineBasicBlock &MBB : MF)
1592     doPRE(MBB);
1593 
1594   // Phase 3 - add any vsetvli instructions needed in the block. Use the
1595   // Phase 2 information to avoid adding vsetvlis before the first vector
1596   // instruction in the block if the VL/VTYPE is satisfied by its
1597   // predecessors.
1598   for (MachineBasicBlock &MBB : MF)
1599     emitVSETVLIs(MBB);
1600 
1601   // Now that all vsetvlis are explicit, go through and do block local
1602   // DSE and peephole based demanded fields based transforms.  Note that
1603   // this *must* be done outside the main dataflow so long as we allow
1604   // any cross block analysis within the dataflow.  We can't have both
1605   // demanded fields based mutation and non-local analysis in the
1606   // dataflow at the same time without introducing inconsistencies.
1607   for (MachineBasicBlock &MBB : MF)
1608     doLocalPostpass(MBB);
1609 
1610   // Once we're fully done rewriting all the instructions, do a final pass
1611   // through to check for VSETVLIs which write to an unused destination.
1612   // For the non X0, X0 variant, we can replace the destination register
1613   // with X0 to reduce register pressure.  This is really a generic
1614   // optimization which can be applied to any dead def (TODO: generalize).
1615   for (MachineBasicBlock &MBB : MF) {
1616     for (MachineInstr &MI : MBB) {
1617       if (MI.getOpcode() == RISCV::PseudoVSETVLI ||
1618           MI.getOpcode() == RISCV::PseudoVSETIVLI) {
1619         Register VRegDef = MI.getOperand(0).getReg();
1620         if (VRegDef != RISCV::X0 && MRI->use_nodbg_empty(VRegDef))
1621           MI.getOperand(0).setReg(RISCV::X0);
1622       }
1623     }
1624   }
1625 
1626   // Insert PseudoReadVL after VLEFF/VLSEGFF and replace it with the vl output
1627   // of VLEFF/VLSEGFF.
1628   for (MachineBasicBlock &MBB : MF)
1629     insertReadVL(MBB);
1630 
1631   BlockInfo.clear();
1632   return HaveVectorOp;
1633 }
1634 
1635 /// Returns an instance of the Insert VSETVLI pass.
1636 FunctionPass *llvm::createRISCVInsertVSETVLIPass() {
1637   return new RISCVInsertVSETVLI();
1638 }
1639