1 //===- RISCVInsertVSETVLI.cpp - Insert VSETVLI instructions ---------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This file implements a function pass that inserts VSETVLI instructions where 10 // needed and expands the vl outputs of VLEFF/VLSEGFF to PseudoReadVL 11 // instructions. 12 // 13 // This pass consists of 3 phases: 14 // 15 // Phase 1 collects how each basic block affects VL/VTYPE. 16 // 17 // Phase 2 uses the information from phase 1 to do a data flow analysis to 18 // propagate the VL/VTYPE changes through the function. This gives us the 19 // VL/VTYPE at the start of each basic block. 20 // 21 // Phase 3 inserts VSETVLI instructions in each basic block. Information from 22 // phase 2 is used to prevent inserting a VSETVLI before the first vector 23 // instruction in the block if possible. 24 // 25 //===----------------------------------------------------------------------===// 26 27 #include "RISCV.h" 28 #include "RISCVSubtarget.h" 29 #include "llvm/CodeGen/LiveIntervals.h" 30 #include "llvm/CodeGen/MachineFunctionPass.h" 31 #include <queue> 32 using namespace llvm; 33 34 #define DEBUG_TYPE "riscv-insert-vsetvli" 35 #define RISCV_INSERT_VSETVLI_NAME "RISCV Insert VSETVLI pass" 36 37 static cl::opt<bool> DisableInsertVSETVLPHIOpt( 38 "riscv-disable-insert-vsetvl-phi-opt", cl::init(false), cl::Hidden, 39 cl::desc("Disable looking through phis when inserting vsetvlis.")); 40 41 static cl::opt<bool> UseStrictAsserts( 42 "riscv-insert-vsetvl-strict-asserts", cl::init(true), cl::Hidden, 43 cl::desc("Enable strict assertion checking for the dataflow algorithm")); 44 45 namespace { 46 47 static unsigned getVLOpNum(const MachineInstr &MI) { 48 return RISCVII::getVLOpNum(MI.getDesc()); 49 } 50 51 static unsigned getSEWOpNum(const MachineInstr &MI) { 52 return RISCVII::getSEWOpNum(MI.getDesc()); 53 } 54 55 static bool isScalarMoveInstr(const MachineInstr &MI) { 56 switch (MI.getOpcode()) { 57 default: 58 return false; 59 case RISCV::PseudoVMV_S_X_M1: 60 case RISCV::PseudoVMV_S_X_M2: 61 case RISCV::PseudoVMV_S_X_M4: 62 case RISCV::PseudoVMV_S_X_M8: 63 case RISCV::PseudoVMV_S_X_MF2: 64 case RISCV::PseudoVMV_S_X_MF4: 65 case RISCV::PseudoVMV_S_X_MF8: 66 case RISCV::PseudoVFMV_S_F16_M1: 67 case RISCV::PseudoVFMV_S_F16_M2: 68 case RISCV::PseudoVFMV_S_F16_M4: 69 case RISCV::PseudoVFMV_S_F16_M8: 70 case RISCV::PseudoVFMV_S_F16_MF2: 71 case RISCV::PseudoVFMV_S_F16_MF4: 72 case RISCV::PseudoVFMV_S_F32_M1: 73 case RISCV::PseudoVFMV_S_F32_M2: 74 case RISCV::PseudoVFMV_S_F32_M4: 75 case RISCV::PseudoVFMV_S_F32_M8: 76 case RISCV::PseudoVFMV_S_F32_MF2: 77 case RISCV::PseudoVFMV_S_F64_M1: 78 case RISCV::PseudoVFMV_S_F64_M2: 79 case RISCV::PseudoVFMV_S_F64_M4: 80 case RISCV::PseudoVFMV_S_F64_M8: 81 return true; 82 } 83 } 84 85 static bool isSplatMoveInstr(const MachineInstr &MI) { 86 switch (MI.getOpcode()) { 87 default: 88 return false; 89 case RISCV::PseudoVMV_V_X_M1: 90 case RISCV::PseudoVMV_V_X_M2: 91 case RISCV::PseudoVMV_V_X_M4: 92 case RISCV::PseudoVMV_V_X_M8: 93 case RISCV::PseudoVMV_V_X_MF2: 94 case RISCV::PseudoVMV_V_X_MF4: 95 case RISCV::PseudoVMV_V_X_MF8: 96 case RISCV::PseudoVMV_V_I_M1: 97 case RISCV::PseudoVMV_V_I_M2: 98 case RISCV::PseudoVMV_V_I_M4: 99 case RISCV::PseudoVMV_V_I_M8: 100 case RISCV::PseudoVMV_V_I_MF2: 101 case RISCV::PseudoVMV_V_I_MF4: 102 case RISCV::PseudoVMV_V_I_MF8: 103 return true; 104 } 105 } 106 107 static bool isSplatOfZeroOrMinusOne(const MachineInstr &MI) { 108 if (!isSplatMoveInstr(MI)) 109 return false; 110 111 const MachineOperand &SrcMO = MI.getOperand(1); 112 if (SrcMO.isImm()) 113 return SrcMO.getImm() == 0 || SrcMO.getImm() == -1; 114 return SrcMO.isReg() && SrcMO.getReg() == RISCV::X0; 115 } 116 117 /// Get the EEW for a load or store instruction. Return None if MI is not 118 /// a load or store which ignores SEW. 119 static Optional<unsigned> getEEWForLoadStore(const MachineInstr &MI) { 120 switch (MI.getOpcode()) { 121 default: 122 return None; 123 case RISCV::PseudoVLE8_V_M1: 124 case RISCV::PseudoVLE8_V_M1_MASK: 125 case RISCV::PseudoVLE8_V_M2: 126 case RISCV::PseudoVLE8_V_M2_MASK: 127 case RISCV::PseudoVLE8_V_M4: 128 case RISCV::PseudoVLE8_V_M4_MASK: 129 case RISCV::PseudoVLE8_V_M8: 130 case RISCV::PseudoVLE8_V_M8_MASK: 131 case RISCV::PseudoVLE8_V_MF2: 132 case RISCV::PseudoVLE8_V_MF2_MASK: 133 case RISCV::PseudoVLE8_V_MF4: 134 case RISCV::PseudoVLE8_V_MF4_MASK: 135 case RISCV::PseudoVLE8_V_MF8: 136 case RISCV::PseudoVLE8_V_MF8_MASK: 137 case RISCV::PseudoVLSE8_V_M1: 138 case RISCV::PseudoVLSE8_V_M1_MASK: 139 case RISCV::PseudoVLSE8_V_M2: 140 case RISCV::PseudoVLSE8_V_M2_MASK: 141 case RISCV::PseudoVLSE8_V_M4: 142 case RISCV::PseudoVLSE8_V_M4_MASK: 143 case RISCV::PseudoVLSE8_V_M8: 144 case RISCV::PseudoVLSE8_V_M8_MASK: 145 case RISCV::PseudoVLSE8_V_MF2: 146 case RISCV::PseudoVLSE8_V_MF2_MASK: 147 case RISCV::PseudoVLSE8_V_MF4: 148 case RISCV::PseudoVLSE8_V_MF4_MASK: 149 case RISCV::PseudoVLSE8_V_MF8: 150 case RISCV::PseudoVLSE8_V_MF8_MASK: 151 case RISCV::PseudoVSE8_V_M1: 152 case RISCV::PseudoVSE8_V_M1_MASK: 153 case RISCV::PseudoVSE8_V_M2: 154 case RISCV::PseudoVSE8_V_M2_MASK: 155 case RISCV::PseudoVSE8_V_M4: 156 case RISCV::PseudoVSE8_V_M4_MASK: 157 case RISCV::PseudoVSE8_V_M8: 158 case RISCV::PseudoVSE8_V_M8_MASK: 159 case RISCV::PseudoVSE8_V_MF2: 160 case RISCV::PseudoVSE8_V_MF2_MASK: 161 case RISCV::PseudoVSE8_V_MF4: 162 case RISCV::PseudoVSE8_V_MF4_MASK: 163 case RISCV::PseudoVSE8_V_MF8: 164 case RISCV::PseudoVSE8_V_MF8_MASK: 165 case RISCV::PseudoVSSE8_V_M1: 166 case RISCV::PseudoVSSE8_V_M1_MASK: 167 case RISCV::PseudoVSSE8_V_M2: 168 case RISCV::PseudoVSSE8_V_M2_MASK: 169 case RISCV::PseudoVSSE8_V_M4: 170 case RISCV::PseudoVSSE8_V_M4_MASK: 171 case RISCV::PseudoVSSE8_V_M8: 172 case RISCV::PseudoVSSE8_V_M8_MASK: 173 case RISCV::PseudoVSSE8_V_MF2: 174 case RISCV::PseudoVSSE8_V_MF2_MASK: 175 case RISCV::PseudoVSSE8_V_MF4: 176 case RISCV::PseudoVSSE8_V_MF4_MASK: 177 case RISCV::PseudoVSSE8_V_MF8: 178 case RISCV::PseudoVSSE8_V_MF8_MASK: 179 return 8; 180 case RISCV::PseudoVLE16_V_M1: 181 case RISCV::PseudoVLE16_V_M1_MASK: 182 case RISCV::PseudoVLE16_V_M2: 183 case RISCV::PseudoVLE16_V_M2_MASK: 184 case RISCV::PseudoVLE16_V_M4: 185 case RISCV::PseudoVLE16_V_M4_MASK: 186 case RISCV::PseudoVLE16_V_M8: 187 case RISCV::PseudoVLE16_V_M8_MASK: 188 case RISCV::PseudoVLE16_V_MF2: 189 case RISCV::PseudoVLE16_V_MF2_MASK: 190 case RISCV::PseudoVLE16_V_MF4: 191 case RISCV::PseudoVLE16_V_MF4_MASK: 192 case RISCV::PseudoVLSE16_V_M1: 193 case RISCV::PseudoVLSE16_V_M1_MASK: 194 case RISCV::PseudoVLSE16_V_M2: 195 case RISCV::PseudoVLSE16_V_M2_MASK: 196 case RISCV::PseudoVLSE16_V_M4: 197 case RISCV::PseudoVLSE16_V_M4_MASK: 198 case RISCV::PseudoVLSE16_V_M8: 199 case RISCV::PseudoVLSE16_V_M8_MASK: 200 case RISCV::PseudoVLSE16_V_MF2: 201 case RISCV::PseudoVLSE16_V_MF2_MASK: 202 case RISCV::PseudoVLSE16_V_MF4: 203 case RISCV::PseudoVLSE16_V_MF4_MASK: 204 case RISCV::PseudoVSE16_V_M1: 205 case RISCV::PseudoVSE16_V_M1_MASK: 206 case RISCV::PseudoVSE16_V_M2: 207 case RISCV::PseudoVSE16_V_M2_MASK: 208 case RISCV::PseudoVSE16_V_M4: 209 case RISCV::PseudoVSE16_V_M4_MASK: 210 case RISCV::PseudoVSE16_V_M8: 211 case RISCV::PseudoVSE16_V_M8_MASK: 212 case RISCV::PseudoVSE16_V_MF2: 213 case RISCV::PseudoVSE16_V_MF2_MASK: 214 case RISCV::PseudoVSE16_V_MF4: 215 case RISCV::PseudoVSE16_V_MF4_MASK: 216 case RISCV::PseudoVSSE16_V_M1: 217 case RISCV::PseudoVSSE16_V_M1_MASK: 218 case RISCV::PseudoVSSE16_V_M2: 219 case RISCV::PseudoVSSE16_V_M2_MASK: 220 case RISCV::PseudoVSSE16_V_M4: 221 case RISCV::PseudoVSSE16_V_M4_MASK: 222 case RISCV::PseudoVSSE16_V_M8: 223 case RISCV::PseudoVSSE16_V_M8_MASK: 224 case RISCV::PseudoVSSE16_V_MF2: 225 case RISCV::PseudoVSSE16_V_MF2_MASK: 226 case RISCV::PseudoVSSE16_V_MF4: 227 case RISCV::PseudoVSSE16_V_MF4_MASK: 228 return 16; 229 case RISCV::PseudoVLE32_V_M1: 230 case RISCV::PseudoVLE32_V_M1_MASK: 231 case RISCV::PseudoVLE32_V_M2: 232 case RISCV::PseudoVLE32_V_M2_MASK: 233 case RISCV::PseudoVLE32_V_M4: 234 case RISCV::PseudoVLE32_V_M4_MASK: 235 case RISCV::PseudoVLE32_V_M8: 236 case RISCV::PseudoVLE32_V_M8_MASK: 237 case RISCV::PseudoVLE32_V_MF2: 238 case RISCV::PseudoVLE32_V_MF2_MASK: 239 case RISCV::PseudoVLSE32_V_M1: 240 case RISCV::PseudoVLSE32_V_M1_MASK: 241 case RISCV::PseudoVLSE32_V_M2: 242 case RISCV::PseudoVLSE32_V_M2_MASK: 243 case RISCV::PseudoVLSE32_V_M4: 244 case RISCV::PseudoVLSE32_V_M4_MASK: 245 case RISCV::PseudoVLSE32_V_M8: 246 case RISCV::PseudoVLSE32_V_M8_MASK: 247 case RISCV::PseudoVLSE32_V_MF2: 248 case RISCV::PseudoVLSE32_V_MF2_MASK: 249 case RISCV::PseudoVSE32_V_M1: 250 case RISCV::PseudoVSE32_V_M1_MASK: 251 case RISCV::PseudoVSE32_V_M2: 252 case RISCV::PseudoVSE32_V_M2_MASK: 253 case RISCV::PseudoVSE32_V_M4: 254 case RISCV::PseudoVSE32_V_M4_MASK: 255 case RISCV::PseudoVSE32_V_M8: 256 case RISCV::PseudoVSE32_V_M8_MASK: 257 case RISCV::PseudoVSE32_V_MF2: 258 case RISCV::PseudoVSE32_V_MF2_MASK: 259 case RISCV::PseudoVSSE32_V_M1: 260 case RISCV::PseudoVSSE32_V_M1_MASK: 261 case RISCV::PseudoVSSE32_V_M2: 262 case RISCV::PseudoVSSE32_V_M2_MASK: 263 case RISCV::PseudoVSSE32_V_M4: 264 case RISCV::PseudoVSSE32_V_M4_MASK: 265 case RISCV::PseudoVSSE32_V_M8: 266 case RISCV::PseudoVSSE32_V_M8_MASK: 267 case RISCV::PseudoVSSE32_V_MF2: 268 case RISCV::PseudoVSSE32_V_MF2_MASK: 269 return 32; 270 case RISCV::PseudoVLE64_V_M1: 271 case RISCV::PseudoVLE64_V_M1_MASK: 272 case RISCV::PseudoVLE64_V_M2: 273 case RISCV::PseudoVLE64_V_M2_MASK: 274 case RISCV::PseudoVLE64_V_M4: 275 case RISCV::PseudoVLE64_V_M4_MASK: 276 case RISCV::PseudoVLE64_V_M8: 277 case RISCV::PseudoVLE64_V_M8_MASK: 278 case RISCV::PseudoVLSE64_V_M1: 279 case RISCV::PseudoVLSE64_V_M1_MASK: 280 case RISCV::PseudoVLSE64_V_M2: 281 case RISCV::PseudoVLSE64_V_M2_MASK: 282 case RISCV::PseudoVLSE64_V_M4: 283 case RISCV::PseudoVLSE64_V_M4_MASK: 284 case RISCV::PseudoVLSE64_V_M8: 285 case RISCV::PseudoVLSE64_V_M8_MASK: 286 case RISCV::PseudoVSE64_V_M1: 287 case RISCV::PseudoVSE64_V_M1_MASK: 288 case RISCV::PseudoVSE64_V_M2: 289 case RISCV::PseudoVSE64_V_M2_MASK: 290 case RISCV::PseudoVSE64_V_M4: 291 case RISCV::PseudoVSE64_V_M4_MASK: 292 case RISCV::PseudoVSE64_V_M8: 293 case RISCV::PseudoVSE64_V_M8_MASK: 294 case RISCV::PseudoVSSE64_V_M1: 295 case RISCV::PseudoVSSE64_V_M1_MASK: 296 case RISCV::PseudoVSSE64_V_M2: 297 case RISCV::PseudoVSSE64_V_M2_MASK: 298 case RISCV::PseudoVSSE64_V_M4: 299 case RISCV::PseudoVSSE64_V_M4_MASK: 300 case RISCV::PseudoVSSE64_V_M8: 301 case RISCV::PseudoVSSE64_V_M8_MASK: 302 return 64; 303 } 304 } 305 306 /// Return true if this is an operation on mask registers. Note that 307 /// this includes both arithmetic/logical ops and load/store (vlm/vsm). 308 static bool isMaskRegOp(const MachineInstr &MI) { 309 if (RISCVII::hasSEWOp(MI.getDesc().TSFlags)) { 310 const unsigned Log2SEW = MI.getOperand(getSEWOpNum(MI)).getImm(); 311 // A Log2SEW of 0 is an operation on mask registers only. 312 return Log2SEW == 0; 313 } 314 return false; 315 } 316 317 static unsigned getSEWLMULRatio(unsigned SEW, RISCVII::VLMUL VLMul) { 318 unsigned LMul; 319 bool Fractional; 320 std::tie(LMul, Fractional) = RISCVVType::decodeVLMUL(VLMul); 321 322 // Convert LMul to a fixed point value with 3 fractional bits. 323 LMul = Fractional ? (8 / LMul) : (LMul * 8); 324 325 assert(SEW >= 8 && "Unexpected SEW value"); 326 return (SEW * 8) / LMul; 327 } 328 329 /// Which subfields of VL or VTYPE have values we need to preserve? 330 struct DemandedFields { 331 bool VL = false; 332 bool SEW = false; 333 bool LMUL = false; 334 bool SEWLMULRatio = false; 335 bool TailPolicy = false; 336 bool MaskPolicy = false; 337 338 // Return true if any part of VTYPE was used 339 bool usedVTYPE() { 340 return SEW || LMUL || SEWLMULRatio || TailPolicy || MaskPolicy; 341 } 342 343 // Mark all VTYPE subfields and properties as demanded 344 void demandVTYPE() { 345 SEW = true; 346 LMUL = true; 347 SEWLMULRatio = true; 348 TailPolicy = true; 349 MaskPolicy = true; 350 } 351 }; 352 353 /// Return true if the two values of the VTYPE register provided are 354 /// indistinguishable from the perspective of an instruction (or set of 355 /// instructions) which use only the Used subfields and properties. 356 static bool areCompatibleVTYPEs(uint64_t VType1, 357 uint64_t VType2, 358 const DemandedFields &Used) { 359 if (Used.SEW && 360 RISCVVType::getSEW(VType1) != RISCVVType::getSEW(VType2)) 361 return false; 362 363 if (Used.LMUL && 364 RISCVVType::getVLMUL(VType1) != RISCVVType::getVLMUL(VType2)) 365 return false; 366 367 if (Used.SEWLMULRatio) { 368 auto Ratio1 = getSEWLMULRatio(RISCVVType::getSEW(VType1), 369 RISCVVType::getVLMUL(VType1)); 370 auto Ratio2 = getSEWLMULRatio(RISCVVType::getSEW(VType2), 371 RISCVVType::getVLMUL(VType2)); 372 if (Ratio1 != Ratio2) 373 return false; 374 } 375 376 if (Used.TailPolicy && 377 RISCVVType::isTailAgnostic(VType1) != RISCVVType::isTailAgnostic(VType2)) 378 return false; 379 if (Used.MaskPolicy && 380 RISCVVType::isMaskAgnostic(VType1) != RISCVVType::isMaskAgnostic(VType2)) 381 return false; 382 return true; 383 } 384 385 /// Return the fields and properties demanded by the provided instruction. 386 static DemandedFields getDemanded(const MachineInstr &MI) { 387 // Warning: This function has to work on both the lowered (i.e. post 388 // emitVSETVLIs) and pre-lowering forms. The main implication of this is 389 // that it can't use the value of a SEW, VL, or Policy operand as they might 390 // be stale after lowering. 391 392 // Most instructions don't use any of these subfeilds. 393 DemandedFields Res; 394 // Start conservative if registers are used 395 if (MI.isCall() || MI.isInlineAsm() || MI.readsRegister(RISCV::VL)) 396 Res.VL = true; 397 if (MI.isCall() || MI.isInlineAsm() || MI.readsRegister(RISCV::VTYPE)) 398 Res.demandVTYPE(); 399 // Start conservative on the unlowered form too 400 uint64_t TSFlags = MI.getDesc().TSFlags; 401 if (RISCVII::hasSEWOp(TSFlags)) { 402 Res.demandVTYPE(); 403 if (RISCVII::hasVLOp(TSFlags)) 404 Res.VL = true; 405 } 406 407 // Loads and stores with implicit EEW do not demand SEW or LMUL directly. 408 // They instead demand the ratio of the two which is used in computing 409 // EMUL, but which allows us the flexibility to change SEW and LMUL 410 // provided we don't change the ratio. 411 if (getEEWForLoadStore(MI)) { 412 Res.SEW = false; 413 Res.LMUL = false; 414 } 415 416 // Store instructions don't use the policy fields. 417 if (RISCVII::hasSEWOp(TSFlags) && MI.getNumExplicitDefs() == 0) { 418 Res.TailPolicy = false; 419 Res.MaskPolicy = false; 420 } 421 422 // A splat of 0/-1 is always a splat of 0/-1, regardless of etype. 423 // TODO: We're currently demanding VL + SEWLMULRatio which is sufficient 424 // but not neccessary. What we really need is VLInBytes. 425 if (isSplatOfZeroOrMinusOne(MI)) { 426 Res.SEW = false; 427 Res.LMUL = false; 428 } 429 430 // If this is a mask reg operation, it only cares about VLMAX. 431 // TODO: Possible extensions to this logic 432 // * Probably ok if available VLMax is larger than demanded 433 // * The policy bits can probably be ignored.. 434 if (isMaskRegOp(MI)) { 435 Res.SEW = false; 436 Res.LMUL = false; 437 } 438 439 return Res; 440 } 441 442 /// Defines the abstract state with which the forward dataflow models the 443 /// values of the VL and VTYPE registers after insertion. 444 class VSETVLIInfo { 445 union { 446 Register AVLReg; 447 unsigned AVLImm; 448 }; 449 450 enum : uint8_t { 451 Uninitialized, 452 AVLIsReg, 453 AVLIsImm, 454 Unknown, 455 } State = Uninitialized; 456 457 // Fields from VTYPE. 458 RISCVII::VLMUL VLMul = RISCVII::LMUL_1; 459 uint8_t SEW = 0; 460 uint8_t TailAgnostic : 1; 461 uint8_t MaskAgnostic : 1; 462 uint8_t SEWLMULRatioOnly : 1; 463 464 public: 465 VSETVLIInfo() 466 : AVLImm(0), TailAgnostic(false), MaskAgnostic(false), 467 SEWLMULRatioOnly(false) {} 468 469 static VSETVLIInfo getUnknown() { 470 VSETVLIInfo Info; 471 Info.setUnknown(); 472 return Info; 473 } 474 475 bool isValid() const { return State != Uninitialized; } 476 void setUnknown() { State = Unknown; } 477 bool isUnknown() const { return State == Unknown; } 478 479 void setAVLReg(Register Reg) { 480 AVLReg = Reg; 481 State = AVLIsReg; 482 } 483 484 void setAVLImm(unsigned Imm) { 485 AVLImm = Imm; 486 State = AVLIsImm; 487 } 488 489 bool hasAVLImm() const { return State == AVLIsImm; } 490 bool hasAVLReg() const { return State == AVLIsReg; } 491 Register getAVLReg() const { 492 assert(hasAVLReg()); 493 return AVLReg; 494 } 495 unsigned getAVLImm() const { 496 assert(hasAVLImm()); 497 return AVLImm; 498 } 499 500 unsigned getSEW() const { return SEW; } 501 RISCVII::VLMUL getVLMUL() const { return VLMul; } 502 503 bool hasZeroAVL() const { 504 if (hasAVLImm()) 505 return getAVLImm() == 0; 506 return false; 507 } 508 bool hasNonZeroAVL() const { 509 if (hasAVLImm()) 510 return getAVLImm() > 0; 511 if (hasAVLReg()) 512 return getAVLReg() == RISCV::X0; 513 return false; 514 } 515 516 bool hasSameAVL(const VSETVLIInfo &Other) const { 517 assert(isValid() && Other.isValid() && 518 "Can't compare invalid VSETVLIInfos"); 519 assert(!isUnknown() && !Other.isUnknown() && 520 "Can't compare AVL in unknown state"); 521 if (hasAVLReg() && Other.hasAVLReg()) 522 return getAVLReg() == Other.getAVLReg(); 523 524 if (hasAVLImm() && Other.hasAVLImm()) 525 return getAVLImm() == Other.getAVLImm(); 526 527 return false; 528 } 529 530 void setVTYPE(unsigned VType) { 531 assert(isValid() && !isUnknown() && 532 "Can't set VTYPE for uninitialized or unknown"); 533 VLMul = RISCVVType::getVLMUL(VType); 534 SEW = RISCVVType::getSEW(VType); 535 TailAgnostic = RISCVVType::isTailAgnostic(VType); 536 MaskAgnostic = RISCVVType::isMaskAgnostic(VType); 537 } 538 void setVTYPE(RISCVII::VLMUL L, unsigned S, bool TA, bool MA) { 539 assert(isValid() && !isUnknown() && 540 "Can't set VTYPE for uninitialized or unknown"); 541 VLMul = L; 542 SEW = S; 543 TailAgnostic = TA; 544 MaskAgnostic = MA; 545 } 546 547 unsigned encodeVTYPE() const { 548 assert(isValid() && !isUnknown() && !SEWLMULRatioOnly && 549 "Can't encode VTYPE for uninitialized or unknown"); 550 return RISCVVType::encodeVTYPE(VLMul, SEW, TailAgnostic, MaskAgnostic); 551 } 552 553 bool hasSEWLMULRatioOnly() const { return SEWLMULRatioOnly; } 554 555 bool hasSameSEW(const VSETVLIInfo &Other) const { 556 assert(isValid() && Other.isValid() && 557 "Can't compare invalid VSETVLIInfos"); 558 assert(!isUnknown() && !Other.isUnknown() && 559 "Can't compare VTYPE in unknown state"); 560 assert(!SEWLMULRatioOnly && !Other.SEWLMULRatioOnly && 561 "Can't compare when only LMUL/SEW ratio is valid."); 562 return SEW == Other.SEW; 563 } 564 565 bool hasSameVTYPE(const VSETVLIInfo &Other) const { 566 assert(isValid() && Other.isValid() && 567 "Can't compare invalid VSETVLIInfos"); 568 assert(!isUnknown() && !Other.isUnknown() && 569 "Can't compare VTYPE in unknown state"); 570 assert(!SEWLMULRatioOnly && !Other.SEWLMULRatioOnly && 571 "Can't compare when only LMUL/SEW ratio is valid."); 572 return std::tie(VLMul, SEW, TailAgnostic, MaskAgnostic) == 573 std::tie(Other.VLMul, Other.SEW, Other.TailAgnostic, 574 Other.MaskAgnostic); 575 } 576 577 unsigned getSEWLMULRatio() const { 578 assert(isValid() && !isUnknown() && 579 "Can't use VTYPE for uninitialized or unknown"); 580 return ::getSEWLMULRatio(SEW, VLMul); 581 } 582 583 // Check if the VTYPE for these two VSETVLIInfos produce the same VLMAX. 584 // Note that having the same VLMAX ensures that both share the same 585 // function from AVL to VL; that is, they must produce the same VL value 586 // for any given AVL value. 587 bool hasSameVLMAX(const VSETVLIInfo &Other) const { 588 assert(isValid() && Other.isValid() && 589 "Can't compare invalid VSETVLIInfos"); 590 assert(!isUnknown() && !Other.isUnknown() && 591 "Can't compare VTYPE in unknown state"); 592 return getSEWLMULRatio() == Other.getSEWLMULRatio(); 593 } 594 595 bool hasSamePolicy(const VSETVLIInfo &Other) const { 596 assert(isValid() && Other.isValid() && 597 "Can't compare invalid VSETVLIInfos"); 598 assert(!isUnknown() && !Other.isUnknown() && 599 "Can't compare VTYPE in unknown state"); 600 return TailAgnostic == Other.TailAgnostic && 601 MaskAgnostic == Other.MaskAgnostic; 602 } 603 604 bool hasCompatibleVTYPE(const MachineInstr &MI, 605 const VSETVLIInfo &Require) const { 606 const DemandedFields Used = getDemanded(MI); 607 return areCompatibleVTYPEs(encodeVTYPE(), Require.encodeVTYPE(), Used); 608 } 609 610 // Determine whether the vector instructions requirements represented by 611 // Require are compatible with the previous vsetvli instruction represented 612 // by this. MI is the instruction whose requirements we're considering. 613 bool isCompatible(const MachineInstr &MI, const VSETVLIInfo &Require) const { 614 assert(isValid() && Require.isValid() && 615 "Can't compare invalid VSETVLIInfos"); 616 assert(!Require.SEWLMULRatioOnly && 617 "Expected a valid VTYPE for instruction!"); 618 // Nothing is compatible with Unknown. 619 if (isUnknown() || Require.isUnknown()) 620 return false; 621 622 // If only our VLMAX ratio is valid, then this isn't compatible. 623 if (SEWLMULRatioOnly) 624 return false; 625 626 // If the instruction doesn't need an AVLReg and the SEW matches, consider 627 // it compatible. 628 if (Require.hasAVLReg() && Require.AVLReg == RISCV::NoRegister) 629 if (SEW == Require.SEW) 630 return true; 631 632 return hasSameAVL(Require) && hasCompatibleVTYPE(MI, Require); 633 } 634 635 bool operator==(const VSETVLIInfo &Other) const { 636 // Uninitialized is only equal to another Uninitialized. 637 if (!isValid()) 638 return !Other.isValid(); 639 if (!Other.isValid()) 640 return !isValid(); 641 642 // Unknown is only equal to another Unknown. 643 if (isUnknown()) 644 return Other.isUnknown(); 645 if (Other.isUnknown()) 646 return isUnknown(); 647 648 if (!hasSameAVL(Other)) 649 return false; 650 651 // If the SEWLMULRatioOnly bits are different, then they aren't equal. 652 if (SEWLMULRatioOnly != Other.SEWLMULRatioOnly) 653 return false; 654 655 // If only the VLMAX is valid, check that it is the same. 656 if (SEWLMULRatioOnly) 657 return hasSameVLMAX(Other); 658 659 // If the full VTYPE is valid, check that it is the same. 660 return hasSameVTYPE(Other); 661 } 662 663 bool operator!=(const VSETVLIInfo &Other) const { 664 return !(*this == Other); 665 } 666 667 // Calculate the VSETVLIInfo visible to a block assuming this and Other are 668 // both predecessors. 669 VSETVLIInfo intersect(const VSETVLIInfo &Other) const { 670 // If the new value isn't valid, ignore it. 671 if (!Other.isValid()) 672 return *this; 673 674 // If this value isn't valid, this must be the first predecessor, use it. 675 if (!isValid()) 676 return Other; 677 678 // If either is unknown, the result is unknown. 679 if (isUnknown() || Other.isUnknown()) 680 return VSETVLIInfo::getUnknown(); 681 682 // If we have an exact, match return this. 683 if (*this == Other) 684 return *this; 685 686 // Not an exact match, but maybe the AVL and VLMAX are the same. If so, 687 // return an SEW/LMUL ratio only value. 688 if (hasSameAVL(Other) && hasSameVLMAX(Other)) { 689 VSETVLIInfo MergeInfo = *this; 690 MergeInfo.SEWLMULRatioOnly = true; 691 return MergeInfo; 692 } 693 694 // Otherwise the result is unknown. 695 return VSETVLIInfo::getUnknown(); 696 } 697 698 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) 699 /// Support for debugging, callable in GDB: V->dump() 700 LLVM_DUMP_METHOD void dump() const { 701 print(dbgs()); 702 dbgs() << "\n"; 703 } 704 705 /// Implement operator<<. 706 /// @{ 707 void print(raw_ostream &OS) const { 708 OS << "{"; 709 if (!isValid()) 710 OS << "Uninitialized"; 711 if (isUnknown()) 712 OS << "unknown";; 713 if (hasAVLReg()) 714 OS << "AVLReg=" << (unsigned)AVLReg; 715 if (hasAVLImm()) 716 OS << "AVLImm=" << (unsigned)AVLImm; 717 OS << ", " 718 << "VLMul=" << (unsigned)VLMul << ", " 719 << "SEW=" << (unsigned)SEW << ", " 720 << "TailAgnostic=" << (bool)TailAgnostic << ", " 721 << "MaskAgnostic=" << (bool)MaskAgnostic << ", " 722 << "SEWLMULRatioOnly=" << (bool)SEWLMULRatioOnly << "}"; 723 } 724 #endif 725 }; 726 727 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) 728 LLVM_ATTRIBUTE_USED 729 inline raw_ostream &operator<<(raw_ostream &OS, const VSETVLIInfo &V) { 730 V.print(OS); 731 return OS; 732 } 733 #endif 734 735 struct BlockData { 736 // The VSETVLIInfo that represents the net changes to the VL/VTYPE registers 737 // made by this block. Calculated in Phase 1. 738 VSETVLIInfo Change; 739 740 // The VSETVLIInfo that represents the VL/VTYPE settings on exit from this 741 // block. Calculated in Phase 2. 742 VSETVLIInfo Exit; 743 744 // The VSETVLIInfo that represents the VL/VTYPE settings from all predecessor 745 // blocks. Calculated in Phase 2, and used by Phase 3. 746 VSETVLIInfo Pred; 747 748 // Keeps track of whether the block is already in the queue. 749 bool InQueue = false; 750 751 BlockData() = default; 752 }; 753 754 class RISCVInsertVSETVLI : public MachineFunctionPass { 755 const TargetInstrInfo *TII; 756 MachineRegisterInfo *MRI; 757 758 std::vector<BlockData> BlockInfo; 759 std::queue<const MachineBasicBlock *> WorkList; 760 761 public: 762 static char ID; 763 764 RISCVInsertVSETVLI() : MachineFunctionPass(ID) { 765 initializeRISCVInsertVSETVLIPass(*PassRegistry::getPassRegistry()); 766 } 767 bool runOnMachineFunction(MachineFunction &MF) override; 768 769 void getAnalysisUsage(AnalysisUsage &AU) const override { 770 AU.setPreservesCFG(); 771 MachineFunctionPass::getAnalysisUsage(AU); 772 } 773 774 StringRef getPassName() const override { return RISCV_INSERT_VSETVLI_NAME; } 775 776 private: 777 bool needVSETVLI(const MachineInstr &MI, const VSETVLIInfo &Require, 778 const VSETVLIInfo &CurInfo) const; 779 bool needVSETVLIPHI(const VSETVLIInfo &Require, 780 const MachineBasicBlock &MBB) const; 781 void insertVSETVLI(MachineBasicBlock &MBB, MachineInstr &MI, 782 const VSETVLIInfo &Info, const VSETVLIInfo &PrevInfo); 783 void insertVSETVLI(MachineBasicBlock &MBB, 784 MachineBasicBlock::iterator InsertPt, DebugLoc DL, 785 const VSETVLIInfo &Info, const VSETVLIInfo &PrevInfo); 786 787 void transferBefore(VSETVLIInfo &Info, const MachineInstr &MI); 788 void transferAfter(VSETVLIInfo &Info, const MachineInstr &MI); 789 bool computeVLVTYPEChanges(const MachineBasicBlock &MBB); 790 void computeIncomingVLVTYPE(const MachineBasicBlock &MBB); 791 void emitVSETVLIs(MachineBasicBlock &MBB); 792 void doLocalPrepass(MachineBasicBlock &MBB); 793 void doLocalPostpass(MachineBasicBlock &MBB); 794 void doPRE(MachineBasicBlock &MBB); 795 void insertReadVL(MachineBasicBlock &MBB); 796 }; 797 798 } // end anonymous namespace 799 800 char RISCVInsertVSETVLI::ID = 0; 801 802 INITIALIZE_PASS(RISCVInsertVSETVLI, DEBUG_TYPE, RISCV_INSERT_VSETVLI_NAME, 803 false, false) 804 805 static bool isVectorConfigInstr(const MachineInstr &MI) { 806 return MI.getOpcode() == RISCV::PseudoVSETVLI || 807 MI.getOpcode() == RISCV::PseudoVSETVLIX0 || 808 MI.getOpcode() == RISCV::PseudoVSETIVLI; 809 } 810 811 /// Return true if this is 'vsetvli x0, x0, vtype' which preserves 812 /// VL and only sets VTYPE. 813 static bool isVLPreservingConfig(const MachineInstr &MI) { 814 if (MI.getOpcode() != RISCV::PseudoVSETVLIX0) 815 return false; 816 assert(RISCV::X0 == MI.getOperand(1).getReg()); 817 return RISCV::X0 == MI.getOperand(0).getReg(); 818 } 819 820 static VSETVLIInfo computeInfoForInstr(const MachineInstr &MI, uint64_t TSFlags, 821 const MachineRegisterInfo *MRI) { 822 VSETVLIInfo InstrInfo; 823 824 // If the instruction has policy argument, use the argument. 825 // If there is no policy argument, default to tail agnostic unless the 826 // destination is tied to a source. Unless the source is undef. In that case 827 // the user would have some control over the policy values. 828 bool TailAgnostic = true; 829 bool UsesMaskPolicy = RISCVII::usesMaskPolicy(TSFlags); 830 // FIXME: Could we look at the above or below instructions to choose the 831 // matched mask policy to reduce vsetvli instructions? Default mask policy is 832 // agnostic if instructions use mask policy, otherwise is undisturbed. Because 833 // most mask operations are mask undisturbed, so we could possibly reduce the 834 // vsetvli between mask and nomasked instruction sequence. 835 bool MaskAgnostic = UsesMaskPolicy; 836 unsigned UseOpIdx; 837 if (RISCVII::hasVecPolicyOp(TSFlags)) { 838 const MachineOperand &Op = MI.getOperand(MI.getNumExplicitOperands() - 1); 839 uint64_t Policy = Op.getImm(); 840 assert(Policy <= (RISCVII::TAIL_AGNOSTIC | RISCVII::MASK_AGNOSTIC) && 841 "Invalid Policy Value"); 842 // Although in some cases, mismatched passthru/maskedoff with policy value 843 // does not make sense (ex. tied operand is IMPLICIT_DEF with non-TAMA 844 // policy, or tied operand is not IMPLICIT_DEF with TAMA policy), but users 845 // have set the policy value explicitly, so compiler would not fix it. 846 TailAgnostic = Policy & RISCVII::TAIL_AGNOSTIC; 847 MaskAgnostic = Policy & RISCVII::MASK_AGNOSTIC; 848 } else if (MI.isRegTiedToUseOperand(0, &UseOpIdx)) { 849 TailAgnostic = false; 850 if (UsesMaskPolicy) 851 MaskAgnostic = false; 852 // If the tied operand is an IMPLICIT_DEF we can keep TailAgnostic. 853 const MachineOperand &UseMO = MI.getOperand(UseOpIdx); 854 MachineInstr *UseMI = MRI->getVRegDef(UseMO.getReg()); 855 if (UseMI && UseMI->isImplicitDef()) { 856 TailAgnostic = true; 857 if (UsesMaskPolicy) 858 MaskAgnostic = true; 859 } 860 // Some pseudo instructions force a tail agnostic policy despite having a 861 // tied def. 862 if (RISCVII::doesForceTailAgnostic(TSFlags)) 863 TailAgnostic = true; 864 } 865 866 RISCVII::VLMUL VLMul = RISCVII::getLMul(TSFlags); 867 868 unsigned Log2SEW = MI.getOperand(getSEWOpNum(MI)).getImm(); 869 // A Log2SEW of 0 is an operation on mask registers only. 870 unsigned SEW = Log2SEW ? 1 << Log2SEW : 8; 871 assert(RISCVVType::isValidSEW(SEW) && "Unexpected SEW"); 872 873 if (RISCVII::hasVLOp(TSFlags)) { 874 const MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI)); 875 if (VLOp.isImm()) { 876 int64_t Imm = VLOp.getImm(); 877 // Conver the VLMax sentintel to X0 register. 878 if (Imm == RISCV::VLMaxSentinel) 879 InstrInfo.setAVLReg(RISCV::X0); 880 else 881 InstrInfo.setAVLImm(Imm); 882 } else { 883 InstrInfo.setAVLReg(VLOp.getReg()); 884 } 885 } else { 886 InstrInfo.setAVLReg(RISCV::NoRegister); 887 } 888 InstrInfo.setVTYPE(VLMul, SEW, TailAgnostic, MaskAgnostic); 889 890 return InstrInfo; 891 } 892 893 void RISCVInsertVSETVLI::insertVSETVLI(MachineBasicBlock &MBB, MachineInstr &MI, 894 const VSETVLIInfo &Info, 895 const VSETVLIInfo &PrevInfo) { 896 DebugLoc DL = MI.getDebugLoc(); 897 insertVSETVLI(MBB, MachineBasicBlock::iterator(&MI), DL, Info, PrevInfo); 898 } 899 900 void RISCVInsertVSETVLI::insertVSETVLI(MachineBasicBlock &MBB, 901 MachineBasicBlock::iterator InsertPt, DebugLoc DL, 902 const VSETVLIInfo &Info, const VSETVLIInfo &PrevInfo) { 903 904 // Use X0, X0 form if the AVL is the same and the SEW+LMUL gives the same 905 // VLMAX. 906 if (PrevInfo.isValid() && !PrevInfo.isUnknown() && 907 Info.hasSameAVL(PrevInfo) && Info.hasSameVLMAX(PrevInfo)) { 908 BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETVLIX0)) 909 .addReg(RISCV::X0, RegState::Define | RegState::Dead) 910 .addReg(RISCV::X0, RegState::Kill) 911 .addImm(Info.encodeVTYPE()) 912 .addReg(RISCV::VL, RegState::Implicit); 913 return; 914 } 915 916 if (Info.hasAVLImm()) { 917 BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETIVLI)) 918 .addReg(RISCV::X0, RegState::Define | RegState::Dead) 919 .addImm(Info.getAVLImm()) 920 .addImm(Info.encodeVTYPE()); 921 return; 922 } 923 924 Register AVLReg = Info.getAVLReg(); 925 if (AVLReg == RISCV::NoRegister) { 926 // We can only use x0, x0 if there's no chance of the vtype change causing 927 // the previous vl to become invalid. 928 if (PrevInfo.isValid() && !PrevInfo.isUnknown() && 929 Info.hasSameVLMAX(PrevInfo)) { 930 BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETVLIX0)) 931 .addReg(RISCV::X0, RegState::Define | RegState::Dead) 932 .addReg(RISCV::X0, RegState::Kill) 933 .addImm(Info.encodeVTYPE()) 934 .addReg(RISCV::VL, RegState::Implicit); 935 return; 936 } 937 // Otherwise use an AVL of 0 to avoid depending on previous vl. 938 BuildMI(MBB, InsertPt, DL, TII->get(RISCV::PseudoVSETIVLI)) 939 .addReg(RISCV::X0, RegState::Define | RegState::Dead) 940 .addImm(0) 941 .addImm(Info.encodeVTYPE()); 942 return; 943 } 944 945 if (AVLReg.isVirtual()) 946 MRI->constrainRegClass(AVLReg, &RISCV::GPRNoX0RegClass); 947 948 // Use X0 as the DestReg unless AVLReg is X0. We also need to change the 949 // opcode if the AVLReg is X0 as they have different register classes for 950 // the AVL operand. 951 Register DestReg = RISCV::X0; 952 unsigned Opcode = RISCV::PseudoVSETVLI; 953 if (AVLReg == RISCV::X0) { 954 DestReg = MRI->createVirtualRegister(&RISCV::GPRRegClass); 955 Opcode = RISCV::PseudoVSETVLIX0; 956 } 957 BuildMI(MBB, InsertPt, DL, TII->get(Opcode)) 958 .addReg(DestReg, RegState::Define | RegState::Dead) 959 .addReg(AVLReg) 960 .addImm(Info.encodeVTYPE()); 961 } 962 963 // Return a VSETVLIInfo representing the changes made by this VSETVLI or 964 // VSETIVLI instruction. 965 static VSETVLIInfo getInfoForVSETVLI(const MachineInstr &MI) { 966 VSETVLIInfo NewInfo; 967 if (MI.getOpcode() == RISCV::PseudoVSETIVLI) { 968 NewInfo.setAVLImm(MI.getOperand(1).getImm()); 969 } else { 970 assert(MI.getOpcode() == RISCV::PseudoVSETVLI || 971 MI.getOpcode() == RISCV::PseudoVSETVLIX0); 972 Register AVLReg = MI.getOperand(1).getReg(); 973 assert((AVLReg != RISCV::X0 || MI.getOperand(0).getReg() != RISCV::X0) && 974 "Can't handle X0, X0 vsetvli yet"); 975 NewInfo.setAVLReg(AVLReg); 976 } 977 NewInfo.setVTYPE(MI.getOperand(2).getImm()); 978 979 return NewInfo; 980 } 981 982 /// Return true if a VSETVLI is required to transition from CurInfo to Require 983 /// before MI. 984 bool RISCVInsertVSETVLI::needVSETVLI(const MachineInstr &MI, 985 const VSETVLIInfo &Require, 986 const VSETVLIInfo &CurInfo) const { 987 assert(Require == computeInfoForInstr(MI, MI.getDesc().TSFlags, MRI)); 988 989 if (CurInfo.isCompatible(MI, Require)) 990 return false; 991 992 // For vmv.s.x and vfmv.s.f, there is only two behaviors, VL = 0 and VL > 0. 993 // So it's compatible when we could make sure that both VL be the same 994 // situation. Additionally, if writing to an implicit_def operand, we 995 // don't need to preserve any other bits and are thus compatible with any 996 // larger etype, and can disregard policy bits. 997 if (isScalarMoveInstr(MI) && 998 ((CurInfo.hasNonZeroAVL() && Require.hasNonZeroAVL()) || 999 (CurInfo.hasZeroAVL() && Require.hasZeroAVL()))) { 1000 auto *VRegDef = MRI->getVRegDef(MI.getOperand(1).getReg()); 1001 if (VRegDef && VRegDef->isImplicitDef() && 1002 CurInfo.getSEW() >= Require.getSEW()) 1003 return false; 1004 if (CurInfo.hasSameSEW(Require) && CurInfo.hasSamePolicy(Require)) 1005 return false; 1006 } 1007 1008 // We didn't find a compatible value. If our AVL is a virtual register, 1009 // it might be defined by a VSET(I)VLI. If it has the same VLMAX we need 1010 // and the last VL/VTYPE we observed is the same, we don't need a 1011 // VSETVLI here. 1012 if (!CurInfo.isUnknown() && Require.hasAVLReg() && 1013 Require.getAVLReg().isVirtual() && !CurInfo.hasSEWLMULRatioOnly() && 1014 CurInfo.hasCompatibleVTYPE(MI, Require)) { 1015 if (MachineInstr *DefMI = MRI->getVRegDef(Require.getAVLReg())) { 1016 if (isVectorConfigInstr(*DefMI)) { 1017 VSETVLIInfo DefInfo = getInfoForVSETVLI(*DefMI); 1018 if (DefInfo.hasSameAVL(CurInfo) && DefInfo.hasSameVLMAX(CurInfo)) 1019 return false; 1020 } 1021 } 1022 } 1023 1024 return true; 1025 } 1026 1027 // Given an incoming state reaching MI, modifies that state so that it is minimally 1028 // compatible with MI. The resulting state is guaranteed to be semantically legal 1029 // for MI, but may not be the state requested by MI. 1030 void RISCVInsertVSETVLI::transferBefore(VSETVLIInfo &Info, const MachineInstr &MI) { 1031 uint64_t TSFlags = MI.getDesc().TSFlags; 1032 if (!RISCVII::hasSEWOp(TSFlags)) 1033 return; 1034 VSETVLIInfo NewInfo = computeInfoForInstr(MI, TSFlags, MRI); 1035 1036 if (!Info.isValid()) { 1037 Info = NewInfo; 1038 } else { 1039 // If this instruction isn't compatible with the previous VL/VTYPE 1040 // we need to insert a VSETVLI. 1041 // NOTE: We only do this if the vtype we're comparing against was 1042 // created in this block. We need the first and third phase to treat 1043 // the store the same way. 1044 if (needVSETVLI(MI, NewInfo, Info)) 1045 Info = NewInfo; 1046 } 1047 } 1048 1049 // Given a state with which we evaluated MI (see transferBefore above for why 1050 // this might be different that the state MI requested), modify the state to 1051 // reflect the changes MI might make. 1052 void RISCVInsertVSETVLI::transferAfter(VSETVLIInfo &Info, const MachineInstr &MI) { 1053 if (isVectorConfigInstr(MI)) { 1054 Info = getInfoForVSETVLI(MI); 1055 return; 1056 } 1057 1058 if (RISCV::isFaultFirstLoad(MI)) { 1059 // Update AVL to vl-output of the fault first load. 1060 Info.setAVLReg(MI.getOperand(1).getReg()); 1061 return; 1062 } 1063 1064 // If this is something that updates VL/VTYPE that we don't know about, set 1065 // the state to unknown. 1066 if (MI.isCall() || MI.isInlineAsm() || MI.modifiesRegister(RISCV::VL) || 1067 MI.modifiesRegister(RISCV::VTYPE)) 1068 Info = VSETVLIInfo::getUnknown(); 1069 } 1070 1071 bool RISCVInsertVSETVLI::computeVLVTYPEChanges(const MachineBasicBlock &MBB) { 1072 bool HadVectorOp = false; 1073 1074 BlockData &BBInfo = BlockInfo[MBB.getNumber()]; 1075 BBInfo.Change = BBInfo.Pred; 1076 for (const MachineInstr &MI : MBB) { 1077 transferBefore(BBInfo.Change, MI); 1078 1079 if (isVectorConfigInstr(MI) || RISCVII::hasSEWOp(MI.getDesc().TSFlags)) 1080 HadVectorOp = true; 1081 1082 transferAfter(BBInfo.Change, MI); 1083 } 1084 1085 return HadVectorOp; 1086 } 1087 1088 void RISCVInsertVSETVLI::computeIncomingVLVTYPE(const MachineBasicBlock &MBB) { 1089 1090 BlockData &BBInfo = BlockInfo[MBB.getNumber()]; 1091 1092 BBInfo.InQueue = false; 1093 1094 VSETVLIInfo InInfo; 1095 if (MBB.pred_empty()) { 1096 // There are no predecessors, so use the default starting status. 1097 InInfo.setUnknown(); 1098 } else { 1099 for (MachineBasicBlock *P : MBB.predecessors()) 1100 InInfo = InInfo.intersect(BlockInfo[P->getNumber()].Exit); 1101 } 1102 1103 // If we don't have any valid predecessor value, wait until we do. 1104 if (!InInfo.isValid()) 1105 return; 1106 1107 // If no change, no need to rerun block 1108 if (InInfo == BBInfo.Pred) 1109 return; 1110 1111 BBInfo.Pred = InInfo; 1112 LLVM_DEBUG(dbgs() << "Entry state of " << printMBBReference(MBB) 1113 << " changed to " << BBInfo.Pred << "\n"); 1114 1115 // Note: It's tempting to cache the state changes here, but due to the 1116 // compatibility checks performed a blocks output state can change based on 1117 // the input state. To cache, we'd have to add logic for finding 1118 // never-compatible state changes. 1119 computeVLVTYPEChanges(MBB); 1120 VSETVLIInfo TmpStatus = BBInfo.Change; 1121 1122 // If the new exit value matches the old exit value, we don't need to revisit 1123 // any blocks. 1124 if (BBInfo.Exit == TmpStatus) 1125 return; 1126 1127 BBInfo.Exit = TmpStatus; 1128 LLVM_DEBUG(dbgs() << "Exit state of " << printMBBReference(MBB) 1129 << " changed to " << BBInfo.Exit << "\n"); 1130 1131 // Add the successors to the work list so we can propagate the changed exit 1132 // status. 1133 for (MachineBasicBlock *S : MBB.successors()) 1134 if (!BlockInfo[S->getNumber()].InQueue) 1135 WorkList.push(S); 1136 } 1137 1138 // If we weren't able to prove a vsetvli was directly unneeded, it might still 1139 // be unneeded if the AVL is a phi node where all incoming values are VL 1140 // outputs from the last VSETVLI in their respective basic blocks. 1141 bool RISCVInsertVSETVLI::needVSETVLIPHI(const VSETVLIInfo &Require, 1142 const MachineBasicBlock &MBB) const { 1143 if (DisableInsertVSETVLPHIOpt) 1144 return true; 1145 1146 if (!Require.hasAVLReg()) 1147 return true; 1148 1149 Register AVLReg = Require.getAVLReg(); 1150 if (!AVLReg.isVirtual()) 1151 return true; 1152 1153 // We need the AVL to be produce by a PHI node in this basic block. 1154 MachineInstr *PHI = MRI->getVRegDef(AVLReg); 1155 if (!PHI || PHI->getOpcode() != RISCV::PHI || PHI->getParent() != &MBB) 1156 return true; 1157 1158 for (unsigned PHIOp = 1, NumOps = PHI->getNumOperands(); PHIOp != NumOps; 1159 PHIOp += 2) { 1160 Register InReg = PHI->getOperand(PHIOp).getReg(); 1161 MachineBasicBlock *PBB = PHI->getOperand(PHIOp + 1).getMBB(); 1162 const BlockData &PBBInfo = BlockInfo[PBB->getNumber()]; 1163 // If the exit from the predecessor has the VTYPE we are looking for 1164 // we might be able to avoid a VSETVLI. 1165 if (PBBInfo.Exit.isUnknown() || !PBBInfo.Exit.hasSameVTYPE(Require)) 1166 return true; 1167 1168 // We need the PHI input to the be the output of a VSET(I)VLI. 1169 MachineInstr *DefMI = MRI->getVRegDef(InReg); 1170 if (!DefMI || !isVectorConfigInstr(*DefMI)) 1171 return true; 1172 1173 // We found a VSET(I)VLI make sure it matches the output of the 1174 // predecessor block. 1175 VSETVLIInfo DefInfo = getInfoForVSETVLI(*DefMI); 1176 if (!DefInfo.hasSameAVL(PBBInfo.Exit) || 1177 !DefInfo.hasSameVTYPE(PBBInfo.Exit)) 1178 return true; 1179 } 1180 1181 // If all the incoming values to the PHI checked out, we don't need 1182 // to insert a VSETVLI. 1183 return false; 1184 } 1185 1186 void RISCVInsertVSETVLI::emitVSETVLIs(MachineBasicBlock &MBB) { 1187 VSETVLIInfo CurInfo = BlockInfo[MBB.getNumber()].Pred; 1188 // Track whether the prefix of the block we've scanned is transparent 1189 // (meaning has not yet changed the abstract state). 1190 bool PrefixTransparent = true; 1191 for (MachineInstr &MI : MBB) { 1192 const VSETVLIInfo PrevInfo = CurInfo; 1193 transferBefore(CurInfo, MI); 1194 1195 // If this is an explicit VSETVLI or VSETIVLI, update our state. 1196 if (isVectorConfigInstr(MI)) { 1197 // Conservatively, mark the VL and VTYPE as live. 1198 assert(MI.getOperand(3).getReg() == RISCV::VL && 1199 MI.getOperand(4).getReg() == RISCV::VTYPE && 1200 "Unexpected operands where VL and VTYPE should be"); 1201 MI.getOperand(3).setIsDead(false); 1202 MI.getOperand(4).setIsDead(false); 1203 PrefixTransparent = false; 1204 } 1205 1206 uint64_t TSFlags = MI.getDesc().TSFlags; 1207 if (RISCVII::hasSEWOp(TSFlags)) { 1208 if (PrevInfo != CurInfo) { 1209 // If this is the first implicit state change, and the state change 1210 // requested can be proven to produce the same register contents, we 1211 // can skip emitting the actual state change and continue as if we 1212 // had since we know the GPR result of the implicit state change 1213 // wouldn't be used and VL/VTYPE registers are correct. Note that 1214 // we *do* need to model the state as if it changed as while the 1215 // register contents are unchanged, the abstract model can change. 1216 if (!PrefixTransparent || needVSETVLIPHI(CurInfo, MBB)) 1217 insertVSETVLI(MBB, MI, CurInfo, PrevInfo); 1218 PrefixTransparent = false; 1219 } 1220 1221 if (RISCVII::hasVLOp(TSFlags)) { 1222 MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI)); 1223 if (VLOp.isReg()) { 1224 // Erase the AVL operand from the instruction. 1225 VLOp.setReg(RISCV::NoRegister); 1226 VLOp.setIsKill(false); 1227 } 1228 MI.addOperand(MachineOperand::CreateReg(RISCV::VL, /*isDef*/ false, 1229 /*isImp*/ true)); 1230 } 1231 MI.addOperand(MachineOperand::CreateReg(RISCV::VTYPE, /*isDef*/ false, 1232 /*isImp*/ true)); 1233 } 1234 1235 if (MI.isCall() || MI.isInlineAsm() || MI.modifiesRegister(RISCV::VL) || 1236 MI.modifiesRegister(RISCV::VTYPE)) 1237 PrefixTransparent = false; 1238 1239 transferAfter(CurInfo, MI); 1240 } 1241 1242 // If we reach the end of the block and our current info doesn't match the 1243 // expected info, insert a vsetvli to correct. 1244 if (!UseStrictAsserts) { 1245 const VSETVLIInfo &ExitInfo = BlockInfo[MBB.getNumber()].Exit; 1246 if (CurInfo.isValid() && ExitInfo.isValid() && !ExitInfo.isUnknown() && 1247 CurInfo != ExitInfo) { 1248 // Note there's an implicit assumption here that terminators never use 1249 // or modify VL or VTYPE. Also, fallthrough will return end(). 1250 auto InsertPt = MBB.getFirstInstrTerminator(); 1251 insertVSETVLI(MBB, InsertPt, MBB.findDebugLoc(InsertPt), ExitInfo, 1252 CurInfo); 1253 CurInfo = ExitInfo; 1254 } 1255 } 1256 1257 if (UseStrictAsserts && CurInfo.isValid()) { 1258 const auto &Info = BlockInfo[MBB.getNumber()]; 1259 if (CurInfo != Info.Exit) { 1260 LLVM_DEBUG(dbgs() << "in block " << printMBBReference(MBB) << "\n"); 1261 LLVM_DEBUG(dbgs() << " begin state: " << Info.Pred << "\n"); 1262 LLVM_DEBUG(dbgs() << " expected end state: " << Info.Exit << "\n"); 1263 LLVM_DEBUG(dbgs() << " actual end state: " << CurInfo << "\n"); 1264 } 1265 assert(CurInfo == Info.Exit && 1266 "InsertVSETVLI dataflow invariant violated"); 1267 } 1268 } 1269 1270 void RISCVInsertVSETVLI::doLocalPrepass(MachineBasicBlock &MBB) { 1271 VSETVLIInfo CurInfo = VSETVLIInfo::getUnknown(); 1272 for (MachineInstr &MI : MBB) { 1273 // If this is an explicit VSETVLI or VSETIVLI, update our state. 1274 if (isVectorConfigInstr(MI)) { 1275 CurInfo = getInfoForVSETVLI(MI); 1276 continue; 1277 } 1278 1279 const uint64_t TSFlags = MI.getDesc().TSFlags; 1280 if (isScalarMoveInstr(MI)) { 1281 assert(RISCVII::hasSEWOp(TSFlags) && RISCVII::hasVLOp(TSFlags)); 1282 const VSETVLIInfo NewInfo = computeInfoForInstr(MI, TSFlags, MRI); 1283 1284 // For vmv.s.x and vfmv.s.f, there are only two behaviors, VL = 0 and 1285 // VL > 0. We can discard the user requested AVL and just use the last 1286 // one if we can prove it equally zero. This removes a vsetvli entirely 1287 // if the types match or allows use of cheaper avl preserving variant 1288 // if VLMAX doesn't change. If VLMAX might change, we couldn't use 1289 // the 'vsetvli x0, x0, vtype" variant, so we avoid the transform to 1290 // prevent extending live range of an avl register operand. 1291 // TODO: We can probably relax this for immediates. 1292 if (((CurInfo.hasNonZeroAVL() && NewInfo.hasNonZeroAVL()) || 1293 (CurInfo.hasZeroAVL() && NewInfo.hasZeroAVL())) && 1294 NewInfo.hasSameVLMAX(CurInfo)) { 1295 MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI)); 1296 if (CurInfo.hasAVLImm()) 1297 VLOp.ChangeToImmediate(CurInfo.getAVLImm()); 1298 else 1299 VLOp.ChangeToRegister(CurInfo.getAVLReg(), /*IsDef*/ false); 1300 CurInfo = computeInfoForInstr(MI, TSFlags, MRI); 1301 continue; 1302 } 1303 } 1304 1305 if (RISCVII::hasSEWOp(TSFlags)) { 1306 if (RISCVII::hasVLOp(TSFlags)) { 1307 const auto Require = computeInfoForInstr(MI, TSFlags, MRI); 1308 // Two cases involving an AVL resulting from a previous vsetvli. 1309 // 1) If the AVL is the result of a previous vsetvli which has the 1310 // same AVL and VLMAX as our current state, we can reuse the AVL 1311 // from the current state for the new one. This allows us to 1312 // generate 'vsetvli x0, x0, vtype" or possible skip the transition 1313 // entirely. 1314 // 2) If AVL is defined by a vsetvli with the same VLMAX, we can 1315 // replace the AVL operand with the AVL of the defining vsetvli. 1316 // We avoid general register AVLs to avoid extending live ranges 1317 // without being sure we can kill the original source reg entirely. 1318 if (Require.hasAVLReg() && Require.getAVLReg().isVirtual()) { 1319 if (MachineInstr *DefMI = MRI->getVRegDef(Require.getAVLReg())) { 1320 if (isVectorConfigInstr(*DefMI)) { 1321 VSETVLIInfo DefInfo = getInfoForVSETVLI(*DefMI); 1322 // case 1 1323 if (!CurInfo.isUnknown() && DefInfo.hasSameAVL(CurInfo) && 1324 DefInfo.hasSameVLMAX(CurInfo)) { 1325 MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI)); 1326 if (CurInfo.hasAVLImm()) 1327 VLOp.ChangeToImmediate(CurInfo.getAVLImm()); 1328 else { 1329 MRI->clearKillFlags(CurInfo.getAVLReg()); 1330 VLOp.ChangeToRegister(CurInfo.getAVLReg(), /*IsDef*/ false); 1331 } 1332 CurInfo = computeInfoForInstr(MI, TSFlags, MRI); 1333 continue; 1334 } 1335 // case 2 1336 if (DefInfo.hasSameVLMAX(Require) && 1337 (DefInfo.hasAVLImm() || DefInfo.getAVLReg() == RISCV::X0)) { 1338 MachineOperand &VLOp = MI.getOperand(getVLOpNum(MI)); 1339 if (DefInfo.hasAVLImm()) 1340 VLOp.ChangeToImmediate(DefInfo.getAVLImm()); 1341 else 1342 VLOp.ChangeToRegister(DefInfo.getAVLReg(), /*IsDef*/ false); 1343 CurInfo = computeInfoForInstr(MI, TSFlags, MRI); 1344 continue; 1345 } 1346 } 1347 } 1348 } 1349 } 1350 CurInfo = computeInfoForInstr(MI, TSFlags, MRI); 1351 continue; 1352 } 1353 1354 transferAfter(CurInfo, MI); 1355 } 1356 } 1357 1358 /// Return true if the VL value configured must be equal to the requested one. 1359 static bool hasFixedResult(const VSETVLIInfo &Info, const RISCVSubtarget &ST) { 1360 if (!Info.hasAVLImm()) 1361 // VLMAX is always the same value. 1362 // TODO: Could extend to other registers by looking at the associated vreg 1363 // def placement. 1364 return RISCV::X0 == Info.getAVLReg(); 1365 1366 unsigned AVL = Info.getAVLImm(); 1367 unsigned SEW = Info.getSEW(); 1368 unsigned AVLInBits = AVL * SEW; 1369 1370 unsigned LMul; 1371 bool Fractional; 1372 std::tie(LMul, Fractional) = RISCVVType::decodeVLMUL(Info.getVLMUL()); 1373 1374 if (Fractional) 1375 return ST.getRealMinVLen() / LMul >= AVLInBits; 1376 return ST.getRealMinVLen() * LMul >= AVLInBits; 1377 } 1378 1379 /// Perform simple partial redundancy elimination of the VSETVLI instructions 1380 /// we're about to insert by looking for cases where we can PRE from the 1381 /// beginning of one block to the end of one of its predecessors. Specifically, 1382 /// this is geared to catch the common case of a fixed length vsetvl in a single 1383 /// block loop when it could execute once in the preheader instead. 1384 void RISCVInsertVSETVLI::doPRE(MachineBasicBlock &MBB) { 1385 const MachineFunction &MF = *MBB.getParent(); 1386 const RISCVSubtarget &ST = MF.getSubtarget<RISCVSubtarget>(); 1387 1388 if (!BlockInfo[MBB.getNumber()].Pred.isUnknown()) 1389 return; 1390 1391 MachineBasicBlock *UnavailablePred = nullptr; 1392 VSETVLIInfo AvailableInfo; 1393 for (MachineBasicBlock *P : MBB.predecessors()) { 1394 const VSETVLIInfo &PredInfo = BlockInfo[P->getNumber()].Exit; 1395 if (PredInfo.isUnknown()) { 1396 if (UnavailablePred) 1397 return; 1398 UnavailablePred = P; 1399 } else if (!AvailableInfo.isValid()) { 1400 AvailableInfo = PredInfo; 1401 } else if (AvailableInfo != PredInfo) { 1402 return; 1403 } 1404 } 1405 1406 // Unreachable, single pred, or full redundancy. Note that FRE is handled by 1407 // phase 3. 1408 if (!UnavailablePred || !AvailableInfo.isValid()) 1409 return; 1410 1411 // Critical edge - TODO: consider splitting? 1412 if (UnavailablePred->succ_size() != 1) 1413 return; 1414 1415 // If VL can be less than AVL, then we can't reduce the frequency of exec. 1416 if (!hasFixedResult(AvailableInfo, ST)) 1417 return; 1418 1419 // Does it actually let us remove an implicit transition in MBB? 1420 bool Found = false; 1421 for (auto &MI : MBB) { 1422 if (isVectorConfigInstr(MI)) 1423 return; 1424 1425 const uint64_t TSFlags = MI.getDesc().TSFlags; 1426 if (RISCVII::hasSEWOp(TSFlags)) { 1427 if (AvailableInfo != computeInfoForInstr(MI, TSFlags, MRI)) 1428 return; 1429 Found = true; 1430 break; 1431 } 1432 } 1433 if (!Found) 1434 return; 1435 1436 // Finally, update both data flow state and insert the actual vsetvli. 1437 // Doing both keeps the code in sync with the dataflow results, which 1438 // is critical for correctness of phase 3. 1439 auto OldInfo = BlockInfo[UnavailablePred->getNumber()].Exit; 1440 LLVM_DEBUG(dbgs() << "PRE VSETVLI from " << MBB.getName() << " to " 1441 << UnavailablePred->getName() << " with state " 1442 << AvailableInfo << "\n"); 1443 BlockInfo[UnavailablePred->getNumber()].Exit = AvailableInfo; 1444 BlockInfo[MBB.getNumber()].Pred = AvailableInfo; 1445 1446 // Note there's an implicit assumption here that terminators never use 1447 // or modify VL or VTYPE. Also, fallthrough will return end(). 1448 auto InsertPt = UnavailablePred->getFirstInstrTerminator(); 1449 insertVSETVLI(*UnavailablePred, InsertPt, 1450 UnavailablePred->findDebugLoc(InsertPt), 1451 AvailableInfo, OldInfo); 1452 } 1453 1454 static void doUnion(DemandedFields &A, DemandedFields B) { 1455 A.VL |= B.VL; 1456 A.SEW |= B.SEW; 1457 A.LMUL |= B.LMUL; 1458 A.SEWLMULRatio |= B.SEWLMULRatio; 1459 A.TailPolicy |= B.TailPolicy; 1460 A.MaskPolicy |= B.MaskPolicy; 1461 } 1462 1463 // Return true if we can mutate PrevMI's VTYPE to match MI's 1464 // without changing any the fields which have been used. 1465 // TODO: Restructure code to allow code reuse between this and isCompatible 1466 // above. 1467 static bool canMutatePriorConfig(const MachineInstr &PrevMI, 1468 const MachineInstr &MI, 1469 const DemandedFields &Used) { 1470 // TODO: Extend this to handle cases where VL does change, but VL 1471 // has not been used. (e.g. over a vmv.x.s) 1472 if (!isVLPreservingConfig(MI)) 1473 // Note: `vsetvli x0, x0, vtype' is the canonical instruction 1474 // for this case. If you find yourself wanting to add other forms 1475 // to this "unused VTYPE" case, we're probably missing a 1476 // canonicalization earlier. 1477 return false; 1478 1479 if (!PrevMI.getOperand(2).isImm() || !MI.getOperand(2).isImm()) 1480 return false; 1481 1482 auto PriorVType = PrevMI.getOperand(2).getImm(); 1483 auto VType = MI.getOperand(2).getImm(); 1484 return areCompatibleVTYPEs(PriorVType, VType, Used); 1485 } 1486 1487 void RISCVInsertVSETVLI::doLocalPostpass(MachineBasicBlock &MBB) { 1488 MachineInstr *PrevMI = nullptr; 1489 DemandedFields Used; 1490 SmallVector<MachineInstr*> ToDelete; 1491 for (MachineInstr &MI : MBB) { 1492 // Note: Must be *before* vsetvli handling to account for config cases 1493 // which only change some subfields. 1494 doUnion(Used, getDemanded(MI)); 1495 1496 if (!isVectorConfigInstr(MI)) 1497 continue; 1498 1499 if (PrevMI) { 1500 if (!Used.VL && !Used.usedVTYPE()) { 1501 ToDelete.push_back(PrevMI); 1502 // fallthrough 1503 } else if (canMutatePriorConfig(*PrevMI, MI, Used)) { 1504 PrevMI->getOperand(2).setImm(MI.getOperand(2).getImm()); 1505 ToDelete.push_back(&MI); 1506 // Leave PrevMI unchanged 1507 continue; 1508 } 1509 } 1510 PrevMI = &MI; 1511 Used = getDemanded(MI); 1512 Register VRegDef = MI.getOperand(0).getReg(); 1513 if (VRegDef != RISCV::X0 && 1514 !(VRegDef.isVirtual() && MRI->use_nodbg_empty(VRegDef))) 1515 Used.VL = true; 1516 } 1517 1518 for (auto *MI : ToDelete) 1519 MI->eraseFromParent(); 1520 } 1521 1522 void RISCVInsertVSETVLI::insertReadVL(MachineBasicBlock &MBB) { 1523 for (auto I = MBB.begin(), E = MBB.end(); I != E;) { 1524 MachineInstr &MI = *I++; 1525 if (RISCV::isFaultFirstLoad(MI)) { 1526 Register VLOutput = MI.getOperand(1).getReg(); 1527 if (!MRI->use_nodbg_empty(VLOutput)) 1528 BuildMI(MBB, I, MI.getDebugLoc(), TII->get(RISCV::PseudoReadVL), 1529 VLOutput); 1530 // We don't use the vl output of the VLEFF/VLSEGFF anymore. 1531 MI.getOperand(1).setReg(RISCV::X0); 1532 } 1533 } 1534 } 1535 1536 bool RISCVInsertVSETVLI::runOnMachineFunction(MachineFunction &MF) { 1537 // Skip if the vector extension is not enabled. 1538 const RISCVSubtarget &ST = MF.getSubtarget<RISCVSubtarget>(); 1539 if (!ST.hasVInstructions()) 1540 return false; 1541 1542 LLVM_DEBUG(dbgs() << "Entering InsertVSETVLI for " << MF.getName() << "\n"); 1543 1544 TII = ST.getInstrInfo(); 1545 MRI = &MF.getRegInfo(); 1546 1547 assert(BlockInfo.empty() && "Expect empty block infos"); 1548 BlockInfo.resize(MF.getNumBlockIDs()); 1549 1550 // Scan the block locally for cases where we can mutate the operands 1551 // of the instructions to reduce state transitions. Critically, this 1552 // must be done before we start propagating data flow states as these 1553 // transforms are allowed to change the contents of VTYPE and VL so 1554 // long as the semantics of the program stays the same. 1555 for (MachineBasicBlock &MBB : MF) 1556 doLocalPrepass(MBB); 1557 1558 bool HaveVectorOp = false; 1559 1560 // Phase 1 - determine how VL/VTYPE are affected by the each block. 1561 for (const MachineBasicBlock &MBB : MF) { 1562 HaveVectorOp |= computeVLVTYPEChanges(MBB); 1563 // Initial exit state is whatever change we found in the block. 1564 BlockData &BBInfo = BlockInfo[MBB.getNumber()]; 1565 BBInfo.Exit = BBInfo.Change; 1566 LLVM_DEBUG(dbgs() << "Initial exit state of " << printMBBReference(MBB) 1567 << " is " << BBInfo.Exit << "\n"); 1568 1569 } 1570 1571 // If we didn't find any instructions that need VSETVLI, we're done. 1572 if (!HaveVectorOp) { 1573 BlockInfo.clear(); 1574 return false; 1575 } 1576 1577 // Phase 2 - determine the exit VL/VTYPE from each block. We add all 1578 // blocks to the list here, but will also add any that need to be revisited 1579 // during Phase 2 processing. 1580 for (const MachineBasicBlock &MBB : MF) { 1581 WorkList.push(&MBB); 1582 BlockInfo[MBB.getNumber()].InQueue = true; 1583 } 1584 while (!WorkList.empty()) { 1585 const MachineBasicBlock &MBB = *WorkList.front(); 1586 WorkList.pop(); 1587 computeIncomingVLVTYPE(MBB); 1588 } 1589 1590 // Perform partial redundancy elimination of vsetvli transitions. 1591 for (MachineBasicBlock &MBB : MF) 1592 doPRE(MBB); 1593 1594 // Phase 3 - add any vsetvli instructions needed in the block. Use the 1595 // Phase 2 information to avoid adding vsetvlis before the first vector 1596 // instruction in the block if the VL/VTYPE is satisfied by its 1597 // predecessors. 1598 for (MachineBasicBlock &MBB : MF) 1599 emitVSETVLIs(MBB); 1600 1601 // Now that all vsetvlis are explicit, go through and do block local 1602 // DSE and peephole based demanded fields based transforms. Note that 1603 // this *must* be done outside the main dataflow so long as we allow 1604 // any cross block analysis within the dataflow. We can't have both 1605 // demanded fields based mutation and non-local analysis in the 1606 // dataflow at the same time without introducing inconsistencies. 1607 for (MachineBasicBlock &MBB : MF) 1608 doLocalPostpass(MBB); 1609 1610 // Once we're fully done rewriting all the instructions, do a final pass 1611 // through to check for VSETVLIs which write to an unused destination. 1612 // For the non X0, X0 variant, we can replace the destination register 1613 // with X0 to reduce register pressure. This is really a generic 1614 // optimization which can be applied to any dead def (TODO: generalize). 1615 for (MachineBasicBlock &MBB : MF) { 1616 for (MachineInstr &MI : MBB) { 1617 if (MI.getOpcode() == RISCV::PseudoVSETVLI || 1618 MI.getOpcode() == RISCV::PseudoVSETIVLI) { 1619 Register VRegDef = MI.getOperand(0).getReg(); 1620 if (VRegDef != RISCV::X0 && MRI->use_nodbg_empty(VRegDef)) 1621 MI.getOperand(0).setReg(RISCV::X0); 1622 } 1623 } 1624 } 1625 1626 // Insert PseudoReadVL after VLEFF/VLSEGFF and replace it with the vl output 1627 // of VLEFF/VLSEGFF. 1628 for (MachineBasicBlock &MBB : MF) 1629 insertReadVL(MBB); 1630 1631 BlockInfo.clear(); 1632 return HaveVectorOp; 1633 } 1634 1635 /// Returns an instance of the Insert VSETVLI pass. 1636 FunctionPass *llvm::createRISCVInsertVSETVLIPass() { 1637 return new RISCVInsertVSETVLI(); 1638 } 1639