Conversion/VectorToGPU/VectorToGPU.cpp

edd9515bSthomasraoux//===- VectorToGPU.cpp - Convert vector to GPU dialect ----------*- C++ -*-===//
edd9515bSthomasraoux//
edd9515bSthomasraoux// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
edd9515bSthomasraoux// See https://llvm.org/LICENSE.txt for license information.
edd9515bSthomasraoux// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
edd9515bSthomasraoux//
edd9515bSthomasraoux//===----------------------------------------------------------------------===//
edd9515bSthomasraoux//
edd9515bSthomasraoux// This file implements lowering of vector operations to GPU dialect ops.
edd9515bSthomasraoux//
edd9515bSthomasraoux//===----------------------------------------------------------------------===//
edd9515bSthomasraoux
edd9515bSthomasraoux#include <type_traits>
edd9515bSthomasraoux
edd9515bSthomasraoux#include "mlir/Conversion/VectorToGPU/VectorToGPU.h"
edd9515bSthomasraoux
edd9515bSthomasraoux#include "../PassDetail.h"
edd9515bSthomasraoux#include "mlir/Analysis/SliceAnalysis.h"
edd9515bSthomasraoux#include "mlir/Dialect/GPU/GPUDialect.h"
66f878ceSMatthias Springer#include "mlir/Dialect/MemRef/IR/MemRef.h"
*1a865592Sthomasraoux#include "mlir/Dialect/SCF/SCF.h"
edd9515bSthomasraoux#include "mlir/Dialect/Utils/StructuredOpsUtils.h"
edd9515bSthomasraoux#include "mlir/Dialect/Vector/VectorOps.h"
edd9515bSthomasraoux#include "mlir/Dialect/Vector/VectorUtils.h"
edd9515bSthomasraoux#include "mlir/IR/Builders.h"
edd9515bSthomasraoux#include "mlir/Pass/Pass.h"
edd9515bSthomasraoux#include "mlir/Transforms/GreedyPatternRewriteDriver.h"
edd9515bSthomasraoux#include "mlir/Transforms/Passes.h"
edd9515bSthomasraoux
edd9515bSthomasraouxusing namespace mlir;
edd9515bSthomasraoux
edd9515bSthomasraoux// Return true if the contract op can be convert to MMA matmul.
edd9515bSthomasraouxstatic bool contractSupportsMMAMatrixType(vector::ContractionOp contract) {
edd9515bSthomasraoux  if (llvm::size(contract.masks()) != 0)
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux
edd9515bSthomasraoux  using MapList = ArrayRef<ArrayRef<AffineExpr>>;
edd9515bSthomasraoux  auto infer = [](MapList m) { return AffineMap::inferFromExprList(m); };
edd9515bSthomasraoux  AffineExpr m, n, k;
edd9515bSthomasraoux  bindDims(contract.getContext(), m, n, k);
edd9515bSthomasraoux  auto iteratorTypes = contract.iterator_types().getValue();
edd9515bSthomasraoux  if (!(isParallelIterator(iteratorTypes[0]) &&
edd9515bSthomasraoux        isParallelIterator(iteratorTypes[1]) &&
edd9515bSthomasraoux        isReductionIterator(iteratorTypes[2])))
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux
edd9515bSthomasraoux  // The contract needs to represent a matmul to be able to convert to
edd9515bSthomasraoux  // MMAMatrix matmul.
edd9515bSthomasraoux  if (contract.getIndexingMaps() != infer({{m, k}, {k, n}, {m, n}}))
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux
edd9515bSthomasraoux  // Check that the size matches what is natively supported.
edd9515bSthomasraoux  VectorType lhsType = contract.lhs().getType().cast<VectorType>();
edd9515bSthomasraoux  VectorType rhsType = contract.rhs().getType().cast<VectorType>();
edd9515bSthomasraoux  VectorType accType = contract.acc().getType().cast<VectorType>();
edd9515bSthomasraoux
edd9515bSthomasraoux  std::tuple<int, int, int> dim(lhsType.getDimSize(0), rhsType.getDimSize(1),
edd9515bSthomasraoux                                lhsType.getDimSize(1));
edd9515bSthomasraoux  if (lhsType.getElementType().isInteger(8) &&
edd9515bSthomasraoux      rhsType.getElementType().isInteger(8) &&
edd9515bSthomasraoux      accType.getElementType().isInteger(32) &&
edd9515bSthomasraoux      (dim == std::make_tuple(8, 8, 32) || dim == std::make_tuple(16, 16, 32) ||
edd9515bSthomasraoux       dim == std::make_tuple(16, 8, 32)))
edd9515bSthomasraoux    return true;
edd9515bSthomasraoux
edd9515bSthomasraoux  if (lhsType.getElementType().isF16() && rhsType.getElementType().isF16() &&
edd9515bSthomasraoux      (accType.getElementType().isF16() || accType.getElementType().isF32()) &&
edd9515bSthomasraoux      (dim == std::make_tuple(8, 8, 16) || dim == std::make_tuple(16, 16, 16) ||
edd9515bSthomasraoux       dim == std::make_tuple(16, 8, 16)))
edd9515bSthomasraoux    return true;
edd9515bSthomasraoux  return false;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux// Return the stide for the dimension 0 of |type| if it is a memref and has a
edd9515bSthomasraoux// constant stride.
edd9515bSthomasraouxstatic llvm::Optional<int64_t>
edd9515bSthomasraouxgetMemrefConstantHorizontalStride(ShapedType type) {
edd9515bSthomasraoux  auto memrefType = type.dyn_cast<MemRefType>();
edd9515bSthomasraoux  if (!memrefType)
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  int64_t offset = 0;
edd9515bSthomasraoux  SmallVector<int64_t, 2> strides;
edd9515bSthomasraoux  if (failed(getStridesAndOffset(memrefType, strides, offset)))
edd9515bSthomasraoux    return llvm::None;
edd9515bSthomasraoux  if (strides[0] == ShapedType::kDynamicStrideOrOffset)
edd9515bSthomasraoux    return llvm::None;
edd9515bSthomasraoux  return strides[0];
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux// Return true if the transfer op can be converted to a MMA matrix load.
edd9515bSthomasraouxstatic bool transferReadSupportsMMAMatrixType(vector::TransferReadOp readOp) {
edd9515bSthomasraoux  if (readOp.mask() || readOp.hasOutOfBoundsDim() ||
edd9515bSthomasraoux      readOp.getVectorType().getRank() != 2)
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  if (!getMemrefConstantHorizontalStride(readOp.getShapedType()))
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  // TODO: Support transpose once it is added to GPU dialect ops.
edd9515bSthomasraoux  if (!readOp.permutation_map().isMinorIdentity())
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  return true;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux// Return true if the transfer op can be converted to a MMA matrix store.
edd9515bSthomasraouxstatic bool
edd9515bSthomasraouxtransferWriteSupportsMMAMatrixType(vector::TransferWriteOp writeOp) {
edd9515bSthomasraoux  if (writeOp.mask() || writeOp.hasOutOfBoundsDim() ||
edd9515bSthomasraoux      writeOp.getVectorType().getRank() != 2)
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  if (!getMemrefConstantHorizontalStride(writeOp.getShapedType()))
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  // TODO: Support transpose once it is added to GPU dialect ops.
edd9515bSthomasraoux  if (!writeOp.permutation_map().isMinorIdentity())
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  return true;
edd9515bSthomasraoux}
edd9515bSthomasraoux
6413226dSthomasraoux/// Return true if the constant is a splat to a 2D vector so that it can be
6413226dSthomasraoux/// converted to a MMA constant matrix op.
6413226dSthomasraouxstatic bool constantSupportsMMAMatrixType(ConstantOp constantOp) {
6413226dSthomasraoux  auto vecType = constantOp.getType().dyn_cast<VectorType>();
6413226dSthomasraoux  if (!vecType || vecType.getRank() != 2)
6413226dSthomasraoux    return false;
6413226dSthomasraoux  return constantOp.value().isa<SplatElementsAttr>();
6413226dSthomasraoux}
6413226dSthomasraoux
edd9515bSthomasraouxstatic bool supportsMMaMatrixType(Operation *op) {
*1a865592Sthomasraoux  if (isa<scf::ForOp, scf::YieldOp>(op))
*1a865592Sthomasraoux    return true;
edd9515bSthomasraoux  if (auto transferRead = dyn_cast<vector::TransferReadOp>(op))
edd9515bSthomasraoux    return transferReadSupportsMMAMatrixType(transferRead);
edd9515bSthomasraoux  if (auto transferWrite = dyn_cast<vector::TransferWriteOp>(op))
edd9515bSthomasraoux    return transferWriteSupportsMMAMatrixType(transferWrite);
edd9515bSthomasraoux  if (auto contract = dyn_cast<vector::ContractionOp>(op))
edd9515bSthomasraoux    return contractSupportsMMAMatrixType(contract);
6413226dSthomasraoux  if (auto constant = dyn_cast<ConstantOp>(op))
6413226dSthomasraoux    return constantSupportsMMAMatrixType(constant);
edd9515bSthomasraoux  return false;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux// Analyze slice of operations based on convert op to figure out if the whole
edd9515bSthomasraoux// slice can be converted to MMA operations.
edd9515bSthomasraouxstatic SetVector<Operation *> getOpToConvert(mlir::Operation *op) {
edd9515bSthomasraoux  auto hasVectorDest = [](Operation *op) {
edd9515bSthomasraoux    return op->getNumResults() == 0 ||
edd9515bSthomasraoux           llvm::any_of(op->getResultTypes(),
edd9515bSthomasraoux                        [](Type t) { return t.isa<VectorType>(); });
edd9515bSthomasraoux  };
edd9515bSthomasraoux  SetVector<Operation *> opToConvert;
edd9515bSthomasraoux  op->walk([&](vector::ContractionOp contract) {
edd9515bSthomasraoux    if (opToConvert.contains(contract.getOperation()))
edd9515bSthomasraoux      return;
edd9515bSthomasraoux    SetVector<Operation *> dependentOps =
edd9515bSthomasraoux        getSlice(contract, hasVectorDest, hasVectorDest);
edd9515bSthomasraoux    // If any instruction cannot use MMA matrix type drop the whole
edd9515bSthomasraoux    // chaine. MMA matrix are stored in an opaque type so they cannot be used
edd9515bSthomasraoux    // by all operations.
edd9515bSthomasraoux    if (llvm::any_of(dependentOps,
edd9515bSthomasraoux                     [](Operation *op) { return !supportsMMaMatrixType(op); }))
edd9515bSthomasraoux      return;
edd9515bSthomasraoux    opToConvert.insert(dependentOps.begin(), dependentOps.end());
edd9515bSthomasraoux  });
edd9515bSthomasraoux  return opToConvert;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxnamespace {
edd9515bSthomasraoux// Transform contract into (m, k)x(k, n)x(m, n) form so that it can be converted
edd9515bSthomasraoux// to MMA matmul.
edd9515bSthomasraouxstruct PrepareContractToGPUMMA
edd9515bSthomasraoux    : public OpRewritePattern<vector::ContractionOp> {
edd9515bSthomasraoux  using OpRewritePattern<vector::ContractionOp>::OpRewritePattern;
edd9515bSthomasraoux
edd9515bSthomasraoux  LogicalResult matchAndRewrite(vector::ContractionOp op,
edd9515bSthomasraoux                                PatternRewriter &rewriter) const override {
edd9515bSthomasraoux    Location loc = op.getLoc();
edd9515bSthomasraoux    Value lhs = op.lhs(), rhs = op.rhs(), res = op.acc();
edd9515bSthomasraoux
edd9515bSthomasraoux    // Set up the parallel/reduction structure in right form.
edd9515bSthomasraoux    using MapList = ArrayRef<ArrayRef<AffineExpr>>;
edd9515bSthomasraoux    auto infer = [](MapList m) { return AffineMap::inferFromExprList(m); };
edd9515bSthomasraoux    AffineExpr m, n, k;
edd9515bSthomasraoux    bindDims(rewriter.getContext(), m, n, k);
edd9515bSthomasraoux    static constexpr std::array<int64_t, 2> perm = {1, 0};
edd9515bSthomasraoux    auto iteratorTypes = op.iterator_types().getValue();
edd9515bSthomasraoux    SmallVector<AffineMap, 4> maps = op.getIndexingMaps();
edd9515bSthomasraoux    if (!(isParallelIterator(iteratorTypes[0]) &&
edd9515bSthomasraoux          isParallelIterator(iteratorTypes[1]) &&
edd9515bSthomasraoux          isReductionIterator(iteratorTypes[2])))
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    //
edd9515bSthomasraoux    // Two outer parallel, one inner reduction (matmat flavor).
edd9515bSthomasraoux    //
edd9515bSthomasraoux    if (maps == infer({{m, k}, {k, n}, {m, n}})) {
edd9515bSthomasraoux      // This is the classical row-major matmul, nothing to do.
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    }
edd9515bSthomasraoux    if (maps == infer({{m, k}, {n, k}, {m, n}})) {
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {k, n}, {m, n}})) {
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {n, k}, {m, n}})) {
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{m, k}, {k, n}, {n, m}})) {
edd9515bSthomasraoux      std::swap(rhs, lhs);
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{m, k}, {n, k}, {n, m}})) {
edd9515bSthomasraoux      std::swap(rhs, lhs);
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {k, n}, {n, m}})) {
edd9515bSthomasraoux      std::swap(lhs, rhs);
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {n, k}, {n, m}})) {
edd9515bSthomasraoux      std::swap(lhs, rhs);
edd9515bSthomasraoux    } else {
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    }
edd9515bSthomasraoux    rewriter.replaceOpWithNewOp<vector::ContractionOp>(
edd9515bSthomasraoux        op, lhs, rhs, res,
edd9515bSthomasraoux        rewriter.getAffineMapArrayAttr(infer({{m, k}, {k, n}, {m, n}})),
edd9515bSthomasraoux        op.iterator_types());
edd9515bSthomasraoux    return success();
edd9515bSthomasraoux  }
edd9515bSthomasraoux};
edd9515bSthomasraoux
edd9515bSthomasraoux// Merge transpose op into the transfer read op. Transpose are not supported on
edd9515bSthomasraoux// MMA types but MMA load can transpose the matrix when loading.
edd9515bSthomasraouxstruct CombineTransferReadOpTranspose final
edd9515bSthomasraoux    : public OpRewritePattern<vector::TransposeOp> {
edd9515bSthomasraoux  using OpRewritePattern<vector::TransposeOp>::OpRewritePattern;
edd9515bSthomasraoux
edd9515bSthomasraoux  LogicalResult matchAndRewrite(vector::TransposeOp op,
edd9515bSthomasraoux                                PatternRewriter &rewriter) const override {
edd9515bSthomasraoux    auto transferReadOp = op.vector().getDefiningOp<vector::TransferReadOp>();
edd9515bSthomasraoux    if (!transferReadOp)
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    if (transferReadOp.mask() || transferReadOp.hasOutOfBoundsDim())
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    SmallVector<int64_t, 2> perm;
edd9515bSthomasraoux    op.getTransp(perm);
edd9515bSthomasraoux    SmallVector<unsigned, 2> permU;
edd9515bSthomasraoux    for (int64_t o : perm)
edd9515bSthomasraoux      permU.push_back(unsigned(o));
edd9515bSthomasraoux    AffineMap permutationMap =
edd9515bSthomasraoux        AffineMap::getPermutationMap(permU, op.getContext());
edd9515bSthomasraoux    AffineMap newMap = permutationMap.compose(transferReadOp.permutation_map());
edd9515bSthomasraoux    rewriter.replaceOpWithNewOp<vector::TransferReadOp>(
edd9515bSthomasraoux        op, op.getType(), transferReadOp.source(), transferReadOp.indices(),
edd9515bSthomasraoux        newMap, transferReadOp.padding(), transferReadOp.mask(),
edd9515bSthomasraoux        transferReadOp.in_boundsAttr());
edd9515bSthomasraoux    return success();
edd9515bSthomasraoux  }
edd9515bSthomasraoux};
edd9515bSthomasraoux
edd9515bSthomasraoux} // namespace
edd9515bSthomasraoux
edd9515bSthomasraoux// MMA types have different layout based on how they are used in matmul ops.
6413226dSthomasraoux// Figure the right layout to use by looking at op uses.
edd9515bSthomasraoux// TODO: Change the GPU dialect to abstract the layout at the this level and
edd9515bSthomasraoux// only care about it during lowering to NVVM.
6413226dSthomasraouxtemplate <typename OpTy>
6413226dSthomasraouxstatic const char *inferFragType(OpTy op) {
edd9515bSthomasraoux  for (Operation *users : op->getUsers()) {
edd9515bSthomasraoux    auto contract = dyn_cast<vector::ContractionOp>(users);
edd9515bSthomasraoux    if (!contract)
edd9515bSthomasraoux      continue;
edd9515bSthomasraoux    if (contract.lhs() == op.getResult())
edd9515bSthomasraoux      return "AOp";
edd9515bSthomasraoux    if (contract.rhs() == op.getResult())
edd9515bSthomasraoux      return "BOp";
edd9515bSthomasraoux  }
edd9515bSthomasraoux  return "COp";
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxstatic void convertTransferReadOp(vector::TransferReadOp op,
edd9515bSthomasraoux                                  llvm::DenseMap<Value, Value> &valueMapping) {
edd9515bSthomasraoux  assert(transferReadSupportsMMAMatrixType(op));
edd9515bSthomasraoux  Optional<int64_t> stride =
edd9515bSthomasraoux      getMemrefConstantHorizontalStride(op.getShapedType());
edd9515bSthomasraoux  assert(stride);
edd9515bSthomasraoux  const char *fragType = inferFragType(op);
edd9515bSthomasraoux  gpu::MMAMatrixType type =
edd9515bSthomasraoux      gpu::MMAMatrixType::get(op.getVectorType().getShape(),
edd9515bSthomasraoux                              op.getVectorType().getElementType(), fragType);
edd9515bSthomasraoux  OpBuilder b(op);
edd9515bSthomasraoux  Value load = b.create<gpu::SubgroupMmaLoadMatrixOp>(
edd9515bSthomasraoux      op.getLoc(), type, op.source(), op.indices(), b.getIndexAttr(*stride));
edd9515bSthomasraoux  valueMapping[op.getResult()] = load;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxstatic void convertTransferWriteOp(vector::TransferWriteOp op,
edd9515bSthomasraoux                                   llvm::DenseMap<Value, Value> &valueMapping) {
edd9515bSthomasraoux  assert(transferWriteSupportsMMAMatrixType(op));
edd9515bSthomasraoux  Optional<int64_t> stride =
edd9515bSthomasraoux      getMemrefConstantHorizontalStride(op.getShapedType());
edd9515bSthomasraoux  assert(stride);
edd9515bSthomasraoux  OpBuilder b(op);
edd9515bSthomasraoux  Value matrix = valueMapping.find(op.vector())->second;
edd9515bSthomasraoux  b.create<gpu::SubgroupMmaStoreMatrixOp>(
edd9515bSthomasraoux      op.getLoc(), matrix, op.source(), op.indices(), b.getIndexAttr(*stride));
edd9515bSthomasraoux  op.erase();
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxstatic void convertContractOp(vector::ContractionOp op,
edd9515bSthomasraoux                              llvm::DenseMap<Value, Value> &valueMapping) {
edd9515bSthomasraoux  OpBuilder b(op);
edd9515bSthomasraoux  Value opA = valueMapping.find(op.lhs())->second;
edd9515bSthomasraoux  Value opB = valueMapping.find(op.rhs())->second;
edd9515bSthomasraoux  Value opC = valueMapping.find(op.acc())->second;
edd9515bSthomasraoux  Value matmul = b.create<gpu::SubgroupMmaComputeOp>(op.getLoc(), opC.getType(),
edd9515bSthomasraoux                                                     opA, opB, opC);
edd9515bSthomasraoux  valueMapping[op.getResult()] = matmul;
edd9515bSthomasraoux}
edd9515bSthomasraoux
6413226dSthomasraoux/// Convert a 2D splat ConstantOp to a SubgroupMmaConstantMatrix op.
6413226dSthomasraouxstatic void convertConstantOp(ConstantOp op,
6413226dSthomasraoux                              llvm::DenseMap<Value, Value> &valueMapping) {
6413226dSthomasraoux  assert(constantSupportsMMAMatrixType(op));
6413226dSthomasraoux  OpBuilder b(op);
6413226dSthomasraoux  Attribute splat = op.getValue().cast<SplatElementsAttr>().getSplatValue();
6413226dSthomasraoux  auto scalarConstant =
6413226dSthomasraoux      b.create<ConstantOp>(op.getLoc(), splat.getType(), splat);
6413226dSthomasraoux  const char *fragType = inferFragType(op);
6413226dSthomasraoux  auto vecType = op.getType().cast<VectorType>();
6413226dSthomasraoux  gpu::MMAMatrixType type = gpu::MMAMatrixType::get(
6413226dSthomasraoux      vecType.getShape(), vecType.getElementType(), llvm::StringRef(fragType));
6413226dSthomasraoux  auto matrix = b.create<gpu::SubgroupMmaConstantMatrixOp>(op.getLoc(), type,
6413226dSthomasraoux                                                           scalarConstant);
6413226dSthomasraoux  valueMapping[op.getResult()] = matrix;
6413226dSthomasraoux}
6413226dSthomasraoux
*1a865592Sthomasraoux// Replace ForOp with a new ForOp with extra operands. The YieldOp is not
*1a865592Sthomasraoux// updated and needs to be updated separatly for the loop to be correct.
*1a865592Sthomasraouxstatic scf::ForOp replaceForOpWithNewSignature(OpBuilder &b, scf::ForOp loop,
*1a865592Sthomasraoux                                               ValueRange newIterOperands) {
*1a865592Sthomasraoux  // Create a new loop before the existing one, with the extra operands.
*1a865592Sthomasraoux  OpBuilder::InsertionGuard g(b);
*1a865592Sthomasraoux  b.setInsertionPoint(loop);
*1a865592Sthomasraoux  auto operands = llvm::to_vector<4>(loop.getIterOperands());
*1a865592Sthomasraoux  operands.append(newIterOperands.begin(), newIterOperands.end());
*1a865592Sthomasraoux  scf::ForOp newLoop =
*1a865592Sthomasraoux      b.create<scf::ForOp>(loop.getLoc(), loop.lowerBound(), loop.upperBound(),
*1a865592Sthomasraoux                           loop.step(), operands);
*1a865592Sthomasraoux  newLoop.getBody()->erase();
*1a865592Sthomasraoux  newLoop.getLoopBody().getBlocks().splice(
*1a865592Sthomasraoux      newLoop.getLoopBody().getBlocks().begin(),
*1a865592Sthomasraoux      loop.getLoopBody().getBlocks());
*1a865592Sthomasraoux  for (auto operand : newIterOperands)
*1a865592Sthomasraoux    newLoop.getBody()->addArgument(operand.getType());
*1a865592Sthomasraoux
*1a865592Sthomasraoux  for (auto it : llvm::zip(loop.getResults(), newLoop.getResults().take_front(
*1a865592Sthomasraoux                                                  loop.getNumResults())))
*1a865592Sthomasraoux    std::get<0>(it).replaceAllUsesWith(std::get<1>(it));
*1a865592Sthomasraoux  loop.erase();
*1a865592Sthomasraoux  return newLoop;
*1a865592Sthomasraoux}
*1a865592Sthomasraoux
*1a865592Sthomasraouxstatic void convertForOp(scf::ForOp op,
*1a865592Sthomasraoux                         llvm::DenseMap<Value, Value> &valueMapping) {
*1a865592Sthomasraoux  SmallVector<Value> newOperands;
*1a865592Sthomasraoux  SmallVector<std::pair<size_t, size_t>> argMapping;
*1a865592Sthomasraoux  for (auto operand : llvm::enumerate(op.getIterOperands())) {
*1a865592Sthomasraoux    auto it = valueMapping.find(operand.value());
*1a865592Sthomasraoux    if (it == valueMapping.end())
*1a865592Sthomasraoux      continue;
*1a865592Sthomasraoux    argMapping.push_back(std::make_pair(
*1a865592Sthomasraoux        operand.index(), op.getNumIterOperands() + newOperands.size()));
*1a865592Sthomasraoux    newOperands.push_back(it->second);
*1a865592Sthomasraoux  }
*1a865592Sthomasraoux  OpBuilder b(op);
*1a865592Sthomasraoux  scf::ForOp newForOp = replaceForOpWithNewSignature(b, op, newOperands);
*1a865592Sthomasraoux  Block &loopBody = *newForOp.getBody();
*1a865592Sthomasraoux  for (auto mapping : argMapping) {
*1a865592Sthomasraoux    valueMapping[newForOp.getResult(mapping.first)] =
*1a865592Sthomasraoux        newForOp.getResult(mapping.second);
*1a865592Sthomasraoux    valueMapping[loopBody.getArgument(mapping.first +
*1a865592Sthomasraoux                                      newForOp.getNumInductionVars())] =
*1a865592Sthomasraoux        loopBody.getArgument(mapping.second + newForOp.getNumInductionVars());
*1a865592Sthomasraoux  }
*1a865592Sthomasraoux}
*1a865592Sthomasraoux
*1a865592Sthomasraouxstatic void convertYieldOp(scf::YieldOp op,
*1a865592Sthomasraoux                           llvm::DenseMap<Value, Value> &valueMapping) {
*1a865592Sthomasraoux  OpBuilder b(op);
*1a865592Sthomasraoux  auto loop = cast<scf::ForOp>(op->getParentOp());
*1a865592Sthomasraoux  auto yieldOperands = llvm::to_vector<4>(op.getOperands());
*1a865592Sthomasraoux  for (auto operand : llvm::enumerate(op.getOperands())) {
*1a865592Sthomasraoux    auto it = valueMapping.find(operand.value());
*1a865592Sthomasraoux    if (it == valueMapping.end())
*1a865592Sthomasraoux      continue;
*1a865592Sthomasraoux    // Replace the yield of old value with the for op argument to make it easier
*1a865592Sthomasraoux    // to remove the dead code.
*1a865592Sthomasraoux    yieldOperands[operand.index()] = loop.getIterOperands()[operand.index()];
*1a865592Sthomasraoux    yieldOperands.push_back(it->second);
*1a865592Sthomasraoux  }
*1a865592Sthomasraoux  b.create<scf::YieldOp>(op.getLoc(), yieldOperands);
*1a865592Sthomasraoux  op.erase();
*1a865592Sthomasraoux}
*1a865592Sthomasraoux
edd9515bSthomasraouxnamespace mlir {
edd9515bSthomasraoux
edd9515bSthomasraouxvoid populatePrepareVectorToMMAPatterns(RewritePatternSet &patterns) {
edd9515bSthomasraoux  patterns.add<PrepareContractToGPUMMA, CombineTransferReadOpTranspose>(
edd9515bSthomasraoux      patterns.getContext());
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxvoid convertVectorToMMAOps(FuncOp funcOp) {
edd9515bSthomasraoux  SetVector<Operation *> ops = getOpToConvert(funcOp);
edd9515bSthomasraoux  llvm::DenseMap<Value, Value> valueMapping;
edd9515bSthomasraoux  for (Operation *op : ops) {
edd9515bSthomasraoux    if (auto transferRead = dyn_cast<vector::TransferReadOp>(op)) {
edd9515bSthomasraoux      convertTransferReadOp(transferRead, valueMapping);
edd9515bSthomasraoux    } else if (auto transferWrite = dyn_cast<vector::TransferWriteOp>(op)) {
edd9515bSthomasraoux      convertTransferWriteOp(transferWrite, valueMapping);
edd9515bSthomasraoux    } else if (auto contractOp = dyn_cast<vector::ContractionOp>(op)) {
edd9515bSthomasraoux      convertContractOp(contractOp, valueMapping);
6413226dSthomasraoux    } else if (auto constantOp = dyn_cast<ConstantOp>(op)) {
6413226dSthomasraoux      convertConstantOp(constantOp, valueMapping);
*1a865592Sthomasraoux    } else if (auto forOp = dyn_cast<scf::ForOp>(op)) {
*1a865592Sthomasraoux      convertForOp(forOp, valueMapping);
*1a865592Sthomasraoux    } else if (auto yiledOp = dyn_cast<scf::YieldOp>(op)) {
*1a865592Sthomasraoux      convertYieldOp(yiledOp, valueMapping);
edd9515bSthomasraoux    }
edd9515bSthomasraoux  }
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux} // namespace mlir
edd9515bSthomasraouxnamespace {
edd9515bSthomasraoux
edd9515bSthomasraouxstruct ConvertVectorToGPUPass
edd9515bSthomasraoux    : public ConvertVectorToGPUBase<ConvertVectorToGPUPass> {
edd9515bSthomasraoux  void runOnFunction() override {
edd9515bSthomasraoux    RewritePatternSet patterns(getFunction().getContext());
edd9515bSthomasraoux    populatePrepareVectorToMMAPatterns(patterns);
edd9515bSthomasraoux    (void)applyPatternsAndFoldGreedily(getFunction(), std::move(patterns));
edd9515bSthomasraoux
edd9515bSthomasraoux    convertVectorToMMAOps(getFunction());
edd9515bSthomasraoux  }
edd9515bSthomasraoux};
edd9515bSthomasraoux
edd9515bSthomasraoux} // namespace
edd9515bSthomasraoux
edd9515bSthomasraouxstd::unique_ptr<Pass> mlir::createConvertVectorToGPUPass() {
edd9515bSthomasraoux  return std::make_unique<ConvertVectorToGPUPass>();
edd9515bSthomasraoux}