Conversion/VectorToGPU/VectorToGPU.cpp

edd9515bSthomasraoux//===- VectorToGPU.cpp - Convert vector to GPU dialect ----------*- C++ -*-===//
edd9515bSthomasraoux//
edd9515bSthomasraoux// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
edd9515bSthomasraoux// See https://llvm.org/LICENSE.txt for license information.
edd9515bSthomasraoux// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
edd9515bSthomasraoux//
edd9515bSthomasraoux//===----------------------------------------------------------------------===//
edd9515bSthomasraoux//
edd9515bSthomasraoux// This file implements lowering of vector operations to GPU dialect ops.
edd9515bSthomasraoux//
edd9515bSthomasraoux//===----------------------------------------------------------------------===//
edd9515bSthomasraoux
edd9515bSthomasraoux#include <type_traits>
edd9515bSthomasraoux
*1ca772edSChristopher Bate#include "NvGpuSupport.h"
edd9515bSthomasraoux#include "mlir/Conversion/VectorToGPU/VectorToGPU.h"
edd9515bSthomasraoux
edd9515bSthomasraoux#include "../PassDetail.h"
edd9515bSthomasraoux#include "mlir/Analysis/SliceAnalysis.h"
a54f4eaeSMogball#include "mlir/Dialect/Arithmetic/IR/Arithmetic.h"
edd9515bSthomasraoux#include "mlir/Dialect/GPU/GPUDialect.h"
66f878ceSMatthias Springer#include "mlir/Dialect/MemRef/IR/MemRef.h"
*1ca772edSChristopher Bate#include "mlir/Dialect/NVGPU/NVGPUDialect.h"
1a865592Sthomasraoux#include "mlir/Dialect/SCF/SCF.h"
edd9515bSthomasraoux#include "mlir/Dialect/Utils/StructuredOpsUtils.h"
99ef9eebSMatthias Springer#include "mlir/Dialect/Vector/IR/VectorOps.h"
99ef9eebSMatthias Springer#include "mlir/Dialect/Vector/Utils/VectorUtils.h"
edd9515bSthomasraoux#include "mlir/IR/Builders.h"
edd9515bSthomasraoux#include "mlir/Pass/Pass.h"
edd9515bSthomasraoux#include "mlir/Transforms/GreedyPatternRewriteDriver.h"
edd9515bSthomasraoux#include "mlir/Transforms/Passes.h"
*1ca772edSChristopher Bate#include "llvm/ADT/TypeSwitch.h"
edd9515bSthomasraoux
edd9515bSthomasraouxusing namespace mlir;
edd9515bSthomasraoux
*1ca772edSChristopher Bate/// For a vector TransferOpType `xferOp`, an empty `indices` vector, and an
*1ca772edSChristopher Bate/// AffineMap representing offsets to apply to indices, the function fills
*1ca772edSChristopher Bate/// `indices` with the original indices plus the offsets. The offsets are
*1ca772edSChristopher Bate/// applied by taking into account the permutation map of the transfer op. If
*1ca772edSChristopher Bate/// the `offsetMap` has dimension placeholders, those should be provided in
*1ca772edSChristopher Bate/// `dimValues`.
*1ca772edSChristopher Batetemplate <typename TransferOpType>
*1ca772edSChristopher Batestatic void getXferIndices(OpBuilder &b, TransferOpType xferOp,
*1ca772edSChristopher Bate                           AffineMap offsetMap, ArrayRef<Value> dimValues,
*1ca772edSChristopher Bate                           SmallVector<Value, 4> &indices) {
*1ca772edSChristopher Bate  indices.append(xferOp.getIndices().begin(), xferOp.getIndices().end());
*1ca772edSChristopher Bate  Location loc = xferOp.getLoc();
*1ca772edSChristopher Bate  unsigned offsetsIdx = 0;
*1ca772edSChristopher Bate  for (auto expr : xferOp.getPermutationMap().getResults()) {
*1ca772edSChristopher Bate    if (auto dim = expr.template dyn_cast<AffineDimExpr>()) {
*1ca772edSChristopher Bate      Value prevIdx = indices[dim.getPosition()];
*1ca772edSChristopher Bate      SmallVector<Value, 3> dims(dimValues.begin(), dimValues.end());
*1ca772edSChristopher Bate      dims.push_back(prevIdx);
*1ca772edSChristopher Bate      AffineExpr d0 = b.getAffineDimExpr(offsetMap.getNumDims());
*1ca772edSChristopher Bate      indices[dim.getPosition()] = makeComposedAffineApply(
*1ca772edSChristopher Bate          b, loc, d0 + offsetMap.getResult(offsetsIdx++), dims);
*1ca772edSChristopher Bate      continue;
*1ca772edSChristopher Bate    }
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
edd9515bSthomasraoux// Return true if the contract op can be convert to MMA matmul.
*1ca772edSChristopher Batestatic bool contractSupportsMMAMatrixType(vector::ContractionOp contract,
*1ca772edSChristopher Bate                                          bool useNvGpu) {
7c38fd60SJacques Pienaar  if (llvm::size(contract.getMasks()) != 0)
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux
edd9515bSthomasraoux  using MapList = ArrayRef<ArrayRef<AffineExpr>>;
edd9515bSthomasraoux  auto infer = [](MapList m) { return AffineMap::inferFromExprList(m); };
edd9515bSthomasraoux  AffineExpr m, n, k;
edd9515bSthomasraoux  bindDims(contract.getContext(), m, n, k);
7c38fd60SJacques Pienaar  auto iteratorTypes = contract.getIteratorTypes().getValue();
edd9515bSthomasraoux  if (!(isParallelIterator(iteratorTypes[0]) &&
edd9515bSthomasraoux        isParallelIterator(iteratorTypes[1]) &&
edd9515bSthomasraoux        isReductionIterator(iteratorTypes[2])))
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux
edd9515bSthomasraoux  // The contract needs to represent a matmul to be able to convert to
edd9515bSthomasraoux  // MMAMatrix matmul.
*1ca772edSChristopher Bate  if (!useNvGpu &&
*1ca772edSChristopher Bate      contract.getIndexingMaps() != infer({{m, k}, {k, n}, {m, n}}))
*1ca772edSChristopher Bate    return false;
*1ca772edSChristopher Bate  if (useNvGpu && contract.getIndexingMaps() != infer({{m, k}, {n, k}, {m, n}}))
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux
edd9515bSthomasraoux  return true;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux// Return the stide for the dimension 0 of |type| if it is a memref and has a
edd9515bSthomasraoux// constant stride.
edd9515bSthomasraouxstatic llvm::Optional<int64_t>
edd9515bSthomasraouxgetMemrefConstantHorizontalStride(ShapedType type) {
edd9515bSthomasraoux  auto memrefType = type.dyn_cast<MemRefType>();
edd9515bSthomasraoux  if (!memrefType)
edd9515bSthomasraoux    return false;
a57ccad5SThomas Raoux  // If the memref is 0 or 1D the horizontal stride is 0.
a57ccad5SThomas Raoux  if (memrefType.getRank() < 2)
a57ccad5SThomas Raoux    return 0;
edd9515bSthomasraoux  int64_t offset = 0;
edd9515bSthomasraoux  SmallVector<int64_t, 2> strides;
d77f4836SThomas Raoux  if (failed(getStridesAndOffset(memrefType, strides, offset)) ||
d77f4836SThomas Raoux      strides.back() != 1)
edd9515bSthomasraoux    return llvm::None;
a57ccad5SThomas Raoux  int64_t stride = strides[strides.size() - 2];
a57ccad5SThomas Raoux  if (stride == ShapedType::kDynamicStrideOrOffset)
edd9515bSthomasraoux    return llvm::None;
a57ccad5SThomas Raoux  return stride;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux// Return true if the transfer op can be converted to a MMA matrix load.
*1ca772edSChristopher Batestatic bool transferReadSupportsMMAMatrixType(vector::TransferReadOp readOp,
*1ca772edSChristopher Bate                                              bool useNvGpu) {
7c38fd60SJacques Pienaar  if (readOp.getMask() || readOp.hasOutOfBoundsDim() ||
edd9515bSthomasraoux      readOp.getVectorType().getRank() != 2)
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  if (!getMemrefConstantHorizontalStride(readOp.getShapedType()))
edd9515bSthomasraoux    return false;
7c38fd60SJacques Pienaar  AffineMap map = readOp.getPermutationMap();
e7969240SThomas Raoux  OpBuilder b(readOp.getContext());
e7969240SThomas Raoux  AffineExpr innerDim = b.getAffineDimExpr(map.getNumDims() - 1);
e7969240SThomas Raoux  AffineExpr zero = b.getAffineConstantExpr(0);
e7969240SThomas Raoux  auto broadcastInnerDim = AffineMap::get(map.getNumDims(), 0, {zero, innerDim},
e7969240SThomas Raoux                                          readOp.getContext());
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  if (!useNvGpu) {
edd9515bSthomasraoux    // TODO: Support transpose once it is added to GPU dialect ops.
e7969240SThomas Raoux    // For now we only support (d0, d1) -> (d0, d1) and (d0, d1) -> (0, d1).
*1ca772edSChristopher Bate    return map.isMinorIdentity() || map == broadcastInnerDim;
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  return true;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraoux// Return true if the transfer op can be converted to a MMA matrix store.
edd9515bSthomasraouxstatic bool
edd9515bSthomasraouxtransferWriteSupportsMMAMatrixType(vector::TransferWriteOp writeOp) {
c537a943SNicolas Vasilache  // TODO: support 0-d corner case.
c537a943SNicolas Vasilache  if (writeOp.getTransferRank() == 0)
c537a943SNicolas Vasilache    return false;
c537a943SNicolas Vasilache
7c38fd60SJacques Pienaar  if (writeOp.getMask() || writeOp.hasOutOfBoundsDim() ||
edd9515bSthomasraoux      writeOp.getVectorType().getRank() != 2)
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  if (!getMemrefConstantHorizontalStride(writeOp.getShapedType()))
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  // TODO: Support transpose once it is added to GPU dialect ops.
7c38fd60SJacques Pienaar  if (!writeOp.getPermutationMap().isMinorIdentity())
edd9515bSthomasraoux    return false;
edd9515bSthomasraoux  return true;
edd9515bSthomasraoux}
edd9515bSthomasraoux
6413226dSthomasraoux/// Return true if the constant is a splat to a 2D vector so that it can be
6413226dSthomasraoux/// converted to a MMA constant matrix op.
a54f4eaeSMogballstatic bool constantSupportsMMAMatrixType(arith::ConstantOp constantOp) {
6413226dSthomasraoux  auto vecType = constantOp.getType().dyn_cast<VectorType>();
6413226dSthomasraoux  if (!vecType || vecType.getRank() != 2)
6413226dSthomasraoux    return false;
cfb72fd3SJacques Pienaar  return constantOp.getValue().isa<SplatElementsAttr>();
6413226dSthomasraoux}
6413226dSthomasraoux
43928419Sthomasraoux/// Return true if this is a broadcast from scalar to a 2D vector.
43928419Sthomasraouxstatic bool broadcastSupportsMMAMatrixType(vector::BroadcastOp broadcastOp) {
43928419Sthomasraoux  return broadcastOp.getVectorType().getRank() == 2 &&
7c38fd60SJacques Pienaar         broadcastOp.getSource().getType().isa<FloatType>();
43928419Sthomasraoux}
43928419Sthomasraoux
7fbb0678Sthomasraoux/// Return the MMA elementwise enum associated with `op` if it is supported.
7fbb0678Sthomasraoux/// Return `llvm::None` otherwise.
7fbb0678Sthomasraouxstatic llvm::Optional<gpu::MMAElementwiseOp>
7fbb0678SthomasraouxconvertElementwiseOpToMMA(Operation *op) {
7fbb0678Sthomasraoux  if (isa<arith::AddFOp>(op))
7fbb0678Sthomasraoux    return gpu::MMAElementwiseOp::ADDF;
7fbb0678Sthomasraoux  if (isa<arith::MulFOp>(op))
7fbb0678Sthomasraoux    return gpu::MMAElementwiseOp::MULF;
9b1d90e8SAlexander Belyaev  if (isa<arith::MaxFOp>(op))
7fbb0678Sthomasraoux    return gpu::MMAElementwiseOp::MAXF;
9b1d90e8SAlexander Belyaev  if (isa<arith::MinFOp>(op))
7fbb0678Sthomasraoux    return gpu::MMAElementwiseOp::MINF;
e7969240SThomas Raoux  if (isa<arith::DivFOp>(op))
e7969240SThomas Raoux    return gpu::MMAElementwiseOp::DIVF;
7fbb0678Sthomasraoux  return llvm::None;
7fbb0678Sthomasraoux}
7fbb0678Sthomasraoux
7fbb0678Sthomasraoux/// Return true if the op is supported as elementwise op on MMAMatrix type.
7fbb0678Sthomasraouxstatic bool elementwiseSupportsMMAMatrixType(Operation *op) {
7fbb0678Sthomasraoux  return convertElementwiseOpToMMA(op).hasValue();
7fbb0678Sthomasraoux}
7fbb0678Sthomasraoux
*1ca772edSChristopher Batestatic bool supportsMMaMatrixType(Operation *op, bool useNvGpu) {
1a865592Sthomasraoux  if (isa<scf::ForOp, scf::YieldOp>(op))
1a865592Sthomasraoux    return true;
edd9515bSthomasraoux  if (auto transferRead = dyn_cast<vector::TransferReadOp>(op))
*1ca772edSChristopher Bate    return transferReadSupportsMMAMatrixType(transferRead, useNvGpu);
edd9515bSthomasraoux  if (auto transferWrite = dyn_cast<vector::TransferWriteOp>(op))
edd9515bSthomasraoux    return transferWriteSupportsMMAMatrixType(transferWrite);
edd9515bSthomasraoux  if (auto contract = dyn_cast<vector::ContractionOp>(op))
*1ca772edSChristopher Bate    return contractSupportsMMAMatrixType(contract, useNvGpu);
a54f4eaeSMogball  if (auto constant = dyn_cast<arith::ConstantOp>(op))
6413226dSthomasraoux    return constantSupportsMMAMatrixType(constant);
43928419Sthomasraoux  if (auto broadcast = dyn_cast<vector::BroadcastOp>(op))
43928419Sthomasraoux    return broadcastSupportsMMAMatrixType(broadcast);
7fbb0678Sthomasraoux  return elementwiseSupportsMMAMatrixType(op);
edd9515bSthomasraoux}
edd9515bSthomasraoux
e7969240SThomas Raoux/// Return an unsorted slice handling scf.for region differently than
e7969240SThomas Raoux/// `getSlice`. In scf.for we only want to include as part of the slice elements
e7969240SThomas Raoux/// that are part of the use/def chain.
e7969240SThomas Raouxstatic SetVector<Operation *> getSliceContract(Operation *op,
e7969240SThomas Raoux                                               TransitiveFilter backwardFilter,
e7969240SThomas Raoux                                               TransitiveFilter forwardFilter) {
e7969240SThomas Raoux  SetVector<Operation *> slice;
e7969240SThomas Raoux  slice.insert(op);
e7969240SThomas Raoux  unsigned currentIndex = 0;
e7969240SThomas Raoux  SetVector<Operation *> backwardSlice;
e7969240SThomas Raoux  SetVector<Operation *> forwardSlice;
e7969240SThomas Raoux  while (currentIndex != slice.size()) {
e7969240SThomas Raoux    auto *currentOp = (slice)[currentIndex];
e7969240SThomas Raoux    // Compute and insert the backwardSlice starting from currentOp.
e7969240SThomas Raoux    backwardSlice.clear();
e7969240SThomas Raoux    getBackwardSlice(currentOp, &backwardSlice, backwardFilter);
e7969240SThomas Raoux    slice.insert(backwardSlice.begin(), backwardSlice.end());
e7969240SThomas Raoux
e7969240SThomas Raoux    // Compute and insert the forwardSlice starting from currentOp.
e7969240SThomas Raoux    forwardSlice.clear();
e7969240SThomas Raoux    // Special case for ForOp, we don't want to include the whole region but
e7969240SThomas Raoux    // only the value using the region arguments.
e7969240SThomas Raoux    // TODO: We should refine this to only care about the region arguments being
e7969240SThomas Raoux    // converted to matrix type.
e7969240SThomas Raoux    if (auto forOp = dyn_cast<scf::ForOp>(currentOp)) {
e7969240SThomas Raoux      for (Value forOpResult : forOp.getResults())
e7969240SThomas Raoux        getForwardSlice(forOpResult, &forwardSlice, forwardFilter);
e7969240SThomas Raoux      for (BlockArgument &arg : forOp.getRegionIterArgs())
e7969240SThomas Raoux        getForwardSlice(arg, &forwardSlice, forwardFilter);
e7969240SThomas Raoux    } else {
e7969240SThomas Raoux      getForwardSlice(currentOp, &forwardSlice, forwardFilter);
e7969240SThomas Raoux    }
e7969240SThomas Raoux    slice.insert(forwardSlice.begin(), forwardSlice.end());
e7969240SThomas Raoux    ++currentIndex;
e7969240SThomas Raoux  }
e7969240SThomas Raoux  return slice;
e7969240SThomas Raoux}
e7969240SThomas Raoux
edd9515bSthomasraoux// Analyze slice of operations based on convert op to figure out if the whole
edd9515bSthomasraoux// slice can be converted to MMA operations.
*1ca772edSChristopher Batestatic SetVector<Operation *> getOpToConvert(mlir::Operation *op,
*1ca772edSChristopher Bate                                             bool useNvGpu) {
edd9515bSthomasraoux  auto hasVectorDest = [](Operation *op) {
43928419Sthomasraoux    return llvm::any_of(op->getResultTypes(),
43928419Sthomasraoux                        [](Type t) { return t.isa<VectorType>(); });
43928419Sthomasraoux  };
43928419Sthomasraoux  auto hasVectorSrc = [](Operation *op) {
43928419Sthomasraoux    return llvm::any_of(op->getOperandTypes(),
edd9515bSthomasraoux                        [](Type t) { return t.isa<VectorType>(); });
edd9515bSthomasraoux  };
edd9515bSthomasraoux  SetVector<Operation *> opToConvert;
edd9515bSthomasraoux  op->walk([&](vector::ContractionOp contract) {
edd9515bSthomasraoux    if (opToConvert.contains(contract.getOperation()))
edd9515bSthomasraoux      return;
edd9515bSthomasraoux    SetVector<Operation *> dependentOps =
e7969240SThomas Raoux        getSliceContract(contract, hasVectorDest, hasVectorSrc);
edd9515bSthomasraoux    // If any instruction cannot use MMA matrix type drop the whole
e7969240SThomas Raoux    // chain. MMA matrix are stored in an opaque type so they cannot be used
edd9515bSthomasraoux    // by all operations.
*1ca772edSChristopher Bate    if (llvm::any_of(dependentOps, [useNvGpu](Operation *op) {
*1ca772edSChristopher Bate          return !supportsMMaMatrixType(op, useNvGpu);
*1ca772edSChristopher Bate        }))
edd9515bSthomasraoux      return;
edd9515bSthomasraoux    opToConvert.insert(dependentOps.begin(), dependentOps.end());
edd9515bSthomasraoux  });
e7969240SThomas Raoux  // Sort the operations so that we can convert them in topological order.
e7969240SThomas Raoux  return topologicalSort(opToConvert);
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxnamespace {
edd9515bSthomasraoux// Transform contract into (m, k)x(k, n)x(m, n) form so that it can be converted
edd9515bSthomasraoux// to MMA matmul.
edd9515bSthomasraouxstruct PrepareContractToGPUMMA
edd9515bSthomasraoux    : public OpRewritePattern<vector::ContractionOp> {
edd9515bSthomasraoux  using OpRewritePattern<vector::ContractionOp>::OpRewritePattern;
edd9515bSthomasraoux
edd9515bSthomasraoux  LogicalResult matchAndRewrite(vector::ContractionOp op,
edd9515bSthomasraoux                                PatternRewriter &rewriter) const override {
edd9515bSthomasraoux    Location loc = op.getLoc();
7c38fd60SJacques Pienaar    Value lhs = op.getLhs(), rhs = op.getRhs(), res = op.getAcc();
edd9515bSthomasraoux
edd9515bSthomasraoux    // Set up the parallel/reduction structure in right form.
edd9515bSthomasraoux    using MapList = ArrayRef<ArrayRef<AffineExpr>>;
edd9515bSthomasraoux    auto infer = [](MapList m) { return AffineMap::inferFromExprList(m); };
edd9515bSthomasraoux    AffineExpr m, n, k;
edd9515bSthomasraoux    bindDims(rewriter.getContext(), m, n, k);
edd9515bSthomasraoux    static constexpr std::array<int64_t, 2> perm = {1, 0};
7c38fd60SJacques Pienaar    auto iteratorTypes = op.getIteratorTypes().getValue();
edd9515bSthomasraoux    SmallVector<AffineMap, 4> maps = op.getIndexingMaps();
edd9515bSthomasraoux    if (!(isParallelIterator(iteratorTypes[0]) &&
edd9515bSthomasraoux          isParallelIterator(iteratorTypes[1]) &&
edd9515bSthomasraoux          isReductionIterator(iteratorTypes[2])))
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    //
edd9515bSthomasraoux    // Two outer parallel, one inner reduction (matmat flavor).
edd9515bSthomasraoux    //
edd9515bSthomasraoux    if (maps == infer({{m, k}, {k, n}, {m, n}})) {
edd9515bSthomasraoux      // This is the classical row-major matmul, nothing to do.
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    }
edd9515bSthomasraoux    if (maps == infer({{m, k}, {n, k}, {m, n}})) {
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {k, n}, {m, n}})) {
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {n, k}, {m, n}})) {
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{m, k}, {k, n}, {n, m}})) {
edd9515bSthomasraoux      std::swap(rhs, lhs);
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{m, k}, {n, k}, {n, m}})) {
edd9515bSthomasraoux      std::swap(rhs, lhs);
edd9515bSthomasraoux      rhs = rewriter.create<vector::TransposeOp>(loc, rhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {k, n}, {n, m}})) {
edd9515bSthomasraoux      std::swap(lhs, rhs);
edd9515bSthomasraoux      lhs = rewriter.create<vector::TransposeOp>(loc, lhs, perm);
edd9515bSthomasraoux    } else if (maps == infer({{k, m}, {n, k}, {n, m}})) {
edd9515bSthomasraoux      std::swap(lhs, rhs);
edd9515bSthomasraoux    } else {
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    }
edd9515bSthomasraoux    rewriter.replaceOpWithNewOp<vector::ContractionOp>(
edd9515bSthomasraoux        op, lhs, rhs, res,
edd9515bSthomasraoux        rewriter.getAffineMapArrayAttr(infer({{m, k}, {k, n}, {m, n}})),
7c38fd60SJacques Pienaar        op.getIteratorTypes());
edd9515bSthomasraoux    return success();
edd9515bSthomasraoux  }
edd9515bSthomasraoux};
edd9515bSthomasraoux
edd9515bSthomasraoux// Merge transpose op into the transfer read op. Transpose are not supported on
edd9515bSthomasraoux// MMA types but MMA load can transpose the matrix when loading.
edd9515bSthomasraouxstruct CombineTransferReadOpTranspose final
edd9515bSthomasraoux    : public OpRewritePattern<vector::TransposeOp> {
edd9515bSthomasraoux  using OpRewritePattern<vector::TransposeOp>::OpRewritePattern;
edd9515bSthomasraoux
edd9515bSthomasraoux  LogicalResult matchAndRewrite(vector::TransposeOp op,
edd9515bSthomasraoux                                PatternRewriter &rewriter) const override {
7c38fd60SJacques Pienaar    auto transferReadOp =
7c38fd60SJacques Pienaar        op.getVector().getDefiningOp<vector::TransferReadOp>();
edd9515bSthomasraoux    if (!transferReadOp)
edd9515bSthomasraoux      return failure();
c537a943SNicolas Vasilache
c537a943SNicolas Vasilache    // TODO: support 0-d corner case.
c537a943SNicolas Vasilache    if (transferReadOp.getTransferRank() == 0)
c537a943SNicolas Vasilache      return failure();
c537a943SNicolas Vasilache
7c38fd60SJacques Pienaar    if (transferReadOp.getMask() || transferReadOp.hasOutOfBoundsDim())
edd9515bSthomasraoux      return failure();
edd9515bSthomasraoux    SmallVector<int64_t, 2> perm;
edd9515bSthomasraoux    op.getTransp(perm);
edd9515bSthomasraoux    SmallVector<unsigned, 2> permU;
edd9515bSthomasraoux    for (int64_t o : perm)
edd9515bSthomasraoux      permU.push_back(unsigned(o));
edd9515bSthomasraoux    AffineMap permutationMap =
edd9515bSthomasraoux        AffineMap::getPermutationMap(permU, op.getContext());
7c38fd60SJacques Pienaar    AffineMap newMap =
7c38fd60SJacques Pienaar        permutationMap.compose(transferReadOp.getPermutationMap());
edd9515bSthomasraoux    rewriter.replaceOpWithNewOp<vector::TransferReadOp>(
7c38fd60SJacques Pienaar        op, op.getType(), transferReadOp.getSource(),
7c38fd60SJacques Pienaar        transferReadOp.getIndices(), AffineMapAttr::get(newMap),
7c38fd60SJacques Pienaar        transferReadOp.getPadding(), transferReadOp.getMask(),
7c38fd60SJacques Pienaar        transferReadOp.getInBoundsAttr());
edd9515bSthomasraoux    return success();
edd9515bSthomasraoux  }
edd9515bSthomasraoux};
edd9515bSthomasraoux
edd9515bSthomasraoux} // namespace
edd9515bSthomasraoux
edd9515bSthomasraoux// MMA types have different layout based on how they are used in matmul ops.
6413226dSthomasraoux// Figure the right layout to use by looking at op uses.
edd9515bSthomasraoux// TODO: Change the GPU dialect to abstract the layout at the this level and
edd9515bSthomasraoux// only care about it during lowering to NVVM.
6413226dSthomasraouxtemplate <typename OpTy>
6413226dSthomasraouxstatic const char *inferFragType(OpTy op) {
edd9515bSthomasraoux  for (Operation *users : op->getUsers()) {
edd9515bSthomasraoux    auto contract = dyn_cast<vector::ContractionOp>(users);
edd9515bSthomasraoux    if (!contract)
edd9515bSthomasraoux      continue;
7c38fd60SJacques Pienaar    if (contract.getLhs() == op.getResult())
edd9515bSthomasraoux      return "AOp";
7c38fd60SJacques Pienaar    if (contract.getRhs() == op.getResult())
edd9515bSthomasraoux      return "BOp";
edd9515bSthomasraoux  }
edd9515bSthomasraoux  return "COp";
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxstatic void convertTransferReadOp(vector::TransferReadOp op,
edd9515bSthomasraoux                                  llvm::DenseMap<Value, Value> &valueMapping) {
c537a943SNicolas Vasilache  assert(op.getTransferRank() > 0 && "unexpected 0-d transfer");
*1ca772edSChristopher Bate  assert(transferReadSupportsMMAMatrixType(op, /*useNvGpu=*/false));
edd9515bSthomasraoux  Optional<int64_t> stride =
edd9515bSthomasraoux      getMemrefConstantHorizontalStride(op.getShapedType());
7c38fd60SJacques Pienaar  AffineMap map = op.getPermutationMap();
e7969240SThomas Raoux  // Handle broadcast by setting the stride to 0.
e7969240SThomas Raoux  if (map.getResult(0).isa<AffineConstantExpr>()) {
e7969240SThomas Raoux    assert(map.getResult(0).cast<AffineConstantExpr>().getValue() == 0);
e7969240SThomas Raoux    stride = 0;
e7969240SThomas Raoux  }
edd9515bSthomasraoux  assert(stride);
edd9515bSthomasraoux  const char *fragType = inferFragType(op);
edd9515bSthomasraoux  gpu::MMAMatrixType type =
edd9515bSthomasraoux      gpu::MMAMatrixType::get(op.getVectorType().getShape(),
edd9515bSthomasraoux                              op.getVectorType().getElementType(), fragType);
edd9515bSthomasraoux  OpBuilder b(op);
edd9515bSthomasraoux  Value load = b.create<gpu::SubgroupMmaLoadMatrixOp>(
7c38fd60SJacques Pienaar      op.getLoc(), type, op.getSource(), op.getIndices(),
7c38fd60SJacques Pienaar      b.getIndexAttr(*stride));
edd9515bSthomasraoux  valueMapping[op.getResult()] = load;
edd9515bSthomasraoux}
edd9515bSthomasraoux
edd9515bSthomasraouxstatic void convertTransferWriteOp(vector::TransferWriteOp op,
edd9515bSthomasraoux                                   llvm::DenseMap<Value, Value> &valueMapping) {
edd9515bSthomasraoux  assert(transferWriteSupportsMMAMatrixType(op));
edd9515bSthomasraoux  Optional<int64_t> stride =
edd9515bSthomasraoux      getMemrefConstantHorizontalStride(op.getShapedType());
edd9515bSthomasraoux  assert(stride);
edd9515bSthomasraoux  OpBuilder b(op);
7c38fd60SJacques Pienaar  Value matrix = valueMapping.find(op.getVector())->second;
7c38fd60SJacques Pienaar  b.create<gpu::SubgroupMmaStoreMatrixOp>(op.getLoc(), matrix, op.getSource(),
7c38fd60SJacques Pienaar                                          op.getIndices(),
7c38fd60SJacques Pienaar                                          b.getIndexAttr(*stride));
edd9515bSthomasraoux  op.erase();
edd9515bSthomasraoux}
edd9515bSthomasraoux
*1ca772edSChristopher Bate/// Returns the vector type which represents a matrix fragment.
*1ca772edSChristopher Batestatic VectorType
*1ca772edSChristopher BategetMmaSyncVectorOperandType(const nvgpu::FragmentElementInfo &regInfo) {
*1ca772edSChristopher Bate  SmallVector<int64_t> shape{regInfo.numRegistersPerFragment,
*1ca772edSChristopher Bate                             regInfo.elementsPerRegister};
*1ca772edSChristopher Bate  Type elType = regInfo.registerLLVMType;
*1ca772edSChristopher Bate  if (auto vecType = elType.dyn_cast<VectorType>())
*1ca772edSChristopher Bate    elType = vecType.getElementType();
*1ca772edSChristopher Bate  return VectorType::get(shape, elType);
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate/// Convert a 2D splat ConstantOp to a SubgroupMmaConstantMatrix op.
*1ca772edSChristopher Batestatic LogicalResult
*1ca772edSChristopher BateconvertConstantOpMmaSync(arith::ConstantOp op,
*1ca772edSChristopher Bate                         llvm::DenseMap<Value, Value> &valueMapping) {
*1ca772edSChristopher Bate  OpBuilder b(op);
*1ca772edSChristopher Bate  FailureOr<nvgpu::WarpMatrixInfo> warpMatrixInfo =
*1ca772edSChristopher Bate      nvgpu::getWarpMatrixInfo(op);
*1ca772edSChristopher Bate  if (failed(warpMatrixInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  FailureOr<nvgpu::FragmentElementInfo> regInfo =
*1ca772edSChristopher Bate      nvgpu::getMmaSyncRegisterType(*warpMatrixInfo);
*1ca772edSChristopher Bate  if (failed(regInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  VectorType vectorType = getMmaSyncVectorOperandType(*regInfo);
*1ca772edSChristopher Bate  auto dense = op.getValue().dyn_cast<SplatElementsAttr>();
*1ca772edSChristopher Bate  if (!dense)
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate  Value result = b.create<arith::ConstantOp>(
*1ca772edSChristopher Bate      op.getLoc(), vectorType,
*1ca772edSChristopher Bate      DenseElementsAttr::get(vectorType, dense.getSplatValue<Attribute>()));
*1ca772edSChristopher Bate  valueMapping[op.getResult()] = result;
*1ca772edSChristopher Bate  return success();
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
*1ca772edSChristopher Batestatic LogicalResult
*1ca772edSChristopher BatecreatLdMatrixCompatibleLoads(vector::TransferReadOp op, OpBuilder &builder,
*1ca772edSChristopher Bate                             llvm::DenseMap<Value, Value> &valueMapping) {
*1ca772edSChristopher Bate  Location loc = op->getLoc();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  FailureOr<nvgpu::WarpMatrixInfo> warpMatrixInfo =
*1ca772edSChristopher Bate      nvgpu::getWarpMatrixInfo(op);
*1ca772edSChristopher Bate  if (failed(warpMatrixInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  FailureOr<nvgpu::FragmentElementInfo> regInfo =
*1ca772edSChristopher Bate      nvgpu::getMmaSyncRegisterType(*warpMatrixInfo);
*1ca772edSChristopher Bate  if (failed(regInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  FailureOr<nvgpu::LdMatrixParams> params = nvgpu::getLdMatrixParams(
*1ca772edSChristopher Bate      *warpMatrixInfo,
*1ca772edSChristopher Bate      /*transpose=*/!op.getPermutationMap().isMinorIdentity());
*1ca772edSChristopher Bate  if (failed(params)) {
*1ca772edSChristopher Bate    return op->emitError()
*1ca772edSChristopher Bate           << "failed to convert vector.transfer_read to ldmatrix; this op "
*1ca772edSChristopher Bate              "likely "
*1ca772edSChristopher Bate              "should not be converted to a nvgpu.ldmatrix call.";
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  // Adjust the load offset.
*1ca772edSChristopher Bate  auto laneId = builder.create<gpu::LaneIdOp>(loc);
*1ca772edSChristopher Bate  FailureOr<AffineMap> offsets =
*1ca772edSChristopher Bate      nvgpu::getLaneIdToLdMatrixMatrixCoord(loc, builder, *params);
*1ca772edSChristopher Bate  if (failed(offsets))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  VectorType vectorType = getMmaSyncVectorOperandType(*regInfo);
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  SmallVector<Value, 4> indices;
*1ca772edSChristopher Bate  getXferIndices<vector::TransferReadOp>(builder, op, *offsets, {laneId},
*1ca772edSChristopher Bate                                         indices);
*1ca772edSChristopher Bate  nvgpu::LdMatrixOp newOp = builder.create<nvgpu::LdMatrixOp>(
*1ca772edSChristopher Bate      loc, vectorType, op.getSource(), indices,
*1ca772edSChristopher Bate      !op.getPermutationMap().isMinorIdentity(), params->numTiles);
*1ca772edSChristopher Bate  valueMapping[op] = newOp->getResult(0);
*1ca772edSChristopher Bate  return success();
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
*1ca772edSChristopher Batestatic LogicalResult
*1ca772edSChristopher BatecreateNonLdMatrixLoads(vector::TransferReadOp op, OpBuilder &builder,
*1ca772edSChristopher Bate                       llvm::DenseMap<Value, Value> &valueMapping) {
*1ca772edSChristopher Bate  Location loc = op.getLoc();
*1ca772edSChristopher Bate  FailureOr<nvgpu::WarpMatrixInfo> warpMatrixInfo =
*1ca772edSChristopher Bate      nvgpu::getWarpMatrixInfo(op);
*1ca772edSChristopher Bate  if (failed(warpMatrixInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate  FailureOr<nvgpu::FragmentElementInfo> regInfo =
*1ca772edSChristopher Bate      nvgpu::getMmaSyncRegisterType(*warpMatrixInfo);
*1ca772edSChristopher Bate  if (failed(regInfo)) {
*1ca772edSChristopher Bate    op->emitError() << "Failed to deduce register fragment type during "
*1ca772edSChristopher Bate                       "conversion to distributed non-ldmatrix compatible load";
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  NVVM::MMALayout targetLayout =
*1ca772edSChristopher Bate      warpMatrixInfo->operandRole == nvgpu::MatMulOperandRole::B
*1ca772edSChristopher Bate          ? NVVM::MMALayout::col
*1ca772edSChristopher Bate          : NVVM::MMALayout::row;
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  Value laneId = builder.create<gpu::LaneIdOp>(loc);
*1ca772edSChristopher Bate  SmallVector<Value, 4> elements;
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  // This is the individual element type.
*1ca772edSChristopher Bate  Type loadedElType = regInfo->registerLLVMType;
*1ca772edSChristopher Bate  VectorType vectorType = getMmaSyncVectorOperandType(*regInfo);
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  Value fill = builder.create<arith::ConstantOp>(
*1ca772edSChristopher Bate      op.getLoc(), vectorType.getElementType(),
*1ca772edSChristopher Bate      builder.getZeroAttr(vectorType.getElementType()));
*1ca772edSChristopher Bate  Value result = builder.create<vector::SplatOp>(op.getLoc(), fill, vectorType);
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  bool isTransposeLoad = !op.getPermutationMap().isMinorIdentity();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  // Vectorized loads.
*1ca772edSChristopher Bate  if (!isTransposeLoad && targetLayout == NVVM::MMALayout::row) {
*1ca772edSChristopher Bate    if (!loadedElType.isa<VectorType>()) {
*1ca772edSChristopher Bate      loadedElType = VectorType::get({1}, loadedElType);
*1ca772edSChristopher Bate    }
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate    for (int i = 0; i < vectorType.getShape()[0]; i++) {
*1ca772edSChristopher Bate      FailureOr<AffineMap> coords = nvgpu::getLaneIdAndValueIdToOperandCoord(
*1ca772edSChristopher Bate          op.getLoc(), builder, *warpMatrixInfo);
*1ca772edSChristopher Bate      if (failed(coords))
*1ca772edSChristopher Bate        return failure();
*1ca772edSChristopher Bate      Value logicalValueId = builder.create<arith::ConstantOp>(
*1ca772edSChristopher Bate          loc, builder.getIndexType(),
*1ca772edSChristopher Bate          builder.getIndexAttr(i * regInfo->elementsPerRegister));
*1ca772edSChristopher Bate      SmallVector<Value, 4> newIndices;
*1ca772edSChristopher Bate      getXferIndices<vector::TransferReadOp>(
*1ca772edSChristopher Bate          builder, op, *coords, {laneId, logicalValueId}, newIndices);
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate      Value el = builder.create<vector::LoadOp>(loc, loadedElType,
*1ca772edSChristopher Bate                                                op.getSource(), newIndices);
*1ca772edSChristopher Bate      result = builder.create<vector::InsertOp>(loc, el, result,
*1ca772edSChristopher Bate                                                builder.getI64ArrayAttr(i));
*1ca772edSChristopher Bate    }
*1ca772edSChristopher Bate  } else if (isTransposeLoad && targetLayout == NVVM::MMALayout::col) {
*1ca772edSChristopher Bate    if (auto vecType = loadedElType.dyn_cast<VectorType>()) {
*1ca772edSChristopher Bate      loadedElType = vecType.getElementType();
*1ca772edSChristopher Bate    }
*1ca772edSChristopher Bate    // Load each element individually.
*1ca772edSChristopher Bate    for (int i = 0; i < vectorType.getShape()[0]; i++) {
*1ca772edSChristopher Bate      for (unsigned innerIdx = 0; innerIdx < vectorType.getShape()[1];
*1ca772edSChristopher Bate           innerIdx++) {
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate        Value logicalValueId = builder.create<arith::ConstantOp>(
*1ca772edSChristopher Bate            loc, builder.getIndexType(),
*1ca772edSChristopher Bate            builder.getIndexAttr(i * regInfo->elementsPerRegister + innerIdx));
*1ca772edSChristopher Bate        FailureOr<AffineMap> coords = nvgpu::getLaneIdAndValueIdToOperandCoord(
*1ca772edSChristopher Bate            op.getLoc(), builder, *warpMatrixInfo);
*1ca772edSChristopher Bate        if (failed(coords))
*1ca772edSChristopher Bate          return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate        SmallVector<Value, 4> newIndices;
*1ca772edSChristopher Bate        getXferIndices<vector::TransferReadOp>(
*1ca772edSChristopher Bate            builder, op, *coords, {laneId, logicalValueId}, newIndices);
*1ca772edSChristopher Bate        Value el = builder.create<memref::LoadOp>(op.getLoc(), loadedElType,
*1ca772edSChristopher Bate                                                  op.getSource(), newIndices);
*1ca772edSChristopher Bate        result = builder.create<vector::InsertOp>(
*1ca772edSChristopher Bate            op.getLoc(), el, result, builder.getI64ArrayAttr({i, innerIdx}));
*1ca772edSChristopher Bate      }
*1ca772edSChristopher Bate    }
*1ca772edSChristopher Bate  } else {
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  valueMapping[op.getResult()] = result;
*1ca772edSChristopher Bate  return success();
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate/// Converts a `vector.transfer_read` operation directly to either a
*1ca772edSChristopher Bate/// `vector.load` or a `nvgpu.ldmatrix` operation. This function should only be
*1ca772edSChristopher Bate/// used when converting to `nvgpu.mma.sync` operations.
*1ca772edSChristopher Batestatic LogicalResult
*1ca772edSChristopher BateconvertTransferReadToLoads(vector::TransferReadOp op,
*1ca772edSChristopher Bate                           llvm::DenseMap<Value, Value> &valueMapping) {
*1ca772edSChristopher Bate  OpBuilder b(op);
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  FailureOr<nvgpu::WarpMatrixInfo> warpMatrixInfo =
*1ca772edSChristopher Bate      nvgpu::getWarpMatrixInfo(op);
*1ca772edSChristopher Bate  if (failed(warpMatrixInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  bool isLdMatrixCompatible =
*1ca772edSChristopher Bate      op.getSource().getType().cast<MemRefType>().getMemorySpaceAsInt() == 3 &&
*1ca772edSChristopher Bate      nvgpu::inferTileWidthInBits(*warpMatrixInfo) == 128;
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  VectorType vecTy = op.getVectorType();
*1ca772edSChristopher Bate  int64_t bitWidth = vecTy.getElementType().getIntOrFloatBitWidth();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  // When we are transposing the B operand, ldmatrix will only work if we have
*1ca772edSChristopher Bate  // at least 8 rows to read and  the width to read for the transpose is 128
*1ca772edSChristopher Bate  // bits.
*1ca772edSChristopher Bate  if (!op.getPermutationMap().isMinorIdentity() &&
*1ca772edSChristopher Bate      (vecTy.getDimSize(1) < 8 || vecTy.getDimSize(0) * bitWidth < 128))
*1ca772edSChristopher Bate    isLdMatrixCompatible = false;
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  if (!isLdMatrixCompatible)
*1ca772edSChristopher Bate    return createNonLdMatrixLoads(op, b, valueMapping);
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  return creatLdMatrixCompatibleLoads(op, b, valueMapping);
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
*1ca772edSChristopher Batestatic LogicalResult
*1ca772edSChristopher BateconvertTransferWriteToStores(vector::TransferWriteOp op,
*1ca772edSChristopher Bate                             llvm::DenseMap<Value, Value> &valueMapping) {
*1ca772edSChristopher Bate  OpBuilder b(op);
*1ca772edSChristopher Bate  Location loc = op->getLoc();
*1ca772edSChristopher Bate  Value matrix = valueMapping.find(op.getVector())->second;
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  FailureOr<nvgpu::WarpMatrixInfo> warpMatrixInfo =
*1ca772edSChristopher Bate      nvgpu::getWarpMatrixInfo(op);
*1ca772edSChristopher Bate  if (failed(warpMatrixInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate  FailureOr<nvgpu::FragmentElementInfo> regInfo =
*1ca772edSChristopher Bate      nvgpu::getMmaSyncRegisterType(*warpMatrixInfo);
*1ca772edSChristopher Bate  if (failed(regInfo))
*1ca772edSChristopher Bate    return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  VectorType vectorType = getMmaSyncVectorOperandType(*regInfo);
*1ca772edSChristopher Bate  Value laneId = b.create<gpu::LaneIdOp>(loc);
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  for (unsigned i = 0; i < vectorType.getShape()[0]; i++) {
*1ca772edSChristopher Bate    Value logicalValueId = b.create<arith::ConstantOp>(
*1ca772edSChristopher Bate        loc, b.getIndexType(),
*1ca772edSChristopher Bate        b.getIndexAttr(i * regInfo->elementsPerRegister));
*1ca772edSChristopher Bate    FailureOr<AffineMap> coords = nvgpu::getLaneIdAndValueIdToOperandCoord(
*1ca772edSChristopher Bate        op.getLoc(), b, *warpMatrixInfo);
*1ca772edSChristopher Bate    if (failed(coords))
*1ca772edSChristopher Bate      return failure();
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate    Value el = b.create<vector::ExtractOp>(loc, matrix, ArrayRef<int64_t>{i});
*1ca772edSChristopher Bate    SmallVector<Value, 4> newIndices;
*1ca772edSChristopher Bate    getXferIndices<vector::TransferWriteOp>(
*1ca772edSChristopher Bate        b, op, *coords, {laneId, logicalValueId}, newIndices);
*1ca772edSChristopher Bate    b.create<vector::StoreOp>(loc, el, op.getSource(), newIndices);
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate  op->erase();
*1ca772edSChristopher Bate  return success();
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
edd9515bSthomasraouxstatic void convertContractOp(vector::ContractionOp op,
edd9515bSthomasraoux                              llvm::DenseMap<Value, Value> &valueMapping) {
edd9515bSthomasraoux  OpBuilder b(op);
7c38fd60SJacques Pienaar  Value opA = valueMapping.find(op.getLhs())->second;
7c38fd60SJacques Pienaar  Value opB = valueMapping.find(op.getRhs())->second;
7c38fd60SJacques Pienaar  Value opC = valueMapping.find(op.getAcc())->second;
edd9515bSthomasraoux  Value matmul = b.create<gpu::SubgroupMmaComputeOp>(op.getLoc(), opC.getType(),
edd9515bSthomasraoux                                                     opA, opB, opC);
edd9515bSthomasraoux  valueMapping[op.getResult()] = matmul;
edd9515bSthomasraoux}
edd9515bSthomasraoux
*1ca772edSChristopher Batestatic LogicalResult
*1ca772edSChristopher BateconvertContractOpToMmaSync(vector::ContractionOp op,
*1ca772edSChristopher Bate                           llvm::DenseMap<Value, Value> &valueMapping) {
*1ca772edSChristopher Bate  OpBuilder b(op);
*1ca772edSChristopher Bate  Value opA = valueMapping.find(op.getLhs())->second;
*1ca772edSChristopher Bate  Value opB = valueMapping.find(op.getRhs())->second;
*1ca772edSChristopher Bate  Value opC = valueMapping.find(op.getAcc())->second;
*1ca772edSChristopher Bate  int64_t m = op.getLhs().getType().cast<VectorType>().getShape()[0];
*1ca772edSChristopher Bate  int64_t n = op.getRhs().getType().cast<VectorType>().getShape()[0];
*1ca772edSChristopher Bate  int64_t k = op.getLhs().getType().cast<VectorType>().getShape()[1];
*1ca772edSChristopher Bate  Value matmul = b.create<nvgpu::MmaSyncOp>(
*1ca772edSChristopher Bate      op.getLoc(), opC.getType(), opA, opB, opC, b.getI64ArrayAttr({m, n, k}));
*1ca772edSChristopher Bate  valueMapping[op.getResult()] = matmul;
*1ca772edSChristopher Bate  return success();
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
6413226dSthomasraoux/// Convert a 2D splat ConstantOp to a SubgroupMmaConstantMatrix op.
a54f4eaeSMogballstatic void convertConstantOp(arith::ConstantOp op,
6413226dSthomasraoux                              llvm::DenseMap<Value, Value> &valueMapping) {
6413226dSthomasraoux  assert(constantSupportsMMAMatrixType(op));
6413226dSthomasraoux  OpBuilder b(op);
937e40a8SRiver Riddle  Attribute splat =
937e40a8SRiver Riddle      op.getValue().cast<SplatElementsAttr>().getSplatValue<Attribute>();
6413226dSthomasraoux  auto scalarConstant =
a54f4eaeSMogball      b.create<arith::ConstantOp>(op.getLoc(), splat.getType(), splat);
6413226dSthomasraoux  const char *fragType = inferFragType(op);
6413226dSthomasraoux  auto vecType = op.getType().cast<VectorType>();
6413226dSthomasraoux  gpu::MMAMatrixType type = gpu::MMAMatrixType::get(
6413226dSthomasraoux      vecType.getShape(), vecType.getElementType(), llvm::StringRef(fragType));
6413226dSthomasraoux  auto matrix = b.create<gpu::SubgroupMmaConstantMatrixOp>(op.getLoc(), type,
6413226dSthomasraoux                                                           scalarConstant);
6413226dSthomasraoux  valueMapping[op.getResult()] = matrix;
6413226dSthomasraoux}
6413226dSthomasraoux
43928419Sthomasraoux/// Convert a vector.broadcast from scalar to a SubgroupMmaConstantMatrix op.
43928419Sthomasraouxstatic void convertBroadcastOp(vector::BroadcastOp op,
43928419Sthomasraoux                               llvm::DenseMap<Value, Value> &valueMapping) {
43928419Sthomasraoux  assert(broadcastSupportsMMAMatrixType(op));
43928419Sthomasraoux  OpBuilder b(op);
43928419Sthomasraoux  const char *fragType = inferFragType(op);
43928419Sthomasraoux  auto vecType = op.getVectorType();
43928419Sthomasraoux  gpu::MMAMatrixType type = gpu::MMAMatrixType::get(
43928419Sthomasraoux      vecType.getShape(), vecType.getElementType(), llvm::StringRef(fragType));
43928419Sthomasraoux  auto matrix = b.create<gpu::SubgroupMmaConstantMatrixOp>(op.getLoc(), type,
7c38fd60SJacques Pienaar                                                           op.getSource());
43928419Sthomasraoux  valueMapping[op.getResult()] = matrix;
43928419Sthomasraoux}
43928419Sthomasraoux
1a865592Sthomasraoux// Replace ForOp with a new ForOp with extra operands. The YieldOp is not
1a865592Sthomasraoux// updated and needs to be updated separatly for the loop to be correct.
1a865592Sthomasraouxstatic scf::ForOp replaceForOpWithNewSignature(OpBuilder &b, scf::ForOp loop,
1a865592Sthomasraoux                                               ValueRange newIterOperands) {
1a865592Sthomasraoux  // Create a new loop before the existing one, with the extra operands.
1a865592Sthomasraoux  OpBuilder::InsertionGuard g(b);
1a865592Sthomasraoux  b.setInsertionPoint(loop);
1a865592Sthomasraoux  auto operands = llvm::to_vector<4>(loop.getIterOperands());
1a865592Sthomasraoux  operands.append(newIterOperands.begin(), newIterOperands.end());
1a865592Sthomasraoux  scf::ForOp newLoop =
c0342a2dSJacques Pienaar      b.create<scf::ForOp>(loop.getLoc(), loop.getLowerBound(),
c0342a2dSJacques Pienaar                           loop.getUpperBound(), loop.getStep(), operands);
1a865592Sthomasraoux  newLoop.getBody()->erase();
1a865592Sthomasraoux  newLoop.getLoopBody().getBlocks().splice(
1a865592Sthomasraoux      newLoop.getLoopBody().getBlocks().begin(),
1a865592Sthomasraoux      loop.getLoopBody().getBlocks());
e084679fSRiver Riddle  for (Value operand : newIterOperands)
e084679fSRiver Riddle    newLoop.getBody()->addArgument(operand.getType(), operand.getLoc());
1a865592Sthomasraoux
1a865592Sthomasraoux  for (auto it : llvm::zip(loop.getResults(), newLoop.getResults().take_front(
1a865592Sthomasraoux                                                  loop.getNumResults())))
1a865592Sthomasraoux    std::get<0>(it).replaceAllUsesWith(std::get<1>(it));
1a865592Sthomasraoux  loop.erase();
1a865592Sthomasraoux  return newLoop;
1a865592Sthomasraoux}
1a865592Sthomasraoux
1a865592Sthomasraouxstatic void convertForOp(scf::ForOp op,
1a865592Sthomasraoux                         llvm::DenseMap<Value, Value> &valueMapping) {
1a865592Sthomasraoux  SmallVector<Value> newOperands;
1a865592Sthomasraoux  SmallVector<std::pair<size_t, size_t>> argMapping;
e4853be2SMehdi Amini  for (const auto &operand : llvm::enumerate(op.getIterOperands())) {
1a865592Sthomasraoux    auto it = valueMapping.find(operand.value());
1a865592Sthomasraoux    if (it == valueMapping.end())
1a865592Sthomasraoux      continue;
1a865592Sthomasraoux    argMapping.push_back(std::make_pair(
1a865592Sthomasraoux        operand.index(), op.getNumIterOperands() + newOperands.size()));
1a865592Sthomasraoux    newOperands.push_back(it->second);
1a865592Sthomasraoux  }
1a865592Sthomasraoux  OpBuilder b(op);
1a865592Sthomasraoux  scf::ForOp newForOp = replaceForOpWithNewSignature(b, op, newOperands);
1a865592Sthomasraoux  Block &loopBody = *newForOp.getBody();
1a865592Sthomasraoux  for (auto mapping : argMapping) {
1a865592Sthomasraoux    valueMapping[newForOp.getResult(mapping.first)] =
1a865592Sthomasraoux        newForOp.getResult(mapping.second);
1a865592Sthomasraoux    valueMapping[loopBody.getArgument(mapping.first +
1a865592Sthomasraoux                                      newForOp.getNumInductionVars())] =
1a865592Sthomasraoux        loopBody.getArgument(mapping.second + newForOp.getNumInductionVars());
1a865592Sthomasraoux  }
1a865592Sthomasraoux}
1a865592Sthomasraoux
1a865592Sthomasraouxstatic void convertYieldOp(scf::YieldOp op,
1a865592Sthomasraoux                           llvm::DenseMap<Value, Value> &valueMapping) {
1a865592Sthomasraoux  OpBuilder b(op);
1a865592Sthomasraoux  auto loop = cast<scf::ForOp>(op->getParentOp());
1a865592Sthomasraoux  auto yieldOperands = llvm::to_vector<4>(op.getOperands());
e4853be2SMehdi Amini  for (const auto &operand : llvm::enumerate(op.getOperands())) {
1a865592Sthomasraoux    auto it = valueMapping.find(operand.value());
1a865592Sthomasraoux    if (it == valueMapping.end())
1a865592Sthomasraoux      continue;
1a865592Sthomasraoux    // Replace the yield of old value with the for op argument to make it easier
1a865592Sthomasraoux    // to remove the dead code.
1a865592Sthomasraoux    yieldOperands[operand.index()] = loop.getIterOperands()[operand.index()];
1a865592Sthomasraoux    yieldOperands.push_back(it->second);
1a865592Sthomasraoux  }
1a865592Sthomasraoux  b.create<scf::YieldOp>(op.getLoc(), yieldOperands);
1a865592Sthomasraoux  op.erase();
1a865592Sthomasraoux}
1a865592Sthomasraoux
7fbb0678Sthomasraoux/// Convert an elementwise op to the equivalent elementwise op on MMA matrix.
7fbb0678Sthomasraouxstatic void convertElementwiseOp(Operation *op, gpu::MMAElementwiseOp opType,
7fbb0678Sthomasraoux                                 llvm::DenseMap<Value, Value> &valueMapping) {
7fbb0678Sthomasraoux  OpBuilder b(op);
7fbb0678Sthomasraoux  SmallVector<Value> matrixOperands;
7fbb0678Sthomasraoux  for (Value operand : op->getOperands())
7fbb0678Sthomasraoux    matrixOperands.push_back(valueMapping.find(operand)->second);
7fbb0678Sthomasraoux  Value newOp = b.create<gpu::SubgroupMmaElementwiseOp>(
7fbb0678Sthomasraoux      op->getLoc(), matrixOperands[0].getType(), matrixOperands, opType);
7fbb0678Sthomasraoux  valueMapping[op->getResult(0)] = newOp;
7fbb0678Sthomasraoux}
7fbb0678Sthomasraoux
*1ca772edSChristopher Batevoid mlir::populatePrepareVectorToMMAPatterns(RewritePatternSet &patterns,
*1ca772edSChristopher Bate                                              bool useNvGpu) {
*1ca772edSChristopher Bate  if (!useNvGpu) {
edd9515bSthomasraoux    patterns.add<PrepareContractToGPUMMA, CombineTransferReadOpTranspose>(
edd9515bSthomasraoux        patterns.getContext());
*1ca772edSChristopher Bate    return;
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate  patterns
*1ca772edSChristopher Bate      .add<nvgpu::PrepareContractToGPUMMASync, CombineTransferReadOpTranspose>(
*1ca772edSChristopher Bate          patterns.getContext());
edd9515bSthomasraoux}
edd9515bSthomasraoux
47f175b0SRiver Riddlevoid mlir::convertVectorToMMAOps(Operation *rootOp) {
*1ca772edSChristopher Bate  SetVector<Operation *> ops = getOpToConvert(rootOp, /*useNvGpu=*/false);
edd9515bSthomasraoux  llvm::DenseMap<Value, Value> valueMapping;
edd9515bSthomasraoux  for (Operation *op : ops) {
edd9515bSthomasraoux    if (auto transferRead = dyn_cast<vector::TransferReadOp>(op)) {
edd9515bSthomasraoux      convertTransferReadOp(transferRead, valueMapping);
edd9515bSthomasraoux    } else if (auto transferWrite = dyn_cast<vector::TransferWriteOp>(op)) {
edd9515bSthomasraoux      convertTransferWriteOp(transferWrite, valueMapping);
edd9515bSthomasraoux    } else if (auto contractOp = dyn_cast<vector::ContractionOp>(op)) {
edd9515bSthomasraoux      convertContractOp(contractOp, valueMapping);
a54f4eaeSMogball    } else if (auto constantOp = dyn_cast<arith::ConstantOp>(op)) {
6413226dSthomasraoux      convertConstantOp(constantOp, valueMapping);
43928419Sthomasraoux    } else if (auto broadcastOp = dyn_cast<vector::BroadcastOp>(op)) {
43928419Sthomasraoux      convertBroadcastOp(broadcastOp, valueMapping);
1a865592Sthomasraoux    } else if (auto forOp = dyn_cast<scf::ForOp>(op)) {
1a865592Sthomasraoux      convertForOp(forOp, valueMapping);
1a865592Sthomasraoux    } else if (auto yiledOp = dyn_cast<scf::YieldOp>(op)) {
1a865592Sthomasraoux      convertYieldOp(yiledOp, valueMapping);
7fbb0678Sthomasraoux    } else if (auto elementwiseType = convertElementwiseOpToMMA(op)) {
7fbb0678Sthomasraoux      convertElementwiseOp(op, *elementwiseType, valueMapping);
edd9515bSthomasraoux    }
edd9515bSthomasraoux  }
edd9515bSthomasraoux}
edd9515bSthomasraoux
*1ca772edSChristopher BateLogicalResult mlir::convertVectorToNVVMCompatibleMMASync(Operation *rootOp) {
*1ca772edSChristopher Bate  SetVector<Operation *> ops = getOpToConvert(rootOp, /*useNvGpu=*/true);
*1ca772edSChristopher Bate  llvm::DenseMap<Value, Value> valueMapping;
*1ca772edSChristopher Bate  for (Operation *op : ops) {
*1ca772edSChristopher Bate    if (llvm::TypeSwitch<Operation *, LogicalResult>(op)
*1ca772edSChristopher Bate            .Case([&](vector::TransferReadOp transferReadOp) {
*1ca772edSChristopher Bate              return convertTransferReadToLoads(transferReadOp, valueMapping);
*1ca772edSChristopher Bate            })
*1ca772edSChristopher Bate            .Case([&](vector::TransferWriteOp transferWriteOp) {
*1ca772edSChristopher Bate              return convertTransferWriteToStores(transferWriteOp,
*1ca772edSChristopher Bate                                                  valueMapping);
*1ca772edSChristopher Bate            })
*1ca772edSChristopher Bate            .Case([&](vector::ContractionOp contractionOp) {
*1ca772edSChristopher Bate              return convertContractOpToMmaSync(contractionOp, valueMapping);
*1ca772edSChristopher Bate            })
*1ca772edSChristopher Bate            .Case([&](scf::ForOp forOp) {
*1ca772edSChristopher Bate              convertForOp(forOp, valueMapping);
*1ca772edSChristopher Bate              return success();
*1ca772edSChristopher Bate            })
*1ca772edSChristopher Bate            .Case([&](scf::YieldOp yieldOp) {
*1ca772edSChristopher Bate              convertYieldOp(yieldOp, valueMapping);
*1ca772edSChristopher Bate              return success();
*1ca772edSChristopher Bate            })
*1ca772edSChristopher Bate            .Case([&](arith::ConstantOp constOp) {
*1ca772edSChristopher Bate              return convertConstantOpMmaSync(constOp, valueMapping);
*1ca772edSChristopher Bate            })
*1ca772edSChristopher Bate            .Default([&](Operation *op) {
*1ca772edSChristopher Bate              op->emitError() << "unhandled vector to mma type: " << *op;
*1ca772edSChristopher Bate              return failure();
*1ca772edSChristopher Bate            })
*1ca772edSChristopher Bate            .failed()) {
*1ca772edSChristopher Bate      op->emitError() << "Failed to convert op " << *op;
*1ca772edSChristopher Bate      return failure();
*1ca772edSChristopher Bate    }
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate  return success();
*1ca772edSChristopher Bate}
*1ca772edSChristopher Bate
edd9515bSthomasraouxnamespace {
edd9515bSthomasraoux
edd9515bSthomasraouxstruct ConvertVectorToGPUPass
edd9515bSthomasraoux    : public ConvertVectorToGPUBase<ConvertVectorToGPUPass> {
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate  explicit ConvertVectorToGPUPass(bool useNvGpu_) {
*1ca772edSChristopher Bate    useNvGpu.setValue(useNvGpu_);
*1ca772edSChristopher Bate  }
*1ca772edSChristopher Bate
41574554SRiver Riddle  void runOnOperation() override {
47f175b0SRiver Riddle    RewritePatternSet patterns(&getContext());
*1ca772edSChristopher Bate    populatePrepareVectorToMMAPatterns(patterns, useNvGpu.getValue());
*1ca772edSChristopher Bate    if (failed(
*1ca772edSChristopher Bate            applyPatternsAndFoldGreedily(getOperation(), std::move(patterns))))
*1ca772edSChristopher Bate      return signalPassFailure();
edd9515bSthomasraoux
*1ca772edSChristopher Bate    if (useNvGpu.getValue()) {
*1ca772edSChristopher Bate      if (failed(convertVectorToNVVMCompatibleMMASync(getOperation())))
*1ca772edSChristopher Bate        return signalPassFailure();
*1ca772edSChristopher Bate    }
*1ca772edSChristopher Bate
*1ca772edSChristopher Bate    (void)convertVectorToMMAOps(getOperation());
edd9515bSthomasraoux  }
edd9515bSthomasraoux};
edd9515bSthomasraoux
edd9515bSthomasraoux} // namespace
edd9515bSthomasraoux
*1ca772edSChristopher Batestd::unique_ptr<Pass> mlir::createConvertVectorToGPUPass(bool useNvGpu) {
*1ca772edSChristopher Bate  return std::make_unique<ConvertVectorToGPUPass>(useNvGpu);
edd9515bSthomasraoux}