|
MLIR 24.0.0git
|
#include "mlir/Dialect/XeGPU/Transforms/XeGPULayoutImpl.h"#include "mlir/Dialect/Func/IR/FuncOps.h"#include "mlir/Dialect/GPU/IR/GPUDialect.h"#include "mlir/Dialect/LLVMIR/XeVMDialect.h"#include "mlir/Dialect/SCF/Transforms/Patterns.h"#include "mlir/Dialect/Utils/IndexingUtils.h"#include "mlir/Dialect/Vector/IR/VectorOps.h"#include "mlir/Dialect/XeGPU/IR/XeGPU.h"#include "mlir/IR/Builders.h"#include "mlir/IR/Operation.h"#include "mlir/IR/ValueRange.h"#include "mlir/Interfaces/ControlFlowInterfaces.h"#include "mlir/Interfaces/LoopLikeInterface.h"#include "mlir/Interfaces/SideEffectInterfaces.h"#include "mlir/Transforms/DialectConversion.h"#include "llvm/ADT/PostOrderIterator.h"#include "llvm/Support/FormatVariadic.h"#include <cstdint>#include <numeric>Go to the source code of this file.
Typedefs | |
| using | LayoutRepresentation = SmallVector<int64_t> |
Functions | |
| static void | setTensorDescLayout (Value val, xegpu::DistributeLayoutAttr layout) |
| static void | walkRegionBackward (Region ®ion, llvm::function_ref< void(Operation *)> visit) |
| static xegpu::DistributeLayoutAttr | getLayoutFromUsePoints (Value result) |
| static void | propagateResultsToRegularOperands (Operation *op) |
| static void | propagateRegionResultsToYieldOperands (mlir::RegionBranchTerminatorOpInterface yieldOp) |
| template void | xegpu::removeLayoutAttr< mlir::OpResult > (const mlir::OpResult &result) |
| template void | xegpu::removeLayoutAttr< mlir::OpOperand > (const mlir::OpOperand &operand) |
| static xegpu::LayoutAttr | buildInstDataLayoutWithLane (mlir::MLIRContext *context, ArrayRef< int64_t > instData, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData, DenseI32ArrayAttr orderAttr=nullptr) |
| static bool | isValidLaneLayout (ArrayRef< int64_t > dataShape, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData) |
| static xegpu::LayoutAttr | buildLaneLayout (mlir::MLIRContext *context, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData, DenseI32ArrayAttr orderAttr=nullptr) |
| static xegpu::LayoutAttr | buildLayout (mlir::MLIRContext *context, ArrayRef< int64_t > sgLayout, ArrayRef< int64_t > sgData, ArrayRef< int64_t > instData, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData, DenseI32ArrayAttr orderAttr=nullptr) |
| static xegpu::LayoutAttr | buildSgLayout (mlir::MLIRContext *context, ArrayRef< int64_t > wgTileShape, ArrayRef< int64_t > sgLayout, int dimK=-1, DenseI32ArrayAttr orderAttr=nullptr) |
| static SmallVector< LayoutRepresentation > | enumerateFactorizations (int64_t total, int64_t rank) |
| Enumerates all ways to split total into rank factors whose product equals total. | |
| static SmallVector< LayoutRepresentation > | getSgLayoutCandidates (ArrayRef< int64_t > wgShape, ArrayRef< int64_t > instData, int64_t sgCount, int64_t broadcastDim=-1) |
| static std::optional< SmallVector< int64_t > > | get2DBlockIOInstDataLayout (ArrayRef< int64_t > dataShape, Type elemTy, const xegpu::uArch::BlockIOInstructionInterface *uArchInstruction, bool transform=false, bool transpose=false) |
| Helper function to compute inst_data vectors for DPAS operands A, B, and C/D. | |
| static std::optional< std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > > | getDpasInstDataLayouts (VectorType aTy, VectorType bTy, VectorType cdTy, const xegpu::uArch::MMAInstructionInterface *uArchInstruction) |
| Helper function to compute inst_data vectors for DPAS operands A, B, and C/D. | |
| static std::pair< SmallVector< int64_t >, SmallVector< int64_t > > | computeScatterIOLaneLayoutAndData (ArrayRef< int64_t > instShape, int64_t subgroupSize, int64_t maxChunkSize) |
| Computes lane_layout and lane_data for scatter-style store anchor layouts (store scatter, store matrix). | |
| static std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > | compute2DBlockIOLaneLayout (ArrayRef< int64_t > instShape, int64_t subgroupSize, int64_t bitwidth, int64_t packingSize, bool transform=false, bool transpose=false) |
| static std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > | computeReductionLaneLayoutAndData (ArrayRef< int64_t > srcShape, ArrayRef< int64_t > reductionDims, int subgroupSize, int64_t maxReduceVectorSize, xegpu::DistributeLayoutAttr consumerLayout=nullptr) |
| Computes a multi_reduction's source layout as (lane_layout, lane_data, inst_data), where inst_data = lane_layout * lane_data on every dim. | |
| static std::optional< std::tuple< xegpu::DistributeLayoutAttr, xegpu::DistributeLayoutAttr, xegpu::DistributeLayoutAttr > > | getDpasSubgroupLayouts (mlir::MLIRContext *context, VectorType aTy, VectorType bTy, VectorType cdTy, xegpu::DistributeLayoutAttr consumerLayout, int numSg, std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > instDataVecs) |
| Helper function to set up subgroup layouts for DPAS operands A, B, and C/D. | |
| static xegpu::DistributeLayoutAttr | createScaleLayout (mlir::MLIRContext *context, VectorType matrixTy, VectorType scaleTy, xegpu::DistributeLayoutAttr matrixLayout, bool isBScale, const xegpu::uArch::uArch *uArch) |
| Helper to create a scale layout derived from a matrix operand layout. | |
| static xegpu::DistributeLayoutAttr | setupGenericLoadAnchorLayout (xegpu::LayoutKind layoutKind, mlir::MLIRContext *context, xegpu::DistributeLayoutAttr consumerLayout, int maxChunkSize, ArrayRef< int64_t > resShape, int subgroupSize) |
| Sets up the anchor layout for load gather and load matrix operation. | |
| static xegpu::DistributeLayoutAttr | getStoreSubgroupLayouts (mlir::MLIRContext *context, ArrayRef< int64_t > wgShape, ArrayRef< int64_t > instData, int numSg) |
| Picks the subgroup layout for a scatter-style store (store_scatter / store_matrix): the most balanced numSg factorization that divides wgShape with sg_data a multiple of instData. | |
| static xegpu::DistributeLayoutAttr | setupGenericStoreAnchorLayout (xegpu::LayoutKind layoutKind, mlir::MLIRContext *context, int maxChunkSize, ArrayRef< int64_t > srcShape, int subgroupSize, int numSg) |
| Sets up the anchor layout for store scatter and store matrix operation, which share the same logic. | |
| static xegpu::DistributeLayoutAttr | adjustInnermostDimForDivisibility (xegpu::DistributeLayoutAttr consumerLayout, xegpu::LayoutKind layoutKind, size_t innerMostDim, int ratio, int64_t bound, const xegpu::uArch::uArch *uArch) |
| Adjusts consumerLayout's innermost-dim data field selected by layoutKind so that the source layout can be safely inferred by dividing that value by ratio. | |
| static bool | isContiguousTiling (ArrayRef< int64_t > dimGroup, ArrayRef< int64_t > shape, ArrayRef< int64_t > counts, ArrayRef< int64_t > tileShape, int64_t expectedStride) |
| Checks that the tiles a split group is cut into walk the collapsed source dim with one constant stride, which is what collapseDims assumes when it multiplies a layout field across the group. | |
| static xegpu::DistributeLayoutAttr | getLoopCarriedLayoutForYieldOperand (RegionBranchTerminatorOpInterface terminator, OpOperand &operand) |
| static xegpu::DistributeLayoutAttr | getParentResultLayoutForYieldOperand (RegionBranchTerminatorOpInterface terminator, OpOperand &operand) |
| using LayoutRepresentation = SmallVector<int64_t> |
Definition at line 1022 of file XeGPULayoutImpl.cpp.
|
static |
Adjusts consumerLayout's innermost-dim data field selected by layoutKind so that the source layout can be safely inferred by dividing that value by ratio.
Doubles the value until the divisibility constraint is met, bounded above by bound like result-shape.
Used by ops whose source relates to the result by a fixed factor along the innermost dim (e.g., bitcast: bitwidth ratio; interleave: 2x).
Divisibility constraints per LayoutKind:
Definition at line 2687 of file XeGPULayoutImpl.cpp.
References mlir::xegpu::uArch::uArch::getSubgroupSize(), mlir::xegpu::InstData, mlir::xegpu::Lane, and mlir::xegpu::Subgroup.
Referenced by mlir::xegpu::setupBitCastResultLayout(), and mlir::xegpu::setupInterleaveResultLayout().
|
static |
Definition at line 438 of file XeGPULayoutImpl.cpp.
References mlir::detail::DenseArrayAttrImpl< int32_t >::get().
Referenced by setupGenericLoadAnchorLayout(), setupGenericStoreAnchorLayout(), mlir::xegpu::setupMultiReductionResultLayout(), and mlir::xegpu::setupStoreNdAnchorLayout().
|
static |
Definition at line 461 of file XeGPULayoutImpl.cpp.
References mlir::detail::DenseArrayAttrImpl< int32_t >::get().
Referenced by setupGenericLoadAnchorLayout(), setupGenericStoreAnchorLayout(), mlir::xegpu::setupMultiReductionResultLayout(), mlir::xegpu::setupReductionResultLayout(), and mlir::xegpu::setupStoreNdAnchorLayout().
|
static |
Definition at line 475 of file XeGPULayoutImpl.cpp.
References mlir::detail::DenseArrayAttrImpl< int32_t >::get().
Referenced by buildSgLayout(), createScaleLayout(), and mlir::xegpu::setupMultiReductionResultLayout().
|
static |
Definition at line 491 of file XeGPULayoutImpl.cpp.
References buildLayout().
Referenced by getDpasSubgroupLayouts(), getStoreSubgroupLayouts(), and mlir::xegpu::setupStoreNdAnchorLayout().
|
static |
Definition at line 1231 of file XeGPULayoutImpl.cpp.
Referenced by mlir::xegpu::setupStoreNdAnchorLayout().
|
static |
Computes a multi_reduction's source layout as (lane_layout, lane_data, inst_data), where inst_data = lane_layout * lane_data on every dim.
maxReduceVectorSize is the per-lane vector budget.
Limit lane_layout by srcShape: a dim can spread over no more lanes than it has elements to give.
Then hand any unused lanes to the innermost dim, if that dim is reduced. This is a heuristic: the innermost dim of the source typically already carries several lanes, whether the source is loaded from memory or is a dpas result, so putting them there matches what the producer did.
Raise lane_data on the innermost dim, so the lane holds a run of elements along it and reduces the run with one vector op instead of one element at a time. The run is at most maxReduceVectorSize long. The dim must also be reduced and have lane_layout 1.
How much: maxReduceVectorSize / product(lane_data), multiplied onto the dim's existing lane_data and capped by its extent.
e.g. src=[16,32,32] reduce [2], consumer lane=[1,2] -> lane_layout=[1,2,8], lane_data=[1,1,1], inst_data=[1,2,8]. e.g. src=[16,16] reduce [1], consumer lane=[16] -> lane_layout=[16,1], lane_data=[1,16], inst_data=[16,16].
Definition at line 1294 of file XeGPULayoutImpl.cpp.
References mlir::computeProduct().
Referenced by mlir::xegpu::setupMultiReductionResultLayout().
|
static |
Computes lane_layout and lane_data for scatter-style store anchor layouts (store scatter, store matrix).
Lanes and the per-lane vector both live on the innermost dim:
Definition at line 1210 of file XeGPULayoutImpl.cpp.
Referenced by setupGenericStoreAnchorLayout().
|
static |
Helper to create a scale layout derived from a matrix operand layout.
The scale layout is computed by mapping each dimension of the matrix layout to the corresponding scale tensor dimension using the ratio between the matrix and scale shapes.
Definition at line 1558 of file XeGPULayoutImpl.cpp.
References buildLayout(), mlir::xegpu::uArch::uArch::getInstruction(), and mlir::xegpu::uArch::SubgroupScaledMatrixMultiplyAcc.
|
static |
Enumerates all ways to split total into rank factors whose product equals total.
Returns the list of all such factorizations.
Definition at line 1026 of file XeGPULayoutImpl.cpp.
Referenced by getSgLayoutCandidates().
|
static |
Helper function to compute inst_data vectors for DPAS operands A, B, and C/D.
Definition at line 1128 of file XeGPULayoutImpl.cpp.
References mlir::xegpu::uArch::BlockIOInstructionInterface::getBlockWidthHeightCount(), and mlir::xegpu::getLargestDivisor().
Referenced by mlir::xegpu::setupStoreNdAnchorLayout().
|
static |
Helper function to compute inst_data vectors for DPAS operands A, B, and C/D.
Look up the uArch table and search for the largest supported block size that divides the data shape
Definition at line 1163 of file XeGPULayoutImpl.cpp.
References mlir::xegpu::getLargestDivisor(), mlir::xegpu::uArch::MMAInstructionInterface::getSupportedK(), mlir::xegpu::uArch::MMAInstructionInterface::getSupportedM(), and mlir::xegpu::uArch::MMAInstructionInterface::getSupportedN().
|
static |
Helper function to set up subgroup layouts for DPAS operands A, B, and C/D.
Compute subgroup layout candidates based on wgtile and instData, and then pick the best one that satisfies all operands and the consumer (if specified).
Definition at line 1445 of file XeGPULayoutImpl.cpp.
References buildSgLayout(), and getSgLayoutCandidates().
|
static |
Definition at line 123 of file XeGPULayoutImpl.cpp.
References mlir::xegpu::getDistributeLayoutAttr(), and result.
Referenced by propagateRegionResultsToYieldOperands(), propagateResultsToRegularOperands(), and mlir::xegpu::recoverTemporaryLayouts().
|
static |
Definition at line 3085 of file XeGPULayoutImpl.cpp.
References mlir::xegpu::getDistributeLayoutAttr().
Referenced by mlir::xegpu::getConsumerLayoutAt().
|
static |
Definition at line 3109 of file XeGPULayoutImpl.cpp.
References mlir::xegpu::getDistributeLayoutAttr(), and result.
Referenced by mlir::xegpu::getConsumerLayoutAt().
|
static |
Definition at line 1078 of file XeGPULayoutImpl.cpp.
References enumerateFactorizations().
Referenced by getDpasSubgroupLayouts(), getStoreSubgroupLayouts(), and mlir::xegpu::setupStoreNdAnchorLayout().
|
static |
Picks the subgroup layout for a scatter-style store (store_scatter / store_matrix): the most balanced numSg factorization that divides wgShape with sg_data a multiple of instData.
A store has no consumer.
Definition at line 2061 of file XeGPULayoutImpl.cpp.
References buildSgLayout(), and getSgLayoutCandidates().
Referenced by setupGenericStoreAnchorLayout().
|
static |
Checks that the tiles a split group is cut into walk the collapsed source dim with one constant stride, which is what collapseDims assumes when it multiplies a layout field across the group.
Along each dim the group is cut into counts[dim] tiles of tileShape[dim] elements (empty for single-element tiles) taken from shape[dim], which gives the row-major strides. Walking from the innermost dim out, a dim cut into several tiles steps by tileShape[dim] * stride(dim) and has to pick up where the inner dims left off, expectedStride; a dim cut into one tile does not step at all, so it only contributes its size.
Definition at line 2818 of file XeGPULayoutImpl.cpp.
Referenced by mlir::xegpu::setupShapeCastResultLayout().
|
static |
Definition at line 452 of file XeGPULayoutImpl.cpp.
Referenced by mlir::xegpu::setupStoreNdAnchorLayout().
|
static |
Definition at line 195 of file XeGPULayoutImpl.cpp.
References mlir::OperandRange::getBeginOperandIndex(), getLayoutFromUsePoints(), mlir::OperandRange::getType(), mlir::xegpu::setTemporaryLayout(), and setTensorDescLayout().
Referenced by mlir::xegpu::recoverTemporaryLayouts().
Definition at line 156 of file XeGPULayoutImpl.cpp.
References getLayoutFromUsePoints(), mlir::Operation::getNumResults(), mlir::Operation::getOpOperands(), mlir::Operation::getResult(), mlir::xegpu::inferSourceLayoutFromResultForNonAnchorOp(), result, mlir::xegpu::setTemporaryLayout(), and setTensorDescLayout().
Referenced by mlir::xegpu::recoverTemporaryLayouts().
Definition at line 82 of file XeGPULayoutImpl.cpp.
References mlir::Value::getType(), and mlir::Value::setType().
Referenced by propagateRegionResultsToYieldOperands(), and propagateResultsToRegularOperands().
|
static |
Sets up the anchor layout for load gather and load matrix operation.
load matrix lowers to load gather and 1d block load. All of them share the same layout setup logic.
For Subgroup layout, uses the consumer layout directly.
For InstData layout, takes consumer's inst_data as-is; lane_layout and lane_data are taken from the consumer.
For Lane layout, lane_layout/lane_data are taken from the consumer.
A consumer layout that carries lane_layout and lane_data is required. maxChunkSize is not read yet: the consumer's lane_data already fixes the per-lane chunk.
TODO: derive lane_layout/lane_data here when the consumer has none, capped by maxChunkSize, the way setupGenericStoreAnchorLayout does via computeScatterIOLaneLayoutAndData. That path is missing today, so the assert below stands in for it.
Definition at line 1980 of file XeGPULayoutImpl.cpp.
References buildInstDataLayoutWithLane(), buildLaneLayout(), mlir::xegpu::InstData, mlir::xegpu::Lane, and mlir::xegpu::Subgroup.
|
static |
Sets up the anchor layout for store scatter and store matrix operation, which share the same logic.
Lane layout comes from computeScatterIOLaneLayoutAndData; inst_data is lane_layout * lane_data.
Definition at line 2073 of file XeGPULayoutImpl.cpp.
References buildInstDataLayoutWithLane(), buildLaneLayout(), computeScatterIOLaneLayoutAndData(), getStoreSubgroupLayouts(), mlir::xegpu::InstData, mlir::xegpu::Lane, and mlir::xegpu::Subgroup.
Referenced by mlir::xegpu::setupStoreMatrixAnchorLayout(), and mlir::xegpu::setupStoreScatterAnchorLayout().
|
static |
Definition at line 95 of file XeGPULayoutImpl.cpp.
References mlir::Region::empty(), visit(), and walkRegionBackward().
Referenced by mlir::xegpu::recoverTemporaryLayouts(), and walkRegionBackward().
| template void xegpu::removeLayoutAttr< mlir::OpOperand > | ( | const mlir::OpOperand & | operand | ) |
References mlir::xegpu::removeLayoutAttr().
| template void xegpu::removeLayoutAttr< mlir::OpResult > | ( | const mlir::OpResult & | result | ) |
References mlir::xegpu::removeLayoutAttr(), and result.