MLIR 24.0.0git
XeGPULayoutImpl.cpp File Reference

Go to the source code of this file.

Typedefs

using LayoutRepresentation = SmallVector<int64_t>

Functions

static void setTensorDescLayout (Value val, xegpu::DistributeLayoutAttr layout)
static void walkRegionBackward (Region &region, llvm::function_ref< void(Operation *)> visit)
static xegpu::DistributeLayoutAttr getLayoutFromUsePoints (Value result)
static void propagateResultsToRegularOperands (Operation *op)
static void propagateRegionResultsToYieldOperands (mlir::RegionBranchTerminatorOpInterface yieldOp)
template void xegpu::removeLayoutAttr< mlir::OpResult > (const mlir::OpResult &result)
template void xegpu::removeLayoutAttr< mlir::OpOperand > (const mlir::OpOperand &operand)
static xegpu::LayoutAttr buildInstDataLayoutWithLane (mlir::MLIRContext *context, ArrayRef< int64_t > instData, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData, DenseI32ArrayAttr orderAttr=nullptr)
static bool isValidLaneLayout (ArrayRef< int64_t > dataShape, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData)
static xegpu::LayoutAttr buildLaneLayout (mlir::MLIRContext *context, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData, DenseI32ArrayAttr orderAttr=nullptr)
static xegpu::LayoutAttr buildLayout (mlir::MLIRContext *context, ArrayRef< int64_t > sgLayout, ArrayRef< int64_t > sgData, ArrayRef< int64_t > instData, ArrayRef< int64_t > laneLayout, ArrayRef< int64_t > laneData, DenseI32ArrayAttr orderAttr=nullptr)
static xegpu::LayoutAttr buildSgLayout (mlir::MLIRContext *context, ArrayRef< int64_t > wgTileShape, ArrayRef< int64_t > sgLayout, int dimK=-1, DenseI32ArrayAttr orderAttr=nullptr)
static SmallVector< LayoutRepresentation > enumerateFactorizations (int64_t total, int64_t rank)
 Enumerates all ways to split total into rank factors whose product equals total.
static SmallVector< LayoutRepresentation > getSgLayoutCandidates (ArrayRef< int64_t > wgShape, ArrayRef< int64_t > instData, int64_t sgCount, int64_t broadcastDim=-1)
static std::optional< SmallVector< int64_t > > get2DBlockIOInstDataLayout (ArrayRef< int64_t > dataShape, Type elemTy, const xegpu::uArch::BlockIOInstructionInterface *uArchInstruction, bool transform=false, bool transpose=false)
 Helper function to compute inst_data vectors for DPAS operands A, B, and C/D.
static std::optional< std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > > getDpasInstDataLayouts (VectorType aTy, VectorType bTy, VectorType cdTy, const xegpu::uArch::MMAInstructionInterface *uArchInstruction)
 Helper function to compute inst_data vectors for DPAS operands A, B, and C/D.
static std::pair< SmallVector< int64_t >, SmallVector< int64_t > > computeScatterIOLaneLayoutAndData (ArrayRef< int64_t > instShape, int64_t subgroupSize, int64_t maxChunkSize)
 Computes lane_layout and lane_data for scatter-style store anchor layouts (store scatter, store matrix).
static std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > compute2DBlockIOLaneLayout (ArrayRef< int64_t > instShape, int64_t subgroupSize, int64_t bitwidth, int64_t packingSize, bool transform=false, bool transpose=false)
static std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > computeReductionLaneLayoutAndData (ArrayRef< int64_t > srcShape, ArrayRef< int64_t > reductionDims, int subgroupSize, int64_t maxReduceVectorSize, xegpu::DistributeLayoutAttr consumerLayout=nullptr)
 Computes a multi_reduction's source layout as (lane_layout, lane_data, inst_data), where inst_data = lane_layout * lane_data on every dim.
static std::optional< std::tuple< xegpu::DistributeLayoutAttr, xegpu::DistributeLayoutAttr, xegpu::DistributeLayoutAttr > > getDpasSubgroupLayouts (mlir::MLIRContext *context, VectorType aTy, VectorType bTy, VectorType cdTy, xegpu::DistributeLayoutAttr consumerLayout, int numSg, std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > instDataVecs)
 Helper function to set up subgroup layouts for DPAS operands A, B, and C/D.
static xegpu::DistributeLayoutAttr createScaleLayout (mlir::MLIRContext *context, VectorType matrixTy, VectorType scaleTy, xegpu::DistributeLayoutAttr matrixLayout, bool isBScale, const xegpu::uArch::uArch *uArch)
 Helper to create a scale layout derived from a matrix operand layout.
static xegpu::DistributeLayoutAttr setupGenericLoadAnchorLayout (xegpu::LayoutKind layoutKind, mlir::MLIRContext *context, xegpu::DistributeLayoutAttr consumerLayout, int maxChunkSize, ArrayRef< int64_t > resShape, int subgroupSize)
 Sets up the anchor layout for load gather and load matrix operation.
static xegpu::DistributeLayoutAttr getStoreSubgroupLayouts (mlir::MLIRContext *context, ArrayRef< int64_t > wgShape, ArrayRef< int64_t > instData, int numSg)
 Picks the subgroup layout for a scatter-style store (store_scatter / store_matrix): the most balanced numSg factorization that divides wgShape with sg_data a multiple of instData.
static xegpu::DistributeLayoutAttr setupGenericStoreAnchorLayout (xegpu::LayoutKind layoutKind, mlir::MLIRContext *context, int maxChunkSize, ArrayRef< int64_t > srcShape, int subgroupSize, int numSg)
 Sets up the anchor layout for store scatter and store matrix operation, which share the same logic.
static xegpu::DistributeLayoutAttr adjustInnermostDimForDivisibility (xegpu::DistributeLayoutAttr consumerLayout, xegpu::LayoutKind layoutKind, size_t innerMostDim, int ratio, int64_t bound, const xegpu::uArch::uArch *uArch)
 Adjusts consumerLayout's innermost-dim data field selected by layoutKind so that the source layout can be safely inferred by dividing that value by ratio.
static bool isContiguousTiling (ArrayRef< int64_t > dimGroup, ArrayRef< int64_t > shape, ArrayRef< int64_t > counts, ArrayRef< int64_t > tileShape, int64_t expectedStride)
 Checks that the tiles a split group is cut into walk the collapsed source dim with one constant stride, which is what collapseDims assumes when it multiplies a layout field across the group.
static xegpu::DistributeLayoutAttr getLoopCarriedLayoutForYieldOperand (RegionBranchTerminatorOpInterface terminator, OpOperand &operand)
static xegpu::DistributeLayoutAttr getParentResultLayoutForYieldOperand (RegionBranchTerminatorOpInterface terminator, OpOperand &operand)

Typedef Documentation

◆ LayoutRepresentation

Definition at line 1022 of file XeGPULayoutImpl.cpp.

Function Documentation

◆ adjustInnermostDimForDivisibility()

xegpu::DistributeLayoutAttr adjustInnermostDimForDivisibility ( xegpu::DistributeLayoutAttr consumerLayout,
xegpu::LayoutKind layoutKind,
size_t innerMostDim,
int ratio,
int64_t bound,
const xegpu::uArch::uArch * uArch )
static

Adjusts consumerLayout's innermost-dim data field selected by layoutKind so that the source layout can be safely inferred by dividing that value by ratio.

Doubles the value until the divisibility constraint is met, bounded above by bound like result-shape.

Used by ops whose source relates to the result by a fixed factor along the innermost dim (e.g., bitcast: bitwidth ratio; interleave: 2x).

Divisibility constraints per LayoutKind:

  • Subgroup: sgData[innermost] % ratio == 0
  • InstData: instData[innermost] % (laneLayout[innermost] * ratio) == 0 (laneLayout falls back to subgroupSize if absent)
  • Lane: laneData[innermost] % ratio == 0

Definition at line 2687 of file XeGPULayoutImpl.cpp.

References mlir::xegpu::uArch::uArch::getSubgroupSize(), mlir::xegpu::InstData, mlir::xegpu::Lane, and mlir::xegpu::Subgroup.

Referenced by mlir::xegpu::setupBitCastResultLayout(), and mlir::xegpu::setupInterleaveResultLayout().

◆ buildInstDataLayoutWithLane()

◆ buildLaneLayout()

◆ buildLayout()

xegpu::LayoutAttr buildLayout ( mlir::MLIRContext * context,
ArrayRef< int64_t > sgLayout,
ArrayRef< int64_t > sgData,
ArrayRef< int64_t > instData,
ArrayRef< int64_t > laneLayout,
ArrayRef< int64_t > laneData,
DenseI32ArrayAttr orderAttr = nullptr )
static

◆ buildSgLayout()

xegpu::LayoutAttr buildSgLayout ( mlir::MLIRContext * context,
ArrayRef< int64_t > wgTileShape,
ArrayRef< int64_t > sgLayout,
int dimK = -1,
DenseI32ArrayAttr orderAttr = nullptr )
static

◆ compute2DBlockIOLaneLayout()

std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > compute2DBlockIOLaneLayout ( ArrayRef< int64_t > instShape,
int64_t subgroupSize,
int64_t bitwidth,
int64_t packingSize,
bool transform = false,
bool transpose = false )
static

Definition at line 1231 of file XeGPULayoutImpl.cpp.

Referenced by mlir::xegpu::setupStoreNdAnchorLayout().

◆ computeReductionLaneLayoutAndData()

std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > computeReductionLaneLayoutAndData ( ArrayRef< int64_t > srcShape,
ArrayRef< int64_t > reductionDims,
int subgroupSize,
int64_t maxReduceVectorSize,
xegpu::DistributeLayoutAttr consumerLayout = nullptr )
static

Computes a multi_reduction's source layout as (lane_layout, lane_data, inst_data), where inst_data = lane_layout * lane_data on every dim.

maxReduceVectorSize is the per-lane vector budget.

  1. Take lane_layout and lane_data from the consumer. A consumer sliced over exactly reductionDims is a view of a source-rank parent, and that parent is reused as-is. Any other consumer has the result's rank: its lane_layout and lane_data hold one element per surviving dim, taken in order. Reduced dims keep lane_layout 1 for now.
  2. Limit lane_layout by srcShape: a dim can spread over no more lanes than it has elements to give.

    Then hand any unused lanes to the innermost dim, if that dim is reduced. This is a heuristic: the innermost dim of the source typically already carries several lanes, whether the source is loaded from memory or is a dpas result, so putting them there matches what the producer did.

  3. Raise lane_data on the innermost dim, so the lane holds a run of elements along it and reduces the run with one vector op instead of one element at a time. The run is at most maxReduceVectorSize long. The dim must also be reduced and have lane_layout 1.

    How much: maxReduceVectorSize / product(lane_data), multiplied onto the dim's existing lane_data and capped by its extent.

e.g. src=[16,32,32] reduce [2], consumer lane=[1,2] -> lane_layout=[1,2,8], lane_data=[1,1,1], inst_data=[1,2,8]. e.g. src=[16,16] reduce [1], consumer lane=[16] -> lane_layout=[16,1], lane_data=[1,16], inst_data=[16,16].

Definition at line 1294 of file XeGPULayoutImpl.cpp.

References mlir::computeProduct().

Referenced by mlir::xegpu::setupMultiReductionResultLayout().

◆ computeScatterIOLaneLayoutAndData()

std::pair< SmallVector< int64_t >, SmallVector< int64_t > > computeScatterIOLaneLayoutAndData ( ArrayRef< int64_t > instShape,
int64_t subgroupSize,
int64_t maxChunkSize )
static

Computes lane_layout and lane_data for scatter-style store anchor layouts (store scatter, store matrix).

Lanes and the per-lane vector both live on the innermost dim:

  • laneLayout[innermost] = min(subgroupSize, srcShape[innermost])
  • laneData[innermost] = min(srcShape[innermost] / laneLayout[innermost], maxChunkSize) All other entries are 1.

Definition at line 1210 of file XeGPULayoutImpl.cpp.

Referenced by setupGenericStoreAnchorLayout().

◆ createScaleLayout()

xegpu::DistributeLayoutAttr createScaleLayout ( mlir::MLIRContext * context,
VectorType matrixTy,
VectorType scaleTy,
xegpu::DistributeLayoutAttr matrixLayout,
bool isBScale,
const xegpu::uArch::uArch * uArch )
static

Helper to create a scale layout derived from a matrix operand layout.

The scale layout is computed by mapping each dimension of the matrix layout to the corresponding scale tensor dimension using the ratio between the matrix and scale shapes.

Definition at line 1558 of file XeGPULayoutImpl.cpp.

References buildLayout(), mlir::xegpu::uArch::uArch::getInstruction(), and mlir::xegpu::uArch::SubgroupScaledMatrixMultiplyAcc.

◆ enumerateFactorizations()

SmallVector< LayoutRepresentation > enumerateFactorizations ( int64_t total,
int64_t rank )
static

Enumerates all ways to split total into rank factors whose product equals total.

Returns the list of all such factorizations.

Definition at line 1026 of file XeGPULayoutImpl.cpp.

Referenced by getSgLayoutCandidates().

◆ get2DBlockIOInstDataLayout()

std::optional< SmallVector< int64_t > > get2DBlockIOInstDataLayout ( ArrayRef< int64_t > dataShape,
Type elemTy,
const xegpu::uArch::BlockIOInstructionInterface * uArchInstruction,
bool transform = false,
bool transpose = false )
static

Helper function to compute inst_data vectors for DPAS operands A, B, and C/D.

Definition at line 1128 of file XeGPULayoutImpl.cpp.

References mlir::xegpu::uArch::BlockIOInstructionInterface::getBlockWidthHeightCount(), and mlir::xegpu::getLargestDivisor().

Referenced by mlir::xegpu::setupStoreNdAnchorLayout().

◆ getDpasInstDataLayouts()

std::optional< std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > > getDpasInstDataLayouts ( VectorType aTy,
VectorType bTy,
VectorType cdTy,
const xegpu::uArch::MMAInstructionInterface * uArchInstruction )
static

Helper function to compute inst_data vectors for DPAS operands A, B, and C/D.

Look up the uArch table and search for the largest supported block size that divides the data shape

Definition at line 1163 of file XeGPULayoutImpl.cpp.

References mlir::xegpu::getLargestDivisor(), mlir::xegpu::uArch::MMAInstructionInterface::getSupportedK(), mlir::xegpu::uArch::MMAInstructionInterface::getSupportedM(), and mlir::xegpu::uArch::MMAInstructionInterface::getSupportedN().

◆ getDpasSubgroupLayouts()

std::optional< std::tuple< xegpu::DistributeLayoutAttr, xegpu::DistributeLayoutAttr, xegpu::DistributeLayoutAttr > > getDpasSubgroupLayouts ( mlir::MLIRContext * context,
VectorType aTy,
VectorType bTy,
VectorType cdTy,
xegpu::DistributeLayoutAttr consumerLayout,
int numSg,
std::tuple< SmallVector< int64_t >, SmallVector< int64_t >, SmallVector< int64_t > > instDataVecs )
static

Helper function to set up subgroup layouts for DPAS operands A, B, and C/D.

Compute subgroup layout candidates based on wgtile and instData, and then pick the best one that satisfies all operands and the consumer (if specified).

Definition at line 1445 of file XeGPULayoutImpl.cpp.

References buildSgLayout(), and getSgLayoutCandidates().

◆ getLayoutFromUsePoints()

xegpu::DistributeLayoutAttr getLayoutFromUsePoints ( Value result)
static

◆ getLoopCarriedLayoutForYieldOperand()

xegpu::DistributeLayoutAttr getLoopCarriedLayoutForYieldOperand ( RegionBranchTerminatorOpInterface terminator,
OpOperand & operand )
static

◆ getParentResultLayoutForYieldOperand()

xegpu::DistributeLayoutAttr getParentResultLayoutForYieldOperand ( RegionBranchTerminatorOpInterface terminator,
OpOperand & operand )
static

◆ getSgLayoutCandidates()

SmallVector< LayoutRepresentation > getSgLayoutCandidates ( ArrayRef< int64_t > wgShape,
ArrayRef< int64_t > instData,
int64_t sgCount,
int64_t broadcastDim = -1 )
static

◆ getStoreSubgroupLayouts()

xegpu::DistributeLayoutAttr getStoreSubgroupLayouts ( mlir::MLIRContext * context,
ArrayRef< int64_t > wgShape,
ArrayRef< int64_t > instData,
int numSg )
static

Picks the subgroup layout for a scatter-style store (store_scatter / store_matrix): the most balanced numSg factorization that divides wgShape with sg_data a multiple of instData.

A store has no consumer.

Definition at line 2061 of file XeGPULayoutImpl.cpp.

References buildSgLayout(), and getSgLayoutCandidates().

Referenced by setupGenericStoreAnchorLayout().

◆ isContiguousTiling()

bool isContiguousTiling ( ArrayRef< int64_t > dimGroup,
ArrayRef< int64_t > shape,
ArrayRef< int64_t > counts,
ArrayRef< int64_t > tileShape,
int64_t expectedStride )
static

Checks that the tiles a split group is cut into walk the collapsed source dim with one constant stride, which is what collapseDims assumes when it multiplies a layout field across the group.

Along each dim the group is cut into counts[dim] tiles of tileShape[dim] elements (empty for single-element tiles) taken from shape[dim], which gives the row-major strides. Walking from the innermost dim out, a dim cut into several tiles steps by tileShape[dim] * stride(dim) and has to pick up where the inner dims left off, expectedStride; a dim cut into one tile does not step at all, so it only contributes its size.

Definition at line 2818 of file XeGPULayoutImpl.cpp.

Referenced by mlir::xegpu::setupShapeCastResultLayout().

◆ isValidLaneLayout()

bool isValidLaneLayout ( ArrayRef< int64_t > dataShape,
ArrayRef< int64_t > laneLayout,
ArrayRef< int64_t > laneData )
static

Definition at line 452 of file XeGPULayoutImpl.cpp.

Referenced by mlir::xegpu::setupStoreNdAnchorLayout().

◆ propagateRegionResultsToYieldOperands()

void propagateRegionResultsToYieldOperands ( mlir::RegionBranchTerminatorOpInterface yieldOp)
static

◆ propagateResultsToRegularOperands()

◆ setTensorDescLayout()

void setTensorDescLayout ( Value val,
xegpu::DistributeLayoutAttr layout )
static

◆ setupGenericLoadAnchorLayout()

xegpu::DistributeLayoutAttr setupGenericLoadAnchorLayout ( xegpu::LayoutKind layoutKind,
mlir::MLIRContext * context,
xegpu::DistributeLayoutAttr consumerLayout,
int maxChunkSize,
ArrayRef< int64_t > resShape,
int subgroupSize )
static

Sets up the anchor layout for load gather and load matrix operation.

load matrix lowers to load gather and 1d block load. All of them share the same layout setup logic.

For Subgroup layout, uses the consumer layout directly.

For InstData layout, takes consumer's inst_data as-is; lane_layout and lane_data are taken from the consumer.

For Lane layout, lane_layout/lane_data are taken from the consumer.

A consumer layout that carries lane_layout and lane_data is required. maxChunkSize is not read yet: the consumer's lane_data already fixes the per-lane chunk.

TODO: derive lane_layout/lane_data here when the consumer has none, capped by maxChunkSize, the way setupGenericStoreAnchorLayout does via computeScatterIOLaneLayoutAndData. That path is missing today, so the assert below stands in for it.

Definition at line 1980 of file XeGPULayoutImpl.cpp.

References buildInstDataLayoutWithLane(), buildLaneLayout(), mlir::xegpu::InstData, mlir::xegpu::Lane, and mlir::xegpu::Subgroup.

◆ setupGenericStoreAnchorLayout()

xegpu::DistributeLayoutAttr setupGenericStoreAnchorLayout ( xegpu::LayoutKind layoutKind,
mlir::MLIRContext * context,
int maxChunkSize,
ArrayRef< int64_t > srcShape,
int subgroupSize,
int numSg )
static

Sets up the anchor layout for store scatter and store matrix operation, which share the same logic.

Lane layout comes from computeScatterIOLaneLayoutAndData; inst_data is lane_layout * lane_data.

Definition at line 2073 of file XeGPULayoutImpl.cpp.

References buildInstDataLayoutWithLane(), buildLaneLayout(), computeScatterIOLaneLayoutAndData(), getStoreSubgroupLayouts(), mlir::xegpu::InstData, mlir::xegpu::Lane, and mlir::xegpu::Subgroup.

Referenced by mlir::xegpu::setupStoreMatrixAnchorLayout(), and mlir::xegpu::setupStoreScatterAnchorLayout().

◆ walkRegionBackward()

void walkRegionBackward ( Region & region,
llvm::function_ref< void(Operation *)> visit )
static

◆ xegpu::removeLayoutAttr< mlir::OpOperand >()

template void xegpu::removeLayoutAttr< mlir::OpOperand > ( const mlir::OpOperand & operand)

◆ xegpu::removeLayoutAttr< mlir::OpResult >()

template void xegpu::removeLayoutAttr< mlir::OpResult > ( const mlir::OpResult & result)