29#include "llvm/ADT/STLExtras.h"
30#include "llvm/Support/Debug.h"
35#define GEN_PASS_DEF_AFFINEVECTORIZE
36#include "mlir/Dialect/Affine/Transforms/Passes.h.inc"
575#define DEBUG_TYPE "early-vect"
582 int fastestVaryingMemRefDimension);
588static std::optional<NestedPattern>
592 int64_t d0 = fastestVaryingPattern.empty() ? -1 : fastestVaryingPattern[0];
593 int64_t d1 = fastestVaryingPattern.size() < 2 ? -1 : fastestVaryingPattern[1];
594 int64_t d2 = fastestVaryingPattern.size() < 3 ? -1 : fastestVaryingPattern[2];
595 switch (vectorRank) {
613 llvm::IsaPred<vector::TransferReadOp, vector::TransferWriteOp>);
624 void runOnOperation()
override;
630 unsigned patternDepth,
631 VectorizationStrategy *strategy) {
632 assert(patternDepth > depthInPattern &&
633 "patternDepth is greater than depthInPattern");
634 if (patternDepth - depthInPattern > strategy->vectorSizes.size()) {
638 strategy->loopToVectorDim[loop] =
639 strategy->vectorSizes.size() - (patternDepth - depthInPattern);
658 unsigned depthInPattern,
659 unsigned patternDepth,
660 VectorizationStrategy *strategy) {
661 for (
auto m : matches) {
663 patternDepth, strategy))) {
667 patternDepth, strategy);
676struct VectorizationState {
689 void registerOpVectorReplacement(Operation *replaced, Operation *
replacement);
702 void registerValueVectorReplacement(Value replaced, Operation *
replacement);
709 void registerBlockArgVectorReplacement(BlockArgument replaced,
722 void registerValueScalarReplacement(Value replaced, Value
replacement);
734 void registerLoopResultScalarReplacement(Value replaced, Value
replacement);
738 void getScalarValueReplacementsFor(
ValueRange inputVals,
739 SmallVectorImpl<Value> &replacedVals);
742 void finishVectorizationPattern(AffineForOp rootLoop);
751 IRMapping valueVectorReplacement;
754 IRMapping valueScalarReplacement;
756 DenseMap<Value, Value> loopResultScalarReplacement;
766 const VectorizationStrategy *strategy =
nullptr;
771 void registerValueVectorReplacementImpl(Value replaced, Value
replacement);
785void VectorizationState::registerOpVectorReplacement(
Operation *replaced,
787 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ commit vectorized op:\n");
788 LLVM_DEBUG(dbgs() << *replaced <<
"\n");
789 LLVM_DEBUG(dbgs() <<
"into\n");
793 "Unexpected replaced and replacement results");
794 assert(opVectorReplacement.count(replaced) == 0 &&
"already registered");
797 for (
auto resultTuple :
799 registerValueVectorReplacementImpl(std::get<0>(resultTuple),
800 std::get<1>(resultTuple));
813void VectorizationState::registerValueVectorReplacement(
816 "Expected single-result replacement");
820 registerValueVectorReplacementImpl(replaced,
replacement->getResult(0));
828void VectorizationState::registerBlockArgVectorReplacement(
829 BlockArgument replaced, BlockArgument
replacement) {
830 registerValueVectorReplacementImpl(replaced,
replacement);
833void VectorizationState::registerValueVectorReplacementImpl(Value replaced,
835 assert(!valueVectorReplacement.
contains(replaced) &&
836 "Vector replacement already registered");
838 "Expected vector type in vector replacement");
852void VectorizationState::registerValueScalarReplacement(Value replaced,
854 assert(!valueScalarReplacement.
contains(replaced) &&
855 "Scalar value replacement already registered");
857 "Expected scalar type in scalar replacement");
870void VectorizationState::registerLoopResultScalarReplacement(
873 assert(loopResultScalarReplacement.count(replaced) == 0 &&
874 "already registered");
875 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ will replace a result of the loop "
878 loopResultScalarReplacement[replaced] =
replacement;
882void VectorizationState::getScalarValueReplacementsFor(
883 ValueRange inputVals, SmallVectorImpl<Value> &replacedVals) {
884 for (Value inputVal : inputVals)
885 replacedVals.push_back(valueScalarReplacement.
lookupOrDefault(inputVal));
890 LLVM_DEBUG(dbgs() <<
"[early-vect]+++++ erasing:\n" << forOp <<
"\n");
899 LLVM_DEBUG(dbgs() <<
"[early-vect]+++++ erasing user:\n" << *user <<
"\n");
905void VectorizationState::finishVectorizationPattern(AffineForOp rootLoop) {
906 LLVM_DEBUG(dbgs() <<
"\n[early-vect] Finalizing vectorization\n");
913 VectorizationState &state,
918 auto afOp = AffineApplyOp::create(state.builder, op->
getLoc(), singleResMap,
920 results.push_back(afOp);
929 int fastestVaryingMemRefDimension) {
930 return [¶llelLoops, fastestVaryingMemRefDimension](
Operation &forOp) {
931 auto loop = cast<AffineForOp>(forOp);
932 if (!parallelLoops.contains(loop))
935 auto vectorizableBody =
937 if (!vectorizableBody)
939 return memRefDim == -1 || fastestVaryingMemRefDimension == -1 ||
940 memRefDim == fastestVaryingMemRefDimension;
947 const VectorizationStrategy *strategy) {
948 assert(!isa<VectorType>(scalarTy) &&
"Expected scalar type");
949 return VectorType::get(strategy->vectorSizes, scalarTy);
956 VectorizationState &state) {
957 Type scalarTy = constOp.getType();
958 if (!VectorType::isValidElementType(scalarTy))
967 while (parentOp && !state.vecLoopToVecDim.count(parentOp))
969 assert(parentOp && state.vecLoopToVecDim.count(parentOp) &&
970 isa<AffineForOp>(parentOp) &&
"Expected a vectorized for op");
971 auto vecForOp = cast<AffineForOp>(parentOp);
974 arith::ConstantOp::create(state.builder, constOp.getLoc(), vecAttr);
977 state.registerOpVectorReplacement(constOp, newConstOp);
984 if (isa<IndexType>(scalarTy)) {
985 auto scalarConstOp = arith::ConstantOp::create(
986 state.builder, constOp.getLoc(), constOp.getValue());
987 state.registerValueScalarReplacement(constOp.getResult(),
988 scalarConstOp.getResult());
997 VectorizationState &state) {
999 for (
Value operand : applyOp.getOperands()) {
1000 if (state.valueVectorReplacement.
contains(operand)) {
1002 dbgs() <<
"\n[early-vect]+++++ affine.apply on vector operand\n");
1006 if (!updatedOperand)
1007 updatedOperand = operand;
1008 updatedOperands.push_back(updatedOperand);
1011 auto newApplyOp = AffineApplyOp::create(
1012 state.builder, applyOp.getLoc(), applyOp.getAffineMap(), updatedOperands);
1015 state.registerValueScalarReplacement(applyOp.getResult(),
1016 newApplyOp.getResult());
1027 if (!VectorType::isValidElementType(scalarTy))
1030 Attribute valueAttr = getIdentityValueAttr(
1031 reductionKind, scalarTy, state.builder, oldOperand.
getLoc());
1035 arith::ConstantOp::create(state.builder, oldOperand.
getLoc(), vecAttr);
1048 assert(state.strategy->vectorSizes.size() == 1 &&
1049 "Creating a mask non-1-D vectors is not supported.");
1050 assert(vecForOp.getStep() == state.strategy->vectorSizes[0] &&
1051 "Creating a mask for loops with non-unit original step size is not "
1055 if (
Value mask = state.vecLoopToMask.lookup(vecForOp))
1060 if (vecForOp.hasConstantBounds()) {
1062 vecForOp.getConstantUpperBound() - vecForOp.getConstantLowerBound();
1063 if (originalTripCount % vecForOp.getStepAsInt() == 0)
1084 AffineMap ubMap = vecForOp.getUpperBoundMap();
1087 ub = AffineApplyOp::create(state.builder, loc, vecForOp.getUpperBoundMap(),
1088 vecForOp.getUpperBoundOperands());
1090 ub = AffineMinOp::create(state.builder, loc, vecForOp.getUpperBoundMap(),
1091 vecForOp.getUpperBoundOperands());
1097 {ub, vecForOp.getInductionVar()});
1100 ub.getDefiningOp()->erase();
1102 Type maskTy = VectorType::get(state.strategy->vectorSizes,
1105 vector::CreateMaskOp::create(state.builder, loc, maskTy, itersLeft);
1107 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ creating a mask:\n"
1108 << itersLeft <<
"\n"
1111 state.vecLoopToMask[vecForOp] = mask;
1121 const VectorizationStrategy *strategy) {
1123 if (forOp && strategy->loopToVectorDim.count(forOp) == 0)
1126 for (
auto loopToDim : strategy->loopToVectorDim) {
1127 auto loop = cast<AffineForOp>(loopToDim.first);
1128 if (!loop.isDefinedOutsideOfLoop(value))
1138 VectorizationState &state) {
1140 Value uniformScalarRepl =
1145 auto bcastOp = BroadcastOp::create(state.builder, uniformVal.
getLoc(),
1146 vectorTy, uniformScalarRepl);
1147 state.registerValueVectorReplacement(uniformVal, bcastOp);
1169 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ vectorize operand: " << operand);
1172 LLVM_DEBUG(dbgs() <<
" -> already vectorized: " << vecRepl);
1179 assert(!isa<VectorType>(operand.
getType()) &&
1180 "Vector op not found in replacement map");
1183 if (
auto constOp = operand.
getDefiningOp<arith::ConstantOp>()) {
1185 LLVM_DEBUG(dbgs() <<
"-> constant: " << vecConstant);
1186 return vecConstant.getResult();
1192 LLVM_DEBUG(dbgs() <<
"-> uniform: " << *vecUniform);
1199 LLVM_DEBUG(dbgs() <<
"-> unsupported block argument\n");
1202 LLVM_DEBUG(dbgs() <<
"-> non-vectorizable\n");
1211 for (
auto &kvp : loopToVectorDim) {
1212 AffineForOp forOp = cast<AffineForOp>(kvp.first);
1217 unsigned nonInvariant = 0;
1219 if (invariants.count(idx))
1222 if (++nonInvariant > 1) {
1223 LLVM_DEBUG(dbgs() <<
"[early‑vect] Bail out: IV "
1224 << forOp.getInductionVar() <<
" drives "
1225 << nonInvariant <<
" indices\n");
1240 VectorizationState &state) {
1241 MemRefType memRefType = loadOp.getMemRefType();
1242 Type elementType = memRefType.getElementType();
1243 auto vectorType = VectorType::get(state.strategy->vectorSizes, elementType);
1247 state.getScalarValueReplacementsFor(loadOp.getMapOperands(), mapOperands);
1251 indices.reserve(memRefType.getRank());
1252 if (loadOp.getAffineMap() !=
1255 for (
auto op : mapOperands) {
1256 if (op.getDefiningOp<AffineApplyOp>())
1262 indices.append(mapOperands.begin(), mapOperands.end());
1270 indices, state.vecLoopToVecDim);
1271 if (!permutationMap) {
1272 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ can't compute permutationMap\n");
1275 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ permutationMap: ");
1276 LLVM_DEBUG(permutationMap.print(dbgs()));
1279 state.builder, loadOp.getLoc(), loadOp.getMemRef(), vectorType,
1285 state.registerOpVectorReplacement(loadOp, transfer);
1296 VectorizationState &state) {
1297 MemRefType memRefType = storeOp.getMemRefType();
1304 state.getScalarValueReplacementsFor(storeOp.getMapOperands(), mapOperands);
1308 indices.reserve(memRefType.getRank());
1309 if (storeOp.getAffineMap() !=
1314 indices.append(mapOperands.begin(), mapOperands.end());
1321 indices, state.vecLoopToVecDim);
1322 if (!permutationMap)
1324 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ permutationMap: ");
1325 LLVM_DEBUG(permutationMap.print(dbgs()));
1329 if (llvm::any_of(permutationMap.getResults(),
1330 llvm::IsaPred<AffineConstantExpr>)) {
1331 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ store permutation map has "
1332 "broadcast dims, bailing out\n");
1337 state.builder, storeOp.getLoc(), vectorValue, storeOp.getMemRef(),
1339 true, permutationMap);
1340 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ vectorized store: " << *transfer);
1343 state.registerOpVectorReplacement(storeOp, transfer);
1350 Value value, VectorizationState &state) {
1352 if (!VectorType::isValidElementType(scalarTy))
1354 Attribute valueAttr = getIdentityValueAttr(reductionKind, scalarTy,
1355 state.builder, value.
getLoc());
1356 if (
auto constOp = value.
getDefiningOp<arith::ConstantOp>())
1357 return constOp.getValue() == valueAttr;
1368 VectorizationState &state) {
1369 const VectorizationStrategy &strategy = *state.strategy;
1370 auto loopToVecDimIt = strategy.loopToVectorDim.find(forOp);
1371 bool isLoopVecDim = loopToVecDimIt != strategy.loopToVectorDim.end();
1374 if (isLoopVecDim && forOp.getNumIterOperands() > 0 && forOp.getStep() != 1) {
1377 <<
"\n[early-vect]+++++ unsupported step size for reduction loop: "
1378 << forOp.getStep() <<
"\n");
1387 unsigned vectorDim = loopToVecDimIt->second;
1388 assert(vectorDim < strategy.vectorSizes.size() &&
"vector dim overflow");
1389 int64_t forOpVecFactor = strategy.vectorSizes[vectorDim];
1390 newStep = forOp.getStepAsInt() * forOpVecFactor;
1392 newStep = forOp.getStepAsInt();
1397 if (isLoopVecDim && forOp.getNumIterOperands() > 0) {
1398 auto it = strategy.reductionLoops.find(forOp);
1399 assert(it != strategy.reductionLoops.end() &&
1400 "Reduction descriptors not found when vectorizing a reduction loop");
1401 reductions = it->second;
1402 assert(reductions.size() == forOp.getNumIterOperands() &&
1403 "The size of reductions array must match the number of iter_args");
1408 if (!isLoopVecDim) {
1409 for (
auto operand : forOp.getInits())
1415 for (
auto redAndOperand : llvm::zip(reductions, forOp.getInits())) {
1417 std::get<0>(redAndOperand).kind, std::get<1>(redAndOperand), state));
1425 state.getScalarValueReplacementsFor(forOp.getLowerBoundOperands(),
1427 state.getScalarValueReplacementsFor(forOp.getUpperBoundOperands(),
1429 auto vecForOp = AffineForOp::create(
1430 state.builder, forOp.getLoc(), lbOperands, forOp.getLowerBoundMap(),
1431 ubOperands, forOp.getUpperBoundMap(), newStep, vecIterOperands,
1450 state.registerOpVectorReplacement(forOp, vecForOp);
1451 state.registerValueScalarReplacement(forOp.getInductionVar(),
1452 vecForOp.getInductionVar());
1453 for (
auto iterTuple :
1454 llvm ::zip(forOp.getRegionIterArgs(), vecForOp.getRegionIterArgs()))
1455 state.registerBlockArgVectorReplacement(std::get<0>(iterTuple),
1456 std::get<1>(iterTuple));
1459 for (
unsigned i = 0; i < vecForOp.getNumIterOperands(); ++i) {
1463 vecForOp.getLoc(), vecForOp.getResult(i));
1464 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ creating a vector reduction: "
1468 Value origInit = forOp.getOperand(forOp.getNumControlOperands() + i);
1469 Value finalRes = reducedRes;
1473 reducedRes.
getLoc(), reducedRes, origInit);
1474 state.registerLoopResultScalarReplacement(forOp.getResult(i), finalRes);
1479 state.vecLoopToVecDim[vecForOp] = loopToVecDimIt->second;
1487 if (isLoopVecDim && forOp.getNumIterOperands() > 0)
1499 vectorTypes.push_back(
1500 VectorType::get(state.strategy->vectorSizes,
result.getType()));
1506 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ an operand failed vectorize\n");
1509 vectorOperands.push_back(vecOperand);
1522 state.registerOpVectorReplacement(op, vecOp);
1531 VectorizationState &state) {
1544 if (
Value mask = state.vecLoopToMask.lookup(newParentOp)) {
1549 cast<AffineForOp>(newParentOp).getRegionIterArgs(), i, combinerOps);
1550 assert(reducedVal &&
"expect non-null value for parallel reduction loop");
1551 assert(combinerOps.size() == 1 &&
"expect only one combiner op");
1553 Value neutralVal = cast<AffineForOp>(newParentOp).getInits()[i];
1555 Value maskedReducedVal = arith::SelectOp::create(
1556 state.builder, reducedVal.
getLoc(), mask, reducedVal, neutralVal);
1558 dbgs() <<
"\n[early-vect]+++++ masking an input to a binary op that"
1559 "produces value for a yield Op: "
1560 << maskedReducedVal);
1561 combinerOps.back()->replaceUsesOfWith(reducedVal, maskedReducedVal);
1579 VectorizationState &state) {
1581 assert(!isa<vector::TransferReadOp>(op) &&
1582 "vector.transfer_read cannot be further vectorized");
1583 assert(!isa<vector::TransferWriteOp>(op) &&
1584 "vector.transfer_write cannot be further vectorized");
1586 if (
auto loadOp = dyn_cast<AffineLoadOp>(op))
1588 if (
auto storeOp = dyn_cast<AffineStoreOp>(op))
1590 if (
auto forOp = dyn_cast<AffineForOp>(op))
1592 if (
auto yieldOp = dyn_cast<AffineYieldOp>(op))
1594 if (
auto constant = dyn_cast<arith::ConstantOp>(op))
1596 if (
auto applyOp = dyn_cast<AffineApplyOp>(op))
1614 assert(currentLevel <= loops.size() &&
"Unexpected currentLevel");
1615 if (currentLevel == loops.size())
1616 loops.emplace_back();
1640 const VectorizationStrategy &strategy) {
1641 assert(loops[0].size() == 1 &&
"Expected single root loop");
1642 AffineForOp rootLoop = loops[0][0];
1643 VectorizationState state(rootLoop.getContext());
1645 state.strategy = &strategy;
1655 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ loop is not vectorizable");
1668 LLVM_DEBUG(dbgs() <<
"[early-vect]+++++ Vectorizing: " << *op);
1672 dbgs() <<
"[early-vect]+++++ failed vectorizing the operation: "
1680 if (opVecResult.wasInterrupted()) {
1681 LLVM_DEBUG(dbgs() <<
"[early-vect]+++++ failed vectorization for: "
1682 << rootLoop <<
"\n");
1684 auto vecRootLoopIt = state.opVectorReplacement.find(rootLoop);
1685 if (vecRootLoopIt != state.opVectorReplacement.end()) {
1686 auto vecRootLoop = cast<AffineForOp>(vecRootLoopIt->second);
1696 for (
auto resPair : state.loopResultScalarReplacement)
1697 resPair.first.replaceAllUsesWith(resPair.second);
1699 assert(state.opVectorReplacement.count(rootLoop) == 1 &&
1700 "Expected vector replacement for loop nest");
1701 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ success vectorizing pattern");
1702 LLVM_DEBUG(dbgs() <<
"\n[early-vect]+++++ vectorization result:\n"
1703 << *state.opVectorReplacement[rootLoop]);
1706 state.finishVectorizationPattern(rootLoop);
1714 const VectorizationStrategy &strategy) {
1715 std::vector<SmallVector<AffineForOp, 2>> loopsToVectorize;
1727 assert(intersectionBuckets.empty() &&
"Expected empty output");
1732 AffineForOp matchRoot = cast<AffineForOp>(match.getMatchedOperation());
1733 bool intersects =
false;
1734 for (
int i = 0, end = intersectionBuckets.size(); i < end; ++i) {
1735 AffineForOp bucketRoot = bucketRoots[i];
1737 if (bucketRoot->isAncestor(matchRoot)) {
1738 intersectionBuckets[i].push_back(match);
1744 if (matchRoot->isAncestor(bucketRoot)) {
1745 bucketRoots[i] = matchRoot;
1746 intersectionBuckets[i].push_back(match);
1755 bucketRoots.push_back(matchRoot);
1756 intersectionBuckets.emplace_back();
1757 intersectionBuckets.back().push_back(match);
1772 assert((reductionLoops.empty() || vectorSizes.size() == 1) &&
1773 "Vectorizing reductions is supported only for 1-D vectors");
1776 std::optional<NestedPattern> pattern =
1777 makePattern(loops, vectorSizes.size(), fastestVaryingPattern);
1779 LLVM_DEBUG(dbgs() <<
"\n[early-vect] pattern couldn't be computed\n");
1783 LLVM_DEBUG(dbgs() <<
"\n******************************************");
1784 LLVM_DEBUG(dbgs() <<
"\n******************************************");
1785 LLVM_DEBUG(dbgs() <<
"\n[early-vect] new pattern on parent op\n");
1786 LLVM_DEBUG(dbgs() << *parentOp <<
"\n");
1788 unsigned patternDepth = pattern->getDepth();
1793 pattern->match(parentOp, &allMatches);
1794 std::vector<SmallVector<NestedMatch, 8>> intersectionBuckets;
1800 for (
auto &intersectingMatches : intersectionBuckets) {
1802 VectorizationStrategy strategy;
1804 strategy.vectorSizes.assign(vectorSizes.begin(), vectorSizes.end());
1805 strategy.reductionLoops = reductionLoops;
1807 patternDepth, &strategy))) {
1821 LLVM_DEBUG(dbgs() <<
"\n");
1824void affine::vectorizeChildAffineLoops(
1825 Operation *parentOp,
bool vectorizeReductions,
1826 ArrayRef<int64_t> vectorSizes, ArrayRef<int64_t> fastestVaryingPattern) {
1832 if (vectorizeReductions) {
1833 parentOp->
walk([¶llelLoops, &reductionLoops](AffineForOp loop) {
1834 SmallVector<LoopReduction, 2> reductions;
1835 if (isLoopParallel(loop, &reductions)) {
1836 parallelLoops.insert(loop);
1838 if (!reductions.empty())
1839 reductionLoops[loop] = reductions;
1843 parentOp->
walk([¶llelLoops](AffineForOp loop) {
1844 if (isLoopParallel(loop))
1845 parallelLoops.insert(loop);
1850 NestedPatternContext mlContext;
1851 vectorizeLoops(parentOp, parallelLoops, vectorSizes, fastestVaryingPattern,
1857void Vectorize::runOnOperation() {
1858 func::FuncOp f = getOperation();
1859 if (vectorSizes.empty()) {
1860 f.emitError(
"The 'virtual-vector-size' option must be specified.");
1861 return signalPassFailure();
1864 if (llvm::any_of(vectorSizes, [](int64_t size) {
return size <= 0; })) {
1866 "The 'virtual-vector-size' option must contain only positive values.");
1867 return signalPassFailure();
1870 if (!fastestVaryingPattern.empty() &&
1871 fastestVaryingPattern.size() != vectorSizes.size()) {
1872 f.emitRemark(
"Fastest varying pattern specified with different size than "
1873 "the vector size.");
1874 return signalPassFailure();
1877 if (vectorizeReductions && vectorSizes.size() != 1) {
1878 f.emitError(
"Vectorizing reductions is supported only for 1-D vectors.");
1879 return signalPassFailure();
1882 vectorizeChildAffineLoops(f, vectorizeReductions, vectorSizes,
1883 fastestVaryingPattern);
1899 if (loops[0].size() != 1)
1903 for (
int i = 1, end = loops.size(); i < end; ++i) {
1904 for (AffineForOp loop : loops[i]) {
1907 if (none_of(loops[i - 1], [&](AffineForOp maybeParent) {
1908 return maybeParent->isProperAncestor(loop);
1914 for (AffineForOp sibling : loops[i]) {
1915 if (sibling->isProperAncestor(loop))
1932void mlir::affine::vectorizeAffineLoops(
1934 ArrayRef<int64_t> vectorSizes, ArrayRef<int64_t> fastestVaryingPattern,
1937 NestedPatternContext mlContext;
1938 vectorizeLoops(parentOp, loops, vectorSizes, fastestVaryingPattern,
1977LogicalResult mlir::affine::vectorizeAffineLoopNest(
1978 std::vector<SmallVector<AffineForOp, 2>> &loops,
1979 const VectorizationStrategy &strategy) {
1981 NestedPatternContext mlContext;
*if copies could not be generated due to yet unimplemented cases *copyInPlacementStart and copyOutPlacementStart in copyPlacementBlock *specify the insertion points where the incoming copies and outgoing should be the output argument nBegin is set to its * replacement(set to `begin` if no invalidation happens). Since outgoing *copies could have been inserted at `end`
static Operation * vectorizeUniform(Value uniformVal, VectorizationState &state)
Generates a broadcast op for the provided uniform value using the vectorization strategy in 'state'.
static std::optional< NestedPattern > makePattern(const DenseSet< Operation * > ¶llelLoops, int vectorRank, ArrayRef< int64_t > fastestVaryingPattern)
Creates a vectorization pattern from the command line arguments.
static LogicalResult vectorizeRootMatch(NestedMatch m, const VectorizationStrategy &strategy)
Extracts the matched loops and vectorizes them following a topological order.
static void vectorizeLoopIfProfitable(Operation *loop, unsigned depthInPattern, unsigned patternDepth, VectorizationStrategy *strategy)
static LogicalResult verifyLoopNesting(const std::vector< SmallVector< AffineForOp, 2 > > &loops)
Verify that affine loops in 'loops' meet the nesting criteria expected by SuperVectorizer:
static Operation * vectorizeOneOperation(Operation *op, VectorizationState &state)
Encodes Operation-specific behavior for vectorization.
static bool isNeutralElementConst(arith::AtomicRMWKind reductionKind, Value value, VectorizationState &state)
Returns true if value is a constant equal to the neutral element of the given vectorizable reduction.
static void eraseUsers(Operation *op)
Recursively erases the users of the results of 'op'.
static LogicalResult vectorizeLoopNest(std::vector< SmallVector< AffineForOp, 2 > > &loops, const VectorizationStrategy &strategy)
Internal implementation to vectorize affine loops from a single loop nest using an n-D vectorization ...
static Operation * vectorizeAffineLoad(AffineLoadOp loadOp, VectorizationState &state)
Vectorizes an affine load with the vectorization strategy in 'state' by generating a 'vector....
static Operation * vectorizeAffineForOp(AffineForOp forOp, VectorizationState &state)
Vectorizes a loop with the vectorization strategy in 'state'.
static Operation * vectorizeAffineApplyOp(AffineApplyOp applyOp, VectorizationState &state)
We have no need to vectorize affine.apply.
static LogicalResult analyzeProfitability(ArrayRef< NestedMatch > matches, unsigned depthInPattern, unsigned patternDepth, VectorizationStrategy *strategy)
Implements a simple strawman strategy for vectorization.
static FilterFunctionType isVectorizableLoopPtrFactory(const DenseSet< Operation * > ¶llelLoops, int fastestVaryingMemRefDimension)
Forward declaration.
static bool isIVMappedToMultipleIndices(ArrayRef< Value > indices, const DenseMap< Operation *, unsigned > &loopToVectorDim)
Returns true if any vectorized loop IV drives more than one index.
static arith::ConstantOp vectorizeConstant(arith::ConstantOp constOp, VectorizationState &state)
Tries to transform a scalar constant into a vector constant.
static bool isUniformDefinition(Value value, const VectorizationStrategy *strategy)
Returns true if the provided value is vector uniform given the vectorization strategy.
static void eraseLoopNest(AffineForOp forOp)
Erases a loop nest, including all its nested operations.
static VectorType getVectorType(Type scalarTy, const VectorizationStrategy *strategy)
Returns the vector type resulting from applying the provided vectorization strategy on the scalar typ...
static void getMatchedAffineLoops(NestedMatch match, std::vector< SmallVector< AffineForOp, 2 > > &loops)
Converts all the nested loops in 'match' to a 2D vector container that preserves the relative nesting...
static Value vectorizeOperand(Value operand, VectorizationState &state)
Tries to vectorize a given operand by applying the following logic:
static void getMatchedAffineLoopsRec(NestedMatch match, unsigned currentLevel, std::vector< SmallVector< AffineForOp, 2 > > &loops)
Recursive implementation to convert all the nested loops in 'match' to a 2D vector container that pre...
static Operation * vectorizeAffineYieldOp(AffineYieldOp yieldOp, VectorizationState &state)
Vectorizes a yield operation by widening its types.
static arith::ConstantOp createInitialVector(arith::AtomicRMWKind reductionKind, Value oldOperand, VectorizationState &state)
Creates a constant vector filled with the neutral elements of the given reduction.
static Operation * widenOp(Operation *op, VectorizationState &state)
Vectorizes arbitrary operation by plain widening.
static Operation * vectorizeAffineStore(AffineStoreOp storeOp, VectorizationState &state)
Vectorizes an affine store with the vectorization strategy in 'state' by generating a 'vector....
static NestedPattern & vectorTransferPattern()
static void vectorizeLoops(Operation *parentOp, DenseSet< Operation * > &loops, ArrayRef< int64_t > vectorSizes, ArrayRef< int64_t > fastestVaryingPattern, const ReductionLoopMap &reductionLoops)
Internal implementation to vectorize affine loops in 'loops' using the n-D vectorization factors in '...
static void computeMemoryOpIndices(Operation *op, AffineMap map, ValueRange mapOperands, VectorizationState &state, SmallVectorImpl< Value > &results)
static void computeIntersectionBuckets(ArrayRef< NestedMatch > matches, std::vector< SmallVector< NestedMatch, 8 > > &intersectionBuckets)
Traverses all the loop matches and classifies them into intersection buckets.
static Value createMask(AffineForOp vecForOp, VectorizationState &state)
Creates a mask used to filter out garbage elements in the last iteration of unaligned loops.
static AffineMap makePermutationMap(ArrayRef< Value > indices, const DenseMap< Operation *, unsigned > &enclosingLoopToVectorDim)
Constructs a permutation map from memref indices to vector dimension.
Base type for affine expression.
A multi-dimensional affine map Affine map's are immutable like Type's, and they are uniqued.
static AffineMap get(MLIRContext *context)
Returns a zero result affine map with no dimensions or symbols: () -> ().
unsigned getNumSymbols() const
unsigned getNumDims() const
ArrayRef< AffineExpr > getResults() const
unsigned getNumResults() const
Attributes are known-constant values of operations.
Operation * getParentOp()
Returns the closest surrounding operation that contains this block.
AffineMap getMultiDimIdentityMap(unsigned rank)
IntegerType getIntegerType(unsigned width)
AffineExpr getAffineDimExpr(unsigned position)
static DenseElementsAttr get(ShapedType type, ArrayRef< Attribute > values)
Constructs a dense elements attribute from an array of element values.
auto lookupOrDefault(T from) const
Lookup a mapped value within the map.
void map(Value from, Value to)
Inserts a new mapping for 'from' to 'to'.
bool contains(T from) const
Checks to see if a mapping for 'from' exists.
auto lookupOrNull(T from) const
Lookup a mapped value within the map.
This class defines the main interface for locations in MLIR and acts as a non-nullable wrapper around...
RAII guard to reset the insertion point of the builder when destroyed.
This class helps build Operations.
void setInsertionPointToStart(Block *block)
Sets the insertion point to the start of the specified block.
void setInsertionPoint(Block *block, Block::iterator insertPoint)
Set the insertion point to the specified location.
Block * getInsertionBlock() const
Return the block the current insertion point belongs to.
void setInsertionPointAfterValue(Value val)
Sets the insertion point to the node after the specified value.
Operation * create(const OperationState &state)
Creates an operation given the fields represented as an OperationState.
void setInsertionPointAfter(Operation *op)
Sets the insertion point to the node after the specified operation, which will cause subsequent inser...
Operation is the basic unit of execution within MLIR.
bool use_empty()
Returns true if this operation has no uses.
OpResult getResult(unsigned idx)
Get the 'idx'th result of this operation.
unsigned getNumRegions()
Returns the number of regions held by this operation.
Location getLoc()
The source location the operation was defined or derived from.
Operation * getParentOp()
Returns the closest surrounding operation that contains this operation or nullptr if this is a top-le...
unsigned getNumOperands()
Attribute getPropertiesAsAttribute()
Return the properties converted to an attribute.
OperationName getName()
The name of an operation is the key identifier for it.
DictionaryAttr getDiscardableAttrDictionary()
Return all of the discardable attributes on this operation as a DictionaryAttr.
operand_range getOperands()
Returns an iterator on the underlying Value's.
std::enable_if_t< llvm::function_traits< std::decay_t< FnT > >::num_args==1, RetT > walk(FnT &&callback)
Walk the operation by calling the callback for each nested operation (including this one),...
result_range getResults()
user_iterator user_begin()
void erase()
Remove this operation from its parent block and delete it.
unsigned getNumResults()
Return the number of results held by this operation.
Instances of the Type class are uniqued, have an immutable identifier and an optional mutable compone...
bool isIntOrIndexOrFloat() const
Return true if this is an integer (of any signedness), index, or float type.
This class provides an abstraction over the different types of ranges over Values.
This class represents an instance of an SSA value in the MLIR system, representing a computable value...
Type getType() const
Return the type of this value.
Location getLoc() const
Return the location of this value.
Operation * getDefiningOp() const
If this value is the result of an operation, return the operation that defines it.
static WalkResult advance()
static WalkResult interrupt()
An NestedPattern captures nested patterns in the IR.
ArrayRef< NestedMatch > getMatchedChildren()
Operation * getMatchedOperation() const
NestedPattern For(const NestedPattern &child)
NestedPattern Op(FilterFunctionType filter=defaultFilterFunction)
AffineApplyOp makeComposedAffineApply(OpBuilder &b, Location loc, AffineMap map, ArrayRef< OpFoldResult > operands, bool composeAffineMin=false)
Returns a composed AffineApplyOp by composing map and operands with other AffineApplyOps supplying th...
DenseMap< Operation *, SmallVector< LoopReduction, 2 > > ReductionLoopMap
bool isVectorizableLoopBody(AffineForOp loop, NestedPattern &vectorTransferMatcher)
Checks whether the loop is structurally vectorizable; i.e.:
DenseSet< Value, DenseMapInfo< Value > > getInvariantAccesses(Value iv, ArrayRef< Value > indices)
Given an induction variable iv of type AffineForOp and indices of type IndexType, returns the set of ...
AffineForOp getForInductionVarOwner(Value val)
Returns the loop parent of an induction variable.
std::function< bool(Operation &)> FilterFunctionType
A NestedPattern is a nested operation walker that:
Value getReductionOp(AtomicRMWKind op, OpBuilder &builder, Location loc, Value lhs, Value rhs)
Returns the value obtained by applying the reduction operation kind associated with a binary AtomicRM...
Operation * createWriteOrMaskedWrite(OpBuilder &builder, Location loc, Value vecToStore, Value dest, SmallVector< Value > writeIndices={}, bool useInBoundsInsteadOfMasking=false, AffineMap permutationMap=AffineMap())
Create a TransferWriteOp of vecToStore into dest.
Value createReadOrMaskedRead(OpBuilder &builder, Location loc, Value source, const VectorType &vecToReadTy, std::optional< Value > padValue=std::nullopt, bool useInBoundsInsteadOfMasking=false, ArrayRef< Value > indices={}, AffineMap permutationMap=AffineMap())
Creates a TransferReadOp from source.
Value getVectorReductionOp(arith::AtomicRMWKind op, OpBuilder &builder, Location loc, Value vector)
Returns the value obtained by reducing the vector into a scalar using the operation kind associated w...
Include the generated interface declarations.
llvm::DenseSet< ValueT, ValueInfoT > DenseSet
Value matchReduction(ArrayRef< BlockArgument > iterCarriedArgs, unsigned redPos, SmallVectorImpl< Operation * > &combinerOps)
Utility to match a generic reduction given a list of iteration-carried arguments, iterCarriedArgs and...
llvm::DenseMap< KeyT, ValueT, KeyInfoT, BucketT > DenseMap
Contains the vectorization state and related methods used across the vectorization process of a given...
VectorizationState(RewriterBase &rewriter)
This represents an operation in an abstracted form, suitable for use with the builder APIs.
Attribute propertiesAttr
This Attribute is used to opaquely construct the properties of the operation.