26#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/SmallVector.h"
28#include "llvm/ADT/TypeSwitch.h"
29#include "llvm/Support/MathExtras.h"
44 if (
auto mod = llvm::dyn_cast<ModuleOp>(current)) {
45 if (mod.getDataLayoutSpec())
47 }
else if (
auto dataLayoutOp =
48 llvm::dyn_cast<DataLayoutOpInterface>(current)) {
50 if (dataLayoutOp.getDataLayoutSpec())
59 if (
auto mod = llvm::dyn_cast<ModuleOp>(op))
78 for (
auto val : output)
79 resultTypes.push_back(val.getType());
81 ComputeRegionOp::create(rewriter, loc, resultTypes, launchArgs, inputArgs,
82 stream, origin, kernelFuncName, kernelModuleName);
84 assert(!regionToClone.
getBlocks().empty() &&
85 "empty region for acc.compute_region");
88 ValueRange mapKeys = inputArgsToMap.empty() ? inputArgs : inputArgsToMap;
89 assert(mapKeys.size() == inputArgs.size() &&
90 "inputArgsToMap must have same size as inputArgs when provided");
94 for (
size_t i = 0; i < launchArgs.size(); ++i)
96 for (
Value input : inputArgs)
98 for (
size_t i = 0; i < inputArgs.size(); ++i)
99 mapping.
map(mapKeys[i], entryBlock->
getArgument(launchArgs.size() + i));
101 if (regionToClone.
getBlocks().size() == 1) {
105 rewriter.
clone(op, mapping);
108 for (
auto val : output)
109 yieldOperands.push_back(mapping.
lookup(val));
111 YieldOp::create(rewriter, loc, yieldOperands);
114 regionToClone, mapping, loc, rewriter);
116 rewriter.
eraseOp(computeRegion);
120 llvm::to_vector(exeRegion.getOps<scf::YieldOp>()));
121 assert(!yieldOps.empty() &&
122 "multi-block region must contain at least one scf.yield");
123 assert(llvm::all_of(yieldOps,
124 [&output](scf::YieldOp yieldOp) {
125 return yieldOp.getNumOperands() ==
126 static_cast<int64_t>(output.size()) &&
128 llvm::zip(yieldOp.getOperands(), output),
130 return std::get<0>(pair).getType() ==
131 std::get<1>(pair).getType();
134 "each scf.yield operand count and types must match output");
136 YieldOp::create(rewriter, loc, exeRegion.getResults());
139 return computeRegion;
144 GPUParallelDimAttr parDim) {
145 return llvm::lower_bound(
147 [](
const GPUParallelDimAttr &
lhs,
const GPUParallelDimAttr &
rhs) {
148 return lhs.getOrder() >
rhs.getOrder();
153 GPUParallelDimAttr parDim) {
155 if (lb == parDims.end() || *lb != parDim)
156 parDims.insert(lb, parDim);
160 GPUParallelDimAttr parDim) {
162 if (lb != parDims.end() && *lb == parDim)
166#define ACC_OP_WITH_PAR_DIMS_LIST \
167 PrivatizeOp, ReductionAccumulateOp, ReductionAccumulateArrayOp
172 [](
auto parOp) {
return parOp.getParDimsAttr(); })
173 .Default([](
Operation *op) -> GPUParallelDimsAttr {
175 GPUParallelDimsAttr parDimsAttr = dyn_cast<GPUParallelDimsAttr>(attr);
176 assert(parDimsAttr &&
"acc.par_dims must be a GPUParallelDimsAttr");
187 return parDimsAttr.isSeq();
192 assert(!
hasParDimsAttr(op) &&
"parallel dimensions attribute is already set");
195 [&](
auto parOp) { parOp.setParDimsAttr(attr); })
202 "expected parallel dimensions attribute to already be set");
205 [&](
auto parOp) { parOp.setParDimsAttr(attr); })
210#undef ACC_OP_WITH_PAR_DIMS_LIST
213 return op->
hasAttrOfType<GPUBlockRedundantAttr>(GPUBlockRedundantAttr::name);
217 op->
setAttr(GPUBlockRedundantAttr::name,
218 GPUBlockRedundantAttr::get(op->
getContext()));
223 "expected parallel dimensions attribute to already be set");
228 return op->
getAttrOfType<ActiveParDimsAttr>(ActiveParDimsAttr::name);
236 op->
setAttr(ActiveParDimsAttr::name, attr);
244 assert(alignment > 0 && llvm::isPowerOf2_64(alignment) &&
245 "alignment must be a power of two");
246 return (offset + alignment - 1) & ~(alignment - 1);
251 if (aligned + bytes > maxTotalBytes_) {
254 bytesUsed_ = aligned + bytes;
260 region.
walk([&](GPUSharedMemoryOp op) {
261 int64_t upperBound = op.getStaticUpperBoundBytes();
268 ComputeRegionOp computeRegion) {
269 Value value = privateLocal.getPrivatized();
270 if (
BlockArgument blockArg = dyn_cast<BlockArgument>(value)) {
271 auto owner = dyn_cast<ComputeRegionOp>(blockArg.getOwner()->getParentOp());
272 value = (owner ? owner : computeRegion).getOperand(blockArg);
274 PrivatizeOp privatizeOp = value.
getDefiningOp<PrivatizeOp>();
275 assert(privatizeOp &&
"expected privatize op to be the defining op");
280 if (GPUParallelDimsAttr parDimsAttr = privatize.getParDimsAttr())
281 return llvm::any_of(parDimsAttr.getArray(),
282 [](GPUParallelDimAttr d) { return d.isThreadX(); });
287 auto memrefTy = cast<PointerLikeType>(baseTy).getAsMemRefType(module);
288 assert(memrefTy &&
"private base type must be convertible to memref");
294 ComputeRegionOp computeRegion) {
299 auto parentLoop = privateLocal->getParentOfType<scf::ParallelOp>();
300 while (parentLoop && computeRegion->isProperAncestor(parentLoop)) {
302 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
304 parentLoop = parentLoop->getParentOfType<scf::ParallelOp>();
306 if (GPUParallelDimsAttr parDimsAttr =
getParDimsAttr(computeRegion))
307 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
309 if (parDims.empty()) {
310 for (GPUParallelDimAttr parDim : computeRegion.getLaunchParDims()) {
311 if (parDim.isAnyBlock())
317 if (
auto accumulateOp = dyn_cast<ReductionAccumulateOp>(user)) {
318 if (accumulateOp.getMemref() == privateLocal.getResult())
319 for (GPUParallelDimAttr parDim : accumulateOp.getParDims().getArray())
322 if (
auto combineOp = dyn_cast<ReductionCombineOp>(user)) {
323 if (combineOp.getSrcMemref() == privateLocal.getResult())
327 if (
auto combineRegionOp = dyn_cast<ReductionCombineRegionOp>(user)) {
328 if (combineRegionOp.getSrcVar() == privateLocal.getResult())
329 for (GPUParallelDimAttr parDim :
338 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion,
340 if (!isWorkerPrivate)
341 return std::optional<int64_t>(1);
343 GPUParallelDimAttr threadY =
344 GPUParallelDimAttr::threadYDim(privateLocal.getContext());
345 std::optional<Value> workerArg = computeRegion.getKnownLaunchArg(threadY);
347 return std::optional<int64_t>();
351 return std::optional<int64_t>(workerArgConst.value());
353 FailureOr<int64_t> workerArgBound =
356 if (succeeded(workerArgBound))
357 return std::optional<int64_t>(*workerArgBound);
361 "worker-private variables in shared memory "
362 "require compile-time constant num_workers");
365 return std::optional<int64_t>();
374 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
382 bool isReductionAccumulator =
383 llvm::any_of(privateLocal.getResult().getUsers(), [](
Operation *user) {
384 return isa<ReductionAccumulateOp>(user);
390 llvm::any_of(parDims, [&](
auto parDim) {
return policy.
isGang(parDim); });
391 bool isWorkerPrivate = llvm::any_of(
392 parDims, [&](
auto parDim) {
return policy.
isWorker(parDim); });
393 bool isVectorPrivate = llvm::any_of(
394 parDims, [&](
auto parDim) {
return policy.
isVector(parDim); });
397 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
400 bool isBlockLevelPrivate =
403 (isWorkerPrivate && baseTy.getRank() > 0 && !isReductionAccumulator));
404 if (!isBlockLevelPrivate)
407 for (
int64_t dim : baseTy.getShape())
408 if (dim == ShapedType::kDynamic)
411 auto resultMemRefTy = dyn_cast<MemRefType>(privateLocal.getType());
412 if (!resultMemRefTy || !resultMemRefTy.getLayout().isIdentity() ||
413 resultMemRefTy.getMemorySpace())
416 if (isGangPrivate && isWorkerPrivate && !isReductionAccumulator)
419 FailureOr<std::optional<int64_t>> numCopies =
421 isWorkerPrivate, support);
422 if (failed(numCopies))
424 return numCopies->has_value();
428 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
431 privateLocal, computeRegion, module, policy);
432 if (failed(isCandidate) || !*isCandidate)
437 bool isWorkerPrivate = llvm::any_of(
438 parDims, [&](
auto parDim) {
return policy.
isWorker(parDim); });
440 FailureOr<std::optional<int64_t>> numCopies =
442 privateLocal, computeRegion, isWorkerPrivate,
nullptr);
443 if (failed(numCopies) || !numCopies->has_value())
447 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
449 std::optional<TypeSizeAndAlignment> elementSizeAndAlignment =
451 if (!elementSizeAndAlignment)
455 for (
int64_t dim : baseTy.getShape())
457 return elementSizeAndAlignment->first.getFixedValue() * numElements *
#define ACC_OP_WITH_PAR_DIMS_LIST
Attributes are known-constant values of operations.
This class represents an argument of a Block.
Block represents an ordered list of Operations.
BlockArgument getArgument(unsigned i)
OpListType & getOperations()
BlockArgument addArgument(Type type, Location loc)
Add one value to the argument list.
The main mechanism for performing data layout queries.
A symbol reference with a reference path containing a single element.
This is a utility class for mapping one set of IR entities to another.
auto lookup(T from) const
Lookup a mapped value within the map.
void map(Value from, Value to)
Inserts a new mapping for 'from' to 'to'.
This class defines the main interface for locations in MLIR and acts as a non-nullable wrapper around...
RAII guard to reset the insertion point of the builder when destroyed.
Block * createBlock(Region *parent, Region::iterator insertPt={}, TypeRange argTypes={}, ArrayRef< Location > locs={})
Add new block with 'argTypes' arguments and set the insertion point to the end of it.
Operation * clone(Operation &op, IRMapping &mapper)
Creates a deep copy of the specified operation, remapping any operands that use values outside of the...
void setInsertionPointToStart(Block *block)
Sets the insertion point to the start of the specified block.
void setInsertionPointToEnd(Block *block)
Sets the insertion point to the end of the specified block.
This class provides the API for ops that are known to be terminators.
Operation is the basic unit of execution within MLIR.
AttrClass getAttrOfType(StringAttr name)
Attribute getAttr(StringAttr name)
Return the specified attribute if present, null otherwise.
bool hasAttrOfType(NameT &&name)
OpResult getResult(unsigned idx)
Get the 'idx'th result of this operation.
Operation * getParentOp()
Returns the closest surrounding operation that contains this operation or nullptr if this is a top-le...
OpTy getParentOfType()
Return the closest surrounding parent operation that is of type 'OpTy'.
void setAttr(StringAttr name, Attribute value)
If the an attribute exists with the specified name, change it to the new value.
MLIRContext * getContext()
Return the context this operation is associated with.
This class contains a list of basic blocks and a link to the parent operation it is attached to.
BlockListType & getBlocks()
RetT walk(FnT &&callback)
Walk all nested operations, blocks or regions (including this region), depending on the type of callb...
This class coordinates the application of a rewrite on a set of IR, providing a way for clients to tr...
virtual void eraseOp(Operation *op)
This method erases an operation that is known to have no uses.
Instances of the Type class are uniqued, have an immutable identifier and an optional mutable compone...
static FailureOr< int64_t > computeConstantBound(presburger::BoundType type, const Variable &var, const StopConditionFn &stopCondition=nullptr, ValueBoundsOptions options={})
Compute a constant bound for the given variable.
This class provides an abstraction over the different types of ranges over Values.
This class represents an instance of an SSA value in the MLIR system, representing a computable value...
user_range getUsers() const
Operation * getDefiningOp() const
If this value is the result of an operation, return the operation that defines it.
virtual bool isWorker(ParDimAttrT attr) const =0
Check if the attribute represents worker parallelism.
virtual bool isVector(ParDimAttrT attr) const =0
Check if the attribute represents vector parallelism.
virtual bool isGang(ParDimAttrT attr) const =0
Check if the attribute represents gang parallelism (any gang dimension).
InFlightDiagnostic emitNYI(Location loc, const Twine &message)
Report a case that is not yet supported by the implementation.
bool tryAllocate(int64_t bytes, int64_t alignment=kDefaultAlignmentBytes)
Reserve bytes, rounding the current offset up to alignment first.
static int64_t alignOffset(int64_t offset, int64_t alignment=kDefaultAlignmentBytes)
Round offset up to the next multiple of alignment, which must be a power of two.
Specialization of arith.constant op that returns an integer of index type.
GPUParallelDimsAttr getParDimsAttr(Operation *op)
Obtain the parallel dimensions carried by op, if any.
std::optional< DataLayout > getDataLayout(Operation *op, bool allowDefault=true)
Get the data layout for an operation.
MemRefType getPrivateBaseMemRefType(Type baseTy, ModuleOp module)
Returns the ranked MemRef type used to allocate privatized storage.
SmallVector< GPUParallelDimAttr > getReductionCombineParDims(ReductionCombineOp op)
Returns the parallel dimensions that participate in op's combine step.
void setActiveParDimsAttr(Operation *op, ActiveParDimsAttr attr)
Set active parallel dimensions on op.
void insertParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Insert parDim into parDims while preserving dimension ordering.
bool hasActiveParDimsAttr(Operation *op)
Return whether op carries active parallel dimensions.
bool hasParDimsAttr(Operation *op)
Return whether op carries parallel dimensions.
ComputeRegionOp buildComputeRegion(Location loc, ValueRange launchArgs, ValueRange inputArgs, llvm::StringRef origin, Region ®ionToClone, RewriterBase &rewriter, IRMapping &mapping, ValueRange output={}, FlatSymbolRefAttr kernelFuncName={}, FlatSymbolRefAttr kernelModuleName={}, Value stream={}, ValueRange inputArgsToMap={})
Build an acc.compute_region operation by cloning a source region.
void setGPUBlockRedundantAttr(Operation *op)
Mark op with the acc.gpu_block_redundant attribute.
static FailureOr< std::optional< int64_t > > getWorkerPrivateSharedMemoryNumCopies(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, bool isWorkerPrivate, OpenACCSupport *support)
bool isSpecializedAccRoutine(mlir::Operation *op)
Used to check whether this is a specialized accelerator version of acc routine function.
static bool isInsideACCSpecializedRoutine(Operation *op)
std::optional< TypeSizeAndAlignment > getTypeSizeAndAlignment(Type ty, ModuleOp module, const DataLayout &dl, OpenACCSupport *support=nullptr)
Returns the size and ABI alignment in bytes.
FailureOr< bool > isPrivateLocalSharedMemoryCandidate(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
True when privateLocal may be placed in shared memory.
int64_t sumExistingSharedMemoryBytes(Region ®ion)
Sum aligned static_upper_bound_bytes for all acc.gpu_shared_memory in region.
scf::ExecuteRegionOp wrapMultiBlockRegionWithSCFExecuteRegion(Region ®ion, IRMapping &mapping, Location loc, RewriterBase &rewriter)
Wrap a multi-block region in an scf.execute_region.
void updateParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Update parallel dimensions on op.
PrivatizeOp getPrivatizeOp(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Resolve the acc.privatize operation associated with a private local.
bool hasSeqParDims(Operation *op)
Return whether op carries sequential parallel dimensions.
void copyParDimsAttr(Operation *from, Operation *to)
Copy parallel dimensions from from to to.
bool hasGPUBlockRedundantAttr(Operation *op)
Return whether op is marked with the acc.gpu_block_redundant attribute, i.e.
void removeParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Remove parDim from parDims if present.
void setParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Set parallel dimensions on op.
ActiveParDimsAttr getActiveParDimsAttr(Operation *op)
Obtain the active parallel dimensions carried by op, if any.
std::optional< int64_t > getPrivateLocalSharedMemoryUpperBoundBytes(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
Upper-bound byte size for a shared-memory private_local candidate, or std::nullopt when not eligible ...
static bool isThreadXPrivatize(PrivatizeOp privatize)
static SmallVector< GPUParallelDimAttr >::iterator findParDim(SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
SmallVector< GPUParallelDimAttr > collectPrivateLocalParDims(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Collect parallel dimensions that govern privatization of privateLocal.
ACCParMappingPolicy< mlir::acc::GPUParallelDimAttr > ACCToGPUMappingPolicy
Type alias for the GPU-specific mapping policy.
Include the generated interface declarations.