26#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/SmallVector.h"
28#include "llvm/ADT/TypeSwitch.h"
29#include "llvm/Support/MathExtras.h"
44 if (
auto mod = llvm::dyn_cast<ModuleOp>(current)) {
45 if (mod.getDataLayoutSpec())
47 }
else if (
auto dataLayoutOp =
48 llvm::dyn_cast<DataLayoutOpInterface>(current)) {
50 if (dataLayoutOp.getDataLayoutSpec())
59 if (
auto mod = llvm::dyn_cast<ModuleOp>(op))
78 for (
auto val : output)
79 resultTypes.push_back(val.getType());
81 ComputeRegionOp::create(rewriter, loc, resultTypes, launchArgs, inputArgs,
82 stream, origin, kernelFuncName, kernelModuleName);
84 assert(!regionToClone.
getBlocks().empty() &&
85 "empty region for acc.compute_region");
88 ValueRange mapKeys = inputArgsToMap.empty() ? inputArgs : inputArgsToMap;
89 assert(mapKeys.size() == inputArgs.size() &&
90 "inputArgsToMap must have same size as inputArgs when provided");
94 for (
size_t i = 0; i < launchArgs.size(); ++i)
96 for (
Value input : inputArgs)
98 for (
size_t i = 0; i < inputArgs.size(); ++i)
99 mapping.
map(mapKeys[i], entryBlock->
getArgument(launchArgs.size() + i));
101 if (regionToClone.
getBlocks().size() == 1) {
105 rewriter.
clone(op, mapping);
108 for (
auto val : output)
109 yieldOperands.push_back(mapping.
lookup(val));
111 YieldOp::create(rewriter, loc, yieldOperands);
114 regionToClone, mapping, loc, rewriter);
116 rewriter.
eraseOp(computeRegion);
120 llvm::to_vector(exeRegion.getOps<scf::YieldOp>()));
121 assert(!yieldOps.empty() &&
122 "multi-block region must contain at least one scf.yield");
123 assert(llvm::all_of(yieldOps,
124 [&output](scf::YieldOp yieldOp) {
125 return yieldOp.getNumOperands() ==
126 static_cast<int64_t>(output.size()) &&
128 llvm::zip(yieldOp.getOperands(), output),
130 return std::get<0>(pair).getType() ==
131 std::get<1>(pair).getType();
134 "each scf.yield operand count and types must match output");
136 YieldOp::create(rewriter, loc, exeRegion.getResults());
139 return computeRegion;
144 GPUParallelDimAttr parDim) {
145 return llvm::lower_bound(
147 [](
const GPUParallelDimAttr &
lhs,
const GPUParallelDimAttr &
rhs) {
148 return lhs.getOrder() >
rhs.getOrder();
153 GPUParallelDimAttr parDim) {
155 if (lb == parDims.end() || *lb != parDim)
156 parDims.insert(lb, parDim);
160 GPUParallelDimAttr parDim) {
162 if (lb != parDims.end() && *lb == parDim)
166#define ACC_OP_WITH_PAR_DIMS_LIST \
167 PrivatizeOp, ReductionAccumulateOp, ReductionAccumulateArrayOp, \
173 [](
auto parOp) {
return parOp.getParDimsAttr(); })
174 .Default([](
Operation *op) -> GPUParallelDimsAttr {
176 GPUParallelDimsAttr parDimsAttr = dyn_cast<GPUParallelDimsAttr>(attr);
177 assert(parDimsAttr &&
"acc.par_dims must be a GPUParallelDimsAttr");
188 return parDimsAttr.isSeq();
193 assert(!
hasParDimsAttr(op) &&
"parallel dimensions attribute is already set");
196 [&](
auto parOp) { parOp.setParDimsAttr(attr); })
203 "expected parallel dimensions attribute to already be set");
206 [&](
auto parOp) { parOp.setParDimsAttr(attr); })
211#undef ACC_OP_WITH_PAR_DIMS_LIST
214 return op->
hasAttrOfType<GPUBlockRedundantAttr>(GPUBlockRedundantAttr::name);
218 op->
setAttr(GPUBlockRedundantAttr::name,
219 GPUBlockRedundantAttr::get(op->
getContext()));
224 "expected parallel dimensions attribute to already be set");
229 return op->
getAttrOfType<ActiveParDimsAttr>(ActiveParDimsAttr::name);
237 op->
setAttr(ActiveParDimsAttr::name, attr);
245 assert(alignment > 0 && llvm::isPowerOf2_64(alignment) &&
246 "alignment must be a power of two");
247 return (offset + alignment - 1) & ~(alignment - 1);
252 if (aligned + bytes > maxTotalBytes_) {
255 bytesUsed_ = aligned + bytes;
261 region.
walk([&](GPUSharedMemoryOp op) {
262 int64_t upperBound = op.getStaticUpperBoundBytes();
269 ComputeRegionOp computeRegion) {
270 Value value = privateLocal.getPrivatized();
271 if (
BlockArgument blockArg = dyn_cast<BlockArgument>(value)) {
272 auto owner = dyn_cast<ComputeRegionOp>(blockArg.getOwner()->getParentOp());
273 value = (owner ? owner : computeRegion).getOperand(blockArg);
275 PrivatizeOp privatizeOp = value.
getDefiningOp<PrivatizeOp>();
276 assert(privatizeOp &&
"expected privatize op to be the defining op");
281 if (GPUParallelDimsAttr parDimsAttr = privatize.getParDimsAttr())
282 return llvm::any_of(parDimsAttr.getArray(),
283 [](GPUParallelDimAttr d) { return d.isThreadX(); });
288 auto memrefTy = cast<PointerLikeType>(baseTy).getAsMemRefType(module);
289 assert(memrefTy &&
"private base type must be convertible to memref");
295 ComputeRegionOp computeRegion) {
300 auto parentLoop = privateLocal->getParentOfType<scf::ParallelOp>();
301 while (parentLoop && computeRegion->isProperAncestor(parentLoop)) {
303 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
305 parentLoop = parentLoop->getParentOfType<scf::ParallelOp>();
307 if (GPUParallelDimsAttr parDimsAttr =
getParDimsAttr(computeRegion))
308 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
310 if (parDims.empty()) {
311 for (GPUParallelDimAttr parDim : computeRegion.getLaunchParDims()) {
312 if (parDim.isAnyBlock())
318 if (
auto accumulateOp = dyn_cast<ReductionAccumulateOp>(user)) {
319 if (accumulateOp.getMemref() == privateLocal.getResult())
320 for (GPUParallelDimAttr parDim : accumulateOp.getParDims().getArray())
323 if (
auto combineOp = dyn_cast<ReductionCombineOp>(user)) {
324 if (combineOp.getSrcMemref() == privateLocal.getResult())
328 if (
auto combineRegionOp = dyn_cast<ReductionCombineRegionOp>(user)) {
329 if (combineRegionOp.getSrcVar() == privateLocal.getResult())
330 for (GPUParallelDimAttr parDim :
339 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion,
341 if (!isWorkerPrivate)
342 return std::optional<int64_t>(1);
344 GPUParallelDimAttr threadY =
345 GPUParallelDimAttr::threadYDim(privateLocal.getContext());
346 std::optional<Value> workerArg = computeRegion.getKnownLaunchArg(threadY);
348 return std::optional<int64_t>();
352 return std::optional<int64_t>(workerArgConst.value());
354 FailureOr<int64_t> workerArgBound =
357 if (succeeded(workerArgBound))
358 return std::optional<int64_t>(*workerArgBound);
362 "worker-private variables in shared memory "
363 "require compile-time constant num_workers");
366 return std::optional<int64_t>();
375 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
383 bool isReductionAccumulator =
384 llvm::any_of(privateLocal.getResult().getUsers(), [](
Operation *user) {
385 return isa<ReductionAccumulateOp>(user);
391 llvm::any_of(parDims, [&](
auto parDim) {
return policy.
isGang(parDim); });
392 bool isWorkerPrivate = llvm::any_of(
393 parDims, [&](
auto parDim) {
return policy.
isWorker(parDim); });
394 bool isVectorPrivate = llvm::any_of(
395 parDims, [&](
auto parDim) {
return policy.
isVector(parDim); });
398 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
401 bool isBlockLevelPrivate =
404 (isWorkerPrivate && baseTy.getRank() > 0 && !isReductionAccumulator));
405 if (!isBlockLevelPrivate)
408 for (
int64_t dim : baseTy.getShape())
409 if (dim == ShapedType::kDynamic)
412 auto resultMemRefTy = dyn_cast<MemRefType>(privateLocal.getType());
413 if (!resultMemRefTy || !resultMemRefTy.getLayout().isIdentity() ||
414 resultMemRefTy.getMemorySpace())
417 if (isGangPrivate && isWorkerPrivate && !isReductionAccumulator)
420 FailureOr<std::optional<int64_t>> numCopies =
422 isWorkerPrivate, support);
423 if (failed(numCopies))
425 return numCopies->has_value();
429 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
432 privateLocal, computeRegion, module, policy);
433 if (failed(isCandidate) || !*isCandidate)
438 bool isWorkerPrivate = llvm::any_of(
439 parDims, [&](
auto parDim) {
return policy.
isWorker(parDim); });
441 FailureOr<std::optional<int64_t>> numCopies =
443 privateLocal, computeRegion, isWorkerPrivate,
nullptr);
444 if (failed(numCopies) || !numCopies->has_value())
448 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
450 std::optional<TypeSizeAndAlignment> elementSizeAndAlignment =
452 if (!elementSizeAndAlignment)
456 for (
int64_t dim : baseTy.getShape())
458 return elementSizeAndAlignment->first.getFixedValue() * numElements *
#define ACC_OP_WITH_PAR_DIMS_LIST
Attributes are known-constant values of operations.
This class represents an argument of a Block.
Block represents an ordered list of Operations.
BlockArgument getArgument(unsigned i)
OpListType & getOperations()
BlockArgument addArgument(Type type, Location loc)
Add one value to the argument list.
The main mechanism for performing data layout queries.
A symbol reference with a reference path containing a single element.
This is a utility class for mapping one set of IR entities to another.
auto lookup(T from) const
Lookup a mapped value within the map.
void map(Value from, Value to)
Inserts a new mapping for 'from' to 'to'.
This class defines the main interface for locations in MLIR and acts as a non-nullable wrapper around...
RAII guard to reset the insertion point of the builder when destroyed.
Block * createBlock(Region *parent, Region::iterator insertPt={}, TypeRange argTypes={}, ArrayRef< Location > locs={})
Add new block with 'argTypes' arguments and set the insertion point to the end of it.
Operation * clone(Operation &op, IRMapping &mapper)
Creates a deep copy of the specified operation, remapping any operands that use values outside of the...
void setInsertionPointToStart(Block *block)
Sets the insertion point to the start of the specified block.
void setInsertionPointToEnd(Block *block)
Sets the insertion point to the end of the specified block.
This class provides the API for ops that are known to be terminators.
Operation is the basic unit of execution within MLIR.
AttrClass getAttrOfType(StringAttr name)
Attribute getAttr(StringAttr name)
Return the specified attribute if present, null otherwise.
bool hasAttrOfType(NameT &&name)
OpResult getResult(unsigned idx)
Get the 'idx'th result of this operation.
Operation * getParentOp()
Returns the closest surrounding operation that contains this operation or nullptr if this is a top-le...
OpTy getParentOfType()
Return the closest surrounding parent operation that is of type 'OpTy'.
void setAttr(StringAttr name, Attribute value)
If the an attribute exists with the specified name, change it to the new value.
MLIRContext * getContext()
Return the context this operation is associated with.
This class contains a list of basic blocks and a link to the parent operation it is attached to.
BlockListType & getBlocks()
RetT walk(FnT &&callback)
Walk all nested operations, blocks or regions (including this region), depending on the type of callb...
This class coordinates the application of a rewrite on a set of IR, providing a way for clients to tr...
virtual void eraseOp(Operation *op)
This method erases an operation that is known to have no uses.
Instances of the Type class are uniqued, have an immutable identifier and an optional mutable compone...
static FailureOr< int64_t > computeConstantBound(presburger::BoundType type, const Variable &var, const StopConditionFn &stopCondition=nullptr, ValueBoundsOptions options={})
Compute a constant bound for the given variable.
This class provides an abstraction over the different types of ranges over Values.
This class represents an instance of an SSA value in the MLIR system, representing a computable value...
user_range getUsers() const
Operation * getDefiningOp() const
If this value is the result of an operation, return the operation that defines it.
virtual bool isWorker(ParDimAttrT attr) const =0
Check if the attribute represents worker parallelism.
virtual bool isVector(ParDimAttrT attr) const =0
Check if the attribute represents vector parallelism.
virtual bool isGang(ParDimAttrT attr) const =0
Check if the attribute represents gang parallelism (any gang dimension).
InFlightDiagnostic emitNYI(Location loc, const Twine &message)
Report a case that is not yet supported by the implementation.
bool tryAllocate(int64_t bytes, int64_t alignment=kDefaultAlignmentBytes)
Reserve bytes, rounding the current offset up to alignment first.
static int64_t alignOffset(int64_t offset, int64_t alignment=kDefaultAlignmentBytes)
Round offset up to the next multiple of alignment, which must be a power of two.
Specialization of arith.constant op that returns an integer of index type.
GPUParallelDimsAttr getParDimsAttr(Operation *op)
Obtain the parallel dimensions carried by op, if any.
std::optional< DataLayout > getDataLayout(Operation *op, bool allowDefault=true)
Get the data layout for an operation.
MemRefType getPrivateBaseMemRefType(Type baseTy, ModuleOp module)
Returns the ranked MemRef type used to allocate privatized storage.
SmallVector< GPUParallelDimAttr > getReductionCombineParDims(ReductionCombineOp op)
Returns the parallel dimensions that participate in op's combine step.
void setActiveParDimsAttr(Operation *op, ActiveParDimsAttr attr)
Set active parallel dimensions on op.
void insertParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Insert parDim into parDims while preserving dimension ordering.
bool hasActiveParDimsAttr(Operation *op)
Return whether op carries active parallel dimensions.
bool hasParDimsAttr(Operation *op)
Return whether op carries parallel dimensions.
ComputeRegionOp buildComputeRegion(Location loc, ValueRange launchArgs, ValueRange inputArgs, llvm::StringRef origin, Region ®ionToClone, RewriterBase &rewriter, IRMapping &mapping, ValueRange output={}, FlatSymbolRefAttr kernelFuncName={}, FlatSymbolRefAttr kernelModuleName={}, Value stream={}, ValueRange inputArgsToMap={})
Build an acc.compute_region operation by cloning a source region.
void setGPUBlockRedundantAttr(Operation *op)
Mark op with the acc.gpu_block_redundant attribute.
static FailureOr< std::optional< int64_t > > getWorkerPrivateSharedMemoryNumCopies(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, bool isWorkerPrivate, OpenACCSupport *support)
bool isSpecializedAccRoutine(mlir::Operation *op)
Used to check whether this is a specialized accelerator version of acc routine function.
static bool isInsideACCSpecializedRoutine(Operation *op)
std::optional< TypeSizeAndAlignment > getTypeSizeAndAlignment(Type ty, ModuleOp module, const DataLayout &dl, OpenACCSupport *support=nullptr)
Returns the size and ABI alignment in bytes.
FailureOr< bool > isPrivateLocalSharedMemoryCandidate(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
True when privateLocal may be placed in shared memory.
int64_t sumExistingSharedMemoryBytes(Region ®ion)
Sum aligned static_upper_bound_bytes for all acc.gpu_shared_memory in region.
scf::ExecuteRegionOp wrapMultiBlockRegionWithSCFExecuteRegion(Region ®ion, IRMapping &mapping, Location loc, RewriterBase &rewriter)
Wrap a multi-block region in an scf.execute_region.
void updateParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Update parallel dimensions on op.
PrivatizeOp getPrivatizeOp(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Resolve the acc.privatize operation associated with a private local.
bool hasSeqParDims(Operation *op)
Return whether op carries sequential parallel dimensions.
void copyParDimsAttr(Operation *from, Operation *to)
Copy parallel dimensions from from to to.
bool hasGPUBlockRedundantAttr(Operation *op)
Return whether op is marked with the acc.gpu_block_redundant attribute, i.e.
void removeParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Remove parDim from parDims if present.
void setParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Set parallel dimensions on op.
ActiveParDimsAttr getActiveParDimsAttr(Operation *op)
Obtain the active parallel dimensions carried by op, if any.
std::optional< int64_t > getPrivateLocalSharedMemoryUpperBoundBytes(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
Upper-bound byte size for a shared-memory private_local candidate, or std::nullopt when not eligible ...
static bool isThreadXPrivatize(PrivatizeOp privatize)
static SmallVector< GPUParallelDimAttr >::iterator findParDim(SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
SmallVector< GPUParallelDimAttr > collectPrivateLocalParDims(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Collect parallel dimensions that govern privatization of privateLocal.
ACCParMappingPolicy< mlir::acc::GPUParallelDimAttr > ACCToGPUMappingPolicy
Type alias for the GPU-specific mapping policy.
Include the generated interface declarations.