26#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/SmallVector.h"
28#include "llvm/ADT/TypeSwitch.h"
29#include "llvm/Support/MathExtras.h"
44 if (
auto mod = llvm::dyn_cast<ModuleOp>(current)) {
45 if (mod.getDataLayoutSpec())
47 }
else if (
auto dataLayoutOp =
48 llvm::dyn_cast<DataLayoutOpInterface>(current)) {
50 if (dataLayoutOp.getDataLayoutSpec())
59 if (
auto mod = llvm::dyn_cast<ModuleOp>(op))
78 for (
auto val : output)
79 resultTypes.push_back(val.getType());
81 ComputeRegionOp::create(rewriter, loc, resultTypes, launchArgs, inputArgs,
82 stream, origin, kernelFuncName, kernelModuleName);
84 assert(!regionToClone.
getBlocks().empty() &&
85 "empty region for acc.compute_region");
88 ValueRange mapKeys = inputArgsToMap.empty() ? inputArgs : inputArgsToMap;
89 assert(mapKeys.size() == inputArgs.size() &&
90 "inputArgsToMap must have same size as inputArgs when provided");
94 for (
size_t i = 0; i < launchArgs.size(); ++i)
96 for (
Value input : inputArgs)
98 for (
size_t i = 0; i < inputArgs.size(); ++i)
99 mapping.
map(mapKeys[i], entryBlock->
getArgument(launchArgs.size() + i));
101 if (regionToClone.
getBlocks().size() == 1) {
105 rewriter.
clone(op, mapping);
108 for (
auto val : output)
109 yieldOperands.push_back(mapping.
lookup(val));
111 YieldOp::create(rewriter, loc, yieldOperands);
114 regionToClone, mapping, loc, rewriter);
116 rewriter.
eraseOp(computeRegion);
120 llvm::to_vector(exeRegion.getOps<scf::YieldOp>()));
121 assert(!yieldOps.empty() &&
122 "multi-block region must contain at least one scf.yield");
123 assert(llvm::all_of(yieldOps,
124 [&output](scf::YieldOp yieldOp) {
125 return yieldOp.getNumOperands() ==
126 static_cast<int64_t>(output.size()) &&
128 llvm::zip(yieldOp.getOperands(), output),
130 return std::get<0>(pair).getType() ==
131 std::get<1>(pair).getType();
134 "each scf.yield operand count and types must match output");
136 YieldOp::create(rewriter, loc, exeRegion.getResults());
139 return computeRegion;
144 GPUParallelDimAttr parDim) {
145 return llvm::lower_bound(
147 [](
const GPUParallelDimAttr &
lhs,
const GPUParallelDimAttr &
rhs) {
148 return lhs.getOrder() >
rhs.getOrder();
153 GPUParallelDimAttr parDim) {
155 if (lb == parDims.end() || *lb != parDim)
156 parDims.insert(lb, parDim);
160 GPUParallelDimAttr parDim) {
162 if (lb != parDims.end() && *lb == parDim)
166#define ACC_OP_WITH_PAR_DIMS_LIST \
167 PrivatizeOp, ReductionAccumulateOp, ReductionAccumulateArrayOp
172 [](
auto parOp) {
return parOp.getParDimsAttr(); })
173 .Default([](
Operation *op) -> GPUParallelDimsAttr {
175 GPUParallelDimsAttr parDimsAttr = dyn_cast<GPUParallelDimsAttr>(attr);
176 assert(parDimsAttr &&
"acc.par_dims must be a GPUParallelDimsAttr");
187 return parDimsAttr.isSeq();
192 assert(!
hasParDimsAttr(op) &&
"parallel dimensions attribute is already set");
195 [&](
auto parOp) { parOp.setParDimsAttr(attr); })
202 "expected parallel dimensions attribute to already be set");
205 [&](
auto parOp) { parOp.setParDimsAttr(attr); })
210#undef ACC_OP_WITH_PAR_DIMS_LIST
213 return op->
hasAttrOfType<GPUBlockRedundantAttr>(GPUBlockRedundantAttr::name);
217 op->
setAttr(GPUBlockRedundantAttr::name,
218 GPUBlockRedundantAttr::get(op->
getContext()));
223 "expected parallel dimensions attribute to already be set");
228 assert(alignment > 0 && llvm::isPowerOf2_64(alignment) &&
229 "alignment must be a power of two");
230 return (offset + alignment - 1) & ~(alignment - 1);
235 if (aligned + bytes > maxTotalBytes_) {
238 bytesUsed_ = aligned + bytes;
244 region.
walk([&](GPUSharedMemoryOp op) {
245 int64_t upperBound = op.getStaticUpperBoundBytes();
252 ComputeRegionOp computeRegion) {
253 Value value = privateLocal.getPrivatized();
254 if (
BlockArgument blockArg = dyn_cast<BlockArgument>(value)) {
255 auto owner = dyn_cast<ComputeRegionOp>(blockArg.getOwner()->getParentOp());
256 value = (owner ? owner : computeRegion).getOperand(blockArg);
258 PrivatizeOp privatizeOp = value.
getDefiningOp<PrivatizeOp>();
259 assert(privatizeOp &&
"expected privatize op to be the defining op");
264 if (GPUParallelDimsAttr parDimsAttr = privatize.getParDimsAttr())
265 return llvm::any_of(parDimsAttr.getArray(),
266 [](GPUParallelDimAttr d) { return d.isThreadX(); });
271 auto memrefTy = cast<PointerLikeType>(baseTy).getAsMemRefType(module);
272 assert(memrefTy &&
"private base type must be convertible to memref");
278 ComputeRegionOp computeRegion) {
283 auto parentLoop = privateLocal->getParentOfType<scf::ParallelOp>();
284 while (parentLoop && computeRegion->isProperAncestor(parentLoop)) {
286 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
288 parentLoop = parentLoop->getParentOfType<scf::ParallelOp>();
290 if (GPUParallelDimsAttr parDimsAttr =
getParDimsAttr(computeRegion))
291 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
293 if (parDims.empty()) {
294 for (GPUParallelDimAttr parDim : computeRegion.getLaunchParDims()) {
295 if (parDim.isAnyBlock())
301 if (
auto accumulateOp = dyn_cast<ReductionAccumulateOp>(user)) {
302 if (accumulateOp.getMemref() == privateLocal.getResult())
303 for (GPUParallelDimAttr parDim : accumulateOp.getParDims().getArray())
306 if (
auto combineOp = dyn_cast<ReductionCombineOp>(user)) {
307 if (combineOp.getSrcMemref() == privateLocal.getResult())
311 if (
auto combineRegionOp = dyn_cast<ReductionCombineRegionOp>(user)) {
312 if (combineRegionOp.getSrcVar() == privateLocal.getResult())
313 for (GPUParallelDimAttr parDim :
322 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion,
324 if (!isWorkerPrivate)
325 return std::optional<int64_t>(1);
327 GPUParallelDimAttr threadY =
328 GPUParallelDimAttr::threadYDim(privateLocal.getContext());
329 std::optional<Value> workerArg = computeRegion.getKnownLaunchArg(threadY);
331 return std::optional<int64_t>();
335 return std::optional<int64_t>(workerArgConst.value());
337 FailureOr<int64_t> workerArgBound =
340 if (succeeded(workerArgBound))
341 return std::optional<int64_t>(*workerArgBound);
345 "worker-private variables in shared memory "
346 "require compile-time constant num_workers");
349 return std::optional<int64_t>();
358 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
366 bool isReductionAccumulator =
367 llvm::any_of(privateLocal.getResult().getUsers(), [](
Operation *user) {
368 return isa<ReductionAccumulateOp>(user);
374 llvm::any_of(parDims, [&](
auto parDim) {
return policy.
isGang(parDim); });
375 bool isWorkerPrivate = llvm::any_of(
376 parDims, [&](
auto parDim) {
return policy.
isWorker(parDim); });
377 bool isVectorPrivate = llvm::any_of(
378 parDims, [&](
auto parDim) {
return policy.
isVector(parDim); });
381 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
384 bool isBlockLevelPrivate =
387 (isWorkerPrivate && baseTy.getRank() > 0 && !isReductionAccumulator));
388 if (!isBlockLevelPrivate)
391 for (
int64_t dim : baseTy.getShape())
392 if (dim == ShapedType::kDynamic)
395 auto resultMemRefTy = dyn_cast<MemRefType>(privateLocal.getType());
396 if (!resultMemRefTy || !resultMemRefTy.getLayout().isIdentity() ||
397 resultMemRefTy.getMemorySpace())
400 if (isGangPrivate && isWorkerPrivate && !isReductionAccumulator)
403 FailureOr<std::optional<int64_t>> numCopies =
405 isWorkerPrivate, support);
406 if (failed(numCopies))
408 return numCopies->has_value();
412 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
415 privateLocal, computeRegion, module, policy);
416 if (failed(isCandidate) || !*isCandidate)
421 bool isWorkerPrivate = llvm::any_of(
422 parDims, [&](
auto parDim) {
return policy.
isWorker(parDim); });
424 FailureOr<std::optional<int64_t>> numCopies =
426 privateLocal, computeRegion, isWorkerPrivate,
nullptr);
427 if (failed(numCopies) || !numCopies->has_value())
431 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
433 std::optional<TypeSizeAndAlignment> elementSizeAndAlignment =
435 if (!elementSizeAndAlignment)
439 for (
int64_t dim : baseTy.getShape())
441 return elementSizeAndAlignment->first.getFixedValue() * numElements *
#define ACC_OP_WITH_PAR_DIMS_LIST
Attributes are known-constant values of operations.
This class represents an argument of a Block.
Block represents an ordered list of Operations.
BlockArgument getArgument(unsigned i)
OpListType & getOperations()
BlockArgument addArgument(Type type, Location loc)
Add one value to the argument list.
The main mechanism for performing data layout queries.
A symbol reference with a reference path containing a single element.
This is a utility class for mapping one set of IR entities to another.
auto lookup(T from) const
Lookup a mapped value within the map.
void map(Value from, Value to)
Inserts a new mapping for 'from' to 'to'.
This class defines the main interface for locations in MLIR and acts as a non-nullable wrapper around...
RAII guard to reset the insertion point of the builder when destroyed.
Block * createBlock(Region *parent, Region::iterator insertPt={}, TypeRange argTypes={}, ArrayRef< Location > locs={})
Add new block with 'argTypes' arguments and set the insertion point to the end of it.
Operation * clone(Operation &op, IRMapping &mapper)
Creates a deep copy of the specified operation, remapping any operands that use values outside of the...
void setInsertionPointToStart(Block *block)
Sets the insertion point to the start of the specified block.
void setInsertionPointToEnd(Block *block)
Sets the insertion point to the end of the specified block.
This class provides the API for ops that are known to be terminators.
Operation is the basic unit of execution within MLIR.
Attribute getAttr(StringAttr name)
Return the specified attribute if present, null otherwise.
bool hasAttrOfType(NameT &&name)
OpResult getResult(unsigned idx)
Get the 'idx'th result of this operation.
Operation * getParentOp()
Returns the closest surrounding operation that contains this operation or nullptr if this is a top-le...
OpTy getParentOfType()
Return the closest surrounding parent operation that is of type 'OpTy'.
void setAttr(StringAttr name, Attribute value)
If the an attribute exists with the specified name, change it to the new value.
MLIRContext * getContext()
Return the context this operation is associated with.
This class contains a list of basic blocks and a link to the parent operation it is attached to.
BlockListType & getBlocks()
RetT walk(FnT &&callback)
Walk all nested operations, blocks or regions (including this region), depending on the type of callb...
This class coordinates the application of a rewrite on a set of IR, providing a way for clients to tr...
virtual void eraseOp(Operation *op)
This method erases an operation that is known to have no uses.
Instances of the Type class are uniqued, have an immutable identifier and an optional mutable compone...
static FailureOr< int64_t > computeConstantBound(presburger::BoundType type, const Variable &var, const StopConditionFn &stopCondition=nullptr, ValueBoundsOptions options={})
Compute a constant bound for the given variable.
This class provides an abstraction over the different types of ranges over Values.
This class represents an instance of an SSA value in the MLIR system, representing a computable value...
user_range getUsers() const
Operation * getDefiningOp() const
If this value is the result of an operation, return the operation that defines it.
virtual bool isWorker(ParDimAttrT attr) const =0
Check if the attribute represents worker parallelism.
virtual bool isVector(ParDimAttrT attr) const =0
Check if the attribute represents vector parallelism.
virtual bool isGang(ParDimAttrT attr) const =0
Check if the attribute represents gang parallelism (any gang dimension).
InFlightDiagnostic emitNYI(Location loc, const Twine &message)
Report a case that is not yet supported by the implementation.
bool tryAllocate(int64_t bytes, int64_t alignment=kDefaultAlignmentBytes)
Reserve bytes, rounding the current offset up to alignment first.
static int64_t alignOffset(int64_t offset, int64_t alignment=kDefaultAlignmentBytes)
Round offset up to the next multiple of alignment, which must be a power of two.
Specialization of arith.constant op that returns an integer of index type.
GPUParallelDimsAttr getParDimsAttr(Operation *op)
Obtain the parallel dimensions carried by op, if any.
std::optional< DataLayout > getDataLayout(Operation *op, bool allowDefault=true)
Get the data layout for an operation.
MemRefType getPrivateBaseMemRefType(Type baseTy, ModuleOp module)
Returns the ranked MemRef type used to allocate privatized storage.
SmallVector< GPUParallelDimAttr > getReductionCombineParDims(ReductionCombineOp op)
Returns the parallel dimensions that participate in op's combine step.
void insertParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Insert parDim into parDims while preserving dimension ordering.
bool hasParDimsAttr(Operation *op)
Return whether op carries parallel dimensions.
ComputeRegionOp buildComputeRegion(Location loc, ValueRange launchArgs, ValueRange inputArgs, llvm::StringRef origin, Region ®ionToClone, RewriterBase &rewriter, IRMapping &mapping, ValueRange output={}, FlatSymbolRefAttr kernelFuncName={}, FlatSymbolRefAttr kernelModuleName={}, Value stream={}, ValueRange inputArgsToMap={})
Build an acc.compute_region operation by cloning a source region.
void setGPUBlockRedundantAttr(Operation *op)
Mark op with the acc.gpu_block_redundant attribute.
static FailureOr< std::optional< int64_t > > getWorkerPrivateSharedMemoryNumCopies(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, bool isWorkerPrivate, OpenACCSupport *support)
bool isSpecializedAccRoutine(mlir::Operation *op)
Used to check whether this is a specialized accelerator version of acc routine function.
static bool isInsideACCSpecializedRoutine(Operation *op)
std::optional< TypeSizeAndAlignment > getTypeSizeAndAlignment(Type ty, ModuleOp module, const DataLayout &dl, OpenACCSupport *support=nullptr)
Returns the size and ABI alignment in bytes.
FailureOr< bool > isPrivateLocalSharedMemoryCandidate(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
True when privateLocal may be placed in shared memory.
int64_t sumExistingSharedMemoryBytes(Region ®ion)
Sum aligned static_upper_bound_bytes for all acc.gpu_shared_memory in region.
scf::ExecuteRegionOp wrapMultiBlockRegionWithSCFExecuteRegion(Region ®ion, IRMapping &mapping, Location loc, RewriterBase &rewriter)
Wrap a multi-block region in an scf.execute_region.
void updateParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Update parallel dimensions on op.
PrivatizeOp getPrivatizeOp(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Resolve the acc.privatize operation associated with a private local.
bool hasSeqParDims(Operation *op)
Return whether op carries sequential parallel dimensions.
void copyParDimsAttr(Operation *from, Operation *to)
Copy parallel dimensions from from to to.
bool hasGPUBlockRedundantAttr(Operation *op)
Return whether op is marked with the acc.gpu_block_redundant attribute, i.e.
void removeParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Remove parDim from parDims if present.
void setParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Set parallel dimensions on op.
std::optional< int64_t > getPrivateLocalSharedMemoryUpperBoundBytes(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
Upper-bound byte size for a shared-memory private_local candidate, or std::nullopt when not eligible ...
static bool isThreadXPrivatize(PrivatizeOp privatize)
static SmallVector< GPUParallelDimAttr >::iterator findParDim(SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
SmallVector< GPUParallelDimAttr > collectPrivateLocalParDims(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Collect parallel dimensions that govern privatization of privateLocal.
ACCParMappingPolicy< mlir::acc::GPUParallelDimAttr > ACCToGPUMappingPolicy
Type alias for the GPU-specific mapping policy.
Include the generated interface declarations.