MLIR 24.0.0git
OpenACCUtilsCG.cpp
Go to the documentation of this file.
1//===- OpenACCUtilsCG.cpp - OpenACC Code Generation Utilities -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements utility functions for OpenACC code generation.
10//
11//===----------------------------------------------------------------------===//
12
14
22#include "mlir/IR/BuiltinOps.h"
23#include "mlir/IR/IRMapping.h"
26#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/SmallVector.h"
28#include "llvm/ADT/TypeSwitch.h"
29#include "llvm/Support/MathExtras.h"
30
31namespace mlir {
32namespace acc {
33
34std::optional<DataLayout> getDataLayout(Operation *op, bool allowDefault) {
35 if (!op)
36 return std::nullopt;
37
38 // Walk up the parent chain to find the nearest operation with an explicit
39 // data layout spec. Check ModuleOp explicitly since it does not actually
40 // implement DataLayoutOpInterface as a trait (it just has the same methods).
41 Operation *current = op;
42 while (current) {
43 // Check for ModuleOp with explicit data layout spec
44 if (auto mod = llvm::dyn_cast<ModuleOp>(current)) {
45 if (mod.getDataLayoutSpec())
46 return DataLayout(mod);
47 } else if (auto dataLayoutOp =
48 llvm::dyn_cast<DataLayoutOpInterface>(current)) {
49 // Check other DataLayoutOpInterface implementations
50 if (dataLayoutOp.getDataLayoutSpec())
51 return DataLayout(dataLayoutOp);
52 }
53 current = current->getParentOp();
54 }
55
56 // No explicit data layout found; return default if allowed
57 if (allowDefault) {
58 // Check if op itself is a ModuleOp
59 if (auto mod = llvm::dyn_cast<ModuleOp>(op))
60 return DataLayout(mod);
61 // Otherwise check parents
62 if (auto mod = op->getParentOfType<ModuleOp>())
63 return DataLayout(mod);
64 }
65
66 return std::nullopt;
67}
68
69ComputeRegionOp buildComputeRegion(Location loc, ValueRange launchArgs,
70 ValueRange inputArgs, llvm::StringRef origin,
71 Region &regionToClone,
72 RewriterBase &rewriter, IRMapping &mapping,
73 ValueRange output,
74 FlatSymbolRefAttr kernelFuncName,
75 FlatSymbolRefAttr kernelModuleName,
76 Value stream, ValueRange inputArgsToMap) {
77 SmallVector<Type> resultTypes;
78 for (auto val : output)
79 resultTypes.push_back(val.getType());
80 auto computeRegion =
81 ComputeRegionOp::create(rewriter, loc, resultTypes, launchArgs, inputArgs,
82 stream, origin, kernelFuncName, kernelModuleName);
83
84 assert(!regionToClone.getBlocks().empty() &&
85 "empty region for acc.compute_region");
86 OpBuilder::InsertionGuard guard(rewriter);
87
88 ValueRange mapKeys = inputArgsToMap.empty() ? inputArgs : inputArgsToMap;
89 assert(mapKeys.size() == inputArgs.size() &&
90 "inputArgsToMap must have same size as inputArgs when provided");
91
92 Type indexType = rewriter.getIndexType();
93 Block *entryBlock = rewriter.createBlock(&computeRegion.getRegion());
94 for (size_t i = 0; i < launchArgs.size(); ++i)
95 entryBlock->addArgument(indexType, loc);
96 for (Value input : inputArgs)
97 entryBlock->addArgument(input.getType(), loc);
98 for (size_t i = 0; i < inputArgs.size(); ++i)
99 mapping.map(mapKeys[i], entryBlock->getArgument(launchArgs.size() + i));
100 rewriter.setInsertionPointToStart(entryBlock);
101 if (regionToClone.getBlocks().size() == 1) {
102 for (auto &op : regionToClone.front().getOperations()) {
103 if (op.hasTrait<OpTrait::IsTerminator>())
104 break;
105 rewriter.clone(op, mapping);
106 }
107 SmallVector<Value> yieldOperands;
108 for (auto val : output)
109 yieldOperands.push_back(mapping.lookup(val));
110 rewriter.setInsertionPointToEnd(entryBlock);
111 YieldOp::create(rewriter, loc, yieldOperands);
112 } else {
114 regionToClone, mapping, loc, rewriter);
115 if (!exeRegion) {
116 rewriter.eraseOp(computeRegion);
117 return nullptr;
118 }
120 llvm::to_vector(exeRegion.getOps<scf::YieldOp>()));
121 assert(!yieldOps.empty() &&
122 "multi-block region must contain at least one scf.yield");
123 assert(llvm::all_of(yieldOps,
124 [&output](scf::YieldOp yieldOp) {
125 return yieldOp.getNumOperands() ==
126 static_cast<int64_t>(output.size()) &&
127 llvm::all_of(
128 llvm::zip(yieldOp.getOperands(), output),
129 [](auto pair) {
130 return std::get<0>(pair).getType() ==
131 std::get<1>(pair).getType();
132 });
133 }) &&
134 "each scf.yield operand count and types must match output");
135 rewriter.setInsertionPointToEnd(entryBlock);
136 YieldOp::create(rewriter, loc, exeRegion.getResults());
137 }
138
139 return computeRegion;
140}
141
144 GPUParallelDimAttr parDim) {
145 return llvm::lower_bound(
146 parDims, parDim,
147 [](const GPUParallelDimAttr &lhs, const GPUParallelDimAttr &rhs) {
148 return lhs.getOrder() > rhs.getOrder();
149 });
150}
151
153 GPUParallelDimAttr parDim) {
155 if (lb == parDims.end() || *lb != parDim)
156 parDims.insert(lb, parDim);
157}
158
160 GPUParallelDimAttr parDim) {
162 if (lb != parDims.end() && *lb == parDim)
163 parDims.erase(lb);
164}
165
166#define ACC_OP_WITH_PAR_DIMS_LIST \
167 PrivatizeOp, ReductionAccumulateOp, ReductionAccumulateArrayOp, \
168 ReductionCombineOp
169
170GPUParallelDimsAttr getParDimsAttr(Operation *op) {
173 [](auto parOp) { return parOp.getParDimsAttr(); })
174 .Default([](Operation *op) -> GPUParallelDimsAttr {
175 if (Attribute attr =
176 op->getDiscardableAttr(GPUParallelDimsAttr::name)) {
177 GPUParallelDimsAttr parDimsAttr = dyn_cast<GPUParallelDimsAttr>(attr);
178 assert(parDimsAttr && "acc.par_dims must be a GPUParallelDimsAttr");
179 return parDimsAttr;
180 }
181 return nullptr;
182 });
183}
184
185bool hasParDimsAttr(Operation *op) { return getParDimsAttr(op) != nullptr; }
186
188 if (GPUParallelDimsAttr parDimsAttr = getParDimsAttr(op))
189 return parDimsAttr.isSeq();
190 return false;
191}
192
193void setParDimsAttr(Operation *op, GPUParallelDimsAttr attr) {
194 assert(!hasParDimsAttr(op) && "parallel dimensions attribute is already set");
197 [&](auto parOp) { parOp.setParDimsAttr(attr); })
198 .Default([&](Operation *op) {
199 op->setDiscardableAttr(GPUParallelDimsAttr::name, attr);
200 });
201}
202
203void updateParDimsAttr(Operation *op, GPUParallelDimsAttr attr) {
204 assert(hasParDimsAttr(op) &&
205 "expected parallel dimensions attribute to already be set");
208 [&](auto parOp) { parOp.setParDimsAttr(attr); })
209 .Default([&](Operation *op) {
210 op->setDiscardableAttr(GPUParallelDimsAttr::name, attr);
211 });
212}
213
214#undef ACC_OP_WITH_PAR_DIMS_LIST
215
217 return op->hasDiscardableAttrOfType<GPUBlockRedundantAttr>(
218 GPUBlockRedundantAttr::name);
219}
220
222 op->setDiscardableAttr(GPUBlockRedundantAttr::name,
223 GPUBlockRedundantAttr::get(op->getContext()));
224}
225
226ChunkSizeAttr getChunkSizeAttr(Operation *op) {
227 return op->getDiscardableAttrOfType<ChunkSizeAttr>(ChunkSizeAttr::name);
228}
229
230bool hasChunkSizeAttr(Operation *op) { return getChunkSizeAttr(op) != nullptr; }
231
232void setChunkSizeAttr(Operation *op, ChunkSizeAttr attr) {
233 op->setDiscardableAttr(ChunkSizeAttr::name, attr);
234}
235
236void setChunkSizeAttr(Operation *op, int64_t chunkSize) {
237 setChunkSizeAttr(op, ChunkSizeAttr::get(op->getContext(), chunkSize));
238}
239
240std::optional<int64_t> getChunkSize(Operation *op) {
241 if (ChunkSizeAttr attr = getChunkSizeAttr(op))
242 return attr.getChunkSize();
243 return std::nullopt;
244}
245
247 assert(hasParDimsAttr(from) &&
248 "expected parallel dimensions attribute to already be set");
250}
251
252ActiveParDimsAttr getActiveParDimsAttr(Operation *op) {
253 return op->getDiscardableAttrOfType<ActiveParDimsAttr>(
254 ActiveParDimsAttr::name);
255}
256
258 return getActiveParDimsAttr(op) != nullptr;
259}
260
261void setActiveParDimsAttr(Operation *op, ActiveParDimsAttr attr) {
262 op->setDiscardableAttr(ActiveParDimsAttr::name, attr);
263}
264
266 setActiveParDimsAttr(op, ActiveParDimsAttr::get(op->getContext(), dims));
267}
268
270 assert(alignment > 0 && llvm::isPowerOf2_64(alignment) &&
271 "alignment must be a power of two");
272 return (offset + alignment - 1) & ~(alignment - 1);
273}
274
276 int64_t aligned = alignOffset(bytesUsed_, alignment);
277 if (aligned + bytes > maxTotalBytes_) {
278 return false;
279 }
280 bytesUsed_ = aligned + bytes;
281 return true;
282}
283
285 int64_t total = 0;
286 region.walk([&](GPUSharedMemoryOp op) {
287 int64_t upperBound = op.getStaticUpperBoundBytes();
288 total = SharedMemoryBudget::alignOffset(total) + upperBound;
289 });
290 return total;
291}
292
293PrivatizeOp getPrivatizeOp(PrivateLocalOp privateLocal,
294 ComputeRegionOp computeRegion) {
295 Value value = privateLocal.getPrivatized();
296 if (BlockArgument blockArg = dyn_cast<BlockArgument>(value)) {
297 auto owner = dyn_cast<ComputeRegionOp>(blockArg.getOwner()->getParentOp());
298 value = (owner ? owner : computeRegion).getOperand(blockArg);
299 }
300 PrivatizeOp privatizeOp = value.getDefiningOp<PrivatizeOp>();
301 assert(privatizeOp && "expected privatize op to be the defining op");
302 return privatizeOp;
303}
304
305static bool isThreadXPrivatize(PrivatizeOp privatize) {
306 if (GPUParallelDimsAttr parDimsAttr = privatize.getParDimsAttr())
307 return llvm::any_of(parDimsAttr.getArray(),
308 [](GPUParallelDimAttr d) { return d.isThreadX(); });
309 return false;
310}
311
312MemRefType getPrivateBaseMemRefType(Type baseTy, ModuleOp module) {
313 auto memrefTy = cast<PointerLikeType>(baseTy).getAsMemRefType(module);
314 assert(memrefTy && "private base type must be convertible to memref");
315 return memrefTy;
316}
317
319collectPrivateLocalParDims(PrivateLocalOp privateLocal,
320 ComputeRegionOp computeRegion) {
322 // Walk the enclosing scf.parallel loops, but stop at the compute region
323 // boundary: loops outside the compute region do not contribute parallel
324 // dimensions to this privatization.
325 auto parentLoop = privateLocal->getParentOfType<scf::ParallelOp>();
326 while (parentLoop && computeRegion->isProperAncestor(parentLoop)) {
327 if (GPUParallelDimsAttr parDimsAttr = getParDimsAttr(parentLoop))
328 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
329 insertParDim(parDims, parDim);
330 parentLoop = parentLoop->getParentOfType<scf::ParallelOp>();
331 }
332 if (GPUParallelDimsAttr parDimsAttr = getParDimsAttr(computeRegion))
333 for (GPUParallelDimAttr parDim : parDimsAttr.getArray())
334 insertParDim(parDims, parDim);
335 if (parDims.empty()) {
336 for (GPUParallelDimAttr parDim : computeRegion.getLaunchParDims()) {
337 if (parDim.isAnyBlock())
338 insertParDim(parDims, parDim);
339 }
340 }
341
342 for (Operation *user : privateLocal.getResult().getUsers()) {
343 if (auto accumulateOp = dyn_cast<ReductionAccumulateOp>(user)) {
344 if (accumulateOp.getMemref() == privateLocal.getResult())
345 for (GPUParallelDimAttr parDim : accumulateOp.getParDims().getArray())
346 insertParDim(parDims, parDim);
347 }
348 if (auto combineOp = dyn_cast<ReductionCombineOp>(user)) {
349 if (combineOp.getSrcMemref() == privateLocal.getResult())
350 for (GPUParallelDimAttr parDim : getReductionCombineParDims(combineOp))
351 insertParDim(parDims, parDim);
352 }
353 if (auto combineRegionOp = dyn_cast<ReductionCombineRegionOp>(user)) {
354 if (combineRegionOp.getSrcVar() == privateLocal.getResult())
355 for (GPUParallelDimAttr parDim :
356 getReductionCombineParDims(combineRegionOp))
357 insertParDim(parDims, parDim);
358 }
359 }
360 return parDims;
361}
362
363static FailureOr<std::optional<int64_t>> getWorkerPrivateSharedMemoryNumCopies(
364 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion,
365 bool isWorkerPrivate, OpenACCSupport *support) {
366 if (!isWorkerPrivate)
367 return std::optional<int64_t>(1);
368
369 GPUParallelDimAttr threadY =
370 GPUParallelDimAttr::threadYDim(privateLocal.getContext());
371 std::optional<Value> workerArg = computeRegion.getKnownLaunchArg(threadY);
372 if (!workerArg)
373 return std::optional<int64_t>();
374
375 auto workerArgConst = workerArg->getDefiningOp<arith::ConstantIndexOp>();
376 if (workerArgConst)
377 return std::optional<int64_t>(workerArgConst.value());
378
379 FailureOr<int64_t> workerArgBound =
381 *workerArg);
382 if (succeeded(workerArgBound))
383 return std::optional<int64_t>(*workerArgBound);
384
385 if (support) {
386 (void)support->emitNYI(privateLocal.getLoc(),
387 "worker-private variables in shared memory "
388 "require compile-time constant num_workers");
389 return failure();
390 }
391 return std::optional<int64_t>();
392}
393
395 auto funcOp = op->getParentOfType<FunctionOpInterface>();
396 return funcOp && isSpecializedAccRoutine(funcOp);
397}
398
400 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
401 const ACCToGPUMappingPolicy &policy, OpenACCSupport *support) {
402 if (isInsideACCSpecializedRoutine(computeRegion))
403 return false;
404
405 if (isThreadXPrivatize(getPrivatizeOp(privateLocal, computeRegion)))
406 return false;
407
408 bool isReductionAccumulator =
409 llvm::any_of(privateLocal.getResult().getUsers(), [](Operation *user) {
410 return isa<ReductionAccumulateOp>(user);
411 });
412
414 collectPrivateLocalParDims(privateLocal, computeRegion);
415 bool isGangPrivate =
416 llvm::any_of(parDims, [&](auto parDim) { return policy.isGang(parDim); });
417 bool isWorkerPrivate = llvm::any_of(
418 parDims, [&](auto parDim) { return policy.isWorker(parDim); });
419 bool isVectorPrivate = llvm::any_of(
420 parDims, [&](auto parDim) { return policy.isVector(parDim); });
421
422 auto baseTy = getPrivateBaseMemRefType(
423 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
424 module);
425
426 bool isBlockLevelPrivate =
427 !isVectorPrivate &&
428 (isGangPrivate ||
429 (isWorkerPrivate && baseTy.getRank() > 0 && !isReductionAccumulator));
430 if (!isBlockLevelPrivate)
431 return false;
432
433 for (int64_t dim : baseTy.getShape())
434 if (dim == ShapedType::kDynamic)
435 return false;
436
437 auto resultMemRefTy = dyn_cast<MemRefType>(privateLocal.getType());
438 if (!resultMemRefTy || !resultMemRefTy.getLayout().isIdentity() ||
439 resultMemRefTy.getMemorySpace())
440 return false;
441
442 if (isGangPrivate && isWorkerPrivate && !isReductionAccumulator)
443 return false;
444
445 FailureOr<std::optional<int64_t>> numCopies =
446 getWorkerPrivateSharedMemoryNumCopies(privateLocal, computeRegion,
447 isWorkerPrivate, support);
448 if (failed(numCopies))
449 return failure();
450 return numCopies->has_value();
451}
452
454 PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module,
455 const ACCToGPUMappingPolicy &policy, OpenACCSupport *support) {
456 FailureOr<bool> isCandidate = isPrivateLocalSharedMemoryCandidate(
457 privateLocal, computeRegion, module, policy);
458 if (failed(isCandidate) || !*isCandidate)
459 return std::nullopt;
460
462 collectPrivateLocalParDims(privateLocal, computeRegion);
463 bool isWorkerPrivate = llvm::any_of(
464 parDims, [&](auto parDim) { return policy.isWorker(parDim); });
465
466 FailureOr<std::optional<int64_t>> numCopies =
468 privateLocal, computeRegion, isWorkerPrivate, /*support=*/nullptr);
469 if (failed(numCopies) || !numCopies->has_value())
470 return std::nullopt;
471
472 auto baseTy = getPrivateBaseMemRefType(
473 cast<PrivateType>(privateLocal.getPrivatized().getType()).getBaseTy(),
474 module);
475 std::optional<TypeSizeAndAlignment> elementSizeAndAlignment =
476 getTypeSizeAndAlignment(baseTy.getElementType(), module, support);
477 if (!elementSizeAndAlignment)
478 return std::nullopt;
479
480 int64_t numElements = 1;
481 for (int64_t dim : baseTy.getShape())
482 numElements *= dim;
483 return elementSizeAndAlignment->first.getFixedValue() * numElements *
484 numCopies->value();
485}
486
487bool hasAttachPoint(Operation *mapEntryOp) {
488 if (!mapEntryOp)
489 return false;
490 if (auto mapInfo = dyn_cast<MapInfoOp>(mapEntryOp))
491 return mapInfo.getVarPtrPtr() != nullptr;
492 if (isa<AttachOp>(mapEntryOp))
493 return true;
494 if (std::optional<DataClause> clause = getDataClause(mapEntryOp)) {
495 if (*clause == DataClause::acc_attach || *clause == DataClause::acc_detach)
496 return true;
497 }
498 return getVarPtrPtr(mapEntryOp) != nullptr;
499}
500
501DataDescKind getDataDescKind(Operation *mapEntryOp) {
502 if (auto mapInfo = dyn_cast<MapInfoOp>(mapEntryOp))
503 return mapInfo.getDescKind();
504 return DataDescKind::none;
505}
506
507Value getDesc(Operation *mapEntryOp) {
508 auto mapInfo = dyn_cast<MapInfoOp>(mapEntryOp);
509 if (!mapInfo)
510 return {};
511 if (Value desc = mapInfo.getDesc())
512 return desc;
513 // When the mapped var is itself the descriptor, map_info omits a redundant
514 // `desc` operand; recover it from `var` whenever a descriptor kind is set.
515 if (mapInfo.getDescKind() != DataDescKind::none)
516 return mapInfo.getVar();
517 return {};
518}
519
520std::optional<int64_t> getMapElementSize(Operation *mapEntryOp) {
521 if (auto mapInfo = dyn_cast<MapInfoOp>(mapEntryOp))
522 if (auto attr = mapInfo.getElementSizeAttr())
523 return attr.getInt();
524 return std::nullopt;
525}
526
528 if (auto mapInfo = dyn_cast<MapInfoOp>(mapEntryOp))
529 return mapInfo.getSize();
530 return {};
531}
532
533std::optional<MapFlags> getMapFlags(Operation *mapEntryOp) {
534 if (auto mapInfo = dyn_cast<MapInfoOp>(mapEntryOp))
535 return mapInfo.getMapFlags();
536 return std::nullopt;
537}
538
541 for (OpOperand &use : entryResult.getUses()) {
542 Operation *op = use.getOwner();
543 if (!isa<ACC_DATA_EXIT_OPS>(op))
544 continue;
545 Value accVar;
547 [&](auto exit) { accVar = exit.getAccVar(); });
548 // The entry result can also be used as another operand, such as async.
549 if (accVar == entryResult)
550 exitOps.push_back(op);
551 }
552 return exitOps;
553}
554
556 SmallVector<Operation *> exitOps = getPairedDataExitOps(entryResult);
557 return exitOps.empty() ? nullptr : exitOps.front();
558}
559
560static std::optional<Location> getMappingExitLoc(Value entryResult) {
561 if (auto mapInfo = entryResult.getDefiningOp<MapInfoOp>())
562 if (std::optional<Location> exitLoc = mapInfo.getExitLoc())
563 return exitLoc;
564 if (Operation *exitOp = findCorrespondingDataExit(entryResult))
565 return exitOp->getLoc();
566 return std::nullopt;
567}
568
569std::optional<Location> getMappingExitLoc(ValueRange dataClauseOperands) {
570 for (Value operand : dataClauseOperands)
571 if (std::optional<Location> exitLoc = getMappingExitLoc(operand))
572 return exitLoc;
573 return std::nullopt;
574}
575
576static std::optional<DataClause> getExitDataClause(Operation *exitOp) {
578 .Case<ACC_DATA_EXIT_OPS>([&](auto exit) { return exit.getDataClause(); })
579 .Default([&](Operation *) { return std::nullopt; });
580}
581
583 auto getMappedVar = [](Operation *op) {
584 Value var = getVar(op);
585 return var ? var : getVarPtr(op);
586 };
587 Value var = getMappedVar(entryOp);
588 if (!var)
589 return false;
590
591 auto copiesOutOnly = [&](Value sibling) {
592 Operation *siblingOp = sibling.getDefiningOp();
593 if (!siblingOp || getMappedVar(siblingOp) != var)
594 return false;
595 if (std::optional<MapFlags> siblingFlags = getMapFlags(siblingOp))
596 return bitEnumContainsAny(*siblingFlags, MapFlags::from) &&
597 !bitEnumContainsAny(*siblingFlags, MapFlags::to);
598 std::optional<DataClause> siblingClause = getDataClause(siblingOp);
599 return siblingClause && (*siblingClause == DataClause::acc_copyout ||
600 *siblingClause == DataClause::acc_copyout_zero);
601 };
602
603 Value entryResult = entryOp->getResult(0);
604 auto mapsSameVarOnConstruct = [&](auto construct) {
605 return llvm::any_of(construct.getDataClauseOperands(), [&](Value sibling) {
606 return sibling != entryResult && copiesOutOnly(sibling);
607 });
608 };
609 return llvm::any_of(entryResult.getUsers(), [&](Operation *user) {
610 return llvm::TypeSwitch<Operation *, bool>(user)
611 .Case<KernelEnvironmentOp, KernelsOp, ParallelOp, SerialOp, DataOp>(
612 mapsSameVarOnConstruct)
613 .Default(false);
614 });
615}
616
617static DataClauseModifier getEntryModifiers(Operation *entryOp) {
619 .Case<ACC_DATA_ENTRY_OPS>(
620 [&](auto entry) { return entry.getModifiers(); })
621 .Default([&](Operation *) { return DataClauseModifier::none; });
622}
623
624MapFlags computePrivatizeMapFlags(PrivatizeOp privatizeOp,
625 const ACCToGPUMappingPolicy &policy) {
626 MapFlags flags = MapFlags::private_;
627
628 // Storage is private without being replicated per parallel level when the
629 // privatization does not name any parallel dimension.
630 GPUParallelDimsAttr parDims = privatizeOp.getParDimsAttr();
631 if (!parDims)
632 return flags;
633
634 for (GPUParallelDimAttr parDim : parDims.getArray()) {
635 if (policy.isGang(parDim))
636 flags = flags | MapFlags::gang_private;
637 else if (policy.isWorker(parDim))
638 flags = flags | MapFlags::worker_private;
639 else if (policy.isVector(parDim))
640 flags = flags | MapFlags::vector_private;
641 }
642 return flags;
643}
644
645MapFlags computeDataClauseMapFlags(Operation *entryOp, bool ptrAndObj) {
646 MapFlags flags = MapFlags::none;
647 std::optional<DataClause> enterClause = getDataClause(entryOp);
648 if (!enterClause)
649 return flags;
650
651 switch (*enterClause) {
652 case DataClause::acc_create:
653 case DataClause::acc_copyout:
654 case DataClause::acc_present:
655 case DataClause::acc_private:
656 case DataClause::acc_firstprivate:
657 case DataClause::acc_delete:
658 case DataClause::acc_update_host:
659 case DataClause::acc_update_self:
660 case DataClause::acc_declare_device_resident:
661 if (*enterClause == DataClause::acc_declare_device_resident)
662 flags = flags | MapFlags::device_resident;
663 if (*enterClause == DataClause::acc_present)
664 flags = flags | MapFlags::present;
665 if (*enterClause == DataClause::acc_private ||
666 *enterClause == DataClause::acc_firstprivate)
667 flags = flags | MapFlags::private_;
668 if (*enterClause == DataClause::acc_firstprivate)
669 flags = flags | MapFlags::to;
670 break;
671 case DataClause::acc_deviceptr:
672 flags = flags | MapFlags::devptr;
673 break;
674 case DataClause::acc_create_zero:
675 case DataClause::acc_copyout_zero:
676 flags = flags | MapFlags::init_zero;
677 break;
678 case DataClause::acc_copy:
679 case DataClause::acc_copyin:
680 case DataClause::acc_copyin_readonly:
681 case DataClause::acc_reduction:
682 case DataClause::acc_update_device:
683 flags = flags | MapFlags::to;
684 break;
685 case DataClause::acc_no_create:
686 flags = flags | MapFlags::no_create;
687 break;
688 case DataClause::acc_attach:
689 flags = flags | MapFlags::attach;
690 break;
691 case DataClause::acc_detach:
692 flags = flags | MapFlags::detach;
693 break;
694 default:
695 break;
696 }
697 if (*enterClause == DataClause::acc_reduction)
698 flags = flags | MapFlags::reduction;
699
700 std::optional<DataClause> exitClause;
701 if (Operation *exitOp = findCorrespondingDataExit(entryOp->getResult(0)))
702 exitClause = getExitDataClause(exitOp);
703 if (exitClause) {
704 switch (*exitClause) {
705 case DataClause::acc_copy:
706 case DataClause::acc_reduction:
707 case DataClause::acc_copyout:
708 case DataClause::acc_copyout_zero:
709 case DataClause::acc_update_host:
710 case DataClause::acc_update_self:
711 flags = flags | MapFlags::from;
712 break;
713 case DataClause::acc_declare_device_resident:
714 flags = flags | MapFlags::device_resident;
715 break;
716 case DataClause::acc_present:
717 flags = flags | MapFlags::present;
718 break;
719 // `delete` only decrements the dynamic reference counter, so it must not
720 // request a forced unmap: the device copy has to survive while an
721 // enclosing region still references it. Only `finalize` zeroes the counter,
722 // and that is handled from the exit_data op below.
723 case DataClause::acc_delete:
724 break;
725 // An exit that repeats its entry clause only releases the device copy.
726 case DataClause::acc_create:
727 case DataClause::acc_create_zero:
728 case DataClause::acc_copyin:
729 case DataClause::acc_copyin_readonly:
730 if (hasCopyOutSibling(entryOp))
731 flags = flags | MapFlags::from;
732 break;
733 default:
734 break;
735 }
736 if (*exitClause == DataClause::acc_reduction)
737 flags = flags | MapFlags::reduction;
738 }
739
740 if (ptrAndObj)
741 flags = flags | MapFlags::ptr_and_obj;
742 if (getImplicitFlag(entryOp))
743 flags = flags | MapFlags::implicit;
744 if (bitEnumContainsAny(getEntryModifiers(entryOp), DataClauseModifier::zero))
745 flags = flags | MapFlags::init_zero;
746
747 for (OpOperand &use : entryOp->getResult(0).getUses()) {
748 if (auto exitDataOp = dyn_cast<ExitDataOp>(use.getOwner())) {
749 if (exitDataOp.getFinalize())
750 flags = flags | MapFlags::delete_;
751 }
752 if (auto updateOp = dyn_cast<UpdateOp>(use.getOwner())) {
753 if (updateOp.getIfPresent())
754 flags = flags | MapFlags::if_present;
755 }
756 }
757
758 return flags;
759}
760
761/// Returns the module \p var lives in.
762static ModuleOp getEnclosingModule(Value var) {
763 if (Operation *def = var.getDefiningOp())
764 return def->getParentOfType<ModuleOp>();
765 if (Region *region = var.getParentRegion())
766 if (Operation *parent = region->getParentOp())
767 return parent->getParentOfType<ModuleOp>();
768 return {};
769}
770
771int64_t computeMapInfoSizeBytes(Value var, Type varType, DataDescKind descKind,
772 ValueRange bounds, const DataLayout &dataLayout,
773 OpenACCSupport *support) {
774 // Bounds-driven and descriptor-driven maps report size 0: the extents and
775 // element size already state the size, and restating it here could disagree.
776 if (!bounds.empty() || descKind != DataDescKind::none)
777 return 0;
778
779 ModuleOp module = getEnclosingModule(var);
780 if (!module)
781 return -1;
782
783 auto tryUtilsSize = [&](Type ty) -> std::optional<int64_t> {
784 std::optional<TypeSizeAndAlignment> sizeAndAlign =
785 getTypeSizeAndAlignment(ty, module, dataLayout, support, var);
786 if (!sizeAndAlign || sizeAndAlign->first.isScalable())
787 return std::nullopt;
788 return static_cast<int64_t>(sizeAndAlign->first.getFixedValue());
789 };
790 if (std::optional<int64_t> size = tryUtilsSize(varType))
791 return *size;
792 if (std::optional<int64_t> size = tryUtilsSize(var.getType()))
793 return *size;
794
795 return -1;
796}
797
798int64_t computeMapInfoSizeBytes(Value var, Type varType, DataDescKind descKind,
799 ValueRange bounds, OpenACCSupport *support) {
800 ModuleOp module = getEnclosingModule(var);
801 if (!module)
802 return -1;
803 std::optional<DataLayout> dataLayout = getDataLayout(module);
804 if (!dataLayout)
805 return -1;
806 return computeMapInfoSizeBytes(var, varType, descKind, bounds, *dataLayout,
807 support);
808}
809
811 OpBuilder &builder) {
812 if (shape.size() != bounds.size())
813 return;
814 for (auto [boundValue, extent] : llvm::zip_equal(bounds, shape)) {
815 auto bound = boundValue.getDefiningOp<DataBoundsOp>();
816 if (!bound || bound.getSourceExtent() || extent < 0)
817 continue;
818 OpBuilder::InsertionGuard guard(builder);
819 builder.setInsertionPoint(bound);
820 Value sourceExtent =
821 arith::ConstantIndexOp::create(builder, bound.getLoc(), extent);
822 bound.getSourceExtentMutable().assign(sourceExtent);
823 }
824}
825
826} // namespace acc
827} // namespace mlir
#define ACC_OP_WITH_PAR_DIMS_LIST
Attributes are known-constant values of operations.
Definition Attributes.h:25
This class represents an argument of a Block.
Definition Value.h:306
Block represents an ordered list of Operations.
Definition Block.h:34
BlockArgument getArgument(unsigned i)
Definition Block.h:154
OpListType & getOperations()
Definition Block.h:162
BlockArgument addArgument(Type type, Location loc)
Add one value to the argument list.
Definition Block.cpp:158
IndexType getIndexType()
Definition Builders.cpp:59
The main mechanism for performing data layout queries.
A symbol reference with a reference path containing a single element.
This is a utility class for mapping one set of IR entities to another.
Definition IRMapping.h:26
auto lookup(T from) const
Lookup a mapped value within the map.
Definition IRMapping.h:72
void map(Value from, Value to)
Inserts a new mapping for 'from' to 'to'.
Definition IRMapping.h:30
This class defines the main interface for locations in MLIR and acts as a non-nullable wrapper around...
Definition Location.h:76
RAII guard to reset the insertion point of the builder when destroyed.
Definition Builders.h:351
This class helps build Operations.
Definition Builders.h:210
Block * createBlock(Region *parent, Region::iterator insertPt={}, TypeRange argTypes={}, ArrayRef< Location > locs={})
Add new block with 'argTypes' arguments and set the insertion point to the end of it.
Definition Builders.cpp:439
Operation * clone(Operation &op, IRMapping &mapper)
Creates a deep copy of the specified operation, remapping any operands that use values outside of the...
Definition Builders.cpp:581
void setInsertionPointToStart(Block *block)
Sets the insertion point to the start of the specified block.
Definition Builders.h:434
void setInsertionPoint(Block *block, Block::iterator insertPoint)
Set the insertion point to the specified location.
Definition Builders.h:401
void setInsertionPointToEnd(Block *block)
Sets the insertion point to the end of the specified block.
Definition Builders.h:439
This class represents an operand of an operation.
Definition Value.h:254
This class provides the API for ops that are known to be terminators.
Operation is the basic unit of execution within MLIR.
Definition Operation.h:87
bool hasDiscardableAttrOfType(NameT &&name)
Definition Operation.h:506
Attribute getDiscardableAttr(StringRef name)
Access a discardable attribute by name, returns a null Attribute if the discardable attribute does no...
Definition Operation.h:485
void setDiscardableAttr(StringAttr name, Attribute value)
Set a discardable attribute by name.
Definition Operation.h:512
OpResult getResult(unsigned idx)
Get the 'idx'th result of this operation.
Definition Operation.h:432
Operation * getParentOp()
Returns the closest surrounding operation that contains this operation or nullptr if this is a top-le...
Definition Operation.h:251
OpTy getParentOfType()
Return the closest surrounding parent operation that is of type 'OpTy'.
Definition Operation.h:255
AttrClass getDiscardableAttrOfType(StringRef name)
Access a discardable attribute by name and cast it to AttrClass.
Definition Operation.h:493
MLIRContext * getContext()
Return the context this operation is associated with.
Definition Operation.h:233
This class contains a list of basic blocks and a link to the parent operation it is attached to.
Definition Region.h:26
Block & front()
Definition Region.h:65
BlockListType & getBlocks()
Definition Region.h:45
RetT walk(FnT &&callback)
Walk all nested operations, blocks or regions (including this region), depending on the type of callb...
Definition Region.h:297
This class coordinates the application of a rewrite on a set of IR, providing a way for clients to tr...
virtual void eraseOp(Operation *op)
This method erases an operation that is known to have no uses.
Instances of the Type class are uniqued, have an immutable identifier and an optional mutable compone...
Definition Types.h:74
static FailureOr< int64_t > computeConstantBound(presburger::BoundType type, const Variable &var, const StopConditionFn &stopCondition=nullptr, ValueBoundsOptions options={})
Compute a constant bound for the given variable.
This class provides an abstraction over the different types of ranges over Values.
Definition ValueRange.h:389
This class represents an instance of an SSA value in the MLIR system, representing a computable value...
Definition Value.h:96
Type getType() const
Return the type of this value.
Definition Value.h:105
use_range getUses() const
Returns a range of all uses, which is useful for iterating over all uses.
Definition Value.h:188
user_range getUsers() const
Definition Value.h:218
Operation * getDefiningOp() const
If this value is the result of an operation, return the operation that defines it.
Definition Value.cpp:18
Region * getParentRegion()
Return the Region in which this Value is defined.
Definition Value.cpp:39
virtual bool isWorker(ParDimAttrT attr) const =0
Check if the attribute represents worker parallelism.
virtual bool isVector(ParDimAttrT attr) const =0
Check if the attribute represents vector parallelism.
virtual bool isGang(ParDimAttrT attr) const =0
Check if the attribute represents gang parallelism (any gang dimension).
InFlightDiagnostic emitNYI(Location loc, const Twine &message)
Report a case that is not yet supported by the implementation.
bool tryAllocate(int64_t bytes, int64_t alignment=kDefaultAlignmentBytes)
Reserve bytes, rounding the current offset up to alignment first.
static int64_t alignOffset(int64_t offset, int64_t alignment=kDefaultAlignmentBytes)
Round offset up to the next multiple of alignment, which must be a power of two.
Specialization of arith.constant op that returns an integer of index type.
Definition Arith.h:93
static ConstantIndexOp create(OpBuilder &builder, Location location, int64_t value)
Definition ArithOps.cpp:398
#define ACC_DATA_ENTRY_OPS
Definition OpenACC.h:49
#define ACC_DATA_EXIT_OPS
Definition OpenACC.h:59
MapFlags computePrivatizeMapFlags(PrivatizeOp privatizeOp, const ACCToGPUMappingPolicy &policy)
Compute the private and parallel-level map flags for privatized storage.
SmallVector< Operation * > getPairedDataExitOps(Value entryResult)
Returns the data exit operations paired with the data entry result entryResult, which take it as thei...
std::optional< int64_t > getChunkSize(Operation *op)
Return the chunk size carried by op, if any.
GPUParallelDimsAttr getParDimsAttr(Operation *op)
Obtain the parallel dimensions carried by op, if any.
static std::optional< DataClause > getExitDataClause(Operation *exitOp)
std::optional< DataLayout > getDataLayout(Operation *op, bool allowDefault=true)
Get the data layout for an operation.
std::optional< int64_t > getMapElementSize(Operation *mapEntryOp)
Returns element size in bytes from acc.map_info, if present.
MemRefType getPrivateBaseMemRefType(Type baseTy, ModuleOp module)
Returns the ranked MemRef type used to allocate privatized storage.
SmallVector< GPUParallelDimAttr > getReductionCombineParDims(ReductionCombineOp op)
Returns the parallel dimensions that participate in op's combine step.
void setActiveParDimsAttr(Operation *op, ActiveParDimsAttr attr)
Set active parallel dimensions on op.
mlir::Value getVar(mlir::Operation *accDataClauseOp)
Used to obtain the var from a data clause operation.
Definition OpenACC.cpp:5403
std::optional< Location > getMappingExitLoc(ValueRange dataClauseOperands)
Returns where the mappings of dataClauseOperands end, taken from the first of them that says.
void insertParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Insert parDim into parDims while preserving dimension ordering.
bool hasActiveParDimsAttr(Operation *op)
Return whether op carries active parallel dimensions.
std::optional< mlir::acc::DataClause > getDataClause(mlir::Operation *accDataEntryOp)
Used to obtain the dataClause from a data entry operation.
Definition OpenACC.cpp:5512
bool hasParDimsAttr(Operation *op)
Return whether op carries parallel dimensions.
static Operation * findCorrespondingDataExit(Value entryResult)
MapFlags computeDataClauseMapFlags(Operation *entryOp, bool ptrAndObj)
Fold enter (+ paired exit) data-clause semantics into offload map flags.
bool hasCopyOutSibling(Operation *entryOp)
True when another data clause of the same construct maps the same variable with a copy-back and no co...
Value getMapSize(Operation *mapEntryOp)
Returns the optional size operand from acc.map_info, or null.
ComputeRegionOp buildComputeRegion(Location loc, ValueRange launchArgs, ValueRange inputArgs, llvm::StringRef origin, Region &regionToClone, RewriterBase &rewriter, IRMapping &mapping, ValueRange output={}, FlatSymbolRefAttr kernelFuncName={}, FlatSymbolRefAttr kernelModuleName={}, Value stream={}, ValueRange inputArgsToMap={})
Build an acc.compute_region operation by cloning a source region.
ChunkSizeAttr getChunkSizeAttr(Operation *op)
Obtain the acc.chunk_size attribute carried by op, if any.
static ModuleOp getEnclosingModule(Value var)
Returns the module var lives in.
void setGPUBlockRedundantAttr(Operation *op)
Mark op with the acc.gpu_block_redundant attribute.
bool hasChunkSizeAttr(Operation *op)
Return whether op carries an acc.chunk_size attribute.
static FailureOr< std::optional< int64_t > > getWorkerPrivateSharedMemoryNumCopies(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, bool isWorkerPrivate, OpenACCSupport *support)
void setChunkSizeAttr(Operation *op, ChunkSizeAttr attr)
Set the acc.chunk_size attribute on op.
bool isSpecializedAccRoutine(mlir::Operation *op)
Used to check whether this is a specialized accelerator version of acc routine function.
Definition OpenACC.h:209
static bool isInsideACCSpecializedRoutine(Operation *op)
FailureOr< bool > isPrivateLocalSharedMemoryCandidate(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
True when privateLocal may be placed in shared memory.
int64_t sumExistingSharedMemoryBytes(Region &region)
Sum aligned static_upper_bound_bytes for all acc.gpu_shared_memory in region.
scf::ExecuteRegionOp wrapMultiBlockRegionWithSCFExecuteRegion(Region &region, IRMapping &mapping, Location loc, RewriterBase &rewriter)
Wrap a multi-block region in an scf.execute_region.
void updateParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Update parallel dimensions on op.
DataDescKind getDataDescKind(Operation *mapEntryOp)
Returns descriptor kind from acc.map_info, or none for other ops.
bool getImplicitFlag(mlir::Operation *accDataEntryOp)
Used to find out whether data operation is implicit.
Definition OpenACC.cpp:5522
Value getDesc(Operation *mapEntryOp)
Returns descriptor value from acc.map_info.
void populateSourceExtents(ValueRange bounds, ArrayRef< int64_t > shape, OpBuilder &builder)
Record known extents of the source array on bounds that may describe a section.
mlir::Value getVarPtrPtr(mlir::Operation *accDataClauseOp)
Used to obtain the varPtrPtr from a data clause operation.
Definition OpenACC.cpp:5444
PrivatizeOp getPrivatizeOp(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Resolve the acc.privatize operation associated with a private local.
bool hasSeqParDims(Operation *op)
Return whether op carries sequential parallel dimensions.
void copyParDimsAttr(Operation *from, Operation *to)
Copy parallel dimensions from from to to.
bool hasGPUBlockRedundantAttr(Operation *op)
Return whether op is marked with the acc.gpu_block_redundant attribute, i.e.
void removeParDim(llvm::SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
Remove parDim from parDims if present.
void setParDimsAttr(Operation *op, GPUParallelDimsAttr attr)
Set parallel dimensions on op.
ActiveParDimsAttr getActiveParDimsAttr(Operation *op)
Obtain the active parallel dimensions carried by op, if any.
std::optional< int64_t > getPrivateLocalSharedMemoryUpperBoundBytes(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion, ModuleOp module, const ACCToGPUMappingPolicy &policy, OpenACCSupport *support=nullptr)
Upper-bound byte size for a shared-memory private_local candidate, or std::nullopt when not eligible ...
static bool isThreadXPrivatize(PrivatizeOp privatize)
std::optional< MapFlags > getMapFlags(Operation *mapEntryOp)
Returns offload map-type flags from acc.map_info, if present.
int64_t computeMapInfoSizeBytes(Value var, Type varType, DataDescKind descKind, ValueRange bounds, const DataLayout &dataLayout, OpenACCSupport *support=nullptr)
Compute total mapped byte size for acc.map_info.
static SmallVector< GPUParallelDimAttr >::iterator findParDim(SmallVector< GPUParallelDimAttr > &parDims, GPUParallelDimAttr parDim)
mlir::TypedValue< mlir::acc::PointerLikeType > getVarPtr(mlir::Operation *accDataClauseOp)
Used to obtain the var from a data clause operation if it implements PointerLikeType.
Definition OpenACC.cpp:5389
SmallVector< GPUParallelDimAttr > collectPrivateLocalParDims(PrivateLocalOp privateLocal, ComputeRegionOp computeRegion)
Collect parallel dimensions that govern privatization of privateLocal.
bool hasAttachPoint(Operation *mapEntryOp)
Returns true when mapEntryOp carries an attach point (varPtrPtr).
ACCParMappingPolicy< mlir::acc::GPUParallelDimAttr > ACCToGPUMappingPolicy
Type alias for the GPU-specific mapping policy.
static DataClauseModifier getEntryModifiers(Operation *entryOp)
std::optional< TypeSizeAndAlignment > getTypeSizeAndAlignment(Type ty, ModuleOp module, const DataLayout &dl, OpenACCSupport *support=nullptr, Value var={})
Returns the size and ABI alignment in bytes.
Include the generated interface declarations.