MLIR 24.0.0git
ROCDLTargetInfo.cpp
Go to the documentation of this file.
1//===- ROCDLTargetInfo.cpp - AMDGPU target description --------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
12#include "mlir/IR/Builders.h"
13#include "llvm/ADT/SmallVector.h"
14#include "llvm/ADT/StringRef.h"
15#include "llvm/ADT/Twine.h"
16
17using namespace mlir;
18using namespace mlir::ROCDL;
19
20namespace AMDGPU = ::llvm::AMDGPU;
21using ::llvm::Triple;
22
23/// Reports \p message through \p emitError if it is non-null, and returns
24/// failure.
26 const Twine &message) {
27 if (emitError)
28 emitError() << message;
29 return failure();
30}
31
32std::optional<AMDGPU::TargetID> TargetInfo::parseTargetID(StringRef arch) {
33 // If we see a five-component triple, that's maximally authoritative.
35 arch.split(parts, '-', /*MaxSplit=*/4);
36 if (parts.size() == 5)
37 return AMDGPU::TargetID::parseTargetIDString(arch);
38
39 // Otherwise, handle bare triples.
40 Triple triple(Triple::normalize(arch));
41 if (triple.isAMDGCN())
42 return AMDGPU::TargetID::parse(triple, "");
43
44 // Otherwise, take the implicit amdgcn-amd-amdhsa legacy triple and pair it
45 // with an arch name.
46 return AMDGPU::TargetID::parse(Triple("amdgcn-amd-amdhsa"), arch);
47}
48
49/// Pins the wavefront size in \p bits, mirroring the policy LLVM applies in
50/// fillAMDGCNFeatureMap: a target that only runs at one size rejects a request
51/// for the other, and a target that runs at either defaults to wave32.
52static LogicalResult
53resolveWavefrontSize(AMDGPU::AMDGPUFeatureBitset &bits, unsigned waveSize,
55 bool targetWave32 = bits.test(AMDGPU::FEAT_WAVEFRONTSIZE32);
56 bool targetWave64 = bits.test(AMDGPU::FEAT_WAVEFRONTSIZE64);
57
58 switch (waveSize) {
59 case 0:
60 // A target that runs at either size and was not asked for one runs wave32.
61 if (!targetWave32 && !targetWave64)
62 bits.set(AMDGPU::FEAT_WAVEFRONTSIZE32);
63 return success();
64 case 32:
65 if (targetWave64)
66 return fail(emitError, "target only supports a wavefront size of 64");
67 bits.set(AMDGPU::FEAT_WAVEFRONTSIZE32);
68 return success();
69 case 64:
70 if (targetWave32)
71 return fail(emitError, "target only supports a wavefront size of 32");
72 bits.set(AMDGPU::FEAT_WAVEFRONTSIZE64);
73 return success();
74 default:
75 return fail(emitError,
76 "wavefront size must be 32 or 64, got " + Twine(waveSize));
77 }
78}
79
80FailureOr<TargetInfo>
81TargetInfo::get(StringRef arch, unsigned waveSize,
83 if (arch.empty())
84 return fail(emitError, "target architecture cannot be empty");
85
86 std::optional<AMDGPU::TargetID> id = parseTargetID(arch);
87 if (!id)
88 return fail(emitError, "'" + arch +
89 "' is not a valid AMDGPU architecture: expected "
90 "a GPU name, a triple, or a target ID");
91
92 TargetInfo info;
93 info.kind = id->getGPUKind();
94 info.subArch = AMDGPU::getSubArch(info.kind);
95 info.featureBits = AMDGPU::getFeatureBitset(info.kind);
96 info.xnackSetting = id->getXnackSetting();
97 info.sramEccSetting = id->getSramEccSetting();
98
99 // Recorded before we record the user's choice since that lives in the same
100 // bitmap.
101 info.dualWavefrontSize =
102 !info.isUnknown() &&
103 !info.featureBits.test(AMDGPU::FEAT_WAVEFRONTSIZE32) &&
104 !info.featureBits.test(AMDGPU::FEAT_WAVEFRONTSIZE64);
105
106 if (info.isUnknown() && waveSize != 0 && waveSize != 32 && waveSize != 64)
107 return fail(emitError,
108 "wavefront size must be 32 or 64, got " + Twine(waveSize));
109 if (!info.isUnknown() &&
110 failed(resolveWavefrontSize(info.featureBits, waveSize, emitError)))
111 return failure();
112
113 return info;
114}
115
116StringRef mlir::ROCDL::resolveArchOption(StringRef arch,
117 StringRef deprecatedAlias) {
118 if (!arch.empty() && arch != "invalid")
119 return arch;
120 return deprecatedAlias.empty() ? arch : deprecatedAlias;
121}
122
123bool TargetInfo::isGeneration(unsigned major) const {
124 // The generation features are cumulative: a gfx12 target has every
125 // FEAT_GFX*_INSTS bit from gfx8 up to gfx12. So a target is *in* generation N
126 // when it has N's bit but not N+1's. This holds for generic targets too,
127 // unlike comparing ISA versions.
128 auto hasGen = [&](unsigned gen) {
129 switch (gen) {
130 case 7:
131 return has(AMDGPU::FEAT_CI_INSTS);
132 case 8:
133 return has(AMDGPU::FEAT_GFX8_INSTS);
134 case 9:
135 return has(AMDGPU::FEAT_GFX9_INSTS);
136 case 10:
137 return has(AMDGPU::FEAT_GFX10_INSTS);
138 case 11:
139 return has(AMDGPU::FEAT_GFX11_INSTS);
140 case 12:
141 return has(AMDGPU::FEAT_GFX12_INSTS);
142 case 13:
143 return has(AMDGPU::FEAT_GFX13_INSTS);
144 default:
145 return false;
146 }
147 };
148
149 if (isUnknown())
150 return false;
151 // gfx6 is the base: it has none of the generation features.
152 if (major == 6)
153 return !hasGen(7);
154 return hasGen(major) && !hasGen(major + 1);
155}
156
157std::optional<unsigned> TargetInfo::getBufferResourceNumRecordsWidth() const {
158 return AMDGPU::getBufferResourceNumRecordsWidth(kind);
159}
160
161std::optional<unsigned> TargetInfo::getMaxAddressableLocalMemorySize() const {
162 if (isUnknown())
163 return std::nullopt;
164 return AMDGPU::getMaxHWAddressableLocalMemorySize(kind);
165}
166
167std::optional<unsigned> TargetInfo::getTotalNumSGPRs() const {
168 if (isUnknown())
169 return std::nullopt;
170 return AMDGPU::getTotalNumSGPRs(kind);
171}
172
173std::optional<unsigned> TargetInfo::getAddressableNumSGPRs() const {
174 if (isUnknown())
175 return std::nullopt;
176 return AMDGPU::getAddressableNumSGPRs(kind);
177}
178
179std::optional<unsigned> TargetInfo::getSGPRAllocGranule() const {
180 if (isUnknown())
181 return std::nullopt;
182 return AMDGPU::getSGPRAllocGranule(kind);
183}
184
185std::optional<unsigned> TargetInfo::getVGPRAllocGranule() const {
186 // The granule depends on the wavefront size, which get() has already pinned.
187 std::optional<unsigned> waveSize = getWavefrontSize();
188 if (isUnknown() || !waveSize)
189 return std::nullopt;
190 return AMDGPU::getVGPRAllocGranule(kind, /*IsWave32=*/*waveSize == 32);
191}
192
193std::optional<unsigned> TargetInfo::getLDSBankCount() const {
194 if (isUnknown())
195 return std::nullopt;
196 return AMDGPU::getLDSBankCount(kind);
197}
198
199std::optional<unsigned> TargetInfo::getMaxWavesPerEU() const {
200 if (isUnknown())
201 return std::nullopt;
202 return AMDGPU::getMaxWavesPerEU(kind);
203}
204
205std::optional<unsigned> TargetInfo::getWavefrontSize() const {
206 if (has(AMDGPU::FEAT_WAVEFRONTSIZE64))
207 return 64;
208 if (has(AMDGPU::FEAT_WAVEFRONTSIZE32))
209 return 32;
210 return std::nullopt;
211}
212
214 assert(LLVM::satisfiesLLVMModule(op) &&
215 "xnack and sramecc describe a whole code object, so they can only be "
216 "recorded on a module");
217 ROCDLDialect *dialect =
218 op->getContext()->getOrLoadDialect<ROCDL::ROCDLDialect>();
219 Builder builder(op->getContext());
220 // The helpers differ in type, hence the generic lambda.
221 auto migrate = [&](AMDGPU::TargetIDSetting setting, auto helper) {
222 if (setting != AMDGPU::TargetIDSetting::On &&
223 setting != AMDGPU::TargetIDSetting::Off)
224 return;
225 helper.setAttr(op,
226 builder.getBoolAttr(setting == AMDGPU::TargetIDSetting::On));
227 };
228 migrate(xnackSetting, dialect->getXnackAttrHelper());
229 migrate(sramEccSetting, dialect->getSrameccAttrHelper());
230}
231
232AMDGPU::IsaVersion TargetInfo::getIsaVersion() const {
233 return AMDGPU::getIsaVersion(subArch);
234}
235
236StringRef TargetInfo::getArchName() const {
237 return AMDGPU::getArchNameAMDGCN(kind);
238}
239
241 return !isUnknown() && AMDGPU::getMajorSubArch(subArch) == subArch;
242}
return success()
static LogicalResult fail(function_ref< InFlightDiagnostic()> emitError, const Twine &message)
Reports message through emitError if it is non-null, and returns failure.
static LogicalResult resolveWavefrontSize(AMDGPU::AMDGPUFeatureBitset &bits, unsigned waveSize, function_ref< InFlightDiagnostic()> emitError)
Pins the wavefront size in bits, mirroring the policy LLVM applies in fillAMDGCNFeatureMap: a target ...
This class is a general helper class for creating context-global objects like types,...
Definition Builders.h:51
This class represents a diagnostic that is inflight and set to be reported.
T * getOrLoadDialect()
Get (or create) a dialect for the given derived dialect type.
Operation is the basic unit of execution within MLIR.
Definition Operation.h:87
MLIRContext * getContext()
Return the context this operation is associated with.
Definition Operation.h:233
bool has(Feature feature) const
Returns whether the target has feature.
std::optional< unsigned > getTotalNumSGPRs() const
Returns the total number of SGPRs, or nullopt for an unknown target.
std::optional< unsigned > getSGPRAllocGranule() const
Returns the SGPR allocation granularity in registers, or nullopt for an unknown target.
std::optional< unsigned > getAddressableNumSGPRs() const
Returns the number of SGPRs addressable by a kernel, or nullopt for an unknown target.
bool isGeneric() const
Returns whether this is a "gfxN-generic" target, which carries only the features common to every GPU ...
::llvm::AMDGPU::IsaVersion getIsaVersion() const
Returns the ISA version.
bool isGeneration(unsigned major) const
Returns whether the target belongs to gfx generation major (9 for any gfx9xx, 12 for any gfx12xx,...
StringRef getArchName() const
Returns the canonical GPU name ("gfx942", "gfx9-4-generic"), or "" if the target is unknown.
std::optional< unsigned > getBufferResourceNumRecordsWidth() const
Returns the width in bits of the num_records field of the buffer resource (V#), or nullopt for an unk...
bool isUnknown() const
Returns whether no GPU was identified, in which case every feature query answers false.
std::optional< unsigned > getLDSBankCount() const
Returns the number of LDS banks per compute unit, or nullopt for an unknown target.
void migrateArchFeaturesToModuleFlags(Operation *op) const
Records the xnack and sramecc settings this target's ID pinned onto the module op,...
std::optional< unsigned > getWavefrontSize() const
Returns the wavefront size, or nullopt for an unknown target.
static FailureOr< TargetInfo > get(StringRef arch, unsigned waveSize=0, function_ref< InFlightDiagnostic()> emitError=nullptr)
Resolves a target description.
std::optional< unsigned > getMaxAddressableLocalMemorySize() const
Returns the maximum LDS in bytes a single workgroup can address, or nullopt for an unknown target.
std::optional< unsigned > getMaxWavesPerEU() const
Returns the maximum number of waves per execution unit, ignoring any limits a particular kernel impos...
TargetInfo()=default
Constructs an unknown target: no subarch, and every feature query answers false.
std::optional< unsigned > getVGPRAllocGranule() const
Returns the VGPR allocation granularity in registers, or nullopt for an unknown target.
static std::optional<::llvm::AMDGPU::TargetID > parseTargetID(StringRef arch)
Parses arch into a target ID, accepting the spellings get() documents, or returns nullopt if it names...
bool satisfiesLLVMModule(Operation *op)
LLVM requires some operations to be inside of a Module operation.
StringRef resolveArchOption(StringRef arch, StringRef deprecatedAlias)
Returns the target architecture that a pass should parse, given its arch option and the value of the ...
Include the generated interface declarations.
InFlightDiagnostic emitError(Location loc)
Utility method to emit an error message using this location.
llvm::function_ref< Fn > function_ref
Definition LLVM.h:147