MLIR 24.0.0git
ROCDLTargetInfo.h
Go to the documentation of this file.
1//===- ROCDLTargetInfo.h - AMDGPU target description ------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8#ifndef MLIR_DIALECT_LLVMIR_ROCDLTARGETINFO_H_
9#define MLIR_DIALECT_LLVMIR_ROCDLTARGETINFO_H_
10
11#include "mlir/IR/Diagnostics.h"
12#include "mlir/Support/LLVM.h"
13#include "llvm/TargetParser/AMDGPUTargetParser.h"
14#include "llvm/TargetParser/Triple.h"
15#include <optional>
16
17namespace mlir::ROCDL {
18
19/// Describes the AMDGPU target a lowering is producing code for: the triple's
20/// subarch (which identifies the GPU) together with the resolved set of
21/// frontend-visible target features.
22///
23/// Lowerings should gate on features (`has(FEAT_...)`) rather than on ISA
24/// version arithmetic, and add features if necessary.
26public:
27 using Feature = ::llvm::AMDGPU::AMDGPUFeature;
28
29 /// Constructs an unknown target: no subarch, and every feature query answers
30 /// false.
31 TargetInfo() = default;
32
33 /// Resolves a target description.
34 ///
35 /// \p arch names the architecture the way Clang does, and accepts any of:
36 ///
37 /// - a full target ID, "<triple>-<processor>[:<feature><+|->]*", such as
38 /// "amdgcn-amd-amdhsa--gfx90a:sramecc+:xnack-" (what `rocminfo` prints
39 /// for a device's ISA) or "amdgpu9.0a-amd-amdhsa--gfx90a";
40 /// - a triple on its own, such as "amdgpu9.42-amd-amdhsa" or the legacy
41 /// subarch-less "amdgcn-amd-amdhsa";
42 /// - a processor on its own, with optional target-ID modifiers: "gfx942",
43 /// "gfx942:xnack+", "gfx9-4-generic".
44 ///
45 /// Only xnack and sramecc may be given as modifiers, and only on a processor
46 /// that supports them; this is the same grammar `clang::parseTargetID`
47 /// accepts, and it is validated by `llvm::AMDGPU::TargetID`.
48 ///
49 /// \p waveSize pins the wavefront size for targets that run at either, and
50 /// must be 0 (meaning the target's own default), 32, or 64.
51 ///
52 /// Diagnostics are emitted via `emitError`.
53 static FailureOr<TargetInfo>
54 get(StringRef arch, unsigned waveSize = 0,
56
57 /// Parses \p arch into a target ID, accepting the spellings `get()`
58 /// documents, or returns nullopt if it names no valid target.
59 ///
60 /// Use only if you need to get the individual components of the target ID.
61 static std::optional<::llvm::AMDGPU::TargetID> parseTargetID(StringRef arch);
62
63 /// Returns whether the target has \p feature.
64 bool has(Feature feature) const { return featureBits.test(feature); }
65
66 /// Returns whether the target's fp8 conversions exist and use the OCP formats
67 /// (E4M3FN/E5M2) rather than the FNUZ ones.
68 bool hasOcpFp8() const {
69 return has(::llvm::AMDGPU::FEAT_OCP_FP8_CONVERSION_INSTS);
70 }
71
72 /// Returns whether the target has fp8 conversions that use the FNUZ formats
73 /// (E4M3FNUZ/E5M2FNUZ).
74 bool hasFnuzFp8() const {
75 return has(::llvm::AMDGPU::FEAT_FP8_CONVERSION_INSTS) && !hasOcpFp8();
76 }
77
78 /// Returns whether the target belongs to gfx generation \p major (9 for any
79 /// gfx9xx, 12 for any gfx12xx, ...).
80 ///
81 /// Prefer `has()` where a feature expresses the condition; this is used when
82 /// no feature exists and the property being checked is a function of the
83 /// major ISA generation (such as the details of buffer encoding).
84 bool isGeneration(unsigned major) const;
85
86 /// Returns the width in bits of the num_records field of the buffer resource
87 /// (V#), or nullopt for an unknown target.
88 std::optional<unsigned> getBufferResourceNumRecordsWidth() const;
89
90 /// Returns the maximum LDS in bytes a single workgroup can address, or
91 /// nullopt for an unknown target.
92 std::optional<unsigned> getMaxAddressableLocalMemorySize() const;
93
94 /// Returns the wavefront size, or nullopt for an unknown target. Targets that
95 /// support both sizes report 32 unless "+wavefrontsize64" was requested.
96 std::optional<unsigned> getWavefrontSize() const;
97
98 /// Returns whether the GPU can be configured for 32-lane or 64-lane
99 /// wavefronts.
100 bool supportsBothWavefrontSizes() const { return dualWavefrontSize; }
101
102 /// Returns the total number of SGPRs, or nullopt for an unknown target.
103 std::optional<unsigned> getTotalNumSGPRs() const;
104
105 /// Returns the number of SGPRs addressable by a kernel, or nullopt for an
106 /// unknown target. This is below getTotalNumSGPRs() where some are reserved.
107 std::optional<unsigned> getAddressableNumSGPRs() const;
108
109 /// Returns the SGPR allocation granularity in registers, or nullopt for an
110 /// unknown target.
111 std::optional<unsigned> getSGPRAllocGranule() const;
112
113 /// Returns the VGPR allocation granularity in registers, or nullopt for an
114 /// unknown target. This property is wavesize-dependent.
115 std::optional<unsigned> getVGPRAllocGranule() const;
116
117 /// Returns the number of LDS banks per compute unit, or nullopt for an
118 /// unknown target.
119 std::optional<unsigned> getLDSBankCount() const;
120
121 /// Returns the maximum number of waves per execution unit, ignoring any
122 /// limits a particular kernel imposes, or nullopt for an unknown target.
123 std::optional<unsigned> getMaxWavesPerEU() const;
124
125 /// Returns whether xnack is on, off, either, or unsupported on this target.
126 /// "Any" means the target supports both and no `:xnack+/-` modifier was used.
127 ::llvm::AMDGPU::TargetIDSetting getXnackSetting() const {
128 return xnackSetting;
129 }
130
131 /// Returns whether sramecc is on, off, either, or unsupported, as for
132 /// getXnackSetting().
133 ::llvm::AMDGPU::TargetIDSetting getSramEccSetting() const {
134 return sramEccSetting;
135 }
136
137 /// Records the xnack and sramecc settings this target's ID pinned onto the
138 /// module \p op, as the `rocdl.xnack` and `rocdl.sramecc` attributes that
139 /// translate to the `amdgpu.xnack` and `amdgpu.sramecc` module flags.
140 ///
141 /// These flags are given as `:{xnack,sramecc}` target-ID "modifiers",
142 /// since they used to be subtarget features, but now frontends (like us and
143 /// Clang) need to migrate them into module flags. This representation keeps
144 /// us compatible with Clang and the output of tools like `rocminfo`.
145 ///
146 /// If a particular modifier is not given, no attribute is set for it, putting
147 /// that value into its "any" state if it is controllable.
149
150 /// Returns the ISA version. For a generic target this is the floor of the
151 /// family it covers (gfx9-4-generic reports 9.4.0), so it must not be used to
152 /// decide whether an instruction is available.
153 ::llvm::AMDGPU::IsaVersion getIsaVersion() const;
154
155 ::llvm::Triple::SubArchType getSubArch() const { return subArch; }
156 ::llvm::AMDGPU::GPUKind getGPUKind() const { return kind; }
157
158 /// Returns the canonical GPU name ("gfx942", "gfx9-4-generic"), or "" if the
159 /// target is unknown.
160 StringRef getArchName() const;
161
162 /// Returns whether this is a "gfxN-generic" target, which carries only the
163 /// features common to every GPU it covers.
164 bool isGeneric() const;
165
166 /// Returns whether no GPU was identified, in which case every feature query
167 /// answers false.
168 bool isUnknown() const { return kind == ::llvm::AMDGPU::GK_NONE; }
169
170 const ::llvm::AMDGPU::AMDGPUFeatureBitset &getFeatures() const {
171 return featureBits;
172 }
173
174private:
175 ::llvm::Triple::SubArchType subArch = ::llvm::Triple::NoSubArch;
176 ::llvm::AMDGPU::GPUKind kind = ::llvm::AMDGPU::GK_NONE;
177 ::llvm::AMDGPU::AMDGPUFeatureBitset featureBits;
178 ::llvm::AMDGPU::TargetIDSetting xnackSetting =
179 ::llvm::AMDGPU::TargetIDSetting::Unsupported;
180 ::llvm::AMDGPU::TargetIDSetting sramEccSetting =
181 ::llvm::AMDGPU::TargetIDSetting::Unsupported;
182 bool dualWavefrontSize = false;
183};
184
185/// Returns the target architecture that a pass should parse, given its `arch`
186/// option and the value of the deprecated alias that `arch` replaced.
187///
188/// The alias is only consulted when `arch` is left at "invalid".
189StringRef resolveArchOption(StringRef arch, StringRef deprecatedAlias);
190
191} // namespace mlir::ROCDL
192
193#endif // MLIR_DIALECT_LLVMIR_ROCDLTARGETINFO_H_
This class represents a diagnostic that is inflight and set to be reported.
Operation is the basic unit of execution within MLIR.
Definition Operation.h:87
bool has(Feature feature) const
Returns whether the target has feature.
std::optional< unsigned > getTotalNumSGPRs() const
Returns the total number of SGPRs, or nullopt for an unknown target.
bool hasFnuzFp8() const
Returns whether the target has fp8 conversions that use the FNUZ formats (E4M3FNUZ/E5M2FNUZ).
std::optional< unsigned > getSGPRAllocGranule() const
Returns the SGPR allocation granularity in registers, or nullopt for an unknown target.
std::optional< unsigned > getAddressableNumSGPRs() const
Returns the number of SGPRs addressable by a kernel, or nullopt for an unknown target.
bool isGeneric() const
Returns whether this is a "gfxN-generic" target, which carries only the features common to every GPU ...
::llvm::Triple::SubArchType getSubArch() const
::llvm::AMDGPU::IsaVersion getIsaVersion() const
Returns the ISA version.
bool isGeneration(unsigned major) const
Returns whether the target belongs to gfx generation major (9 for any gfx9xx, 12 for any gfx12xx,...
StringRef getArchName() const
Returns the canonical GPU name ("gfx942", "gfx9-4-generic"), or "" if the target is unknown.
std::optional< unsigned > getBufferResourceNumRecordsWidth() const
Returns the width in bits of the num_records field of the buffer resource (V#), or nullopt for an unk...
::llvm::AMDGPU::AMDGPUFeature Feature
bool isUnknown() const
Returns whether no GPU was identified, in which case every feature query answers false.
std::optional< unsigned > getLDSBankCount() const
Returns the number of LDS banks per compute unit, or nullopt for an unknown target.
::llvm::AMDGPU::TargetIDSetting getXnackSetting() const
Returns whether xnack is on, off, either, or unsupported on this target.
bool hasOcpFp8() const
Returns whether the target's fp8 conversions exist and use the OCP formats (E4M3FN/E5M2) rather than ...
bool supportsBothWavefrontSizes() const
Returns whether the GPU can be configured for 32-lane or 64-lane wavefronts.
::llvm::AMDGPU::TargetIDSetting getSramEccSetting() const
Returns whether sramecc is on, off, either, or unsupported, as for getXnackSetting().
void migrateArchFeaturesToModuleFlags(Operation *op) const
Records the xnack and sramecc settings this target's ID pinned onto the module op,...
std::optional< unsigned > getWavefrontSize() const
Returns the wavefront size, or nullopt for an unknown target.
static FailureOr< TargetInfo > get(StringRef arch, unsigned waveSize=0, function_ref< InFlightDiagnostic()> emitError=nullptr)
Resolves a target description.
::llvm::AMDGPU::GPUKind getGPUKind() const
std::optional< unsigned > getMaxAddressableLocalMemorySize() const
Returns the maximum LDS in bytes a single workgroup can address, or nullopt for an unknown target.
const ::llvm::AMDGPU::AMDGPUFeatureBitset & getFeatures() const
std::optional< unsigned > getMaxWavesPerEU() const
Returns the maximum number of waves per execution unit, ignoring any limits a particular kernel impos...
TargetInfo()=default
Constructs an unknown target: no subarch, and every feature query answers false.
std::optional< unsigned > getVGPRAllocGranule() const
Returns the VGPR allocation granularity in registers, or nullopt for an unknown target.
static std::optional<::llvm::AMDGPU::TargetID > parseTargetID(StringRef arch)
Parses arch into a target ID, accepting the spellings get() documents, or returns nullopt if it names...
StringRef resolveArchOption(StringRef arch, StringRef deprecatedAlias)
Returns the target architecture that a pass should parse, given its arch option and the value of the ...
InFlightDiagnostic emitError(Location loc)
Utility method to emit an error message using this location.
llvm::function_ref< Fn > function_ref
Definition LLVM.h:147