clang 24.0.0git
AMDGPU.cpp
Go to the documentation of this file.
1//===--- AMDGPU.cpp - Implement AMDGPU target feature support -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements AMDGPU TargetInfo objects.
10//
11//===----------------------------------------------------------------------===//
12
13#include "AMDGPU.h"
19#include "llvm/ADT/SmallString.h"
20#include "llvm/TargetParser/AMDGPUTargetParser.h"
21using namespace clang;
22using namespace clang::targets;
23
24namespace clang {
25namespace targets {
26
27// If you edit the description strings, make sure you update
28// getPointerWidthV().
29
30const LangASMap AMDGPUTargetInfo::AMDGPUAddrSpaceMap = {
31 {LangAS::Default, llvm::AMDGPUAS::FLAT_ADDRESS},
32 {LangAS::opencl_global, llvm::AMDGPUAS::GLOBAL_ADDRESS},
33 {LangAS::opencl_local, llvm::AMDGPUAS::LOCAL_ADDRESS},
34 {LangAS::opencl_constant, llvm::AMDGPUAS::CONSTANT_ADDRESS},
35 {LangAS::opencl_private, llvm::AMDGPUAS::PRIVATE_ADDRESS},
36 {LangAS::opencl_generic, llvm::AMDGPUAS::FLAT_ADDRESS},
37 {LangAS::opencl_global_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
38 {LangAS::opencl_global_host, llvm::AMDGPUAS::GLOBAL_ADDRESS},
39 {LangAS::cuda_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
40 {LangAS::cuda_constant, llvm::AMDGPUAS::CONSTANT_ADDRESS},
41 {LangAS::cuda_shared, llvm::AMDGPUAS::LOCAL_ADDRESS},
42 {LangAS::sycl_global, llvm::AMDGPUAS::GLOBAL_ADDRESS},
43 {LangAS::sycl_global_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
44 {LangAS::sycl_global_host, llvm::AMDGPUAS::GLOBAL_ADDRESS},
45 {LangAS::sycl_local, llvm::AMDGPUAS::LOCAL_ADDRESS},
46 {LangAS::sycl_private, llvm::AMDGPUAS::PRIVATE_ADDRESS},
47 {LangAS::ptr32_sptr, llvm::AMDGPUAS::FLAT_ADDRESS},
48 {LangAS::ptr32_uptr, llvm::AMDGPUAS::FLAT_ADDRESS},
49 {LangAS::ptr64, llvm::AMDGPUAS::FLAT_ADDRESS},
50 {LangAS::hlsl_groupshared, llvm::AMDGPUAS::FLAT_ADDRESS},
51 {LangAS::hlsl_constant, llvm::AMDGPUAS::CONSTANT_ADDRESS},
52 // FIXME(pr/122103): hlsl_private -> PRIVATE is wrong, but at least this
53 // will break loudly.
54 {LangAS::hlsl_private, llvm::AMDGPUAS::PRIVATE_ADDRESS},
55 {LangAS::hlsl_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
56 {LangAS::hlsl_input, llvm::AMDGPUAS::PRIVATE_ADDRESS},
57 {LangAS::hlsl_output, llvm::AMDGPUAS::PRIVATE_ADDRESS},
58 {LangAS::hlsl_push_constant, llvm::AMDGPUAS::GLOBAL_ADDRESS},
59 {LangAS::amdgpu_barrier, llvm::AMDGPUAS::LOCAL_ADDRESS},
60};
61
62} // namespace targets
63} // namespace clang
64
65static constexpr int NumBuiltins =
67
68#define GET_BUILTIN_STR_TABLE
69#include "clang/Basic/BuiltinsAMDGPU.inc"
70#undef GET_BUILTIN_STR_TABLE
71
72static constexpr Builtin::Info BuiltinInfos[] = {
73#define GET_BUILTIN_INFOS
74#include "clang/Basic/BuiltinsAMDGPU.inc"
75#undef GET_BUILTIN_INFOS
76};
77static_assert(std::size(BuiltinInfos) == NumBuiltins);
78
79const char *const AMDGPUTargetInfo::GCCRegNames[] = {
80 "v0", "v1", "v2", "v3", "v4", "v5", "v6", "v7", "v8",
81 "v9", "v10", "v11", "v12", "v13", "v14", "v15", "v16", "v17",
82 "v18", "v19", "v20", "v21", "v22", "v23", "v24", "v25", "v26",
83 "v27", "v28", "v29", "v30", "v31", "v32", "v33", "v34", "v35",
84 "v36", "v37", "v38", "v39", "v40", "v41", "v42", "v43", "v44",
85 "v45", "v46", "v47", "v48", "v49", "v50", "v51", "v52", "v53",
86 "v54", "v55", "v56", "v57", "v58", "v59", "v60", "v61", "v62",
87 "v63", "v64", "v65", "v66", "v67", "v68", "v69", "v70", "v71",
88 "v72", "v73", "v74", "v75", "v76", "v77", "v78", "v79", "v80",
89 "v81", "v82", "v83", "v84", "v85", "v86", "v87", "v88", "v89",
90 "v90", "v91", "v92", "v93", "v94", "v95", "v96", "v97", "v98",
91 "v99", "v100", "v101", "v102", "v103", "v104", "v105", "v106", "v107",
92 "v108", "v109", "v110", "v111", "v112", "v113", "v114", "v115", "v116",
93 "v117", "v118", "v119", "v120", "v121", "v122", "v123", "v124", "v125",
94 "v126", "v127", "v128", "v129", "v130", "v131", "v132", "v133", "v134",
95 "v135", "v136", "v137", "v138", "v139", "v140", "v141", "v142", "v143",
96 "v144", "v145", "v146", "v147", "v148", "v149", "v150", "v151", "v152",
97 "v153", "v154", "v155", "v156", "v157", "v158", "v159", "v160", "v161",
98 "v162", "v163", "v164", "v165", "v166", "v167", "v168", "v169", "v170",
99 "v171", "v172", "v173", "v174", "v175", "v176", "v177", "v178", "v179",
100 "v180", "v181", "v182", "v183", "v184", "v185", "v186", "v187", "v188",
101 "v189", "v190", "v191", "v192", "v193", "v194", "v195", "v196", "v197",
102 "v198", "v199", "v200", "v201", "v202", "v203", "v204", "v205", "v206",
103 "v207", "v208", "v209", "v210", "v211", "v212", "v213", "v214", "v215",
104 "v216", "v217", "v218", "v219", "v220", "v221", "v222", "v223", "v224",
105 "v225", "v226", "v227", "v228", "v229", "v230", "v231", "v232", "v233",
106 "v234", "v235", "v236", "v237", "v238", "v239", "v240", "v241", "v242",
107 "v243", "v244", "v245", "v246", "v247", "v248", "v249", "v250", "v251",
108 "v252", "v253", "v254", "v255", "s0", "s1", "s2", "s3", "s4",
109 "s5", "s6", "s7", "s8", "s9", "s10", "s11", "s12", "s13",
110 "s14", "s15", "s16", "s17", "s18", "s19", "s20", "s21", "s22",
111 "s23", "s24", "s25", "s26", "s27", "s28", "s29", "s30", "s31",
112 "s32", "s33", "s34", "s35", "s36", "s37", "s38", "s39", "s40",
113 "s41", "s42", "s43", "s44", "s45", "s46", "s47", "s48", "s49",
114 "s50", "s51", "s52", "s53", "s54", "s55", "s56", "s57", "s58",
115 "s59", "s60", "s61", "s62", "s63", "s64", "s65", "s66", "s67",
116 "s68", "s69", "s70", "s71", "s72", "s73", "s74", "s75", "s76",
117 "s77", "s78", "s79", "s80", "s81", "s82", "s83", "s84", "s85",
118 "s86", "s87", "s88", "s89", "s90", "s91", "s92", "s93", "s94",
119 "s95", "s96", "s97", "s98", "s99", "s100", "s101", "s102", "s103",
120 "s104", "s105", "s106", "s107", "s108", "s109", "s110", "s111", "s112",
121 "s113", "s114", "s115", "s116", "s117", "s118", "s119", "s120", "s121",
122 "s122", "s123", "s124", "s125", "s126", "s127", "exec", "vcc", "scc",
123 "m0", "flat_scratch", "exec_lo", "exec_hi", "vcc_lo", "vcc_hi",
124 "flat_scratch_lo", "flat_scratch_hi",
125 "a0", "a1", "a2", "a3", "a4", "a5", "a6", "a7", "a8",
126 "a9", "a10", "a11", "a12", "a13", "a14", "a15", "a16", "a17",
127 "a18", "a19", "a20", "a21", "a22", "a23", "a24", "a25", "a26",
128 "a27", "a28", "a29", "a30", "a31", "a32", "a33", "a34", "a35",
129 "a36", "a37", "a38", "a39", "a40", "a41", "a42", "a43", "a44",
130 "a45", "a46", "a47", "a48", "a49", "a50", "a51", "a52", "a53",
131 "a54", "a55", "a56", "a57", "a58", "a59", "a60", "a61", "a62",
132 "a63", "a64", "a65", "a66", "a67", "a68", "a69", "a70", "a71",
133 "a72", "a73", "a74", "a75", "a76", "a77", "a78", "a79", "a80",
134 "a81", "a82", "a83", "a84", "a85", "a86", "a87", "a88", "a89",
135 "a90", "a91", "a92", "a93", "a94", "a95", "a96", "a97", "a98",
136 "a99", "a100", "a101", "a102", "a103", "a104", "a105", "a106", "a107",
137 "a108", "a109", "a110", "a111", "a112", "a113", "a114", "a115", "a116",
138 "a117", "a118", "a119", "a120", "a121", "a122", "a123", "a124", "a125",
139 "a126", "a127", "a128", "a129", "a130", "a131", "a132", "a133", "a134",
140 "a135", "a136", "a137", "a138", "a139", "a140", "a141", "a142", "a143",
141 "a144", "a145", "a146", "a147", "a148", "a149", "a150", "a151", "a152",
142 "a153", "a154", "a155", "a156", "a157", "a158", "a159", "a160", "a161",
143 "a162", "a163", "a164", "a165", "a166", "a167", "a168", "a169", "a170",
144 "a171", "a172", "a173", "a174", "a175", "a176", "a177", "a178", "a179",
145 "a180", "a181", "a182", "a183", "a184", "a185", "a186", "a187", "a188",
146 "a189", "a190", "a191", "a192", "a193", "a194", "a195", "a196", "a197",
147 "a198", "a199", "a200", "a201", "a202", "a203", "a204", "a205", "a206",
148 "a207", "a208", "a209", "a210", "a211", "a212", "a213", "a214", "a215",
149 "a216", "a217", "a218", "a219", "a220", "a221", "a222", "a223", "a224",
150 "a225", "a226", "a227", "a228", "a229", "a230", "a231", "a232", "a233",
151 "a234", "a235", "a236", "a237", "a238", "a239", "a240", "a241", "a242",
152 "a243", "a244", "a245", "a246", "a247", "a248", "a249", "a250", "a251",
153 "a252", "a253", "a254", "a255"
154};
155
159
161 llvm::StringMap<bool> &Features, DiagnosticsEngine &Diags, StringRef CPU,
162 const std::vector<std::string> &FeatureVec) const {
163
164 using namespace llvm::AMDGPU;
165
166 if (!TargetInfo::initFeatureMap(Features, Diags, CPU, FeatureVec))
167 return false;
168
169 auto HasError = fillAMDGPUFeatureMap(CPU, getTriple(), Features);
170 switch (HasError.first) {
171 default:
172 break;
173 case llvm::AMDGPU::INVALID_FEATURE_COMBINATION:
174 Diags.Report(diag::err_invalid_feature_combination) << HasError.second;
175 return false;
176 case llvm::AMDGPU::UNSUPPORTED_TARGET_FEATURE:
177 Diags.Report(diag::err_opt_not_valid_on_target) << HasError.second;
178 return false;
179 }
180
181 return true;
182}
183
185 SmallVectorImpl<StringRef> &Values) const {
186 if (getTriple().isAMDGCN())
187 llvm::AMDGPU::fillValidArchListAMDGCN(Values, getTriple().getSubArch());
188 else
189 llvm::AMDGPU::fillValidArchListR600(Values);
190}
191
192AMDGPUTargetInfo::AMDGPUTargetInfo(const llvm::Triple &Triple,
193 const TargetOptions &Opts)
194 : TargetInfo(Triple),
195 GPUKind(Triple.isAMDGCN()
196 ? (Opts.CPU.empty() ? llvm::AMDGPU::getGPUKindFromSubArch(
197 Triple.getSubArch())
198 : llvm::AMDGPU::parseArchAMDGCN(Opts.CPU))
199 : llvm::AMDGPU::parseArchR600(Opts.CPU)),
200 GPUFeatures(Triple.isAMDGCN() ? llvm::AMDGPU::getArchAttrAMDGCN(GPUKind)
201 : llvm::AMDGPU::getArchAttrR600(GPUKind)) {
203
204 AddrSpaceMap = &AMDGPUAddrSpaceMap;
206 HasAMDGPUTypes = true;
207
208 if (Triple.isAMDGCN()) {
209 // __bf16 is always available as a load/store only type on AMDGCN.
211 BFloat16Format = &llvm::APFloat::BFloat();
212 }
213
214 // TODO: This is not really true for targets without half support, but also
215 // should just be assumed true for the dummy target.
216 HasFastHalfType = true;
217 HasFloat16 = true;
218 WavefrontSize = (GPUFeatures & llvm::AMDGPU::FEATURE_WAVE32) ? 32 : 64;
219
220 // Set pointer width and alignment for the generic address space.
222 if (getMaxPointerWidth() == 64) {
223 LongWidth = LongAlign = 64;
227 }
228
230 CUMode = !(GPUFeatures & llvm::AMDGPU::FEATURE_WGP);
231
232 for (auto F : {"image-insts", "gws", "vmem-to-lds-load-insts"}) {
233 if (GPUKind != llvm::AMDGPU::GK_NONE)
234 ReadOnlyFeatures.insert(F);
235 }
236 HalfArgsAndReturns = true;
237
239 OffloadArchFeatures["xnack"] =
241 }
242
244 OffloadArchFeatures["sramecc"] =
246 }
247}
248
250 const TargetInfo *Aux) {
251 TargetInfo::adjust(Diags, Opts, Aux);
253}
254
259
261 MacroBuilder &Builder) const {
262 Builder.defineMacro("__AMD__");
263 Builder.defineMacro("__AMDGPU__");
264
265 if (getTriple().isAMDGCN())
266 Builder.defineMacro("__AMDGCN__");
267 else
268 Builder.defineMacro("__R600__");
269
270 // TODO: __HAS_FMAF__, __HAS_LDEXPF__, __HAS_FP64__ are deprecated and will be
271 // removed in the near future.
272 if (hasFMAF())
273 Builder.defineMacro("__HAS_FMAF__");
274 if (hasFastFMAF())
275 Builder.defineMacro("FP_FAST_FMAF");
276 if (hasLDEXPF())
277 Builder.defineMacro("__HAS_LDEXPF__");
278 if (hasFP64())
279 Builder.defineMacro("__HAS_FP64__");
280 if (hasFastFMA())
281 Builder.defineMacro("FP_FAST_FMA");
282 if (HasFastHalfType)
283 Builder.defineMacro("FP_FAST_FMA_HALF");
284
285 Builder.defineMacro("__AMDGCN_CUMODE__", Twine(CUMode));
286
287 // Legacy HIP host code relies on these default attributes to be defined.
288 bool IsHIPHost = Opts.HIP && !Opts.CUDAIsDevice;
289 if (GPUKind == llvm::AMDGPU::GK_NONE && !IsHIPHost)
290 return;
291
292 llvm::SmallString<16> CanonName =
293 (getTriple().isAMDGCN() ? getArchNameAMDGCN(GPUKind)
294 : getArchNameR600(GPUKind));
295
296 // Sanitize the name of generic targets, the only names containing '-'.
297 // e.g. gfx10-1-generic -> gfx10_1_generic
298 llvm::replace(CanonName, '-', '_');
299
300 Builder.defineMacro(Twine("__") + Twine(CanonName) + Twine("__"));
301 // Emit macros for gfx family e.g. gfx906 -> __GFX9__, gfx1030 -> __GFX10___
302 if (getTriple().isAMDGCN() && !IsHIPHost) {
303 assert(StringRef(CanonName).starts_with("gfx") &&
304 "Invalid amdgcn canonical name");
305 StringRef CanonFamilyName = getArchFamilyNameAMDGCN(GPUKind);
306 Builder.defineMacro(Twine("__") + Twine(CanonFamilyName.upper()) +
307 Twine("__"));
308 Builder.defineMacro("__amdgcn_processor__",
309 Twine("\"") + Twine(CanonName) + Twine("\""));
310 Builder.defineMacro(
311 "__amdgcn_target_id__",
312 Twine("\"") +
313 Twine(getCanonicalTargetID(getArchNameAMDGCN(GPUKind),
314 OffloadArchFeatures)) +
315 Twine("\""));
316 for (auto F : getAllPossibleTargetIDFeatures(getTriple(), CanonName)) {
317 auto Loc = OffloadArchFeatures.find(F);
318 if (Loc != OffloadArchFeatures.end()) {
319 std::string NewF = F.str();
320 llvm::replace(NewF, '-', '_');
321 Builder.defineMacro(Twine("__amdgcn_feature_") + Twine(NewF) +
322 Twine("__"),
323 Loc->second ? "1" : "0");
324 }
325 }
326 }
327
329 Builder.defineMacro("__AMDGCN_UNSAFE_FP_ATOMICS__");
330}
331
333 assert(HalfFormat == Aux->HalfFormat);
334 assert(FloatFormat == Aux->FloatFormat);
335 assert(DoubleFormat == Aux->DoubleFormat);
336
337 // On x86_64 long double is 80-bit extended precision format, which is
338 // not supported by AMDGPU. 128-bit floating point format is also not
339 // supported by AMDGPU. Therefore keep its own format for these two types.
340 auto SaveLongDoubleFormat = LongDoubleFormat;
341 auto SaveFloat128Format = Float128Format;
342 auto SaveLongDoubleWidth = LongDoubleWidth;
343 auto SaveLongDoubleAlign = LongDoubleAlign;
344 copyAuxTarget(Aux);
345 LongDoubleFormat = SaveLongDoubleFormat;
346 Float128Format = SaveFloat128Format;
347 LongDoubleWidth = SaveLongDoubleWidth;
348 LongDoubleAlign = SaveLongDoubleAlign;
349 // For certain builtin types support on the host target, claim they are
350 // support to pass the compilation of the host code during the device-side
351 // compilation.
352 // FIXME: As the side effect, we also accept `__float128` uses in the device
353 // code. To rejct these builtin types supported in the host target but not in
354 // the device target, one approach would support `device_builtin` attribute
355 // so that we could tell the device builtin types from the host ones. The
356 // also solves the different representations of the same builtin type, such
357 // as `size_t` in the MSVC environment.
358 if (Aux->hasFloat128Type()) {
359 HasFloat128 = true;
361 }
362}
Defines the Diagnostic-related interfaces.
static constexpr llvm::StringTable BuiltinStrings
Definition ARM.cpp:1115
static constexpr Builtin::Info BuiltinInfos[]
Definition Builtins.cpp:39
static constexpr unsigned NumBuiltins
Definition Builtins.cpp:33
Defines enum values for all the target-independent builtin functions.
Defines the clang::LangOptions interface.
Defines the clang::MacroBuilder utility class.
Enumerates target-specific builtins in their own namespaces within namespace clang.
Concrete class used by the front-end to report problems and issues.
Definition Diagnostic.h:234
DiagnosticBuilder Report(SourceLocation Loc, unsigned DiagID)
Issue the message to the client.
Keeps track of the various options that can be enabled, which controls the dialect of C or C++ that i...
void copyAuxTarget(const TargetInfo *Aux)
Copy type and layout related info.
TargetInfo(const llvm::Triple &T)
const llvm::Triple & getTriple() const
Returns the target triple of the primary target.
const LangASMap * AddrSpaceMap
Definition TargetInfo.h:260
unsigned HasAMDGPUTypes
Definition TargetInfo.h:291
AtomicOptions AtomicOpts
Definition TargetInfo.h:319
virtual void adjust(DiagnosticsEngine &Diags, LangOptions &Opts, const TargetInfo *Aux)
Set forced language options.
unsigned char MaxAtomicPromoteWidth
Definition TargetInfo.h:253
bool UseAddrSpaceMapMangling
Specify if mangling based on address space map should be used or not for language specific address sp...
Definition TargetInfo.h:392
void resetDataLayout()
Set the data layout based on current triple and ABI.
llvm::StringSet ReadOnlyFeatures
Definition TargetInfo.h:316
virtual bool hasFloat128Type() const
Determine whether the __float128 type is supported on this target.
Definition TargetInfo.h:724
virtual bool initFeatureMap(llvm::StringMap< bool > &Features, DiagnosticsEngine &Diags, StringRef CPU, const std::vector< std::string > &FeatureVec) const
Initialize the map with the default set of target features for the CPU this should include all legal ...
unsigned char MaxAtomicInlineWidth
Definition TargetInfo.h:253
Options for controlling the target.
AMDGPUFeatureState AMDGPUSramEccState
AMDGPU sramecc setting from -msramecc/-mno-sramecc.
AMDGPUFeatureState AMDGPUXnackState
AMDGPU xnack setting from -mxnack/-mno-xnack.
@ Enabled
Feature explicitly enabled.
@ Any
Feature state not specified and should generate most compatible code.
void setAuxTarget(const TargetInfo *Aux) override
Definition AMDGPU.cpp:332
ArrayRef< const char * > getGCCRegNames() const override
Definition AMDGPU.cpp:156
AMDGPUTargetInfo(const llvm::Triple &Triple, const TargetOptions &Opts)
Definition AMDGPU.cpp:192
uint64_t getPointerWidthV(LangAS AS) const override
Definition AMDGPU.h:98
void fillValidCPUList(SmallVectorImpl< StringRef > &Values) const override
Fill a SmallVectorImpl with the valid values to setCPU.
Definition AMDGPU.cpp:184
void adjust(DiagnosticsEngine &Diags, LangOptions &Opts, const TargetInfo *Aux) override
Set forced language options.
Definition AMDGPU.cpp:249
bool initFeatureMap(llvm::StringMap< bool > &Features, DiagnosticsEngine &Diags, StringRef CPU, const std::vector< std::string > &FeatureVec) const override
Initialize the map with the default set of target features for the CPU this should include all legal ...
Definition AMDGPU.cpp:160
void getTargetDefines(const LangOptions &Opts, MacroBuilder &Builder) const override
===-— Other target property query methods -----------------------—===//
Definition AMDGPU.cpp:260
llvm::SmallVector< Builtin::InfosShard > getTargetBuiltins() const override
Return information about target-specific builtins for the current primary target, and info about whic...
Definition AMDGPU.cpp:256
uint64_t getMaxPointerWidth() const override
Return the maximum width of pointers on this target.
Definition AMDGPU.h:127
The JSON file list parser is used to communicate input to InstallAPI.
llvm::SmallVector< llvm::StringRef, 4 > getAllPossibleTargetIDFeatures(const llvm::Triple &T, llvm::StringRef Processor)
Get all feature strings that can be used in target ID for Processor.
Definition TargetID.cpp:42
std::string getCanonicalTargetID(llvm::StringRef Processor, const llvm::StringMap< bool > &Features)
Returns canonical target ID, assuming Processor is canonical and all entries in Features are valid.
Definition TargetID.cpp:133
Diagnostic wrappers for TextAPI types for error reporting.
Definition Dominators.h:30
The info used to represent each builtin.
Definition Builtins.h:80
const llvm::fltSemantics * DoubleFormat
Definition TargetInfo.h:144
const llvm::fltSemantics * LongDoubleFormat
Definition TargetInfo.h:144
const llvm::fltSemantics * Float128Format
Definition TargetInfo.h:144
const llvm::fltSemantics * FloatFormat
Definition TargetInfo.h:143
const llvm::fltSemantics * HalfFormat
Definition TargetInfo.h:143
const llvm::fltSemantics * BFloat16Format
Definition TargetInfo.h:143