clang 24.0.0git
AMDGPU.cpp
Go to the documentation of this file.
1//===--- AMDGPU.cpp - Implement AMDGPU target feature support -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements AMDGPU TargetInfo objects.
10//
11//===----------------------------------------------------------------------===//
12
13#include "AMDGPU.h"
20#include "llvm/ADT/SmallString.h"
21#include "llvm/TargetParser/AMDGPUTargetParser.h"
22using namespace clang;
23using namespace clang::targets;
24
25namespace clang {
26namespace targets {
27
28// If you edit the description strings, make sure you update
29// getPointerWidthV().
30
31const LangASMap AMDGPUTargetInfo::AMDGPUAddrSpaceMap = {
32 {LangAS::Default, llvm::AMDGPUAS::FLAT_ADDRESS},
33 {LangAS::opencl_global, llvm::AMDGPUAS::GLOBAL_ADDRESS},
34 {LangAS::opencl_local, llvm::AMDGPUAS::LOCAL_ADDRESS},
35 {LangAS::opencl_constant, llvm::AMDGPUAS::CONSTANT_ADDRESS},
36 {LangAS::opencl_private, llvm::AMDGPUAS::PRIVATE_ADDRESS},
37 {LangAS::opencl_generic, llvm::AMDGPUAS::FLAT_ADDRESS},
38 {LangAS::opencl_global_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
39 {LangAS::opencl_global_host, llvm::AMDGPUAS::GLOBAL_ADDRESS},
40 {LangAS::cuda_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
41 {LangAS::cuda_constant, llvm::AMDGPUAS::CONSTANT_ADDRESS},
42 {LangAS::cuda_shared, llvm::AMDGPUAS::LOCAL_ADDRESS},
43 {LangAS::sycl_global, llvm::AMDGPUAS::GLOBAL_ADDRESS},
44 {LangAS::sycl_global_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
45 {LangAS::sycl_global_host, llvm::AMDGPUAS::GLOBAL_ADDRESS},
46 {LangAS::sycl_local, llvm::AMDGPUAS::LOCAL_ADDRESS},
47 {LangAS::sycl_private, llvm::AMDGPUAS::PRIVATE_ADDRESS},
48 {LangAS::ptr32_sptr, llvm::AMDGPUAS::FLAT_ADDRESS},
49 {LangAS::ptr32_uptr, llvm::AMDGPUAS::FLAT_ADDRESS},
50 {LangAS::ptr64, llvm::AMDGPUAS::FLAT_ADDRESS},
51 {LangAS::hlsl_groupshared, llvm::AMDGPUAS::FLAT_ADDRESS},
52 {LangAS::hlsl_constant, llvm::AMDGPUAS::CONSTANT_ADDRESS},
53 // FIXME(pr/122103): hlsl_private -> PRIVATE is wrong, but at least this
54 // will break loudly.
55 {LangAS::hlsl_private, llvm::AMDGPUAS::PRIVATE_ADDRESS},
56 {LangAS::hlsl_device, llvm::AMDGPUAS::GLOBAL_ADDRESS},
57 {LangAS::hlsl_input, llvm::AMDGPUAS::PRIVATE_ADDRESS},
58 {LangAS::hlsl_output, llvm::AMDGPUAS::PRIVATE_ADDRESS},
59 {LangAS::hlsl_push_constant, llvm::AMDGPUAS::GLOBAL_ADDRESS},
60 {LangAS::amdgpu_barrier, llvm::AMDGPUAS::BARRIER},
61};
62
63} // namespace targets
64} // namespace clang
65
66static constexpr int NumBuiltins =
68
69#define GET_BUILTIN_STR_TABLE
70#include "clang/Basic/BuiltinsAMDGPU.inc"
71#undef GET_BUILTIN_STR_TABLE
72
73static constexpr Builtin::Info BuiltinInfos[] = {
74#define GET_BUILTIN_INFOS
75#include "clang/Basic/BuiltinsAMDGPU.inc"
76#undef GET_BUILTIN_INFOS
77};
78static_assert(std::size(BuiltinInfos) == NumBuiltins);
79
80const char *const AMDGPUTargetInfo::GCCRegNames[] = {
81 "v0", "v1", "v2", "v3", "v4", "v5", "v6", "v7", "v8",
82 "v9", "v10", "v11", "v12", "v13", "v14", "v15", "v16", "v17",
83 "v18", "v19", "v20", "v21", "v22", "v23", "v24", "v25", "v26",
84 "v27", "v28", "v29", "v30", "v31", "v32", "v33", "v34", "v35",
85 "v36", "v37", "v38", "v39", "v40", "v41", "v42", "v43", "v44",
86 "v45", "v46", "v47", "v48", "v49", "v50", "v51", "v52", "v53",
87 "v54", "v55", "v56", "v57", "v58", "v59", "v60", "v61", "v62",
88 "v63", "v64", "v65", "v66", "v67", "v68", "v69", "v70", "v71",
89 "v72", "v73", "v74", "v75", "v76", "v77", "v78", "v79", "v80",
90 "v81", "v82", "v83", "v84", "v85", "v86", "v87", "v88", "v89",
91 "v90", "v91", "v92", "v93", "v94", "v95", "v96", "v97", "v98",
92 "v99", "v100", "v101", "v102", "v103", "v104", "v105", "v106", "v107",
93 "v108", "v109", "v110", "v111", "v112", "v113", "v114", "v115", "v116",
94 "v117", "v118", "v119", "v120", "v121", "v122", "v123", "v124", "v125",
95 "v126", "v127", "v128", "v129", "v130", "v131", "v132", "v133", "v134",
96 "v135", "v136", "v137", "v138", "v139", "v140", "v141", "v142", "v143",
97 "v144", "v145", "v146", "v147", "v148", "v149", "v150", "v151", "v152",
98 "v153", "v154", "v155", "v156", "v157", "v158", "v159", "v160", "v161",
99 "v162", "v163", "v164", "v165", "v166", "v167", "v168", "v169", "v170",
100 "v171", "v172", "v173", "v174", "v175", "v176", "v177", "v178", "v179",
101 "v180", "v181", "v182", "v183", "v184", "v185", "v186", "v187", "v188",
102 "v189", "v190", "v191", "v192", "v193", "v194", "v195", "v196", "v197",
103 "v198", "v199", "v200", "v201", "v202", "v203", "v204", "v205", "v206",
104 "v207", "v208", "v209", "v210", "v211", "v212", "v213", "v214", "v215",
105 "v216", "v217", "v218", "v219", "v220", "v221", "v222", "v223", "v224",
106 "v225", "v226", "v227", "v228", "v229", "v230", "v231", "v232", "v233",
107 "v234", "v235", "v236", "v237", "v238", "v239", "v240", "v241", "v242",
108 "v243", "v244", "v245", "v246", "v247", "v248", "v249", "v250", "v251",
109 "v252", "v253", "v254", "v255", "s0", "s1", "s2", "s3", "s4",
110 "s5", "s6", "s7", "s8", "s9", "s10", "s11", "s12", "s13",
111 "s14", "s15", "s16", "s17", "s18", "s19", "s20", "s21", "s22",
112 "s23", "s24", "s25", "s26", "s27", "s28", "s29", "s30", "s31",
113 "s32", "s33", "s34", "s35", "s36", "s37", "s38", "s39", "s40",
114 "s41", "s42", "s43", "s44", "s45", "s46", "s47", "s48", "s49",
115 "s50", "s51", "s52", "s53", "s54", "s55", "s56", "s57", "s58",
116 "s59", "s60", "s61", "s62", "s63", "s64", "s65", "s66", "s67",
117 "s68", "s69", "s70", "s71", "s72", "s73", "s74", "s75", "s76",
118 "s77", "s78", "s79", "s80", "s81", "s82", "s83", "s84", "s85",
119 "s86", "s87", "s88", "s89", "s90", "s91", "s92", "s93", "s94",
120 "s95", "s96", "s97", "s98", "s99", "s100", "s101", "s102", "s103",
121 "s104", "s105", "s106", "s107", "s108", "s109", "s110", "s111", "s112",
122 "s113", "s114", "s115", "s116", "s117", "s118", "s119", "s120", "s121",
123 "s122", "s123", "s124", "s125", "s126", "s127", "exec", "vcc", "scc",
124 "m0", "flat_scratch", "exec_lo", "exec_hi", "vcc_lo", "vcc_hi",
125 "flat_scratch_lo", "flat_scratch_hi",
126 "a0", "a1", "a2", "a3", "a4", "a5", "a6", "a7", "a8",
127 "a9", "a10", "a11", "a12", "a13", "a14", "a15", "a16", "a17",
128 "a18", "a19", "a20", "a21", "a22", "a23", "a24", "a25", "a26",
129 "a27", "a28", "a29", "a30", "a31", "a32", "a33", "a34", "a35",
130 "a36", "a37", "a38", "a39", "a40", "a41", "a42", "a43", "a44",
131 "a45", "a46", "a47", "a48", "a49", "a50", "a51", "a52", "a53",
132 "a54", "a55", "a56", "a57", "a58", "a59", "a60", "a61", "a62",
133 "a63", "a64", "a65", "a66", "a67", "a68", "a69", "a70", "a71",
134 "a72", "a73", "a74", "a75", "a76", "a77", "a78", "a79", "a80",
135 "a81", "a82", "a83", "a84", "a85", "a86", "a87", "a88", "a89",
136 "a90", "a91", "a92", "a93", "a94", "a95", "a96", "a97", "a98",
137 "a99", "a100", "a101", "a102", "a103", "a104", "a105", "a106", "a107",
138 "a108", "a109", "a110", "a111", "a112", "a113", "a114", "a115", "a116",
139 "a117", "a118", "a119", "a120", "a121", "a122", "a123", "a124", "a125",
140 "a126", "a127", "a128", "a129", "a130", "a131", "a132", "a133", "a134",
141 "a135", "a136", "a137", "a138", "a139", "a140", "a141", "a142", "a143",
142 "a144", "a145", "a146", "a147", "a148", "a149", "a150", "a151", "a152",
143 "a153", "a154", "a155", "a156", "a157", "a158", "a159", "a160", "a161",
144 "a162", "a163", "a164", "a165", "a166", "a167", "a168", "a169", "a170",
145 "a171", "a172", "a173", "a174", "a175", "a176", "a177", "a178", "a179",
146 "a180", "a181", "a182", "a183", "a184", "a185", "a186", "a187", "a188",
147 "a189", "a190", "a191", "a192", "a193", "a194", "a195", "a196", "a197",
148 "a198", "a199", "a200", "a201", "a202", "a203", "a204", "a205", "a206",
149 "a207", "a208", "a209", "a210", "a211", "a212", "a213", "a214", "a215",
150 "a216", "a217", "a218", "a219", "a220", "a221", "a222", "a223", "a224",
151 "a225", "a226", "a227", "a228", "a229", "a230", "a231", "a232", "a233",
152 "a234", "a235", "a236", "a237", "a238", "a239", "a240", "a241", "a242",
153 "a243", "a244", "a245", "a246", "a247", "a248", "a249", "a250", "a251",
154 "a252", "a253", "a254", "a255"
155};
156
160
162 llvm::StringMap<bool> &Features, DiagnosticsEngine &Diags, StringRef CPU,
163 const std::vector<std::string> &FeatureVec) const {
164
165 using namespace llvm::AMDGPU;
166
167 if (!TargetInfo::initFeatureMap(Features, Diags, CPU, FeatureVec))
168 return false;
169
170 auto HasError = fillAMDGPUFeatureMap(CPU, getTriple(), Features);
171 switch (HasError.first) {
172 default:
173 break;
174 case llvm::AMDGPU::INVALID_FEATURE_COMBINATION:
175 Diags.Report(diag::err_invalid_feature_combination) << HasError.second;
176 return false;
177 case llvm::AMDGPU::UNSUPPORTED_TARGET_FEATURE:
178 Diags.Report(diag::err_opt_not_valid_on_target) << HasError.second;
179 return false;
180 }
181
182 return true;
183}
184
186 SmallVectorImpl<StringRef> &Values) const {
187 if (getTriple().isAMDGCN())
188 llvm::AMDGPU::fillValidArchListAMDGCN(Values, getTriple().getSubArch());
189 else
190 llvm::AMDGPU::fillValidArchListR600(Values);
191}
192
193AMDGPUTargetInfo::AMDGPUTargetInfo(const llvm::Triple &Triple,
194 const TargetOptions &Opts)
195 : TargetInfo(Triple),
196 GPUKind(Triple.isAMDGCN()
197 ? (Opts.CPU.empty() ? llvm::AMDGPU::getGPUKindFromSubArch(
198 Triple.getSubArch())
199 : llvm::AMDGPU::parseArchAMDGCN(Opts.CPU))
200 : llvm::AMDGPU::parseArchR600(Opts.CPU)) {
202
203 AddrSpaceMap = &AMDGPUAddrSpaceMap;
205 HasAMDGPUTypes = true;
206
207 if (Triple.isAMDGCN()) {
208 // __bf16 is always available as a load/store only type on AMDGCN.
210 BFloat16Format = &llvm::APFloat::BFloat();
211 }
212
213 // TODO: This is not really true for targets without half support, but also
214 // should just be assumed true for the dummy target.
215 HasFastHalfType = true;
216 HasFloat16 = true;
217 WavefrontSize = llvm::AMDGPU::getFeatureBitset(GPUKind).test(
218 llvm::AMDGPU::FEAT_SUPPORTS_WAVE32)
219 ? 32
220 : 64;
221
222 // Set pointer width and alignment for the generic address space.
224 if (getMaxPointerWidth() == 64) {
225 LongWidth = LongAlign = 64;
231 }
232
234 CUMode = !llvm::AMDGPU::getFeatureBitset(GPUKind).test(
235 llvm::AMDGPU::FEAT_SUPPORTS_WGP);
236
237 for (auto F : {"image-insts", "gws", "vmem-to-lds-load-insts", "supports-wgp",
238 "supports-wave32", "xnack-support", "sramecc-support",
239 "xnack-on-off-modes"}) {
240 if (GPUKind != llvm::AMDGPU::GK_NONE)
241 ReadOnlyFeatures.insert(F);
242 }
243 HalfArgsAndReturns = true;
244
246 OffloadArchFeatures["xnack"] =
248 }
249
251 OffloadArchFeatures["sramecc"] =
253 }
254}
255
257 const TargetInfo *Aux) {
258 TargetInfo::adjust(Diags, Opts, Aux);
260}
261
266
268 MacroBuilder &Builder) const {
269 Builder.defineMacro("__AMD__");
270 Builder.defineMacro("__AMDGPU__");
271
272 if (getTriple().isAMDGCN())
273 Builder.defineMacro("__AMDGCN__");
274 else
275 Builder.defineMacro("__R600__");
276
277 // TODO: __HAS_FMAF__, __HAS_LDEXPF__, __HAS_FP64__ are deprecated and will be
278 // removed in the near future.
279 if (hasFMAF())
280 Builder.defineMacro("__HAS_FMAF__");
281 if (hasFastFMAF())
282 Builder.defineMacro("FP_FAST_FMAF");
283 if (hasLDEXPF())
284 Builder.defineMacro("__HAS_LDEXPF__");
285 if (hasFP64())
286 Builder.defineMacro("__HAS_FP64__");
287 if (hasFastFMA())
288 Builder.defineMacro("FP_FAST_FMA");
289 if (HasFastHalfType)
290 Builder.defineMacro("FP_FAST_FMA_HALF");
291
292 Builder.defineMacro("__AMDGCN_CUMODE__", Twine(CUMode));
293
294 // Legacy HIP host code relies on these default attributes to be defined.
295 bool IsHIPHost = Opts.HIP && !Opts.CUDAIsDevice;
296 if (GPUKind == llvm::AMDGPU::GK_NONE && !IsHIPHost)
297 return;
298
299 llvm::SmallString<16> CanonName =
300 (getTriple().isAMDGCN() ? getArchNameAMDGCN(GPUKind)
301 : getArchNameR600(GPUKind));
302
303 // Sanitize the name of generic targets, the only names containing '-'.
304 // e.g. gfx10-1-generic -> gfx10_1_generic
305 llvm::replace(CanonName, '-', '_');
306
307 Builder.defineMacro(Twine("__") + Twine(CanonName) + Twine("__"));
308 // Emit macros for gfx family e.g. gfx906 -> __GFX9__, gfx1030 -> __GFX10___
309 if (getTriple().isAMDGCN() && !IsHIPHost) {
310 assert(StringRef(CanonName).starts_with("gfx") &&
311 "Invalid amdgcn canonical name");
312 StringRef CanonFamilyName = getArchFamilyNameAMDGCN(GPUKind);
313 Builder.defineMacro(Twine("__") + Twine(CanonFamilyName.upper()) +
314 Twine("__"));
315 Builder.defineMacro("__amdgcn_processor__",
316 Twine("\"") + Twine(CanonName) + Twine("\""));
317 Builder.defineMacro(
318 "__amdgcn_target_id__",
319 Twine("\"") +
320 Twine(getCanonicalTargetID(getArchNameAMDGCN(GPUKind),
321 OffloadArchFeatures)) +
322 Twine("\""));
323 for (auto F : getAllPossibleTargetIDFeatures(getTriple(), CanonName)) {
324 auto Loc = OffloadArchFeatures.find(F);
325 if (Loc != OffloadArchFeatures.end()) {
326 std::string NewF = F.str();
327 llvm::replace(NewF, '-', '_');
328 Builder.defineMacro(Twine("__amdgcn_feature_") + Twine(NewF) +
329 Twine("__"),
330 Loc->second ? "1" : "0");
331 }
332 }
333 }
334
336 Builder.defineMacro("__AMDGCN_UNSAFE_FP_ATOMICS__");
337}
338
340 assert(HalfFormat == Aux->HalfFormat);
341 assert(FloatFormat == Aux->FloatFormat);
342 assert(DoubleFormat == Aux->DoubleFormat);
343
344 // On x86_64 long double is 80-bit extended precision format, which is
345 // not supported by AMDGPU. 128-bit floating point format is also not
346 // supported by AMDGPU. Therefore keep its own format for these two types.
347 auto SaveLongDoubleFormat = LongDoubleFormat;
348 auto SaveFloat128Format = Float128Format;
349 auto SaveLongDoubleWidth = LongDoubleWidth;
350 auto SaveLongDoubleAlign = LongDoubleAlign;
351 copyAuxTarget(Aux);
352 LongDoubleFormat = SaveLongDoubleFormat;
353 Float128Format = SaveFloat128Format;
354 LongDoubleWidth = SaveLongDoubleWidth;
355 LongDoubleAlign = SaveLongDoubleAlign;
356 // For certain builtin types support on the host target, claim they are
357 // support to pass the compilation of the host code during the device-side
358 // compilation.
359 // FIXME: As the side effect, we also accept `__float128` uses in the device
360 // code. To rejct these builtin types supported in the host target but not in
361 // the device target, one approach would support `device_builtin` attribute
362 // so that we could tell the device builtin types from the host ones. The
363 // also solves the different representations of the same builtin type, such
364 // as `size_t` in the MSVC environment.
365 if (Aux->hasFloat128Type()) {
366 HasFloat128 = true;
368 }
369}
Defines the Diagnostic-related interfaces.
static constexpr llvm::StringTable BuiltinStrings
Definition AVR.cpp:24
static constexpr Builtin::Info BuiltinInfos[]
Definition Builtins.cpp:39
static constexpr unsigned NumBuiltins
Definition Builtins.cpp:33
Defines enum values for all the target-independent builtin functions.
Defines the clang::LangOptions interface.
Defines the clang::MacroBuilder utility class.
Enumerates target-specific builtins in their own namespaces within namespace clang.
Concrete class used by the front-end to report problems and issues.
Definition Diagnostic.h:234
DiagnosticBuilder Report(SourceLocation Loc, unsigned DiagID)
Issue the message to the client.
Keeps track of the various options that can be enabled, which controls the dialect of C or C++ that i...
void copyAuxTarget(const TargetInfo *Aux)
Copy type and layout related info.
TargetInfo(const llvm::Triple &T)
const llvm::Triple & getTriple() const
Returns the target triple of the primary target.
const LangASMap * AddrSpaceMap
Definition TargetInfo.h:259
unsigned HasAMDGPUTypes
Definition TargetInfo.h:290
AtomicOptions AtomicOpts
Definition TargetInfo.h:318
virtual void adjust(DiagnosticsEngine &Diags, LangOptions &Opts, const TargetInfo *Aux)
Set forced language options.
unsigned char MaxAtomicPromoteWidth
Definition TargetInfo.h:252
bool UseAddrSpaceMapMangling
Specify if mangling based on address space map should be used or not for language specific address sp...
Definition TargetInfo.h:391
void resetDataLayout()
Set the data layout based on current triple and ABI.
llvm::StringSet ReadOnlyFeatures
Definition TargetInfo.h:315
virtual bool hasFloat128Type() const
Determine whether the __float128 type is supported on this target.
Definition TargetInfo.h:711
virtual bool initFeatureMap(llvm::StringMap< bool > &Features, DiagnosticsEngine &Diags, StringRef CPU, const std::vector< std::string > &FeatureVec) const
Initialize the map with the default set of target features for the CPU this should include all legal ...
unsigned char MaxAtomicInlineWidth
Definition TargetInfo.h:252
Options for controlling the target.
AMDGPUFeatureState AMDGPUSramEccState
AMDGPU sramecc setting from -msramecc/-mno-sramecc.
AMDGPUFeatureState AMDGPUXnackState
AMDGPU xnack setting from -mxnack/-mno-xnack.
@ Enabled
Feature explicitly enabled.
@ Any
Feature state not specified and should generate most compatible code.
void setAuxTarget(const TargetInfo *Aux) override
Definition AMDGPU.cpp:339
ArrayRef< const char * > getGCCRegNames() const override
Definition AMDGPU.cpp:157
AMDGPUTargetInfo(const llvm::Triple &Triple, const TargetOptions &Opts)
Definition AMDGPU.cpp:193
uint64_t getPointerWidthV(LangAS AS) const override
Definition AMDGPU.h:100
void fillValidCPUList(SmallVectorImpl< StringRef > &Values) const override
Fill a SmallVectorImpl with the valid values to setCPU.
Definition AMDGPU.cpp:185
void adjust(DiagnosticsEngine &Diags, LangOptions &Opts, const TargetInfo *Aux) override
Set forced language options.
Definition AMDGPU.cpp:256
bool initFeatureMap(llvm::StringMap< bool > &Features, DiagnosticsEngine &Diags, StringRef CPU, const std::vector< std::string > &FeatureVec) const override
Initialize the map with the default set of target features for the CPU this should include all legal ...
Definition AMDGPU.cpp:161
void getTargetDefines(const LangOptions &Opts, MacroBuilder &Builder) const override
===-— Other target property query methods -----------------------—===//
Definition AMDGPU.cpp:267
llvm::SmallVector< Builtin::InfosShard > getTargetBuiltins() const override
Return information about target-specific builtins for the current primary target, and info about whic...
Definition AMDGPU.cpp:263
uint64_t getMaxPointerWidth() const override
Return the maximum width of pointers on this target.
Definition AMDGPU.h:129
Top level wrappers for InstallAPI frontend operations.
llvm::SmallVector< llvm::StringRef, 4 > getAllPossibleTargetIDFeatures(const llvm::Triple &T, llvm::StringRef Processor)
Get all feature strings that can be used in target ID for Processor.
Definition TargetID.cpp:44
std::string getCanonicalTargetID(llvm::StringRef Processor, const llvm::StringMap< bool > &Features)
Returns canonical target ID, assuming Processor is canonical and all entries in Features are valid.
Definition TargetID.cpp:146
Diagnostic wrappers for TextAPI types for error reporting.
Definition Dominators.h:30
The info used to represent each builtin.
Definition Builtins.h:80
const llvm::fltSemantics * DoubleFormat
Definition TargetInfo.h:143
const llvm::fltSemantics * LongDoubleFormat
Definition TargetInfo.h:143
const llvm::fltSemantics * Float128Format
Definition TargetInfo.h:143
const llvm::fltSemantics * FloatFormat
Definition TargetInfo.h:142
const llvm::fltSemantics * HalfFormat
Definition TargetInfo.h:142
const llvm::fltSemantics * BFloat16Format
Definition TargetInfo.h:142