clang 24.0.0git
CGOpenMPRuntime.cpp
Go to the documentation of this file.
1//===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This provides a class for OpenMP runtime code generation.
10//
11//===----------------------------------------------------------------------===//
12
13#include "CGOpenMPRuntime.h"
14#include "ABIInfoImpl.h"
15#include "CGCXXABI.h"
16#include "CGCleanup.h"
17#include "CGDebugInfo.h"
18#include "CGRecordLayout.h"
19#include "CodeGenFunction.h"
20#include "TargetInfo.h"
21#include "clang/AST/APValue.h"
22#include "clang/AST/Attr.h"
23#include "clang/AST/Decl.h"
31#include "llvm/ADT/ArrayRef.h"
32#include "llvm/ADT/SmallSet.h"
33#include "llvm/ADT/SmallVector.h"
34#include "llvm/ADT/StringExtras.h"
35#include "llvm/Bitcode/BitcodeReader.h"
36#include "llvm/IR/Constants.h"
37#include "llvm/IR/DerivedTypes.h"
38#include "llvm/IR/GlobalValue.h"
39#include "llvm/IR/InstrTypes.h"
40#include "llvm/IR/Value.h"
41#include "llvm/Support/AtomicOrdering.h"
42#include "llvm/Support/VirtualFileSystem.h"
43#include "llvm/Support/raw_ostream.h"
44#include <cassert>
45#include <cstdint>
46#include <numeric>
47#include <optional>
48
49using namespace clang;
50using namespace CodeGen;
51using namespace llvm::omp;
52
53namespace {
54/// Base class for handling code generation inside OpenMP regions.
55class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
56public:
57 /// Kinds of OpenMP regions used in codegen.
58 enum CGOpenMPRegionKind {
59 /// Region with outlined function for standalone 'parallel'
60 /// directive.
61 ParallelOutlinedRegion,
62 /// Region with outlined function for standalone 'task' directive.
63 TaskOutlinedRegion,
64 /// Region for constructs that do not require function outlining,
65 /// like 'for', 'sections', 'atomic' etc. directives.
66 InlinedRegion,
67 /// Region with outlined function for standalone 'target' directive.
68 TargetRegion,
69 };
70
71 CGOpenMPRegionInfo(const CapturedStmt &CS,
72 const CGOpenMPRegionKind RegionKind,
73 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
74 bool HasCancel)
75 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
76 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {}
77
78 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
79 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
80 bool HasCancel)
81 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen),
82 Kind(Kind), HasCancel(HasCancel) {}
83
84 /// Get a variable or parameter for storing global thread id
85 /// inside OpenMP construct.
86 virtual const VarDecl *getThreadIDVariable() const = 0;
87
88 /// Emit the captured statement body.
89 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
90
91 /// Get an LValue for the current ThreadID variable.
92 /// \return LValue for thread id variable. This LValue always has type int32*.
93 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
94
95 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {}
96
97 CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
98
99 OpenMPDirectiveKind getDirectiveKind() const { return Kind; }
100
101 bool hasCancel() const { return HasCancel; }
102
103 static bool classof(const CGCapturedStmtInfo *Info) {
104 return Info->getKind() == CR_OpenMP;
105 }
106
107 ~CGOpenMPRegionInfo() override = default;
108
109protected:
110 CGOpenMPRegionKind RegionKind;
111 RegionCodeGenTy CodeGen;
113 bool HasCancel;
114};
115
116/// API for captured statement code generation in OpenMP constructs.
117class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo {
118public:
119 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
120 const RegionCodeGenTy &CodeGen,
121 OpenMPDirectiveKind Kind, bool HasCancel,
122 StringRef HelperName)
123 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind,
124 HasCancel),
125 ThreadIDVar(ThreadIDVar), HelperName(HelperName) {
126 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
127 }
128
129 /// Get a variable or parameter for storing global thread id
130 /// inside OpenMP construct.
131 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
132
133 /// Get the name of the capture helper.
134 StringRef getHelperName() const override { return HelperName; }
135
136 static bool classof(const CGCapturedStmtInfo *Info) {
137 return CGOpenMPRegionInfo::classof(Info) &&
138 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
139 ParallelOutlinedRegion;
140 }
141
142private:
143 /// A variable or parameter storing global thread id for OpenMP
144 /// constructs.
145 const VarDecl *ThreadIDVar;
146 StringRef HelperName;
147};
148
149/// API for captured statement code generation in OpenMP constructs.
150class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo {
151public:
152 class UntiedTaskActionTy final : public PrePostActionTy {
153 bool Untied;
154 const VarDecl *PartIDVar;
155 const RegionCodeGenTy UntiedCodeGen;
156 llvm::SwitchInst *UntiedSwitch = nullptr;
157
158 public:
159 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar,
160 const RegionCodeGenTy &UntiedCodeGen)
161 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {}
162 void Enter(CodeGenFunction &CGF) override {
163 if (Untied) {
164 // Emit task switching point.
165 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
166 CGF.GetAddrOfLocalVar(PartIDVar),
167 PartIDVar->getType()->castAs<PointerType>());
168 llvm::Value *Res =
169 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation());
170 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done.");
171 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB);
172 CGF.EmitBlock(DoneBB);
174 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
175 UntiedSwitch->addCase(CGF.Builder.getInt32(0),
176 CGF.Builder.GetInsertBlock());
177 emitUntiedSwitch(CGF);
178 }
179 }
180 void emitUntiedSwitch(CodeGenFunction &CGF) const {
181 if (Untied) {
182 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
183 CGF.GetAddrOfLocalVar(PartIDVar),
184 PartIDVar->getType()->castAs<PointerType>());
185 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
186 PartIdLVal);
187 UntiedCodeGen(CGF);
188 CodeGenFunction::JumpDest CurPoint =
189 CGF.getJumpDestInCurrentScope(".untied.next.");
191 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
192 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
193 CGF.Builder.GetInsertBlock());
194 CGF.EmitBranchThroughCleanup(CurPoint);
195 CGF.EmitBlock(CurPoint.getBlock());
196 }
197 }
198 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); }
199 };
200 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
201 const VarDecl *ThreadIDVar,
202 const RegionCodeGenTy &CodeGen,
203 OpenMPDirectiveKind Kind, bool HasCancel,
204 const UntiedTaskActionTy &Action)
205 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel),
206 ThreadIDVar(ThreadIDVar), Action(Action) {
207 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
208 }
209
210 /// Get a variable or parameter for storing global thread id
211 /// inside OpenMP construct.
212 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
213
214 /// Get an LValue for the current ThreadID variable.
215 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
216
217 /// Get the name of the capture helper.
218 StringRef getHelperName() const override { return ".omp_outlined."; }
219
220 void emitUntiedSwitch(CodeGenFunction &CGF) override {
221 Action.emitUntiedSwitch(CGF);
222 }
223
224 static bool classof(const CGCapturedStmtInfo *Info) {
225 return CGOpenMPRegionInfo::classof(Info) &&
226 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
227 TaskOutlinedRegion;
228 }
229
230private:
231 /// A variable or parameter storing global thread id for OpenMP
232 /// constructs.
233 const VarDecl *ThreadIDVar;
234 /// Action for emitting code for untied tasks.
235 const UntiedTaskActionTy &Action;
236};
237
238/// API for inlined captured statement code generation in OpenMP
239/// constructs.
240class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
241public:
242 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
243 const RegionCodeGenTy &CodeGen,
244 OpenMPDirectiveKind Kind, bool HasCancel)
245 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel),
246 OldCSI(OldCSI),
247 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
248
249 // Retrieve the value of the context parameter.
250 llvm::Value *getContextValue() const override {
251 if (OuterRegionInfo)
252 return OuterRegionInfo->getContextValue();
253 llvm_unreachable("No context value for inlined OpenMP region");
254 }
255
256 void setContextValue(llvm::Value *V) override {
257 if (OuterRegionInfo) {
258 OuterRegionInfo->setContextValue(V);
259 return;
260 }
261 llvm_unreachable("No context value for inlined OpenMP region");
262 }
263
264 /// Lookup the captured field decl for a variable.
265 const FieldDecl *lookup(const VarDecl *VD) const override {
266 if (OuterRegionInfo)
267 return OuterRegionInfo->lookup(VD);
268 // If there is no outer outlined region,no need to lookup in a list of
269 // captured variables, we can use the original one.
270 return nullptr;
271 }
272
273 FieldDecl *getThisFieldDecl() const override {
274 if (OuterRegionInfo)
275 return OuterRegionInfo->getThisFieldDecl();
276 return nullptr;
277 }
278
279 /// Get a variable or parameter for storing global thread id
280 /// inside OpenMP construct.
281 const VarDecl *getThreadIDVariable() const override {
282 if (OuterRegionInfo)
283 return OuterRegionInfo->getThreadIDVariable();
284 return nullptr;
285 }
286
287 /// Get an LValue for the current ThreadID variable.
288 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override {
289 if (OuterRegionInfo)
290 return OuterRegionInfo->getThreadIDVariableLValue(CGF);
291 llvm_unreachable("No LValue for inlined OpenMP construct");
292 }
293
294 /// Get the name of the capture helper.
295 StringRef getHelperName() const override {
296 if (auto *OuterRegionInfo = getOldCSI())
297 return OuterRegionInfo->getHelperName();
298 llvm_unreachable("No helper name for inlined OpenMP construct");
299 }
300
301 void emitUntiedSwitch(CodeGenFunction &CGF) override {
302 if (OuterRegionInfo)
303 OuterRegionInfo->emitUntiedSwitch(CGF);
304 }
305
306 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
307
308 static bool classof(const CGCapturedStmtInfo *Info) {
309 return CGOpenMPRegionInfo::classof(Info) &&
310 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion;
311 }
312
313 ~CGOpenMPInlinedRegionInfo() override = default;
314
315private:
316 /// CodeGen info about outer OpenMP region.
317 CodeGenFunction::CGCapturedStmtInfo *OldCSI;
318 CGOpenMPRegionInfo *OuterRegionInfo;
319};
320
321/// API for captured statement code generation in OpenMP target
322/// constructs. For this captures, implicit parameters are used instead of the
323/// captured fields. The name of the target region has to be unique in a given
324/// application so it is provided by the client, because only the client has
325/// the information to generate that.
326class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo {
327public:
328 CGOpenMPTargetRegionInfo(const CapturedStmt &CS,
329 const RegionCodeGenTy &CodeGen, StringRef HelperName)
330 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target,
331 /*HasCancel=*/false),
332 HelperName(HelperName) {}
333
334 /// This is unused for target regions because each starts executing
335 /// with a single thread.
336 const VarDecl *getThreadIDVariable() const override { return nullptr; }
337
338 /// Get the name of the capture helper.
339 StringRef getHelperName() const override { return HelperName; }
340
341 static bool classof(const CGCapturedStmtInfo *Info) {
342 return CGOpenMPRegionInfo::classof(Info) &&
343 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion;
344 }
345
346private:
347 StringRef HelperName;
348};
349
350static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) {
351 llvm_unreachable("No codegen for expressions");
352}
353/// API for generation of expressions captured in a innermost OpenMP
354/// region.
355class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo {
356public:
357 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS)
358 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen,
359 OMPD_unknown,
360 /*HasCancel=*/false),
361 PrivScope(CGF) {
362 // Make sure the globals captured in the provided statement are local by
363 // using the privatization logic. We assume the same variable is not
364 // captured more than once.
365 for (const auto &C : CS.captures()) {
366 if (!C.capturesVariable() && !C.capturesVariableByCopy())
367 continue;
368
369 const VarDecl *VD = C.getCapturedVar();
370 if (VD->isLocalVarDeclOrParm())
371 continue;
372
373 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
374 /*RefersToEnclosingVariableOrCapture=*/false,
375 VD->getType().getNonReferenceType(), VK_LValue,
376 C.getLocation());
377 PrivScope.addPrivate(VD, CGF.EmitLValue(&DRE).getAddress());
378 }
379 (void)PrivScope.Privatize();
380 }
381
382 /// Lookup the captured field decl for a variable.
383 const FieldDecl *lookup(const VarDecl *VD) const override {
384 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD))
385 return FD;
386 return nullptr;
387 }
388
389 /// Emit the captured statement body.
390 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override {
391 llvm_unreachable("No body for expressions");
392 }
393
394 /// Get a variable or parameter for storing global thread id
395 /// inside OpenMP construct.
396 const VarDecl *getThreadIDVariable() const override {
397 llvm_unreachable("No thread id for expressions");
398 }
399
400 /// Get the name of the capture helper.
401 StringRef getHelperName() const override {
402 llvm_unreachable("No helper name for expressions");
403 }
404
405 static bool classof(const CGCapturedStmtInfo *Info) { return false; }
406
407private:
408 /// Private scope to capture global variables.
409 CodeGenFunction::OMPPrivateScope PrivScope;
410};
411
412/// RAII for emitting code of OpenMP constructs.
413class InlinedOpenMPRegionRAII {
414 CodeGenFunction &CGF;
415 llvm::DenseMap<const ValueDecl *, FieldDecl *> LambdaCaptureFields;
416 FieldDecl *LambdaThisCaptureField = nullptr;
417 const CodeGen::CGBlockInfo *BlockInfo = nullptr;
418 bool NoInheritance = false;
419
420public:
421 /// Constructs region for combined constructs.
422 /// \param CodeGen Code generation sequence for combined directives. Includes
423 /// a list of functions used for code generation of implicitly inlined
424 /// regions.
425 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen,
426 OpenMPDirectiveKind Kind, bool HasCancel,
427 bool NoInheritance = true)
428 : CGF(CGF), NoInheritance(NoInheritance) {
429 // Start emission for the construct.
430 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(
431 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel);
432 if (NoInheritance) {
433 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
434 LambdaThisCaptureField = CGF.LambdaThisCaptureField;
435 CGF.LambdaThisCaptureField = nullptr;
436 BlockInfo = CGF.BlockInfo;
437 CGF.BlockInfo = nullptr;
438 }
439 }
440
441 ~InlinedOpenMPRegionRAII() {
442 // Restore original CapturedStmtInfo only if we're done with code emission.
443 auto *OldCSI =
444 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
445 delete CGF.CapturedStmtInfo;
446 CGF.CapturedStmtInfo = OldCSI;
447 if (NoInheritance) {
448 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
449 CGF.LambdaThisCaptureField = LambdaThisCaptureField;
450 CGF.BlockInfo = BlockInfo;
451 }
452 }
453};
454
455/// Values for bit flags used in the ident_t to describe the fields.
456/// All enumeric elements are named and described in accordance with the code
457/// from https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h
458enum OpenMPLocationFlags : unsigned {
459 /// Use trampoline for internal microtask.
460 OMP_IDENT_IMD = 0x01,
461 /// Use c-style ident structure.
462 OMP_IDENT_KMPC = 0x02,
463 /// Atomic reduction option for kmpc_reduce.
464 OMP_ATOMIC_REDUCE = 0x10,
465 /// Explicit 'barrier' directive.
466 OMP_IDENT_BARRIER_EXPL = 0x20,
467 /// Implicit barrier in code.
468 OMP_IDENT_BARRIER_IMPL = 0x40,
469 /// Implicit barrier in 'for' directive.
470 OMP_IDENT_BARRIER_IMPL_FOR = 0x40,
471 /// Implicit barrier in 'sections' directive.
472 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0,
473 /// Implicit barrier in 'single' directive.
474 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140,
475 /// Call of __kmp_for_static_init for static loop.
476 OMP_IDENT_WORK_LOOP = 0x200,
477 /// Call of __kmp_for_static_init for sections.
478 OMP_IDENT_WORK_SECTIONS = 0x400,
479 /// Call of __kmp_for_static_init for distribute.
480 OMP_IDENT_WORK_DISTRIBUTE = 0x800,
481 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE)
482};
483
484/// Describes ident structure that describes a source location.
485/// All descriptions are taken from
486/// https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h
487/// Original structure:
488/// typedef struct ident {
489/// kmp_int32 reserved_1; /**< might be used in Fortran;
490/// see above */
491/// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags;
492/// KMP_IDENT_KMPC identifies this union
493/// member */
494/// kmp_int32 reserved_2; /**< not really used in Fortran any more;
495/// see above */
496///#if USE_ITT_BUILD
497/// /* but currently used for storing
498/// region-specific ITT */
499/// /* contextual information. */
500///#endif /* USE_ITT_BUILD */
501/// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for
502/// C++ */
503/// char const *psource; /**< String describing the source location.
504/// The string is composed of semi-colon separated
505// fields which describe the source file,
506/// the function and a pair of line numbers that
507/// delimit the construct.
508/// */
509/// } ident_t;
510enum IdentFieldIndex {
511 /// might be used in Fortran
512 IdentField_Reserved_1,
513 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member.
514 IdentField_Flags,
515 /// Not really used in Fortran any more
516 IdentField_Reserved_2,
517 /// Source[4] in Fortran, do not use for C++
518 IdentField_Reserved_3,
519 /// String describing the source location. The string is composed of
520 /// semi-colon separated fields which describe the source file, the function
521 /// and a pair of line numbers that delimit the construct.
522 IdentField_PSource
523};
524
525/// Schedule types for 'omp for' loops (these enumerators are taken from
526/// the enum sched_type in kmp.h).
527enum OpenMPSchedType {
528 /// Lower bound for default (unordered) versions.
529 OMP_sch_lower = 32,
530 OMP_sch_static_chunked = 33,
531 OMP_sch_static = 34,
532 OMP_sch_dynamic_chunked = 35,
533 OMP_sch_guided_chunked = 36,
534 OMP_sch_runtime = 37,
535 OMP_sch_auto = 38,
536 /// static with chunk adjustment (e.g., simd)
537 OMP_sch_static_balanced_chunked = 45,
538 /// Lower bound for 'ordered' versions.
539 OMP_ord_lower = 64,
540 OMP_ord_static_chunked = 65,
541 OMP_ord_static = 66,
542 OMP_ord_dynamic_chunked = 67,
543 OMP_ord_guided_chunked = 68,
544 OMP_ord_runtime = 69,
545 OMP_ord_auto = 70,
546 OMP_sch_default = OMP_sch_static,
547 /// dist_schedule types
548 OMP_dist_sch_static_chunked = 91,
549 OMP_dist_sch_static = 92,
550 /// Fused distribute+for static schedule (entityId = team*nthreads + tid,
551 /// num_entities = nteams*nthreads). One for_static_init call, no
552 /// surrounding distribute_static_init. Matches
553 /// kmp_sched_distr_static_chunk_sched_static_chunkone in the device RTL
554 /// (openmp/device/include/DeviceTypes.h).
555 OMP_dist_sch_static_chunked_sch_static_chunkone = 93,
556 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers.
557 /// Set if the monotonic schedule modifier was present.
558 OMP_sch_modifier_monotonic = (1 << 29),
559 /// Set if the nonmonotonic schedule modifier was present.
560 OMP_sch_modifier_nonmonotonic = (1 << 30),
561};
562
563/// A basic class for pre|post-action for advanced codegen sequence for OpenMP
564/// region.
565class CleanupTy final : public EHScopeStack::Cleanup {
566 PrePostActionTy *Action;
567
568public:
569 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {}
570 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
571 if (!CGF.HaveInsertPoint())
572 return;
573 Action->Exit(CGF);
574 }
575};
576
577} // anonymous namespace
578
581 if (PrePostAction) {
582 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction);
583 Callback(CodeGen, CGF, *PrePostAction);
584 } else {
585 PrePostActionTy Action;
586 Callback(CodeGen, CGF, Action);
587 }
588}
589
590/// Check if the combiner is a call to UDR combiner and if it is so return the
591/// UDR decl used for reduction.
592static const OMPDeclareReductionDecl *
593getReductionInit(const Expr *ReductionOp) {
594 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
595 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
596 if (const auto *DRE =
597 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
598 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl()))
599 return DRD;
600 return nullptr;
601}
602
604 const OMPDeclareReductionDecl *DRD,
605 const Expr *InitOp,
606 Address Private, Address Original,
607 QualType Ty) {
608 if (DRD->getInitializer()) {
609 std::pair<llvm::Function *, llvm::Function *> Reduction =
611 const auto *CE = cast<CallExpr>(InitOp);
612 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee());
613 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts();
614 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts();
615 const auto *LHSDRE =
616 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr());
617 const auto *RHSDRE =
618 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr());
619 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
620 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), Private);
621 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), Original);
622 (void)PrivateScope.Privatize();
625 CGF.EmitIgnoredExpr(InitOp);
626 } else {
627 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty);
628 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"});
629 auto *GV = new llvm::GlobalVariable(
630 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true,
631 llvm::GlobalValue::PrivateLinkage, Init, Name);
632 LValue LV = CGF.MakeNaturalAlignRawAddrLValue(GV, Ty);
633 RValue InitRVal;
634 switch (CGF.getEvaluationKind(Ty)) {
635 case TEK_Scalar:
636 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation());
637 break;
638 case TEK_Complex:
639 InitRVal =
641 break;
642 case TEK_Aggregate: {
643 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_LValue);
644 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, LV);
645 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
646 /*IsInitializer=*/false);
647 return;
648 }
649 }
650 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_PRValue);
651 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal);
652 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
653 /*IsInitializer=*/false);
654 }
655}
656
657/// Emit initialization of arrays of complex types.
658/// \param DestAddr Address of the array.
659/// \param Type Type of array.
660/// \param Init Initial expression of array.
661/// \param SrcAddr Address of the original array.
663 QualType Type, bool EmitDeclareReductionInit,
664 const Expr *Init,
665 const OMPDeclareReductionDecl *DRD,
666 Address SrcAddr = Address::invalid()) {
667 // Perform element-by-element initialization.
668 QualType ElementTy;
669
670 // Drill down to the base element type on both arrays.
671 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
672 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr);
673 if (DRD)
674 SrcAddr = SrcAddr.withElementType(DestAddr.getElementType());
675
676 llvm::Value *SrcBegin = nullptr;
677 if (DRD)
678 SrcBegin = SrcAddr.emitRawPointer(CGF);
679 llvm::Value *DestBegin = DestAddr.emitRawPointer(CGF);
680 // Cast from pointer to array type to pointer to single element.
681 llvm::Value *DestEnd =
682 CGF.Builder.CreateGEP(DestAddr.getElementType(), DestBegin, NumElements);
683 // The basic structure here is a while-do loop.
684 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body");
685 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done");
686 llvm::Value *IsEmpty =
687 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty");
688 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
689
690 // Enter the loop body, making that address the current address.
691 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
692 CGF.EmitBlock(BodyBB);
693
694 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
695
696 llvm::PHINode *SrcElementPHI = nullptr;
697 Address SrcElementCurrent = Address::invalid();
698 if (DRD) {
699 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2,
700 "omp.arraycpy.srcElementPast");
701 SrcElementPHI->addIncoming(SrcBegin, EntryBB);
702 SrcElementCurrent =
703 Address(SrcElementPHI, SrcAddr.getElementType(),
704 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize));
705 }
706 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI(
707 DestBegin->getType(), 2, "omp.arraycpy.destElementPast");
708 DestElementPHI->addIncoming(DestBegin, EntryBB);
709 Address DestElementCurrent =
710 Address(DestElementPHI, DestAddr.getElementType(),
711 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize));
712
713 // Emit copy.
714 {
716 if (EmitDeclareReductionInit) {
717 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent,
718 SrcElementCurrent, ElementTy);
719 } else
720 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(),
721 /*IsInitializer=*/false);
722 }
723
724 if (DRD) {
725 // Shift the address forward by one element.
726 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32(
727 SrcAddr.getElementType(), SrcElementPHI, /*Idx0=*/1,
728 "omp.arraycpy.dest.element");
729 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock());
730 }
731
732 // Shift the address forward by one element.
733 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32(
734 DestAddr.getElementType(), DestElementPHI, /*Idx0=*/1,
735 "omp.arraycpy.dest.element");
736 // Check whether we've reached the end.
737 llvm::Value *Done =
738 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done");
739 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
740 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock());
741
742 // Done.
743 CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
744}
745
746LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) {
747 return CGF.EmitOMPSharedLValue(E);
748}
749
750LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF,
751 const Expr *E) {
752 if (const auto *OASE = dyn_cast<ArraySectionExpr>(E))
753 return CGF.EmitArraySectionExpr(OASE, /*IsLowerBound=*/false);
754 return LValue();
755}
756
757void ReductionCodeGen::emitAggregateInitialization(
758 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr,
759 const OMPDeclareReductionDecl *DRD) {
760 // Emit VarDecl with copy init for arrays.
761 // Get the address of the original variable captured in current
762 // captured region.
763 const auto *PrivateVD =
764 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
765 bool EmitDeclareReductionInit =
766 DRD && (DRD->getInitializer() || !PrivateVD->hasInit());
767 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(),
768 EmitDeclareReductionInit,
769 EmitDeclareReductionInit ? ClausesData[N].ReductionOp
770 : PrivateVD->getInit(),
771 DRD, SharedAddr);
772}
773
777 ArrayRef<const Expr *> ReductionOps) {
778 ClausesData.reserve(Shareds.size());
779 SharedAddresses.reserve(Shareds.size());
780 Sizes.reserve(Shareds.size());
781 BaseDecls.reserve(Shareds.size());
782 const auto *IOrig = Origs.begin();
783 const auto *IPriv = Privates.begin();
784 const auto *IRed = ReductionOps.begin();
785 for (const Expr *Ref : Shareds) {
786 ClausesData.emplace_back(Ref, *IOrig, *IPriv, *IRed);
787 std::advance(IOrig, 1);
788 std::advance(IPriv, 1);
789 std::advance(IRed, 1);
790 }
791}
792
794 assert(SharedAddresses.size() == N && OrigAddresses.size() == N &&
795 "Number of generated lvalues must be exactly N.");
796 LValue First = emitSharedLValue(CGF, ClausesData[N].Shared);
797 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Shared);
798 SharedAddresses.emplace_back(First, Second);
799 if (ClausesData[N].Shared == ClausesData[N].Ref) {
800 OrigAddresses.emplace_back(First, Second);
801 } else {
802 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref);
803 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref);
804 OrigAddresses.emplace_back(First, Second);
805 }
806}
807
809 QualType PrivateType = getPrivateType(N);
810 bool AsArraySection = isa<ArraySectionExpr>(ClausesData[N].Ref);
811 if (!PrivateType->isVariablyModifiedType()) {
812 Sizes.emplace_back(
813 CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType()),
814 nullptr);
815 return;
816 }
817 llvm::Value *Size;
818 llvm::Value *SizeInChars;
819 auto *ElemType = OrigAddresses[N].first.getAddress().getElementType();
820 auto *ElemSizeOf = llvm::ConstantInt::get(
821 CGF.SizeTy, CGF.CGM.getDataLayout().getTypeAllocSize(ElemType));
822 if (AsArraySection) {
823 SizeInChars =
824 CGF.Builder.CreatePtrDiff(OrigAddresses[N].second.getPointer(CGF),
825 OrigAddresses[N].first.getPointer(CGF));
826 SizeInChars = CGF.Builder.CreateNUWAdd(SizeInChars, ElemSizeOf);
827 } else {
828 SizeInChars =
829 CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType());
830 }
831 Size = ElemSizeOf->isOne()
832 ? SizeInChars
833 : CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf);
834 Sizes.emplace_back(SizeInChars, Size);
836 CGF,
838 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
839 RValue::get(Size));
840 CGF.EmitVariablyModifiedType(PrivateType);
841}
842
844 llvm::Value *Size) {
845 QualType PrivateType = getPrivateType(N);
846 if (!PrivateType->isVariablyModifiedType()) {
847 assert(!Size && !Sizes[N].second &&
848 "Size should be nullptr for non-variably modified reduction "
849 "items.");
850 return;
851 }
853 CGF,
855 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
856 RValue::get(Size));
857 CGF.EmitVariablyModifiedType(PrivateType);
858}
859
861 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr,
862 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) {
863 assert(SharedAddresses.size() > N && "No variable was generated");
864 const auto *PrivateVD =
865 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
866 const OMPDeclareReductionDecl *DRD =
867 getReductionInit(ClausesData[N].ReductionOp);
868 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) {
869 if (DRD && DRD->getInitializer())
870 (void)DefaultInit(CGF);
871 emitAggregateInitialization(CGF, N, PrivateAddr, SharedAddr, DRD);
872 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) {
873 (void)DefaultInit(CGF);
874 QualType SharedType = SharedAddresses[N].first.getType();
875 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp,
876 PrivateAddr, SharedAddr, SharedType);
877 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() &&
878 !CGF.isTrivialInitializer(PrivateVD->getInit())) {
879 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr,
880 PrivateVD->getType().getQualifiers(),
881 /*IsInitializer=*/false);
882 }
883}
884
886 QualType PrivateType = getPrivateType(N);
887 QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
888 return DTorKind != QualType::DK_none;
889}
890
892 Address PrivateAddr) {
893 QualType PrivateType = getPrivateType(N);
894 QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
895 if (needCleanups(N)) {
896 PrivateAddr =
897 PrivateAddr.withElementType(CGF.ConvertTypeForMem(PrivateType));
898 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType);
899 }
900}
901
902static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
903 LValue BaseLV) {
904 BaseTy = BaseTy.getNonReferenceType();
905 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
906 !CGF.getContext().hasSameType(BaseTy, ElTy)) {
907 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) {
908 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(), PtrTy);
909 } else {
910 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(), BaseTy);
911 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal);
912 }
913 BaseTy = BaseTy->getPointeeType();
914 }
915 return CGF.MakeAddrLValue(
916 BaseLV.getAddress().withElementType(CGF.ConvertTypeForMem(ElTy)),
917 BaseLV.getType(), BaseLV.getBaseInfo(),
918 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType()));
919}
920
922 Address OriginalBaseAddress, llvm::Value *Addr) {
924 Address TopTmp = Address::invalid();
925 Address MostTopTmp = Address::invalid();
926 BaseTy = BaseTy.getNonReferenceType();
927 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
928 !CGF.getContext().hasSameType(BaseTy, ElTy)) {
929 Tmp = CGF.CreateMemTempWithoutCast(BaseTy);
930 if (TopTmp.isValid())
931 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp);
932 else
933 MostTopTmp = Tmp;
934 TopTmp = Tmp;
935 BaseTy = BaseTy->getPointeeType();
936 }
937
938 if (Tmp.isValid()) {
940 Addr, Tmp.getElementType());
941 CGF.Builder.CreateStore(Addr, Tmp);
942 return MostTopTmp;
943 }
944
946 Addr, OriginalBaseAddress.getType());
947 return OriginalBaseAddress.withPointer(Addr, NotKnownNonNull);
948}
949
950static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) {
951 const VarDecl *OrigVD = nullptr;
952 if (const auto *OASE = dyn_cast<ArraySectionExpr>(Ref)) {
953 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
954 while (const auto *TempOASE = dyn_cast<ArraySectionExpr>(Base))
955 Base = TempOASE->getBase()->IgnoreParenImpCasts();
956 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
957 Base = TempASE->getBase()->IgnoreParenImpCasts();
959 OrigVD = cast<VarDecl>(DE->getDecl());
960 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) {
961 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts();
962 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
963 Base = TempASE->getBase()->IgnoreParenImpCasts();
965 OrigVD = cast<VarDecl>(DE->getDecl());
966 }
967 return OrigVD;
968}
969
971 Address PrivateAddr) {
972 const DeclRefExpr *DE;
973 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) {
974 BaseDecls.emplace_back(OrigVD);
975 LValue OriginalBaseLValue = CGF.EmitLValue(DE);
976 LValue BaseLValue =
977 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(),
978 OriginalBaseLValue);
979 Address SharedAddr = SharedAddresses[N].first.getAddress();
980 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff(
981 SharedAddr.getElementType(), BaseLValue.getPointer(CGF),
982 SharedAddr.emitRawPointer(CGF));
983 llvm::Value *PrivatePointer =
985 PrivateAddr.emitRawPointer(CGF), SharedAddr.getType());
986 llvm::Value *Ptr = CGF.Builder.CreateGEP(
987 SharedAddr.getElementType(), PrivatePointer, Adjustment);
988 return castToBase(CGF, OrigVD->getType(),
989 SharedAddresses[N].first.getType(),
990 OriginalBaseLValue.getAddress(), Ptr);
991 }
992 BaseDecls.emplace_back(
993 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl()));
994 return PrivateAddr;
995}
996
998 const OMPDeclareReductionDecl *DRD =
999 getReductionInit(ClausesData[N].ReductionOp);
1000 return DRD && DRD->getInitializer();
1001}
1002
1003LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
1004 return CGF.EmitLoadOfPointerLValue(
1005 CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1006 getThreadIDVariable()->getType()->castAs<PointerType>());
1007}
1008
1009void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt *S) {
1010 if (!CGF.HaveInsertPoint())
1011 return;
1012 // 1.2.2 OpenMP Language Terminology
1013 // Structured block - An executable statement with a single entry at the
1014 // top and a single exit at the bottom.
1015 // The point of exit cannot be a branch out of the structured block.
1016 // longjmp() and throw() must not violate the entry/exit criteria.
1017 CGF.EHStack.pushTerminate();
1018 if (S)
1020 CodeGen(CGF);
1021 CGF.EHStack.popTerminate();
1022}
1023
1024LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
1025 CodeGenFunction &CGF) {
1026 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1027 getThreadIDVariable()->getType(),
1029}
1030
1032 QualType FieldTy) {
1033 auto *Field = FieldDecl::Create(
1034 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1035 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1036 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1037 Field->setAccess(AS_public);
1038 DC->addDecl(Field);
1039 return Field;
1040}
1041
1043 : CGM(CGM), OMPBuilder(CGM.getModule()) {
1044 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
1045 llvm::OpenMPIRBuilderConfig Config(
1046 CGM.getLangOpts().OpenMPIsTargetDevice, isGPU(),
1047 CGM.getLangOpts().OpenMPOffloadMandatory,
1048 /*HasRequiresReverseOffload*/ false, /*HasRequiresUnifiedAddress*/ false,
1049 hasRequiresUnifiedSharedMemory(), /*HasRequiresDynamicAllocators*/ false);
1050 Config.setDefaultTargetAS(
1051 CGM.getContext().getTargetInfo().getTargetAddressSpace(LangAS::Default));
1052 Config.setRuntimeCC(CGM.getRuntimeCC());
1053
1054 OMPBuilder.setConfig(Config);
1055 OMPBuilder.initialize();
1056 OMPBuilder.loadOffloadInfoMetadata(*CGM.getFileSystem(),
1057 CGM.getLangOpts().OpenMPIsTargetDevice
1058 ? CGM.getLangOpts().OMPHostIRFile
1059 : StringRef{});
1060
1061 // The user forces the compiler to behave as if omp requires
1062 // unified_shared_memory was given.
1063 if (CGM.getLangOpts().OpenMPForceUSM) {
1065 OMPBuilder.Config.setHasRequiresUnifiedSharedMemory(true);
1066 }
1067}
1068
1070 InternalVars.clear();
1071 // Clean non-target variable declarations possibly used only in debug info.
1072 for (const auto &Data : EmittedNonTargetVariables) {
1073 if (!Data.getValue().pointsToAliveValue())
1074 continue;
1075 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue());
1076 if (!GV)
1077 continue;
1078 if (!GV->isDeclaration() || GV->getNumUses() > 0)
1079 continue;
1080 GV->eraseFromParent();
1081 }
1082}
1083
1085 return OMPBuilder.createPlatformSpecificName(Parts);
1086}
1087
1088static llvm::Function *
1090 const Expr *CombinerInitializer, const VarDecl *In,
1091 const VarDecl *Out, bool IsCombiner) {
1092 // void .omp_combiner.(Ty *in, Ty *out);
1093 ASTContext &C = CGM.getContext();
1094 QualType PtrTy = C.getPointerType(Ty).withRestrict();
1095 auto *OmpOutParm = ImplicitParamDecl::Create(
1096 C, /*DC=*/nullptr, Out->getLocation(),
1097 /*Id=*/nullptr, PtrTy, ImplicitParamKind::Other);
1098 auto *OmpInParm = ImplicitParamDecl::Create(
1099 C, /*DC=*/nullptr, In->getLocation(),
1100 /*Id=*/nullptr, PtrTy, ImplicitParamKind::Other);
1101 FunctionArgList Args{OmpOutParm, OmpInParm};
1102 const CGFunctionInfo &FnInfo =
1103 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
1104 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
1105 std::string Name = CGM.getOpenMPRuntime().getName(
1106 {IsCombiner ? "omp_combiner" : "omp_initializer", ""});
1107 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
1108 Name, &CGM.getModule());
1109 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
1110 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
1111 Fn->addFnAttr("sample-profile-suffix-elision-policy", "selected");
1112 if (CGM.getCodeGenOpts().OptimizationLevel != 0) {
1113 Fn->removeFnAttr(llvm::Attribute::NoInline);
1114 Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
1115 Fn->addFnAttr(llvm::Attribute::AlwaysInline);
1116 }
1117 CodeGenFunction CGF(CGM);
1118 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions.
1119 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions.
1120 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(),
1121 Out->getLocation());
1123 Address AddrIn = CGF.GetAddrOfLocalVar(OmpInParm);
1124 Scope.addPrivate(
1125 In, CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>())
1126 .getAddress());
1127 Address AddrOut = CGF.GetAddrOfLocalVar(OmpOutParm);
1128 Scope.addPrivate(
1129 Out, CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>())
1130 .getAddress());
1131 (void)Scope.Privatize();
1132 if (!IsCombiner && Out->hasInit() &&
1133 !CGF.isTrivialInitializer(Out->getInit())) {
1134 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out),
1135 Out->getType().getQualifiers(),
1136 /*IsInitializer=*/true);
1137 }
1138 if (CombinerInitializer)
1139 CGF.EmitIgnoredExpr(CombinerInitializer);
1140 Scope.ForceCleanup();
1141 CGF.FinishFunction();
1142 return Fn;
1143}
1144
1147 if (UDRMap.count(D) > 0)
1148 return;
1149 llvm::Function *Combiner = emitCombinerOrInitializer(
1150 CGM, D->getType(), D->getCombiner(),
1153 /*IsCombiner=*/true);
1154 llvm::Function *Initializer = nullptr;
1155 if (const Expr *Init = D->getInitializer()) {
1157 CGM, D->getType(),
1159 : nullptr,
1162 /*IsCombiner=*/false);
1163 }
1164 UDRMap.try_emplace(D, Combiner, Initializer);
1165 if (CGF)
1166 FunctionUDRMap[CGF->CurFn].push_back(D);
1167}
1168
1169std::pair<llvm::Function *, llvm::Function *>
1171 auto I = UDRMap.find(D);
1172 if (I != UDRMap.end())
1173 return I->second;
1174 emitUserDefinedReduction(/*CGF=*/nullptr, D);
1175 return UDRMap.lookup(D);
1176}
1177
1178namespace {
1179// Temporary RAII solution to perform a push/pop stack event on the OpenMP IR
1180// Builder if one is present.
1181struct PushAndPopStackRAII {
1182 PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF,
1183 bool HasCancel, llvm::omp::Directive Kind)
1184 : OMPBuilder(OMPBuilder) {
1185 if (!OMPBuilder)
1186 return;
1187
1188 // The following callback is the crucial part of clangs cleanup process.
1189 //
1190 // NOTE:
1191 // Once the OpenMPIRBuilder is used to create parallel regions (and
1192 // similar), the cancellation destination (Dest below) is determined via
1193 // IP. That means if we have variables to finalize we split the block at IP,
1194 // use the new block (=BB) as destination to build a JumpDest (via
1195 // getJumpDestInCurrentScope(BB)) which then is fed to
1196 // EmitBranchThroughCleanup. Furthermore, there will not be the need
1197 // to push & pop an FinalizationInfo object.
1198 // The FiniCB will still be needed but at the point where the
1199 // OpenMPIRBuilder is asked to construct a parallel (or similar) construct.
1200 auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) {
1201 assert(IP == IP.getNodeParent()->end() &&
1202 "Clang CG should cause non-terminated block!");
1203 CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1204 CGF.Builder.restoreIP(IP);
1206 CGF.getOMPCancelDestination(OMPD_parallel);
1207 CGF.EmitBranchThroughCleanup(Dest);
1208 return llvm::Error::success();
1209 };
1210
1211 // TODO: Remove this once we emit parallel regions through the
1212 // OpenMPIRBuilder as it can do this setup internally.
1213 llvm::OpenMPIRBuilder::FinalizationInfo FI({FiniCB, Kind, HasCancel});
1214 OMPBuilder->pushFinalizationCB(std::move(FI));
1215 }
1216 ~PushAndPopStackRAII() {
1217 if (OMPBuilder)
1218 OMPBuilder->popFinalizationCB();
1219 }
1220 llvm::OpenMPIRBuilder *OMPBuilder;
1221};
1222} // namespace
1223
1225 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS,
1226 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1227 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) {
1228 assert(ThreadIDVar->getType()->isPointerType() &&
1229 "thread id variable must be of type kmp_int32 *");
1230 CodeGenFunction CGF(CGM, true);
1231 bool HasCancel = false;
1232 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D))
1233 HasCancel = OPD->hasCancel();
1234 else if (const auto *OPD = dyn_cast<OMPTargetParallelDirective>(&D))
1235 HasCancel = OPD->hasCancel();
1236 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D))
1237 HasCancel = OPSD->hasCancel();
1238 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D))
1239 HasCancel = OPFD->hasCancel();
1240 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D))
1241 HasCancel = OPFD->hasCancel();
1242 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D))
1243 HasCancel = OPFD->hasCancel();
1244 else if (const auto *OPFD =
1245 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D))
1246 HasCancel = OPFD->hasCancel();
1247 else if (const auto *OPFD =
1248 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D))
1249 HasCancel = OPFD->hasCancel();
1250
1251 // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new
1252 // parallel region to make cancellation barriers work properly.
1253 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
1254 PushAndPopStackRAII PSR(&OMPBuilder, CGF, HasCancel, InnermostKind);
1255 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind,
1256 HasCancel, OutlinedHelperName);
1257 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1258 return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D);
1259}
1260
1261std::string CGOpenMPRuntime::getOutlinedHelperName(StringRef Name) const {
1262 std::string Suffix = getName({"omp_outlined"});
1263 return (Name + Suffix).str();
1264}
1265
1267 return getOutlinedHelperName(CGF.CurFn->getName());
1268}
1269
1270std::string CGOpenMPRuntime::getReductionFuncName(StringRef Name) const {
1271 std::string Suffix = getName({"omp", "reduction", "reduction_func"});
1272 return (Name + Suffix).str();
1273}
1274
1277 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1278 const RegionCodeGenTy &CodeGen) {
1279 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel);
1281 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(CGF),
1282 CodeGen);
1283}
1284
1287 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1288 const RegionCodeGenTy &CodeGen) {
1289 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams);
1290 llvm::Function *OutlinedFn = emitParallelOrTeamsOutlinedFunction(
1291 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(CGF),
1292 CodeGen);
1293 // A teams body is called once per team and is not handed back to the runtime
1294 // as a callback, so unlike a parallel body it cannot be re-entered while a
1295 // call to it is live.
1296 OutlinedFn->setDoesNotRecurse();
1297 return OutlinedFn;
1298}
1299
1301 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1302 const VarDecl *PartIDVar, const VarDecl *TaskTVar,
1303 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
1304 bool Tied, unsigned &NumberOfParts) {
1305 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF,
1306 PrePostActionTy &) {
1307 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc());
1308 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc());
1309 llvm::Value *TaskArgs[] = {
1310 UpLoc, ThreadID,
1311 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar),
1312 TaskTVar->getType()->castAs<PointerType>())
1313 .getPointer(CGF)};
1314 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
1315 CGM.getModule(), OMPRTL___kmpc_omp_task),
1316 TaskArgs);
1317 };
1318 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar,
1319 UntiedCodeGen);
1320 CodeGen.setAction(Action);
1321 assert(!ThreadIDVar->getType()->isPointerType() &&
1322 "thread id variable must be of type kmp_int32 for tasks");
1323 const OpenMPDirectiveKind Region =
1324 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop
1325 : OMPD_task;
1326 const CapturedStmt *CS = D.getCapturedStmt(Region);
1327 bool HasCancel = false;
1328 if (const auto *TD = dyn_cast<OMPTaskDirective>(&D))
1329 HasCancel = TD->hasCancel();
1330 else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D))
1331 HasCancel = TD->hasCancel();
1332 else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D))
1333 HasCancel = TD->hasCancel();
1334 else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D))
1335 HasCancel = TD->hasCancel();
1336
1337 CodeGenFunction CGF(CGM, true);
1338 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen,
1339 InnermostKind, HasCancel, Action);
1340 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1341 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS);
1342 if (!Tied)
1343 NumberOfParts = Action.getNumberOfParts();
1344 return Res;
1345}
1346
1348 bool AtCurrentPoint) {
1349 auto &Elem = OpenMPLocThreadIDMap[CGF.CurFn];
1350 assert(!Elem.ServiceInsertPt && "Insert point is set already.");
1351
1352 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty);
1353 if (AtCurrentPoint) {
1354 Elem.ServiceInsertPt = new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt",
1355 CGF.Builder.GetInsertBlock());
1356 } else {
1357 Elem.ServiceInsertPt = new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt");
1358 Elem.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt->getIterator());
1359 }
1360}
1361
1363 auto &Elem = OpenMPLocThreadIDMap[CGF.CurFn];
1364 if (Elem.ServiceInsertPt) {
1365 llvm::Instruction *Ptr = Elem.ServiceInsertPt;
1366 Elem.ServiceInsertPt = nullptr;
1367 Ptr->eraseFromParent();
1368 }
1369}
1370
1372 SourceLocation Loc,
1373 SmallString<128> &Buffer) {
1374 llvm::raw_svector_ostream OS(Buffer);
1375 // Build debug location
1377 OS << ";";
1378 if (auto *DbgInfo = CGF.getDebugInfo())
1379 OS << DbgInfo->remapDIPath(PLoc.getFilename());
1380 else
1381 OS << PLoc.getFilename();
1382 OS << ";";
1383 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1384 OS << FD->getQualifiedNameAsString();
1385 OS << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
1386 return OS.str();
1387}
1388
1390 SourceLocation Loc,
1391 unsigned Flags, bool EmitLoc) {
1392 uint32_t SrcLocStrSize;
1393 llvm::Constant *SrcLocStr;
1394 if ((!EmitLoc && CGM.getCodeGenOpts().getDebugInfo() ==
1395 llvm::codegenoptions::NoDebugInfo) ||
1396 Loc.isInvalid()) {
1397 SrcLocStr = OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize);
1398 } else {
1399 std::string FunctionName;
1400 std::string FileName;
1401 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1402 FunctionName = FD->getQualifiedNameAsString();
1404 if (auto *DbgInfo = CGF.getDebugInfo())
1405 FileName = DbgInfo->remapDIPath(PLoc.getFilename());
1406 else
1407 FileName = PLoc.getFilename();
1408 unsigned Line = PLoc.getLine();
1409 unsigned Column = PLoc.getColumn();
1410 SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(FunctionName, FileName, Line,
1411 Column, SrcLocStrSize);
1412 }
1413 unsigned Reserved2Flags = getDefaultLocationReserved2Flags();
1414 return OMPBuilder.getOrCreateIdent(
1415 SrcLocStr, SrcLocStrSize, llvm::omp::IdentFlag(Flags), Reserved2Flags);
1416}
1417
1419 SourceLocation Loc) {
1420 assert(CGF.CurFn && "No function in current CodeGenFunction.");
1421 // If the OpenMPIRBuilder is used we need to use it for all thread id calls as
1422 // the clang invariants used below might be broken.
1423 if (CGM.getLangOpts().OpenMPIRBuilder) {
1424 SmallString<128> Buffer;
1425 OMPBuilder.updateToLocation(CGF.Builder);
1426 uint32_t SrcLocStrSize;
1427 auto *SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(
1428 getIdentStringFromSourceLocation(CGF, Loc, Buffer), SrcLocStrSize);
1429 return OMPBuilder.getOrCreateThreadID(
1430 OMPBuilder.getOrCreateIdent(SrcLocStr, SrcLocStrSize));
1431 }
1432
1433 llvm::Value *ThreadID = nullptr;
1434 // Check whether we've already cached a load of the thread id in this
1435 // function.
1436 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1437 if (I != OpenMPLocThreadIDMap.end()) {
1438 ThreadID = I->second.ThreadID;
1439 if (ThreadID != nullptr)
1440 return ThreadID;
1441 }
1442 // If exceptions are enabled, do not use parameter to avoid possible crash.
1443 if (auto *OMPRegionInfo =
1444 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
1445 if (OMPRegionInfo->getThreadIDVariable()) {
1446 // Check if this an outlined function with thread id passed as argument.
1447 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
1448 llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent();
1449 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions ||
1450 !CGF.getLangOpts().CXXExceptions ||
1451 CGF.Builder.GetInsertBlock() == TopBlock ||
1452 !isa<llvm::Instruction>(LVal.getPointer(CGF)) ||
1453 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1454 TopBlock ||
1455 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1456 CGF.Builder.GetInsertBlock()) {
1457 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc);
1458 // If value loaded in entry block, cache it and use it everywhere in
1459 // function.
1460 if (CGF.Builder.GetInsertBlock() == TopBlock)
1461 OpenMPLocThreadIDMap[CGF.CurFn].ThreadID = ThreadID;
1462 return ThreadID;
1463 }
1464 }
1465 }
1466
1467 // This is not an outlined function region - need to call __kmpc_int32
1468 // kmpc_global_thread_num(ident_t *loc).
1469 // Generate thread id value and cache this value for use across the
1470 // function.
1471 auto &Elem = OpenMPLocThreadIDMap[CGF.CurFn];
1472 if (!Elem.ServiceInsertPt)
1474 CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1475 CGF.Builder.SetInsertPoint(Elem.ServiceInsertPt);
1477 llvm::CallInst *Call = CGF.Builder.CreateCall(
1478 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
1479 OMPRTL___kmpc_global_thread_num),
1480 emitUpdateLocation(CGF, Loc));
1481 Call->setCallingConv(CGF.getRuntimeCC());
1482 Elem.ThreadID = Call;
1483 return Call;
1484}
1485
1487 assert(CGF.CurFn && "No function in current CodeGenFunction.");
1488 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) {
1490 OpenMPLocThreadIDMap.erase(CGF.CurFn);
1491 }
1492 if (auto I = FunctionUDRMap.find(CGF.CurFn); I != FunctionUDRMap.end()) {
1493 for (const auto *D : I->second)
1494 UDRMap.erase(D);
1495 FunctionUDRMap.erase(I);
1496 }
1497 if (auto I = FunctionUDMMap.find(CGF.CurFn); I != FunctionUDMMap.end()) {
1498 for (const auto *D : I->second)
1499 UDMMap.erase(D);
1500 FunctionUDMMap.erase(I);
1501 }
1504}
1505
1507 return OMPBuilder.IdentPtr;
1508}
1509
1510static llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseKind
1512 std::optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
1513 OMPDeclareTargetDeclAttr::getDeviceType(VD);
1514 if (!DevTy)
1515 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseNone;
1516
1517 switch ((int)*DevTy) { // Avoid -Wcovered-switch-default
1518 case OMPDeclareTargetDeclAttr::DT_Host:
1519 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseHost;
1520 break;
1521 case OMPDeclareTargetDeclAttr::DT_NoHost:
1522 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseNoHost;
1523 break;
1524 case OMPDeclareTargetDeclAttr::DT_Any:
1525 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseAny;
1526 break;
1527 default:
1528 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseNone;
1529 break;
1530 }
1531}
1532
1533static llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind
1535 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> MapType =
1536 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
1537 if (!MapType)
1538 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryNone;
1539 switch ((int)*MapType) { // Avoid -Wcovered-switch-default
1540 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_To:
1541 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryTo;
1542 break;
1543 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_Enter:
1544 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryEnter;
1545 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_Link:
1546 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryLink;
1547 break;
1548 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_Local:
1549 // MT_Local variables don't need offload entry (device-local).
1550 llvm_unreachable("MT_Local should not reach convertCaptureClause");
1551 break;
1552 default:
1553 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryNone;
1554 break;
1555 }
1556}
1557
1558static llvm::TargetRegionEntryInfo getEntryInfoFromPresumedLoc(
1559 CodeGenModule &CGM, llvm::OpenMPIRBuilder &OMPBuilder,
1560 SourceLocation BeginLoc, llvm::StringRef ParentName = "") {
1561
1562 auto FileInfoCallBack = [&]() {
1564 PresumedLoc PLoc = SM.getPresumedLoc(BeginLoc);
1565
1566 if (!CGM.getFileSystem()->exists(PLoc.getFilename()))
1567 PLoc = SM.getPresumedLoc(BeginLoc, /*UseLineDirectives=*/false);
1568
1569 return std::pair<std::string, uint64_t>(PLoc.getFilename(), PLoc.getLine());
1570 };
1571
1572 return OMPBuilder.getTargetEntryUniqueInfo(FileInfoCallBack,
1573 *CGM.getFileSystem(), ParentName);
1574}
1575
1577 auto AddrOfGlobal = [&VD, this]() { return CGM.GetAddrOfGlobal(VD); };
1578
1579 auto LinkageForVariable = [&VD, this]() {
1580 return CGM.getLLVMLinkageVarDefinition(VD);
1581 };
1582
1583 std::vector<llvm::GlobalVariable *> GeneratedRefs;
1584
1585 llvm::Type *LlvmPtrTy = CGM.getTypes().ConvertTypeForMem(
1586 CGM.getContext().getPointerType(VD->getType()));
1587 llvm::Constant *addr = OMPBuilder.getAddrOfDeclareTargetVar(
1589 VD->hasDefinition(CGM.getContext()) == VarDecl::DeclarationOnly,
1590 VD->isExternallyVisible(),
1592 VD->getCanonicalDecl()->getBeginLoc()),
1593 CGM.getMangledName(VD), GeneratedRefs, CGM.getLangOpts().OpenMPSimd,
1594 CGM.getLangOpts().OMPTargetTriples, LlvmPtrTy, AddrOfGlobal,
1595 LinkageForVariable);
1596
1597 if (!addr)
1598 return ConstantAddress::invalid();
1599 return ConstantAddress(addr, LlvmPtrTy, CGM.getContext().getDeclAlign(VD));
1600}
1601
1602llvm::Constant *
1604 assert(!CGM.getLangOpts().OpenMPUseTLS ||
1605 !CGM.getContext().getTargetInfo().isTLSSupported());
1606 // Lookup the entry, lazily creating it if necessary.
1607 std::string Suffix = getName({"cache", ""});
1608 return OMPBuilder.getOrCreateInternalVariable(
1609 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix).str());
1610}
1611
1613 const VarDecl *VD,
1614 Address VDAddr,
1615 SourceLocation Loc) {
1616 if (CGM.getLangOpts().OpenMPUseTLS &&
1617 CGM.getContext().getTargetInfo().isTLSSupported())
1618 return VDAddr;
1619
1620 llvm::Type *VarTy = VDAddr.getElementType();
1621 llvm::Value *Args[] = {
1622 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
1623 CGF.Builder.CreatePointerCast(VDAddr.emitRawPointer(CGF), CGM.Int8PtrTy),
1624 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
1626 return Address(
1627 CGF.EmitRuntimeCall(
1628 OMPBuilder.getOrCreateRuntimeFunction(
1629 CGM.getModule(), OMPRTL___kmpc_threadprivate_cached),
1630 Args),
1631 CGF.Int8Ty, VDAddr.getAlignment());
1632}
1633
1635 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor,
1636 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
1637 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
1638 // library.
1639 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc);
1640 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
1641 CGM.getModule(), OMPRTL___kmpc_global_thread_num),
1642 OMPLoc);
1643 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
1644 // to register constructor/destructor for variable.
1645 llvm::Value *Args[] = {
1646 OMPLoc,
1647 CGF.Builder.CreatePointerCast(VDAddr.emitRawPointer(CGF), CGM.VoidPtrTy),
1648 Ctor, CopyCtor, Dtor};
1649 CGF.EmitRuntimeCall(
1650 OMPBuilder.getOrCreateRuntimeFunction(
1651 CGM.getModule(), OMPRTL___kmpc_threadprivate_register),
1652 Args);
1653}
1654
1656 const VarDecl *VD, Address VDAddr, SourceLocation Loc,
1657 bool PerformInit, CodeGenFunction *CGF) {
1658 if (CGM.getLangOpts().OpenMPUseTLS &&
1659 CGM.getContext().getTargetInfo().isTLSSupported())
1660 return nullptr;
1661
1662 VD = VD->getDefinition(CGM.getContext());
1663 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) {
1664 QualType ASTTy = VD->getType();
1665
1666 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
1667 const Expr *Init = VD->getAnyInitializer();
1668 if (CGM.getLangOpts().CPlusPlus && PerformInit) {
1669 // Generate function that re-emits the declaration's initializer into the
1670 // threadprivate copy of the variable VD
1671 CodeGenFunction CtorCGF(CGM);
1672 auto *Dst = ImplicitParamDecl::Create(
1673 CGM.getContext(), /*DC=*/nullptr, Loc,
1674 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, ImplicitParamKind::Other);
1675
1676 FunctionArgList Args{Dst};
1677 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
1678 CGM.getContext().VoidPtrTy, Args);
1679 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
1680 std::string Name = getName({"__kmpc_global_ctor_", ""});
1681 llvm::Function *Fn =
1682 CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc);
1683 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
1684 Args, Loc, Loc);
1685 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar(
1686 CtorCGF.GetAddrOfLocalVar(Dst), /*Volatile=*/false,
1687 CGM.getContext().VoidPtrTy, Dst->getLocation());
1688 Address Arg(ArgVal, CtorCGF.ConvertTypeForMem(ASTTy),
1689 VDAddr.getAlignment());
1690 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
1691 /*IsInitializer=*/true);
1692 ArgVal = CtorCGF.EmitLoadOfScalar(
1693 CtorCGF.GetAddrOfLocalVar(Dst), /*Volatile=*/false,
1694 CGM.getContext().VoidPtrTy, Dst->getLocation());
1695 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
1696 CtorCGF.FinishFunction();
1697 Ctor = Fn;
1698 }
1700 // Generate function that emits destructor call for the threadprivate copy
1701 // of the variable VD
1702 CodeGenFunction DtorCGF(CGM);
1703 auto *Dst = ImplicitParamDecl::Create(
1704 CGM.getContext(), /*DC=*/nullptr, Loc,
1705 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, ImplicitParamKind::Other);
1706
1707 FunctionArgList Args{Dst};
1708 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
1709 CGM.getContext().VoidTy, Args);
1710 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
1711 std::string Name = getName({"__kmpc_global_dtor_", ""});
1712 llvm::Function *Fn =
1713 CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc);
1714 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
1715 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
1716 Loc, Loc);
1717 // Create a scope with an artificial location for the body of this function.
1718 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
1719 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar(
1720 DtorCGF.GetAddrOfLocalVar(Dst),
1721 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst->getLocation());
1722 DtorCGF.emitDestroy(
1723 Address(ArgVal, DtorCGF.Int8Ty, VDAddr.getAlignment()), ASTTy,
1724 DtorCGF.getDestroyer(ASTTy.isDestructedType()),
1725 DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
1726 DtorCGF.FinishFunction();
1727 Dtor = Fn;
1728 }
1729 // Do not emit init function if it is not required.
1730 if (!Ctor && !Dtor)
1731 return nullptr;
1732
1733 // Copying constructor for the threadprivate variable.
1734 // Must be NULL - reserved by runtime, but currently it requires that this
1735 // parameter is always NULL. Otherwise it fires assertion.
1736 CopyCtor = llvm::Constant::getNullValue(CGM.DefaultPtrTy);
1737 if (Ctor == nullptr) {
1738 Ctor = llvm::Constant::getNullValue(CGM.DefaultPtrTy);
1739 }
1740 if (Dtor == nullptr) {
1741 Dtor = llvm::Constant::getNullValue(CGM.DefaultPtrTy);
1742 }
1743 if (!CGF) {
1744 auto *InitFunctionTy =
1745 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
1746 std::string Name = getName({"__omp_threadprivate_init_", ""});
1747 llvm::Function *InitFunction = CGM.CreateGlobalInitOrCleanUpFunction(
1748 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction());
1749 CodeGenFunction InitCGF(CGM);
1750 FunctionArgList ArgList;
1751 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
1752 CGM.getTypes().arrangeNullaryFunction(), ArgList,
1753 Loc, Loc);
1754 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
1755 InitCGF.FinishFunction();
1756 return InitFunction;
1757 }
1758 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
1759 }
1760 return nullptr;
1761}
1762
1764 llvm::GlobalValue *GV) {
1765 std::optional<OMPDeclareTargetDeclAttr *> ActiveAttr =
1766 OMPDeclareTargetDeclAttr::getActiveAttr(FD);
1767
1768 // We only need to handle active 'indirect' declare target functions.
1769 if (!ActiveAttr || !(*ActiveAttr)->getIndirect())
1770 return;
1771
1772 // Get a mangled name to store the new device global in.
1773 llvm::TargetRegionEntryInfo EntryInfo = getEntryInfoFromPresumedLoc(
1775 SmallString<128> Name;
1776 OMPBuilder.OffloadInfoManager.getTargetRegionEntryFnName(Name, EntryInfo);
1777
1778 // We need to generate a new global to hold the address of the indirectly
1779 // called device function. Doing this allows us to keep the visibility and
1780 // linkage of the associated function unchanged while allowing the runtime to
1781 // access its value.
1782 llvm::GlobalValue *Addr = GV;
1783 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
1784 llvm::PointerType *FnPtrTy = llvm::PointerType::get(
1785 CGM.getLLVMContext(),
1786 CGM.getModule().getDataLayout().getProgramAddressSpace());
1787 Addr = new llvm::GlobalVariable(
1788 CGM.getModule(), FnPtrTy,
1789 /*isConstant=*/true, llvm::GlobalValue::ExternalLinkage, GV, Name,
1790 nullptr, llvm::GlobalValue::NotThreadLocal,
1791 CGM.getModule().getDataLayout().getDefaultGlobalsAddressSpace());
1792 Addr->setVisibility(llvm::GlobalValue::ProtectedVisibility);
1793 }
1794
1795 // Register the indirect Vtable:
1796 // This is similar to OMPTargetGlobalVarEntryIndirect, except that the
1797 // size field refers to the size of memory pointed to, not the size of
1798 // the pointer symbol itself (which is implicitly the size of a pointer).
1799 OMPBuilder.OffloadInfoManager.registerDeviceGlobalVarEntryInfo(
1800 Name, Addr, CGM.GetTargetTypeStoreSize(CGM.VoidPtrTy).getQuantity(),
1801 llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryIndirect,
1802 llvm::GlobalValue::WeakODRLinkage);
1803}
1804
1805void CGOpenMPRuntime::registerVTableOffloadEntry(llvm::GlobalVariable *VTable,
1806 const VarDecl *VD) {
1807 // TODO: add logic to avoid duplicate vtable registrations per
1808 // translation unit; though for external linkage, this should no
1809 // longer be an issue - or at least we can avoid the issue by
1810 // checking for an existing offloading entry. But, perhaps the
1811 // better approach is to defer emission of the vtables and offload
1812 // entries until later (by tracking a list of items that need to be
1813 // emitted).
1814
1815 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
1816
1817 // Generate a new externally visible global to point to the
1818 // internally visible vtable. Doing this allows us to keep the
1819 // visibility and linkage of the associated vtable unchanged while
1820 // allowing the runtime to access its value. The externally
1821 // visible global var needs to be emitted with a unique mangled
1822 // name that won't conflict with similarly named (internal)
1823 // vtables in other translation units.
1824
1825 // Register vtable with source location of dynamic object in map
1826 // clause.
1827 llvm::TargetRegionEntryInfo EntryInfo = getEntryInfoFromPresumedLoc(
1829 VTable->getName());
1830
1831 llvm::GlobalVariable *Addr = VTable;
1832 SmallString<128> AddrName;
1833 OMPBuilder.OffloadInfoManager.getTargetRegionEntryFnName(AddrName, EntryInfo);
1834 AddrName.append("addr");
1835
1836 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
1837 Addr = new llvm::GlobalVariable(
1838 CGM.getModule(), VTable->getType(),
1839 /*isConstant=*/true, llvm::GlobalValue::ExternalLinkage, VTable,
1840 AddrName,
1841 /*InsertBefore=*/nullptr, llvm::GlobalValue::NotThreadLocal,
1842 CGM.getModule().getDataLayout().getDefaultGlobalsAddressSpace());
1843 Addr->setVisibility(llvm::GlobalValue::ProtectedVisibility);
1844 }
1845 OMPBuilder.OffloadInfoManager.registerDeviceGlobalVarEntryInfo(
1846 AddrName, VTable,
1847 CGM.getDataLayout().getTypeAllocSize(VTable->getInitializer()->getType()),
1848 llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryIndirectVTable,
1849 llvm::GlobalValue::WeakODRLinkage);
1850}
1851
1854 const VarDecl *VD) {
1855 // Register C++ VTable to OpenMP Offload Entry if it's a new
1856 // CXXRecordDecl.
1857 if (CXXRecord && CXXRecord->isDynamicClass() &&
1858 !CGM.getOpenMPRuntime().VTableDeclMap.contains(CXXRecord)) {
1859 auto Res = CGM.getOpenMPRuntime().VTableDeclMap.try_emplace(CXXRecord, VD);
1860 if (Res.second) {
1861 CGM.EmitVTable(CXXRecord);
1862 CodeGenVTables VTables = CGM.getVTables();
1863 llvm::GlobalVariable *VTablesAddr = VTables.GetAddrOfVTable(CXXRecord);
1864 assert(VTablesAddr && "Expected non-null VTable address");
1865 // Must set VTables to weak since we're emitting them in multiple TUs now
1866 if (VTablesAddr->hasExternalLinkage())
1867 VTablesAddr->setLinkage(llvm::GlobalValue::WeakODRLinkage);
1868 CGM.getOpenMPRuntime().registerVTableOffloadEntry(VTablesAddr, VD);
1869 // Emit VTable for all the fields containing dynamic CXXRecord
1870 for (const FieldDecl *Field : CXXRecord->fields()) {
1871 if (CXXRecordDecl *RecordDecl = Field->getType()->getAsCXXRecordDecl())
1873 }
1874 // Emit VTable for all dynamic parent class
1875 for (CXXBaseSpecifier &Base : CXXRecord->bases()) {
1876 if (CXXRecordDecl *BaseDecl = Base.getType()->getAsCXXRecordDecl())
1877 emitAndRegisterVTable(CGM, BaseDecl, VD);
1878 }
1879 }
1880 }
1881}
1882
1884 // Register VTable by scanning through the map clause of OpenMP target region.
1885 // Get CXXRecordDecl and VarDecl from Expr.
1886 auto GetVTableDecl = [](const Expr *E) {
1887 QualType VDTy = E->getType();
1888 CXXRecordDecl *CXXRecord = nullptr;
1889 if (const auto *RefType = VDTy->getAs<LValueReferenceType>())
1890 VDTy = RefType->getPointeeType();
1891 if (VDTy->isPointerType())
1893 else
1894 CXXRecord = VDTy->getAsCXXRecordDecl();
1895
1896 const VarDecl *VD = nullptr;
1897 if (auto *DRE = dyn_cast<DeclRefExpr>(E)) {
1898 // Handle BindingDecls by redirecting to their DecompositionDecl.
1899 if (auto *BD = dyn_cast<BindingDecl>(DRE->getDecl()))
1900 VD = cast<VarDecl>(BD->getDecomposedDecl());
1901 else
1902 VD = cast<VarDecl>(DRE->getDecl());
1903 } else if (auto *MRE = dyn_cast<MemberExpr>(E)) {
1904 if (auto *BaseDRE = dyn_cast<DeclRefExpr>(MRE->getBase())) {
1905 if (auto *BaseVD = dyn_cast<VarDecl>(BaseDRE->getDecl()))
1906 VD = BaseVD;
1907 }
1908 }
1909 return std::pair<CXXRecordDecl *, const VarDecl *>(CXXRecord, VD);
1910 };
1911 // Collect VTable from OpenMP map clause.
1912 for (const auto *C : D.getClausesOfKind<OMPMapClause>()) {
1913 for (const auto *E : C->varlist()) {
1914 auto DeclPair = GetVTableDecl(E);
1915 // Ensure VD is not null
1916 if (DeclPair.second)
1917 emitAndRegisterVTable(CGM, DeclPair.first, DeclPair.second);
1918 }
1919 }
1920}
1921
1923 QualType VarType,
1924 StringRef Name) {
1925 std::string Suffix = getName({"artificial", ""});
1926 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType);
1927 llvm::GlobalVariable *GAddr = OMPBuilder.getOrCreateInternalVariable(
1928 VarLVType, Twine(Name).concat(Suffix).str());
1929 if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS &&
1930 CGM.getTarget().isTLSSupported()) {
1931 GAddr->setThreadLocal(/*Val=*/true);
1932 return Address(GAddr, GAddr->getValueType(),
1933 CGM.getContext().getTypeAlignInChars(VarType));
1934 }
1935 std::string CacheSuffix = getName({"cache", ""});
1936 llvm::Value *Args[] = {
1939 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy),
1940 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy,
1941 /*isSigned=*/false),
1942 OMPBuilder.getOrCreateInternalVariable(
1943 CGM.VoidPtrPtrTy,
1944 Twine(Name).concat(Suffix).concat(CacheSuffix).str())};
1945 return Address(
1947 CGF.EmitRuntimeCall(
1948 OMPBuilder.getOrCreateRuntimeFunction(
1949 CGM.getModule(), OMPRTL___kmpc_threadprivate_cached),
1950 Args),
1951 CGF.Builder.getPtrTy(0)),
1952 VarLVType, CGM.getContext().getTypeAlignInChars(VarType));
1953}
1954
1956 const RegionCodeGenTy &ThenGen,
1957 const RegionCodeGenTy &ElseGen) {
1958 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
1959
1960 // If the condition constant folds and can be elided, try to avoid emitting
1961 // the condition and the dead arm of the if/else.
1962 bool CondConstant;
1963 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
1964 if (CondConstant)
1965 ThenGen(CGF);
1966 else
1967 ElseGen(CGF);
1968 return;
1969 }
1970
1971 // Otherwise, the condition did not fold, or we couldn't elide it. Just
1972 // emit the conditional branch.
1973 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then");
1974 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else");
1975 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end");
1976 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0);
1977
1978 // Emit the 'then' code.
1979 CGF.EmitBlock(ThenBlock);
1980 ThenGen(CGF);
1981 CGF.EmitBranch(ContBlock);
1982 // Emit the 'else' code if present.
1983 // There is no need to emit line number for unconditional branch.
1985 CGF.EmitBlock(ElseBlock);
1986 ElseGen(CGF);
1987 // There is no need to emit line number for unconditional branch.
1989 CGF.EmitBranch(ContBlock);
1990 // Emit the continuation block for code after the if.
1991 CGF.EmitBlock(ContBlock, /*IsFinished=*/true);
1992}
1993
1995 CodeGenFunction &CGF, SourceLocation Loc, llvm::Function *OutlinedFn,
1996 ArrayRef<llvm::Value *> CapturedVars, const Expr *IfCond,
1997 llvm::Value *NumThreads, OpenMPNumThreadsClauseModifier NumThreadsModifier,
1998 OpenMPSeverityClauseKind Severity, const Expr *Message) {
1999 if (!CGF.HaveInsertPoint())
2000 return;
2001 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
2002 auto &M = CGM.getModule();
2003 auto &&ThenGen = [&M, OutlinedFn, CapturedVars, RTLoc,
2004 this](CodeGenFunction &CGF, PrePostActionTy &) {
2005 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn);
2006 llvm::Value *Args[] = {
2007 RTLoc,
2008 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
2009 OutlinedFn};
2011 RealArgs.append(std::begin(Args), std::end(Args));
2012 RealArgs.append(CapturedVars.begin(), CapturedVars.end());
2013
2014 llvm::FunctionCallee RTLFn =
2015 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_fork_call);
2016 CGF.EmitRuntimeCall(RTLFn, RealArgs);
2017 };
2018 auto &&ElseGen = [&M, OutlinedFn, CapturedVars, RTLoc, Loc,
2019 this](CodeGenFunction &CGF, PrePostActionTy &) {
2021 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc);
2022 // Build calls:
2023 // __kmpc_serialized_parallel(&Loc, GTid);
2024 llvm::Value *Args[] = {RTLoc, ThreadID};
2025 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2026 M, OMPRTL___kmpc_serialized_parallel),
2027 Args);
2028
2029 // OutlinedFn(&GTid, &zero_bound, CapturedStruct);
2030 Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc);
2031 RawAddress ZeroAddrBound =
2033 /*Name=*/".bound.zero.addr");
2034 CGF.Builder.CreateStore(CGF.Builder.getInt32(/*C*/ 0), ZeroAddrBound);
2036 // ThreadId for serialized parallels is 0.
2037 OutlinedFnArgs.push_back(ThreadIDAddr.emitRawPointer(CGF));
2038 OutlinedFnArgs.push_back(ZeroAddrBound.getPointer());
2039 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end());
2040
2041 // Ensure we do not inline the function. This is trivially true for the ones
2042 // passed to __kmpc_fork_call but the ones called in serialized regions
2043 // could be inlined. This is not a perfect but it is closer to the invariant
2044 // we want, namely, every data environment starts with a new function.
2045 // TODO: We should pass the if condition to the runtime function and do the
2046 // handling there. Much cleaner code.
2047 OutlinedFn->removeFnAttr(llvm::Attribute::AlwaysInline);
2048 OutlinedFn->addFnAttr(llvm::Attribute::NoInline);
2049 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs);
2050
2051 // __kmpc_end_serialized_parallel(&Loc, GTid);
2052 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID};
2053 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2054 M, OMPRTL___kmpc_end_serialized_parallel),
2055 EndArgs);
2056 };
2057 if (IfCond) {
2058 emitIfClause(CGF, IfCond, ThenGen, ElseGen);
2059 } else {
2060 RegionCodeGenTy ThenRCG(ThenGen);
2061 ThenRCG(CGF);
2062 }
2063}
2064
2065// If we're inside an (outlined) parallel region, use the region info's
2066// thread-ID variable (it is passed in a first argument of the outlined function
2067// as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
2068// regular serial code region, get thread ID by calling kmp_int32
2069// kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
2070// return the address of that temp.
2072 SourceLocation Loc) {
2073 if (auto *OMPRegionInfo =
2074 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
2075 if (OMPRegionInfo->getThreadIDVariable())
2076 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress();
2077
2078 llvm::Value *ThreadID = getThreadID(CGF, Loc);
2079 QualType Int32Ty =
2080 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
2081 Address ThreadIDTemp =
2082 CGF.CreateMemTempWithoutCast(Int32Ty, /*Name*/ ".threadid_temp.");
2083 CGF.EmitStoreOfScalar(ThreadID,
2084 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty));
2085
2086 return ThreadIDTemp;
2087}
2088
2089llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
2090 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str();
2091 std::string Name = getName({Prefix, "var"});
2092 llvm::GlobalVariable *GV =
2093 OMPBuilder.getOrCreateInternalVariable(KmpCriticalNameTy, Name);
2094 CGM.setDSOLocal(GV);
2095 return GV;
2096}
2097
2098namespace {
2099/// Common pre(post)-action for different OpenMP constructs.
2100class CommonActionTy final : public PrePostActionTy {
2101 llvm::FunctionCallee EnterCallee;
2102 ArrayRef<llvm::Value *> EnterArgs;
2103 llvm::FunctionCallee ExitCallee;
2104 ArrayRef<llvm::Value *> ExitArgs;
2105 bool Conditional;
2106 llvm::BasicBlock *ContBlock = nullptr;
2107
2108public:
2109 CommonActionTy(llvm::FunctionCallee EnterCallee,
2110 ArrayRef<llvm::Value *> EnterArgs,
2111 llvm::FunctionCallee ExitCallee,
2112 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false)
2113 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee),
2114 ExitArgs(ExitArgs), Conditional(Conditional) {}
2115 void Enter(CodeGenFunction &CGF) override {
2116 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs);
2117 if (Conditional) {
2118 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes);
2119 auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
2120 ContBlock = CGF.createBasicBlock("omp_if.end");
2121 // Generate the branch (If-stmt)
2122 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
2123 CGF.EmitBlock(ThenBlock);
2124 }
2125 }
2126 void Done(CodeGenFunction &CGF) {
2127 // Emit the rest of blocks/branches
2128 CGF.EmitBranch(ContBlock);
2129 CGF.EmitBlock(ContBlock, true);
2130 }
2131 void Exit(CodeGenFunction &CGF) override {
2132 CGF.EmitRuntimeCall(ExitCallee, ExitArgs);
2133 }
2134};
2135} // anonymous namespace
2136
2138 StringRef CriticalName,
2139 const RegionCodeGenTy &CriticalOpGen,
2140 SourceLocation Loc, const Expr *Hint) {
2141 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]);
2142 // CriticalOpGen();
2143 // __kmpc_end_critical(ident_t *, gtid, Lock);
2144 // Prepare arguments and build a call to __kmpc_critical
2145 if (!CGF.HaveInsertPoint())
2146 return;
2147 llvm::FunctionCallee RuntimeFcn = OMPBuilder.getOrCreateRuntimeFunction(
2148 CGM.getModule(),
2149 Hint ? OMPRTL___kmpc_critical_with_hint : OMPRTL___kmpc_critical);
2150 llvm::Value *LockVar = getCriticalRegionLock(CriticalName);
2151 unsigned LockVarArgIdx = 2;
2152 if (cast<llvm::GlobalVariable>(LockVar)->getAddressSpace() !=
2153 RuntimeFcn.getFunctionType()
2154 ->getParamType(LockVarArgIdx)
2155 ->getPointerAddressSpace())
2156 LockVar = CGF.Builder.CreateAddrSpaceCast(
2157 LockVar, RuntimeFcn.getFunctionType()->getParamType(LockVarArgIdx));
2158 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2159 LockVar};
2160 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args),
2161 std::end(Args));
2162 if (Hint) {
2163 EnterArgs.push_back(CGF.Builder.CreateIntCast(
2164 CGF.EmitScalarExpr(Hint), CGM.Int32Ty, /*isSigned=*/false));
2165 }
2166 CommonActionTy Action(RuntimeFcn, EnterArgs,
2167 OMPBuilder.getOrCreateRuntimeFunction(
2168 CGM.getModule(), OMPRTL___kmpc_end_critical),
2169 Args);
2170 CriticalOpGen.setAction(Action);
2171 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen);
2172}
2173
2175 const RegionCodeGenTy &MasterOpGen,
2176 SourceLocation Loc) {
2177 if (!CGF.HaveInsertPoint())
2178 return;
2179 // if(__kmpc_master(ident_t *, gtid)) {
2180 // MasterOpGen();
2181 // __kmpc_end_master(ident_t *, gtid);
2182 // }
2183 // Prepare arguments and build a call to __kmpc_master
2184 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2185 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2186 CGM.getModule(), OMPRTL___kmpc_master),
2187 Args,
2188 OMPBuilder.getOrCreateRuntimeFunction(
2189 CGM.getModule(), OMPRTL___kmpc_end_master),
2190 Args,
2191 /*Conditional=*/true);
2192 MasterOpGen.setAction(Action);
2193 emitInlinedDirective(CGF, OMPD_master, MasterOpGen);
2194 Action.Done(CGF);
2195}
2196
2198 const RegionCodeGenTy &MaskedOpGen,
2199 SourceLocation Loc, const Expr *Filter) {
2200 if (!CGF.HaveInsertPoint())
2201 return;
2202 // if(__kmpc_masked(ident_t *, gtid, filter)) {
2203 // MaskedOpGen();
2204 // __kmpc_end_masked(iden_t *, gtid);
2205 // }
2206 // Prepare arguments and build a call to __kmpc_masked
2207 llvm::Value *FilterVal = Filter
2208 ? CGF.EmitScalarExpr(Filter, CGF.Int32Ty)
2209 : llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/0);
2210 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2211 FilterVal};
2212 llvm::Value *ArgsEnd[] = {emitUpdateLocation(CGF, Loc),
2213 getThreadID(CGF, Loc)};
2214 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2215 CGM.getModule(), OMPRTL___kmpc_masked),
2216 Args,
2217 OMPBuilder.getOrCreateRuntimeFunction(
2218 CGM.getModule(), OMPRTL___kmpc_end_masked),
2219 ArgsEnd,
2220 /*Conditional=*/true);
2221 MaskedOpGen.setAction(Action);
2222 emitInlinedDirective(CGF, OMPD_masked, MaskedOpGen);
2223 Action.Done(CGF);
2224}
2225
2227 SourceLocation Loc) {
2228 if (!CGF.HaveInsertPoint())
2229 return;
2230 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2231 OMPBuilder.createTaskyield(CGF.Builder);
2232 } else {
2233 // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
2234 llvm::Value *Args[] = {
2235 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2236 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
2237 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2238 CGM.getModule(), OMPRTL___kmpc_omp_taskyield),
2239 Args);
2240 }
2241
2242 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
2243 Region->emitUntiedSwitch(CGF);
2244}
2245
2247 const RegionCodeGenTy &TaskgroupOpGen,
2248 SourceLocation Loc) {
2249 if (!CGF.HaveInsertPoint())
2250 return;
2251 // __kmpc_taskgroup(ident_t *, gtid);
2252 // TaskgroupOpGen();
2253 // __kmpc_end_taskgroup(ident_t *, gtid);
2254 // Prepare arguments and build a call to __kmpc_taskgroup
2255 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2256 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2257 CGM.getModule(), OMPRTL___kmpc_taskgroup),
2258 Args,
2259 OMPBuilder.getOrCreateRuntimeFunction(
2260 CGM.getModule(), OMPRTL___kmpc_end_taskgroup),
2261 Args);
2262 TaskgroupOpGen.setAction(Action);
2263 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen);
2264}
2265
2266/// Given an array of pointers to variables, project the address of a
2267/// given variable.
2269 unsigned Index, const VarDecl *Var) {
2270 // Pull out the pointer to the variable.
2271 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index);
2272 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr);
2273
2274 llvm::Type *ElemTy = CGF.ConvertTypeForMem(Var->getType());
2275 return Address(Ptr, ElemTy, CGF.getContext().getDeclAlign(Var));
2276}
2277
2279 CodeGenModule &CGM, llvm::Type *ArgsElemType,
2280 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
2281 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps,
2282 SourceLocation Loc) {
2283 ASTContext &C = CGM.getContext();
2284 // void copy_func(void *LHSArg, void *RHSArg);
2285
2286 auto *LHSArg =
2287 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
2288 C.VoidPtrTy, ImplicitParamKind::Other);
2289 auto *RHSArg =
2290 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
2291 C.VoidPtrTy, ImplicitParamKind::Other);
2292 FunctionArgList Args{LHSArg, RHSArg};
2293 const auto &CGFI =
2294 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
2295 std::string Name =
2296 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"});
2297 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
2298 llvm::GlobalValue::InternalLinkage, Name,
2299 &CGM.getModule());
2301 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
2302 Fn->addFnAttr("sample-profile-suffix-elision-policy", "selected");
2303 Fn->setDoesNotRecurse();
2304 CodeGenFunction CGF(CGM);
2305 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
2306 // Dest = (void*[n])(LHSArg);
2307 // Src = (void*[n])(RHSArg);
2309 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(LHSArg)),
2310 CGF.Builder.getPtrTy(0)),
2311 ArgsElemType, CGF.getPointerAlign());
2313 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(RHSArg)),
2314 CGF.Builder.getPtrTy(0)),
2315 ArgsElemType, CGF.getPointerAlign());
2316 // *(Type0*)Dst[0] = *(Type0*)Src[0];
2317 // *(Type1*)Dst[1] = *(Type1*)Src[1];
2318 // ...
2319 // *(Typen*)Dst[n] = *(Typen*)Src[n];
2320 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
2321 const auto *DestVar =
2322 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl());
2323 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar);
2324
2325 const auto *SrcVar =
2326 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl());
2327 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar);
2328
2329 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl();
2330 QualType Type = VD->getType();
2331 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]);
2332 }
2333 CGF.FinishFunction();
2334 return Fn;
2335}
2336
2338 const RegionCodeGenTy &SingleOpGen,
2339 SourceLocation Loc,
2340 ArrayRef<const Expr *> CopyprivateVars,
2341 ArrayRef<const Expr *> SrcExprs,
2342 ArrayRef<const Expr *> DstExprs,
2343 ArrayRef<const Expr *> AssignmentOps) {
2344 if (!CGF.HaveInsertPoint())
2345 return;
2346 assert(CopyprivateVars.size() == SrcExprs.size() &&
2347 CopyprivateVars.size() == DstExprs.size() &&
2348 CopyprivateVars.size() == AssignmentOps.size());
2349 ASTContext &C = CGM.getContext();
2350 // int32 did_it = 0;
2351 // if(__kmpc_single(ident_t *, gtid)) {
2352 // SingleOpGen();
2353 // __kmpc_end_single(ident_t *, gtid);
2354 // did_it = 1;
2355 // }
2356 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
2357 // <copy_func>, did_it);
2358
2359 Address DidIt = Address::invalid();
2360 if (!CopyprivateVars.empty()) {
2361 // int32 did_it = 0;
2362 QualType KmpInt32Ty =
2363 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
2364 DidIt = CGF.CreateMemTempWithoutCast(KmpInt32Ty, ".omp.copyprivate.did_it");
2365 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt);
2366 }
2367 // Prepare arguments and build a call to __kmpc_single
2368 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2369 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2370 CGM.getModule(), OMPRTL___kmpc_single),
2371 Args,
2372 OMPBuilder.getOrCreateRuntimeFunction(
2373 CGM.getModule(), OMPRTL___kmpc_end_single),
2374 Args,
2375 /*Conditional=*/true);
2376 SingleOpGen.setAction(Action);
2377 emitInlinedDirective(CGF, OMPD_single, SingleOpGen);
2378 if (DidIt.isValid()) {
2379 // did_it = 1;
2380 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt);
2381 }
2382 Action.Done(CGF);
2383 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
2384 // <copy_func>, did_it);
2385 if (DidIt.isValid()) {
2386 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
2387 QualType CopyprivateArrayTy = C.getConstantArrayType(
2388 C.VoidPtrTy, ArraySize, nullptr, ArraySizeModifier::Normal,
2389 /*IndexTypeQuals=*/0);
2390 // Create a list of all private variables for copyprivate.
2391 Address CopyprivateList = CGF.CreateMemTempWithoutCast(
2392 CopyprivateArrayTy, ".omp.copyprivate.cpr_list");
2393 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
2394 Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I);
2395 CGF.Builder.CreateStore(
2397 CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF),
2398 CGF.VoidPtrTy),
2399 Elem);
2400 }
2401 // Build function that copies private values from single region to all other
2402 // threads in the corresponding parallel region.
2403 llvm::Value *CpyFn = emitCopyprivateCopyFunction(
2404 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy), CopyprivateVars,
2405 SrcExprs, DstExprs, AssignmentOps, Loc);
2406 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy);
2408 CopyprivateList, CGF.VoidPtrTy, CGF.Int8Ty);
2409 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt);
2410 llvm::Value *Args[] = {
2411 emitUpdateLocation(CGF, Loc), // ident_t *<loc>
2412 getThreadID(CGF, Loc), // i32 <gtid>
2413 BufSize, // size_t <buf_size>
2414 CL.emitRawPointer(CGF), // void *<copyprivate list>
2415 CpyFn, // void (*) (void *, void *) <copy_func>
2416 DidItVal // i32 did_it
2417 };
2418 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2419 CGM.getModule(), OMPRTL___kmpc_copyprivate),
2420 Args);
2421 }
2422}
2423
2425 const RegionCodeGenTy &OrderedOpGen,
2426 SourceLocation Loc, bool IsThreads) {
2427 if (!CGF.HaveInsertPoint())
2428 return;
2429 // __kmpc_ordered(ident_t *, gtid);
2430 // OrderedOpGen();
2431 // __kmpc_end_ordered(ident_t *, gtid);
2432 // Prepare arguments and build a call to __kmpc_ordered
2433 if (IsThreads) {
2434 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2435 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2436 CGM.getModule(), OMPRTL___kmpc_ordered),
2437 Args,
2438 OMPBuilder.getOrCreateRuntimeFunction(
2439 CGM.getModule(), OMPRTL___kmpc_end_ordered),
2440 Args);
2441 OrderedOpGen.setAction(Action);
2442 emitInlinedDirective(CGF, OMPD_ordered_blockassoc, OrderedOpGen);
2443 return;
2444 }
2445 emitInlinedDirective(CGF, OMPD_ordered_blockassoc, OrderedOpGen);
2446}
2447
2449 unsigned Flags;
2450 if (Kind == OMPD_for)
2451 Flags = OMP_IDENT_BARRIER_IMPL_FOR;
2452 else if (Kind == OMPD_sections)
2453 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS;
2454 else if (Kind == OMPD_single)
2455 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE;
2456 else if (Kind == OMPD_barrier)
2457 Flags = OMP_IDENT_BARRIER_EXPL;
2458 else
2459 Flags = OMP_IDENT_BARRIER_IMPL;
2460 return Flags;
2461}
2462
2464 CodeGenFunction &CGF, const OMPLoopDirective &S,
2465 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const {
2466 // Check if the loop directive is actually a doacross loop directive. In this
2467 // case choose static, 1 schedule.
2468 if (llvm::any_of(
2469 S.getClausesOfKind<OMPOrderedClause>(),
2470 [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) {
2471 ScheduleKind = OMPC_SCHEDULE_static;
2472 // Chunk size is 1 in this case.
2473 llvm::APInt ChunkSize(32, 1);
2474 ChunkExpr = IntegerLiteral::Create(
2475 CGF.getContext(), ChunkSize,
2476 CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0),
2477 SourceLocation());
2478 }
2479}
2480
2482 OpenMPDirectiveKind Kind, bool EmitChecks,
2483 bool ForceSimpleCall) {
2484 // Check if we should use the OMPBuilder
2485 auto *OMPRegionInfo =
2486 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo);
2487 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2488 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
2489 cantFail(OMPBuilder.createBarrier(CGF.Builder, Kind, ForceSimpleCall,
2490 EmitChecks));
2491 CGF.Builder.restoreIP(AfterIP);
2492 return;
2493 }
2494
2495 if (!CGF.HaveInsertPoint())
2496 return;
2497 // Build call __kmpc_cancel_barrier(loc, thread_id);
2498 // Build call __kmpc_barrier(loc, thread_id);
2499 unsigned Flags = getDefaultFlagsForBarriers(Kind);
2500 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc,
2501 // thread_id);
2502 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
2503 getThreadID(CGF, Loc)};
2504 if (OMPRegionInfo) {
2505 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) {
2506 llvm::Value *Result = CGF.EmitRuntimeCall(
2507 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
2508 OMPRTL___kmpc_cancel_barrier),
2509 Args);
2510 if (EmitChecks) {
2511 // if (__kmpc_cancel_barrier()) {
2512 // exit from construct;
2513 // }
2514 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
2515 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
2516 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
2517 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
2518 CGF.EmitBlock(ExitBB);
2519 // exit from construct;
2520 CodeGenFunction::JumpDest CancelDestination =
2521 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
2522 CGF.EmitBranchThroughCleanup(CancelDestination);
2523 CGF.EmitBlock(ContBB, /*IsFinished=*/true);
2524 }
2525 return;
2526 }
2527 }
2528 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2529 CGM.getModule(), OMPRTL___kmpc_barrier),
2530 Args);
2531}
2532
2534 Expr *ME, bool IsFatal) {
2535 llvm::Value *MVL = ME ? CGF.EmitScalarExpr(ME)
2536 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
2537 // Build call void __kmpc_error(ident_t *loc, int severity, const char
2538 // *message)
2539 llvm::Value *Args[] = {
2540 emitUpdateLocation(CGF, Loc, /*Flags=*/0, /*GenLoc=*/true),
2541 llvm::ConstantInt::get(CGM.Int32Ty, IsFatal ? 2 : 1),
2542 CGF.Builder.CreatePointerCast(MVL, CGM.Int8PtrTy)};
2543 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2544 CGM.getModule(), OMPRTL___kmpc_error),
2545 Args);
2546}
2547
2548/// Map the OpenMP loop schedule to the runtime enumeration.
2549static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
2550 bool Chunked, bool Ordered) {
2551 switch (ScheduleKind) {
2552 case OMPC_SCHEDULE_static:
2553 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked)
2554 : (Ordered ? OMP_ord_static : OMP_sch_static);
2555 case OMPC_SCHEDULE_dynamic:
2556 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked;
2557 case OMPC_SCHEDULE_guided:
2558 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked;
2559 case OMPC_SCHEDULE_runtime:
2560 return Ordered ? OMP_ord_runtime : OMP_sch_runtime;
2561 case OMPC_SCHEDULE_auto:
2562 return Ordered ? OMP_ord_auto : OMP_sch_auto;
2564 assert(!Chunked && "chunk was specified but schedule kind not known");
2565 return Ordered ? OMP_ord_static : OMP_sch_static;
2566 }
2567 llvm_unreachable("Unexpected runtime schedule");
2568}
2569
2570/// Map the OpenMP distribute schedule to the runtime enumeration.
2571static OpenMPSchedType
2573 // only static is allowed for dist_schedule
2574 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static;
2575}
2576
2578 bool Chunked) const {
2579 OpenMPSchedType Schedule =
2580 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
2581 return Schedule == OMP_sch_static;
2582}
2583
2585 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
2586 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
2587 return Schedule == OMP_dist_sch_static;
2588}
2589
2591 bool Chunked) const {
2592 OpenMPSchedType Schedule =
2593 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
2594 return Schedule == OMP_sch_static_chunked;
2595}
2596
2598 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
2599 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
2600 return Schedule == OMP_dist_sch_static_chunked;
2601}
2602
2604 OpenMPSchedType Schedule =
2605 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false);
2606 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
2607 return Schedule != OMP_sch_static;
2608}
2609
2610static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule,
2613 int Modifier = 0;
2614 switch (M1) {
2615 case OMPC_SCHEDULE_MODIFIER_monotonic:
2616 Modifier = OMP_sch_modifier_monotonic;
2617 break;
2618 case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
2619 Modifier = OMP_sch_modifier_nonmonotonic;
2620 break;
2621 case OMPC_SCHEDULE_MODIFIER_simd:
2622 if (Schedule == OMP_sch_static_chunked)
2623 Schedule = OMP_sch_static_balanced_chunked;
2624 break;
2627 break;
2628 }
2629 switch (M2) {
2630 case OMPC_SCHEDULE_MODIFIER_monotonic:
2631 Modifier = OMP_sch_modifier_monotonic;
2632 break;
2633 case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
2634 Modifier = OMP_sch_modifier_nonmonotonic;
2635 break;
2636 case OMPC_SCHEDULE_MODIFIER_simd:
2637 if (Schedule == OMP_sch_static_chunked)
2638 Schedule = OMP_sch_static_balanced_chunked;
2639 break;
2642 break;
2643 }
2644 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription.
2645 // If the static schedule kind is specified or if the ordered clause is
2646 // specified, and if the nonmonotonic modifier is not specified, the effect is
2647 // as if the monotonic modifier is specified. Otherwise, unless the monotonic
2648 // modifier is specified, the effect is as if the nonmonotonic modifier is
2649 // specified.
2650 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) {
2651 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static ||
2652 Schedule == OMP_sch_static_balanced_chunked ||
2653 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static ||
2654 Schedule == OMP_dist_sch_static_chunked ||
2655 Schedule == OMP_dist_sch_static ||
2656 Schedule == OMP_dist_sch_static_chunked_sch_static_chunkone))
2657 Modifier = OMP_sch_modifier_nonmonotonic;
2658 }
2659 return Schedule | Modifier;
2660}
2661
2664 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
2665 bool Ordered, const DispatchRTInput &DispatchValues) {
2666 if (!CGF.HaveInsertPoint())
2667 return;
2668 OpenMPSchedType Schedule = getRuntimeSchedule(
2669 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered);
2670 assert(Ordered ||
2671 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked &&
2672 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked &&
2673 Schedule != OMP_sch_static_balanced_chunked));
2674 // Call __kmpc_dispatch_init(
2675 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
2676 // kmp_int[32|64] lower, kmp_int[32|64] upper,
2677 // kmp_int[32|64] stride, kmp_int[32|64] chunk);
2678
2679 // If the Chunk was not specified in the clause - use default value 1.
2680 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk
2681 : CGF.Builder.getIntN(IVSize, 1);
2682 llvm::Value *Args[] = {
2683 emitUpdateLocation(CGF, Loc),
2684 getThreadID(CGF, Loc),
2685 CGF.Builder.getInt32(addMonoNonMonoModifier(
2686 CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type
2687 DispatchValues.LB, // Lower
2688 DispatchValues.UB, // Upper
2689 CGF.Builder.getIntN(IVSize, 1), // Stride
2690 Chunk // Chunk
2691 };
2692 CGF.EmitRuntimeCall(OMPBuilder.createDispatchInitFunction(IVSize, IVSigned),
2693 Args);
2694}
2695
2697 SourceLocation Loc) {
2698 if (!CGF.HaveInsertPoint())
2699 return;
2700 // Call __kmpc_dispatch_deinit(ident_t *loc, kmp_int32 tid);
2701 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2702 CGF.EmitRuntimeCall(OMPBuilder.createDispatchDeinitFunction(), Args);
2703}
2704
2706 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId,
2707 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule,
2709 const CGOpenMPRuntime::StaticRTInput &Values) {
2710 if (!CGF.HaveInsertPoint())
2711 return;
2712
2713 assert(!Values.Ordered);
2714 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked ||
2715 Schedule == OMP_sch_static_balanced_chunked ||
2716 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked ||
2717 Schedule == OMP_dist_sch_static ||
2718 Schedule == OMP_dist_sch_static_chunked ||
2719 Schedule == OMP_dist_sch_static_chunked_sch_static_chunkone);
2720
2721 // Call __kmpc_for_static_init(
2722 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
2723 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
2724 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
2725 // kmp_int[32|64] incr, kmp_int[32|64] chunk);
2726 llvm::Value *Chunk = Values.Chunk;
2727 if (Chunk == nullptr) {
2728 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static ||
2729 Schedule == OMP_dist_sch_static) &&
2730 "expected static non-chunked schedule");
2731 // If the Chunk was not specified in the clause - use default value 1.
2732 Chunk = CGF.Builder.getIntN(Values.IVSize, 1);
2733 } else {
2734 assert((Schedule == OMP_sch_static_chunked ||
2735 Schedule == OMP_sch_static_balanced_chunked ||
2736 Schedule == OMP_ord_static_chunked ||
2737 Schedule == OMP_dist_sch_static_chunked ||
2738 Schedule == OMP_dist_sch_static_chunked_sch_static_chunkone) &&
2739 "expected static chunked schedule");
2740 }
2741 llvm::Value *Args[] = {
2742 UpdateLocation,
2743 ThreadId,
2744 CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1,
2745 M2)), // Schedule type
2746 Values.IL.emitRawPointer(CGF), // &isLastIter
2747 Values.LB.emitRawPointer(CGF), // &LB
2748 Values.UB.emitRawPointer(CGF), // &UB
2749 Values.ST.emitRawPointer(CGF), // &Stride
2750 CGF.Builder.getIntN(Values.IVSize, 1), // Incr
2751 Chunk // Chunk
2752 };
2753 CGF.EmitRuntimeCall(ForStaticInitFunction, Args);
2754}
2755
2757 SourceLocation Loc,
2758 OpenMPDirectiveKind DKind,
2759 const OpenMPScheduleTy &ScheduleKind,
2760 const StaticRTInput &Values) {
2761 OpenMPSchedType ScheduleNum =
2762 ScheduleKind.UseFusedDistChunkSchedule
2763 ? OMP_dist_sch_static_chunked_sch_static_chunkone
2764 : getRuntimeSchedule(ScheduleKind.Schedule, Values.Chunk != nullptr,
2765 Values.Ordered);
2766 assert((isOpenMPWorksharingDirective(DKind) || (DKind == OMPD_loop)) &&
2767 "Expected loop-based or sections-based directive.");
2768 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc,
2770 ? OMP_IDENT_WORK_LOOP
2771 : OMP_IDENT_WORK_SECTIONS);
2772 llvm::Value *ThreadId = getThreadID(CGF, Loc);
2773 llvm::FunctionCallee StaticInitFunction =
2774 OMPBuilder.createForStaticInitFunction(Values.IVSize, Values.IVSigned,
2775 false);
2777 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
2778 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values);
2779}
2780
2784 const CGOpenMPRuntime::StaticRTInput &Values) {
2785 OpenMPSchedType ScheduleNum =
2786 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr);
2787 llvm::Value *UpdatedLocation =
2788 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE);
2789 llvm::Value *ThreadId = getThreadID(CGF, Loc);
2790 llvm::FunctionCallee StaticInitFunction;
2791 bool isGPUDistribute =
2792 CGM.getLangOpts().OpenMPIsTargetDevice && CGM.getTriple().isGPU();
2793 StaticInitFunction = OMPBuilder.createForStaticInitFunction(
2794 Values.IVSize, Values.IVSigned, isGPUDistribute);
2795
2796 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
2797 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown,
2799}
2800
2802 SourceLocation Loc,
2803 OpenMPDirectiveKind DKind) {
2804 assert((DKind == OMPD_distribute || DKind == OMPD_for ||
2805 DKind == OMPD_sections) &&
2806 "Expected distribute, for, or sections directive kind");
2807 if (!CGF.HaveInsertPoint())
2808 return;
2809 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
2810 llvm::Value *Args[] = {
2811 emitUpdateLocation(CGF, Loc,
2813 (DKind == OMPD_target_teams_loop)
2814 ? OMP_IDENT_WORK_DISTRIBUTE
2815 : isOpenMPLoopDirective(DKind)
2816 ? OMP_IDENT_WORK_LOOP
2817 : OMP_IDENT_WORK_SECTIONS),
2818 getThreadID(CGF, Loc)};
2820 if (isOpenMPDistributeDirective(DKind) &&
2821 CGM.getLangOpts().OpenMPIsTargetDevice && CGM.getTriple().isGPU())
2822 CGF.EmitRuntimeCall(
2823 OMPBuilder.getOrCreateRuntimeFunction(
2824 CGM.getModule(), OMPRTL___kmpc_distribute_static_fini),
2825 Args);
2826 else
2827 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2828 CGM.getModule(), OMPRTL___kmpc_for_static_fini),
2829 Args);
2830}
2831
2833 SourceLocation Loc,
2834 unsigned IVSize,
2835 bool IVSigned) {
2836 if (!CGF.HaveInsertPoint())
2837 return;
2838 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
2839 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2840 CGF.EmitRuntimeCall(OMPBuilder.createDispatchFiniFunction(IVSize, IVSigned),
2841 Args);
2842}
2843
2845 SourceLocation Loc, unsigned IVSize,
2846 bool IVSigned, Address IL,
2847 Address LB, Address UB,
2848 Address ST) {
2849 // Call __kmpc_dispatch_next(
2850 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
2851 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
2852 // kmp_int[32|64] *p_stride);
2853 llvm::Value *Args[] = {
2854 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2855 IL.emitRawPointer(CGF), // &isLastIter
2856 LB.emitRawPointer(CGF), // &Lower
2857 UB.emitRawPointer(CGF), // &Upper
2858 ST.emitRawPointer(CGF) // &Stride
2859 };
2860 llvm::Value *Call = CGF.EmitRuntimeCall(
2861 OMPBuilder.createDispatchNextFunction(IVSize, IVSigned), Args);
2862 return CGF.EmitScalarConversion(
2863 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1),
2864 CGF.getContext().BoolTy, Loc);
2865}
2866
2868 const Expr *Message,
2869 SourceLocation Loc) {
2870 if (!Message)
2871 return llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
2872 return CGF.EmitScalarExpr(Message);
2873}
2874
2875llvm::Value *
2877 SourceLocation Loc) {
2878 // OpenMP 6.0, 10.4: "If no severity clause is specified then the effect is
2879 // as if sev-level is fatal."
2880 return llvm::ConstantInt::get(CGM.Int32Ty,
2881 Severity == OMPC_SEVERITY_warning ? 1 : 2);
2882}
2883
2885 CodeGenFunction &CGF, llvm::Value *NumThreads, SourceLocation Loc,
2887 SourceLocation SeverityLoc, const Expr *Message,
2888 SourceLocation MessageLoc) {
2889 if (!CGF.HaveInsertPoint())
2890 return;
2892 {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2893 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)});
2894 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
2895 // or __kmpc_push_num_threads_strict(&loc, global_tid, num_threads, severity,
2896 // messsage) if strict modifier is used.
2897 RuntimeFunction FnID = OMPRTL___kmpc_push_num_threads;
2898 if (Modifier == OMPC_NUMTHREADS_strict) {
2899 FnID = OMPRTL___kmpc_push_num_threads_strict;
2900 Args.push_back(emitSeverityClause(Severity, SeverityLoc));
2901 Args.push_back(emitMessageClause(CGF, Message, MessageLoc));
2902 }
2903 CGF.EmitRuntimeCall(
2904 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), FnID), Args);
2905}
2906
2908 ProcBindKind ProcBind,
2909 SourceLocation Loc) {
2910 if (!CGF.HaveInsertPoint())
2911 return;
2912 assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value.");
2913 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind)
2914 llvm::Value *Args[] = {
2915 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2916 llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)};
2917 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2918 CGM.getModule(), OMPRTL___kmpc_push_proc_bind),
2919 Args);
2920}
2921
2923 SourceLocation Loc, llvm::AtomicOrdering AO) {
2924 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2925 OMPBuilder.createFlush(CGF.Builder);
2926 } else {
2927 if (!CGF.HaveInsertPoint())
2928 return;
2929 // Build call void __kmpc_flush(ident_t *loc)
2930 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2931 CGM.getModule(), OMPRTL___kmpc_flush),
2932 emitUpdateLocation(CGF, Loc));
2933 }
2934}
2935
2936namespace {
2937/// Indexes of fields for type kmp_task_t.
2938enum KmpTaskTFields {
2939 /// List of shared variables.
2940 KmpTaskTShareds,
2941 /// Task routine.
2942 KmpTaskTRoutine,
2943 /// Partition id for the untied tasks.
2944 KmpTaskTPartId,
2945 /// Function with call of destructors for private variables.
2946 Data1,
2947 /// Task priority.
2948 Data2,
2949 /// (Taskloops only) Lower bound.
2950 KmpTaskTLowerBound,
2951 /// (Taskloops only) Upper bound.
2952 KmpTaskTUpperBound,
2953 /// (Taskloops only) Stride.
2954 KmpTaskTStride,
2955 /// (Taskloops only) Is last iteration flag.
2956 KmpTaskTLastIter,
2957 /// (Taskloops only) Reduction data.
2958 KmpTaskTReductions,
2959};
2960} // anonymous namespace
2961
2963 // If we are in simd mode or there are no entries, we don't need to do
2964 // anything.
2965 if (CGM.getLangOpts().OpenMPSimd || OMPBuilder.OffloadInfoManager.empty())
2966 return;
2967
2968 llvm::OpenMPIRBuilder::EmitMetadataErrorReportFunctionTy &&ErrorReportFn =
2969 [this](llvm::OpenMPIRBuilder::EmitMetadataErrorKind Kind,
2970 const llvm::TargetRegionEntryInfo &EntryInfo) -> void {
2971 SourceLocation Loc;
2972 if (Kind != llvm::OpenMPIRBuilder::EMIT_MD_GLOBAL_VAR_LINK_ERROR) {
2973 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(),
2974 E = CGM.getContext().getSourceManager().fileinfo_end();
2975 I != E; ++I) {
2976 if (I->getFirst().getUniqueID().getDevice() == EntryInfo.DeviceID &&
2977 I->getFirst().getUniqueID().getFile() == EntryInfo.FileID) {
2978 Loc = CGM.getContext().getSourceManager().translateFileLineCol(
2979 I->getFirst(), EntryInfo.Line, 1);
2980 break;
2981 }
2982 }
2983 }
2984 switch (Kind) {
2985 case llvm::OpenMPIRBuilder::EMIT_MD_TARGET_REGION_ERROR: {
2986 CGM.getDiags().Report(Loc,
2987 diag::err_target_region_offloading_entry_incorrect)
2988 << EntryInfo.ParentName;
2989 } break;
2990 case llvm::OpenMPIRBuilder::EMIT_MD_DECLARE_TARGET_ERROR: {
2991 CGM.getDiags().Report(
2992 Loc, diag::err_target_var_offloading_entry_incorrect_with_parent)
2993 << EntryInfo.ParentName;
2994 } break;
2995 case llvm::OpenMPIRBuilder::EMIT_MD_GLOBAL_VAR_LINK_ERROR: {
2996 CGM.getDiags().Report(diag::err_target_var_offloading_entry_incorrect);
2997 } break;
2998 case llvm::OpenMPIRBuilder::EMIT_MD_GLOBAL_VAR_INDIRECT_ERROR: {
2999 unsigned DiagID = CGM.getDiags().getCustomDiagID(
3000 DiagnosticsEngine::Error, "Offloading entry for indirect declare "
3001 "target variable is incorrect: the "
3002 "address is invalid.");
3003 CGM.getDiags().Report(DiagID);
3004 } break;
3005 }
3006 };
3007
3008 OMPBuilder.createOffloadEntriesAndInfoMetadata(ErrorReportFn);
3009}
3010
3012 if (!KmpRoutineEntryPtrTy) {
3013 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
3014 ASTContext &C = CGM.getContext();
3015 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
3017 KmpRoutineEntryPtrQTy = C.getPointerType(
3018 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
3019 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
3020 }
3021}
3022
3023namespace {
3024struct PrivateHelpersTy {
3025 PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original,
3026 const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit)
3027 : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy),
3028 PrivateElemInit(PrivateElemInit) {}
3029 PrivateHelpersTy(const VarDecl *Original) : Original(Original) {}
3030 const Expr *OriginalRef = nullptr;
3031 const VarDecl *Original = nullptr;
3032 const VarDecl *PrivateCopy = nullptr;
3033 const VarDecl *PrivateElemInit = nullptr;
3034 bool isLocalPrivate() const {
3035 return !OriginalRef && !PrivateCopy && !PrivateElemInit;
3036 }
3037};
3038typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
3039} // anonymous namespace
3040
3041/// For BindingDecls, returns the DecomposedDecl as the original VarDecl.
3042/// For regular VarDecls, returns the VarDecl itself.
3044 if (const auto *BD = dyn_cast<BindingDecl>(Decl))
3045 return cast<VarDecl>(BD->getDecomposedDecl());
3046 return cast<VarDecl>(Decl);
3047}
3048
3049static bool isAllocatableDecl(const VarDecl *VD) {
3050 const VarDecl *CVD = VD->getCanonicalDecl();
3051 if (!CVD->hasAttr<OMPAllocateDeclAttr>())
3052 return false;
3053 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
3054 // Use the default allocation.
3055 return !(AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc &&
3056 !AA->getAllocator());
3057}
3058
3059static RecordDecl *
3061 if (!Privates.empty()) {
3062 ASTContext &C = CGM.getContext();
3063 // Build struct .kmp_privates_t. {
3064 // /* private vars */
3065 // };
3066 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t");
3067 RD->startDefinition();
3068 for (const auto &Pair : Privates) {
3069 const VarDecl *VD = Pair.second.Original;
3070 const VarDecl *PrivateCopy = Pair.second.PrivateCopy;
3071 // For BindingDecls, use PrivateCopy type (binding's actual type).
3072 // For regular variables, use Original type to preserve qualifiers.
3073 // Check OriginalRef to detect BindingDecls since Original may be the
3074 // DecompositionDecl.
3075 bool IsBinding =
3076 Pair.second.OriginalRef &&
3078 cast<DeclRefExpr>(Pair.second.OriginalRef)->getDecl());
3079 QualType Type = IsBinding ? PrivateCopy->getType().getNonReferenceType()
3080 : VD->getType().getNonReferenceType();
3081
3082 // If the private variable is a local variable with lvalue ref type,
3083 // allocate the pointer instead of the pointee type.
3084 if (Pair.second.isLocalPrivate()) {
3085 if (VD->getType()->isLValueReferenceType())
3086 Type = C.getPointerType(Type);
3087 if (isAllocatableDecl(VD))
3088 Type = C.getPointerType(Type);
3089 }
3091 if (VD->hasAttrs()) {
3092 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()),
3093 E(VD->getAttrs().end());
3094 I != E; ++I)
3095 FD->addAttr(*I);
3096 }
3097 }
3098 RD->completeDefinition();
3099 return RD;
3100 }
3101 return nullptr;
3102}
3103
3104static RecordDecl *
3106 QualType KmpInt32Ty,
3107 QualType KmpRoutineEntryPointerQTy) {
3108 ASTContext &C = CGM.getContext();
3109 // Build struct kmp_task_t {
3110 // void * shareds;
3111 // kmp_routine_entry_t routine;
3112 // kmp_int32 part_id;
3113 // kmp_cmplrdata_t data1;
3114 // kmp_cmplrdata_t data2;
3115 // For taskloops additional fields:
3116 // kmp_uint64 lb;
3117 // kmp_uint64 ub;
3118 // kmp_int64 st;
3119 // kmp_int32 liter;
3120 // void * reductions;
3121 // };
3122 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TagTypeKind::Union);
3123 UD->startDefinition();
3124 addFieldToRecordDecl(C, UD, KmpInt32Ty);
3125 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy);
3126 UD->completeDefinition();
3127 CanQualType KmpCmplrdataTy = C.getCanonicalTagType(UD);
3128 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t");
3129 RD->startDefinition();
3130 addFieldToRecordDecl(C, RD, C.VoidPtrTy);
3131 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
3132 addFieldToRecordDecl(C, RD, KmpInt32Ty);
3133 addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
3134 addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
3135 if (isOpenMPTaskLoopDirective(Kind)) {
3136 QualType KmpUInt64Ty =
3137 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
3138 QualType KmpInt64Ty =
3139 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
3140 addFieldToRecordDecl(C, RD, KmpUInt64Ty);
3141 addFieldToRecordDecl(C, RD, KmpUInt64Ty);
3142 addFieldToRecordDecl(C, RD, KmpInt64Ty);
3143 addFieldToRecordDecl(C, RD, KmpInt32Ty);
3144 addFieldToRecordDecl(C, RD, C.VoidPtrTy);
3145 }
3146 RD->completeDefinition();
3147 return RD;
3148}
3149
3150static RecordDecl *
3153 ASTContext &C = CGM.getContext();
3154 // Build struct kmp_task_t_with_privates {
3155 // kmp_task_t task_data;
3156 // .kmp_privates_t. privates;
3157 // };
3158 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates");
3159 RD->startDefinition();
3160 addFieldToRecordDecl(C, RD, KmpTaskTQTy);
3161 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates))
3162 addFieldToRecordDecl(C, RD, C.getCanonicalTagType(PrivateRD));
3163 RD->completeDefinition();
3164 return RD;
3165}
3166
3167/// Emit a proxy function which accepts kmp_task_t as the second
3168/// argument.
3169/// \code
3170/// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
3171/// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt,
3172/// For taskloops:
3173/// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
3174/// tt->reductions, tt->shareds);
3175/// return 0;
3176/// }
3177/// \endcode
3178static llvm::Function *
3180 OpenMPDirectiveKind Kind, QualType KmpInt32Ty,
3181 QualType KmpTaskTWithPrivatesPtrQTy,
3182 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
3183 QualType SharedsPtrTy, llvm::Function *TaskFunction,
3184 llvm::Value *TaskPrivatesMap) {
3185 ASTContext &C = CGM.getContext();
3186 auto *GtidArg =
3187 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3188 KmpInt32Ty, ImplicitParamKind::Other);
3189 auto *TaskTypeArg = ImplicitParamDecl::Create(
3190 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3191 KmpTaskTWithPrivatesPtrQTy.withRestrict(), ImplicitParamKind::Other);
3192 FunctionArgList Args{GtidArg, TaskTypeArg};
3193 const auto &TaskEntryFnInfo =
3194 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
3195 llvm::FunctionType *TaskEntryTy =
3196 CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
3197 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""});
3198 auto *TaskEntry = llvm::Function::Create(
3199 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
3200 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo);
3201 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3202 TaskEntry->addFnAttr("sample-profile-suffix-elision-policy", "selected");
3203 TaskEntry->setDoesNotRecurse();
3204 CodeGenFunction CGF(CGM);
3205 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args,
3206 Loc, Loc);
3207
3208 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map,
3209 // tt,
3210 // For taskloops:
3211 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
3212 // tt->task_data.shareds);
3213 llvm::Value *GtidParam = CGF.EmitLoadOfScalar(
3214 CGF.GetAddrOfLocalVar(GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc);
3215 LValue TDBase = CGF.EmitLoadOfPointerLValue(
3216 CGF.GetAddrOfLocalVar(TaskTypeArg),
3217 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3218 const auto *KmpTaskTWithPrivatesQTyRD =
3219 KmpTaskTWithPrivatesQTy->castAsRecordDecl();
3220 LValue Base =
3221 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
3222 const auto *KmpTaskTQTyRD = KmpTaskTQTy->castAsRecordDecl();
3223 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
3224 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI);
3225 llvm::Value *PartidParam = PartIdLVal.getPointer(CGF);
3226
3227 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds);
3228 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI);
3229 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3230 CGF.EmitLoadOfScalar(SharedsLVal, Loc),
3231 CGF.ConvertTypeForMem(SharedsPtrTy));
3232
3233 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
3234 llvm::Value *PrivatesParam;
3235 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) {
3236 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI);
3237 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3238 PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy);
3239 } else {
3240 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
3241 }
3242
3243 llvm::Value *CommonArgs[] = {
3244 GtidParam, PartidParam, PrivatesParam, TaskPrivatesMap,
3245 CGF.Builder
3246 .CreatePointerBitCastOrAddrSpaceCast(TDBase.getAddress(),
3247 CGF.VoidPtrTy, CGF.Int8Ty)
3248 .emitRawPointer(CGF)};
3249 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs),
3250 std::end(CommonArgs));
3251 if (isOpenMPTaskLoopDirective(Kind)) {
3252 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound);
3253 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI);
3254 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc);
3255 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound);
3256 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI);
3257 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc);
3258 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride);
3259 LValue StLVal = CGF.EmitLValueForField(Base, *StFI);
3260 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc);
3261 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
3262 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
3263 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc);
3264 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions);
3265 LValue RLVal = CGF.EmitLValueForField(Base, *RFI);
3266 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc);
3267 CallArgs.push_back(LBParam);
3268 CallArgs.push_back(UBParam);
3269 CallArgs.push_back(StParam);
3270 CallArgs.push_back(LIParam);
3271 CallArgs.push_back(RParam);
3272 }
3273 CallArgs.push_back(SharedsParam);
3274
3275 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction,
3276 CallArgs);
3277 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)),
3278 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty));
3279 CGF.FinishFunction();
3280 return TaskEntry;
3281}
3282
3284 SourceLocation Loc,
3285 QualType KmpInt32Ty,
3286 QualType KmpTaskTWithPrivatesPtrQTy,
3287 QualType KmpTaskTWithPrivatesQTy) {
3288 ASTContext &C = CGM.getContext();
3289 auto *GtidArg =
3290 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3291 KmpInt32Ty, ImplicitParamKind::Other);
3292 auto *TaskTypeArg = ImplicitParamDecl::Create(
3293 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3294 KmpTaskTWithPrivatesPtrQTy.withRestrict(), ImplicitParamKind::Other);
3295 FunctionArgList Args{GtidArg, TaskTypeArg};
3296 const auto &DestructorFnInfo =
3297 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
3298 llvm::FunctionType *DestructorFnTy =
3299 CGM.getTypes().GetFunctionType(DestructorFnInfo);
3300 std::string Name =
3301 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""});
3302 auto *DestructorFn =
3303 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage,
3304 Name, &CGM.getModule());
3305 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn,
3306 DestructorFnInfo);
3307 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3308 DestructorFn->addFnAttr("sample-profile-suffix-elision-policy", "selected");
3309 DestructorFn->setDoesNotRecurse();
3310 CodeGenFunction CGF(CGM);
3311 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo,
3312 Args, Loc, Loc);
3313
3314 LValue Base = CGF.EmitLoadOfPointerLValue(
3315 CGF.GetAddrOfLocalVar(TaskTypeArg),
3316 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3317 const auto *KmpTaskTWithPrivatesQTyRD =
3318 KmpTaskTWithPrivatesQTy->castAsRecordDecl();
3319 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
3320 Base = CGF.EmitLValueForField(Base, *FI);
3321 for (const auto *Field : FI->getType()->castAsRecordDecl()->fields()) {
3322 if (QualType::DestructionKind DtorKind =
3323 Field->getType().isDestructedType()) {
3324 LValue FieldLValue = CGF.EmitLValueForField(Base, Field);
3325 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(), Field->getType());
3326 }
3327 }
3328 CGF.FinishFunction();
3329 return DestructorFn;
3330}
3331
3332/// Emit a privates mapping function for correct handling of private and
3333/// firstprivate variables.
3334/// \code
3335/// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1>
3336/// **noalias priv1,..., <tyn> **noalias privn) {
3337/// *priv1 = &.privates.priv1;
3338/// ...;
3339/// *privn = &.privates.privn;
3340/// }
3341/// \endcode
3342static llvm::Value *
3344 const OMPTaskDataTy &Data, QualType PrivatesQTy,
3346 ASTContext &C = CGM.getContext();
3347 FunctionArgList Args;
3348 auto *TaskPrivatesArg = ImplicitParamDecl::Create(
3349 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3350 C.getPointerType(PrivatesQTy).withConst().withRestrict(),
3352 Args.push_back(TaskPrivatesArg);
3353 llvm::SmallDenseMap<CanonicalDeclPtr<const VarDecl>, unsigned> PrivateVarsPos;
3354 // Track BindingDecl positions separately since BindingDecl is not a VarDecl.
3355 llvm::SmallDenseMap<const BindingDecl *, unsigned> BindingDeclPos;
3356 unsigned Counter = 1;
3357 for (const Expr *E : Data.PrivateVars) {
3358 Args.push_back(ImplicitParamDecl::Create(
3359 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3360 C.getPointerType(C.getPointerType(E->getType()))
3361 .withConst()
3362 .withRestrict(),
3364 const ValueDecl *VD = cast<DeclRefExpr>(E)->getDecl();
3365 if (const auto *BD = dyn_cast<BindingDecl>(VD))
3366 BindingDeclPos[cast<BindingDecl>(BD->getCanonicalDecl())] = Counter;
3367 else
3368 PrivateVarsPos[cast<VarDecl>(VD)] = Counter;
3369 ++Counter;
3370 }
3371 for (const Expr *E : Data.FirstprivateVars) {
3372 Args.push_back(ImplicitParamDecl::Create(
3373 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3374 C.getPointerType(C.getPointerType(E->getType()))
3375 .withConst()
3376 .withRestrict(),
3378 const ValueDecl *VD = cast<DeclRefExpr>(E)->getDecl();
3379 if (const auto *BD = dyn_cast<BindingDecl>(VD))
3380 BindingDeclPos[cast<BindingDecl>(BD->getCanonicalDecl())] = Counter;
3381 else
3382 PrivateVarsPos[cast<VarDecl>(VD)] = Counter;
3383 ++Counter;
3384 }
3385 for (const Expr *E : Data.LastprivateVars) {
3386 Args.push_back(ImplicitParamDecl::Create(
3387 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3388 C.getPointerType(C.getPointerType(E->getType()))
3389 .withConst()
3390 .withRestrict(),
3392 const ValueDecl *VD = cast<DeclRefExpr>(E)->getDecl();
3393 if (const auto *BD = dyn_cast<BindingDecl>(VD))
3394 BindingDeclPos[cast<BindingDecl>(BD->getCanonicalDecl())] = Counter;
3395 else
3396 PrivateVarsPos[cast<VarDecl>(VD)] = Counter;
3397 ++Counter;
3398 }
3399 for (const VarDecl *VD : Data.PrivateLocals) {
3401 if (VD->getType()->isLValueReferenceType())
3402 Ty = C.getPointerType(Ty);
3403 if (isAllocatableDecl(VD))
3404 Ty = C.getPointerType(Ty);
3405 Args.push_back(ImplicitParamDecl::Create(
3406 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3407 C.getPointerType(C.getPointerType(Ty)).withConst().withRestrict(),
3409 PrivateVarsPos[VD] = Counter;
3410 ++Counter;
3411 }
3412 const auto &TaskPrivatesMapFnInfo =
3413 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
3414 llvm::FunctionType *TaskPrivatesMapTy =
3415 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo);
3416 std::string Name =
3417 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""});
3418 auto *TaskPrivatesMap = llvm::Function::Create(
3419 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name,
3420 &CGM.getModule());
3421 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap,
3422 TaskPrivatesMapFnInfo);
3423 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3424 TaskPrivatesMap->addFnAttr("sample-profile-suffix-elision-policy",
3425 "selected");
3426 if (CGM.getCodeGenOpts().OptimizationLevel != 0) {
3427 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline);
3428 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone);
3429 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline);
3430 }
3431 CodeGenFunction CGF(CGM);
3432 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap,
3433 TaskPrivatesMapFnInfo, Args, Loc, Loc);
3434
3435 // *privi = &.privates.privi;
3436 LValue Base = CGF.EmitLoadOfPointerLValue(
3437 CGF.GetAddrOfLocalVar(TaskPrivatesArg),
3438 TaskPrivatesArg->getType()->castAs<PointerType>());
3439 const auto *PrivatesQTyRD = PrivatesQTy->castAsRecordDecl();
3440 Counter = 0;
3441 for (const FieldDecl *Field : PrivatesQTyRD->fields()) {
3442 LValue FieldLVal = CGF.EmitLValueForField(Base, Field);
3443 // Lookup by the original declaration (BindingDecl or VarDecl).
3444 const ValueDecl *LookupVD;
3445 if (Privates[Counter].second.OriginalRef) {
3446 LookupVD =
3447 cast<DeclRefExpr>(Privates[Counter].second.OriginalRef)->getDecl();
3448 } else {
3449 LookupVD = Privates[Counter].second.Original;
3450 }
3451
3452 // For BindingDecls, the privates record now stores each binding's type
3453 // directly (not the full DecompositionDecl), so FieldLVal is already
3454 // correct.
3455
3456 unsigned Position;
3457 if (const auto *BD = dyn_cast<BindingDecl>(LookupVD)) {
3458 Position =
3459 BindingDeclPos.lookup(cast<BindingDecl>(BD->getCanonicalDecl()));
3460 assert(Position && "binding not in privates mapping");
3461 } else {
3462 Position = PrivateVarsPos[cast<VarDecl>(LookupVD)];
3463 }
3464 const VarDecl *VD = Args[Position];
3465 LValue RefLVal =
3466 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType());
3467 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue(
3468 RefLVal.getAddress(), RefLVal.getType()->castAs<PointerType>());
3469 CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal);
3470 ++Counter;
3471 }
3472 CGF.FinishFunction();
3473 return TaskPrivatesMap;
3474}
3475
3476/// Emit initialization for private variables in task-based directives.
3478 const OMPExecutableDirective &D,
3479 Address KmpTaskSharedsPtr, LValue TDBase,
3480 const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3481 QualType SharedsTy, QualType SharedsPtrTy,
3482 const OMPTaskDataTy &Data,
3483 ArrayRef<PrivateDataTy> Privates, bool ForDup) {
3484 ASTContext &C = CGF.getContext();
3485 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
3486 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI);
3487 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind())
3488 ? OMPD_taskloop
3489 : OMPD_task;
3490 const CapturedStmt &CS = *D.getCapturedStmt(Kind);
3491 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS);
3492 LValue SrcBase;
3493 bool IsTargetTask =
3494 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) ||
3495 isOpenMPTargetExecutionDirective(D.getDirectiveKind());
3496 // For target-based directives skip 4 firstprivate arrays BasePointersArray,
3497 // PointersArray, SizesArray, and MappersArray. The original variables for
3498 // these arrays are not captured and we get their addresses explicitly.
3499 if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) ||
3500 (IsTargetTask && KmpTaskSharedsPtr.isValid())) {
3501 SrcBase = CGF.MakeAddrLValue(
3503 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy),
3504 CGF.ConvertTypeForMem(SharedsTy)),
3505 SharedsTy);
3506 }
3507 FI = FI->getType()->castAsRecordDecl()->field_begin();
3508 for (const PrivateDataTy &Pair : Privates) {
3509 // Do not initialize private locals.
3510 if (Pair.second.isLocalPrivate()) {
3511 ++FI;
3512 continue;
3513 }
3514 const VarDecl *VD = Pair.second.PrivateCopy;
3515 const Expr *Init = VD->getAnyInitializer();
3516 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) &&
3517 !CGF.isTrivialInitializer(Init)))) {
3518 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
3519 if (const VarDecl *Elem = Pair.second.PrivateElemInit) {
3520 const VarDecl *OriginalVD = Pair.second.Original;
3521 // Check if the variable is the target-based BasePointersArray,
3522 // PointersArray, SizesArray, or MappersArray.
3523 LValue SharedRefLValue;
3524 QualType Type = PrivateLValue.getType();
3525 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD);
3526 if (IsTargetTask && !SharedField) {
3527 assert(isa<ImplicitParamDecl>(OriginalVD) &&
3528 isa<CapturedDecl>(OriginalVD->getDeclContext()) &&
3529 cast<CapturedDecl>(OriginalVD->getDeclContext())
3530 ->getNumParams() == 0 &&
3532 cast<CapturedDecl>(OriginalVD->getDeclContext())
3533 ->getDeclContext()) &&
3534 "Expected artificial target data variable.");
3535 SharedRefLValue =
3536 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type);
3537 } else if (ForDup) {
3538 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField);
3539 // For BindingDecls, access the specific binding field within the
3540 // captured DecompositionDecl.
3541 if (Pair.second.OriginalRef) {
3542 if (const auto *DRE =
3543 dyn_cast<DeclRefExpr>(Pair.second.OriginalRef)) {
3544 if (const auto *BD = dyn_cast<BindingDecl>(DRE->getDecl())) {
3545 // Emit the binding subobject (member or array element) with
3546 // the decomposed decl temporarily mapped to the capture.
3547 const VarDecl *DD = cast<VarDecl>(BD->getDecomposedDecl());
3548 auto It = CGF.findLocalDecl(DD);
3549 bool WasMapped = It != CGF.localDeclMapEnd();
3550 Address Saved = WasMapped ? It->second : Address::invalid();
3551 if (WasMapped)
3552 It->second = SharedRefLValue.getAddress();
3553 else
3554 CGF.insertLocalDecl(DD, SharedRefLValue.getAddress());
3555 SharedRefLValue = CGF.EmitLValue(BD->getBinding());
3556 if (WasMapped) {
3557 auto RestoreIt = CGF.findLocalDecl(DD);
3558 RestoreIt->second = Saved;
3559 } else {
3560 CGF.eraseLocalDecl(DD);
3561 }
3562 }
3563 }
3564 }
3565 bool IsBinding =
3566 Pair.second.OriginalRef &&
3568 cast<DeclRefExpr>(Pair.second.OriginalRef)->getDecl());
3569 SharedRefLValue = CGF.MakeAddrLValue(
3570 SharedRefLValue.getAddress().withAlignment(
3571 IsBinding ? C.toCharUnitsFromBits(
3572 C.getTypeAlign(SharedRefLValue.getType()))
3573 : C.getDeclAlign(OriginalVD)),
3574 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl),
3575 SharedRefLValue.getTBAAInfo());
3576 } else if (CGF.LambdaCaptureFields.count(
3577 Pair.second.Original->getCanonicalDecl()) > 0 ||
3578 isa_and_nonnull<BlockDecl>(CGF.CurCodeDecl)) {
3579 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef);
3580 } else {
3581 // Processing for implicitly captured variables.
3582 InlinedOpenMPRegionRAII Region(
3583 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown,
3584 /*HasCancel=*/false, /*NoInheritance=*/true);
3585 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef);
3586 }
3587 if (Type->isArrayType()) {
3588 // Initialize firstprivate array.
3590 // Perform simple memcpy.
3591 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type);
3592 } else {
3593 // Initialize firstprivate array using element-by-element
3594 // initialization.
3596 PrivateLValue.getAddress(), SharedRefLValue.getAddress(), Type,
3597 [&CGF, Elem, Init, &CapturesInfo](Address DestElement,
3598 Address SrcElement) {
3599 // Clean up any temporaries needed by the initialization.
3600 CodeGenFunction::OMPPrivateScope InitScope(CGF);
3601 InitScope.addPrivate(Elem, SrcElement);
3602 (void)InitScope.Privatize();
3603 // Emit initialization for single element.
3604 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(
3605 CGF, &CapturesInfo);
3606 CGF.EmitAnyExprToMem(Init, DestElement,
3607 Init->getType().getQualifiers(),
3608 /*IsInitializer=*/false);
3609 });
3610 }
3611 } else {
3612 CodeGenFunction::OMPPrivateScope InitScope(CGF);
3613 InitScope.addPrivate(Elem, SharedRefLValue.getAddress());
3614 (void)InitScope.Privatize();
3615 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo);
3616 CGF.EmitExprAsInit(Init, VD, PrivateLValue,
3617 /*capturedByInit=*/false);
3618 }
3619 } else {
3620 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false);
3621 }
3622 } else if (const VarDecl *OriginalVD = Pair.second.Original) {
3623 // Handle array bindings without initializers (firstprivate only).
3624 // For private clause (PrivateElemInit is nullptr), skip initialization.
3625 // Check if OriginalRef is a BindingDecl with array type.
3626 const BindingDecl *BD = nullptr;
3627 if (Pair.second.OriginalRef) {
3628 if (const auto *DRE = dyn_cast<DeclRefExpr>(Pair.second.OriginalRef)) {
3629 BD = dyn_cast<BindingDecl>(DRE->getDecl());
3630 }
3631 }
3632 if (BD && BD->getType()->isArrayType() && Pair.second.PrivateElemInit) {
3633 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
3634 QualType Type = PrivateLValue.getType();
3635 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD);
3636
3637 // Lambda to temporarily map DecompositionDecl and emit the binding
3638 // expression.
3639 auto EmitBindingWithTempMap = [&CGF](const BindingDecl *BD,
3640 Address DDAddr) -> LValue {
3641 const VarDecl *DD = cast<VarDecl>(BD->getDecomposedDecl());
3642 auto It = CGF.findLocalDecl(DD);
3643 bool WasMapped = It != CGF.localDeclMapEnd();
3644 Address Saved = WasMapped ? It->second : Address::invalid();
3645 if (WasMapped)
3646 It->second = DDAddr;
3647 else
3648 CGF.insertLocalDecl(DD, DDAddr);
3649
3650 LValue Result = CGF.EmitLValue(BD->getBinding());
3651
3652 // Restore the mapping
3653 if (WasMapped) {
3654 auto RestoreIt = CGF.findLocalDecl(DD);
3655 RestoreIt->second = Saved;
3656 } else {
3657 CGF.eraseLocalDecl(DD);
3658 }
3659
3660 return Result;
3661 };
3662
3663 LValue SharedRefLValue;
3664 if (ForDup) {
3665 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField);
3666 SharedRefLValue =
3667 EmitBindingWithTempMap(BD, SharedRefLValue.getAddress());
3668 SharedRefLValue = CGF.MakeAddrLValue(
3669 SharedRefLValue.getAddress().withAlignment(
3670 C.getDeclAlign(OriginalVD)),
3671 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl),
3672 SharedRefLValue.getTBAAInfo());
3673 } else {
3674 // For !ForDup (first task), emit binding from parent scope using
3675 // InlinedOpenMPRegionRAII to access the correct scope
3676 InlinedOpenMPRegionRAII Region(
3677 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown,
3678 /*HasCancel=*/false, /*NoInheritance=*/true);
3679
3680 // Get the decomposed decl address from parent scope
3681 const VarDecl *DD = cast<VarDecl>(BD->getDecomposedDecl());
3682 Address DDAddr = CGF.GetAddrOfLocalVar(DD);
3683 SharedRefLValue = EmitBindingWithTempMap(BD, DDAddr);
3684 }
3685 // Perform simple memcpy for array binding
3686 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type);
3687 }
3688 }
3689 ++FI;
3690 }
3691}
3692
3693/// Check if duplication function is required for taskloops.
3696 bool InitRequired = false;
3697 for (const PrivateDataTy &Pair : Privates) {
3698 if (Pair.second.isLocalPrivate())
3699 continue;
3700 const VarDecl *VD = Pair.second.PrivateCopy;
3701 const Expr *Init = VD->getAnyInitializer();
3702 InitRequired = InitRequired || (isa_and_nonnull<CXXConstructExpr>(Init) &&
3704 if (InitRequired)
3705 break;
3706 }
3707 return InitRequired;
3708}
3709
3710
3711/// Emit task_dup function (for initialization of
3712/// private/firstprivate/lastprivate vars and last_iter flag)
3713/// \code
3714/// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int
3715/// lastpriv) {
3716/// // setup lastprivate flag
3717/// task_dst->last = lastpriv;
3718/// // could be constructor calls here...
3719/// }
3720/// \endcode
3721static llvm::Value *
3723 const OMPExecutableDirective &D,
3724 QualType KmpTaskTWithPrivatesPtrQTy,
3725 const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3726 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy,
3727 QualType SharedsPtrTy, const OMPTaskDataTy &Data,
3728 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) {
3729 ASTContext &C = CGM.getContext();
3730 auto *DstArg = ImplicitParamDecl::Create(
3731 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpTaskTWithPrivatesPtrQTy,
3733 auto *SrcArg = ImplicitParamDecl::Create(
3734 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpTaskTWithPrivatesPtrQTy,
3736 auto *LastprivArg =
3737 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy,
3739 FunctionArgList Args{DstArg, SrcArg, LastprivArg};
3740 const auto &TaskDupFnInfo =
3741 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
3742 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo);
3743 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""});
3744 auto *TaskDup = llvm::Function::Create(
3745 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
3746 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo);
3747 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3748 TaskDup->addFnAttr("sample-profile-suffix-elision-policy", "selected");
3749 TaskDup->setDoesNotRecurse();
3750 CodeGenFunction CGF(CGM);
3751 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc,
3752 Loc);
3753
3754 LValue TDBase = CGF.EmitLoadOfPointerLValue(
3755 CGF.GetAddrOfLocalVar(DstArg),
3756 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3757 // task_dst->liter = lastpriv;
3758 if (WithLastIter) {
3759 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
3760 LValue Base = CGF.EmitLValueForField(
3761 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
3762 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
3763 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar(
3764 CGF.GetAddrOfLocalVar(LastprivArg), /*Volatile=*/false, C.IntTy, Loc);
3765 CGF.EmitStoreOfScalar(Lastpriv, LILVal);
3766 }
3767
3768 // Emit initial values for private copies (if any).
3769 assert(!Privates.empty());
3770 Address KmpTaskSharedsPtr = Address::invalid();
3771 if (!Data.FirstprivateVars.empty()) {
3772 LValue TDBase = CGF.EmitLoadOfPointerLValue(
3773 CGF.GetAddrOfLocalVar(SrcArg),
3774 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3775 LValue Base = CGF.EmitLValueForField(
3776 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
3777 KmpTaskSharedsPtr = Address(
3779 Base, *std::next(KmpTaskTQTyRD->field_begin(),
3780 KmpTaskTShareds)),
3781 Loc),
3782 CGF.Int8Ty, CGM.getNaturalTypeAlignment(SharedsTy));
3783 }
3784 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD,
3785 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true);
3786 CGF.FinishFunction();
3787 return TaskDup;
3788}
3789
3790/// Checks if destructor function is required to be generated.
3791/// \return true if cleanups are required, false otherwise.
3792static bool
3793checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3795 for (const PrivateDataTy &P : Privates) {
3796 if (P.second.isLocalPrivate())
3797 continue;
3798 QualType Ty = P.second.Original->getType().getNonReferenceType();
3799 if (Ty.isDestructedType())
3800 return true;
3801 }
3802 return false;
3803}
3804
3805namespace {
3806/// Loop generator for OpenMP iterator expression.
3807class OMPIteratorGeneratorScope final
3809 CodeGenFunction &CGF;
3810 const OMPIteratorExpr *E = nullptr;
3811 SmallVector<CodeGenFunction::JumpDest, 4> ContDests;
3812 SmallVector<CodeGenFunction::JumpDest, 4> ExitDests;
3813 OMPIteratorGeneratorScope() = delete;
3814 OMPIteratorGeneratorScope(OMPIteratorGeneratorScope &) = delete;
3815
3816public:
3817 OMPIteratorGeneratorScope(CodeGenFunction &CGF, const OMPIteratorExpr *E)
3818 : CodeGenFunction::OMPPrivateScope(CGF), CGF(CGF), E(E) {
3819 if (!E)
3820 return;
3821 SmallVector<llvm::Value *, 4> Uppers;
3822 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) {
3823 Uppers.push_back(CGF.EmitScalarExpr(E->getHelper(I).Upper));
3824 const auto *VD = cast<VarDecl>(E->getIteratorDecl(I));
3825 addPrivate(VD, CGF.CreateMemTemp(VD->getType(), VD->getName()));
3826 const OMPIteratorHelperData &HelperData = E->getHelper(I);
3827 addPrivate(
3828 HelperData.CounterVD,
3829 CGF.CreateMemTemp(HelperData.CounterVD->getType(), "counter.addr"));
3830 }
3831 Privatize();
3832
3833 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) {
3834 const OMPIteratorHelperData &HelperData = E->getHelper(I);
3835 LValue CLVal =
3836 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(HelperData.CounterVD),
3837 HelperData.CounterVD->getType());
3838 // Counter = 0;
3839 CGF.EmitStoreOfScalar(
3840 llvm::ConstantInt::get(CLVal.getAddress().getElementType(), 0),
3841 CLVal);
3842 CodeGenFunction::JumpDest &ContDest =
3843 ContDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.cont"));
3844 CodeGenFunction::JumpDest &ExitDest =
3845 ExitDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.exit"));
3846 // N = <number-of_iterations>;
3847 llvm::Value *N = Uppers[I];
3848 // cont:
3849 // if (Counter < N) goto body; else goto exit;
3850 CGF.EmitBlock(ContDest.getBlock());
3851 auto *CVal =
3852 CGF.EmitLoadOfScalar(CLVal, HelperData.CounterVD->getLocation());
3853 llvm::Value *Cmp =
3854 HelperData.CounterVD->getType()->isSignedIntegerOrEnumerationType()
3855 ? CGF.Builder.CreateICmpSLT(CVal, N)
3856 : CGF.Builder.CreateICmpULT(CVal, N);
3857 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("iter.body");
3858 CGF.Builder.CreateCondBr(Cmp, BodyBB, ExitDest.getBlock());
3859 // body:
3860 CGF.EmitBlock(BodyBB);
3861 // Iteri = Begini + Counter * Stepi;
3862 CGF.EmitIgnoredExpr(HelperData.Update);
3863 }
3864 }
3865 ~OMPIteratorGeneratorScope() {
3866 if (!E)
3867 return;
3868 for (unsigned I = E->numOfIterators(); I > 0; --I) {
3869 // Counter = Counter + 1;
3870 const OMPIteratorHelperData &HelperData = E->getHelper(I - 1);
3871 CGF.EmitIgnoredExpr(HelperData.CounterUpdate);
3872 // goto cont;
3873 CGF.EmitBranchThroughCleanup(ContDests[I - 1]);
3874 // exit:
3875 CGF.EmitBlock(ExitDests[I - 1].getBlock(), /*IsFinished=*/I == 1);
3876 }
3877 }
3878};
3879} // namespace
3880
3881static std::pair<llvm::Value *, llvm::Value *>
3883 const auto *OASE = dyn_cast<OMPArrayShapingExpr>(E);
3884 llvm::Value *Addr;
3885 if (OASE) {
3886 const Expr *Base = OASE->getBase();
3887 Addr = CGF.EmitScalarExpr(Base);
3888 } else {
3889 Addr = CGF.EmitLValue(E).getPointer(CGF);
3890 }
3891 llvm::Value *SizeVal;
3892 QualType Ty = E->getType();
3893 if (OASE) {
3894 SizeVal = CGF.getTypeSize(OASE->getBase()->getType()->getPointeeType());
3895 for (const Expr *SE : OASE->getDimensions()) {
3896 llvm::Value *Sz = CGF.EmitScalarExpr(SE);
3897 Sz = CGF.EmitScalarConversion(
3898 Sz, SE->getType(), CGF.getContext().getSizeType(), SE->getExprLoc());
3899 SizeVal = CGF.Builder.CreateNUWMul(SizeVal, Sz);
3900 }
3901 } else if (const auto *ASE =
3902 dyn_cast<ArraySectionExpr>(E->IgnoreParenImpCasts())) {
3903 LValue UpAddrLVal = CGF.EmitArraySectionExpr(ASE, /*IsLowerBound=*/false);
3904 Address UpAddrAddress = UpAddrLVal.getAddress();
3905 llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32(
3906 UpAddrAddress.getElementType(), UpAddrAddress.emitRawPointer(CGF),
3907 /*Idx0=*/1);
3908 SizeVal = CGF.Builder.CreatePtrDiff(UpAddr, Addr, "", /*IsNUW=*/true);
3909 } else {
3910 SizeVal = CGF.getTypeSize(Ty);
3911 }
3912 return std::make_pair(Addr, SizeVal);
3913}
3914
3915/// Builds kmp_depend_info, if it is not built yet, and builds flags type.
3916static void getKmpAffinityType(ASTContext &C, QualType &KmpTaskAffinityInfoTy) {
3917 QualType FlagsTy = C.getIntTypeForBitwidth(32, /*Signed=*/false);
3918 if (KmpTaskAffinityInfoTy.isNull()) {
3919 RecordDecl *KmpAffinityInfoRD =
3920 C.buildImplicitRecord("kmp_task_affinity_info_t");
3921 KmpAffinityInfoRD->startDefinition();
3922 addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getIntPtrType());
3923 addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getSizeType());
3924 addFieldToRecordDecl(C, KmpAffinityInfoRD, FlagsTy);
3925 KmpAffinityInfoRD->completeDefinition();
3926 KmpTaskAffinityInfoTy = C.getCanonicalTagType(KmpAffinityInfoRD);
3927 }
3928}
3929
3932 const OMPExecutableDirective &D,
3933 llvm::Function *TaskFunction, QualType SharedsTy,
3934 Address Shareds, const OMPTaskDataTy &Data) {
3935 ASTContext &C = CGM.getContext();
3937 // Aggregate privates and sort them by the alignment.
3938 const auto *I = Data.PrivateCopies.begin();
3939 for (const Expr *E : Data.PrivateVars) {
3940 const auto *Decl = cast<DeclRefExpr>(E)->getDecl();
3941 const auto *VD = getOriginalVarDecl(Decl);
3942 const auto *CopyVD = cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl());
3943 Privates.emplace_back(C.getDeclAlign(VD),
3944 PrivateHelpersTy(E, VD, CopyVD,
3945 /*PrivateElemInit=*/nullptr));
3946 ++I;
3947 }
3948 I = Data.FirstprivateCopies.begin();
3949 const auto *IElemInitRef = Data.FirstprivateInits.begin();
3950 for (const Expr *E : Data.FirstprivateVars) {
3951 const auto *Decl = cast<DeclRefExpr>(E)->getDecl();
3952 const auto *VD = getOriginalVarDecl(Decl);
3953 const auto *CopyVD = cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl());
3954 const auto *InitVD =
3955 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl());
3956 Privates.emplace_back(C.getDeclAlign(VD),
3957 PrivateHelpersTy(E, VD, CopyVD, InitVD));
3958 ++I;
3959 ++IElemInitRef;
3960 }
3961 I = Data.LastprivateCopies.begin();
3962 for (const Expr *E : Data.LastprivateVars) {
3963 const auto *Decl = cast<DeclRefExpr>(E)->getDecl();
3964 const auto *VD = getOriginalVarDecl(Decl);
3965 const auto *CopyVD = cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl());
3966 Privates.emplace_back(C.getDeclAlign(VD),
3967 PrivateHelpersTy(E, VD, CopyVD,
3968 /*PrivateElemInit=*/nullptr));
3969 ++I;
3970 }
3971 for (const VarDecl *VD : Data.PrivateLocals) {
3972 if (isAllocatableDecl(VD))
3973 Privates.emplace_back(CGM.getPointerAlign(), PrivateHelpersTy(VD));
3974 else
3975 Privates.emplace_back(C.getDeclAlign(VD), PrivateHelpersTy(VD));
3976 }
3977 llvm::stable_sort(Privates,
3978 [](const PrivateDataTy &L, const PrivateDataTy &R) {
3979 return L.first > R.first;
3980 });
3981 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
3982 // Build type kmp_routine_entry_t (if not built yet).
3983 emitKmpRoutineEntryT(KmpInt32Ty);
3984 // Build type kmp_task_t (if not built yet).
3985 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) {
3986 if (SavedKmpTaskloopTQTy.isNull()) {
3987 SavedKmpTaskloopTQTy = C.getCanonicalTagType(createKmpTaskTRecordDecl(
3988 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
3989 }
3991 } else {
3992 assert((D.getDirectiveKind() == OMPD_task ||
3993 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) ||
3994 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) &&
3995 "Expected taskloop, task or target directive");
3996 if (SavedKmpTaskTQTy.isNull()) {
3997 SavedKmpTaskTQTy = C.getCanonicalTagType(createKmpTaskTRecordDecl(
3998 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
3999 }
4001 }
4002 const auto *KmpTaskTQTyRD = KmpTaskTQTy->castAsRecordDecl();
4003 // Build particular struct kmp_task_t for the given task.
4004 const RecordDecl *KmpTaskTWithPrivatesQTyRD =
4006 CanQualType KmpTaskTWithPrivatesQTy =
4007 C.getCanonicalTagType(KmpTaskTWithPrivatesQTyRD);
4008 QualType KmpTaskTWithPrivatesPtrQTy =
4009 C.getPointerType(KmpTaskTWithPrivatesQTy);
4010 llvm::Type *KmpTaskTWithPrivatesPtrTy = CGF.Builder.getPtrTy(0);
4011 llvm::Value *KmpTaskTWithPrivatesTySize =
4012 CGF.getTypeSize(KmpTaskTWithPrivatesQTy);
4013 QualType SharedsPtrTy = C.getPointerType(SharedsTy);
4014
4015 // Emit initial values for private copies (if any).
4016 llvm::Value *TaskPrivatesMap = nullptr;
4017 llvm::Type *TaskPrivatesMapTy =
4018 std::next(TaskFunction->arg_begin(), 3)->getType();
4019 if (!Privates.empty()) {
4020 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4021 TaskPrivatesMap =
4022 emitTaskPrivateMappingFunction(CGM, Loc, Data, FI->getType(), Privates);
4023 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4024 TaskPrivatesMap, TaskPrivatesMapTy);
4025 } else {
4026 TaskPrivatesMap = llvm::ConstantPointerNull::get(
4027 cast<llvm::PointerType>(TaskPrivatesMapTy));
4028 }
4029 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
4030 // kmp_task_t *tt);
4031 llvm::Function *TaskEntry = emitProxyTaskFunction(
4032 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
4033 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction,
4034 TaskPrivatesMap);
4035
4036 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
4037 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
4038 // kmp_routine_entry_t *task_entry);
4039 // Task flags. Format is taken from
4040 // https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h,
4041 // description of kmp_tasking_flags struct.
4042 enum {
4043 TiedFlag = 0x1,
4044 FinalFlag = 0x2,
4045 DestructorsFlag = 0x8,
4046 PriorityFlag = 0x20,
4047 DetachableFlag = 0x40,
4048 FreeAgentFlag = 0x80,
4049 TransparentFlag = 0x100,
4050 };
4051 unsigned Flags = Data.Tied ? TiedFlag : 0;
4052 bool NeedsCleanup = false;
4053 if (!Privates.empty()) {
4054 NeedsCleanup =
4055 checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD, Privates);
4056 if (NeedsCleanup)
4057 Flags = Flags | DestructorsFlag;
4058 }
4059 if (const auto *Clause = D.getSingleClause<OMPThreadsetClause>()) {
4060 OpenMPThreadsetKind Kind = Clause->getThreadsetKind();
4061 if (Kind == OMPC_THREADSET_omp_pool)
4062 Flags = Flags | FreeAgentFlag;
4063 }
4064 if (D.getSingleClause<OMPTransparentClause>())
4065 Flags |= TransparentFlag;
4066
4067 if (Data.Priority.getInt())
4068 Flags = Flags | PriorityFlag;
4069 if (D.hasClausesOfKind<OMPDetachClause>())
4070 Flags = Flags | DetachableFlag;
4071 llvm::Value *TaskFlags =
4072 Data.Final.getPointer()
4073 ? CGF.Builder.CreateSelect(Data.Final.getPointer(),
4074 CGF.Builder.getInt32(FinalFlag),
4075 CGF.Builder.getInt32(/*C=*/0))
4076 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0);
4077 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
4078 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy));
4080 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize,
4082 TaskEntry, KmpRoutineEntryPtrTy)};
4083 llvm::Value *NewTask;
4084 if (D.hasClausesOfKind<OMPNowaitClause>()) {
4085 // Check if we have any device clause associated with the directive.
4086 const Expr *Device = nullptr;
4087 if (auto *C = D.getSingleClause<OMPDeviceClause>())
4088 Device = C->getDevice();
4089 // Emit device ID if any otherwise use default value.
4090 llvm::Value *DeviceID;
4091 if (Device)
4092 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
4093 CGF.Int64Ty, /*isSigned=*/true);
4094 else
4095 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
4096 AllocArgs.push_back(DeviceID);
4097 NewTask = CGF.EmitRuntimeCall(
4098 OMPBuilder.getOrCreateRuntimeFunction(
4099 CGM.getModule(), OMPRTL___kmpc_omp_target_task_alloc),
4100 AllocArgs);
4101 } else {
4102 NewTask =
4103 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
4104 CGM.getModule(), OMPRTL___kmpc_omp_task_alloc),
4105 AllocArgs);
4106 }
4107 // Emit detach clause initialization.
4108 // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid,
4109 // task_descriptor);
4110 if (const auto *DC = D.getSingleClause<OMPDetachClause>()) {
4111 const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts();
4112 LValue EvtLVal = CGF.EmitLValue(Evt);
4113
4114 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
4115 // int gtid, kmp_task_t *task);
4116 llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc());
4117 llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc());
4118 Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false);
4119 llvm::Value *EvtVal = CGF.EmitRuntimeCall(
4120 OMPBuilder.getOrCreateRuntimeFunction(
4121 CGM.getModule(), OMPRTL___kmpc_task_allow_completion_event),
4122 {Loc, Tid, NewTask});
4123 EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(),
4124 Evt->getExprLoc());
4125 CGF.EmitStoreOfScalar(EvtVal, EvtLVal);
4126 }
4127 // Process affinity clauses.
4128 if (D.hasClausesOfKind<OMPAffinityClause>()) {
4129 // Process list of affinity data.
4130 ASTContext &C = CGM.getContext();
4131 Address AffinitiesArray = Address::invalid();
4132 // Calculate number of elements to form the array of affinity data.
4133 llvm::Value *NumOfElements = nullptr;
4134 unsigned NumAffinities = 0;
4135 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4136 if (const Expr *Modifier = C->getModifier()) {
4137 const auto *IE = cast<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts());
4138 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4139 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper);
4140 Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false);
4141 NumOfElements =
4142 NumOfElements ? CGF.Builder.CreateNUWMul(NumOfElements, Sz) : Sz;
4143 }
4144 } else {
4145 NumAffinities += C->varlist_size();
4146 }
4147 }
4149 // Fields ids in kmp_task_affinity_info record.
4150 enum RTLAffinityInfoFieldsTy { BaseAddr, Len, Flags };
4151
4152 QualType KmpTaskAffinityInfoArrayTy;
4153 if (NumOfElements) {
4154 NumOfElements = CGF.Builder.CreateNUWAdd(
4155 llvm::ConstantInt::get(CGF.SizeTy, NumAffinities), NumOfElements);
4156 auto *OVE = new (C) OpaqueValueExpr(
4157 Loc,
4158 C.getIntTypeForBitwidth(C.getTypeSize(C.getSizeType()), /*Signed=*/0),
4159 VK_PRValue);
4160 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE,
4161 RValue::get(NumOfElements));
4162 KmpTaskAffinityInfoArrayTy = C.getVariableArrayType(
4164 /*IndexTypeQuals=*/0);
4165 // Properly emit variable-sized array.
4166 auto *PD = ImplicitParamDecl::Create(C, KmpTaskAffinityInfoArrayTy,
4168 CGF.EmitVarDecl(*PD);
4169 AffinitiesArray = CGF.GetAddrOfLocalVar(PD);
4170 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty,
4171 /*isSigned=*/false);
4172 } else {
4173 KmpTaskAffinityInfoArrayTy = C.getConstantArrayType(
4175 llvm::APInt(C.getTypeSize(C.getSizeType()), NumAffinities), nullptr,
4176 ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
4177 AffinitiesArray = CGF.CreateMemTempWithoutCast(KmpTaskAffinityInfoArrayTy,
4178 ".affs.arr.addr");
4179 AffinitiesArray = CGF.Builder.CreateConstArrayGEP(AffinitiesArray, 0);
4180 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumAffinities,
4181 /*isSigned=*/false);
4182 }
4183
4184 const auto *KmpAffinityInfoRD = KmpTaskAffinityInfoTy->getAsRecordDecl();
4185 // Fill array by elements without iterators.
4186 unsigned Pos = 0;
4187 bool HasIterator = false;
4188 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4189 if (C->getModifier()) {
4190 HasIterator = true;
4191 continue;
4192 }
4193 for (const Expr *E : C->varlist()) {
4194 llvm::Value *Addr;
4195 llvm::Value *Size;
4196 std::tie(Addr, Size) = getPointerAndSize(CGF, E);
4197 LValue Base =
4198 CGF.MakeAddrLValue(CGF.Builder.CreateConstGEP(AffinitiesArray, Pos),
4200 // affs[i].base_addr = &<Affinities[i].second>;
4201 LValue BaseAddrLVal = CGF.EmitLValueForField(
4202 Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr));
4203 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy),
4204 BaseAddrLVal);
4205 // affs[i].len = sizeof(<Affinities[i].second>);
4206 LValue LenLVal = CGF.EmitLValueForField(
4207 Base, *std::next(KmpAffinityInfoRD->field_begin(), Len));
4208 CGF.EmitStoreOfScalar(Size, LenLVal);
4209 ++Pos;
4210 }
4211 }
4212 LValue PosLVal;
4213 if (HasIterator) {
4214 PosLVal = CGF.MakeAddrLValue(
4215 CGF.CreateMemTempWithoutCast(C.getSizeType(), "affs.counter.addr"),
4216 C.getSizeType());
4217 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal);
4218 }
4219 // Process elements with iterators.
4220 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4221 const Expr *Modifier = C->getModifier();
4222 if (!Modifier)
4223 continue;
4224 OMPIteratorGeneratorScope IteratorScope(
4225 CGF, cast_or_null<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts()));
4226 for (const Expr *E : C->varlist()) {
4227 llvm::Value *Addr;
4228 llvm::Value *Size;
4229 std::tie(Addr, Size) = getPointerAndSize(CGF, E);
4230 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4231 LValue Base =
4232 CGF.MakeAddrLValue(CGF.Builder.CreateGEP(CGF, AffinitiesArray, Idx),
4234 // affs[i].base_addr = &<Affinities[i].second>;
4235 LValue BaseAddrLVal = CGF.EmitLValueForField(
4236 Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr));
4237 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy),
4238 BaseAddrLVal);
4239 // affs[i].len = sizeof(<Affinities[i].second>);
4240 LValue LenLVal = CGF.EmitLValueForField(
4241 Base, *std::next(KmpAffinityInfoRD->field_begin(), Len));
4242 CGF.EmitStoreOfScalar(Size, LenLVal);
4243 Idx = CGF.Builder.CreateNUWAdd(
4244 Idx, llvm::ConstantInt::get(Idx->getType(), 1));
4245 CGF.EmitStoreOfScalar(Idx, PosLVal);
4246 }
4247 }
4248 // Call to kmp_int32 __kmpc_omp_reg_task_with_affinity(ident_t *loc_ref,
4249 // kmp_int32 gtid, kmp_task_t *new_task, kmp_int32
4250 // naffins, kmp_task_affinity_info_t *affin_list);
4251 llvm::Value *LocRef = emitUpdateLocation(CGF, Loc);
4252 llvm::Value *GTid = getThreadID(CGF, Loc);
4253 llvm::Value *AffinListPtr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4254 AffinitiesArray.emitRawPointer(CGF), CGM.VoidPtrTy);
4255 // FIXME: Emit the function and ignore its result for now unless the
4256 // runtime function is properly implemented.
4257 (void)CGF.EmitRuntimeCall(
4258 OMPBuilder.getOrCreateRuntimeFunction(
4259 CGM.getModule(), OMPRTL___kmpc_omp_reg_task_with_affinity),
4260 {LocRef, GTid, NewTask, NumOfElements, AffinListPtr});
4261 }
4262 llvm::Value *NewTaskNewTaskTTy =
4264 NewTask, KmpTaskTWithPrivatesPtrTy);
4265 LValue Base = CGF.MakeNaturalAlignRawAddrLValue(NewTaskNewTaskTTy,
4266 KmpTaskTWithPrivatesQTy);
4267 LValue TDBase =
4268 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
4269 // Fill the data in the resulting kmp_task_t record.
4270 // Copy shareds if there are any.
4271 Address KmpTaskSharedsPtr = Address::invalid();
4272 if (!SharedsTy->castAsRecordDecl()->field_empty()) {
4273 KmpTaskSharedsPtr = Address(
4274 CGF.EmitLoadOfScalar(
4276 TDBase,
4277 *std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds)),
4278 Loc),
4279 CGF.Int8Ty, CGM.getNaturalTypeAlignment(SharedsTy));
4280 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy);
4281 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy);
4282 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap);
4283 }
4284 // Emit initial values for private copies (if any).
4286 if (!Privates.empty()) {
4287 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD,
4288 SharedsTy, SharedsPtrTy, Data, Privates,
4289 /*ForDup=*/false);
4290 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) &&
4291 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) {
4292 Result.TaskDupFn = emitTaskDupFunction(
4293 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD,
4294 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates,
4295 /*WithLastIter=*/!Data.LastprivateVars.empty());
4296 }
4297 }
4298 // Fields of union "kmp_cmplrdata_t" for destructors and priority.
4299 enum { Priority = 0, Destructors = 1 };
4300 // Provide pointer to function with destructors for privates.
4301 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1);
4302 const auto *KmpCmplrdataUD = (*FI)->getType()->castAsRecordDecl();
4303 assert(KmpCmplrdataUD->isUnion());
4304 if (NeedsCleanup) {
4305 llvm::Value *DestructorFn = emitDestructorsFunction(
4306 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
4307 KmpTaskTWithPrivatesQTy);
4308 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI);
4309 LValue DestructorsLV = CGF.EmitLValueForField(
4310 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors));
4312 DestructorFn, KmpRoutineEntryPtrTy),
4313 DestructorsLV);
4314 }
4315 // Set priority.
4316 if (Data.Priority.getInt()) {
4317 LValue Data2LV = CGF.EmitLValueForField(
4318 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2));
4319 LValue PriorityLV = CGF.EmitLValueForField(
4320 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority));
4321 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV);
4322 }
4323 Result.NewTask = NewTask;
4324 Result.TaskEntry = TaskEntry;
4325 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy;
4326 Result.TDBase = TDBase;
4327 Result.KmpTaskTQTyRD = KmpTaskTQTyRD;
4328 return Result;
4329}
4330
4331/// Translates internal dependency kind into the runtime kind.
4333 RTLDependenceKindTy DepKind;
4334 switch (K) {
4335 case OMPC_DEPEND_in:
4336 DepKind = RTLDependenceKindTy::DepIn;
4337 break;
4338 // Out and InOut dependencies must use the same code.
4339 case OMPC_DEPEND_out:
4340 case OMPC_DEPEND_inout:
4341 DepKind = RTLDependenceKindTy::DepInOut;
4342 break;
4343 case OMPC_DEPEND_mutexinoutset:
4344 DepKind = RTLDependenceKindTy::DepMutexInOutSet;
4345 break;
4346 case OMPC_DEPEND_inoutset:
4347 DepKind = RTLDependenceKindTy::DepInOutSet;
4348 break;
4349 case OMPC_DEPEND_outallmemory:
4350 DepKind = RTLDependenceKindTy::DepOmpAllMem;
4351 break;
4352 case OMPC_DEPEND_source:
4353 case OMPC_DEPEND_sink:
4354 case OMPC_DEPEND_depobj:
4355 case OMPC_DEPEND_inoutallmemory:
4357 llvm_unreachable("Unknown task dependence type");
4358 }
4359 return DepKind;
4360}
4361
4362/// Builds kmp_depend_info, if it is not built yet, and builds flags type.
4363static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy,
4364 QualType &FlagsTy) {
4365 FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false);
4366 if (KmpDependInfoTy.isNull()) {
4367 RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info");
4368 KmpDependInfoRD->startDefinition();
4369 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType());
4370 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType());
4371 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy);
4372 KmpDependInfoRD->completeDefinition();
4373 KmpDependInfoTy = C.getCanonicalTagType(KmpDependInfoRD);
4374 }
4375}
4376
4377std::pair<llvm::Value *, LValue>
4379 SourceLocation Loc) {
4380 ASTContext &C = CGM.getContext();
4381 QualType FlagsTy;
4382 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4383 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4384 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
4386 DepobjLVal.getAddress().withElementType(
4387 CGF.ConvertTypeForMem(KmpDependInfoPtrTy)),
4388 KmpDependInfoPtrTy->castAs<PointerType>());
4389 Address DepObjAddr = CGF.Builder.CreateGEP(
4390 CGF, Base.getAddress(),
4391 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
4392 LValue NumDepsBase = CGF.MakeAddrLValue(
4393 DepObjAddr, KmpDependInfoTy, Base.getBaseInfo(), Base.getTBAAInfo());
4394 // NumDeps = deps[i].base_addr;
4395 LValue BaseAddrLVal = CGF.EmitLValueForField(
4396 NumDepsBase,
4397 *std::next(KmpDependInfoRD->field_begin(),
4398 static_cast<unsigned int>(RTLDependInfoFields::BaseAddr)));
4399 llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc);
4400 return std::make_pair(NumDeps, Base);
4401}
4402
4403static void emitDependData(CodeGenFunction &CGF, QualType &KmpDependInfoTy,
4404 llvm::PointerUnion<unsigned *, LValue *> Pos,
4406 Address DependenciesArray) {
4407 CodeGenModule &CGM = CGF.CGM;
4408 ASTContext &C = CGM.getContext();
4409 QualType FlagsTy;
4410 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4411 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4412 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
4413
4414 OMPIteratorGeneratorScope IteratorScope(
4415 CGF, cast_or_null<OMPIteratorExpr>(
4416 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4417 : nullptr));
4418 for (const Expr *E : Data.DepExprs) {
4419 llvm::Value *Addr;
4420 llvm::Value *Size;
4421
4422 // The expression will be a nullptr in the 'omp_all_memory' case.
4423 if (E) {
4424 std::tie(Addr, Size) = getPointerAndSize(CGF, E);
4425 Addr = CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy);
4426 } else {
4427 Addr = llvm::ConstantInt::get(CGF.IntPtrTy, 0);
4428 Size = llvm::ConstantInt::get(CGF.SizeTy, 0);
4429 }
4430 LValue Base;
4431 if (unsigned *P = dyn_cast<unsigned *>(Pos)) {
4432 Base = CGF.MakeAddrLValue(
4433 CGF.Builder.CreateConstGEP(DependenciesArray, *P), KmpDependInfoTy);
4434 } else {
4435 assert(E && "Expected a non-null expression");
4436 LValue &PosLVal = *cast<LValue *>(Pos);
4437 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4438 Base = CGF.MakeAddrLValue(
4439 CGF.Builder.CreateGEP(CGF, DependenciesArray, Idx), KmpDependInfoTy);
4440 }
4441 // deps[i].base_addr = &<Dependencies[i].second>;
4442 LValue BaseAddrLVal = CGF.EmitLValueForField(
4443 Base,
4444 *std::next(KmpDependInfoRD->field_begin(),
4445 static_cast<unsigned int>(RTLDependInfoFields::BaseAddr)));
4446 CGF.EmitStoreOfScalar(Addr, BaseAddrLVal);
4447 // deps[i].len = sizeof(<Dependencies[i].second>);
4448 LValue LenLVal = CGF.EmitLValueForField(
4449 Base, *std::next(KmpDependInfoRD->field_begin(),
4450 static_cast<unsigned int>(RTLDependInfoFields::Len)));
4451 CGF.EmitStoreOfScalar(Size, LenLVal);
4452 // deps[i].flags = <Dependencies[i].first>;
4453 RTLDependenceKindTy DepKind = translateDependencyKind(Data.DepKind);
4454 LValue FlagsLVal = CGF.EmitLValueForField(
4455 Base,
4456 *std::next(KmpDependInfoRD->field_begin(),
4457 static_cast<unsigned int>(RTLDependInfoFields::Flags)));
4459 llvm::ConstantInt::get(LLVMFlagsTy, static_cast<unsigned int>(DepKind)),
4460 FlagsLVal);
4461 if (unsigned *P = dyn_cast<unsigned *>(Pos)) {
4462 ++(*P);
4463 } else {
4464 LValue &PosLVal = *cast<LValue *>(Pos);
4465 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4466 Idx = CGF.Builder.CreateNUWAdd(Idx,
4467 llvm::ConstantInt::get(Idx->getType(), 1));
4468 CGF.EmitStoreOfScalar(Idx, PosLVal);
4469 }
4470 }
4471}
4472
4476 assert(Data.DepKind == OMPC_DEPEND_depobj &&
4477 "Expected depobj dependency kind.");
4479 SmallVector<LValue, 4> SizeLVals;
4480 ASTContext &C = CGF.getContext();
4481 {
4482 OMPIteratorGeneratorScope IteratorScope(
4483 CGF, cast_or_null<OMPIteratorExpr>(
4484 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4485 : nullptr));
4486 for (const Expr *E : Data.DepExprs) {
4487 llvm::Value *NumDeps;
4488 LValue Base;
4489 LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts());
4490 std::tie(NumDeps, Base) =
4491 getDepobjElements(CGF, DepobjLVal, E->getExprLoc());
4492 LValue NumLVal = CGF.MakeAddrLValue(
4493 CGF.CreateMemTempWithoutCast(C.getUIntPtrType(), "depobj.size.addr"),
4494 C.getUIntPtrType());
4495 CGF.Builder.CreateStore(llvm::ConstantInt::get(CGF.IntPtrTy, 0),
4496 NumLVal.getAddress());
4497 llvm::Value *PrevVal = CGF.EmitLoadOfScalar(NumLVal, E->getExprLoc());
4498 llvm::Value *Add = CGF.Builder.CreateNUWAdd(PrevVal, NumDeps);
4499 CGF.EmitStoreOfScalar(Add, NumLVal);
4500 SizeLVals.push_back(NumLVal);
4501 }
4502 }
4503 for (unsigned I = 0, E = SizeLVals.size(); I < E; ++I) {
4504 llvm::Value *Size =
4505 CGF.EmitLoadOfScalar(SizeLVals[I], Data.DepExprs[I]->getExprLoc());
4506 Sizes.push_back(Size);
4507 }
4508 return Sizes;
4509}
4510
4513 LValue PosLVal,
4515 Address DependenciesArray) {
4516 assert(Data.DepKind == OMPC_DEPEND_depobj &&
4517 "Expected depobj dependency kind.");
4518 llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy);
4519 {
4520 OMPIteratorGeneratorScope IteratorScope(
4521 CGF, cast_or_null<OMPIteratorExpr>(
4522 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4523 : nullptr));
4524 for (const Expr *E : Data.DepExprs) {
4525 llvm::Value *NumDeps;
4526 LValue Base;
4527 LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts());
4528 std::tie(NumDeps, Base) =
4529 getDepobjElements(CGF, DepobjLVal, E->getExprLoc());
4530
4531 // memcopy dependency data.
4532 llvm::Value *Size = CGF.Builder.CreateNUWMul(
4533 ElSize,
4534 CGF.Builder.CreateIntCast(NumDeps, CGF.SizeTy, /*isSigned=*/false));
4535 llvm::Value *Pos = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4536 Address DepAddr = CGF.Builder.CreateGEP(CGF, DependenciesArray, Pos);
4537 CGF.Builder.CreateMemCpy(DepAddr, Base.getAddress(), Size);
4538
4539 // Increase pos.
4540 // pos += size;
4541 llvm::Value *Add = CGF.Builder.CreateNUWAdd(Pos, NumDeps);
4542 CGF.EmitStoreOfScalar(Add, PosLVal);
4543 }
4544 }
4545}
4546
4547std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause(
4549 SourceLocation Loc) {
4550 if (llvm::all_of(Dependencies, [](const OMPTaskDataTy::DependData &D) {
4551 return D.DepExprs.empty();
4552 }))
4553 return std::make_pair(nullptr, Address::invalid());
4554 // Process list of dependencies.
4555 ASTContext &C = CGM.getContext();
4556 Address DependenciesArray = Address::invalid();
4557 llvm::Value *NumOfElements = nullptr;
4558 unsigned NumDependencies = std::accumulate(
4559 Dependencies.begin(), Dependencies.end(), 0,
4560 [](unsigned V, const OMPTaskDataTy::DependData &D) {
4561 return D.DepKind == OMPC_DEPEND_depobj
4562 ? V
4563 : (V + (D.IteratorExpr ? 0 : D.DepExprs.size()));
4564 });
4565 QualType FlagsTy;
4566 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4567 bool HasDepobjDeps = false;
4568 bool HasRegularWithIterators = false;
4569 llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0);
4570 llvm::Value *NumOfRegularWithIterators =
4571 llvm::ConstantInt::get(CGF.IntPtrTy, 0);
4572 // Calculate number of depobj dependencies and regular deps with the
4573 // iterators.
4574 for (const OMPTaskDataTy::DependData &D : Dependencies) {
4575 if (D.DepKind == OMPC_DEPEND_depobj) {
4578 for (llvm::Value *Size : Sizes) {
4579 NumOfDepobjElements =
4580 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, Size);
4581 }
4582 HasDepobjDeps = true;
4583 continue;
4584 }
4585 // Include number of iterations, if any.
4586
4587 if (const auto *IE = cast_or_null<OMPIteratorExpr>(D.IteratorExpr)) {
4588 llvm::Value *ClauseIteratorSpace =
4589 llvm::ConstantInt::get(CGF.IntPtrTy, 1);
4590 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4591 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper);
4592 Sz = CGF.Builder.CreateIntCast(Sz, CGF.IntPtrTy, /*isSigned=*/false);
4593 ClauseIteratorSpace = CGF.Builder.CreateNUWMul(Sz, ClauseIteratorSpace);
4594 }
4595 llvm::Value *NumClauseDeps = CGF.Builder.CreateNUWMul(
4596 ClauseIteratorSpace,
4597 llvm::ConstantInt::get(CGF.IntPtrTy, D.DepExprs.size()));
4598 NumOfRegularWithIterators =
4599 CGF.Builder.CreateNUWAdd(NumOfRegularWithIterators, NumClauseDeps);
4600 HasRegularWithIterators = true;
4601 continue;
4602 }
4603 }
4604
4605 QualType KmpDependInfoArrayTy;
4606 if (HasDepobjDeps || HasRegularWithIterators) {
4607 NumOfElements = llvm::ConstantInt::get(CGM.IntPtrTy, NumDependencies,
4608 /*isSigned=*/false);
4609 if (HasDepobjDeps) {
4610 NumOfElements =
4611 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumOfElements);
4612 }
4613 if (HasRegularWithIterators) {
4614 NumOfElements =
4615 CGF.Builder.CreateNUWAdd(NumOfRegularWithIterators, NumOfElements);
4616 }
4617 auto *OVE = new (C) OpaqueValueExpr(
4618 Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0),
4619 VK_PRValue);
4620 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE,
4621 RValue::get(NumOfElements));
4622 KmpDependInfoArrayTy =
4623 C.getVariableArrayType(KmpDependInfoTy, OVE, ArraySizeModifier::Normal,
4624 /*IndexTypeQuals=*/0);
4625 // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy);
4626 // Properly emit variable-sized array.
4627 auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy,
4629 CGF.EmitVarDecl(*PD);
4630 DependenciesArray = CGF.GetAddrOfLocalVar(PD);
4631 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty,
4632 /*isSigned=*/false);
4633 } else {
4634 KmpDependInfoArrayTy = C.getConstantArrayType(
4635 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), nullptr,
4636 ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
4637 DependenciesArray =
4638 CGF.CreateMemTempWithoutCast(KmpDependInfoArrayTy, ".dep.arr.addr");
4639 DependenciesArray = CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0);
4640 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies,
4641 /*isSigned=*/false);
4642 }
4643 unsigned Pos = 0;
4644 for (const OMPTaskDataTy::DependData &Dep : Dependencies) {
4645 if (Dep.DepKind == OMPC_DEPEND_depobj || Dep.IteratorExpr)
4646 continue;
4647 emitDependData(CGF, KmpDependInfoTy, &Pos, Dep, DependenciesArray);
4648 }
4649 // Copy regular dependencies with iterators.
4650 LValue PosLVal = CGF.MakeAddrLValue(
4651 CGF.CreateMemTempWithoutCast(C.getSizeType(), "dep.counter.addr"),
4652 C.getSizeType());
4653 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal);
4654 for (const OMPTaskDataTy::DependData &Dep : Dependencies) {
4655 if (Dep.DepKind == OMPC_DEPEND_depobj || !Dep.IteratorExpr)
4656 continue;
4657 emitDependData(CGF, KmpDependInfoTy, &PosLVal, Dep, DependenciesArray);
4658 }
4659 // Copy final depobj arrays without iterators.
4660 if (HasDepobjDeps) {
4661 for (const OMPTaskDataTy::DependData &Dep : Dependencies) {
4662 if (Dep.DepKind != OMPC_DEPEND_depobj)
4663 continue;
4664 emitDepobjElements(CGF, KmpDependInfoTy, PosLVal, Dep, DependenciesArray);
4665 }
4666 }
4667 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4668 DependenciesArray, CGF.VoidPtrTy, CGF.Int8Ty);
4669 return std::make_pair(NumOfElements, DependenciesArray);
4670}
4671
4673 CodeGenFunction &CGF, const OMPTaskDataTy::DependData &Dependencies,
4674 SourceLocation Loc) {
4675 if (Dependencies.DepExprs.empty())
4676 return Address::invalid();
4677 // Process list of dependencies.
4678 ASTContext &C = CGM.getContext();
4679 Address DependenciesArray = Address::invalid();
4680 unsigned NumDependencies = Dependencies.DepExprs.size();
4681 QualType FlagsTy;
4682 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4683 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4684
4685 llvm::Value *Size;
4686 // Define type kmp_depend_info[<Dependencies.size()>];
4687 // For depobj reserve one extra element to store the number of elements.
4688 // It is required to handle depobj(x) update(in) construct.
4689 // kmp_depend_info[<Dependencies.size()>] deps;
4690 llvm::Value *NumDepsVal;
4691 CharUnits Align = C.getTypeAlignInChars(KmpDependInfoTy);
4692 if (const auto *IE =
4693 cast_or_null<OMPIteratorExpr>(Dependencies.IteratorExpr)) {
4694 NumDepsVal = llvm::ConstantInt::get(CGF.SizeTy, 1);
4695 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4696 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper);
4697 Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false);
4698 NumDepsVal = CGF.Builder.CreateNUWMul(NumDepsVal, Sz);
4699 }
4700 Size = CGF.Builder.CreateNUWAdd(llvm::ConstantInt::get(CGF.SizeTy, 1),
4701 NumDepsVal);
4702 CharUnits SizeInBytes =
4703 C.getTypeSizeInChars(KmpDependInfoTy).alignTo(Align);
4704 llvm::Value *RecSize = CGM.getSize(SizeInBytes);
4705 Size = CGF.Builder.CreateNUWMul(Size, RecSize);
4706 NumDepsVal =
4707 CGF.Builder.CreateIntCast(NumDepsVal, CGF.IntPtrTy, /*isSigned=*/false);
4708 } else {
4709 QualType KmpDependInfoArrayTy = C.getConstantArrayType(
4710 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1),
4711 nullptr, ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
4712 CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy);
4713 Size = CGM.getSize(Sz.alignTo(Align));
4714 NumDepsVal = llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies);
4715 }
4716 // Need to allocate on the dynamic memory.
4717 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4718 // Use default allocator.
4719 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4720 llvm::Value *Args[] = {ThreadID, Size, Allocator};
4721
4722 llvm::Value *Addr =
4723 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
4724 CGM.getModule(), OMPRTL___kmpc_alloc),
4725 Args, ".dep.arr.addr");
4726 llvm::Type *KmpDependInfoLlvmTy = CGF.ConvertTypeForMem(KmpDependInfoTy);
4728 Addr, CGF.Builder.getPtrTy(0));
4729 DependenciesArray = Address(Addr, KmpDependInfoLlvmTy, Align);
4730 // Write number of elements in the first element of array for depobj.
4731 LValue Base = CGF.MakeAddrLValue(DependenciesArray, KmpDependInfoTy);
4732 // deps[i].base_addr = NumDependencies;
4733 LValue BaseAddrLVal = CGF.EmitLValueForField(
4734 Base,
4735 *std::next(KmpDependInfoRD->field_begin(),
4736 static_cast<unsigned int>(RTLDependInfoFields::BaseAddr)));
4737 CGF.EmitStoreOfScalar(NumDepsVal, BaseAddrLVal);
4738 llvm::PointerUnion<unsigned *, LValue *> Pos;
4739 unsigned Idx = 1;
4740 LValue PosLVal;
4741 if (Dependencies.IteratorExpr) {
4742 PosLVal = CGF.MakeAddrLValue(
4743 CGF.CreateMemTempWithoutCast(C.getSizeType(), "iterator.counter.addr"),
4744 C.getSizeType());
4745 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Idx), PosLVal,
4746 /*IsInit=*/true);
4747 Pos = &PosLVal;
4748 } else {
4749 Pos = &Idx;
4750 }
4751 emitDependData(CGF, KmpDependInfoTy, Pos, Dependencies, DependenciesArray);
4752 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4753 CGF.Builder.CreateConstGEP(DependenciesArray, 1), CGF.VoidPtrTy,
4754 CGF.Int8Ty);
4755 return DependenciesArray;
4756}
4757
4759 SourceLocation Loc) {
4760 ASTContext &C = CGM.getContext();
4761 QualType FlagsTy;
4762 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4763 LValue Base = CGF.EmitLoadOfPointerLValue(DepobjLVal.getAddress(),
4764 C.VoidPtrTy.castAs<PointerType>());
4765 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
4767 Base.getAddress(), CGF.ConvertTypeForMem(KmpDependInfoPtrTy),
4769 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP(
4770 Addr.getElementType(), Addr.emitRawPointer(CGF),
4771 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
4772 DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr,
4773 CGF.VoidPtrTy);
4774 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4775 // Use default allocator.
4776 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4777 llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator};
4778
4779 // _kmpc_free(gtid, addr, nullptr);
4780 (void)CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
4781 CGM.getModule(), OMPRTL___kmpc_free),
4782 Args);
4783}
4784
4786 CodeGenFunction &CGF, LValue DepobjLVal, OpenMPDependClauseKind NewDepKind,
4787 SourceLocation Loc) {
4788 ASTContext &C = CGM.getContext();
4789 QualType FlagsTy;
4790 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4791 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4792 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
4793 llvm::Value *NumDeps;
4794 LValue Base;
4795 std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc);
4796
4797 Address Begin = Base.getAddress();
4798 // Cast from pointer to array type to pointer to single element.
4799 llvm::Value *End = CGF.Builder.CreateGEP(Begin.getElementType(),
4800 Begin.emitRawPointer(CGF), NumDeps);
4801 // The basic structure here is a while-do loop.
4802 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body");
4803 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done");
4804 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
4805 CGF.EmitBlock(BodyBB);
4806 llvm::PHINode *ElementPHI =
4807 CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast");
4808 ElementPHI->addIncoming(Begin.emitRawPointer(CGF), EntryBB);
4809 Begin = Begin.withPointer(ElementPHI, KnownNonNull);
4810 Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(),
4811 Base.getTBAAInfo());
4812 // deps[i].flags = NewDepKind;
4813 RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind);
4814 LValue FlagsLVal = CGF.EmitLValueForField(
4815 Base, *std::next(KmpDependInfoRD->field_begin(),
4816 static_cast<unsigned int>(RTLDependInfoFields::Flags)));
4818 llvm::ConstantInt::get(LLVMFlagsTy, static_cast<unsigned int>(DepKind)),
4819 FlagsLVal);
4820
4821 // Shift the address forward by one element.
4822 llvm::Value *ElementNext =
4823 CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext")
4824 .emitRawPointer(CGF);
4825 ElementPHI->addIncoming(ElementNext, CGF.Builder.GetInsertBlock());
4826 llvm::Value *IsEmpty =
4827 CGF.Builder.CreateICmpEQ(ElementNext, End, "omp.isempty");
4828 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
4829 // Done.
4830 CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
4831}
4832
4834 const OMPExecutableDirective &D,
4835 llvm::Function *TaskFunction,
4836 QualType SharedsTy, Address Shareds,
4837 const Expr *IfCond,
4838 const OMPTaskDataTy &Data) {
4839 if (!CGF.HaveInsertPoint())
4840 return;
4841
4843 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
4844 llvm::Value *NewTask = Result.NewTask;
4845 llvm::Function *TaskEntry = Result.TaskEntry;
4846 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy;
4847 LValue TDBase = Result.TDBase;
4848 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD;
4849 // Process list of dependences.
4850 Address DependenciesArray = Address::invalid();
4851 llvm::Value *NumOfElements;
4852 std::tie(NumOfElements, DependenciesArray) =
4853 emitDependClause(CGF, Data.Dependences, Loc);
4854
4855 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
4856 // libcall.
4857 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
4858 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
4859 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence
4860 // list is not empty
4861 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4862 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
4863 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask };
4864 llvm::Value *DepTaskArgs[7];
4865 if (!Data.Dependences.empty()) {
4866 DepTaskArgs[0] = UpLoc;
4867 DepTaskArgs[1] = ThreadID;
4868 DepTaskArgs[2] = NewTask;
4869 DepTaskArgs[3] = NumOfElements;
4870 DepTaskArgs[4] = DependenciesArray.emitRawPointer(CGF);
4871 DepTaskArgs[5] = CGF.Builder.getInt32(0);
4872 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4873 }
4874 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs,
4875 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) {
4876 if (!Data.Tied) {
4877 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
4878 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI);
4879 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal);
4880 }
4881 if (!Data.Dependences.empty()) {
4882 CGF.EmitRuntimeCall(
4883 OMPBuilder.getOrCreateRuntimeFunction(
4884 CGM.getModule(), OMPRTL___kmpc_omp_task_with_deps),
4885 DepTaskArgs);
4886 } else {
4887 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
4888 CGM.getModule(), OMPRTL___kmpc_omp_task),
4889 TaskArgs);
4890 }
4891 // Check if parent region is untied and build return for untied task;
4892 if (auto *Region =
4893 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
4894 Region->emitUntiedSwitch(CGF);
4895 };
4896
4897 llvm::Value *DepWaitTaskArgs[7];
4898 if (!Data.Dependences.empty()) {
4899 DepWaitTaskArgs[0] = UpLoc;
4900 DepWaitTaskArgs[1] = ThreadID;
4901 DepWaitTaskArgs[2] = NumOfElements;
4902 DepWaitTaskArgs[3] = DependenciesArray.emitRawPointer(CGF);
4903 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
4904 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4905 DepWaitTaskArgs[6] =
4906 llvm::ConstantInt::get(CGF.Int32Ty, Data.HasNowaitClause);
4907 }
4908 auto &M = CGM.getModule();
4909 auto &&ElseCodeGen = [this, &M, &TaskArgs, ThreadID, NewTaskNewTaskTTy,
4910 TaskEntry, &Data, &DepWaitTaskArgs,
4911 Loc](CodeGenFunction &CGF, PrePostActionTy &) {
4912 CodeGenFunction::RunCleanupsScope LocalScope(CGF);
4913 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
4914 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
4915 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
4916 // is specified.
4917 if (!Data.Dependences.empty())
4918 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
4919 M, OMPRTL___kmpc_omp_taskwait_deps_51),
4920 DepWaitTaskArgs);
4921 // Call proxy_task_entry(gtid, new_task);
4922 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy,
4923 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
4924 Action.Enter(CGF);
4925 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
4926 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry,
4927 OutlinedFnArgs);
4928 };
4929
4930 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
4931 // kmp_task_t *new_task);
4932 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
4933 // kmp_task_t *new_task);
4935 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
4936 M, OMPRTL___kmpc_omp_task_begin_if0),
4937 TaskArgs,
4938 OMPBuilder.getOrCreateRuntimeFunction(
4939 M, OMPRTL___kmpc_omp_task_complete_if0),
4940 TaskArgs);
4941 RCG.setAction(Action);
4942 RCG(CGF);
4943 };
4944
4945 if (IfCond) {
4946 emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen);
4947 } else {
4948 RegionCodeGenTy ThenRCG(ThenCodeGen);
4949 ThenRCG(CGF);
4950 }
4951}
4952
4954 const OMPLoopDirective &D,
4955 llvm::Function *TaskFunction,
4956 QualType SharedsTy, Address Shareds,
4957 const Expr *IfCond,
4958 const OMPTaskDataTy &Data) {
4959 if (!CGF.HaveInsertPoint())
4960 return;
4962 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
4963 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
4964 // libcall.
4965 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
4966 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
4967 // sched, kmp_uint64 grainsize, void *task_dup);
4968 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4969 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
4970 llvm::Value *IfVal;
4971 if (IfCond) {
4972 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy,
4973 /*isSigned=*/true);
4974 } else {
4975 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1);
4976 }
4977
4978 LValue LBLVal = CGF.EmitLValueForField(
4979 Result.TDBase,
4980 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound));
4981 const auto *LBVar =
4983 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(), LBLVal.getQuals(),
4984 /*IsInitializer=*/true);
4985 LValue UBLVal = CGF.EmitLValueForField(
4986 Result.TDBase,
4987 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound));
4988 const auto *UBVar =
4990 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(), UBLVal.getQuals(),
4991 /*IsInitializer=*/true);
4992 LValue StLVal = CGF.EmitLValueForField(
4993 Result.TDBase,
4994 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride));
4995 const auto *StVar =
4997 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(), StLVal.getQuals(),
4998 /*IsInitializer=*/true);
4999 // Store reductions address.
5000 LValue RedLVal = CGF.EmitLValueForField(
5001 Result.TDBase,
5002 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions));
5003 if (Data.Reductions) {
5004 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal);
5005 } else {
5006 CGF.EmitNullInitialization(RedLVal.getAddress(),
5007 CGF.getContext().VoidPtrTy);
5008 }
5009 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 };
5011 UpLoc,
5012 ThreadID,
5013 Result.NewTask,
5014 IfVal,
5015 LBLVal.getPointer(CGF),
5016 UBLVal.getPointer(CGF),
5017 CGF.EmitLoadOfScalar(StLVal, Loc),
5018 llvm::ConstantInt::getSigned(
5019 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler
5020 llvm::ConstantInt::getSigned(
5021 CGF.IntTy, Data.Schedule.getPointer()
5022 ? Data.Schedule.getInt() ? NumTasks : Grainsize
5023 : NoSchedule),
5024 Data.Schedule.getPointer()
5025 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty,
5026 /*isSigned=*/false)
5027 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0)};
5028 if (Data.HasModifier)
5029 TaskArgs.push_back(llvm::ConstantInt::get(CGF.Int32Ty, 1));
5030
5031 TaskArgs.push_back(Result.TaskDupFn
5033 Result.TaskDupFn, CGF.VoidPtrTy)
5034 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy));
5035 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5036 CGM.getModule(), Data.HasModifier
5037 ? OMPRTL___kmpc_taskloop_5
5038 : OMPRTL___kmpc_taskloop),
5039 TaskArgs);
5040}
5041
5042/// Emit reduction operation for each element of array (required for
5043/// array sections) LHS op = RHS.
5044/// \param Type Type of array.
5045/// \param LHSVar Variable on the left side of the reduction operation
5046/// (references element of array in original variable).
5047/// \param RHSVar Variable on the right side of the reduction operation
5048/// (references element of array in original variable).
5049/// \param RedOpGen Generator of reduction operation with use of LHSVar and
5050/// RHSVar.
5052 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar,
5053 const VarDecl *RHSVar,
5054 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *,
5055 const Expr *, const Expr *)> &RedOpGen,
5056 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr,
5057 const Expr *UpExpr = nullptr) {
5058 // Perform element-by-element initialization.
5059 QualType ElementTy;
5060 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar);
5061 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar);
5062
5063 // Drill down to the base element type on both arrays.
5064 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
5065 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr);
5066
5067 llvm::Value *RHSBegin = RHSAddr.emitRawPointer(CGF);
5068 llvm::Value *LHSBegin = LHSAddr.emitRawPointer(CGF);
5069 // Cast from pointer to array type to pointer to single element.
5070 llvm::Value *LHSEnd =
5071 CGF.Builder.CreateGEP(LHSAddr.getElementType(), LHSBegin, NumElements);
5072 // The basic structure here is a while-do loop.
5073 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body");
5074 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done");
5075 llvm::Value *IsEmpty =
5076 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty");
5077 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5078
5079 // Enter the loop body, making that address the current address.
5080 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5081 CGF.EmitBlock(BodyBB);
5082
5083 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
5084
5085 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI(
5086 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast");
5087 RHSElementPHI->addIncoming(RHSBegin, EntryBB);
5088 Address RHSElementCurrent(
5089 RHSElementPHI, RHSAddr.getElementType(),
5090 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5091
5092 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI(
5093 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast");
5094 LHSElementPHI->addIncoming(LHSBegin, EntryBB);
5095 Address LHSElementCurrent(
5096 LHSElementPHI, LHSAddr.getElementType(),
5097 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5098
5099 // Emit copy.
5101 Scope.addPrivate(LHSVar, LHSElementCurrent);
5102 Scope.addPrivate(RHSVar, RHSElementCurrent);
5103 Scope.Privatize();
5104 RedOpGen(CGF, XExpr, EExpr, UpExpr);
5105 Scope.ForceCleanup();
5106
5107 // Shift the address forward by one element.
5108 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32(
5109 LHSAddr.getElementType(), LHSElementPHI, /*Idx0=*/1,
5110 "omp.arraycpy.dest.element");
5111 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32(
5112 RHSAddr.getElementType(), RHSElementPHI, /*Idx0=*/1,
5113 "omp.arraycpy.src.element");
5114 // Check whether we've reached the end.
5115 llvm::Value *Done =
5116 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done");
5117 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
5118 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock());
5119 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock());
5120
5121 // Done.
5122 CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5123}
5124
5125/// Emit reduction combiner. If the combiner is a simple expression emit it as
5126/// is, otherwise consider it as combiner of UDR decl and emit it as a call of
5127/// UDR combiner function.
5129 const Expr *ReductionOp) {
5130 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
5131 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
5132 if (const auto *DRE =
5133 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
5134 if (const auto *DRD =
5135 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) {
5136 std::pair<llvm::Function *, llvm::Function *> Reduction =
5140 CGF.EmitIgnoredExpr(ReductionOp);
5141 return;
5142 }
5143 CGF.EmitIgnoredExpr(ReductionOp);
5144}
5145
5147 StringRef ReducerName, SourceLocation Loc, llvm::Type *ArgsElemType,
5149 ArrayRef<const Expr *> RHSExprs, ArrayRef<const Expr *> ReductionOps) {
5150 ASTContext &C = CGM.getContext();
5151
5152 // void reduction_func(void *LHSArg, void *RHSArg);
5153 auto *LHSArg =
5154 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5155 C.VoidPtrTy, ImplicitParamKind::Other);
5156 auto *RHSArg =
5157 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5158 C.VoidPtrTy, ImplicitParamKind::Other);
5159 FunctionArgList Args{LHSArg, RHSArg};
5160 const auto &CGFI =
5161 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5162 std::string Name = getReductionFuncName(ReducerName);
5163 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
5164 llvm::GlobalValue::InternalLinkage, Name,
5165 &CGM.getModule());
5166 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
5167 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
5168 Fn->addFnAttr("sample-profile-suffix-elision-policy", "selected");
5169 Fn->setDoesNotRecurse();
5170 CodeGenFunction CGF(CGM);
5171 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
5172
5173 // Dst = (void*[n])(LHSArg);
5174 // Src = (void*[n])(RHSArg);
5176 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(LHSArg)),
5177 CGF.Builder.getPtrTy(0)),
5178 ArgsElemType, CGF.getPointerAlign());
5180 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(RHSArg)),
5181 CGF.Builder.getPtrTy(0)),
5182 ArgsElemType, CGF.getPointerAlign());
5183
5184 // ...
5185 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
5186 // ...
5188 const auto *IPriv = Privates.begin();
5189 unsigned Idx = 0;
5190 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) {
5191 const auto *RHSVar =
5192 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl());
5193 Scope.addPrivate(RHSVar, emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar));
5194 const auto *LHSVar =
5195 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl());
5196 Scope.addPrivate(LHSVar, emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar));
5197 QualType PrivTy = (*IPriv)->getType();
5198 if (PrivTy->isVariablyModifiedType()) {
5199 // Get array size and emit VLA type.
5200 ++Idx;
5201 Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx);
5202 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem);
5203 const VariableArrayType *VLA =
5204 CGF.getContext().getAsVariableArrayType(PrivTy);
5205 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr());
5207 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy)));
5208 CGF.EmitVariablyModifiedType(PrivTy);
5209 }
5210 }
5211 Scope.Privatize();
5212 IPriv = Privates.begin();
5213 const auto *ILHS = LHSExprs.begin();
5214 const auto *IRHS = RHSExprs.begin();
5215 for (const Expr *E : ReductionOps) {
5216 if ((*IPriv)->getType()->isArrayType()) {
5217 // Emit reduction for array section.
5218 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5219 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5221 CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5222 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5223 emitReductionCombiner(CGF, E);
5224 });
5225 } else {
5226 // Emit reduction for array subscript or single variable.
5227 emitReductionCombiner(CGF, E);
5228 }
5229 ++IPriv;
5230 ++ILHS;
5231 ++IRHS;
5232 }
5233 Scope.ForceCleanup();
5234 CGF.FinishFunction();
5235 return Fn;
5236}
5237
5239 const Expr *ReductionOp,
5240 const Expr *PrivateRef,
5241 const DeclRefExpr *LHS,
5242 const DeclRefExpr *RHS) {
5243 if (PrivateRef->getType()->isArrayType()) {
5244 // Emit reduction for array section.
5245 const auto *LHSVar = cast<VarDecl>(LHS->getDecl());
5246 const auto *RHSVar = cast<VarDecl>(RHS->getDecl());
5248 CGF, PrivateRef->getType(), LHSVar, RHSVar,
5249 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5250 emitReductionCombiner(CGF, ReductionOp);
5251 });
5252 } else {
5253 // Emit reduction for array subscript or single variable.
5254 emitReductionCombiner(CGF, ReductionOp);
5255 }
5256}
5257
5258static std::string generateUniqueName(CodeGenModule &CGM,
5259 llvm::StringRef Prefix, const Expr *Ref);
5260
5262 CodeGenFunction &CGF, SourceLocation Loc, const Expr *Privates,
5263 const Expr *LHSExprs, const Expr *RHSExprs, const Expr *ReductionOps) {
5264
5265 // Create a shared global variable (__shared_reduction_var) to accumulate the
5266 // final result.
5267 //
5268 // Call __kmpc_barrier to synchronize threads before initialization.
5269 //
5270 // The master thread (thread_id == 0) initializes __shared_reduction_var
5271 // with the identity value or initializer.
5272 //
5273 // Call __kmpc_barrier to synchronize before combining.
5274 // For each i:
5275 // - Thread enters critical section.
5276 // - Reads its private value from LHSExprs[i].
5277 // - Updates __shared_reduction_var[i] = RedOp_i(__shared_reduction_var[i],
5278 // Privates[i]).
5279 // - Exits critical section.
5280 //
5281 // Call __kmpc_barrier after combining.
5282 //
5283 // Each thread copies __shared_reduction_var[i] back to RHSExprs[i].
5284 //
5285 // Final __kmpc_barrier to synchronize after broadcasting
5286 QualType PrivateType = Privates->getType();
5287 llvm::Type *LLVMType = CGF.ConvertTypeForMem(PrivateType);
5288
5289 const OMPDeclareReductionDecl *UDR = getReductionInit(ReductionOps);
5290 std::string ReductionVarNameStr;
5291 if (const auto *DRE = dyn_cast<DeclRefExpr>(Privates->IgnoreParenCasts()))
5292 ReductionVarNameStr =
5293 generateUniqueName(CGM, DRE->getDecl()->getNameAsString(), Privates);
5294 else
5295 ReductionVarNameStr = "unnamed_priv_var";
5296
5297 // Create an internal shared variable
5298 std::string SharedName =
5299 CGM.getOpenMPRuntime().getName({"internal_pivate_", ReductionVarNameStr});
5300 llvm::GlobalVariable *SharedVar = OMPBuilder.getOrCreateInternalVariable(
5301 LLVMType, ".omp.reduction." + SharedName);
5302
5303 SharedVar->setAlignment(
5304 llvm::MaybeAlign(CGF.getContext().getTypeAlign(PrivateType) / 8));
5305
5306 Address SharedResult =
5307 CGF.MakeNaturalAlignRawAddrLValue(SharedVar, PrivateType).getAddress();
5308
5309 llvm::Value *ThreadId = getThreadID(CGF, Loc);
5310 llvm::Value *BarrierLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE);
5311 llvm::Value *BarrierArgs[] = {BarrierLoc, ThreadId};
5312
5313 llvm::BasicBlock *InitBB = CGF.createBasicBlock("init");
5314 llvm::BasicBlock *InitEndBB = CGF.createBasicBlock("init.end");
5315
5316 llvm::Value *IsWorker = CGF.Builder.CreateICmpEQ(
5317 ThreadId, llvm::ConstantInt::get(ThreadId->getType(), 0));
5318 CGF.Builder.CreateCondBr(IsWorker, InitBB, InitEndBB);
5319
5320 CGF.EmitBlock(InitBB);
5321
5322 auto EmitSharedInit = [&]() {
5323 if (UDR) { // Check if it's a User-Defined Reduction
5324 if (const Expr *UDRInitExpr = UDR->getInitializer()) {
5325 std::pair<llvm::Function *, llvm::Function *> FnPair =
5327 llvm::Function *InitializerFn = FnPair.second;
5328 if (InitializerFn) {
5329 if (const auto *CE =
5330 dyn_cast<CallExpr>(UDRInitExpr->IgnoreParenImpCasts())) {
5331 const auto *OutDRE = cast<DeclRefExpr>(
5332 cast<UnaryOperator>(CE->getArg(0)->IgnoreParenImpCasts())
5333 ->getSubExpr());
5334 const VarDecl *OutVD = cast<VarDecl>(OutDRE->getDecl());
5335
5336 CodeGenFunction::OMPPrivateScope LocalScope(CGF);
5337 LocalScope.addPrivate(OutVD, SharedResult);
5338
5339 (void)LocalScope.Privatize();
5340 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(
5341 CE->getCallee()->IgnoreParenImpCasts())) {
5343 CGF, OVE, RValue::get(InitializerFn));
5344 CGF.EmitIgnoredExpr(CE);
5345 } else {
5346 CGF.EmitAnyExprToMem(UDRInitExpr, SharedResult,
5347 PrivateType.getQualifiers(),
5348 /*IsInitializer=*/true);
5349 }
5350 } else {
5351 CGF.EmitAnyExprToMem(UDRInitExpr, SharedResult,
5352 PrivateType.getQualifiers(),
5353 /*IsInitializer=*/true);
5354 }
5355 } else {
5356 CGF.EmitAnyExprToMem(UDRInitExpr, SharedResult,
5357 PrivateType.getQualifiers(),
5358 /*IsInitializer=*/true);
5359 }
5360 } else {
5361 // EmitNullInitialization handles default construction for C++ classes
5362 // and zeroing for scalars, which is a reasonable default.
5363 CGF.EmitNullInitialization(SharedResult, PrivateType);
5364 }
5365 return; // UDR initialization handled
5366 }
5367 if (const auto *DRE = dyn_cast<DeclRefExpr>(Privates)) {
5368 if (const auto *VD = dyn_cast<VarDecl>(DRE->getDecl())) {
5369 if (const Expr *InitExpr = VD->getInit()) {
5370 CGF.EmitAnyExprToMem(InitExpr, SharedResult,
5371 PrivateType.getQualifiers(), true);
5372 return;
5373 }
5374 }
5375 }
5376 CGF.EmitNullInitialization(SharedResult, PrivateType);
5377 };
5378 EmitSharedInit();
5379 CGF.Builder.CreateBr(InitEndBB);
5380 CGF.EmitBlock(InitEndBB);
5381
5382 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5383 CGM.getModule(), OMPRTL___kmpc_barrier),
5384 BarrierArgs);
5385
5386 const Expr *ReductionOp = ReductionOps;
5387 const OMPDeclareReductionDecl *CurrentUDR = getReductionInit(ReductionOp);
5388 LValue SharedLV = CGF.MakeAddrLValue(SharedResult, PrivateType);
5389 LValue LHSLV = CGF.EmitLValue(Privates);
5390
5391 auto EmitCriticalReduction = [&](auto ReductionGen) {
5392 std::string CriticalName = getName({"reduction_critical"});
5393 emitCriticalRegion(CGF, CriticalName, ReductionGen, Loc);
5394 };
5395
5396 if (CurrentUDR) {
5397 // Handle user-defined reduction.
5398 auto ReductionGen = [&](CodeGenFunction &CGF, PrePostActionTy &Action) {
5399 Action.Enter(CGF);
5400 std::pair<llvm::Function *, llvm::Function *> FnPair =
5401 getUserDefinedReduction(CurrentUDR);
5402 if (FnPair.first) {
5403 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) {
5404 const auto *OutDRE = cast<DeclRefExpr>(
5405 cast<UnaryOperator>(CE->getArg(0)->IgnoreParenImpCasts())
5406 ->getSubExpr());
5407 const auto *InDRE = cast<DeclRefExpr>(
5408 cast<UnaryOperator>(CE->getArg(1)->IgnoreParenImpCasts())
5409 ->getSubExpr());
5410 CodeGenFunction::OMPPrivateScope LocalScope(CGF);
5411 LocalScope.addPrivate(cast<VarDecl>(OutDRE->getDecl()),
5412 SharedLV.getAddress());
5413 LocalScope.addPrivate(cast<VarDecl>(InDRE->getDecl()),
5414 LHSLV.getAddress());
5415 (void)LocalScope.Privatize();
5416 emitReductionCombiner(CGF, ReductionOp);
5417 }
5418 }
5419 };
5420 EmitCriticalReduction(ReductionGen);
5421 } else {
5422 // Handle built-in reduction operations.
5423#ifndef NDEBUG
5424 const Expr *ReductionClauseExpr = ReductionOp->IgnoreParenCasts();
5425 if (const auto *Cleanup = dyn_cast<ExprWithCleanups>(ReductionClauseExpr))
5426 ReductionClauseExpr = Cleanup->getSubExpr()->IgnoreParenCasts();
5427
5428 const Expr *AssignRHS = nullptr;
5429 if (const auto *BinOp = dyn_cast<BinaryOperator>(ReductionClauseExpr)) {
5430 if (BinOp->getOpcode() == BO_Assign)
5431 AssignRHS = BinOp->getRHS();
5432 } else if (const auto *OpCall =
5433 dyn_cast<CXXOperatorCallExpr>(ReductionClauseExpr)) {
5434 if (OpCall->getOperator() == OO_Equal)
5435 AssignRHS = OpCall->getArg(1);
5436 }
5437
5438 assert(AssignRHS &&
5439 "Private Variable Reduction : Invalid ReductionOp expression");
5440#endif
5441
5442 auto ReductionGen = [&](CodeGenFunction &CGF, PrePostActionTy &Action) {
5443 Action.Enter(CGF);
5444 const auto *OmpOutDRE =
5445 dyn_cast<DeclRefExpr>(LHSExprs->IgnoreParenImpCasts());
5446 const auto *OmpInDRE =
5447 dyn_cast<DeclRefExpr>(RHSExprs->IgnoreParenImpCasts());
5448 assert(
5449 OmpOutDRE && OmpInDRE &&
5450 "Private Variable Reduction : LHSExpr/RHSExpr must be DeclRefExprs");
5451 const VarDecl *OmpOutVD = cast<VarDecl>(OmpOutDRE->getDecl());
5452 const VarDecl *OmpInVD = cast<VarDecl>(OmpInDRE->getDecl());
5453 CodeGenFunction::OMPPrivateScope LocalScope(CGF);
5454 LocalScope.addPrivate(OmpOutVD, SharedLV.getAddress());
5455 LocalScope.addPrivate(OmpInVD, LHSLV.getAddress());
5456 (void)LocalScope.Privatize();
5457 // Emit the actual reduction operation
5458 CGF.EmitIgnoredExpr(ReductionOp);
5459 };
5460 EmitCriticalReduction(ReductionGen);
5461 }
5462
5463 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5464 CGM.getModule(), OMPRTL___kmpc_barrier),
5465 BarrierArgs);
5466
5467 // Broadcast final result
5468 bool IsAggregate = PrivateType->isAggregateType();
5469 LValue SharedLV1 = CGF.MakeAddrLValue(SharedResult, PrivateType);
5470 llvm::Value *FinalResultVal = nullptr;
5471 Address FinalResultAddr = Address::invalid();
5472
5473 if (IsAggregate)
5474 FinalResultAddr = SharedResult;
5475 else
5476 FinalResultVal = CGF.EmitLoadOfScalar(SharedLV1, Loc);
5477
5478 LValue TargetLHSLV = CGF.EmitLValue(RHSExprs);
5479 if (IsAggregate) {
5480 CGF.EmitAggregateCopy(TargetLHSLV,
5481 CGF.MakeAddrLValue(FinalResultAddr, PrivateType),
5482 PrivateType, AggValueSlot::DoesNotOverlap, false);
5483 } else {
5484 CGF.EmitStoreOfScalar(FinalResultVal, TargetLHSLV);
5485 }
5486 // Final synchronization barrier
5487 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5488 CGM.getModule(), OMPRTL___kmpc_barrier),
5489 BarrierArgs);
5490
5491 // Combiner with original list item
5492 auto OriginalListCombiner = [&](CodeGenFunction &CGF,
5493 PrePostActionTy &Action) {
5494 Action.Enter(CGF);
5495 emitSingleReductionCombiner(CGF, ReductionOps, Privates,
5496 cast<DeclRefExpr>(LHSExprs),
5497 cast<DeclRefExpr>(RHSExprs));
5498 };
5499 EmitCriticalReduction(OriginalListCombiner);
5500}
5501
5503 ArrayRef<const Expr *> OrgPrivates,
5504 ArrayRef<const Expr *> OrgLHSExprs,
5505 ArrayRef<const Expr *> OrgRHSExprs,
5506 ArrayRef<const Expr *> OrgReductionOps,
5507 ReductionOptionsTy Options) {
5508 if (!CGF.HaveInsertPoint())
5509 return;
5510
5511 bool WithNowait = Options.WithNowait;
5512 bool SimpleReduction = Options.SimpleReduction;
5513
5514 // Next code should be emitted for reduction:
5515 //
5516 // static kmp_critical_name lock = { 0 };
5517 //
5518 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
5519 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
5520 // ...
5521 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
5522 // *(Type<n>-1*)rhs[<n>-1]);
5523 // }
5524 //
5525 // ...
5526 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
5527 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5528 // RedList, reduce_func, &<lock>)) {
5529 // case 1:
5530 // ...
5531 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5532 // ...
5533 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5534 // break;
5535 // case 2:
5536 // ...
5537 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5538 // ...
5539 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
5540 // break;
5541 // default:;
5542 // }
5543 //
5544 // if SimpleReduction is true, only the next code is generated:
5545 // ...
5546 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5547 // ...
5548
5549 ASTContext &C = CGM.getContext();
5550
5551 if (SimpleReduction) {
5553 const auto *IPriv = OrgPrivates.begin();
5554 const auto *ILHS = OrgLHSExprs.begin();
5555 const auto *IRHS = OrgRHSExprs.begin();
5556 for (const Expr *E : OrgReductionOps) {
5557 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5558 cast<DeclRefExpr>(*IRHS));
5559 ++IPriv;
5560 ++ILHS;
5561 ++IRHS;
5562 }
5563 return;
5564 }
5565
5566 // Filter out shared reduction variables based on IsPrivateVarReduction flag.
5567 // Only keep entries where the corresponding variable is not private.
5568 SmallVector<const Expr *> FilteredPrivates, FilteredLHSExprs,
5569 FilteredRHSExprs, FilteredReductionOps;
5570 for (unsigned I : llvm::seq<unsigned>(
5571 std::min(OrgReductionOps.size(), OrgLHSExprs.size()))) {
5572 if (!Options.IsPrivateVarReduction[I]) {
5573 FilteredPrivates.emplace_back(OrgPrivates[I]);
5574 FilteredLHSExprs.emplace_back(OrgLHSExprs[I]);
5575 FilteredRHSExprs.emplace_back(OrgRHSExprs[I]);
5576 FilteredReductionOps.emplace_back(OrgReductionOps[I]);
5577 }
5578 }
5579 // Wrap filtered vectors in ArrayRef for downstream shared reduction
5580 // processing.
5581 ArrayRef<const Expr *> Privates = FilteredPrivates;
5582 ArrayRef<const Expr *> LHSExprs = FilteredLHSExprs;
5583 ArrayRef<const Expr *> RHSExprs = FilteredRHSExprs;
5584 ArrayRef<const Expr *> ReductionOps = FilteredReductionOps;
5585
5586 // 1. Build a list of reduction variables.
5587 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
5588 auto Size = RHSExprs.size();
5589 for (const Expr *E : Privates) {
5590 if (E->getType()->isVariablyModifiedType())
5591 // Reserve place for array size.
5592 ++Size;
5593 }
5594 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size);
5595 QualType ReductionArrayTy = C.getConstantArrayType(
5596 C.VoidPtrTy, ArraySize, nullptr, ArraySizeModifier::Normal,
5597 /*IndexTypeQuals=*/0);
5598 RawAddress ReductionList =
5599 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list");
5600 const auto *IPriv = Privates.begin();
5601 unsigned Idx = 0;
5602 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) {
5603 Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5604 CGF.Builder.CreateStore(
5606 CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy),
5607 Elem);
5608 if ((*IPriv)->getType()->isVariablyModifiedType()) {
5609 // Store array size.
5610 ++Idx;
5611 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5612 llvm::Value *Size = CGF.Builder.CreateIntCast(
5613 CGF.getVLASize(
5614 CGF.getContext().getAsVariableArrayType((*IPriv)->getType()))
5615 .NumElts,
5616 CGF.SizeTy, /*isSigned=*/false);
5617 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy),
5618 Elem);
5619 }
5620 }
5621
5622 // 2. Emit reduce_func().
5623 llvm::Function *ReductionFn = emitReductionFunction(
5624 CGF.CurFn->getName(), Loc, CGF.ConvertTypeForMem(ReductionArrayTy),
5625 Privates, LHSExprs, RHSExprs, ReductionOps);
5626
5627 // 3. Create static kmp_critical_name lock = { 0 };
5628 std::string Name = getName({"reduction"});
5629 llvm::Value *Lock = getCriticalRegionLock(Name);
5630
5631 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5632 // RedList, reduce_func, &<lock>);
5633 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE);
5634 llvm::Value *ThreadId = getThreadID(CGF, Loc);
5635 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy);
5636 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5637 ReductionList.getPointer(), CGF.VoidPtrTy);
5638 llvm::Value *Args[] = {
5639 IdentTLoc, // ident_t *<loc>
5640 ThreadId, // i32 <gtid>
5641 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n>
5642 ReductionArrayTySize, // size_type sizeof(RedList)
5643 RL, // void *RedList
5644 ReductionFn, // void (*) (void *, void *) <reduce_func>
5645 Lock // kmp_critical_name *&<lock>
5646 };
5647 llvm::Value *Res = CGF.EmitRuntimeCall(
5648 OMPBuilder.getOrCreateRuntimeFunction(
5649 CGM.getModule(),
5650 WithNowait ? OMPRTL___kmpc_reduce_nowait : OMPRTL___kmpc_reduce),
5651 Args);
5652
5653 // 5. Build switch(res)
5654 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default");
5655 llvm::SwitchInst *SwInst =
5656 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2);
5657
5658 // 6. Build case 1:
5659 // ...
5660 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5661 // ...
5662 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5663 // break;
5664 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1");
5665 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB);
5666 CGF.EmitBlock(Case1BB);
5667
5668 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5669 llvm::Value *EndArgs[] = {
5670 IdentTLoc, // ident_t *<loc>
5671 ThreadId, // i32 <gtid>
5672 Lock // kmp_critical_name *&<lock>
5673 };
5674 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps](
5675 CodeGenFunction &CGF, PrePostActionTy &Action) {
5677 const auto *IPriv = Privates.begin();
5678 const auto *ILHS = LHSExprs.begin();
5679 const auto *IRHS = RHSExprs.begin();
5680 for (const Expr *E : ReductionOps) {
5681 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5682 cast<DeclRefExpr>(*IRHS));
5683 ++IPriv;
5684 ++ILHS;
5685 ++IRHS;
5686 }
5687 };
5689 CommonActionTy Action(
5690 nullptr, {},
5691 OMPBuilder.getOrCreateRuntimeFunction(
5692 CGM.getModule(), WithNowait ? OMPRTL___kmpc_end_reduce_nowait
5693 : OMPRTL___kmpc_end_reduce),
5694 EndArgs);
5695 RCG.setAction(Action);
5696 RCG(CGF);
5697
5698 CGF.EmitBranch(DefaultBB);
5699
5700 // 7. Build case 2:
5701 // ...
5702 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5703 // ...
5704 // break;
5705 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2");
5706 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB);
5707 CGF.EmitBlock(Case2BB);
5708
5709 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps](
5710 CodeGenFunction &CGF, PrePostActionTy &Action) {
5711 const auto *ILHS = LHSExprs.begin();
5712 const auto *IRHS = RHSExprs.begin();
5713 const auto *IPriv = Privates.begin();
5714 for (const Expr *E : ReductionOps) {
5715 const Expr *XExpr = nullptr;
5716 const Expr *EExpr = nullptr;
5717 const Expr *UpExpr = nullptr;
5718 BinaryOperatorKind BO = BO_Comma;
5719 if (const auto *BO = dyn_cast<BinaryOperator>(E)) {
5720 if (BO->getOpcode() == BO_Assign) {
5721 XExpr = BO->getLHS();
5722 UpExpr = BO->getRHS();
5723 }
5724 }
5725 // Try to emit update expression as a simple atomic.
5726 const Expr *RHSExpr = UpExpr;
5727 if (RHSExpr) {
5728 // Analyze RHS part of the whole expression.
5729 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>(
5730 RHSExpr->IgnoreParenImpCasts())) {
5731 // If this is a conditional operator, analyze its condition for
5732 // min/max reduction operator.
5733 RHSExpr = ACO->getCond();
5734 }
5735 if (const auto *BORHS =
5736 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) {
5737 EExpr = BORHS->getRHS();
5738 BO = BORHS->getOpcode();
5739 }
5740 }
5741 if (XExpr) {
5742 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5743 auto &&AtomicRedGen = [BO, VD,
5744 Loc](CodeGenFunction &CGF, const Expr *XExpr,
5745 const Expr *EExpr, const Expr *UpExpr) {
5746 LValue X = CGF.EmitLValue(XExpr);
5747 RValue E;
5748 if (EExpr)
5749 E = CGF.EmitAnyExpr(EExpr);
5750 CGF.EmitOMPAtomicSimpleUpdateExpr(
5751 X, E, BO, /*IsXLHSInRHSPart=*/true,
5752 llvm::AtomicOrdering::Monotonic, Loc,
5753 [&CGF, UpExpr, VD, Loc](RValue XRValue) {
5754 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
5755 Address LHSTemp = CGF.CreateMemTemp(VD->getType());
5756 CGF.emitOMPSimpleStore(
5757 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue,
5758 VD->getType().getNonReferenceType(), Loc);
5759 PrivateScope.addPrivate(VD, LHSTemp);
5760 (void)PrivateScope.Privatize();
5761 return CGF.EmitAnyExpr(UpExpr);
5762 });
5763 };
5764 if ((*IPriv)->getType()->isArrayType()) {
5765 // Emit atomic reduction for array section.
5766 const auto *RHSVar =
5767 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5768 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar,
5769 AtomicRedGen, XExpr, EExpr, UpExpr);
5770 } else {
5771 // Emit atomic reduction for array subscript or single variable.
5772 AtomicRedGen(CGF, XExpr, EExpr, UpExpr);
5773 }
5774 } else {
5775 // Emit as a critical region.
5776 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *,
5777 const Expr *, const Expr *) {
5778 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5779 std::string Name = RT.getName({"atomic_reduction"});
5781 CGF, Name,
5782 [=](CodeGenFunction &CGF, PrePostActionTy &Action) {
5783 Action.Enter(CGF);
5784 emitReductionCombiner(CGF, E);
5785 },
5786 Loc);
5787 };
5788 if ((*IPriv)->getType()->isArrayType()) {
5789 const auto *LHSVar =
5790 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5791 const auto *RHSVar =
5792 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5793 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5794 CritRedGen);
5795 } else {
5796 CritRedGen(CGF, nullptr, nullptr, nullptr);
5797 }
5798 }
5799 ++ILHS;
5800 ++IRHS;
5801 ++IPriv;
5802 }
5803 };
5804 RegionCodeGenTy AtomicRCG(AtomicCodeGen);
5805 if (!WithNowait) {
5806 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
5807 llvm::Value *EndArgs[] = {
5808 IdentTLoc, // ident_t *<loc>
5809 ThreadId, // i32 <gtid>
5810 Lock // kmp_critical_name *&<lock>
5811 };
5812 CommonActionTy Action(nullptr, {},
5813 OMPBuilder.getOrCreateRuntimeFunction(
5814 CGM.getModule(), OMPRTL___kmpc_end_reduce),
5815 EndArgs);
5816 AtomicRCG.setAction(Action);
5817 AtomicRCG(CGF);
5818 } else {
5819 AtomicRCG(CGF);
5820 }
5821
5822 CGF.EmitBranch(DefaultBB);
5823 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true);
5824 assert(OrgLHSExprs.size() == OrgPrivates.size() &&
5825 "PrivateVarReduction: Privates size mismatch");
5826 assert(OrgLHSExprs.size() == OrgReductionOps.size() &&
5827 "PrivateVarReduction: ReductionOps size mismatch");
5828 for (unsigned I : llvm::seq<unsigned>(
5829 std::min(OrgReductionOps.size(), OrgLHSExprs.size()))) {
5830 if (Options.IsPrivateVarReduction[I])
5831 emitPrivateReduction(CGF, Loc, OrgPrivates[I], OrgLHSExprs[I],
5832 OrgRHSExprs[I], OrgReductionOps[I]);
5833 }
5834}
5835
5836/// Generates unique name for artificial threadprivate variables.
5837/// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>"
5838static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix,
5839 const Expr *Ref) {
5840 SmallString<256> Buffer;
5841 llvm::raw_svector_ostream Out(Buffer);
5842 const clang::DeclRefExpr *DE;
5843 const VarDecl *D = ::getBaseDecl(Ref, DE);
5844 if (!D) {
5845 auto *DRE = cast<DeclRefExpr>(Ref);
5846 if (const auto *BD = dyn_cast<BindingDecl>(DRE->getDecl())) {
5847 // For BindingDecls, use the decomposed declaration as the base.
5848 D = cast<VarDecl>(BD->getDecomposedDecl());
5849 } else {
5850 D = cast<VarDecl>(DRE->getDecl());
5851 }
5852 }
5853 D = D->getCanonicalDecl();
5854 std::string Name = CGM.getOpenMPRuntime().getName(
5855 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)});
5856 Out << Prefix << Name << "_"
5858 return std::string(Out.str());
5859}
5860
5861/// Emits reduction initializer function:
5862/// \code
5863/// void @.red_init(void* %arg, void* %orig) {
5864/// %0 = bitcast void* %arg to <type>*
5865/// store <type> <init>, <type>* %0
5866/// ret void
5867/// }
5868/// \endcode
5869static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM,
5870 SourceLocation Loc,
5871 ReductionCodeGen &RCG, unsigned N) {
5872 ASTContext &C = CGM.getContext();
5873 QualType VoidPtrTy = C.VoidPtrTy;
5874 VoidPtrTy.addRestrict();
5875 FunctionArgList Args;
5876 auto *Param =
5877 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5878 VoidPtrTy, ImplicitParamKind::Other);
5879 auto *ParamOrig =
5880 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5881 VoidPtrTy, ImplicitParamKind::Other);
5882 Args.emplace_back(Param);
5883 Args.emplace_back(ParamOrig);
5884 const auto &FnInfo =
5885 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5886 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
5887 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""});
5888 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
5889 Name, &CGM.getModule());
5890 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
5891 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
5892 Fn->addFnAttr("sample-profile-suffix-elision-policy", "selected");
5893 Fn->setDoesNotRecurse();
5894 CodeGenFunction CGF(CGM);
5895 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
5896 QualType PrivateType = RCG.getPrivateType(N);
5897 Address PrivateAddr = CGF.EmitLoadOfPointer(
5898 CGF.GetAddrOfLocalVar(Param).withElementType(CGF.Builder.getPtrTy(0)),
5899 C.getPointerType(PrivateType)->castAs<PointerType>());
5900 llvm::Value *Size = nullptr;
5901 // If the size of the reduction item is non-constant, load it from global
5902 // threadprivate variable.
5903 if (RCG.getSizes(N).second) {
5905 CGF, CGM.getContext().getSizeType(),
5906 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
5907 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
5908 CGM.getContext().getSizeType(), Loc);
5909 }
5910 RCG.emitAggregateType(CGF, N, Size);
5911 Address OrigAddr = Address::invalid();
5912 // If initializer uses initializer from declare reduction construct, emit a
5913 // pointer to the address of the original reduction item (reuired by reduction
5914 // initializer)
5915 if (RCG.usesReductionInitializer(N)) {
5916 Address SharedAddr = CGF.GetAddrOfLocalVar(ParamOrig);
5917 OrigAddr = CGF.EmitLoadOfPointer(
5918 SharedAddr,
5919 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr());
5920 }
5921 // Emit the initializer:
5922 // %0 = bitcast void* %arg to <type>*
5923 // store <type> <init>, <type>* %0
5924 RCG.emitInitialization(CGF, N, PrivateAddr, OrigAddr,
5925 [](CodeGenFunction &) { return false; });
5926 CGF.FinishFunction();
5927 return Fn;
5928}
5929
5930/// Emits reduction combiner function:
5931/// \code
5932/// void @.red_comb(void* %arg0, void* %arg1) {
5933/// %lhs = bitcast void* %arg0 to <type>*
5934/// %rhs = bitcast void* %arg1 to <type>*
5935/// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs)
5936/// store <type> %2, <type>* %lhs
5937/// ret void
5938/// }
5939/// \endcode
5940static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM,
5941 SourceLocation Loc,
5942 ReductionCodeGen &RCG, unsigned N,
5943 const Expr *ReductionOp,
5944 const Expr *LHS, const Expr *RHS,
5945 const Expr *PrivateRef) {
5946 ASTContext &C = CGM.getContext();
5947 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl());
5948 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl());
5949 FunctionArgList Args;
5950 auto *ParamInOut =
5951 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5952 C.VoidPtrTy, ImplicitParamKind::Other);
5953 auto *ParamIn =
5954 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5955 C.VoidPtrTy, ImplicitParamKind::Other);
5956 Args.emplace_back(ParamInOut);
5957 Args.emplace_back(ParamIn);
5958 const auto &FnInfo =
5959 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5960 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
5961 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""});
5962 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
5963 Name, &CGM.getModule());
5964 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
5965 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
5966 Fn->addFnAttr("sample-profile-suffix-elision-policy", "selected");
5967 Fn->setDoesNotRecurse();
5968 CodeGenFunction CGF(CGM);
5969 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
5970 llvm::Value *Size = nullptr;
5971 // If the size of the reduction item is non-constant, load it from global
5972 // threadprivate variable.
5973 if (RCG.getSizes(N).second) {
5975 CGF, CGM.getContext().getSizeType(),
5976 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
5977 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
5978 CGM.getContext().getSizeType(), Loc);
5979 }
5980 RCG.emitAggregateType(CGF, N, Size);
5981 // Remap lhs and rhs variables to the addresses of the function arguments.
5982 // %lhs = bitcast void* %arg0 to <type>*
5983 // %rhs = bitcast void* %arg1 to <type>*
5984 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
5985 PrivateScope.addPrivate(
5986 LHSVD,
5987 // Pull out the pointer to the variable.
5989 CGF.GetAddrOfLocalVar(ParamInOut)
5990 .withElementType(CGF.Builder.getPtrTy(0)),
5991 C.getPointerType(LHSVD->getType())->castAs<PointerType>()));
5992 PrivateScope.addPrivate(
5993 RHSVD,
5994 // Pull out the pointer to the variable.
5997 CGF.Builder.getPtrTy(0)),
5998 C.getPointerType(RHSVD->getType())->castAs<PointerType>()));
5999 PrivateScope.Privatize();
6000 // Emit the combiner body:
6001 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs)
6002 // store <type> %2, <type>* %lhs
6004 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS),
6005 cast<DeclRefExpr>(RHS));
6006 CGF.FinishFunction();
6007 return Fn;
6008}
6009
6010/// Emits reduction finalizer function:
6011/// \code
6012/// void @.red_fini(void* %arg) {
6013/// %0 = bitcast void* %arg to <type>*
6014/// <destroy>(<type>* %0)
6015/// ret void
6016/// }
6017/// \endcode
6018static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM,
6019 SourceLocation Loc,
6020 ReductionCodeGen &RCG, unsigned N) {
6021 if (!RCG.needCleanups(N))
6022 return nullptr;
6023 ASTContext &C = CGM.getContext();
6024 FunctionArgList Args;
6025 auto *Param =
6026 ImplicitParamDecl::Create(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
6027 C.VoidPtrTy, ImplicitParamKind::Other);
6028 Args.emplace_back(Param);
6029 const auto &FnInfo =
6030 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6031 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6032 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""});
6033 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6034 Name, &CGM.getModule());
6035 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6036 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
6037 Fn->addFnAttr("sample-profile-suffix-elision-policy", "selected");
6038 Fn->setDoesNotRecurse();
6039 CodeGenFunction CGF(CGM);
6040 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6041 Address PrivateAddr = CGF.EmitLoadOfPointer(
6042 CGF.GetAddrOfLocalVar(Param), C.VoidPtrTy.castAs<PointerType>());
6043 llvm::Value *Size = nullptr;
6044 // If the size of the reduction item is non-constant, load it from global
6045 // threadprivate variable.
6046 if (RCG.getSizes(N).second) {
6048 CGF, CGM.getContext().getSizeType(),
6049 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6050 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6051 CGM.getContext().getSizeType(), Loc);
6052 }
6053 RCG.emitAggregateType(CGF, N, Size);
6054 // Emit the finalizer body:
6055 // <destroy>(<type>* %0)
6056 RCG.emitCleanups(CGF, N, PrivateAddr);
6057 CGF.FinishFunction(Loc);
6058 return Fn;
6059}
6060
6063 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
6064 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty())
6065 return nullptr;
6066
6067 // Build typedef struct:
6068 // kmp_taskred_input {
6069 // void *reduce_shar; // shared reduction item
6070 // void *reduce_orig; // original reduction item used for initialization
6071 // size_t reduce_size; // size of data item
6072 // void *reduce_init; // data initialization routine
6073 // void *reduce_fini; // data finalization routine
6074 // void *reduce_comb; // data combiner routine
6075 // kmp_task_red_flags_t flags; // flags for additional info from compiler
6076 // } kmp_taskred_input_t;
6077 ASTContext &C = CGM.getContext();
6078 RecordDecl *RD = C.buildImplicitRecord("kmp_taskred_input_t");
6079 RD->startDefinition();
6080 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6081 const FieldDecl *OrigFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6082 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType());
6083 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6084 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6085 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6086 const FieldDecl *FlagsFD = addFieldToRecordDecl(
6087 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false));
6088 RD->completeDefinition();
6089 CanQualType RDType = C.getCanonicalTagType(RD);
6090 unsigned Size = Data.ReductionVars.size();
6091 llvm::APInt ArraySize(/*numBits=*/64, Size);
6092 QualType ArrayRDType =
6093 C.getConstantArrayType(RDType, ArraySize, nullptr,
6094 ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
6095 // kmp_task_red_input_t .rd_input.[Size];
6096 RawAddress TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input.");
6097 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionOrigs,
6098 Data.ReductionCopies, Data.ReductionOps);
6099 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) {
6100 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt];
6101 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0),
6102 llvm::ConstantInt::get(CGM.SizeTy, Cnt)};
6103 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP(
6104 TaskRedInput.getElementType(), TaskRedInput.getPointer(), Idxs,
6105 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc,
6106 ".rd_input.gep.");
6107 LValue ElemLVal = CGF.MakeNaturalAlignRawAddrLValue(GEP, RDType);
6108 // ElemLVal.reduce_shar = &Shareds[Cnt];
6109 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD);
6110 RCG.emitSharedOrigLValue(CGF, Cnt);
6111 llvm::Value *Shared = RCG.getSharedLValue(Cnt).getPointer(CGF);
6112 CGF.EmitStoreOfScalar(Shared, SharedLVal);
6113 // ElemLVal.reduce_orig = &Origs[Cnt];
6114 LValue OrigLVal = CGF.EmitLValueForField(ElemLVal, OrigFD);
6115 llvm::Value *Orig = RCG.getOrigLValue(Cnt).getPointer(CGF);
6116 CGF.EmitStoreOfScalar(Orig, OrigLVal);
6117 RCG.emitAggregateType(CGF, Cnt);
6118 llvm::Value *SizeValInChars;
6119 llvm::Value *SizeVal;
6120 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt);
6121 // We use delayed creation/initialization for VLAs and array sections. It is
6122 // required because runtime does not provide the way to pass the sizes of
6123 // VLAs/array sections to initializer/combiner/finalizer functions. Instead
6124 // threadprivate global variables are used to store these values and use
6125 // them in the functions.
6126 bool DelayedCreation = !!SizeVal;
6127 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy,
6128 /*isSigned=*/false);
6129 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD);
6130 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal);
6131 // ElemLVal.reduce_init = init;
6132 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD);
6133 llvm::Value *InitAddr = emitReduceInitFunction(CGM, Loc, RCG, Cnt);
6134 CGF.EmitStoreOfScalar(InitAddr, InitLVal);
6135 // ElemLVal.reduce_fini = fini;
6136 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD);
6137 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt);
6138 llvm::Value *FiniAddr =
6139 Fini ? Fini : llvm::ConstantPointerNull::get(CGM.VoidPtrTy);
6140 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal);
6141 // ElemLVal.reduce_comb = comb;
6142 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD);
6143 llvm::Value *CombAddr = emitReduceCombFunction(
6144 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt],
6145 RHSExprs[Cnt], Data.ReductionCopies[Cnt]);
6146 CGF.EmitStoreOfScalar(CombAddr, CombLVal);
6147 // ElemLVal.flags = 0;
6148 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD);
6149 if (DelayedCreation) {
6151 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true),
6152 FlagsLVal);
6153 } else
6154 CGF.EmitNullInitialization(FlagsLVal.getAddress(), FlagsLVal.getType());
6155 }
6156 if (Data.IsReductionWithTaskMod) {
6157 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int
6158 // is_ws, int num, void *data);
6159 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc);
6160 llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6161 CGM.IntTy, /*isSigned=*/true);
6162 llvm::Value *Args[] = {
6163 IdentTLoc, GTid,
6164 llvm::ConstantInt::get(CGM.IntTy, Data.IsWorksharingReduction ? 1 : 0,
6165 /*isSigned=*/true),
6166 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6168 TaskRedInput.getPointer(), CGM.VoidPtrTy)};
6169 return CGF.EmitRuntimeCall(
6170 OMPBuilder.getOrCreateRuntimeFunction(
6171 CGM.getModule(), OMPRTL___kmpc_taskred_modifier_init),
6172 Args);
6173 }
6174 // Build call void *__kmpc_taskred_init(int gtid, int num_data, void *data);
6175 llvm::Value *Args[] = {
6176 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy,
6177 /*isSigned=*/true),
6178 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6180 CGM.VoidPtrTy)};
6181 return CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
6182 CGM.getModule(), OMPRTL___kmpc_taskred_init),
6183 Args);
6184}
6185
6187 SourceLocation Loc,
6188 bool IsWorksharingReduction) {
6189 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int
6190 // is_ws, int num, void *data);
6191 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc);
6192 llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6193 CGM.IntTy, /*isSigned=*/true);
6194 llvm::Value *Args[] = {IdentTLoc, GTid,
6195 llvm::ConstantInt::get(CGM.IntTy,
6196 IsWorksharingReduction ? 1 : 0,
6197 /*isSigned=*/true)};
6198 (void)CGF.EmitRuntimeCall(
6199 OMPBuilder.getOrCreateRuntimeFunction(
6200 CGM.getModule(), OMPRTL___kmpc_task_reduction_modifier_fini),
6201 Args);
6202}
6203
6205 SourceLocation Loc,
6206 ReductionCodeGen &RCG,
6207 unsigned N) {
6208 auto Sizes = RCG.getSizes(N);
6209 // Emit threadprivate global variable if the type is non-constant
6210 // (Sizes.second = nullptr).
6211 if (Sizes.second) {
6212 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy,
6213 /*isSigned=*/false);
6215 CGF, CGM.getContext().getSizeType(),
6216 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6217 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false);
6218 }
6219}
6220
6222 SourceLocation Loc,
6223 llvm::Value *ReductionsPtr,
6224 LValue SharedLVal) {
6225 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
6226 // *d);
6227 llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6228 CGM.IntTy,
6229 /*isSigned=*/true),
6230 ReductionsPtr,
6232 SharedLVal.getPointer(CGF), CGM.VoidPtrTy)};
6233 return Address(
6234 CGF.EmitRuntimeCall(
6235 OMPBuilder.getOrCreateRuntimeFunction(
6236 CGM.getModule(), OMPRTL___kmpc_task_reduction_get_th_data),
6237 Args),
6238 CGF.Int8Ty, SharedLVal.getAlignment());
6239}
6240
6242 const OMPTaskDataTy &Data) {
6243 if (!CGF.HaveInsertPoint())
6244 return;
6245
6246 if (CGF.CGM.getLangOpts().OpenMPIRBuilder && Data.Dependences.empty()) {
6247 // TODO: Need to support taskwait with dependences in the OpenMPIRBuilder.
6248 OMPBuilder.createTaskwait(CGF.Builder);
6249 } else {
6250 llvm::Value *ThreadID = getThreadID(CGF, Loc);
6251 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
6252 auto &M = CGM.getModule();
6253 Address DependenciesArray = Address::invalid();
6254 llvm::Value *NumOfElements;
6255 std::tie(NumOfElements, DependenciesArray) =
6256 emitDependClause(CGF, Data.Dependences, Loc);
6257 if (!Data.Dependences.empty()) {
6258 llvm::Value *DepWaitTaskArgs[7];
6259 DepWaitTaskArgs[0] = UpLoc;
6260 DepWaitTaskArgs[1] = ThreadID;
6261 DepWaitTaskArgs[2] = NumOfElements;
6262 DepWaitTaskArgs[3] = DependenciesArray.emitRawPointer(CGF);
6263 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
6264 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
6265 DepWaitTaskArgs[6] =
6266 llvm::ConstantInt::get(CGF.Int32Ty, Data.HasNowaitClause);
6267
6268 CodeGenFunction::RunCleanupsScope LocalScope(CGF);
6269
6270 // Build void __kmpc_omp_taskwait_deps_51(ident_t *, kmp_int32 gtid,
6271 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
6272 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list,
6273 // kmp_int32 has_no_wait); if dependence info is specified.
6274 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
6275 M, OMPRTL___kmpc_omp_taskwait_deps_51),
6276 DepWaitTaskArgs);
6277
6278 } else {
6279
6280 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
6281 // global_tid);
6282 llvm::Value *Args[] = {UpLoc, ThreadID};
6283 // Ignore return result until untied tasks are supported.
6284 CGF.EmitRuntimeCall(
6285 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_taskwait),
6286 Args);
6287 }
6288 }
6289
6290 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
6291 Region->emitUntiedSwitch(CGF);
6292}
6293
6295 OpenMPDirectiveKind InnerKind,
6296 const RegionCodeGenTy &CodeGen,
6297 bool HasCancel) {
6298 if (!CGF.HaveInsertPoint())
6299 return;
6300 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel,
6301 InnerKind != OMPD_critical &&
6302 InnerKind != OMPD_master &&
6303 InnerKind != OMPD_masked);
6304 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
6305}
6306
6307namespace {
6308enum RTCancelKind {
6309 CancelNoreq = 0,
6310 CancelParallel = 1,
6311 CancelLoop = 2,
6312 CancelSections = 3,
6313 CancelTaskgroup = 4
6314};
6315} // anonymous namespace
6316
6317static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) {
6318 RTCancelKind CancelKind = CancelNoreq;
6319 if (CancelRegion == OMPD_parallel)
6320 CancelKind = CancelParallel;
6321 else if (CancelRegion == OMPD_for)
6322 CancelKind = CancelLoop;
6323 else if (CancelRegion == OMPD_sections)
6324 CancelKind = CancelSections;
6325 else {
6326 assert(CancelRegion == OMPD_taskgroup);
6327 CancelKind = CancelTaskgroup;
6328 }
6329 return CancelKind;
6330}
6331
6334 OpenMPDirectiveKind CancelRegion) {
6335 if (!CGF.HaveInsertPoint())
6336 return;
6337 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
6338 // global_tid, kmp_int32 cncl_kind);
6339 if (auto *OMPRegionInfo =
6340 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6341 // For 'cancellation point taskgroup', the task region info may not have a
6342 // cancel. This may instead happen in another adjacent task.
6343 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) {
6344 llvm::Value *Args[] = {
6345 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
6346 CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6347 // Ignore return result until untied tasks are supported.
6348 llvm::Value *Result = CGF.EmitRuntimeCall(
6349 OMPBuilder.getOrCreateRuntimeFunction(
6350 CGM.getModule(), OMPRTL___kmpc_cancellationpoint),
6351 Args);
6352 // if (__kmpc_cancellationpoint()) {
6353 // call i32 @__kmpc_cancel_barrier( // for parallel cancellation only
6354 // exit from construct;
6355 // }
6356 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6357 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6358 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6359 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6360 CGF.EmitBlock(ExitBB);
6361 if (CancelRegion == OMPD_parallel)
6362 emitBarrierCall(CGF, Loc, OMPD_unknown, /*EmitChecks=*/false);
6363 // exit from construct;
6364 CodeGenFunction::JumpDest CancelDest =
6365 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6366 CGF.EmitBranchThroughCleanup(CancelDest);
6367 CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6368 }
6369 }
6370}
6371
6373 const Expr *IfCond,
6374 OpenMPDirectiveKind CancelRegion) {
6375 if (!CGF.HaveInsertPoint())
6376 return;
6377 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
6378 // kmp_int32 cncl_kind);
6379 auto &M = CGM.getModule();
6380 if (auto *OMPRegionInfo =
6381 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6382 auto &&ThenGen = [this, &M, Loc, CancelRegion,
6383 OMPRegionInfo](CodeGenFunction &CGF, PrePostActionTy &) {
6384 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6385 llvm::Value *Args[] = {
6386 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc),
6387 CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6388 // Ignore return result until untied tasks are supported.
6389 llvm::Value *Result = CGF.EmitRuntimeCall(
6390 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_cancel), Args);
6391 // if (__kmpc_cancel()) {
6392 // call i32 @__kmpc_cancel_barrier( // for parallel cancellation only
6393 // exit from construct;
6394 // }
6395 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6396 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6397 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6398 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6399 CGF.EmitBlock(ExitBB);
6400 if (CancelRegion == OMPD_parallel)
6401 RT.emitBarrierCall(CGF, Loc, OMPD_unknown, /*EmitChecks=*/false);
6402 // exit from construct;
6403 CodeGenFunction::JumpDest CancelDest =
6404 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6405 CGF.EmitBranchThroughCleanup(CancelDest);
6406 CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6407 };
6408 if (IfCond) {
6409 emitIfClause(CGF, IfCond, ThenGen,
6410 [](CodeGenFunction &, PrePostActionTy &) {});
6411 } else {
6412 RegionCodeGenTy ThenRCG(ThenGen);
6413 ThenRCG(CGF);
6414 }
6415 }
6416}
6417
6418namespace {
6419/// Cleanup action for uses_allocators support.
6420class OMPUsesAllocatorsActionTy final : public PrePostActionTy {
6422 const OMPExecutableDirective &D;
6423 bool IsOffloadEntry;
6424
6425public:
6426 OMPUsesAllocatorsActionTy(
6427 ArrayRef<std::pair<const Expr *, const Expr *>> Allocators,
6428 const OMPExecutableDirective &D, bool IsOffloadEntry)
6429 : Allocators(Allocators), D(D), IsOffloadEntry(IsOffloadEntry) {}
6430 void Enter(CodeGenFunction &CGF) override {
6431 if (!CGF.HaveInsertPoint())
6432 return;
6433 for (const auto &AllocatorData : Allocators) {
6435 CGF, AllocatorData.first, AllocatorData.second);
6436 }
6437 // This kernel does not go through the device-side runtime
6438 // init/deinit sequence (that is GPU-only), but the runtime still
6439 // needs a '<kernel>_kernel_environment' global to know how the
6440 // kernel was configured, so emit it directly here.
6441 if (IsOffloadEntry)
6443 }
6444 void Exit(CodeGenFunction &CGF) override {
6445 if (!CGF.HaveInsertPoint())
6446 return;
6447 for (const auto &AllocatorData : Allocators) {
6449 AllocatorData.first);
6450 }
6451 }
6452};
6453} // namespace
6454
6456 CodeGenFunction &CGF) {
6457 llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs Attrs;
6458 Attrs.ExecFlags = llvm::omp::OMPTgtExecModeFlags::OMP_TGT_EXEC_MODE_GENERIC;
6459 computeMinAndMaxThreadsAndTeams(D, CGF, Attrs);
6460 OMPBuilder.emitKernelEnvironment(CGF.Builder, Attrs);
6461}
6462
6464 const OMPExecutableDirective &D, StringRef ParentName,
6465 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6466 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6467 assert(!ParentName.empty() && "Invalid target entry parent name!");
6470 for (const auto *C : D.getClausesOfKind<OMPUsesAllocatorsClause>()) {
6471 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) {
6472 const OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I);
6473 if (!D.AllocatorTraits)
6474 continue;
6475 Allocators.emplace_back(D.Allocator, D.AllocatorTraits);
6476 }
6477 }
6478 OMPUsesAllocatorsActionTy UsesAllocatorAction(Allocators, D, IsOffloadEntry);
6479 CodeGen.setAction(UsesAllocatorAction);
6480 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID,
6481 IsOffloadEntry, CodeGen);
6482}
6483
6485 const Expr *Allocator,
6486 const Expr *AllocatorTraits) {
6487 llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc());
6488 ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true);
6489 // Use default memspace handle.
6490 llvm::Value *MemSpaceHandle = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
6491 llvm::Value *NumTraits = llvm::ConstantInt::get(
6493 AllocatorTraits->getType()->getAsArrayTypeUnsafe())
6494 ->getSize()
6495 .getLimitedValue());
6496 LValue AllocatorTraitsLVal = CGF.EmitLValue(AllocatorTraits);
6498 AllocatorTraitsLVal.getAddress(), CGF.VoidPtrPtrTy, CGF.VoidPtrTy);
6499 AllocatorTraitsLVal = CGF.MakeAddrLValue(Addr, CGF.getContext().VoidPtrTy,
6500 AllocatorTraitsLVal.getBaseInfo(),
6501 AllocatorTraitsLVal.getTBAAInfo());
6502 llvm::Value *Traits = Addr.emitRawPointer(CGF);
6503
6504 llvm::Value *AllocatorVal =
6505 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
6506 CGM.getModule(), OMPRTL___kmpc_init_allocator),
6507 {ThreadId, MemSpaceHandle, NumTraits, Traits});
6508 // Store to allocator.
6510 cast<DeclRefExpr>(Allocator->IgnoreParenImpCasts())->getDecl()));
6511 LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts());
6512 AllocatorVal =
6513 CGF.EmitScalarConversion(AllocatorVal, CGF.getContext().VoidPtrTy,
6514 Allocator->getType(), Allocator->getExprLoc());
6515 CGF.EmitStoreOfScalar(AllocatorVal, AllocatorLVal);
6516}
6517
6519 const Expr *Allocator) {
6520 llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc());
6521 ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true);
6522 LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts());
6523 llvm::Value *AllocatorVal =
6524 CGF.EmitLoadOfScalar(AllocatorLVal, Allocator->getExprLoc());
6525 AllocatorVal = CGF.EmitScalarConversion(AllocatorVal, Allocator->getType(),
6526 CGF.getContext().VoidPtrTy,
6527 Allocator->getExprLoc());
6528 (void)CGF.EmitRuntimeCall(
6529 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
6530 OMPRTL___kmpc_destroy_allocator),
6531 {ThreadId, AllocatorVal});
6532}
6533
6536 llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs &Attrs) {
6537 assert(Attrs.MaxTeams.size() == 1 && Attrs.MaxThreads.size() == 1 &&
6538 "invalid default attrs structure");
6539 int32_t &MaxTeamsVal = Attrs.MaxTeams.front();
6540 int32_t &MaxThreadsVal = Attrs.MaxThreads.front();
6541
6542 getNumTeamsExprForTargetDirective(CGF, D, Attrs.MinTeams.front(),
6543 MaxTeamsVal);
6544 getNumThreadsExprForTargetDirective(CGF, D, MaxThreadsVal,
6545 /*UpperBoundOnly=*/true);
6546
6547 for (auto *C : D.getClausesOfKind<OMPXAttributeClause>()) {
6548 for (auto *A : C->getAttrs()) {
6549 int32_t AttrMinThreadsVal = 1, AttrMaxThreadsVal = -1;
6550 int32_t AttrMinBlocksVal = 1, AttrMaxBlocksVal = -1;
6551 if (auto *Attr = dyn_cast<CUDALaunchBoundsAttr>(A))
6552 CGM.handleCUDALaunchBoundsAttr(nullptr, Attr, &AttrMaxThreadsVal,
6553 &AttrMinBlocksVal, &AttrMaxBlocksVal);
6554 else if (auto *Attr = dyn_cast<AMDGPUFlatWorkGroupSizeAttr>(A))
6555 CGM.handleAMDGPUFlatWorkGroupSizeAttr(
6556 nullptr, Attr, /*ReqdWGS=*/nullptr, &AttrMinThreadsVal,
6557 &AttrMaxThreadsVal);
6558 else
6559 continue;
6560
6561 Attrs.MinThreads.front() =
6562 std::max(Attrs.MinThreads.front(), AttrMinThreadsVal);
6563 if (AttrMaxThreadsVal > 0)
6564 MaxThreadsVal = MaxThreadsVal > 0
6565 ? std::min(MaxThreadsVal, AttrMaxThreadsVal)
6566 : AttrMaxThreadsVal;
6567 Attrs.MinTeams.front() =
6568 std::max(Attrs.MinTeams.front(), AttrMinBlocksVal);
6569 if (AttrMaxBlocksVal > 0)
6570 MaxTeamsVal = MaxTeamsVal > 0 ? std::min(MaxTeamsVal, AttrMaxBlocksVal)
6571 : AttrMaxBlocksVal;
6572 }
6573 }
6574}
6575
6577 const OMPExecutableDirective &D, StringRef ParentName,
6578 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6579 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6580
6581 llvm::TargetRegionEntryInfo EntryInfo =
6582 getEntryInfoFromPresumedLoc(CGM, OMPBuilder, D.getBeginLoc(), ParentName);
6583
6584 CodeGenFunction CGF(CGM, true);
6585 llvm::OpenMPIRBuilder::FunctionGenCallback &&GenerateOutlinedFunction =
6586 [&CGF, &D, &CodeGen, this](StringRef EntryFnName) {
6587 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
6588
6589 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName);
6590 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6591 if (CGM.getLangOpts().OpenMPIsTargetDevice && !isGPU())
6593 return CGF.GenerateOpenMPCapturedStmtFunction(CS, D);
6594 };
6595
6596 cantFail(OMPBuilder.emitTargetRegionFunction(
6597 EntryInfo, GenerateOutlinedFunction, IsOffloadEntry, OutlinedFn,
6598 OutlinedFnID));
6599
6600 if (!OutlinedFn)
6601 return;
6602
6603 // A target body is entered once, from the kernel, and never re-entered by
6604 // the runtime, so it cannot occur in a cycle.
6605 OutlinedFn->setDoesNotRecurse();
6606
6607 CGM.getTargetCodeGenInfo().setTargetAttributes(nullptr, OutlinedFn, CGM);
6608
6609 for (auto *C : D.getClausesOfKind<OMPXAttributeClause>()) {
6610 for (auto *A : C->getAttrs()) {
6611 if (auto *Attr = dyn_cast<AMDGPUWavesPerEUAttr>(A))
6612 CGM.handleAMDGPUWavesPerEUAttr(OutlinedFn, Attr);
6613 }
6614 }
6615 registerVTable(D);
6616}
6617
6618/// Checks if the expression is constant or does not have non-trivial function
6619/// calls.
6620static bool isTrivial(ASTContext &Ctx, const Expr * E) {
6621 // We can skip constant expressions.
6622 // We can skip expressions with trivial calls or simple expressions.
6624 !E->hasNonTrivialCall(Ctx)) &&
6625 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true);
6626}
6627
6629 const Stmt *Body) {
6630 const Stmt *Child = Body->IgnoreContainers();
6631 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) {
6632 Child = nullptr;
6633 for (const Stmt *S : C->body()) {
6634 if (const auto *E = dyn_cast<Expr>(S)) {
6635 if (isTrivial(Ctx, E))
6636 continue;
6637 }
6638 // Some of the statements can be ignored.
6641 continue;
6642 // Analyze declarations.
6643 if (const auto *DS = dyn_cast<DeclStmt>(S)) {
6644 if (llvm::all_of(DS->decls(), [](const Decl *D) {
6645 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) ||
6646 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) ||
6647 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) ||
6648 isa<UsingDirectiveDecl>(D) ||
6649 isa<OMPDeclareReductionDecl>(D) ||
6650 isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D))
6651 return true;
6652 const auto *VD = dyn_cast<VarDecl>(D);
6653 if (!VD)
6654 return false;
6655 return VD->hasGlobalStorage() || !VD->isUsed();
6656 }))
6657 continue;
6658 }
6659 // Found multiple children - cannot get the one child only.
6660 if (Child)
6661 return nullptr;
6662 Child = S;
6663 }
6664 if (Child)
6665 Child = Child->IgnoreContainers();
6666 }
6667 return Child;
6668}
6669
6671 CodeGenFunction &CGF, const OMPExecutableDirective &D, int32_t &MinTeamsVal,
6672 int32_t &MaxTeamsVal) {
6673
6674 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6675 assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6676 "Expected target-based executable directive.");
6677 switch (DirectiveKind) {
6678 case OMPD_target: {
6679 const auto *CS = D.getInnermostCapturedStmt();
6680 const auto *Body =
6681 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
6682 const Stmt *ChildStmt =
6684 if (const auto *NestedDir =
6685 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
6686 if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) {
6687 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) {
6688 const Expr *NumTeams = NestedDir->getSingleClause<OMPNumTeamsClause>()
6689 ->getNumTeams()
6690 .front();
6691 if (NumTeams->isIntegerConstantExpr(CGF.getContext()))
6692 if (auto Constant =
6693 NumTeams->getIntegerConstantExpr(CGF.getContext()))
6694 MinTeamsVal = MaxTeamsVal = Constant->getExtValue();
6695 return NumTeams;
6696 }
6697 MinTeamsVal = MaxTeamsVal = 0;
6698 return nullptr;
6699 }
6700 MinTeamsVal = MaxTeamsVal = 1;
6701 return nullptr;
6702 }
6703 // A value of -1 is used to check if we need to emit no teams region
6704 MinTeamsVal = MaxTeamsVal = -1;
6705 return nullptr;
6706 }
6707 case OMPD_target_teams_loop:
6708 case OMPD_target_teams:
6709 case OMPD_target_teams_distribute:
6710 case OMPD_target_teams_distribute_simd:
6711 case OMPD_target_teams_distribute_parallel_for:
6712 case OMPD_target_teams_distribute_parallel_for_simd: {
6713 if (D.hasClausesOfKind<OMPNumTeamsClause>()) {
6714 const Expr *NumTeams =
6715 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams().front();
6716 if (NumTeams->isIntegerConstantExpr(CGF.getContext()))
6717 if (auto Constant = NumTeams->getIntegerConstantExpr(CGF.getContext()))
6718 MinTeamsVal = MaxTeamsVal = Constant->getExtValue();
6719 return NumTeams;
6720 }
6721 MinTeamsVal = MaxTeamsVal = 0;
6722 return nullptr;
6723 }
6724 case OMPD_target_parallel:
6725 case OMPD_target_parallel_for:
6726 case OMPD_target_parallel_for_simd:
6727 case OMPD_target_parallel_loop:
6728 case OMPD_target_simd:
6729 MinTeamsVal = MaxTeamsVal = 1;
6730 return nullptr;
6731 case OMPD_parallel:
6732 case OMPD_for:
6733 case OMPD_parallel_for:
6734 case OMPD_parallel_loop:
6735 case OMPD_parallel_master:
6736 case OMPD_parallel_sections:
6737 case OMPD_for_simd:
6738 case OMPD_parallel_for_simd:
6739 case OMPD_cancel:
6740 case OMPD_cancellation_point:
6741 case OMPD_ordered_standalone:
6742 case OMPD_ordered_blockassoc:
6743 case OMPD_threadprivate:
6744 case OMPD_allocate:
6745 case OMPD_task:
6746 case OMPD_simd:
6747 case OMPD_tile:
6748 case OMPD_unroll:
6749 case OMPD_sections:
6750 case OMPD_section:
6751 case OMPD_single:
6752 case OMPD_master:
6753 case OMPD_critical:
6754 case OMPD_taskyield:
6755 case OMPD_barrier:
6756 case OMPD_taskwait:
6757 case OMPD_taskgroup:
6758 case OMPD_atomic:
6759 case OMPD_flush:
6760 case OMPD_depobj:
6761 case OMPD_scan:
6762 case OMPD_teams:
6763 case OMPD_target_data:
6764 case OMPD_target_exit_data:
6765 case OMPD_target_enter_data:
6766 case OMPD_distribute:
6767 case OMPD_distribute_simd:
6768 case OMPD_distribute_parallel_for:
6769 case OMPD_distribute_parallel_for_simd:
6770 case OMPD_teams_distribute:
6771 case OMPD_teams_distribute_simd:
6772 case OMPD_teams_distribute_parallel_for:
6773 case OMPD_teams_distribute_parallel_for_simd:
6774 case OMPD_target_update:
6775 case OMPD_declare_simd:
6776 case OMPD_declare_variant:
6777 case OMPD_begin_declare_variant:
6778 case OMPD_end_declare_variant:
6779 case OMPD_declare_target:
6780 case OMPD_end_declare_target:
6781 case OMPD_declare_reduction:
6782 case OMPD_declare_mapper:
6783 case OMPD_taskloop:
6784 case OMPD_taskloop_simd:
6785 case OMPD_master_taskloop:
6786 case OMPD_master_taskloop_simd:
6787 case OMPD_parallel_master_taskloop:
6788 case OMPD_parallel_master_taskloop_simd:
6789 case OMPD_requires:
6790 case OMPD_metadirective:
6791 case OMPD_unknown:
6792 break;
6793 default:
6794 break;
6795 }
6796 llvm_unreachable("Unexpected directive kind.");
6797}
6798
6800 CodeGenFunction &CGF, const OMPExecutableDirective &D) {
6801 assert(!CGF.getLangOpts().OpenMPIsTargetDevice &&
6802 "Clauses associated with the teams directive expected to be emitted "
6803 "only for the host!");
6804 CGBuilderTy &Bld = CGF.Builder;
6805 int32_t MinNT = -1, MaxNT = -1;
6806 const Expr *NumTeams =
6807 getNumTeamsExprForTargetDirective(CGF, D, MinNT, MaxNT);
6808 if (NumTeams != nullptr) {
6809 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6810
6811 switch (DirectiveKind) {
6812 case OMPD_target: {
6813 const auto *CS = D.getInnermostCapturedStmt();
6814 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6815 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6816 llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(NumTeams,
6817 /*IgnoreResultAssign*/ true);
6818 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6819 /*isSigned=*/true);
6820 }
6821 case OMPD_target_teams:
6822 case OMPD_target_teams_distribute:
6823 case OMPD_target_teams_distribute_simd:
6824 case OMPD_target_teams_distribute_parallel_for:
6825 case OMPD_target_teams_distribute_parallel_for_simd: {
6826 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF);
6827 llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(NumTeams,
6828 /*IgnoreResultAssign*/ true);
6829 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6830 /*isSigned=*/true);
6831 }
6832 default:
6833 break;
6834 }
6835 }
6836
6837 assert(MinNT == MaxNT && "Num threads ranges require handling here.");
6838 return llvm::ConstantInt::getSigned(CGF.Int32Ty, MinNT);
6839}
6840
6841/// Merge the thread count upper bound \p Val into \p UpperBound.
6842///
6843/// \p UpperBound is -1 while no thread limiting clause has been seen, 0 once
6844/// one has been seen whose value is not known at compile time, and otherwise
6845/// the smallest constant bound found so far.
6846///
6847/// Thread limiting clauses compose by taking the minimum, so a constant bound
6848/// stays valid whatever the clauses that are not compile time constants
6849/// evaluate to. That makes it correct to replace the 0 marker with \p Val, and
6850/// necessary to keep a clause from raising a smaller bound found earlier.
6851static void mergeThreadCountUpperBound(int32_t &UpperBound, int32_t Val) {
6852 UpperBound = UpperBound > 0 ? std::min(UpperBound, Val) : Val;
6853}
6854
6855/// Check for a num threads constant value (stored in \p DefaultVal), or
6856/// expression (stored in \p E). If the value is conditional (via an if-clause),
6857/// store the condition in \p CondVal. If \p E, and \p CondVal respectively, are
6858/// nullptr, no expression evaluation is perfomed.
6859static void getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS,
6860 const Expr **E, int32_t &UpperBound,
6861 bool UpperBoundOnly, llvm::Value **CondVal) {
6863 CGF.getContext(), CS->getCapturedStmt());
6864 const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child);
6865 if (!Dir)
6866 return;
6867
6868 if (isOpenMPParallelDirective(Dir->getDirectiveKind())) {
6869 // Handle if clause. If if clause present, the number of threads is
6870 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
6871 if (CondVal && Dir->hasClausesOfKind<OMPIfClause>()) {
6872 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6873 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6874 const OMPIfClause *IfClause = nullptr;
6875 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) {
6876 if (C->getNameModifier() == OMPD_unknown ||
6877 C->getNameModifier() == OMPD_parallel) {
6878 IfClause = C;
6879 break;
6880 }
6881 }
6882 if (IfClause) {
6883 const Expr *CondExpr = IfClause->getCondition();
6884 bool Result;
6885 if (CondExpr->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
6886 if (!Result) {
6887 UpperBound = 1;
6888 return;
6889 }
6890 } else {
6892 if (const auto *PreInit =
6893 cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) {
6894 for (const auto *I : PreInit->decls()) {
6895 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6896 CGF.EmitVarDecl(cast<VarDecl>(*I));
6897 } else {
6900 CGF.EmitAutoVarCleanups(Emission);
6901 }
6902 }
6903 *CondVal = CGF.EvaluateExprAsBool(CondExpr);
6904 }
6905 }
6906 }
6907 }
6908 // Check the value of num_threads clause iff if clause was not specified
6909 // or is not evaluated to false.
6910 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) {
6911 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6912 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6913 const auto *NumThreadsClause =
6914 Dir->getSingleClause<OMPNumThreadsClause>();
6915 const Expr *NTExpr = NumThreadsClause->getNumThreads().front();
6916 if (NTExpr->isIntegerConstantExpr(CGF.getContext()))
6917 if (auto Constant = NTExpr->getIntegerConstantExpr(CGF.getContext()))
6919 UpperBound, static_cast<int32_t>(Constant->getZExtValue()));
6920 // If we haven't found a upper bound, remember we saw a thread limiting
6921 // clause.
6922 if (UpperBound == -1)
6923 UpperBound = 0;
6924 if (!E)
6925 return;
6926 CodeGenFunction::LexicalScope Scope(CGF, NTExpr->getSourceRange());
6927 if (const auto *PreInit =
6928 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) {
6929 for (const auto *I : PreInit->decls()) {
6930 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6931 CGF.EmitVarDecl(cast<VarDecl>(*I));
6932 } else {
6935 CGF.EmitAutoVarCleanups(Emission);
6936 }
6937 }
6938 }
6939 *E = NTExpr;
6940 }
6941 return;
6942 }
6943 if (isOpenMPSimdDirective(Dir->getDirectiveKind()))
6944 UpperBound = 1;
6945}
6946
6948 CodeGenFunction &CGF, const OMPExecutableDirective &D, int32_t &UpperBound,
6949 bool UpperBoundOnly, llvm::Value **CondVal, const Expr **ThreadLimitExpr) {
6950 assert((!CGF.getLangOpts().OpenMPIsTargetDevice || UpperBoundOnly) &&
6951 "Clauses associated with the teams directive expected to be emitted "
6952 "only for the host!");
6953 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6954 assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6955 "Expected target-based executable directive.");
6956
6957 const Expr *NT = nullptr;
6958 const Expr **NTPtr = UpperBoundOnly ? nullptr : &NT;
6959
6960 auto CheckForConstExpr = [&](const Expr *E, const Expr **EPtr) {
6961 if (E->isIntegerConstantExpr(CGF.getContext())) {
6962 if (auto Constant = E->getIntegerConstantExpr(CGF.getContext()))
6964 UpperBound, static_cast<int32_t>(Constant->getZExtValue()));
6965 }
6966 // If we haven't found a upper bound, remember we saw a thread limiting
6967 // clause.
6968 if (UpperBound == -1)
6969 UpperBound = 0;
6970 if (EPtr)
6971 *EPtr = E;
6972 };
6973
6974 auto ReturnSequential = [&]() {
6975 UpperBound = 1;
6976 return NT;
6977 };
6978
6979 switch (DirectiveKind) {
6980 case OMPD_target: {
6981 const CapturedStmt *CS = D.getInnermostCapturedStmt();
6982 getNumThreads(CGF, CS, NTPtr, UpperBound, UpperBoundOnly, CondVal);
6984 CGF.getContext(), CS->getCapturedStmt());
6985 // TODO: The standard is not clear how to resolve two thread limit clauses,
6986 // let's pick the teams one if it's present, otherwise the target one.
6987 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
6988 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6989 if (const auto *TLC = Dir->getSingleClause<OMPThreadLimitClause>()) {
6990 ThreadLimitClause = TLC;
6991 if (ThreadLimitExpr) {
6992 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6993 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6995 CGF,
6996 ThreadLimitClause->getThreadLimit().front()->getSourceRange());
6997 if (const auto *PreInit =
6998 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) {
6999 for (const auto *I : PreInit->decls()) {
7000 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7001 CGF.EmitVarDecl(cast<VarDecl>(*I));
7002 } else {
7005 CGF.EmitAutoVarCleanups(Emission);
7006 }
7007 }
7008 }
7009 }
7010 }
7011 }
7012 if (ThreadLimitClause)
7013 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7014 ThreadLimitExpr);
7015 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7016 if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) &&
7017 !isOpenMPDistributeDirective(Dir->getDirectiveKind())) {
7018 CS = Dir->getInnermostCapturedStmt();
7019 // Now that the 'teams' level has been peeled off, the remainder is
7020 // shaped like a 'target teams' region, so pick up the num_threads of
7021 // the directive nested in it the same way the OMPD_target_teams case
7022 // below does. Without this the upper bound of a construct written as
7023 // 'target' / 'teams' / 'distribute parallel for' would stay at the
7024 // default, while every combined spelling of the same construct honors
7025 // the clause. Only the bound is taken here: passing null for the
7026 // expression and the condition keeps this from emitting anything, so
7027 // the value the host passes to the kernel launch is left as it was.
7028 getNumThreads(CGF, CS, /*E=*/nullptr, UpperBound, UpperBoundOnly,
7029 /*CondVal=*/nullptr);
7031 CGF.getContext(), CS->getCapturedStmt());
7032 Dir = dyn_cast_or_null<OMPExecutableDirective>(Child);
7033 }
7034 if (Dir && isOpenMPParallelDirective(Dir->getDirectiveKind())) {
7035 CS = Dir->getInnermostCapturedStmt();
7036 getNumThreads(CGF, CS, NTPtr, UpperBound, UpperBoundOnly, CondVal);
7037 } else if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind()))
7038 return ReturnSequential();
7039 }
7040 return NT;
7041 }
7042 case OMPD_target_teams: {
7043 if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7044 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7045 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7046 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7047 ThreadLimitExpr);
7048 }
7049 const CapturedStmt *CS = D.getInnermostCapturedStmt();
7050 getNumThreads(CGF, CS, NTPtr, UpperBound, UpperBoundOnly, CondVal);
7052 CGF.getContext(), CS->getCapturedStmt());
7053 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7054 if (Dir->getDirectiveKind() == OMPD_distribute) {
7055 CS = Dir->getInnermostCapturedStmt();
7056 getNumThreads(CGF, CS, NTPtr, UpperBound, UpperBoundOnly, CondVal);
7057 }
7058 }
7059 return NT;
7060 }
7061 case OMPD_target_teams_distribute:
7062 if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7063 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7064 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7065 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7066 ThreadLimitExpr);
7067 }
7068 getNumThreads(CGF, D.getInnermostCapturedStmt(), NTPtr, UpperBound,
7069 UpperBoundOnly, CondVal);
7070 return NT;
7071 case OMPD_target_teams_loop:
7072 case OMPD_target_parallel_loop:
7073 case OMPD_target_parallel:
7074 case OMPD_target_parallel_for:
7075 case OMPD_target_parallel_for_simd:
7076 case OMPD_target_teams_distribute_parallel_for:
7077 case OMPD_target_teams_distribute_parallel_for_simd: {
7078 if (CondVal && D.hasClausesOfKind<OMPIfClause>()) {
7079 const OMPIfClause *IfClause = nullptr;
7080 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) {
7081 if (C->getNameModifier() == OMPD_unknown ||
7082 C->getNameModifier() == OMPD_parallel) {
7083 IfClause = C;
7084 break;
7085 }
7086 }
7087 if (IfClause) {
7088 const Expr *Cond = IfClause->getCondition();
7089 bool Result;
7090 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
7091 if (!Result)
7092 return ReturnSequential();
7093 } else {
7095 *CondVal = CGF.EvaluateExprAsBool(Cond);
7096 }
7097 }
7098 }
7099 if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7100 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7101 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7102 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7103 ThreadLimitExpr);
7104 }
7105 if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
7106 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
7107 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
7108 CheckForConstExpr(NumThreadsClause->getNumThreads().front(), nullptr);
7109 return NumThreadsClause->getNumThreads().front();
7110 }
7111 return NT;
7112 }
7113 case OMPD_target_teams_distribute_simd:
7114 case OMPD_target_simd:
7115 return ReturnSequential();
7116 default:
7117 break;
7118 }
7119 llvm_unreachable("Unsupported directive kind.");
7120}
7121
7123 CodeGenFunction &CGF, const OMPExecutableDirective &D) {
7124 llvm::Value *NumThreadsVal = nullptr;
7125 llvm::Value *CondVal = nullptr;
7126 llvm::Value *ThreadLimitVal = nullptr;
7127 const Expr *ThreadLimitExpr = nullptr;
7128 int32_t UpperBound = -1;
7129
7131 CGF, D, UpperBound, /* UpperBoundOnly */ false, &CondVal,
7132 &ThreadLimitExpr);
7133
7134 // Thread limit expressions are used below, emit them.
7135 if (ThreadLimitExpr) {
7136 ThreadLimitVal =
7137 CGF.EmitScalarExpr(ThreadLimitExpr, /*IgnoreResultAssign=*/true);
7138 ThreadLimitVal = CGF.Builder.CreateIntCast(ThreadLimitVal, CGF.Int32Ty,
7139 /*isSigned=*/false);
7140 }
7141
7142 // Generate the num teams expression.
7143 if (UpperBound == 1) {
7144 NumThreadsVal = CGF.Builder.getInt32(UpperBound);
7145 } else if (NT) {
7146 NumThreadsVal = CGF.EmitScalarExpr(NT, /*IgnoreResultAssign=*/true);
7147 NumThreadsVal = CGF.Builder.CreateIntCast(NumThreadsVal, CGF.Int32Ty,
7148 /*isSigned=*/false);
7149 } else if (ThreadLimitVal) {
7150 // If we do not have a num threads value but a thread limit, replace the
7151 // former with the latter. We know handled the thread limit expression.
7152 NumThreadsVal = ThreadLimitVal;
7153 ThreadLimitVal = nullptr;
7154 } else {
7155 // Default to "0" which means runtime choice.
7156 assert(!ThreadLimitVal && "Default not applicable with thread limit value");
7157 NumThreadsVal = CGF.Builder.getInt32(0);
7158 }
7159
7160 // Handle if clause. If if clause present, the number of threads is
7161 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
7162 if (CondVal) {
7164 NumThreadsVal = CGF.Builder.CreateSelect(CondVal, NumThreadsVal,
7165 CGF.Builder.getInt32(1));
7166 }
7167
7168 // If the thread limit and num teams expression were present, take the
7169 // minimum.
7170 if (ThreadLimitVal) {
7171 NumThreadsVal = CGF.Builder.CreateSelect(
7172 CGF.Builder.CreateICmpULT(ThreadLimitVal, NumThreadsVal),
7173 ThreadLimitVal, NumThreadsVal);
7174 }
7175
7176 return NumThreadsVal;
7177}
7178
7179namespace {
7181
7182// Utility to handle information from clauses associated with a given
7183// construct that use mappable expressions (e.g. 'map' clause, 'to' clause).
7184// It provides a convenient interface to obtain the information and generate
7185// code for that information.
7186class MappableExprsHandler {
7187public:
7188 /// Custom comparator for attach-pointer expressions that compares them by
7189 /// complexity (i.e. their component-depth) first, then by the order in which
7190 /// they were computed by collectAttachPtrExprInfo(), if they are semantically
7191 /// different.
7192 struct AttachPtrExprComparator {
7193 const MappableExprsHandler &Handler;
7194 // Cache of previous equality comparison results.
7195 mutable llvm::DenseMap<std::pair<const Expr *, const Expr *>, bool>
7196 CachedEqualityComparisons;
7197
7198 AttachPtrExprComparator(const MappableExprsHandler &H) : Handler(H) {}
7199 AttachPtrExprComparator() = delete;
7200
7201 // Return true iff LHS is "less than" RHS.
7202 bool operator()(const Expr *LHS, const Expr *RHS) const {
7203 if (LHS == RHS)
7204 return false;
7205
7206 // First, compare by complexity (depth)
7207 const auto ItLHS = Handler.AttachPtrComponentDepthMap.find(LHS);
7208 const auto ItRHS = Handler.AttachPtrComponentDepthMap.find(RHS);
7209
7210 std::optional<size_t> DepthLHS =
7211 (ItLHS != Handler.AttachPtrComponentDepthMap.end()) ? ItLHS->second
7212 : std::nullopt;
7213 std::optional<size_t> DepthRHS =
7214 (ItRHS != Handler.AttachPtrComponentDepthMap.end()) ? ItRHS->second
7215 : std::nullopt;
7216
7217 // std::nullopt (no attach pointer) has lowest complexity
7218 if (!DepthLHS.has_value() && !DepthRHS.has_value()) {
7219 // Both have same complexity, now check semantic equality
7220 if (areEqual(LHS, RHS))
7221 return false;
7222 // Different semantically, compare by computation order
7223 return wasComputedBefore(LHS, RHS);
7224 }
7225 if (!DepthLHS.has_value())
7226 return true; // LHS has lower complexity
7227 if (!DepthRHS.has_value())
7228 return false; // RHS has lower complexity
7229
7230 // Both have values, compare by depth (lower depth = lower complexity)
7231 if (DepthLHS.value() != DepthRHS.value())
7232 return DepthLHS.value() < DepthRHS.value();
7233
7234 // Same complexity, now check semantic equality
7235 if (areEqual(LHS, RHS))
7236 return false;
7237 // Different semantically, compare by computation order
7238 return wasComputedBefore(LHS, RHS);
7239 }
7240
7241 public:
7242 /// Return true if \p LHS and \p RHS are semantically equal. Uses pre-cached
7243 /// results, if available, otherwise does a recursive semantic comparison.
7244 bool areEqual(const Expr *LHS, const Expr *RHS) const {
7245 // Check cache first for faster lookup
7246 const auto CachedResultIt = CachedEqualityComparisons.find({LHS, RHS});
7247 if (CachedResultIt != CachedEqualityComparisons.end())
7248 return CachedResultIt->second;
7249
7250 bool ComparisonResult = areSemanticallyEqual(LHS, RHS);
7251
7252 // Cache the result for future lookups (both orders since semantic
7253 // equality is commutative)
7254 CachedEqualityComparisons[{LHS, RHS}] = ComparisonResult;
7255 CachedEqualityComparisons[{RHS, LHS}] = ComparisonResult;
7256 return ComparisonResult;
7257 }
7258
7259 /// Compare the two attach-ptr expressions by their computation order.
7260 /// Returns true iff LHS was computed before RHS by
7261 /// collectAttachPtrExprInfo().
7262 bool wasComputedBefore(const Expr *LHS, const Expr *RHS) const {
7263 const size_t &OrderLHS = Handler.AttachPtrComputationOrderMap.at(LHS);
7264 const size_t &OrderRHS = Handler.AttachPtrComputationOrderMap.at(RHS);
7265
7266 return OrderLHS < OrderRHS;
7267 }
7268
7269 private:
7270 /// Helper function to compare attach-pointer expressions semantically.
7271 /// This function handles various expression types that can be part of an
7272 /// attach-pointer.
7273 /// TODO: Not urgent, but we should ideally return true when comparing
7274 /// `p[10]`, `*(p + 10)`, `*(p + 5 + 5)`, `p[10:1]` etc.
7275 bool areSemanticallyEqual(const Expr *LHS, const Expr *RHS) const {
7276 if (LHS == RHS)
7277 return true;
7278
7279 // If only one is null, they aren't equal
7280 if (!LHS || !RHS)
7281 return false;
7282
7283 ASTContext &Ctx = Handler.CGF.getContext();
7284 // Strip away parentheses and no-op casts to get to the core expression
7285 LHS = LHS->IgnoreParenNoopCasts(Ctx);
7286 RHS = RHS->IgnoreParenNoopCasts(Ctx);
7287
7288 // Direct pointer comparison of the underlying expressions
7289 if (LHS == RHS)
7290 return true;
7291
7292 // Check if the expression classes match
7293 if (LHS->getStmtClass() != RHS->getStmtClass())
7294 return false;
7295
7296 // Handle DeclRefExpr (variable references)
7297 if (const auto *LD = dyn_cast<DeclRefExpr>(LHS)) {
7298 const auto *RD = dyn_cast<DeclRefExpr>(RHS);
7299 if (!RD)
7300 return false;
7301 return LD->getDecl()->getCanonicalDecl() ==
7302 RD->getDecl()->getCanonicalDecl();
7303 }
7304
7305 // Handle ArraySubscriptExpr (array indexing like a[i])
7306 if (const auto *LA = dyn_cast<ArraySubscriptExpr>(LHS)) {
7307 const auto *RA = dyn_cast<ArraySubscriptExpr>(RHS);
7308 if (!RA)
7309 return false;
7310 return areSemanticallyEqual(LA->getBase(), RA->getBase()) &&
7311 areSemanticallyEqual(LA->getIdx(), RA->getIdx());
7312 }
7313
7314 // Handle MemberExpr (member access like s.m or p->m)
7315 if (const auto *LM = dyn_cast<MemberExpr>(LHS)) {
7316 const auto *RM = dyn_cast<MemberExpr>(RHS);
7317 if (!RM)
7318 return false;
7319 if (LM->getMemberDecl()->getCanonicalDecl() !=
7320 RM->getMemberDecl()->getCanonicalDecl())
7321 return false;
7322 return areSemanticallyEqual(LM->getBase(), RM->getBase());
7323 }
7324
7325 // Handle UnaryOperator (unary operations like *p, &x, etc.)
7326 if (const auto *LU = dyn_cast<UnaryOperator>(LHS)) {
7327 const auto *RU = dyn_cast<UnaryOperator>(RHS);
7328 if (!RU)
7329 return false;
7330 if (LU->getOpcode() != RU->getOpcode())
7331 return false;
7332 return areSemanticallyEqual(LU->getSubExpr(), RU->getSubExpr());
7333 }
7334
7335 // Handle BinaryOperator (binary operations like p + offset)
7336 if (const auto *LB = dyn_cast<BinaryOperator>(LHS)) {
7337 const auto *RB = dyn_cast<BinaryOperator>(RHS);
7338 if (!RB)
7339 return false;
7340 if (LB->getOpcode() != RB->getOpcode())
7341 return false;
7342 return areSemanticallyEqual(LB->getLHS(), RB->getLHS()) &&
7343 areSemanticallyEqual(LB->getRHS(), RB->getRHS());
7344 }
7345
7346 // Handle ArraySectionExpr (array sections like a[0:1])
7347 // Attach pointers should not contain array-sections, but currently we
7348 // don't emit an error.
7349 if (const auto *LAS = dyn_cast<ArraySectionExpr>(LHS)) {
7350 const auto *RAS = dyn_cast<ArraySectionExpr>(RHS);
7351 if (!RAS)
7352 return false;
7353 return areSemanticallyEqual(LAS->getBase(), RAS->getBase()) &&
7354 areSemanticallyEqual(LAS->getLowerBound(),
7355 RAS->getLowerBound()) &&
7356 areSemanticallyEqual(LAS->getLength(), RAS->getLength());
7357 }
7358
7359 // Handle CastExpr (explicit casts)
7360 if (const auto *LC = dyn_cast<CastExpr>(LHS)) {
7361 const auto *RC = dyn_cast<CastExpr>(RHS);
7362 if (!RC)
7363 return false;
7364 if (LC->getCastKind() != RC->getCastKind())
7365 return false;
7366 return areSemanticallyEqual(LC->getSubExpr(), RC->getSubExpr());
7367 }
7368
7369 // Handle CXXThisExpr (this pointer)
7370 if (isa<CXXThisExpr>(LHS) && isa<CXXThisExpr>(RHS))
7371 return true;
7372
7373 // Handle IntegerLiteral (integer constants)
7374 if (const auto *LI = dyn_cast<IntegerLiteral>(LHS)) {
7375 const auto *RI = dyn_cast<IntegerLiteral>(RHS);
7376 if (!RI)
7377 return false;
7378 return LI->getValue() == RI->getValue();
7379 }
7380
7381 // Handle CharacterLiteral (character constants)
7382 if (const auto *LC = dyn_cast<CharacterLiteral>(LHS)) {
7383 const auto *RC = dyn_cast<CharacterLiteral>(RHS);
7384 if (!RC)
7385 return false;
7386 return LC->getValue() == RC->getValue();
7387 }
7388
7389 // Handle FloatingLiteral (floating point constants)
7390 if (const auto *LF = dyn_cast<FloatingLiteral>(LHS)) {
7391 const auto *RF = dyn_cast<FloatingLiteral>(RHS);
7392 if (!RF)
7393 return false;
7394 // Use bitwise comparison for floating point literals
7395 return LF->getValue().bitwiseIsEqual(RF->getValue());
7396 }
7397
7398 // Handle StringLiteral (string constants)
7399 if (const auto *LS = dyn_cast<StringLiteral>(LHS)) {
7400 const auto *RS = dyn_cast<StringLiteral>(RHS);
7401 if (!RS)
7402 return false;
7403 return LS->getString() == RS->getString();
7404 }
7405
7406 // Handle CXXNullPtrLiteralExpr (nullptr)
7408 return true;
7409
7410 // Handle CXXBoolLiteralExpr (true/false)
7411 if (const auto *LB = dyn_cast<CXXBoolLiteralExpr>(LHS)) {
7412 const auto *RB = dyn_cast<CXXBoolLiteralExpr>(RHS);
7413 if (!RB)
7414 return false;
7415 return LB->getValue() == RB->getValue();
7416 }
7417
7418 // Fallback for other forms - use the existing comparison method
7419 return Expr::isSameComparisonOperand(LHS, RHS);
7420 }
7421 };
7422
7423 /// Get the offset of the OMP_MAP_MEMBER_OF field.
7424 static unsigned getFlagMemberOffset() {
7425 unsigned Offset = 0;
7426 for (uint64_t Remain =
7427 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
7428 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF);
7429 !(Remain & 1); Remain = Remain >> 1)
7430 Offset++;
7431 return Offset;
7432 }
7433
7434 /// Class that holds debugging information for a data mapping to be passed to
7435 /// the runtime library.
7436 class MappingExprInfo {
7437 /// The variable declaration used for the data mapping.
7438 const ValueDecl *MapDecl = nullptr;
7439 /// The original expression used in the map clause, or null if there is
7440 /// none.
7441 const Expr *MapExpr = nullptr;
7442
7443 public:
7444 MappingExprInfo(const ValueDecl *MapDecl, const Expr *MapExpr = nullptr)
7445 : MapDecl(MapDecl), MapExpr(MapExpr) {}
7446
7447 const ValueDecl *getMapDecl() const { return MapDecl; }
7448 const Expr *getMapExpr() const { return MapExpr; }
7449 };
7450
7451 using DeviceInfoTy = llvm::OpenMPIRBuilder::DeviceInfoTy;
7452 using MapBaseValuesArrayTy = llvm::OpenMPIRBuilder::MapValuesArrayTy;
7453 using MapValuesArrayTy = llvm::OpenMPIRBuilder::MapValuesArrayTy;
7454 using MapFlagsArrayTy = llvm::OpenMPIRBuilder::MapFlagsArrayTy;
7455 using MapDimArrayTy = llvm::OpenMPIRBuilder::MapDimArrayTy;
7456 using MapNonContiguousArrayTy =
7457 llvm::OpenMPIRBuilder::MapNonContiguousArrayTy;
7458 using MapExprsArrayTy = SmallVector<MappingExprInfo, 4>;
7459 using MapValueDeclsArrayTy = SmallVector<const ValueDecl *, 4>;
7460 using MapData =
7462 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>,
7463 bool /*IsImplicit*/, const ValueDecl *, const Expr *>;
7464 using MapDataArrayTy = SmallVector<MapData, 4>;
7465
7466 /// This structure contains combined information generated for mappable
7467 /// clauses, including base pointers, pointers, sizes, map types, user-defined
7468 /// mappers, and non-contiguous information.
7469 struct MapCombinedInfoTy : llvm::OpenMPIRBuilder::MapInfosTy {
7470 MapExprsArrayTy Exprs;
7471 MapValueDeclsArrayTy Mappers;
7472 MapValueDeclsArrayTy DevicePtrDecls;
7473
7474 /// Append arrays in \a CurInfo.
7475 void append(MapCombinedInfoTy &CurInfo) {
7476 Exprs.append(CurInfo.Exprs.begin(), CurInfo.Exprs.end());
7477 DevicePtrDecls.append(CurInfo.DevicePtrDecls.begin(),
7478 CurInfo.DevicePtrDecls.end());
7479 Mappers.append(CurInfo.Mappers.begin(), CurInfo.Mappers.end());
7480 llvm::OpenMPIRBuilder::MapInfosTy::append(CurInfo);
7481 }
7482 };
7483
7484 /// Map between a struct and the its lowest & highest elements which have been
7485 /// mapped.
7486 /// [ValueDecl *] --> {LE(FieldIndex, Pointer),
7487 /// HE(FieldIndex, Pointer)}
7488 struct StructRangeInfoTy {
7489 MapCombinedInfoTy PreliminaryMapData;
7490 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = {
7491 0, Address::invalid()};
7492 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = {
7493 0, Address::invalid()};
7496 bool IsArraySection = false;
7497 bool HasCompleteRecord = false;
7498 };
7499
7500 /// A struct to store the attach pointer and pointee information, to be used
7501 /// when emitting an attach entry.
7502 struct AttachInfoTy {
7503 Address AttachPtrAddr = Address::invalid();
7504 Address AttachPteeAddr = Address::invalid();
7505 const ValueDecl *AttachPtrDecl = nullptr;
7506 const Expr *AttachMapExpr = nullptr;
7507
7508 bool isValid() const {
7509 return AttachPtrAddr.isValid() && AttachPteeAddr.isValid();
7510 }
7511 };
7512
7513 /// Check if there's any component list where the attach pointer expression
7514 /// matches the given captured variable.
7515 bool hasAttachEntryForCapturedVar(const ValueDecl *VD) const {
7516 for (const auto &AttachEntry : AttachPtrExprMap) {
7517 if (AttachEntry.second) {
7518 // Check if the attach pointer expression is a DeclRefExpr that
7519 // references the captured variable
7520 if (const auto *DRE = dyn_cast<DeclRefExpr>(AttachEntry.second))
7521 if (DRE->getDecl() == VD)
7522 return true;
7523 }
7524 }
7525 return false;
7526 }
7527
7528 /// Get the previously-cached attach pointer for a component list, if-any.
7529 const Expr *getAttachPtrExpr(
7531 const {
7532 const auto It = AttachPtrExprMap.find(Components);
7533 if (It != AttachPtrExprMap.end())
7534 return It->second;
7535
7536 return nullptr;
7537 }
7538
7539private:
7540 /// Kind that defines how a device pointer has to be returned.
7541 struct MapInfo {
7544 ArrayRef<OpenMPMapModifierKind> MapModifiers;
7545 ArrayRef<OpenMPMotionModifierKind> MotionModifiers;
7546 bool ReturnDevicePointer = false;
7547 bool IsImplicit = false;
7548 const ValueDecl *Mapper = nullptr;
7549 const Expr *VarRef = nullptr;
7550 bool ForDeviceAddr = false;
7551 bool HasUdpFbNullify = false;
7552
7553 MapInfo() = default;
7554 MapInfo(
7556 OpenMPMapClauseKind MapType,
7557 ArrayRef<OpenMPMapModifierKind> MapModifiers,
7558 ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
7559 bool ReturnDevicePointer, bool IsImplicit,
7560 const ValueDecl *Mapper = nullptr, const Expr *VarRef = nullptr,
7561 bool ForDeviceAddr = false, bool HasUdpFbNullify = false)
7562 : Components(Components), MapType(MapType), MapModifiers(MapModifiers),
7563 MotionModifiers(MotionModifiers),
7564 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit),
7565 Mapper(Mapper), VarRef(VarRef), ForDeviceAddr(ForDeviceAddr),
7566 HasUdpFbNullify(HasUdpFbNullify) {}
7567 };
7568
7569 /// The target directive from where the mappable clauses were extracted. It
7570 /// is either a executable directive or a user-defined mapper directive.
7571 llvm::PointerUnion<const OMPExecutableDirective *,
7572 const OMPDeclareMapperDecl *>
7573 CurDir;
7574
7575 /// Function the directive is being generated for.
7576 CodeGenFunction &CGF;
7577
7578 /// Set of all first private variables in the current directive.
7579 /// bool data is set to true if the variable is implicitly marked as
7580 /// firstprivate, false otherwise.
7581 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls;
7582
7583 /// Set of defaultmap clause kinds that use firstprivate behavior.
7584 llvm::SmallSet<OpenMPDefaultmapClauseKind, 4> DefaultmapFirstprivateKinds;
7585
7586 /// Map between device pointer declarations and their expression components.
7587 /// The key value for declarations in 'this' is null.
7588 llvm::DenseMap<
7589 const ValueDecl *,
7590 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7591 DevPointersMap;
7592
7593 /// Map between device addr declarations and their expression components.
7594 /// The key value for declarations in 'this' is null.
7595 llvm::DenseMap<
7596 const ValueDecl *,
7597 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7598 HasDevAddrsMap;
7599
7600 /// Map between lambda declarations and their map type.
7601 llvm::DenseMap<const ValueDecl *, const OMPMapClause *> LambdasMap;
7602
7603 /// Map from component lists to their attach pointer expressions.
7605 const Expr *>
7606 AttachPtrExprMap;
7607
7608 /// Map from attach pointer expressions to their component depth.
7609 /// nullptr key has std::nullopt depth. This can be used to order attach-ptr
7610 /// expressions with increasing/decreasing depth.
7611 /// The component-depth of `nullptr` (i.e. no attach-ptr) is `std::nullopt`.
7612 /// TODO: Not urgent, but we should ideally use the number of pointer
7613 /// dereferences in an expr as an indicator of its complexity, instead of the
7614 /// component-depth. That would be needed for us to treat `p[1]`, `*(p + 10)`,
7615 /// `*(p + 5 + 5)` together.
7616 llvm::DenseMap<const Expr *, std::optional<size_t>>
7617 AttachPtrComponentDepthMap = {{nullptr, std::nullopt}};
7618
7619 /// Map from attach pointer expressions to the order they were computed in, in
7620 /// collectAttachPtrExprInfo().
7621 llvm::DenseMap<const Expr *, size_t> AttachPtrComputationOrderMap = {
7622 {nullptr, 0}};
7623
7624 /// An instance of attach-ptr-expr comparator that can be used throughout the
7625 /// lifetime of this handler.
7626 AttachPtrExprComparator AttachPtrComparator;
7627
7628 llvm::Value *getExprTypeSize(const Expr *E) const {
7629 QualType ExprTy = E->getType().getCanonicalType();
7630
7631 // Calculate the size for array shaping expression.
7632 if (const auto *OAE = dyn_cast<OMPArrayShapingExpr>(E)) {
7633 llvm::Value *Size =
7634 CGF.getTypeSize(OAE->getBase()->getType()->getPointeeType());
7635 for (const Expr *SE : OAE->getDimensions()) {
7636 llvm::Value *Sz = CGF.EmitScalarExpr(SE);
7637 Sz = CGF.EmitScalarConversion(Sz, SE->getType(),
7638 CGF.getContext().getSizeType(),
7639 SE->getExprLoc());
7640 Size = CGF.Builder.CreateNUWMul(Size, Sz);
7641 }
7642 return Size;
7643 }
7644
7645 // Reference types are ignored for mapping purposes.
7646 if (const auto *RefTy = ExprTy->getAs<ReferenceType>())
7647 ExprTy = RefTy->getPointeeType().getCanonicalType();
7648
7649 // Given that an array section is considered a built-in type, we need to
7650 // do the calculation based on the length of the section instead of relying
7651 // on CGF.getTypeSize(E->getType()).
7652 if (const auto *OAE = dyn_cast<ArraySectionExpr>(E)) {
7653 QualType BaseTy = ArraySectionExpr::getBaseOriginalType(
7654 OAE->getBase()->IgnoreParenImpCasts())
7656
7657 // If there is no length associated with the expression and lower bound is
7658 // not specified too, that means we are using the whole length of the
7659 // base.
7660 if (!OAE->getLength() && OAE->getColonLocFirst().isValid() &&
7661 !OAE->getLowerBound())
7662 return CGF.getTypeSize(BaseTy);
7663
7664 llvm::Value *ElemSize;
7665 if (const auto *PTy = BaseTy->getAs<PointerType>()) {
7666 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType());
7667 } else {
7668 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr());
7669 assert(ATy && "Expecting array type if not a pointer type.");
7670 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType());
7671 }
7672
7673 // If we don't have a length at this point, that is because we have an
7674 // array section with a single element.
7675 if (!OAE->getLength() && OAE->getColonLocFirst().isInvalid())
7676 return ElemSize;
7677
7678 if (const Expr *LenExpr = OAE->getLength()) {
7679 llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr);
7680 LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(),
7681 CGF.getContext().getSizeType(),
7682 LenExpr->getExprLoc());
7683 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize);
7684 }
7685 assert(!OAE->getLength() && OAE->getColonLocFirst().isValid() &&
7686 OAE->getLowerBound() && "expected array_section[lb:].");
7687 // Size = sizetype - lb * elemtype;
7688 llvm::Value *LengthVal = CGF.getTypeSize(BaseTy);
7689 llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound());
7690 LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(),
7691 CGF.getContext().getSizeType(),
7692 OAE->getLowerBound()->getExprLoc());
7693 LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize);
7694 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal);
7695 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal);
7696 LengthVal = CGF.Builder.CreateSelect(
7697 Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0));
7698 return LengthVal;
7699 }
7700 return CGF.getTypeSize(ExprTy);
7701 }
7702
7703 /// Return the corresponding bits for a given map clause modifier. Add
7704 /// a flag marking the map as a pointer if requested. Add a flag marking the
7705 /// map as the first one of a series of maps that relate to the same map
7706 /// expression.
7707 OpenMPOffloadMappingFlags getMapTypeBits(
7708 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7709 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, bool IsImplicit,
7710 bool AddPtrFlag, bool AddIsTargetParamFlag, bool IsNonContiguous) const {
7711 OpenMPOffloadMappingFlags Bits =
7712 IsImplicit ? OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT
7713 : OpenMPOffloadMappingFlags::OMP_MAP_NONE;
7714 switch (MapType) {
7715 case OMPC_MAP_alloc:
7716 case OMPC_MAP_release:
7717 // alloc and release is the default behavior in the runtime library, i.e.
7718 // if we don't pass any bits alloc/release that is what the runtime is
7719 // going to do. Therefore, we don't need to signal anything for these two
7720 // type modifiers.
7721 break;
7722 case OMPC_MAP_to:
7723 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_TO;
7724 break;
7725 case OMPC_MAP_from:
7726 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_FROM;
7727 break;
7728 case OMPC_MAP_tofrom:
7729 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_TO |
7730 OpenMPOffloadMappingFlags::OMP_MAP_FROM;
7731 break;
7732 case OMPC_MAP_delete:
7733 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_DELETE;
7734 break;
7735 case OMPC_MAP_unknown:
7736 llvm_unreachable("Unexpected map type!");
7737 }
7738 if (AddPtrFlag)
7739 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ;
7740 if (AddIsTargetParamFlag)
7741 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM;
7742 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_always))
7743 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_ALWAYS;
7744 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_close))
7745 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_CLOSE;
7746 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_present) ||
7747 llvm::is_contained(MotionModifiers, OMPC_MOTION_MODIFIER_present))
7748 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_PRESENT;
7749 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_ompx_hold))
7750 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD;
7751 if (IsNonContiguous)
7752 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_NON_CONTIG;
7753 return Bits;
7754 }
7755
7756 /// Return true if the provided expression is a final array section. A
7757 /// final array section, is one whose length can't be proved to be one.
7758 bool isFinalArraySectionExpression(const Expr *E) const {
7759 const auto *OASE = dyn_cast<ArraySectionExpr>(E);
7760
7761 // It is not an array section and therefore not a unity-size one.
7762 if (!OASE)
7763 return false;
7764
7765 // An array section with no colon always refer to a single element.
7766 if (OASE->getColonLocFirst().isInvalid())
7767 return false;
7768
7769 const Expr *Length = OASE->getLength();
7770
7771 // If we don't have a length we have to check if the array has size 1
7772 // for this dimension. Also, we should always expect a length if the
7773 // base type is pointer.
7774 if (!Length) {
7775 QualType BaseQTy = ArraySectionExpr::getBaseOriginalType(
7776 OASE->getBase()->IgnoreParenImpCasts())
7778 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr()))
7779 return ATy->getSExtSize() != 1;
7780 // If we don't have a constant dimension length, we have to consider
7781 // the current section as having any size, so it is not necessarily
7782 // unitary. If it happen to be unity size, that's user fault.
7783 return true;
7784 }
7785
7786 // Check if the length evaluates to 1.
7787 Expr::EvalResult Result;
7788 if (!Length->EvaluateAsInt(Result, CGF.getContext()))
7789 return true; // Can have more that size 1.
7790
7791 llvm::APSInt ConstLength = Result.Val.getInt();
7792 return ConstLength.getSExtValue() != 1;
7793 }
7794
7795 /// Emit an attach entry into \p CombinedInfo, using the information from \p
7796 /// AttachInfo. For example, for a map of form `int *p; ... map(p[1:10])`,
7797 /// an attach entry has the following form:
7798 /// &p, &p[1], sizeof(void*), ATTACH
7799 void emitAttachEntry(CodeGenFunction &CGF, MapCombinedInfoTy &CombinedInfo,
7800 const AttachInfoTy &AttachInfo) const {
7801 assert(AttachInfo.isValid() &&
7802 "Expected valid attach pointer/pointee information!");
7803
7804 // Size is the size of the pointer itself - use pointer size, not BaseDecl
7805 // size
7806 llvm::Value *PointerSize = CGF.Builder.CreateIntCast(
7807 llvm::ConstantInt::get(
7808 CGF.CGM.SizeTy, CGF.getContext()
7810 .getQuantity()),
7811 CGF.Int64Ty, /*isSigned=*/true);
7812
7813 CombinedInfo.Exprs.emplace_back(AttachInfo.AttachPtrDecl,
7814 AttachInfo.AttachMapExpr);
7815 CombinedInfo.BasePointers.push_back(
7816 AttachInfo.AttachPtrAddr.emitRawPointer(CGF));
7817 CombinedInfo.DevicePtrDecls.push_back(nullptr);
7818 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
7819 CombinedInfo.Pointers.push_back(
7820 AttachInfo.AttachPteeAddr.emitRawPointer(CGF));
7821 CombinedInfo.Sizes.push_back(PointerSize);
7822 CombinedInfo.Types.push_back(OpenMPOffloadMappingFlags::OMP_MAP_ATTACH);
7823 // ATTACH entries themselves don't "have" a base attach-ptr.
7824 CombinedInfo.HasAttachPtr.push_back(false);
7825 CombinedInfo.Mappers.push_back(nullptr);
7826 CombinedInfo.NonContigInfo.Dims.push_back(1);
7827 }
7828
7829 /// A helper class to copy structures with overlapped elements, i.e. those
7830 /// which have mappings of both "s" and "s.mem". Consecutive elements that
7831 /// are not explicitly copied have mapping nodes synthesized for them,
7832 /// taking care to avoid generating zero-sized copies.
7833 class CopyOverlappedEntryGaps {
7834 CodeGenFunction &CGF;
7835 MapCombinedInfoTy &CombinedInfo;
7836 OpenMPOffloadMappingFlags Flags = OpenMPOffloadMappingFlags::OMP_MAP_NONE;
7837 const ValueDecl *MapDecl = nullptr;
7838 const Expr *MapExpr = nullptr;
7840 bool IsNonContiguous = false;
7841 uint64_t DimSize = 0;
7842 // These elements track the position as the struct is iterated over
7843 // (in order of increasing element address).
7844 const RecordDecl *LastParent = nullptr;
7845 uint64_t Cursor = 0;
7846 unsigned LastIndex = -1u;
7848
7849 public:
7850 CopyOverlappedEntryGaps(CodeGenFunction &CGF,
7851 MapCombinedInfoTy &CombinedInfo,
7852 OpenMPOffloadMappingFlags Flags,
7853 const ValueDecl *MapDecl, const Expr *MapExpr,
7854 Address BP, Address LB, bool IsNonContiguous,
7855 uint64_t DimSize)
7856 : CGF(CGF), CombinedInfo(CombinedInfo), Flags(Flags), MapDecl(MapDecl),
7857 MapExpr(MapExpr), BP(BP), IsNonContiguous(IsNonContiguous),
7858 DimSize(DimSize), LB(LB) {}
7859
7860 void processField(
7861 const OMPClauseMappableExprCommon::MappableComponent &MC,
7862 const FieldDecl *FD,
7863 llvm::function_ref<LValue(CodeGenFunction &, const MemberExpr *)>
7864 EmitMemberExprBase) {
7865 const RecordDecl *RD = FD->getParent();
7866 const ASTRecordLayout &RL = CGF.getContext().getASTRecordLayout(RD);
7867 uint64_t FieldOffset = RL.getFieldOffset(FD->getFieldIndex());
7868 uint64_t FieldSize =
7870 Address ComponentLB = Address::invalid();
7871
7872 if (FD->getType()->isLValueReferenceType()) {
7873 const auto *ME = cast<MemberExpr>(MC.getAssociatedExpression());
7874 LValue BaseLVal = EmitMemberExprBase(CGF, ME);
7875 ComponentLB =
7876 CGF.EmitLValueForFieldInitialization(BaseLVal, FD).getAddress();
7877 } else {
7878 ComponentLB =
7880 }
7881
7882 if (!LastParent)
7883 LastParent = RD;
7884 if (FD->getParent() == LastParent) {
7885 if (FD->getFieldIndex() != LastIndex + 1)
7886 copyUntilField(FD, ComponentLB);
7887 } else {
7888 LastParent = FD->getParent();
7889 if (((int64_t)FieldOffset - (int64_t)Cursor) > 0)
7890 copyUntilField(FD, ComponentLB);
7891 }
7892 Cursor = FieldOffset + FieldSize;
7893 LastIndex = FD->getFieldIndex();
7894 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1);
7895 }
7896
7897 void copyUntilField(const FieldDecl *FD, Address ComponentLB) {
7898 llvm::Value *ComponentLBPtr = ComponentLB.emitRawPointer(CGF);
7899 llvm::Value *LBPtr = LB.emitRawPointer(CGF);
7900 llvm::Value *Size = CGF.Builder.CreatePtrDiff(ComponentLBPtr, LBPtr);
7901 copySizedChunk(LBPtr, Size);
7902 }
7903
7904 void copyUntilEnd(Address HB) {
7905 if (LastParent) {
7906 const ASTRecordLayout &RL =
7907 CGF.getContext().getASTRecordLayout(LastParent);
7908 if ((uint64_t)CGF.getContext().toBits(RL.getSize()) <= Cursor)
7909 return;
7910 }
7911 llvm::Value *LBPtr = LB.emitRawPointer(CGF);
7912 llvm::Value *Size = CGF.Builder.CreatePtrDiff(
7913 CGF.Builder.CreateConstGEP(HB, 1).emitRawPointer(CGF), LBPtr);
7914 copySizedChunk(LBPtr, Size);
7915 }
7916
7917 void copySizedChunk(llvm::Value *Base, llvm::Value *Size) {
7918 CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr);
7919 CombinedInfo.BasePointers.push_back(BP.emitRawPointer(CGF));
7920 CombinedInfo.DevicePtrDecls.push_back(nullptr);
7921 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
7922 CombinedInfo.Pointers.push_back(Base);
7923 CombinedInfo.Sizes.push_back(
7924 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/false));
7925 CombinedInfo.Types.push_back(Flags);
7926 CombinedInfo.HasAttachPtr.push_back(false);
7927 CombinedInfo.Mappers.push_back(nullptr);
7928 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize : 1);
7929 }
7930 };
7931
7932 /// Generate the base pointers, section pointers, sizes, map type bits, and
7933 /// user-defined mappers (all included in \a CombinedInfo) for the provided
7934 /// map type, map or motion modifiers, and expression components.
7935 /// \a IsFirstComponent should be set to true if the provided set of
7936 /// components is the first associated with a capture.
7937 void generateInfoForComponentList(
7938 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7939 ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
7941 MapCombinedInfoTy &CombinedInfo,
7942 MapCombinedInfoTy &StructBaseCombinedInfo,
7943 StructRangeInfoTy &PartialStruct, AttachInfoTy &AttachInfo,
7944 bool IsFirstComponentList, bool IsImplicit,
7945 bool GenerateAllInfoForClauses, const ValueDecl *Mapper = nullptr,
7946 bool ForDeviceAddr = false, const ValueDecl *BaseDecl = nullptr,
7947 const Expr *MapExpr = nullptr,
7948 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
7949 OverlappedElements = {}) const {
7950
7951 // The following summarizes what has to be generated for each map and the
7952 // types below. The generated information is expressed in this order:
7953 // base pointer, section pointer, size, flags
7954 // (to add to the ones that come from the map type and modifier).
7955 // Entries annotated with (+) are only generated for "target" constructs,
7956 // and only if the variable at the beginning of the expression is used in
7957 // the region.
7958 //
7959 // double d;
7960 // int i[100];
7961 // float *p;
7962 // int **a = &i;
7963 //
7964 // struct S1 {
7965 // int i;
7966 // float f[50];
7967 // }
7968 // struct S2 {
7969 // int i;
7970 // float f[50];
7971 // S1 s;
7972 // double *p;
7973 // double *&pref;
7974 // struct S2 *ps;
7975 // int &ref;
7976 // }
7977 // S2 s;
7978 // S2 *ps;
7979 //
7980 // map(d)
7981 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM
7982 //
7983 // map(i)
7984 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM
7985 //
7986 // map(i[1:23])
7987 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM
7988 //
7989 // map(p)
7990 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
7991 //
7992 // map(p[1:24])
7993 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM // map pointee
7994 // &p, &p[1], sizeof(void*), ATTACH // attach pointer/pointee, if both
7995 // // are present, and either is new
7996 //
7997 // map(([22])p)
7998 // p, p, 22*sizeof(float), TARGET_PARAM | TO | FROM
7999 // &p, p, sizeof(void*), ATTACH
8000 //
8001 // map((*a)[0:3])
8002 // a, a, 0, TARGET_PARAM | IMPLICIT // (+)
8003 // (*a)[0], &(*a)[0], 3 * sizeof(int), TO | FROM
8004 // &(*a), &(*a)[0], sizeof(void*), ATTACH
8005 // (+) Only on target, if a is used in the region
8006 // Note: Since the attach base-pointer is `*a`, which is not a scalar
8007 // variable, it doesn't determine the clause on `a`. `a` is mapped using
8008 // a zero-length-array-section map by generateDefaultMapInfo, if it is
8009 // referenced in the target region, because it is a pointer.
8010 //
8011 // map(**a)
8012 // a, a, 0, TARGET_PARAM | IMPLICIT // (+)
8013 // &(*a)[0], &(*a)[0], sizeof(int), TO | FROM
8014 // &(*a), &(*a)[0], sizeof(void*), ATTACH
8015 // (+) Only on target, if a is used in the region
8016 //
8017 // map(s)
8018 // FIXME: This needs to also imply map(ref_ptr_ptee: s.ref), since the
8019 // effect is supposed to be same as if the user had a map for every element
8020 // of the struct. We currently do a shallow-map of s.
8021 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM
8022 //
8023 // map(s.i)
8024 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM
8025 //
8026 // map(s.s.f)
8027 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
8028 //
8029 // map(s.p)
8030 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM
8031 //
8032 // map(to: s.p[:22])
8033 // &s, &(s.p), sizeof(double*), TARGET_PARAM | IMPLICIT // (+)
8034 // &(s.p[0]), &(s.p[0]), 22 * sizeof(double*), TO | FROM
8035 // &(s.p), &(s.p[0]), sizeof(void*), ATTACH
8036 //
8037 // map(to: s.ref)
8038 // &s, &(ptr(s.ref)), sizeof(int*), TARGET_PARAM (*)
8039 // &s, &(ptee(s.ref)), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
8040 // (*) alloc space for struct members, only this is a target parameter.
8041 // (**) map the pointer (nothing to be mapped in this example) (the compiler
8042 // optimizes this entry out, same in the examples below)
8043 // (***) map the pointee (map: to)
8044 // Note: ptr(s.ref) represents the referring pointer of s.ref
8045 // ptee(s.ref) represents the referenced pointee of s.ref
8046 //
8047 // map(to: s.pref)
8048 // &s, &(ptr(s.pref)), sizeof(double**), TARGET_PARAM
8049 // &s, &(ptee(s.pref)), sizeof(double*), MEMBER_OF(1) | PTR_AND_OBJ | TO
8050 //
8051 // map(to: s.pref[:22])
8052 // &s, &(ptr(s.pref)), sizeof(double**), TARGET_PARAM | IMPLICIT // (+)
8053 // &s, &(ptee(s.pref)), sizeof(double*), MEMBER_OF(1) | PTR_AND_OBJ | TO |
8054 // FROM | IMPLICIT // (+)
8055 // &(ptee(s.pref)[0]), &(ptee(s.pref)[0]), 22 * sizeof(double), TO
8056 // &(ptee(s.pref)), &(ptee(s.pref)[0]), sizeof(void*), ATTACH
8057 //
8058 // map(s.ps)
8059 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM
8060 //
8061 // map(from: s.ps->s.i)
8062 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8063 // &(s.ps[0]), &(s.ps->s.i), sizeof(int), FROM
8064 // &(s.ps), &(s.ps->s.i), sizeof(void*), ATTACH
8065 //
8066 // map(to: s.ps->ps)
8067 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8068 // &(s.ps[0]), &(s.ps->ps), sizeof(S2*), TO
8069 // &(s.ps), &(s.ps->ps), sizeof(void*), ATTACH
8070 //
8071 // map(s.ps->ps->ps)
8072 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8073 // &(s.ps->ps[0]), &(s.ps->ps->ps), sizeof(S2*), TO
8074 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(void*), ATTACH
8075 //
8076 // map(to: s.ps->ps->s.f[:22])
8077 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8078 // &(s.ps->ps[0]), &(s.ps->ps->s.f[0]), 22*sizeof(float), TO
8079 // &(s.ps->ps), &(s.ps->ps->s.f[0]), sizeof(void*), ATTACH
8080 //
8081 // map(ps)
8082 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM
8083 //
8084 // map(ps->i)
8085 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM
8086 // &ps, &(ps->i), sizeof(void*), ATTACH
8087 //
8088 // map(ps->s.f)
8089 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
8090 // &ps, &(ps->s.f[0]), sizeof(ps), ATTACH
8091 //
8092 // map(from: ps->p)
8093 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM
8094 // &ps, &(ps->p), sizeof(ps), ATTACH
8095 //
8096 // map(to: ps->p[:22])
8097 // ps, &(ps[0]), 0, TARGET_PARAM | IMPLICIT // (+)
8098 // &(ps->p[0]), &(ps->p[0]), 22*sizeof(double), TO
8099 // &(ps->p), &(ps->p[0]), sizeof(void*), ATTACH
8100 //
8101 // map(ps->ps)
8102 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM
8103 // &ps, &(ps->ps), sizeof(ps), ATTACH
8104 //
8105 // map(from: ps->ps->s.i)
8106 // ps, &(ps[0]), 0, TARGET_PARAM | IMPLICIT // (+)
8107 // &(ps->ps[0]), &(ps->ps->s.i), sizeof(int), FROM
8108 // &(ps->ps), &(ps->ps->s.i), sizeof(void*), ATTACH
8109 //
8110 // map(from: ps->ps->ps)
8111 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8112 // &(ps->ps[0]), &(ps->ps->ps), sizeof(S2*), FROM
8113 // &(ps->ps), &(ps->ps->ps), sizeof(void*), ATTACH
8114 //
8115 // map(ps->ps->ps->ps)
8116 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8117 // &(ps->ps->ps[0]), &(ps->ps->ps->ps), sizeof(S2*), FROM
8118 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(void*), ATTACH
8119 //
8120 // map(to: ps->ps->ps->s.f[:22])
8121 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8122 // &(ps->ps->ps[0]), &(ps->ps->ps->s.f[0]), 22*sizeof(float), TO
8123 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), sizeof(void*), ATTACH
8124 //
8125 // map(to: s.f[:22]) map(from: s.p[:33])
8126 // On target, and if s is used in the region:
8127 //
8128 // &s, &(s.f[0]), 50*sizeof(float) +
8129 // sizeof(struct S1) +
8130 // sizeof(double*) (**), TARGET_PARAM
8131 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO
8132 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) | TO |
8133 // FROM | IMPLICIT
8134 // &(s.p[0]), &(s.p[0]), 33*sizeof(double), FROM
8135 // &(s.p), &(s.p[0]), sizeof(void*), ATTACH
8136 // (**) allocate contiguous space needed to fit all mapped members even if
8137 // we allocate space for members not mapped (in this example,
8138 // s.f[22..49] and s.s are not mapped, yet we must allocate space for
8139 // them as well because they fall between &s.f[0] and &s.p)
8140 //
8141 // On other constructs, and, if s is not used in the region, on target:
8142 // &s, &(s.f[0]), 22*sizeof(float), TO
8143 // &(s.p[0]), &(s.p[0]), 33*sizeof(double), FROM
8144 // &(s.p), &(s.p[0]), sizeof(void*), ATTACH
8145 //
8146 // map(from: s.f[:22]) map(to: ps->p[:33])
8147 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM
8148 // &ps[0], &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8149 // &(ps->p[0]), &(ps->p[0]), 33*sizeof(double), TO
8150 // &(ps->p), &(ps->p[0]), sizeof(void*), ATTACH
8151 //
8152 // map(from: s.f[:22], s.s) map(to: ps->p[:33])
8153 // &s, &(s.f[0]), 50*sizeof(float) +
8154 // sizeof(struct S1), TARGET_PARAM
8155 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM
8156 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM
8157 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8158 // &(ps->p[0]), &(ps->p[0]), 33*sizeof(double), TO
8159 // &(ps->p), &(ps->p[0]), sizeof(void*), ATTACH
8160 //
8161 // map(p[:100], p)
8162 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
8163 // p, &p[0], 100*sizeof(float), TO | FROM
8164 // &p, &p[0], sizeof(float*), ATTACH
8165
8166 // Track if the map information being generated is the first for a capture.
8167 bool IsCaptureFirstInfo = IsFirstComponentList;
8168 // When the variable is on a declare target link or in a to clause with
8169 // unified memory, a reference is needed to hold the host/device address
8170 // of the variable.
8171 bool RequiresReference = false;
8172
8173 // Scan the components from the base to the complete expression.
8174 auto CI = Components.rbegin();
8175 auto CE = Components.rend();
8176 auto I = CI;
8177
8178 // Track if the map information being generated is the first for a list of
8179 // components.
8180 bool IsExpressionFirstInfo = true;
8181 bool FirstPointerInComplexData = false;
8183 Address FinalLowestElem = Address::invalid();
8184 const Expr *AssocExpr = I->getAssociatedExpression();
8185 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr);
8186 const auto *OASE = dyn_cast<ArraySectionExpr>(AssocExpr);
8187 const auto *OAShE = dyn_cast<OMPArrayShapingExpr>(AssocExpr);
8188
8189 // Get the pointer-attachment base-pointer for the given list, if any.
8190 const Expr *AttachPtrExpr = getAttachPtrExpr(Components);
8191 auto [AttachPtrAddr, AttachPteeBaseAddr] =
8192 getAttachPtrAddrAndPteeBaseAddr(AttachPtrExpr, CGF);
8193
8194 bool HasAttachPtr = AttachPtrExpr != nullptr;
8195 bool FirstComponentIsForAttachPtr = AssocExpr == AttachPtrExpr;
8196 bool SeenAttachPtr = FirstComponentIsForAttachPtr;
8197
8198 if (FirstComponentIsForAttachPtr) {
8199 // No need to process AttachPtr here. It will be processed at the end
8200 // after we have computed the pointee's address.
8201 ++I;
8202 } else if (isa<MemberExpr>(AssocExpr)) {
8203 // The base is the 'this' pointer. The content of the pointer is going
8204 // to be the base of the field being mapped.
8205 BP = CGF.LoadCXXThisAddress();
8206 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) ||
8207 (OASE &&
8208 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) {
8209 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress();
8210 } else if (OAShE &&
8211 isa<CXXThisExpr>(OAShE->getBase()->IgnoreParenCasts())) {
8212 BP = Address(
8213 CGF.EmitScalarExpr(OAShE->getBase()),
8214 CGF.ConvertTypeForMem(OAShE->getBase()->getType()->getPointeeType()),
8215 CGF.getContext().getTypeAlignInChars(OAShE->getBase()->getType()));
8216 } else {
8217 // The base is the reference to the variable.
8218 // BP = &Var.
8219 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress();
8220 if (const auto *VD =
8221 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) {
8222 if (std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
8223 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) {
8224 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
8225 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
8226 *Res == OMPDeclareTargetDeclAttr::MT_Enter) &&
8228 RequiresReference = true;
8230 }
8231 }
8232 }
8233
8234 // If the variable is a pointer and is being dereferenced (i.e. is not
8235 // the last component), the base has to be the pointer itself, not its
8236 // reference. References are ignored for mapping purposes.
8237 QualType Ty =
8238 I->getAssociatedDeclaration()->getType().getNonReferenceType();
8239 if (Ty->isAnyPointerType() && std::next(I) != CE) {
8240 // No need to generate individual map information for the pointer, it
8241 // can be associated with the combined storage if shared memory mode is
8242 // active or the base declaration is not global variable.
8243 const auto *VD = dyn_cast<VarDecl>(I->getAssociatedDeclaration());
8245 !VD || VD->hasLocalStorage() || HasAttachPtr)
8246 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
8247 else
8248 FirstPointerInComplexData = true;
8249 ++I;
8250 }
8251 }
8252
8253 // Track whether a component of the list should be marked as MEMBER_OF some
8254 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry
8255 // in a component list should be marked as MEMBER_OF, all subsequent entries
8256 // do not belong to the base struct. E.g.
8257 // struct S2 s;
8258 // s.ps->ps->ps->f[:]
8259 // (1) (2) (3) (4)
8260 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a
8261 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3)
8262 // is the pointee of ps(2) which is not member of struct s, so it should not
8263 // be marked as such (it is still PTR_AND_OBJ).
8264 // The variable is initialized to false so that PTR_AND_OBJ entries which
8265 // are not struct members are not considered (e.g. array of pointers to
8266 // data).
8267 bool ShouldBeMemberOf = false;
8268
8269 // Variable keeping track of whether or not we have encountered a component
8270 // in the component list which is a member expression. Useful when we have a
8271 // pointer or a final array section, in which case it is the previous
8272 // component in the list which tells us whether we have a member expression.
8273 // E.g. X.f[:]
8274 // While processing the final array section "[:]" it is "f" which tells us
8275 // whether we are dealing with a member of a declared struct.
8276 const MemberExpr *EncounteredME = nullptr;
8277
8278 // Track for the total number of dimension. Start from one for the dummy
8279 // dimension.
8280 uint64_t DimSize = 1;
8281
8282 // Detects non-contiguous updates due to strided accesses.
8283 // Sets the 'IsNonContiguous' flag so that the 'MapType' bits are set
8284 // correctly when generating information to be passed to the runtime. The
8285 // flag is set to true if any array section has a stride not equal to 1, or
8286 // if the stride is not a constant expression (conservatively assumed
8287 // non-contiguous).
8288 bool IsNonContiguous =
8289 CombinedInfo.NonContigInfo.IsNonContiguous ||
8290 any_of(Components, [&](const auto &Component) {
8291 const auto *OASE =
8292 dyn_cast<ArraySectionExpr>(Component.getAssociatedExpression());
8293 if (!OASE)
8294 return false;
8295
8296 const Expr *StrideExpr = OASE->getStride();
8297 if (!StrideExpr)
8298 return false;
8299
8300 assert(StrideExpr->getType()->isIntegerType() &&
8301 "Stride expression must be of integer type");
8302
8303 // If stride is not evaluatable as a constant, treat as
8304 // non-contiguous.
8305 const auto Constant =
8306 StrideExpr->getIntegerConstantExpr(CGF.getContext());
8307 if (!Constant)
8308 return true;
8309
8310 // Treat non-unitary strides as non-contiguous.
8311 return !Constant->isOne();
8312 });
8313
8314 bool IsPrevMemberReference = false;
8315
8316 bool IsPartialMapped =
8317 !PartialStruct.PreliminaryMapData.BasePointers.empty();
8318
8319 // We need to check if we will be encountering any MEs. If we do not
8320 // encounter any ME expression it means we will be mapping the whole struct.
8321 // In that case we need to skip adding an entry for the struct to the
8322 // CombinedInfo list and instead add an entry to the StructBaseCombinedInfo
8323 // list only when generating all info for clauses.
8324 bool IsMappingWholeStruct = true;
8325 if (!GenerateAllInfoForClauses) {
8326 IsMappingWholeStruct = false;
8327 } else {
8328 for (auto TempI = I; TempI != CE; ++TempI) {
8329 const MemberExpr *PossibleME =
8330 dyn_cast<MemberExpr>(TempI->getAssociatedExpression());
8331 if (PossibleME) {
8332 IsMappingWholeStruct = false;
8333 break;
8334 }
8335 }
8336 }
8337
8338 bool SeenFirstNonBinOpExprAfterAttachPtr = false;
8339 for (; I != CE; ++I) {
8340 // If we have a valid attach-ptr, we skip processing all components until
8341 // after the attach-ptr.
8342 if (HasAttachPtr && !SeenAttachPtr) {
8343 SeenAttachPtr = I->getAssociatedExpression() == AttachPtrExpr;
8344 continue;
8345 }
8346
8347 // After finding the attach pointer, skip binary-ops, to skip past
8348 // expressions like (p + 10), for a map like map(*(p + 10)), where p is
8349 // the attach-ptr.
8350 if (HasAttachPtr && !SeenFirstNonBinOpExprAfterAttachPtr) {
8351 const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression());
8352 if (BO)
8353 continue;
8354
8355 // Found the first non-binary-operator component after attach
8356 SeenFirstNonBinOpExprAfterAttachPtr = true;
8357 BP = AttachPteeBaseAddr;
8358 }
8359
8360 // If the current component is member of a struct (parent struct) mark it.
8361 if (!EncounteredME) {
8362 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression());
8363 // If we encounter a PTR_AND_OBJ entry from now on it should be marked
8364 // as MEMBER_OF the parent struct.
8365 if (EncounteredME) {
8366 ShouldBeMemberOf = true;
8367 // Do not emit as complex pointer if this is actually not array-like
8368 // expression.
8369 if (FirstPointerInComplexData) {
8370 QualType Ty = std::prev(I)
8371 ->getAssociatedDeclaration()
8372 ->getType()
8373 .getNonReferenceType();
8374 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
8375 FirstPointerInComplexData = false;
8376 }
8377 }
8378 }
8379
8380 auto Next = std::next(I);
8381
8382 // We need to generate the addresses and sizes if this is the last
8383 // component, if the component is a pointer or if it is an array section
8384 // whose length can't be proved to be one. If this is a pointer, it
8385 // becomes the base address for the following components.
8386
8387 // A final array section, is one whose length can't be proved to be one.
8388 // If the map item is non-contiguous then we don't treat any array section
8389 // as final array section.
8390 bool IsFinalArraySection =
8391 !IsNonContiguous &&
8392 isFinalArraySectionExpression(I->getAssociatedExpression());
8393
8394 // If we have a declaration for the mapping use that, otherwise use
8395 // the base declaration of the map clause.
8396 const ValueDecl *MapDecl = (I->getAssociatedDeclaration())
8397 ? I->getAssociatedDeclaration()
8398 : BaseDecl;
8399 MapExpr = (I->getAssociatedExpression()) ? I->getAssociatedExpression()
8400 : MapExpr;
8401
8402 // Get information on whether the element is a pointer. Have to do a
8403 // special treatment for array sections given that they are built-in
8404 // types.
8405 const auto *OASE =
8406 dyn_cast<ArraySectionExpr>(I->getAssociatedExpression());
8407 const auto *OAShE =
8408 dyn_cast<OMPArrayShapingExpr>(I->getAssociatedExpression());
8409 const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression());
8410 const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression());
8411 bool IsPointer =
8412 OAShE ||
8415 ->isAnyPointerType()) ||
8416 I->getAssociatedExpression()->getType()->isAnyPointerType();
8417 bool IsMemberReference = isa<MemberExpr>(I->getAssociatedExpression()) &&
8418 MapDecl &&
8419 MapDecl->getType()->isLValueReferenceType();
8420 bool IsNonDerefPointer = IsPointer &&
8421 !(UO && UO->getOpcode() != UO_Deref) && !BO &&
8422 !IsNonContiguous;
8423
8424 if (OASE)
8425 ++DimSize;
8426
8427 if (Next == CE || IsMemberReference || IsNonDerefPointer ||
8428 IsFinalArraySection) {
8429 // If this is not the last component, we expect the pointer to be
8430 // associated with an array expression or member expression.
8431 assert((Next == CE ||
8432 isa<MemberExpr>(Next->getAssociatedExpression()) ||
8433 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) ||
8434 isa<ArraySectionExpr>(Next->getAssociatedExpression()) ||
8435 isa<OMPArrayShapingExpr>(Next->getAssociatedExpression()) ||
8436 isa<UnaryOperator>(Next->getAssociatedExpression()) ||
8437 isa<BinaryOperator>(Next->getAssociatedExpression())) &&
8438 "Unexpected expression");
8439
8441 Address LowestElem = Address::invalid();
8442 auto &&EmitMemberExprBase = [](CodeGenFunction &CGF,
8443 const MemberExpr *E) {
8444 const Expr *BaseExpr = E->getBase();
8445 // If this is s.x, emit s as an lvalue. If it is s->x, emit s as a
8446 // scalar.
8447 LValue BaseLV;
8448 if (E->isArrow()) {
8449 LValueBaseInfo BaseInfo;
8450 TBAAAccessInfo TBAAInfo;
8451 Address Addr =
8452 CGF.EmitPointerWithAlignment(BaseExpr, &BaseInfo, &TBAAInfo);
8453 QualType PtrTy = BaseExpr->getType()->getPointeeType();
8454 BaseLV = CGF.MakeAddrLValue(Addr, PtrTy, BaseInfo, TBAAInfo);
8455 } else {
8456 BaseLV = CGF.EmitOMPSharedLValue(BaseExpr);
8457 }
8458 return BaseLV;
8459 };
8460 if (OAShE) {
8461 LowestElem = LB =
8462 Address(CGF.EmitScalarExpr(OAShE->getBase()),
8464 OAShE->getBase()->getType()->getPointeeType()),
8466 OAShE->getBase()->getType()));
8467 } else if (IsMemberReference) {
8468 const auto *ME = cast<MemberExpr>(I->getAssociatedExpression());
8469 LValue BaseLVal = EmitMemberExprBase(CGF, ME);
8470 LowestElem = CGF.EmitLValueForFieldInitialization(
8471 BaseLVal, cast<FieldDecl>(MapDecl))
8472 .getAddress();
8473 LB = CGF.EmitLoadOfReferenceLValue(LowestElem, MapDecl->getType())
8474 .getAddress();
8475 } else {
8476 LowestElem = LB =
8477 CGF.EmitOMPSharedLValue(I->getAssociatedExpression())
8478 .getAddress();
8479 }
8480
8481 // Save the final LowestElem, to use it as the pointee in attach maps,
8482 // if emitted.
8483 if (Next == CE)
8484 FinalLowestElem = LowestElem;
8485
8486 // If this component is a pointer inside the base struct then we don't
8487 // need to create any entry for it - it will be combined with the object
8488 // it is pointing to into a single PTR_AND_OBJ entry.
8489 bool IsMemberPointerOrAddr =
8490 EncounteredME &&
8491 (((IsPointer || ForDeviceAddr) &&
8492 I->getAssociatedExpression() == EncounteredME) ||
8493 (IsPrevMemberReference && !IsPointer) ||
8494 (IsMemberReference && Next != CE &&
8495 !Next->getAssociatedExpression()->getType()->isPointerType()));
8496 if (!OverlappedElements.empty() && Next == CE) {
8497 // Handle base element with the info for overlapped elements.
8498 assert(!PartialStruct.Base.isValid() && "The base element is set.");
8499 assert(!IsPointer &&
8500 "Unexpected base element with the pointer type.");
8501 // Mark the whole struct as the struct that requires allocation on the
8502 // device.
8503 PartialStruct.LowestElem = {0, LowestElem};
8504 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars(
8505 I->getAssociatedExpression()->getType());
8508 LowestElem, CGF.VoidPtrTy, CGF.Int8Ty),
8509 TypeSize.getQuantity() - 1);
8510 PartialStruct.HighestElem = {
8511 std::numeric_limits<decltype(
8512 PartialStruct.HighestElem.first)>::max(),
8513 HB};
8514 PartialStruct.Base = BP;
8515 PartialStruct.LB = LB;
8516 assert(
8517 PartialStruct.PreliminaryMapData.BasePointers.empty() &&
8518 "Overlapped elements must be used only once for the variable.");
8519 std::swap(PartialStruct.PreliminaryMapData, CombinedInfo);
8520 // Emit data for non-overlapped data.
8521 OpenMPOffloadMappingFlags Flags =
8522 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
8523 getMapTypeBits(MapType, MapModifiers, MotionModifiers, IsImplicit,
8524 /*AddPtrFlag=*/false,
8525 /*AddIsTargetParamFlag=*/false, IsNonContiguous);
8526 CopyOverlappedEntryGaps CopyGaps(CGF, CombinedInfo, Flags, MapDecl,
8527 MapExpr, BP, LB, IsNonContiguous,
8528 DimSize);
8529 // Do bitcopy of all non-overlapped structure elements.
8531 Component : OverlappedElements) {
8532 for (const OMPClauseMappableExprCommon::MappableComponent &MC :
8533 Component) {
8534 if (const ValueDecl *VD = MC.getAssociatedDeclaration()) {
8535 if (const auto *FD = dyn_cast<FieldDecl>(VD)) {
8536 CopyGaps.processField(MC, FD, EmitMemberExprBase);
8537 }
8538 }
8539 }
8540 }
8541 CopyGaps.copyUntilEnd(HB);
8542 break;
8543 }
8544 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression());
8545 // Skip adding an entry in the CurInfo of this combined entry if the
8546 // whole struct is currently being mapped. The struct needs to be added
8547 // in the first position before any data internal to the struct is being
8548 // mapped.
8549 // Skip adding an entry in the CurInfo of this combined entry if the
8550 // PartialStruct.PreliminaryMapData.BasePointers has been mapped.
8551 if ((!IsMemberPointerOrAddr && !IsPartialMapped) ||
8552 (Next == CE && MapType != OMPC_MAP_unknown)) {
8553 if (!IsMappingWholeStruct) {
8554 CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr);
8555 CombinedInfo.BasePointers.push_back(BP.emitRawPointer(CGF));
8556 CombinedInfo.DevicePtrDecls.push_back(nullptr);
8557 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
8558 CombinedInfo.Pointers.push_back(LB.emitRawPointer(CGF));
8559 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
8560 Size, CGF.Int64Ty, /*isSigned=*/true));
8561 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize
8562 : 1);
8563 } else {
8564 StructBaseCombinedInfo.Exprs.emplace_back(MapDecl, MapExpr);
8565 StructBaseCombinedInfo.BasePointers.push_back(
8566 BP.emitRawPointer(CGF));
8567 StructBaseCombinedInfo.DevicePtrDecls.push_back(nullptr);
8568 StructBaseCombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
8569 StructBaseCombinedInfo.Pointers.push_back(LB.emitRawPointer(CGF));
8570 StructBaseCombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
8571 Size, CGF.Int64Ty, /*isSigned=*/true));
8572 StructBaseCombinedInfo.NonContigInfo.Dims.push_back(
8573 IsNonContiguous ? DimSize : 1);
8574 }
8575
8576 // If Mapper is valid, the last component inherits the mapper.
8577 bool HasMapper = Mapper && Next == CE;
8578 if (!IsMappingWholeStruct)
8579 CombinedInfo.Mappers.push_back(HasMapper ? Mapper : nullptr);
8580 else
8581 StructBaseCombinedInfo.Mappers.push_back(HasMapper ? Mapper
8582 : nullptr);
8583
8584 // We need to add a pointer flag for each map that comes from the
8585 // same expression except for the first one. We also need to signal
8586 // this map is the first one that relates with the current capture
8587 // (there is a set of entries for each capture).
8588 OpenMPOffloadMappingFlags Flags = getMapTypeBits(
8589 MapType, MapModifiers, MotionModifiers, IsImplicit,
8590 !IsExpressionFirstInfo || RequiresReference ||
8591 FirstPointerInComplexData || IsMemberReference,
8592 IsCaptureFirstInfo && !RequiresReference, IsNonContiguous);
8593
8594 if (!IsExpressionFirstInfo || IsMemberReference) {
8595 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well,
8596 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags.
8597 if (IsPointer || (IsMemberReference && Next != CE))
8598 Flags &= ~(OpenMPOffloadMappingFlags::OMP_MAP_TO |
8599 OpenMPOffloadMappingFlags::OMP_MAP_FROM |
8600 OpenMPOffloadMappingFlags::OMP_MAP_ALWAYS |
8601 OpenMPOffloadMappingFlags::OMP_MAP_DELETE |
8602 OpenMPOffloadMappingFlags::OMP_MAP_CLOSE);
8603
8604 if (ShouldBeMemberOf) {
8605 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag
8606 // should be later updated with the correct value of MEMBER_OF.
8607 Flags |= OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF;
8608 // From now on, all subsequent PTR_AND_OBJ entries should not be
8609 // marked as MEMBER_OF.
8610 ShouldBeMemberOf = false;
8611 }
8612 }
8613
8614 if (!IsMappingWholeStruct) {
8615 CombinedInfo.Types.push_back(Flags);
8616 // HasAttachPtr marks pointee entries, which have a base attach-ptr.
8617 CombinedInfo.HasAttachPtr.push_back(HasAttachPtr);
8618 } else {
8619 StructBaseCombinedInfo.Types.push_back(Flags);
8620 StructBaseCombinedInfo.HasAttachPtr.push_back(HasAttachPtr);
8621 }
8622 }
8623
8624 // If we have encountered a member expression so far, keep track of the
8625 // mapped member. If the parent is "*this", then the value declaration
8626 // is nullptr.
8627 if (EncounteredME) {
8628 const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl());
8629 unsigned FieldIndex = FD->getFieldIndex();
8630
8631 // Update info about the lowest and highest elements for this struct
8632 if (!PartialStruct.Base.isValid()) {
8633 PartialStruct.LowestElem = {FieldIndex, LowestElem};
8634 if (IsFinalArraySection && OASE) {
8635 Address HB =
8636 CGF.EmitArraySectionExpr(OASE, /*IsLowerBound=*/false)
8637 .getAddress();
8638 PartialStruct.HighestElem = {FieldIndex, HB};
8639 } else {
8640 PartialStruct.HighestElem = {FieldIndex, LowestElem};
8641 }
8642 PartialStruct.Base = BP;
8643 PartialStruct.LB = BP;
8644 } else if (FieldIndex < PartialStruct.LowestElem.first) {
8645 PartialStruct.LowestElem = {FieldIndex, LowestElem};
8646 } else if (FieldIndex > PartialStruct.HighestElem.first) {
8647 if (IsFinalArraySection && OASE) {
8648 Address HB =
8649 CGF.EmitArraySectionExpr(OASE, /*IsLowerBound=*/false)
8650 .getAddress();
8651 PartialStruct.HighestElem = {FieldIndex, HB};
8652 } else {
8653 PartialStruct.HighestElem = {FieldIndex, LowestElem};
8654 }
8655 }
8656 }
8657
8658 // Need to emit combined struct for array sections.
8659 if (IsFinalArraySection || IsNonContiguous)
8660 PartialStruct.IsArraySection = true;
8661
8662 // If we have a final array section, we are done with this expression.
8663 if (IsFinalArraySection)
8664 break;
8665
8666 // The pointer becomes the base for the next element.
8667 if (Next != CE)
8668 BP = IsMemberReference ? LowestElem : LB;
8669 if (!IsPartialMapped)
8670 IsExpressionFirstInfo = false;
8671 IsCaptureFirstInfo = false;
8672 FirstPointerInComplexData = false;
8673 IsPrevMemberReference = IsMemberReference;
8674 } else if (FirstPointerInComplexData) {
8675 QualType Ty = Components.rbegin()
8676 ->getAssociatedDeclaration()
8677 ->getType()
8678 .getNonReferenceType();
8679 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
8680 FirstPointerInComplexData = false;
8681 }
8682 }
8683 // If ran into the whole component - allocate the space for the whole
8684 // record.
8685 if (!EncounteredME)
8686 PartialStruct.HasCompleteRecord = true;
8687
8688 // Populate ATTACH information for later processing by emitAttachEntry.
8689 if (shouldEmitAttachEntry(AttachPtrExpr, BaseDecl, CGF, CurDir)) {
8690 AttachInfo.AttachPtrAddr = AttachPtrAddr;
8691 AttachInfo.AttachPteeAddr = FinalLowestElem;
8692 AttachInfo.AttachPtrDecl = BaseDecl;
8693 AttachInfo.AttachMapExpr = MapExpr;
8694 }
8695
8696 if (!IsNonContiguous)
8697 return;
8698
8699 const ASTContext &Context = CGF.getContext();
8700
8701 // For supporting stride in array section, we need to initialize the first
8702 // dimension size as 1, first offset as 0, and first count as 1
8703 MapValuesArrayTy CurOffsets = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 0)};
8704 MapValuesArrayTy CurCounts;
8705 MapValuesArrayTy CurStrides = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)};
8706 MapValuesArrayTy DimSizes{llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)};
8707 uint64_t ElementTypeSize;
8708
8709 // Collect Size information for each dimension and get the element size as
8710 // the first Stride. For example, for `int arr[10][10]`, the DimSizes
8711 // should be [10, 10] and the first stride is 4 btyes.
8712 for (const OMPClauseMappableExprCommon::MappableComponent &Component :
8713 Components) {
8714 const Expr *AssocExpr = Component.getAssociatedExpression();
8715 const auto *OASE = dyn_cast<ArraySectionExpr>(AssocExpr);
8716
8717 if (!OASE)
8718 continue;
8719
8720 QualType Ty = ArraySectionExpr::getBaseOriginalType(OASE->getBase());
8721 auto *CAT = Context.getAsConstantArrayType(Ty);
8722 auto *VAT = Context.getAsVariableArrayType(Ty);
8723
8724 // We need all the dimension size except for the last dimension.
8725 assert((VAT || CAT || &Component == &*Components.begin()) &&
8726 "Should be either ConstantArray or VariableArray if not the "
8727 "first Component");
8728
8729 // Get element size if CurCounts is empty.
8730 if (CurCounts.empty()) {
8731 const Type *ElementType = nullptr;
8732 if (CAT)
8733 ElementType = CAT->getElementType().getTypePtr();
8734 else if (VAT)
8735 ElementType = VAT->getElementType().getTypePtr();
8736 else if (&Component == &*Components.begin()) {
8737 // If the base is a raw pointer (e.g. T *data with data[a:b:c]),
8738 // there was no earlier CAT/VAT/array handling to establish
8739 // ElementType. Capture the pointee type now so that subsequent
8740 // components (offset/length/stride) have a concrete element type to
8741 // work with. This makes pointer-backed sections behave consistently
8742 // with CAT/VAT/array bases.
8743 if (const auto *PtrType = Ty->getAs<PointerType>())
8744 ElementType = PtrType->getPointeeType().getTypePtr();
8745 } else {
8746 // Any component after the first should never have a raw pointer type;
8747 // by this point. ElementType must already be known (set above or in
8748 // prior array / CAT / VAT handling).
8749 assert(!Ty->isPointerType() &&
8750 "Non-first components should not be raw pointers");
8751 }
8752
8753 // At this stage, if ElementType was a base pointer and we are in the
8754 // first iteration, it has been computed.
8755 if (ElementType) {
8756 // For the case that having pointer as base, we need to remove one
8757 // level of indirection.
8758 if (&Component != &*Components.begin())
8759 ElementType = ElementType->getPointeeOrArrayElementType();
8760 ElementTypeSize =
8761 Context.getTypeSizeInChars(ElementType).getQuantity();
8762 CurCounts.push_back(
8763 llvm::ConstantInt::get(CGF.Int64Ty, ElementTypeSize));
8764 }
8765 }
8766 // Get dimension value except for the last dimension since we don't need
8767 // it.
8768 if (DimSizes.size() < Components.size() - 1) {
8769 if (CAT)
8770 DimSizes.push_back(
8771 llvm::ConstantInt::get(CGF.Int64Ty, CAT->getZExtSize()));
8772 else if (VAT)
8773 DimSizes.push_back(CGF.Builder.CreateIntCast(
8774 CGF.EmitScalarExpr(VAT->getSizeExpr()), CGF.Int64Ty,
8775 /*IsSigned=*/false));
8776 }
8777 }
8778
8779 // Skip the dummy dimension since we have already have its information.
8780 auto *DI = DimSizes.begin() + 1;
8781 // Product of dimension.
8782 llvm::Value *DimProd =
8783 llvm::ConstantInt::get(CGF.CGM.Int64Ty, ElementTypeSize);
8784
8785 // Collect info for non-contiguous. Notice that offset, count, and stride
8786 // are only meaningful for array-section, so we insert a null for anything
8787 // other than array-section.
8788 // Also, the size of offset, count, and stride are not the same as
8789 // pointers, base_pointers, sizes, or dims. Instead, the size of offset,
8790 // count, and stride are the same as the number of non-contiguous
8791 // declaration in target update to/from clause.
8792 for (const OMPClauseMappableExprCommon::MappableComponent &Component :
8793 Components) {
8794 const Expr *AssocExpr = Component.getAssociatedExpression();
8795
8796 if (const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr)) {
8797 llvm::Value *Offset = CGF.Builder.CreateIntCast(
8798 CGF.EmitScalarExpr(AE->getIdx()), CGF.Int64Ty,
8799 /*isSigned=*/false);
8800 CurOffsets.push_back(Offset);
8801 CurCounts.push_back(llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/1));
8802 CurStrides.push_back(CurStrides.back());
8803 continue;
8804 }
8805
8806 const auto *OASE = dyn_cast<ArraySectionExpr>(AssocExpr);
8807
8808 if (!OASE)
8809 continue;
8810
8811 // Offset
8812 const Expr *OffsetExpr = OASE->getLowerBound();
8813 llvm::Value *Offset = nullptr;
8814 if (!OffsetExpr) {
8815 // If offset is absent, then we just set it to zero.
8816 Offset = llvm::ConstantInt::get(CGF.Int64Ty, 0);
8817 } else {
8818 Offset = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(OffsetExpr),
8819 CGF.Int64Ty,
8820 /*isSigned=*/false);
8821 }
8822
8823 // Count
8824 const Expr *CountExpr = OASE->getLength();
8825 llvm::Value *Count = nullptr;
8826 if (!CountExpr) {
8827 // In Clang, once a high dimension is an array section, we construct all
8828 // the lower dimension as array section, however, for case like
8829 // arr[0:2][2], Clang construct the inner dimension as an array section
8830 // but it actually is not in an array section form according to spec.
8831 if (!OASE->getColonLocFirst().isValid() &&
8832 !OASE->getColonLocSecond().isValid()) {
8833 Count = llvm::ConstantInt::get(CGF.Int64Ty, 1);
8834 } else {
8835 // OpenMP 5.0, 2.1.5 Array Sections, Description.
8836 // When the length is absent it defaults to ⌈(size −
8837 // lower-bound)/stride⌉, where size is the size of the array
8838 // dimension.
8839 const Expr *StrideExpr = OASE->getStride();
8840 llvm::Value *Stride =
8841 StrideExpr
8842 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr),
8843 CGF.Int64Ty, /*isSigned=*/false)
8844 : nullptr;
8845 if (Stride)
8846 Count = CGF.Builder.CreateUDiv(
8847 CGF.Builder.CreateNUWSub(*DI, Offset), Stride);
8848 else
8849 Count = CGF.Builder.CreateNUWSub(*DI, Offset);
8850 }
8851 } else {
8852 Count = CGF.EmitScalarExpr(CountExpr);
8853 }
8854 Count = CGF.Builder.CreateIntCast(Count, CGF.Int64Ty, /*isSigned=*/false);
8855 CurCounts.push_back(Count);
8856
8857 // Stride_n' = Stride_n * (D_0 * D_1 ... * D_n-1) * Unit size
8858 // Offset_n' = Offset_n * (D_0 * D_1 ... * D_n-1) * Unit size
8859 // Take `int arr[5][5][5]` and `arr[0:2:2][1:2:1][0:2:2]` as an example:
8860 // Offset Count Stride
8861 // D0 0 4 1 (int) <- dummy dimension
8862 // D1 0 2 8 (2 * (1) * 4)
8863 // D2 100 2 20 (1 * (1 * 5) * 4)
8864 // D3 0 2 200 (2 * (1 * 5 * 4) * 4)
8865 const Expr *StrideExpr = OASE->getStride();
8866 llvm::Value *Stride =
8867 StrideExpr
8868 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr),
8869 CGF.Int64Ty, /*isSigned=*/false)
8870 : nullptr;
8871 DimProd = CGF.Builder.CreateNUWMul(DimProd, *(DI - 1));
8872 if (Stride)
8873 CurStrides.push_back(CGF.Builder.CreateNUWMul(DimProd, Stride));
8874 else
8875 CurStrides.push_back(DimProd);
8876
8877 Offset = CGF.Builder.CreateNUWMul(DimProd, Offset);
8878 CurOffsets.push_back(Offset);
8879
8880 if (DI != DimSizes.end())
8881 ++DI;
8882 }
8883
8884 CombinedInfo.NonContigInfo.Offsets.push_back(CurOffsets);
8885 CombinedInfo.NonContigInfo.Counts.push_back(CurCounts);
8886 CombinedInfo.NonContigInfo.Strides.push_back(CurStrides);
8887 }
8888
8889 /// Return the adjusted map modifiers if the declaration a capture refers to
8890 /// appears in a first-private clause. This is expected to be used only with
8891 /// directives that start with 'target'.
8892 OpenMPOffloadMappingFlags
8893 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const {
8894 assert(Cap.capturesVariable() && "Expected capture by reference only!");
8895
8896 // A first private variable captured by reference will use only the
8897 // 'private ptr' and 'map to' flag. Return the right flags if the captured
8898 // declaration is known as first-private in this handler.
8899 if (FirstPrivateDecls.count(Cap.getCapturedVar())) {
8900 if (Cap.getCapturedVar()->getType()->isAnyPointerType())
8901 return OpenMPOffloadMappingFlags::OMP_MAP_TO |
8902 OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ;
8903 return OpenMPOffloadMappingFlags::OMP_MAP_PRIVATE |
8904 OpenMPOffloadMappingFlags::OMP_MAP_TO;
8905 }
8906 auto I = LambdasMap.find(Cap.getCapturedVar()->getCanonicalDecl());
8907 if (I != LambdasMap.end())
8908 // for map(to: lambda): using user specified map type.
8909 return getMapTypeBits(
8910 I->getSecond()->getMapType(), I->getSecond()->getMapTypeModifiers(),
8911 /*MotionModifiers=*/{}, I->getSecond()->isImplicit(),
8912 /*AddPtrFlag=*/false,
8913 /*AddIsTargetParamFlag=*/false,
8914 /*isNonContiguous=*/false);
8915 return OpenMPOffloadMappingFlags::OMP_MAP_TO |
8916 OpenMPOffloadMappingFlags::OMP_MAP_FROM;
8917 }
8918
8919 void getPlainLayout(const CXXRecordDecl *RD,
8920 llvm::SmallVectorImpl<const FieldDecl *> &Layout,
8921 bool AsBase) const {
8922 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD);
8923
8924 llvm::StructType *St =
8925 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType();
8926
8927 unsigned NumElements = St->getNumElements();
8928 llvm::SmallVector<
8929 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4>
8930 RecordLayout(NumElements);
8931
8932 // Fill bases.
8933 for (const auto &I : RD->bases()) {
8934 if (I.isVirtual())
8935 continue;
8936
8937 QualType BaseTy = I.getType();
8938 const auto *Base = BaseTy->getAsCXXRecordDecl();
8939 // Ignore empty bases.
8940 if (isEmptyRecordForLayout(CGF.getContext(), BaseTy) ||
8941 CGF.getContext()
8942 .getASTRecordLayout(Base)
8944 .isZero())
8945 continue;
8946
8947 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base);
8948 RecordLayout[FieldIndex] = Base;
8949 }
8950 // Fill in virtual bases.
8951 for (const auto &I : RD->vbases()) {
8952 QualType BaseTy = I.getType();
8953 // Ignore empty bases.
8954 if (isEmptyRecordForLayout(CGF.getContext(), BaseTy))
8955 continue;
8956
8957 const auto *Base = BaseTy->getAsCXXRecordDecl();
8958 unsigned FieldIndex = RL.getVirtualBaseIndex(Base);
8959 if (RecordLayout[FieldIndex])
8960 continue;
8961 RecordLayout[FieldIndex] = Base;
8962 }
8963 // Fill in all the fields.
8964 assert(!RD->isUnion() && "Unexpected union.");
8965 for (const auto *Field : RD->fields()) {
8966 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we
8967 // will fill in later.)
8968 if (!Field->isBitField() &&
8969 !isEmptyFieldForLayout(CGF.getContext(), Field)) {
8970 unsigned FieldIndex = RL.getLLVMFieldNo(Field);
8971 RecordLayout[FieldIndex] = Field;
8972 }
8973 }
8974 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>
8975 &Data : RecordLayout) {
8976 if (Data.isNull())
8977 continue;
8978 if (const auto *Base = dyn_cast<const CXXRecordDecl *>(Data))
8979 getPlainLayout(Base, Layout, /*AsBase=*/true);
8980 else
8981 Layout.push_back(cast<const FieldDecl *>(Data));
8982 }
8983 }
8984
8985 /// Returns the address corresponding to \p PointerExpr.
8986 static Address getAttachPtrAddr(const Expr *PointerExpr,
8987 CodeGenFunction &CGF) {
8988 assert(PointerExpr && "Cannot get addr from null attach-ptr expr");
8989 Address AttachPtrAddr = Address::invalid();
8990
8991 if (auto *DRE = dyn_cast<DeclRefExpr>(PointerExpr)) {
8992 // If the pointer is a variable, we can use its address directly.
8993 AttachPtrAddr = CGF.EmitLValue(DRE).getAddress();
8994 } else if (auto *OASE = dyn_cast<ArraySectionExpr>(PointerExpr)) {
8995 AttachPtrAddr =
8996 CGF.EmitArraySectionExpr(OASE, /*IsLowerBound=*/true).getAddress();
8997 } else if (auto *ASE = dyn_cast<ArraySubscriptExpr>(PointerExpr)) {
8998 AttachPtrAddr = CGF.EmitLValue(ASE).getAddress();
8999 } else if (auto *ME = dyn_cast<MemberExpr>(PointerExpr)) {
9000 AttachPtrAddr = CGF.EmitMemberExpr(ME).getAddress();
9001 } else if (auto *UO = dyn_cast<UnaryOperator>(PointerExpr)) {
9002 assert(UO->getOpcode() == UO_Deref &&
9003 "Unexpected unary-operator on attach-ptr-expr");
9004 AttachPtrAddr = CGF.EmitLValue(UO).getAddress();
9005 }
9006 assert(AttachPtrAddr.isValid() &&
9007 "Failed to get address for attach pointer expression");
9008 return AttachPtrAddr;
9009 }
9010
9011 /// Get the address of the attach pointer, and a load from it, to get the
9012 /// pointee base address.
9013 /// \return A pair containing AttachPtrAddr and AttachPteeBaseAddr. The pair
9014 /// contains invalid addresses if \p AttachPtrExpr is null.
9015 static std::pair<Address, Address>
9016 getAttachPtrAddrAndPteeBaseAddr(const Expr *AttachPtrExpr,
9017 CodeGenFunction &CGF) {
9018
9019 if (!AttachPtrExpr)
9020 return {Address::invalid(), Address::invalid()};
9021
9022 Address AttachPtrAddr = getAttachPtrAddr(AttachPtrExpr, CGF);
9023 assert(AttachPtrAddr.isValid() && "Invalid attach pointer addr");
9024
9025 QualType AttachPtrType =
9028
9029 Address AttachPteeBaseAddr = CGF.EmitLoadOfPointer(
9030 AttachPtrAddr, AttachPtrType->castAs<PointerType>());
9031 assert(AttachPteeBaseAddr.isValid() && "Invalid attach pointee base addr");
9032
9033 return {AttachPtrAddr, AttachPteeBaseAddr};
9034 }
9035
9036 /// Returns whether an attach entry should be emitted for a map on
9037 /// \p MapBaseDecl on the directive \p CurDir.
9038 static bool
9039 shouldEmitAttachEntry(const Expr *PointerExpr, const ValueDecl *MapBaseDecl,
9040 CodeGenFunction &CGF,
9041 llvm::PointerUnion<const OMPExecutableDirective *,
9042 const OMPDeclareMapperDecl *>
9043 CurDir) {
9044 if (!PointerExpr)
9045 return false;
9046
9047 // Pointer attachment is needed at map-entering time or for declare
9048 // mappers.
9049 return isa<const OMPDeclareMapperDecl *>(CurDir) ||
9052 ->getDirectiveKind());
9053 }
9054
9055 /// Computes the attach-ptr expr for \p Components, and updates various maps
9056 /// with the information.
9057 /// It internally calls OMPClauseMappableExprCommon::findAttachPtrExpr()
9058 /// with the OpenMPDirectiveKind extracted from \p CurDir.
9059 /// It updates AttachPtrComputationOrderMap, AttachPtrComponentDepthMap, and
9060 /// AttachPtrExprMap.
9061 void collectAttachPtrExprInfo(
9063 llvm::PointerUnion<const OMPExecutableDirective *,
9064 const OMPDeclareMapperDecl *>
9065 CurDir) {
9066
9067 OpenMPDirectiveKind CurDirectiveID =
9069 ? OMPD_declare_mapper
9070 : cast<const OMPExecutableDirective *>(CurDir)->getDirectiveKind();
9071
9072 const auto &[AttachPtrExpr, Depth] =
9074 CurDirectiveID);
9075
9076 AttachPtrComputationOrderMap.try_emplace(
9077 AttachPtrExpr, AttachPtrComputationOrderMap.size());
9078 AttachPtrComponentDepthMap.try_emplace(AttachPtrExpr, Depth);
9079 AttachPtrExprMap.try_emplace(Components, AttachPtrExpr);
9080 }
9081
9082 /// Generate all the base pointers, section pointers, sizes, map types, and
9083 /// mappers for the extracted mappable expressions (all included in \a
9084 /// CombinedInfo). Also, for each item that relates with a device pointer, a
9085 /// pair of the relevant declaration and index where it occurs is appended to
9086 /// the device pointers info array.
9087 void generateAllInfoForClauses(
9088 ArrayRef<const OMPClause *> Clauses, MapCombinedInfoTy &CombinedInfo,
9089 llvm::OpenMPIRBuilder &OMPBuilder,
9090 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet =
9091 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const {
9092 // We have to process the component lists that relate with the same
9093 // declaration in a single chunk so that we can generate the map flags
9094 // correctly. Therefore, we organize all lists in a map.
9095 enum MapKind { Present, Allocs, Other, Total };
9096 llvm::MapVector<CanonicalDeclPtr<const Decl>,
9097 SmallVector<SmallVector<MapInfo, 8>, 4>>
9098 Info;
9099
9100 // Helper function to fill the information map for the different supported
9101 // clauses.
9102 auto &&InfoGen =
9103 [&Info, &SkipVarSet](
9104 const ValueDecl *D, MapKind Kind,
9106 OpenMPMapClauseKind MapType,
9107 ArrayRef<OpenMPMapModifierKind> MapModifiers,
9108 ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
9109 bool ReturnDevicePointer, bool IsImplicit, const ValueDecl *Mapper,
9110 const Expr *VarRef = nullptr, bool ForDeviceAddr = false) {
9111 if (SkipVarSet.contains(D))
9112 return;
9113 auto It = Info.try_emplace(D, Total).first;
9114 It->second[Kind].emplace_back(
9115 L, MapType, MapModifiers, MotionModifiers, ReturnDevicePointer,
9116 IsImplicit, Mapper, VarRef, ForDeviceAddr);
9117 };
9118
9119 for (const auto *Cl : Clauses) {
9120 const auto *C = dyn_cast<OMPMapClause>(Cl);
9121 if (!C)
9122 continue;
9123 MapKind Kind = Other;
9124 if (llvm::is_contained(C->getMapTypeModifiers(),
9125 OMPC_MAP_MODIFIER_present))
9126 Kind = Present;
9127 else if (C->getMapType() == OMPC_MAP_alloc)
9128 Kind = Allocs;
9129 const auto *EI = C->getVarRefs().begin();
9130 for (const auto L : C->component_lists()) {
9131 const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr;
9132 InfoGen(std::get<0>(L), Kind, std::get<1>(L), C->getMapType(),
9133 C->getMapTypeModifiers(), {},
9134 /*ReturnDevicePointer=*/false, C->isImplicit(), std::get<2>(L),
9135 E);
9136 ++EI;
9137 }
9138 }
9139 for (const auto *Cl : Clauses) {
9140 const auto *C = dyn_cast<OMPToClause>(Cl);
9141 if (!C)
9142 continue;
9143 MapKind Kind = Other;
9144 if (llvm::is_contained(C->getMotionModifiers(),
9145 OMPC_MOTION_MODIFIER_present))
9146 Kind = Present;
9147 if (llvm::is_contained(C->getMotionModifiers(),
9148 OMPC_MOTION_MODIFIER_iterator)) {
9149 if (auto *IteratorExpr = dyn_cast<OMPIteratorExpr>(
9150 C->getIteratorModifier()->IgnoreParenImpCasts())) {
9151 const auto *VD = cast<VarDecl>(IteratorExpr->getIteratorDecl(0));
9152 CGF.EmitVarDecl(*VD);
9153 }
9154 }
9155
9156 const auto *EI = C->getVarRefs().begin();
9157 for (const auto L : C->component_lists()) {
9158 InfoGen(std::get<0>(L), Kind, std::get<1>(L), OMPC_MAP_to, {},
9159 C->getMotionModifiers(), /*ReturnDevicePointer=*/false,
9160 C->isImplicit(), std::get<2>(L), *EI);
9161 ++EI;
9162 }
9163 }
9164 for (const auto *Cl : Clauses) {
9165 const auto *C = dyn_cast<OMPFromClause>(Cl);
9166 if (!C)
9167 continue;
9168 MapKind Kind = Other;
9169 if (llvm::is_contained(C->getMotionModifiers(),
9170 OMPC_MOTION_MODIFIER_present))
9171 Kind = Present;
9172 if (llvm::is_contained(C->getMotionModifiers(),
9173 OMPC_MOTION_MODIFIER_iterator)) {
9174 if (auto *IteratorExpr = dyn_cast<OMPIteratorExpr>(
9175 C->getIteratorModifier()->IgnoreParenImpCasts())) {
9176 const auto *VD = cast<VarDecl>(IteratorExpr->getIteratorDecl(0));
9177 CGF.EmitVarDecl(*VD);
9178 }
9179 }
9180
9181 const auto *EI = C->getVarRefs().begin();
9182 for (const auto L : C->component_lists()) {
9183 InfoGen(std::get<0>(L), Kind, std::get<1>(L), OMPC_MAP_from, {},
9184 C->getMotionModifiers(),
9185 /*ReturnDevicePointer=*/false, C->isImplicit(), std::get<2>(L),
9186 *EI);
9187 ++EI;
9188 }
9189 }
9190
9191 // Look at the use_device_ptr and use_device_addr clauses information and
9192 // mark the existing map entries as such. If there is no map information for
9193 // an entry in the use_device_ptr and use_device_addr list, we create one
9194 // with map type 'return_param' and zero size section. It is the user's
9195 // fault if that was not mapped before. If there is no map information, then
9196 // we defer the emission of that entry until all the maps for the same VD
9197 // have been handled.
9198 MapCombinedInfoTy UseDeviceDataCombinedInfo;
9199
9200 auto &&UseDeviceDataCombinedInfoGen =
9201 [&UseDeviceDataCombinedInfo](const ValueDecl *VD, llvm::Value *Ptr,
9202 CodeGenFunction &CGF, bool IsDevAddr,
9203 bool HasUdpFbNullify = false) {
9204 UseDeviceDataCombinedInfo.Exprs.push_back(VD);
9205 UseDeviceDataCombinedInfo.BasePointers.emplace_back(Ptr);
9206 UseDeviceDataCombinedInfo.DevicePtrDecls.emplace_back(VD);
9207 UseDeviceDataCombinedInfo.DevicePointers.emplace_back(
9208 IsDevAddr ? DeviceInfoTy::Address : DeviceInfoTy::Pointer);
9209 // FIXME: For use_device_addr on array-sections, this should
9210 // be the starting address of the section.
9211 // e.g. int *p;
9212 // ... use_device_addr(p[3])
9213 // &p[0], &p[3], /*size=*/0, RETURN_PARAM
9214 UseDeviceDataCombinedInfo.Pointers.push_back(Ptr);
9215 UseDeviceDataCombinedInfo.Sizes.push_back(
9216 llvm::Constant::getNullValue(CGF.Int64Ty));
9217 OpenMPOffloadMappingFlags Flags =
9218 OpenMPOffloadMappingFlags::OMP_MAP_RETURN_PARAM;
9219 if (HasUdpFbNullify)
9220 Flags |= OpenMPOffloadMappingFlags::OMP_MAP_FB_NULLIFY;
9221 UseDeviceDataCombinedInfo.Types.push_back(Flags);
9222 UseDeviceDataCombinedInfo.HasAttachPtr.push_back(false);
9223 UseDeviceDataCombinedInfo.Mappers.push_back(nullptr);
9224 };
9225
9226 auto &&MapInfoGen =
9227 [&UseDeviceDataCombinedInfoGen](
9228 CodeGenFunction &CGF, const Expr *IE, const ValueDecl *VD,
9230 Components,
9231 bool IsDevAddr, bool IEIsAttachPtrForDevAddr = false,
9232 bool HasUdpFbNullify = false) {
9233 // We didn't find any match in our map information - generate a zero
9234 // size array section.
9235 llvm::Value *Ptr;
9236 if (IsDevAddr && !IEIsAttachPtrForDevAddr) {
9237 if (IE->isGLValue())
9238 Ptr = CGF.EmitLValue(IE).getPointer(CGF);
9239 else
9240 Ptr = CGF.EmitScalarExpr(IE);
9241 } else {
9242 Ptr = CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc());
9243 }
9244 bool TreatDevAddrAsDevPtr = IEIsAttachPtrForDevAddr;
9245 // For the purpose of address-translation, treat something like the
9246 // following:
9247 // int *p;
9248 // ... use_device_addr(p[1])
9249 // equivalent to
9250 // ... use_device_ptr(p)
9251 UseDeviceDataCombinedInfoGen(VD, Ptr, CGF, /*IsDevAddr=*/IsDevAddr &&
9252 !TreatDevAddrAsDevPtr,
9253 HasUdpFbNullify);
9254 };
9255
9256 auto &&IsMapInfoExist =
9257 [&Info, this](CodeGenFunction &CGF, const ValueDecl *VD, const Expr *IE,
9258 const Expr *DesiredAttachPtrExpr, bool IsDevAddr,
9259 bool HasUdpFbNullify = false) -> bool {
9260 // We potentially have map information for this declaration already.
9261 // Look for the first set of components that refer to it. If found,
9262 // return true.
9263 // If the first component is a member expression, we have to look into
9264 // 'this', which maps to null in the map of map information. Otherwise
9265 // look directly for the information.
9266 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD);
9267 if (It != Info.end()) {
9268 bool Found = false;
9269 for (auto &Data : It->second) {
9270 MapInfo *CI = nullptr;
9271 // We potentially have multiple maps for the same decl. We need to
9272 // only consider those for which the attach-ptr matches the desired
9273 // attach-ptr.
9274 auto *It = llvm::find_if(Data, [&](const MapInfo &MI) {
9275 if (MI.Components.back().getAssociatedDeclaration() != VD)
9276 return false;
9277
9278 const Expr *MapAttachPtr = getAttachPtrExpr(MI.Components);
9279 bool Match = AttachPtrComparator.areEqual(MapAttachPtr,
9280 DesiredAttachPtrExpr);
9281 return Match;
9282 });
9283
9284 if (It != Data.end())
9285 CI = &*It;
9286
9287 if (CI) {
9288 if (IsDevAddr) {
9289 CI->ForDeviceAddr = true;
9290 CI->ReturnDevicePointer = true;
9291 CI->HasUdpFbNullify = HasUdpFbNullify;
9292 Found = true;
9293 break;
9294 } else {
9295 auto PrevCI = std::next(CI->Components.rbegin());
9296 const auto *VarD = dyn_cast<VarDecl>(VD);
9297 const Expr *AttachPtrExpr = getAttachPtrExpr(CI->Components);
9298 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
9299 isa<MemberExpr>(IE) ||
9300 !VD->getType().getNonReferenceType()->isPointerType() ||
9301 PrevCI == CI->Components.rend() ||
9302 isa<MemberExpr>(PrevCI->getAssociatedExpression()) || !VarD ||
9303 VarD->hasLocalStorage() ||
9304 (isa_and_nonnull<DeclRefExpr>(AttachPtrExpr) &&
9305 VD == cast<DeclRefExpr>(AttachPtrExpr)->getDecl())) {
9306 CI->ForDeviceAddr = IsDevAddr;
9307 CI->ReturnDevicePointer = true;
9308 CI->HasUdpFbNullify = HasUdpFbNullify;
9309 Found = true;
9310 break;
9311 }
9312 }
9313 }
9314 }
9315 return Found;
9316 }
9317 return false;
9318 };
9319
9320 // Look at the use_device_ptr clause information and mark the existing map
9321 // entries as such. If there is no map information for an entry in the
9322 // use_device_ptr list, we create one with map type 'alloc' and zero size
9323 // section. It is the user fault if that was not mapped before. If there is
9324 // no map information and the pointer is a struct member, then we defer the
9325 // emission of that entry until the whole struct has been processed.
9326 for (const auto *Cl : Clauses) {
9327 const auto *C = dyn_cast<OMPUseDevicePtrClause>(Cl);
9328 if (!C)
9329 continue;
9330 bool HasUdpFbNullify =
9331 C->getFallbackModifier() == OMPC_USE_DEVICE_PTR_FALLBACK_fb_nullify;
9332 for (const auto L : C->component_lists()) {
9334 std::get<1>(L);
9335 assert(!Components.empty() &&
9336 "Not expecting empty list of components!");
9337 const ValueDecl *VD = Components.back().getAssociatedDeclaration();
9339 const Expr *IE = Components.back().getAssociatedExpression();
9340 // For use_device_ptr, we match an existing map clause if its attach-ptr
9341 // is same as the use_device_ptr operand. e.g.
9342 // map expr | use_device_ptr expr | current behavior
9343 // ---------|---------------------|-----------------
9344 // p[1] | p | match
9345 // ps->a | ps | match
9346 // p | p | no match
9347 const Expr *UDPOperandExpr =
9348 Components.front().getAssociatedExpression();
9349 if (IsMapInfoExist(CGF, VD, IE,
9350 /*DesiredAttachPtrExpr=*/UDPOperandExpr,
9351 /*IsDevAddr=*/false, HasUdpFbNullify))
9352 continue;
9353 MapInfoGen(CGF, IE, VD, Components, /*IsDevAddr=*/false,
9354 /*IEIsAttachPtrForDevAddr=*/false, HasUdpFbNullify);
9355 }
9356 }
9357
9358 llvm::SmallDenseSet<CanonicalDeclPtr<const Decl>, 4> Processed;
9359 for (const auto *Cl : Clauses) {
9360 const auto *C = dyn_cast<OMPUseDeviceAddrClause>(Cl);
9361 if (!C)
9362 continue;
9363 for (const auto L : C->component_lists()) {
9365 std::get<1>(L);
9366 assert(!std::get<1>(L).empty() &&
9367 "Not expecting empty list of components!");
9368 const ValueDecl *VD = std::get<1>(L).back().getAssociatedDeclaration();
9369 if (!Processed.insert(VD).second)
9370 continue;
9372 // For use_device_addr, we match an existing map clause if the
9373 // use_device_addr operand's attach-ptr matches the map operand's
9374 // attach-ptr.
9375 // We chould also restrict to only match cases when there is a full
9376 // match between the map/use_device_addr clause exprs, but that may be
9377 // unnecessary.
9378 //
9379 // map expr | use_device_addr expr | current | possible restrictive/
9380 // | | behavior | safer behavior
9381 // ---------|----------------------|-----------|-----------------------
9382 // p | p | match | match
9383 // p[0] | p[0] | match | match
9384 // p[0:1] | p[0] | match | no match
9385 // p[0:1] | p[2:1] | match | no match
9386 // p[1] | p[0] | match | no match
9387 // ps->a | ps->b | match | no match
9388 // p | p[0] | no match | no match
9389 // pp | pp[0][0] | no match | no match
9390 const Expr *UDAAttachPtrExpr = getAttachPtrExpr(Components);
9391 const Expr *IE = std::get<1>(L).back().getAssociatedExpression();
9392 assert((!UDAAttachPtrExpr || UDAAttachPtrExpr == IE) &&
9393 "use_device_addr operand has an attach-ptr, but does not match "
9394 "last component's expr.");
9395 if (IsMapInfoExist(CGF, VD, IE,
9396 /*DesiredAttachPtrExpr=*/UDAAttachPtrExpr,
9397 /*IsDevAddr=*/true))
9398 continue;
9399 MapInfoGen(CGF, IE, VD, Components,
9400 /*IsDevAddr=*/true,
9401 /*IEIsAttachPtrForDevAddr=*/UDAAttachPtrExpr != nullptr);
9402 }
9403 }
9404
9405 for (const auto &Data : Info) {
9406 MapCombinedInfoTy CurInfo;
9407 const Decl *D = Data.first;
9408 const ValueDecl *VD = cast_or_null<ValueDecl>(D);
9409 // Group component lists by their AttachPtrExpr and process them in order
9410 // of increasing complexity (nullptr first, then simple expressions like
9411 // p, then more complex ones like p[0], etc.)
9412 //
9413 // This is similar to how generateInfoForCaptureFromClauseInfo handles
9414 // grouping for target constructs.
9415 SmallVector<std::pair<const Expr *, MapInfo>, 16> AttachPtrMapInfoPairs;
9416
9417 // First, collect all MapData entries with their attach-ptr exprs.
9418 for (const auto &M : Data.second) {
9419 for (const MapInfo &L : M) {
9420 assert(!L.Components.empty() &&
9421 "Not expecting declaration with no component lists.");
9422
9423 const Expr *AttachPtrExpr = getAttachPtrExpr(L.Components);
9424 AttachPtrMapInfoPairs.emplace_back(AttachPtrExpr, L);
9425 }
9426 }
9427
9428 // Next, sort by increasing order of their complexity.
9429 llvm::stable_sort(AttachPtrMapInfoPairs,
9430 [this](const auto &LHS, const auto &RHS) {
9431 return AttachPtrComparator(LHS.first, RHS.first);
9432 });
9433
9434 // And finally, process them all in order, grouping those with
9435 // equivalent attach-ptr exprs together.
9436 auto *It = AttachPtrMapInfoPairs.begin();
9437 while (It != AttachPtrMapInfoPairs.end()) {
9438 const Expr *AttachPtrExpr = It->first;
9439
9440 SmallVector<MapInfo, 8> GroupLists;
9441 while (It != AttachPtrMapInfoPairs.end() &&
9442 (It->first == AttachPtrExpr ||
9443 AttachPtrComparator.areEqual(It->first, AttachPtrExpr))) {
9444 GroupLists.push_back(It->second);
9445 ++It;
9446 }
9447 assert(!GroupLists.empty() && "GroupLists should not be empty");
9448
9449 StructRangeInfoTy PartialStruct;
9450 AttachInfoTy AttachInfo;
9451 MapCombinedInfoTy GroupCurInfo;
9452 // Current group's struct base information:
9453 MapCombinedInfoTy GroupStructBaseCurInfo;
9454 for (const MapInfo &L : GroupLists) {
9455 // Remember the current base pointer index.
9456 unsigned CurrentBasePointersIdx = GroupCurInfo.BasePointers.size();
9457 unsigned StructBasePointersIdx =
9458 GroupStructBaseCurInfo.BasePointers.size();
9459
9460 GroupCurInfo.NonContigInfo.IsNonContiguous =
9461 L.Components.back().isNonContiguous();
9462 generateInfoForComponentList(
9463 L.MapType, L.MapModifiers, L.MotionModifiers, L.Components,
9464 GroupCurInfo, GroupStructBaseCurInfo, PartialStruct, AttachInfo,
9465 /*IsFirstComponentList=*/false, L.IsImplicit,
9466 /*GenerateAllInfoForClauses*/ true, L.Mapper, L.ForDeviceAddr, VD,
9467 L.VarRef, /*OverlappedElements*/ {});
9468
9469 // If this entry relates to a device pointer, set the relevant
9470 // declaration and add the 'return pointer' flag.
9471 if (L.ReturnDevicePointer) {
9472 // Check whether a value was added to either GroupCurInfo or
9473 // GroupStructBaseCurInfo and error if no value was added to either
9474 // of them:
9475 assert((CurrentBasePointersIdx < GroupCurInfo.BasePointers.size() ||
9476 StructBasePointersIdx <
9477 GroupStructBaseCurInfo.BasePointers.size()) &&
9478 "Unexpected number of mapped base pointers.");
9479
9480 // Choose a base pointer index which is always valid:
9481 const ValueDecl *RelevantVD =
9482 L.Components.back().getAssociatedDeclaration();
9483 assert(RelevantVD &&
9484 "No relevant declaration related with device pointer??");
9485
9486 // If GroupStructBaseCurInfo has been updated this iteration then
9487 // work on the first new entry added to it i.e. make sure that when
9488 // multiple values are added to any of the lists, the first value
9489 // added is being modified by the assignments below (not the last
9490 // value added).
9491 auto SetDevicePointerInfo = [&](MapCombinedInfoTy &Info,
9492 unsigned Idx) {
9493 Info.DevicePtrDecls[Idx] = RelevantVD;
9494 Info.DevicePointers[Idx] = L.ForDeviceAddr
9495 ? DeviceInfoTy::Address
9496 : DeviceInfoTy::Pointer;
9497 Info.Types[Idx] |=
9498 OpenMPOffloadMappingFlags::OMP_MAP_RETURN_PARAM;
9499 if (L.HasUdpFbNullify)
9500 Info.Types[Idx] |=
9501 OpenMPOffloadMappingFlags::OMP_MAP_FB_NULLIFY;
9502 };
9503
9504 if (StructBasePointersIdx <
9505 GroupStructBaseCurInfo.BasePointers.size())
9506 SetDevicePointerInfo(GroupStructBaseCurInfo,
9507 StructBasePointersIdx);
9508 else
9509 SetDevicePointerInfo(GroupCurInfo, CurrentBasePointersIdx);
9510 }
9511 }
9512
9513 // Unify entries in one list making sure the struct mapping precedes the
9514 // individual fields:
9515 MapCombinedInfoTy GroupUnionCurInfo;
9516 GroupUnionCurInfo.append(GroupStructBaseCurInfo);
9517 GroupUnionCurInfo.append(GroupCurInfo);
9518
9519 // If there is an entry in PartialStruct it means we have a struct with
9520 // individual members mapped. Emit an extra combined entry.
9521 if (PartialStruct.Base.isValid()) {
9522 // Prepend a synthetic dimension of length 1 to represent the
9523 // aggregated struct object. Using 1 (not 0, as 0 produced an
9524 // incorrect non-contiguous descriptor (DimSize==1), causing the
9525 // non-contiguous motion clause path to be skipped.) is important:
9526 // * It preserves the correct rank so targetDataUpdate() computes
9527 // DimSize == 2 for cases like strided array sections originating
9528 // from user-defined mappers (e.g. test with s.data[0:8:2]).
9529 GroupUnionCurInfo.NonContigInfo.Dims.insert(
9530 GroupUnionCurInfo.NonContigInfo.Dims.begin(), 1);
9531 emitCombinedEntry(
9532 CurInfo, GroupUnionCurInfo.Types, PartialStruct, AttachInfo,
9533 /*IsMapThis=*/!VD, OMPBuilder, VD,
9534 /*OffsetForMemberOfFlag=*/CombinedInfo.BasePointers.size(),
9535 /*NotTargetParams=*/true);
9536 }
9537
9538 // Append this group's results to the overall CurInfo in the correct
9539 // order: combined-entry -> original-field-entries -> attach-entry
9540 CurInfo.append(GroupUnionCurInfo);
9541 if (AttachInfo.isValid())
9542 emitAttachEntry(CGF, CurInfo, AttachInfo);
9543 }
9544
9545 // We need to append the results of this capture to what we already have.
9546 CombinedInfo.append(CurInfo);
9547 }
9548 // Append data for use_device_ptr/addr clauses.
9549 CombinedInfo.append(UseDeviceDataCombinedInfo);
9550 }
9551
9552public:
9553 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF)
9554 : CurDir(&Dir), CGF(CGF), AttachPtrComparator(*this) {
9555 // Extract firstprivate clause information.
9556 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>())
9557 for (const auto *D : C->varlist()) {
9558 const ValueDecl *VD = cast<DeclRefExpr>(D)->getDecl();
9559 if (const auto *BD = dyn_cast<BindingDecl>(VD))
9560 VD = cast<VarDecl>(BD->getDecomposedDecl());
9561 FirstPrivateDecls.try_emplace(cast<VarDecl>(VD), C->isImplicit());
9562 }
9563 // Extract implicit firstprivates from uses_allocators clauses.
9564 for (const auto *C : Dir.getClausesOfKind<OMPUsesAllocatorsClause>()) {
9565 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) {
9566 OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I);
9567 if (const auto *DRE = dyn_cast_or_null<DeclRefExpr>(D.AllocatorTraits))
9568 FirstPrivateDecls.try_emplace(cast<VarDecl>(DRE->getDecl()),
9569 /*Implicit=*/true);
9570 else if (const auto *VD = dyn_cast<VarDecl>(
9571 cast<DeclRefExpr>(D.Allocator->IgnoreParenImpCasts())
9572 ->getDecl()))
9573 FirstPrivateDecls.try_emplace(VD, /*Implicit=*/true);
9574 }
9575 }
9576 // Extract defaultmap clause information.
9577 for (const auto *C : Dir.getClausesOfKind<OMPDefaultmapClause>())
9578 if (C->getDefaultmapModifier() == OMPC_DEFAULTMAP_MODIFIER_firstprivate)
9579 DefaultmapFirstprivateKinds.insert(C->getDefaultmapKind());
9580 // Extract device pointer clause information.
9581 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
9582 for (auto L : C->component_lists())
9583 DevPointersMap[std::get<0>(L)].push_back(std::get<1>(L));
9584 // Extract device addr clause information.
9585 for (const auto *C : Dir.getClausesOfKind<OMPHasDeviceAddrClause>())
9586 for (auto L : C->component_lists())
9587 HasDevAddrsMap[std::get<0>(L)].push_back(std::get<1>(L));
9588 // Extract map information.
9589 for (const auto *C : Dir.getClausesOfKind<OMPMapClause>()) {
9590 if (C->getMapType() != OMPC_MAP_to)
9591 continue;
9592 for (auto L : C->component_lists()) {
9593 const ValueDecl *VD = std::get<0>(L);
9594 const auto *RD = VD ? VD->getType()
9595 .getCanonicalType()
9596 .getNonReferenceType()
9597 ->getAsCXXRecordDecl()
9598 : nullptr;
9599 if (RD && RD->isLambda())
9600 LambdasMap.try_emplace(std::get<0>(L), C);
9601 }
9602 }
9603
9604 auto CollectAttachPtrExprsForClauseComponents = [this](const auto *C) {
9605 for (auto L : C->component_lists()) {
9607 std::get<1>(L);
9608 if (!Components.empty())
9609 collectAttachPtrExprInfo(Components, CurDir);
9610 }
9611 };
9612
9613 // Populate the AttachPtrExprMap for all component lists from map-related
9614 // clauses.
9615 for (const auto *C : Dir.getClausesOfKind<OMPMapClause>())
9616 CollectAttachPtrExprsForClauseComponents(C);
9617 for (const auto *C : Dir.getClausesOfKind<OMPToClause>())
9618 CollectAttachPtrExprsForClauseComponents(C);
9619 for (const auto *C : Dir.getClausesOfKind<OMPFromClause>())
9620 CollectAttachPtrExprsForClauseComponents(C);
9621 for (const auto *C : Dir.getClausesOfKind<OMPUseDevicePtrClause>())
9622 CollectAttachPtrExprsForClauseComponents(C);
9623 for (const auto *C : Dir.getClausesOfKind<OMPUseDeviceAddrClause>())
9624 CollectAttachPtrExprsForClauseComponents(C);
9625 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
9626 CollectAttachPtrExprsForClauseComponents(C);
9627 for (const auto *C : Dir.getClausesOfKind<OMPHasDeviceAddrClause>())
9628 CollectAttachPtrExprsForClauseComponents(C);
9629 }
9630
9631 /// Constructor for the declare mapper directive.
9632 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF)
9633 : CurDir(&Dir), CGF(CGF), AttachPtrComparator(*this) {
9634 auto CollectAttachPtrExprsForClauseComponents = [this](const auto *C) {
9635 for (auto L : C->component_lists()) {
9637 std::get<1>(L);
9638 if (!Components.empty())
9639 collectAttachPtrExprInfo(Components, CurDir);
9640 }
9641 };
9642
9643 // Populate the AttachPtrExprMap for all component lists from map-related
9644 // clauses in the declare mapper directive, to enable attach-style mapping
9645 // for mappers.
9646 for (const auto *Cl : Dir.clauses()) {
9647 if (const auto *C = dyn_cast<OMPMapClause>(Cl))
9648 CollectAttachPtrExprsForClauseComponents(C);
9649 else if (const auto *C = dyn_cast<OMPToClause>(Cl))
9650 CollectAttachPtrExprsForClauseComponents(C);
9651 else if (const auto *C = dyn_cast<OMPFromClause>(Cl))
9652 CollectAttachPtrExprsForClauseComponents(C);
9653 }
9654 }
9655
9656 /// Generate code for the combined entry if we have a partially mapped struct
9657 /// and take care of the mapping flags of the arguments corresponding to
9658 /// individual struct members.
9659 /// If a valid \p AttachInfo exists, its pointee addr will be updated to point
9660 /// to the combined-entry's begin address, if emitted.
9661 /// \p PartialStruct contains attach base-pointer information.
9662 /// \returns The index of the combined entry if one was added, std::nullopt
9663 /// otherwise.
9664 void emitCombinedEntry(MapCombinedInfoTy &CombinedInfo,
9665 MapFlagsArrayTy &CurTypes,
9666 const StructRangeInfoTy &PartialStruct,
9667 AttachInfoTy &AttachInfo, bool IsMapThis,
9668 llvm::OpenMPIRBuilder &OMPBuilder, const ValueDecl *VD,
9669 unsigned OffsetForMemberOfFlag,
9670 bool NotTargetParams) const {
9671 if (CurTypes.size() == 1 &&
9672 ((CurTypes.back() & OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF) !=
9673 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF) &&
9674 !PartialStruct.IsArraySection)
9675 return;
9676 Address LBAddr = PartialStruct.LowestElem.second;
9677 Address HBAddr = PartialStruct.HighestElem.second;
9678 if (PartialStruct.HasCompleteRecord) {
9679 LBAddr = PartialStruct.LB;
9680 HBAddr = PartialStruct.LB;
9681 }
9682 CombinedInfo.Exprs.push_back(VD);
9683 // Base is the base of the struct
9684 CombinedInfo.BasePointers.push_back(PartialStruct.Base.emitRawPointer(CGF));
9685 CombinedInfo.DevicePtrDecls.push_back(nullptr);
9686 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
9687 // Pointer is the address of the lowest element
9688 llvm::Value *LB = LBAddr.emitRawPointer(CGF);
9689 const CXXMethodDecl *MD =
9690 CGF.CurFuncDecl ? dyn_cast<CXXMethodDecl>(CGF.CurFuncDecl) : nullptr;
9691 const CXXRecordDecl *RD = MD ? MD->getParent() : nullptr;
9692 bool HasBaseClass = RD && IsMapThis ? RD->getNumBases() > 0 : false;
9693 // There should not be a mapper for a combined entry.
9694 if (HasBaseClass) {
9695 // OpenMP 5.2 148:21:
9696 // If the target construct is within a class non-static member function,
9697 // and a variable is an accessible data member of the object for which the
9698 // non-static data member function is invoked, the variable is treated as
9699 // if the this[:1] expression had appeared in a map clause with a map-type
9700 // of tofrom.
9701 // Emit this[:1]
9702 CombinedInfo.Pointers.push_back(PartialStruct.Base.emitRawPointer(CGF));
9703 QualType Ty = MD->getFunctionObjectParameterType();
9704 llvm::Value *Size =
9705 CGF.Builder.CreateIntCast(CGF.getTypeSize(Ty), CGF.Int64Ty,
9706 /*isSigned=*/true);
9707 CombinedInfo.Sizes.push_back(Size);
9708 } else {
9709 CombinedInfo.Pointers.push_back(LB);
9710 // Size is (addr of {highest+1} element) - (addr of lowest element)
9711 llvm::Value *HB = HBAddr.emitRawPointer(CGF);
9712 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(
9713 HBAddr.getElementType(), HB, /*Idx0=*/1);
9714 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy);
9715 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy);
9716 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr);
9717 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty,
9718 /*isSigned=*/false);
9719 CombinedInfo.Sizes.push_back(Size);
9720 }
9721 CombinedInfo.Mappers.push_back(nullptr);
9722 // Map type is always TARGET_PARAM, if generate info for captures.
9723 CombinedInfo.Types.push_back(
9724 NotTargetParams ? OpenMPOffloadMappingFlags::OMP_MAP_NONE
9725 : !PartialStruct.PreliminaryMapData.BasePointers.empty()
9726 ? OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ
9727 : OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM);
9728 // A combined entry has a base attach-ptr if its constituents do. e.g.:
9729 // map(s2.s1p->x, s2.s1p->y)
9730 // combined entry:
9731 // s2.s1p[0], s2.s1p->x, sizeof(s1p->x..y), ALLOC
9732 // here s2.s1p is the attach-ptr for the combined entry.
9733 // See the inline comments in emitUserDefinedMapper's definition for how
9734 // entries with an attach-ptr are treated.
9735 CombinedInfo.HasAttachPtr.push_back(AttachInfo.isValid());
9736 // If any element has the present modifier, then make sure the runtime
9737 // doesn't attempt to allocate the struct.
9738 if (CurTypes.end() !=
9739 llvm::find_if(CurTypes, [](OpenMPOffloadMappingFlags Type) {
9740 return static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
9741 Type & OpenMPOffloadMappingFlags::OMP_MAP_PRESENT);
9742 }))
9743 CombinedInfo.Types.back() |= OpenMPOffloadMappingFlags::OMP_MAP_PRESENT;
9744 // Remove TARGET_PARAM flag from the first element
9745 (*CurTypes.begin()) &= ~OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM;
9746 // If any element has the ompx_hold modifier, then make sure the runtime
9747 // uses the hold reference count for the struct as a whole so that it won't
9748 // be unmapped by an extra dynamic reference count decrement. Add it to all
9749 // elements as well so the runtime knows which reference count to check
9750 // when determining whether it's time for device-to-host transfers of
9751 // individual elements.
9752 if (CurTypes.end() !=
9753 llvm::find_if(CurTypes, [](OpenMPOffloadMappingFlags Type) {
9754 return static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
9755 Type & OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD);
9756 })) {
9757 CombinedInfo.Types.back() |= OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD;
9758 for (auto &M : CurTypes)
9759 M |= OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD;
9760 }
9761
9762 // All other current entries will be MEMBER_OF the combined entry
9763 // (except for PTR_AND_OBJ entries which do not have a placeholder value
9764 // 0xFFFF in the MEMBER_OF field, or ATTACH entries since they are expected
9765 // to be handled by themselves, after all other maps).
9766 OpenMPOffloadMappingFlags MemberOfFlag = OMPBuilder.getMemberOfFlag(
9767 OffsetForMemberOfFlag + CombinedInfo.BasePointers.size() - 1);
9768 for (auto &M : CurTypes)
9769 OMPBuilder.setCorrectMemberOfFlag(M, MemberOfFlag);
9770
9771 // When we are emitting a combined entry. If there were any pending
9772 // attachments to be done, we do them to the begin address of the combined
9773 // entry. Note that this means only one attachment per combined-entry will
9774 // be done. So, for instance, if we have:
9775 // S *ps;
9776 // ... map(ps->a, ps->b)
9777 // When we are emitting a combined entry. If AttachInfo is valid,
9778 // update the pointee address to point to the begin address of the combined
9779 // entry. This ensures that if we have multiple maps like:
9780 // `map(ps->a, ps->b)`, we still get a single ATTACH entry, like:
9781 //
9782 // &ps[0], &ps->a, sizeof(ps->a to ps->b), ALLOC // combined-entry
9783 // &ps[0], &ps->a, sizeof(ps->a), TO | FROM
9784 // &ps[0], &ps->b, sizeof(ps->b), TO | FROM
9785 // &ps, &ps->a, sizeof(void*), ATTACH // Use combined-entry's LB
9786 if (AttachInfo.isValid())
9787 AttachInfo.AttachPteeAddr = LBAddr;
9788 }
9789
9790 /// Generate all the base pointers, section pointers, sizes, map types, and
9791 /// mappers for the extracted mappable expressions (all included in \a
9792 /// CombinedInfo). Also, for each item that relates with a device pointer, a
9793 /// pair of the relevant declaration and index where it occurs is appended to
9794 /// the device pointers info array.
9795 void generateAllInfo(
9796 MapCombinedInfoTy &CombinedInfo, llvm::OpenMPIRBuilder &OMPBuilder,
9797 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet =
9798 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const {
9799 assert(isa<const OMPExecutableDirective *>(CurDir) &&
9800 "Expect a executable directive");
9801 const auto *CurExecDir = cast<const OMPExecutableDirective *>(CurDir);
9802 generateAllInfoForClauses(CurExecDir->clauses(), CombinedInfo, OMPBuilder,
9803 SkipVarSet);
9804 }
9805
9806 /// Generate all the base pointers, section pointers, sizes, map types, and
9807 /// mappers for the extracted map clauses of user-defined mapper (all included
9808 /// in \a CombinedInfo).
9809 void generateAllInfoForMapper(MapCombinedInfoTy &CombinedInfo,
9810 llvm::OpenMPIRBuilder &OMPBuilder) const {
9811 assert(isa<const OMPDeclareMapperDecl *>(CurDir) &&
9812 "Expect a declare mapper directive");
9813 const auto *CurMapperDir = cast<const OMPDeclareMapperDecl *>(CurDir);
9814 generateAllInfoForClauses(CurMapperDir->clauses(), CombinedInfo,
9815 OMPBuilder);
9816 }
9817
9818 /// Emit capture info for lambdas for variables captured by reference.
9819 void generateInfoForLambdaCaptures(
9820 const ValueDecl *VD, llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo,
9821 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const {
9822 QualType VDType = VD->getType().getCanonicalType().getNonReferenceType();
9823 const auto *RD = VDType->getAsCXXRecordDecl();
9824 if (!RD || !RD->isLambda())
9825 return;
9826 Address VDAddr(Arg, CGF.ConvertTypeForMem(VDType),
9827 CGF.getContext().getDeclAlign(VD));
9828 LValue VDLVal = CGF.MakeAddrLValue(VDAddr, VDType);
9829 llvm::DenseMap<const ValueDecl *, FieldDecl *> Captures;
9830 FieldDecl *ThisCapture = nullptr;
9831 RD->getCaptureFields(Captures, ThisCapture);
9832 if (ThisCapture) {
9833 LValue ThisLVal =
9834 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture);
9835 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture);
9836 LambdaPointers.try_emplace(ThisLVal.getPointer(CGF),
9837 VDLVal.getPointer(CGF));
9838 CombinedInfo.Exprs.push_back(VD);
9839 CombinedInfo.BasePointers.push_back(ThisLVal.getPointer(CGF));
9840 CombinedInfo.DevicePtrDecls.push_back(nullptr);
9841 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
9842 CombinedInfo.Pointers.push_back(ThisLValVal.getPointer(CGF));
9843 CombinedInfo.Sizes.push_back(
9844 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
9845 CGF.Int64Ty, /*isSigned=*/true));
9846 CombinedInfo.Types.push_back(
9847 OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ |
9848 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
9849 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
9850 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT);
9851 CombinedInfo.HasAttachPtr.push_back(false);
9852 CombinedInfo.Mappers.push_back(nullptr);
9853 }
9854 for (const LambdaCapture &LC : RD->captures()) {
9855 if (!LC.capturesVariable())
9856 continue;
9857 const VarDecl *VD = cast<VarDecl>(LC.getCapturedVar());
9858 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType())
9859 continue;
9860 auto It = Captures.find(VD);
9861 assert(It != Captures.end() && "Found lambda capture without field.");
9862 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second);
9863 if (LC.getCaptureKind() == LCK_ByRef) {
9864 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second);
9865 LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
9866 VDLVal.getPointer(CGF));
9867 CombinedInfo.Exprs.push_back(VD);
9868 CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF));
9869 CombinedInfo.DevicePtrDecls.push_back(nullptr);
9870 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
9871 CombinedInfo.Pointers.push_back(VarLValVal.getPointer(CGF));
9872 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
9873 CGF.getTypeSize(
9875 CGF.Int64Ty, /*isSigned=*/true));
9876 } else {
9877 RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation());
9878 LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
9879 VDLVal.getPointer(CGF));
9880 CombinedInfo.Exprs.push_back(VD);
9881 CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF));
9882 CombinedInfo.DevicePtrDecls.push_back(nullptr);
9883 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
9884 CombinedInfo.Pointers.push_back(VarRVal.getScalarVal());
9885 CombinedInfo.Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0));
9886 }
9887 CombinedInfo.Types.push_back(
9888 OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ |
9889 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
9890 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
9891 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT);
9892 CombinedInfo.HasAttachPtr.push_back(false);
9893 CombinedInfo.Mappers.push_back(nullptr);
9894 }
9895 }
9896
9897 /// Set correct indices for lambdas captures.
9898 void adjustMemberOfForLambdaCaptures(
9899 llvm::OpenMPIRBuilder &OMPBuilder,
9900 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers,
9901 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
9902 MapFlagsArrayTy &Types) const {
9903 for (unsigned I = 0, E = Types.size(); I < E; ++I) {
9904 // Set correct member_of idx for all implicit lambda captures.
9905 if (Types[I] != (OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ |
9906 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
9907 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
9908 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT))
9909 continue;
9910 llvm::Value *BasePtr = LambdaPointers.lookup(BasePointers[I]);
9911 assert(BasePtr && "Unable to find base lambda address.");
9912 int TgtIdx = -1;
9913 for (unsigned J = I; J > 0; --J) {
9914 unsigned Idx = J - 1;
9915 if (Pointers[Idx] != BasePtr)
9916 continue;
9917 TgtIdx = Idx;
9918 break;
9919 }
9920 assert(TgtIdx != -1 && "Unable to find parent lambda.");
9921 // All other current entries will be MEMBER_OF the combined entry
9922 // (except for PTR_AND_OBJ entries which do not have a placeholder value
9923 // 0xFFFF in the MEMBER_OF field).
9924 OpenMPOffloadMappingFlags MemberOfFlag =
9925 OMPBuilder.getMemberOfFlag(TgtIdx);
9926 OMPBuilder.setCorrectMemberOfFlag(Types[I], MemberOfFlag);
9927 }
9928 }
9929
9930 /// Populate component lists for non-lambda captured variables from map,
9931 /// is_device_ptr and has_device_addr clause info.
9932 void populateComponentListsForNonLambdaCaptureFromClauses(
9933 const ValueDecl *VD, MapDataArrayTy &DeclComponentLists,
9934 SmallVectorImpl<
9935 SmallVector<OMPClauseMappableExprCommon::MappableComponent, 8>>
9936 &StorageForImplicitlyAddedComponentLists) const {
9937 if (VD && LambdasMap.count(VD))
9938 return;
9939
9940 // For member fields list in is_device_ptr, store it in
9941 // DeclComponentLists for generating components info.
9943 auto It = DevPointersMap.find(VD);
9944 if (It != DevPointersMap.end())
9945 for (const auto &MCL : It->second)
9946 DeclComponentLists.emplace_back(MCL, OMPC_MAP_to, Unknown,
9947 /*IsImpicit = */ true, nullptr,
9948 nullptr);
9949 auto I = HasDevAddrsMap.find(VD);
9950 if (I != HasDevAddrsMap.end())
9951 for (const auto &MCL : I->second)
9952 DeclComponentLists.emplace_back(MCL, OMPC_MAP_tofrom, Unknown,
9953 /*IsImpicit = */ true, nullptr,
9954 nullptr);
9955 assert(isa<const OMPExecutableDirective *>(CurDir) &&
9956 "Expect a executable directive");
9957 const auto *CurExecDir = cast<const OMPExecutableDirective *>(CurDir);
9958 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
9959 const auto *EI = C->getVarRefs().begin();
9960 for (const auto L : C->decl_component_lists(VD)) {
9961 const ValueDecl *VDecl, *Mapper;
9962 // The Expression is not correct if the mapping is implicit
9963 const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr;
9965 std::tie(VDecl, Components, Mapper) = L;
9966 assert(VDecl == VD && "We got information for the wrong declaration??");
9967 assert(!Components.empty() &&
9968 "Not expecting declaration with no component lists.");
9969 DeclComponentLists.emplace_back(Components, C->getMapType(),
9970 C->getMapTypeModifiers(),
9971 C->isImplicit(), Mapper, E);
9972 ++EI;
9973 }
9974 }
9975
9976 // For the target construct, if there's a map with a base-pointer that's
9977 // a member of an implicitly captured struct, of the current class,
9978 // we need to emit an implicit map on the pointer.
9979 if (isOpenMPTargetExecutionDirective(CurExecDir->getDirectiveKind()))
9980 addImplicitMapForAttachPtrBaseIfMemberOfCapturedVD(
9981 VD, DeclComponentLists, StorageForImplicitlyAddedComponentLists);
9982
9983 llvm::stable_sort(DeclComponentLists, [](const MapData &LHS,
9984 const MapData &RHS) {
9985 ArrayRef<OpenMPMapModifierKind> MapModifiers = std::get<2>(LHS);
9986 OpenMPMapClauseKind MapType = std::get<1>(RHS);
9987 bool HasPresent =
9988 llvm::is_contained(MapModifiers, clang::OMPC_MAP_MODIFIER_present);
9989 bool HasAllocs = MapType == OMPC_MAP_alloc;
9990 MapModifiers = std::get<2>(RHS);
9991 MapType = std::get<1>(LHS);
9992 bool HasPresentR =
9993 llvm::is_contained(MapModifiers, clang::OMPC_MAP_MODIFIER_present);
9994 bool HasAllocsR = MapType == OMPC_MAP_alloc;
9995 return (HasPresent && !HasPresentR) || (HasAllocs && !HasAllocsR);
9996 });
9997 }
9998
9999 /// On a target construct, if there's an implicit map on a struct, or that of
10000 /// this[:], and an explicit map with a member of that struct/class as the
10001 /// base-pointer, we need to make sure that base-pointer is implicitly mapped,
10002 /// to make sure we don't map the full struct/class. For example:
10003 ///
10004 /// \code
10005 /// struct S {
10006 /// int dummy[10000];
10007 /// int *p;
10008 /// void f1() {
10009 /// #pragma omp target map(p[0:1])
10010 /// (void)this;
10011 /// }
10012 /// }; S s;
10013 ///
10014 /// void f2() {
10015 /// #pragma omp target map(s.p[0:10])
10016 /// (void)s;
10017 /// }
10018 /// \endcode
10019 ///
10020 /// Only `this-p` and `s.p` should be mapped in the two cases above.
10021 //
10022 // OpenMP 6.0: 7.9.6 map clause, pg 285
10023 // If a list item with an implicitly determined data-mapping attribute does
10024 // not have any corresponding storage in the device data environment prior to
10025 // a task encountering the construct associated with the map clause, and one
10026 // or more contiguous parts of the original storage are either list items or
10027 // base pointers to list items that are explicitly mapped on the construct,
10028 // only those parts of the original storage will have corresponding storage in
10029 // the device data environment as a result of the map clauses on the
10030 // construct.
10031 void addImplicitMapForAttachPtrBaseIfMemberOfCapturedVD(
10032 const ValueDecl *CapturedVD, MapDataArrayTy &DeclComponentLists,
10033 SmallVectorImpl<
10034 SmallVector<OMPClauseMappableExprCommon::MappableComponent, 8>>
10035 &ComponentVectorStorage) const {
10036 bool IsThisCapture = CapturedVD == nullptr;
10037
10038 for (const auto &ComponentsAndAttachPtr : AttachPtrExprMap) {
10040 ComponentsWithAttachPtr = ComponentsAndAttachPtr.first;
10041 const Expr *AttachPtrExpr = ComponentsAndAttachPtr.second;
10042 if (!AttachPtrExpr)
10043 continue;
10044
10045 const auto *ME = dyn_cast<MemberExpr>(AttachPtrExpr);
10046 if (!ME)
10047 continue;
10048
10049 const Expr *Base = ME->getBase()->IgnoreParenImpCasts();
10050
10051 // If we are handling a "this" capture, then we are looking for
10052 // attach-ptrs of form `this->p`, either explicitly or implicitly.
10053 if (IsThisCapture && !ME->isImplicitCXXThis() && !isa<CXXThisExpr>(Base))
10054 continue;
10055
10056 if (!IsThisCapture && (!isa<DeclRefExpr>(Base) ||
10057 cast<DeclRefExpr>(Base)->getDecl() != CapturedVD))
10058 continue;
10059
10060 // For non-this captures, we are looking for attach-ptrs of form
10061 // `s.p`.
10062 // For non-this captures, we are looking for attach-ptrs like `s.p`.
10063 if (!IsThisCapture && (ME->isArrow() || !isa<DeclRefExpr>(Base) ||
10064 cast<DeclRefExpr>(Base)->getDecl() != CapturedVD))
10065 continue;
10066
10067 // Check if we have an existing map on either:
10068 // this[:], s, this->p, or s.p, in which case, we don't need to add
10069 // an implicit one for the attach-ptr s.p/this->p.
10070 bool FoundExistingMap = false;
10071 for (const MapData &ExistingL : DeclComponentLists) {
10073 ExistingComponents = std::get<0>(ExistingL);
10074
10075 if (ExistingComponents.empty())
10076 continue;
10077
10078 // First check if we have a map like map(this->p) or map(s.p).
10079 const auto &FirstComponent = ExistingComponents.front();
10080 const Expr *FirstExpr = FirstComponent.getAssociatedExpression();
10081
10082 if (!FirstExpr)
10083 continue;
10084
10085 // First check if we have a map like map(this->p) or map(s.p).
10086 if (AttachPtrComparator.areEqual(FirstExpr, AttachPtrExpr)) {
10087 FoundExistingMap = true;
10088 break;
10089 }
10090
10091 // Check if we have a map like this[0:1]
10092 if (IsThisCapture) {
10093 if (const auto *OASE = dyn_cast<ArraySectionExpr>(FirstExpr)) {
10094 if (isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts())) {
10095 FoundExistingMap = true;
10096 break;
10097 }
10098 }
10099 continue;
10100 }
10101
10102 // When the attach-ptr is something like `s.p`, check if
10103 // `s` itself is mapped explicitly.
10104 if (const auto *DRE = dyn_cast<DeclRefExpr>(FirstExpr)) {
10105 if (DRE->getDecl() == CapturedVD) {
10106 FoundExistingMap = true;
10107 break;
10108 }
10109 }
10110 }
10111
10112 if (FoundExistingMap)
10113 continue;
10114
10115 // If no base map is found, we need to create an implicit map for the
10116 // attach-pointer expr.
10117
10118 ComponentVectorStorage.emplace_back();
10119 auto &AttachPtrComponents = ComponentVectorStorage.back();
10120
10122 bool SeenAttachPtrComponent = false;
10123 // For creating a map on the attach-ptr `s.p/this->p`, we copy all
10124 // components from the component-list which has `s.p/this->p`
10125 // as the attach-ptr, starting from the component which matches
10126 // `s.p/this->p`. This way, we'll have component-lists of
10127 // `s.p` -> `s`, and `this->p` -> `this`.
10128 for (size_t i = 0; i < ComponentsWithAttachPtr.size(); ++i) {
10129 const auto &Component = ComponentsWithAttachPtr[i];
10130 const Expr *ComponentExpr = Component.getAssociatedExpression();
10131
10132 if (!SeenAttachPtrComponent && ComponentExpr != AttachPtrExpr)
10133 continue;
10134 SeenAttachPtrComponent = true;
10135
10136 AttachPtrComponents.emplace_back(Component.getAssociatedExpression(),
10137 Component.getAssociatedDeclaration(),
10138 Component.isNonContiguous());
10139 }
10140 assert(!AttachPtrComponents.empty() &&
10141 "Could not populate component-lists for mapping attach-ptr");
10142
10143 DeclComponentLists.emplace_back(
10144 AttachPtrComponents, OMPC_MAP_tofrom, Unknown,
10145 /*IsImplicit=*/true, /*mapper=*/nullptr, AttachPtrExpr);
10146 }
10147 }
10148
10149 /// For a capture that has an associated clause, generate the base pointers,
10150 /// section pointers, sizes, map types, and mappers (all included in
10151 /// \a CurCaptureVarInfo).
10152 void generateInfoForCaptureFromClauseInfo(
10153 const MapDataArrayTy &DeclComponentListsFromClauses,
10154 const CapturedStmt::Capture *Cap, llvm::Value *Arg,
10155 MapCombinedInfoTy &CurCaptureVarInfo, llvm::OpenMPIRBuilder &OMPBuilder,
10156 unsigned OffsetForMemberOfFlag) const {
10157 assert(!Cap->capturesVariableArrayType() &&
10158 "Not expecting to generate map info for a variable array type!");
10159
10160 // We need to know when we generating information for the first component
10161 const ValueDecl *VD = Cap->capturesThis()
10162 ? nullptr
10163 : Cap->getCapturedVar()->getCanonicalDecl();
10164
10165 // for map(to: lambda): skip here, processing it in
10166 // generateDefaultMapInfo
10167 if (LambdasMap.count(VD))
10168 return;
10169
10170 // If this declaration appears in a is_device_ptr clause we just have to
10171 // pass the pointer by value. If it is a reference to a declaration, we just
10172 // pass its value.
10173 if (VD && (DevPointersMap.count(VD) || HasDevAddrsMap.count(VD))) {
10174 CurCaptureVarInfo.Exprs.push_back(VD);
10175 CurCaptureVarInfo.BasePointers.emplace_back(Arg);
10176 CurCaptureVarInfo.DevicePtrDecls.emplace_back(VD);
10177 CurCaptureVarInfo.DevicePointers.emplace_back(DeviceInfoTy::Pointer);
10178 CurCaptureVarInfo.Pointers.push_back(Arg);
10179 CurCaptureVarInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
10180 CGF.getTypeSize(CGF.getContext().VoidPtrTy), CGF.Int64Ty,
10181 /*isSigned=*/true));
10182 CurCaptureVarInfo.Types.push_back(
10183 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
10184 OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM);
10185 CurCaptureVarInfo.HasAttachPtr.push_back(false);
10186 CurCaptureVarInfo.Mappers.push_back(nullptr);
10187 return;
10188 }
10189
10190 auto GenerateInfoForComponentLists =
10191 [&](ArrayRef<MapData> DeclComponentListsFromClauses,
10192 bool IsEligibleForTargetParamFlag) {
10193 MapCombinedInfoTy CurInfoForComponentLists;
10194 StructRangeInfoTy PartialStruct;
10195 AttachInfoTy AttachInfo;
10196
10197 if (DeclComponentListsFromClauses.empty())
10198 return;
10199
10200 generateInfoForCaptureFromComponentLists(
10201 VD, DeclComponentListsFromClauses, CurInfoForComponentLists,
10202 PartialStruct, AttachInfo, IsEligibleForTargetParamFlag);
10203
10204 // If there is an entry in PartialStruct it means we have a
10205 // struct with individual members mapped. Emit an extra combined
10206 // entry.
10207 if (PartialStruct.Base.isValid()) {
10208 CurCaptureVarInfo.append(PartialStruct.PreliminaryMapData);
10209 emitCombinedEntry(
10210 CurCaptureVarInfo, CurInfoForComponentLists.Types,
10211 PartialStruct, AttachInfo, Cap->capturesThis(), OMPBuilder,
10212 /*VD=*/nullptr, OffsetForMemberOfFlag,
10213 /*NotTargetParams*/ !IsEligibleForTargetParamFlag);
10214 }
10215
10216 // We do the appends to get the entries in the following order:
10217 // combined-entry -> individual-field-entries -> attach-entry,
10218 CurCaptureVarInfo.append(CurInfoForComponentLists);
10219 if (AttachInfo.isValid())
10220 emitAttachEntry(CGF, CurCaptureVarInfo, AttachInfo);
10221 };
10222
10223 // Group component lists by their AttachPtrExpr and process them in order
10224 // of increasing complexity (nullptr first, then simple expressions like p,
10225 // then more complex ones like p[0], etc.)
10226 //
10227 // This ensure that we:
10228 // * handle maps that can contribute towards setting the kernel argument,
10229 // (e.g. map(ps), or map(ps[0])), before any that cannot (e.g. ps->pt->d).
10230 // * allocate a single contiguous storage for all exprs with the same
10231 // captured var and having the same attach-ptr.
10232 //
10233 // Example: The map clauses below should be handled grouped together based
10234 // on their attachable-base-pointers:
10235 // map-clause | attachable-base-pointer
10236 // --------------------------+------------------------
10237 // map(p, ps) | nullptr
10238 // map(p[0]) | p
10239 // map(p[0]->b, p[0]->c) | p[0]
10240 // map(ps->d, ps->e, ps->pt) | ps
10241 // map(ps->pt->d, ps->pt->e) | ps->pt
10242
10243 // First, collect all MapData entries with their attach-ptr exprs.
10244 SmallVector<std::pair<const Expr *, MapData>, 16> AttachPtrMapDataPairs;
10245
10246 for (const MapData &L : DeclComponentListsFromClauses) {
10248 std::get<0>(L);
10249 const Expr *AttachPtrExpr = getAttachPtrExpr(Components);
10250 AttachPtrMapDataPairs.emplace_back(AttachPtrExpr, L);
10251 }
10252
10253 // Next, sort by increasing order of their complexity.
10254 llvm::stable_sort(AttachPtrMapDataPairs,
10255 [this](const auto &LHS, const auto &RHS) {
10256 return AttachPtrComparator(LHS.first, RHS.first);
10257 });
10258
10259 bool NoDefaultMappingDoneForVD = CurCaptureVarInfo.BasePointers.empty();
10260 bool IsFirstGroup = true;
10261
10262 // And finally, process them all in order, grouping those with
10263 // equivalent attach-ptr exprs together.
10264 auto *It = AttachPtrMapDataPairs.begin();
10265 while (It != AttachPtrMapDataPairs.end()) {
10266 const Expr *AttachPtrExpr = It->first;
10267
10268 MapDataArrayTy GroupLists;
10269 while (It != AttachPtrMapDataPairs.end() &&
10270 (It->first == AttachPtrExpr ||
10271 AttachPtrComparator.areEqual(It->first, AttachPtrExpr))) {
10272 GroupLists.push_back(It->second);
10273 ++It;
10274 }
10275 assert(!GroupLists.empty() && "GroupLists should not be empty");
10276
10277 // Determine if this group of component-lists is eligible for TARGET_PARAM
10278 // flag. Only the first group processed should be eligible, and only if no
10279 // default mapping was done.
10280 bool IsEligibleForTargetParamFlag =
10281 IsFirstGroup && NoDefaultMappingDoneForVD;
10282
10283 GenerateInfoForComponentLists(GroupLists, IsEligibleForTargetParamFlag);
10284 IsFirstGroup = false;
10285 }
10286 }
10287
10288 /// Generate the base pointers, section pointers, sizes, map types, and
10289 /// mappers associated to \a DeclComponentLists for a given capture
10290 /// \a VD (all included in \a CurComponentListInfo).
10291 void generateInfoForCaptureFromComponentLists(
10292 const ValueDecl *VD, ArrayRef<MapData> DeclComponentLists,
10293 MapCombinedInfoTy &CurComponentListInfo, StructRangeInfoTy &PartialStruct,
10294 AttachInfoTy &AttachInfo, bool IsListEligibleForTargetParamFlag) const {
10295 // Find overlapping elements (including the offset from the base element).
10296 llvm::SmallDenseMap<
10297 const MapData *,
10298 llvm::SmallVector<
10300 4>
10301 OverlappedData;
10302 size_t Count = 0;
10303 for (const MapData &L : DeclComponentLists) {
10305 OpenMPMapClauseKind MapType;
10306 ArrayRef<OpenMPMapModifierKind> MapModifiers;
10307 bool IsImplicit;
10308 const ValueDecl *Mapper;
10309 const Expr *VarRef;
10310 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) =
10311 L;
10312 ++Count;
10313 for (const MapData &L1 : ArrayRef(DeclComponentLists).slice(Count)) {
10315 std::tie(Components1, MapType, MapModifiers, IsImplicit, Mapper,
10316 VarRef) = L1;
10317 auto CI = Components.rbegin();
10318 auto CE = Components.rend();
10319 auto SI = Components1.rbegin();
10320 auto SE = Components1.rend();
10321 for (; CI != CE && SI != SE; ++CI, ++SI) {
10322 if (CI->getAssociatedExpression()->getStmtClass() !=
10323 SI->getAssociatedExpression()->getStmtClass())
10324 break;
10325 // Are we dealing with different variables/fields?
10326 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration())
10327 break;
10328 }
10329 // Found overlapping if, at least for one component, reached the head
10330 // of the components list.
10331 if (CI == CE || SI == SE) {
10332 // Ignore it if it is the same component.
10333 if (CI == CE && SI == SE)
10334 continue;
10335 const auto It = (SI == SE) ? CI : SI;
10336 // If one component is a pointer and another one is a kind of
10337 // dereference of this pointer (array subscript, section, dereference,
10338 // etc.), it is not an overlapping.
10339 // Same, if one component is a base and another component is a
10340 // dereferenced pointer memberexpr with the same base.
10341 if (!isa<MemberExpr>(It->getAssociatedExpression()) ||
10342 (std::prev(It)->getAssociatedDeclaration() &&
10343 std::prev(It)
10344 ->getAssociatedDeclaration()
10345 ->getType()
10346 ->isPointerType()) ||
10347 (It->getAssociatedDeclaration() &&
10348 It->getAssociatedDeclaration()->getType()->isPointerType() &&
10349 std::next(It) != CE && std::next(It) != SE))
10350 continue;
10351 const MapData &BaseData = CI == CE ? L : L1;
10353 SI == SE ? Components : Components1;
10354 OverlappedData[&BaseData].push_back(SubData);
10355 }
10356 }
10357 }
10358 // Sort the overlapped elements for each item.
10359 llvm::SmallVector<const FieldDecl *, 4> Layout;
10360 if (!OverlappedData.empty()) {
10361 const Type *BaseType = VD->getType().getCanonicalType().getTypePtr();
10362 const Type *OrigType = BaseType->getPointeeOrArrayElementType();
10363 while (BaseType != OrigType) {
10364 BaseType = OrigType->getCanonicalTypeInternal().getTypePtr();
10365 OrigType = BaseType->getPointeeOrArrayElementType();
10366 }
10367
10368 if (const auto *CRD = BaseType->getAsCXXRecordDecl())
10369 getPlainLayout(CRD, Layout, /*AsBase=*/false);
10370 else {
10371 const auto *RD = BaseType->getAsRecordDecl();
10372 Layout.append(RD->field_begin(), RD->field_end());
10373 }
10374 }
10375 for (auto &Pair : OverlappedData) {
10376 llvm::stable_sort(
10377 Pair.getSecond(),
10378 [&Layout](
10381 Second) {
10382 auto CI = First.rbegin();
10383 auto CE = First.rend();
10384 auto SI = Second.rbegin();
10385 auto SE = Second.rend();
10386 for (; CI != CE && SI != SE; ++CI, ++SI) {
10387 if (CI->getAssociatedExpression()->getStmtClass() !=
10388 SI->getAssociatedExpression()->getStmtClass())
10389 break;
10390 // Are we dealing with different variables/fields?
10391 if (CI->getAssociatedDeclaration() !=
10392 SI->getAssociatedDeclaration())
10393 break;
10394 }
10395
10396 // Lists contain the same elements.
10397 if (CI == CE && SI == SE)
10398 return false;
10399
10400 // List with less elements is less than list with more elements.
10401 if (CI == CE || SI == SE)
10402 return CI == CE;
10403
10404 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration());
10405 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration());
10406 if (FD1->getParent() == FD2->getParent())
10407 return FD1->getFieldIndex() < FD2->getFieldIndex();
10408 const auto *It =
10409 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) {
10410 return FD == FD1 || FD == FD2;
10411 });
10412 return *It == FD1;
10413 });
10414 }
10415
10416 // Associated with a capture, because the mapping flags depend on it.
10417 // Go through all of the elements with the overlapped elements.
10418 bool AddTargetParamFlag = IsListEligibleForTargetParamFlag;
10419 MapCombinedInfoTy StructBaseCombinedInfo;
10420 for (const auto &Pair : OverlappedData) {
10421 const MapData &L = *Pair.getFirst();
10423 OpenMPMapClauseKind MapType;
10424 ArrayRef<OpenMPMapModifierKind> MapModifiers;
10425 bool IsImplicit;
10426 const ValueDecl *Mapper;
10427 const Expr *VarRef;
10428 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) =
10429 L;
10430 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
10431 OverlappedComponents = Pair.getSecond();
10432 generateInfoForComponentList(
10433 MapType, MapModifiers, {}, Components, CurComponentListInfo,
10434 StructBaseCombinedInfo, PartialStruct, AttachInfo, AddTargetParamFlag,
10435 IsImplicit, /*GenerateAllInfoForClauses*/ false, Mapper,
10436 /*ForDeviceAddr=*/false, VD, VarRef, OverlappedComponents);
10437 AddTargetParamFlag = false;
10438 }
10439 // Go through other elements without overlapped elements.
10440 for (const MapData &L : DeclComponentLists) {
10442 OpenMPMapClauseKind MapType;
10443 ArrayRef<OpenMPMapModifierKind> MapModifiers;
10444 bool IsImplicit;
10445 const ValueDecl *Mapper;
10446 const Expr *VarRef;
10447 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) =
10448 L;
10449 auto It = OverlappedData.find(&L);
10450 if (It == OverlappedData.end())
10451 generateInfoForComponentList(
10452 MapType, MapModifiers, {}, Components, CurComponentListInfo,
10453 StructBaseCombinedInfo, PartialStruct, AttachInfo,
10454 AddTargetParamFlag, IsImplicit, /*GenerateAllInfoForClauses*/ false,
10455 Mapper, /*ForDeviceAddr=*/false, VD, VarRef,
10456 /*OverlappedElements*/ {});
10457 AddTargetParamFlag = false;
10458 }
10459 }
10460
10461 /// Check if a variable should be treated as firstprivate due to explicit
10462 /// firstprivate clause or defaultmap(firstprivate:...).
10463 bool isEffectivelyFirstprivate(const VarDecl *VD, QualType Type) const {
10464 // Check explicit firstprivate clauses (not implicit from defaultmap)
10465 auto I = FirstPrivateDecls.find(VD);
10466 if (I != FirstPrivateDecls.end() && !I->getSecond())
10467 return true; // Explicit firstprivate only
10468
10469 // Check defaultmap(firstprivate:scalar) for scalar types
10470 if (DefaultmapFirstprivateKinds.count(OMPC_DEFAULTMAP_scalar)) {
10471 if (Type->isScalarType())
10472 return true;
10473 }
10474
10475 // Check defaultmap(firstprivate:pointer) for pointer types
10476 if (DefaultmapFirstprivateKinds.count(OMPC_DEFAULTMAP_pointer)) {
10477 if (Type->isAnyPointerType())
10478 return true;
10479 }
10480
10481 // Check defaultmap(firstprivate:aggregate) for aggregate types
10482 if (DefaultmapFirstprivateKinds.count(OMPC_DEFAULTMAP_aggregate)) {
10483 if (Type->isAggregateType())
10484 return true;
10485 }
10486
10487 // Check defaultmap(firstprivate:all) for all types
10488 return DefaultmapFirstprivateKinds.count(OMPC_DEFAULTMAP_all);
10489 }
10490
10491 /// Generate the default map information for a given capture \a CI,
10492 /// record field declaration \a RI and captured value \a CV.
10493 void generateDefaultMapInfo(const CapturedStmt::Capture &CI,
10494 const FieldDecl &RI, llvm::Value *CV,
10495 MapCombinedInfoTy &CombinedInfo) const {
10496 bool IsImplicit = true;
10497 // Do the default mapping.
10498 if (CI.capturesThis()) {
10499 CombinedInfo.Exprs.push_back(nullptr);
10500 CombinedInfo.BasePointers.push_back(CV);
10501 CombinedInfo.DevicePtrDecls.push_back(nullptr);
10502 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
10503 CombinedInfo.Pointers.push_back(CV);
10504 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr());
10505 CombinedInfo.Sizes.push_back(
10506 CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()),
10507 CGF.Int64Ty, /*isSigned=*/true));
10508 // Default map type.
10509 CombinedInfo.Types.push_back(OpenMPOffloadMappingFlags::OMP_MAP_TO |
10510 OpenMPOffloadMappingFlags::OMP_MAP_FROM);
10511 } else if (CI.capturesVariableByCopy()) {
10512 const VarDecl *VD = CI.getCapturedVar();
10513 CombinedInfo.Exprs.push_back(VD->getCanonicalDecl());
10514 CombinedInfo.BasePointers.push_back(CV);
10515 CombinedInfo.DevicePtrDecls.push_back(nullptr);
10516 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
10517 CombinedInfo.Pointers.push_back(CV);
10518 bool IsFirstprivate =
10519 isEffectivelyFirstprivate(VD, RI.getType().getNonReferenceType());
10520
10521 if (!RI.getType()->isAnyPointerType()) {
10522 // We have to signal to the runtime captures passed by value that are
10523 // not pointers.
10524 CombinedInfo.Types.push_back(
10525 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
10526 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
10527 CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true));
10528 } else if (IsFirstprivate) {
10529 // Firstprivate pointers should be passed by value (as literals)
10530 // without performing a present table lookup at runtime.
10531 CombinedInfo.Types.push_back(
10532 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
10533 // Use zero size for pointer literals (just passing the pointer value)
10534 CombinedInfo.Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
10535 } else {
10536 // Pointers are implicitly mapped with a zero size and no flags
10537 // (other than first map that is added for all implicit maps).
10538 CombinedInfo.Types.push_back(OpenMPOffloadMappingFlags::OMP_MAP_NONE);
10539 CombinedInfo.Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
10540 }
10541 auto I = FirstPrivateDecls.find(VD);
10542 if (I != FirstPrivateDecls.end())
10543 IsImplicit = I->getSecond();
10544 } else {
10545 assert(CI.capturesVariable() && "Expected captured reference.");
10546 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr());
10547 QualType ElementType = PtrTy->getPointeeType();
10548 const VarDecl *VD = CI.getCapturedVar();
10549 bool IsFirstprivate = isEffectivelyFirstprivate(VD, ElementType);
10550 CombinedInfo.Exprs.push_back(VD->getCanonicalDecl());
10551 CombinedInfo.BasePointers.push_back(CV);
10552 CombinedInfo.DevicePtrDecls.push_back(nullptr);
10553 CombinedInfo.DevicePointers.push_back(DeviceInfoTy::None);
10554
10555 // For firstprivate pointers, pass by value instead of dereferencing
10556 if (IsFirstprivate && ElementType->isAnyPointerType()) {
10557 // Treat as a literal value (pass the pointer value itself)
10558 CombinedInfo.Pointers.push_back(CV);
10559 // Use zero size for pointer literals
10560 CombinedInfo.Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
10561 CombinedInfo.Types.push_back(
10562 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
10563 } else {
10564 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
10565 CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true));
10566 // The default map type for a scalar/complex type is 'to' because by
10567 // default the value doesn't have to be retrieved. For an aggregate
10568 // type, the default is 'tofrom'.
10569 CombinedInfo.Types.push_back(getMapModifiersForPrivateClauses(CI));
10570 CombinedInfo.Pointers.push_back(CV);
10571 }
10572 auto I = FirstPrivateDecls.find(VD);
10573 if (I != FirstPrivateDecls.end())
10574 IsImplicit = I->getSecond();
10575 }
10576 // Every default map produces a single argument which is a target parameter.
10577 CombinedInfo.Types.back() |=
10578 OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM;
10579
10580 // Add flag stating this is an implicit map.
10581 if (IsImplicit)
10582 CombinedInfo.Types.back() |= OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT;
10583
10584 CombinedInfo.HasAttachPtr.push_back(false);
10585 // No user-defined mapper for default mapping.
10586 CombinedInfo.Mappers.push_back(nullptr);
10587 }
10588};
10589} // anonymous namespace
10590
10591// Try to extract the base declaration from a `this->x` expression if possible.
10593 if (!E)
10594 return nullptr;
10595
10596 if (const auto *OASE = dyn_cast<ArraySectionExpr>(E->IgnoreParenCasts()))
10597 if (const MemberExpr *ME =
10598 dyn_cast<MemberExpr>(OASE->getBase()->IgnoreParenImpCasts()))
10599 return ME->getMemberDecl();
10600 return nullptr;
10601}
10602
10603/// Emit a string constant containing the names of the values mapped to the
10604/// offloading runtime library.
10605static llvm::Constant *
10606emitMappingInformation(CodeGenFunction &CGF, llvm::OpenMPIRBuilder &OMPBuilder,
10607 MappableExprsHandler::MappingExprInfo &MapExprs) {
10608
10609 uint32_t SrcLocStrSize;
10610 if (!MapExprs.getMapDecl() && !MapExprs.getMapExpr())
10611 return OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize);
10612
10613 SourceLocation Loc;
10614 if (!MapExprs.getMapDecl() && MapExprs.getMapExpr()) {
10615 if (const ValueDecl *VD = getDeclFromThisExpr(MapExprs.getMapExpr()))
10616 Loc = VD->getLocation();
10617 else
10618 Loc = MapExprs.getMapExpr()->getExprLoc();
10619 } else {
10620 Loc = MapExprs.getMapDecl()->getLocation();
10621 }
10622
10623 std::string ExprName;
10624 if (MapExprs.getMapExpr()) {
10626 llvm::raw_string_ostream OS(ExprName);
10627 MapExprs.getMapExpr()->printPretty(OS, nullptr, P);
10628 } else {
10629 ExprName = MapExprs.getMapDecl()->getNameAsString();
10630 }
10631
10632 std::string FileName;
10634 if (auto *DbgInfo = CGF.getDebugInfo())
10635 FileName = DbgInfo->remapDIPath(PLoc.getFilename());
10636 else
10637 FileName = PLoc.getFilename();
10638 return OMPBuilder.getOrCreateSrcLocStr(FileName, ExprName, PLoc.getLine(),
10639 PLoc.getColumn(), SrcLocStrSize);
10640}
10641/// Emit the arrays used to pass the captures and map information to the
10642/// offloading runtime library. If there is no map or capture information,
10643/// return nullptr by reference.
10645 CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo,
10646 CGOpenMPRuntime::TargetDataInfo &Info, llvm::OpenMPIRBuilder &OMPBuilder,
10647 bool IsNonContiguous = false, bool ForEndCall = false) {
10648 CodeGenModule &CGM = CGF.CGM;
10649
10650 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
10651 InsertPointTy AllocaIP(CGF.AllocaInsertPt->getIterator());
10652 InsertPointTy CodeGenIP(CGF.Builder.GetInsertPoint());
10653
10654 auto DeviceAddrCB = [&](unsigned int I, llvm::Value *NewDecl) {
10655 if (const ValueDecl *DevVD = CombinedInfo.DevicePtrDecls[I]) {
10656 Info.CaptureDeviceAddrMap.try_emplace(DevVD, NewDecl);
10657 }
10658 };
10659
10660 auto CustomMapperCB = [&](unsigned int I) {
10661 llvm::Function *MFunc = nullptr;
10662 if (CombinedInfo.Mappers[I]) {
10663 Info.HasMapper = true;
10665 cast<OMPDeclareMapperDecl>(CombinedInfo.Mappers[I]));
10666 }
10667 return MFunc;
10668 };
10669 cantFail(OMPBuilder.emitOffloadingArraysAndArgs(
10670 AllocaIP, CodeGenIP, Info, Info.RTArgs, CombinedInfo, CustomMapperCB,
10671 IsNonContiguous, ForEndCall, DeviceAddrCB));
10672}
10673
10674/// Check for inner distribute directive.
10675static const OMPExecutableDirective *
10677 const auto *CS = D.getInnermostCapturedStmt();
10678 const auto *Body =
10679 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
10680 const Stmt *ChildStmt =
10682
10683 if (const auto *NestedDir =
10684 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
10685 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind();
10686 switch (D.getDirectiveKind()) {
10687 case OMPD_target:
10688 // For now, treat 'target' with nested 'teams loop' as if it's
10689 // distributed (target teams distribute).
10690 if (isOpenMPDistributeDirective(DKind) || DKind == OMPD_teams_loop)
10691 return NestedDir;
10692 if (DKind == OMPD_teams) {
10693 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers(
10694 /*IgnoreCaptured=*/true);
10695 if (!Body)
10696 return nullptr;
10697 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
10698 if (const auto *NND =
10699 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
10700 DKind = NND->getDirectiveKind();
10701 if (isOpenMPDistributeDirective(DKind))
10702 return NND;
10703 }
10704 }
10705 return nullptr;
10706 case OMPD_target_teams:
10707 if (isOpenMPDistributeDirective(DKind))
10708 return NestedDir;
10709 return nullptr;
10710 case OMPD_target_parallel:
10711 case OMPD_target_simd:
10712 case OMPD_target_parallel_for:
10713 case OMPD_target_parallel_for_simd:
10714 return nullptr;
10715 case OMPD_target_teams_distribute:
10716 case OMPD_target_teams_distribute_simd:
10717 case OMPD_target_teams_distribute_parallel_for:
10718 case OMPD_target_teams_distribute_parallel_for_simd:
10719 case OMPD_parallel:
10720 case OMPD_for:
10721 case OMPD_parallel_for:
10722 case OMPD_parallel_master:
10723 case OMPD_parallel_sections:
10724 case OMPD_for_simd:
10725 case OMPD_parallel_for_simd:
10726 case OMPD_cancel:
10727 case OMPD_cancellation_point:
10728 case OMPD_ordered_standalone:
10729 case OMPD_ordered_blockassoc:
10730 case OMPD_threadprivate:
10731 case OMPD_allocate:
10732 case OMPD_task:
10733 case OMPD_simd:
10734 case OMPD_tile:
10735 case OMPD_unroll:
10736 case OMPD_sections:
10737 case OMPD_section:
10738 case OMPD_single:
10739 case OMPD_master:
10740 case OMPD_critical:
10741 case OMPD_taskyield:
10742 case OMPD_barrier:
10743 case OMPD_taskwait:
10744 case OMPD_taskgroup:
10745 case OMPD_atomic:
10746 case OMPD_flush:
10747 case OMPD_depobj:
10748 case OMPD_scan:
10749 case OMPD_teams:
10750 case OMPD_target_data:
10751 case OMPD_target_exit_data:
10752 case OMPD_target_enter_data:
10753 case OMPD_distribute:
10754 case OMPD_distribute_simd:
10755 case OMPD_distribute_parallel_for:
10756 case OMPD_distribute_parallel_for_simd:
10757 case OMPD_teams_distribute:
10758 case OMPD_teams_distribute_simd:
10759 case OMPD_teams_distribute_parallel_for:
10760 case OMPD_teams_distribute_parallel_for_simd:
10761 case OMPD_target_update:
10762 case OMPD_declare_simd:
10763 case OMPD_declare_variant:
10764 case OMPD_begin_declare_variant:
10765 case OMPD_end_declare_variant:
10766 case OMPD_declare_target:
10767 case OMPD_end_declare_target:
10768 case OMPD_declare_reduction:
10769 case OMPD_declare_mapper:
10770 case OMPD_taskloop:
10771 case OMPD_taskloop_simd:
10772 case OMPD_master_taskloop:
10773 case OMPD_master_taskloop_simd:
10774 case OMPD_parallel_master_taskloop:
10775 case OMPD_parallel_master_taskloop_simd:
10776 case OMPD_requires:
10777 case OMPD_metadirective:
10778 case OMPD_unknown:
10779 default:
10780 llvm_unreachable("Unexpected directive.");
10781 }
10782 }
10783
10784 return nullptr;
10785}
10786
10787/// Emit the user-defined mapper function. The code generation follows the
10788/// pattern in the example below.
10789/// \code
10790/// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle,
10791/// void *base, void *begin,
10792/// int64_t size, int64_t type,
10793/// void *name = nullptr) {
10794/// // Allocate space for an array section first.
10795/// if ((size > 1 || (base != begin)) && !maptype.IsDelete)
10796/// __tgt_push_mapper_component(rt_mapper_handle, base, begin,
10797/// size*sizeof(Ty), clearToFromMember(type));
10798/// // Map members.
10799/// for (unsigned i = 0; i < size; i++) {
10800/// N = __tgt_mapper_num_components(rt_mapper_handle);
10801/// // For each component specified by this mapper:
10802/// for (auto c : begin[i]->all_components) {
10803/// // MEMBER_OF grouping: tie this component to the current array element
10804/// // (component N) by adding N<<48. Exceptions:
10805/// // - ATTACH entries are not members of any struct storage range.
10806/// // - Pointee entries (reached via a pointer member) occupy separate
10807/// // storage; their inner MEMBER_OF bits are shifted by N instead.
10808/// if (c.isAttach() || c.isPointee())
10809/// member_type = c.arg_type + (c.hasInnerMemberOf() ? N<<48 : 0);
10810/// else
10811/// member_type = c.arg_type + N<<48;
10812/// // Map-type-modifying bits (ALWAYS, DELETE, CLOSE) from the outer map
10813/// // clause are propagated to each component, except ATTACH entries
10814/// // (ATTACH|ALWAYS is reserved for attach(always), and other modifier
10815/// // bits have no meaning for ATTACH). PRESENT is additionally
10816/// // propagated to components with HasAttachPtr (the pointee data) at
10817/// // OpenMP >= 6.0.
10818/// present_bit = (v60 && c.hasAttachPtr()) ? PRESENT : 0;
10819/// imported_modifier_bits =
10820/// type & (ALWAYS | DELETE | CLOSE | present_bit);
10821/// effective_type = c.isAttach() ? member_type
10822/// : member_type | imported_modifier_bits;
10823/// if (c.hasMapper())
10824/// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size,
10825/// effective_type, c.arg_name);
10826/// else
10827/// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base,
10828/// c.arg_begin, c.arg_size, effective_type,
10829/// c.arg_name);
10830/// }
10831/// }
10832/// // Delete the array section.
10833/// if (size > 1 && maptype.IsDelete)
10834/// __tgt_push_mapper_component(rt_mapper_handle, base, begin,
10835/// size*sizeof(Ty), clearToFromMember(type));
10836/// }
10837/// \endcode
10839 CodeGenFunction *CGF) {
10840 if (UDMMap.count(D) > 0)
10841 return;
10842 ASTContext &C = CGM.getContext();
10843 QualType Ty = D->getType();
10844 auto *MapperVarDecl =
10846 CharUnits ElementSize = C.getTypeSizeInChars(Ty);
10847 llvm::Type *ElemTy = CGM.getTypes().ConvertTypeForMem(Ty);
10848
10849 CodeGenFunction MapperCGF(CGM);
10850 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
10851 auto PrivatizeAndGenMapInfoCB =
10852 [&](llvm::OpenMPIRBuilder::InsertPointTy CodeGenIP, llvm::Value *PtrPHI,
10853 llvm::Value *BeginArg) -> llvm::OpenMPIRBuilder::MapInfosTy & {
10854 MapperCGF.Builder.restoreIP(CodeGenIP);
10855
10856 // Privatize the declared variable of mapper to be the current array
10857 // element.
10858 Address PtrCurrent(
10859 PtrPHI, ElemTy,
10860 Address(BeginArg, MapperCGF.VoidPtrTy, CGM.getPointerAlign())
10861 .getAlignment()
10862 .alignmentOfArrayElement(ElementSize));
10864 Scope.addPrivate(MapperVarDecl, PtrCurrent);
10865 (void)Scope.Privatize();
10866
10867 // Get map clause information.
10868 MappableExprsHandler MEHandler(*D, MapperCGF);
10869 MEHandler.generateAllInfoForMapper(CombinedInfo, OMPBuilder);
10870
10871 auto FillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) {
10872 return emitMappingInformation(MapperCGF, OMPBuilder, MapExpr);
10873 };
10874 if (CGM.getCodeGenOpts().getDebugInfo() !=
10875 llvm::codegenoptions::NoDebugInfo) {
10876 CombinedInfo.Names.resize(CombinedInfo.Exprs.size());
10877 llvm::transform(CombinedInfo.Exprs, CombinedInfo.Names.begin(),
10878 FillInfoMap);
10879 }
10880
10881 return CombinedInfo;
10882 };
10883
10884 auto CustomMapperCB = [&](unsigned I) {
10885 llvm::Function *MapperFunc = nullptr;
10886 if (CombinedInfo.Mappers[I]) {
10887 // Call the corresponding mapper function.
10889 cast<OMPDeclareMapperDecl>(CombinedInfo.Mappers[I]));
10890 assert(MapperFunc && "Expect a valid mapper function is available.");
10891 }
10892 return MapperFunc;
10893 };
10894
10895 SmallString<64> TyStr;
10896 llvm::raw_svector_ostream Out(TyStr);
10897 CGM.getCXXABI().getMangleContext().mangleCanonicalTypeName(Ty, Out);
10898 std::string Name = getName({"omp_mapper", TyStr, D->getName()});
10899
10900 // Propagate the PRESENT modifier to the pointee entries (those with
10901 // HasAttachPtr) only for OpenMP >= 6.0; before 6.0 the present modifier does
10902 // not apply to the pointee (see the OpenMP 6.0 erratum on the present motion
10903 // vs. map-type modifier divergence).
10904 bool PropagatePresentToPointee = CGM.getLangOpts().OpenMP >= 60;
10905 llvm::Function *NewFn = cantFail(OMPBuilder.emitUserDefinedMapper(
10906 PrivatizeAndGenMapInfoCB, ElemTy, Name, CustomMapperCB,
10907 /*PreserveMemberOfFlags=*/false, PropagatePresentToPointee));
10908 UDMMap.try_emplace(D, NewFn);
10909 if (CGF)
10910 FunctionUDMMap[CGF->CurFn].push_back(D);
10911}
10912
10914 const OMPDeclareMapperDecl *D) {
10915 auto I = UDMMap.find(D);
10916 if (I != UDMMap.end())
10917 return I->second;
10919 return UDMMap.lookup(D);
10920}
10921
10924 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
10925 const OMPLoopDirective &D)>
10926 SizeEmitter) {
10927 OpenMPDirectiveKind Kind = D.getDirectiveKind();
10928 const OMPExecutableDirective *TD = &D;
10929 // Get nested teams distribute kind directive, if any. For now, treat
10930 // 'target_teams_loop' as if it's really a target_teams_distribute.
10931 if ((!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) &&
10932 Kind != OMPD_target_teams_loop)
10933 TD = getNestedDistributeDirective(CGM.getContext(), D);
10934 if (!TD)
10935 return llvm::ConstantInt::get(CGF.Int64Ty, 0);
10936
10937 const auto *LD = cast<OMPLoopDirective>(TD);
10938 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD))
10939 return NumIterations;
10940 return llvm::ConstantInt::get(CGF.Int64Ty, 0);
10941}
10942
10943static void
10944emitTargetCallFallback(CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn,
10945 const OMPExecutableDirective &D,
10947 bool RequiresOuterTask, const CapturedStmt &CS,
10948 bool OffloadingMandatory, CodeGenFunction &CGF) {
10949 if (OffloadingMandatory) {
10950 CGF.Builder.CreateUnreachable();
10951 } else {
10952 if (RequiresOuterTask) {
10953 CapturedVars.clear();
10954 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
10955 }
10956 llvm::SmallVector<llvm::Value *, 16> Args(CapturedVars.begin(),
10957 CapturedVars.end());
10958 Args.push_back(llvm::Constant::getNullValue(CGF.Builder.getPtrTy()));
10959 OMPRuntime->emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn,
10960 Args);
10961 }
10962}
10963
10964static llvm::Value *emitDeviceID(
10965 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
10966 CodeGenFunction &CGF) {
10967 // Emit device ID if any.
10968 llvm::Value *DeviceID;
10969 if (Device.getPointer()) {
10970 assert((Device.getInt() == OMPC_DEVICE_unknown ||
10971 Device.getInt() == OMPC_DEVICE_device_num) &&
10972 "Expected device_num modifier.");
10973 llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer());
10974 DeviceID =
10975 CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true);
10976 } else {
10977 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10978 }
10979 return DeviceID;
10980}
10981
10982static std::pair<llvm::Value *, OMPDynGroupprivateFallbackType>
10984 llvm::Value *DynGP = CGF.Builder.getInt32(0);
10985 auto DynGPFallback = OMPDynGroupprivateFallbackType::Abort;
10986
10987 if (auto *DynGPClause = D.getSingleClause<OMPDynGroupprivateClause>()) {
10988 CodeGenFunction::RunCleanupsScope DynGPScope(CGF);
10989 llvm::Value *DynGPVal =
10990 CGF.EmitScalarExpr(DynGPClause->getSize(), /*IgnoreResultAssign=*/true);
10991 DynGP = CGF.Builder.CreateIntCast(DynGPVal, CGF.Int32Ty,
10992 /*isSigned=*/false);
10993 auto FallbackModifier = DynGPClause->getDynGroupprivateFallbackModifier();
10994 switch (FallbackModifier) {
10995 case OMPC_DYN_GROUPPRIVATE_FALLBACK_abort:
10996 DynGPFallback = OMPDynGroupprivateFallbackType::Abort;
10997 break;
10998 case OMPC_DYN_GROUPPRIVATE_FALLBACK_null:
10999 DynGPFallback = OMPDynGroupprivateFallbackType::Null;
11000 break;
11001 case OMPC_DYN_GROUPPRIVATE_FALLBACK_default_mem:
11003 // This is the default for dyn_groupprivate.
11004 DynGPFallback = OMPDynGroupprivateFallbackType::DefaultMem;
11005 break;
11006 default:
11007 llvm_unreachable("Unknown fallback modifier for OpenMP dyn_groupprivate");
11008 }
11009 } else if (auto *OMPXDynCGClause =
11010 D.getSingleClause<OMPXDynCGroupMemClause>()) {
11011 CodeGenFunction::RunCleanupsScope DynCGMemScope(CGF);
11012 llvm::Value *DynCGMemVal = CGF.EmitScalarExpr(OMPXDynCGClause->getSize(),
11013 /*IgnoreResultAssign=*/true);
11014 DynGP = CGF.Builder.CreateIntCast(DynCGMemVal, CGF.Int32Ty,
11015 /*isSigned=*/false);
11016 }
11017 return {DynGP, DynGPFallback};
11018}
11019
11021 MappableExprsHandler &MEHandler, CodeGenFunction &CGF,
11022 const CapturedStmt &CS, llvm::SmallVectorImpl<llvm::Value *> &CapturedVars,
11023 llvm::OpenMPIRBuilder &OMPBuilder,
11024 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &MappedVarSet,
11025 MappableExprsHandler::MapCombinedInfoTy &CombinedInfo) {
11026
11027 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers;
11028 auto RI = CS.getCapturedRecordDecl()->field_begin();
11029 auto *CV = CapturedVars.begin();
11031 CE = CS.capture_end();
11032 CI != CE; ++CI, ++RI, ++CV) {
11033 MappableExprsHandler::MapCombinedInfoTy CurInfo;
11034
11035 // VLA sizes are passed to the outlined region by copy and do not have map
11036 // information associated.
11037 if (CI->capturesVariableArrayType()) {
11038 CurInfo.Exprs.push_back(nullptr);
11039 CurInfo.BasePointers.push_back(*CV);
11040 CurInfo.DevicePtrDecls.push_back(nullptr);
11041 CurInfo.DevicePointers.push_back(
11042 MappableExprsHandler::DeviceInfoTy::None);
11043 CurInfo.Pointers.push_back(*CV);
11044 CurInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
11045 CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true));
11046 // Copy to the device as an argument. No need to retrieve it.
11047 CurInfo.Types.push_back(OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
11048 OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM |
11049 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT);
11050 CurInfo.HasAttachPtr.push_back(false);
11051 CurInfo.Mappers.push_back(nullptr);
11052 } else {
11053 const ValueDecl *CapturedVD =
11054 CI->capturesThis() ? nullptr
11056 bool HasEntryWithCVAsAttachPtr = false;
11057 if (CapturedVD)
11058 HasEntryWithCVAsAttachPtr =
11059 MEHandler.hasAttachEntryForCapturedVar(CapturedVD);
11060
11061 // Populate component lists for the captured variable from clauses.
11062 MappableExprsHandler::MapDataArrayTy DeclComponentLists;
11065 StorageForImplicitlyAddedComponentLists;
11066 MEHandler.populateComponentListsForNonLambdaCaptureFromClauses(
11067 CapturedVD, DeclComponentLists,
11068 StorageForImplicitlyAddedComponentLists);
11069
11070 // OpenMP 6.0, 15.8, target construct, restrictions:
11071 // * A list item in a map clause that is specified on a target construct
11072 // must have a base variable or base pointer.
11073 //
11074 // Map clauses on a target construct must either have a base pointer, or a
11075 // base-variable. So, if we don't have a base-pointer, that means that it
11076 // must have a base-variable, i.e. we have a map like `map(s)`, `map(s.x)`
11077 // etc. In such cases, we do not need to handle default map generation
11078 // for `s`.
11079 bool HasEntryWithoutAttachPtr =
11080 llvm::any_of(DeclComponentLists, [&](const auto &MapData) {
11082 Components = std::get<0>(MapData);
11083 return !MEHandler.getAttachPtrExpr(Components);
11084 });
11085
11086 // Generate default map info first if there's no direct map with CV as
11087 // the base-variable, or attach pointer.
11088 if (DeclComponentLists.empty() ||
11089 (!HasEntryWithCVAsAttachPtr && !HasEntryWithoutAttachPtr))
11090 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurInfo);
11091
11092 // If we have any information in the map clause, we use it, otherwise we
11093 // just do a default mapping.
11094 MEHandler.generateInfoForCaptureFromClauseInfo(
11095 DeclComponentLists, CI, *CV, CurInfo, OMPBuilder,
11096 /*OffsetForMemberOfFlag=*/CombinedInfo.BasePointers.size());
11097
11098 if (!CI->capturesThis())
11099 MappedVarSet.insert(CI->getCapturedVar());
11100 else
11101 MappedVarSet.insert(nullptr);
11102
11103 // Generate correct mapping for variables captured by reference in
11104 // lambdas.
11105 if (CI->capturesVariable())
11106 MEHandler.generateInfoForLambdaCaptures(CI->getCapturedVar(), *CV,
11107 CurInfo, LambdaPointers);
11108 }
11109 // We expect to have at least an element of information for this capture.
11110 assert(!CurInfo.BasePointers.empty() &&
11111 "Non-existing map pointer for capture!");
11112 assert(CurInfo.BasePointers.size() == CurInfo.Pointers.size() &&
11113 CurInfo.BasePointers.size() == CurInfo.Sizes.size() &&
11114 CurInfo.BasePointers.size() == CurInfo.Types.size() &&
11115 CurInfo.BasePointers.size() == CurInfo.Mappers.size() &&
11116 "Inconsistent map information sizes!");
11117
11118 // We need to append the results of this capture to what we already have.
11119 CombinedInfo.append(CurInfo);
11120 }
11121 // Adjust MEMBER_OF flags for the lambdas captures.
11122 MEHandler.adjustMemberOfForLambdaCaptures(
11123 OMPBuilder, LambdaPointers, CombinedInfo.BasePointers,
11124 CombinedInfo.Pointers, CombinedInfo.Types);
11125}
11126static void
11127genMapInfo(MappableExprsHandler &MEHandler, CodeGenFunction &CGF,
11128 MappableExprsHandler::MapCombinedInfoTy &CombinedInfo,
11129 llvm::OpenMPIRBuilder &OMPBuilder,
11130 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkippedVarSet =
11131 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) {
11132
11133 CodeGenModule &CGM = CGF.CGM;
11134 // Map any list items in a map clause that were not captures because they
11135 // weren't referenced within the construct.
11136 MEHandler.generateAllInfo(CombinedInfo, OMPBuilder, SkippedVarSet);
11137
11138 auto FillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) {
11139 return emitMappingInformation(CGF, OMPBuilder, MapExpr);
11140 };
11141 if (CGM.getCodeGenOpts().getDebugInfo() !=
11142 llvm::codegenoptions::NoDebugInfo) {
11143 CombinedInfo.Names.resize(CombinedInfo.Exprs.size());
11144 llvm::transform(CombinedInfo.Exprs, CombinedInfo.Names.begin(),
11145 FillInfoMap);
11146 }
11147}
11148
11150 const CapturedStmt &CS,
11152 llvm::OpenMPIRBuilder &OMPBuilder,
11153 MappableExprsHandler::MapCombinedInfoTy &CombinedInfo) {
11154 // Get mappable expression information.
11155 MappableExprsHandler MEHandler(D, CGF);
11156 llvm::DenseSet<CanonicalDeclPtr<const Decl>> MappedVarSet;
11157
11158 genMapInfoForCaptures(MEHandler, CGF, CS, CapturedVars, OMPBuilder,
11159 MappedVarSet, CombinedInfo);
11160 genMapInfo(MEHandler, CGF, CombinedInfo, OMPBuilder, MappedVarSet);
11161}
11162
11163template <typename ClauseTy>
11164static void
11166 const OMPExecutableDirective &D,
11168 const auto *C = D.getSingleClause<ClauseTy>();
11169 assert(!C->varlist_empty() &&
11170 "ompx_bare requires explicit num_teams and thread_limit");
11172 for (auto *E : C->varlist()) {
11173 llvm::Value *V = CGF.EmitScalarExpr(E);
11174 Values.push_back(
11175 CGF.Builder.CreateIntCast(V, CGF.Int32Ty, /*isSigned=*/true));
11176 }
11177}
11178
11180 CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn,
11181 const OMPExecutableDirective &D,
11182 llvm::SmallVectorImpl<llvm::Value *> &CapturedVars, bool RequiresOuterTask,
11183 const CapturedStmt &CS, bool OffloadingMandatory,
11184 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
11185 llvm::Value *OutlinedFnID, CodeGenFunction::OMPTargetDataInfo &InputInfo,
11186 llvm::Value *&MapTypesArray, llvm::Value *&MapNamesArray,
11187 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
11188 const OMPLoopDirective &D)>
11189 SizeEmitter,
11190 CodeGenFunction &CGF, CodeGenModule &CGM) {
11191 llvm::OpenMPIRBuilder &OMPBuilder = OMPRuntime->getOMPBuilder();
11192
11193 // Fill up the arrays with all the captured variables.
11194 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
11196 genMapInfo(D, CGF, CS, CapturedVars, OMPBuilder, CombinedInfo);
11197
11198 // Append a null entry for the implicit dyn_ptr argument.
11199 using OpenMPOffloadMappingFlags = llvm::omp::OpenMPOffloadMappingFlags;
11200 auto *NullPtr = llvm::Constant::getNullValue(CGF.Builder.getPtrTy());
11201 CombinedInfo.BasePointers.push_back(NullPtr);
11202 CombinedInfo.Pointers.push_back(NullPtr);
11203 CombinedInfo.DevicePointers.push_back(
11204 llvm::OpenMPIRBuilder::DeviceInfoTy::None);
11205 CombinedInfo.Sizes.push_back(CGF.Builder.getInt64(0));
11206 CombinedInfo.Types.push_back(OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM |
11207 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
11208 CombinedInfo.HasAttachPtr.push_back(false);
11209 if (!CombinedInfo.Names.empty())
11210 CombinedInfo.Names.push_back(NullPtr);
11211 CombinedInfo.Exprs.push_back(nullptr);
11212 CombinedInfo.Mappers.push_back(nullptr);
11213 CombinedInfo.DevicePtrDecls.push_back(nullptr);
11214
11215 emitOffloadingArraysAndArgs(CGF, CombinedInfo, Info, OMPBuilder,
11216 /*IsNonContiguous=*/true, /*ForEndCall=*/false);
11217
11218 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
11219 InputInfo.BasePointersArray = Address(Info.RTArgs.BasePointersArray,
11220 CGF.VoidPtrTy, CGM.getPointerAlign());
11221 InputInfo.PointersArray =
11222 Address(Info.RTArgs.PointersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
11223 InputInfo.SizesArray =
11224 Address(Info.RTArgs.SizesArray, CGF.Int64Ty, CGM.getPointerAlign());
11225 InputInfo.MappersArray =
11226 Address(Info.RTArgs.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
11227 MapTypesArray = Info.RTArgs.MapTypesArray;
11228 MapNamesArray = Info.RTArgs.MapNamesArray;
11229
11230 auto &&ThenGen = [&OMPRuntime, OutlinedFn, &D, &CapturedVars,
11231 RequiresOuterTask, &CS, OffloadingMandatory, Device,
11232 OutlinedFnID, &InputInfo, &MapTypesArray, &MapNamesArray,
11233 SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) {
11234 bool IsReverseOffloading = Device.getInt() == OMPC_DEVICE_ancestor;
11235
11236 if (IsReverseOffloading) {
11237 // Reverse offloading is not supported, so just execute on the host.
11238 // FIXME: This fallback solution is incorrect since it ignores the
11239 // OMP_TARGET_OFFLOAD environment variable. Instead it would be better to
11240 // assert here and ensure SEMA emits an error.
11241 emitTargetCallFallback(OMPRuntime, OutlinedFn, D, CapturedVars,
11242 RequiresOuterTask, CS, OffloadingMandatory, CGF);
11243 return;
11244 }
11245
11246 bool HasNoWait = D.hasClausesOfKind<OMPNowaitClause>();
11247 unsigned NumTargetItems = InputInfo.NumberOfTargetItems;
11248
11249 llvm::Value *BasePointersArray =
11250 InputInfo.BasePointersArray.emitRawPointer(CGF);
11251 llvm::Value *PointersArray = InputInfo.PointersArray.emitRawPointer(CGF);
11252 llvm::Value *SizesArray = InputInfo.SizesArray.emitRawPointer(CGF);
11253 llvm::Value *MappersArray = InputInfo.MappersArray.emitRawPointer(CGF);
11254
11255 auto &&EmitTargetCallFallbackCB =
11256 [&OMPRuntime, OutlinedFn, &D, &CapturedVars, RequiresOuterTask, &CS,
11257 OffloadingMandatory, &CGF](llvm::OpenMPIRBuilder::InsertPointTy IP)
11258 -> llvm::OpenMPIRBuilder::InsertPointTy {
11259 CGF.Builder.restoreIP(IP);
11260 emitTargetCallFallback(OMPRuntime, OutlinedFn, D, CapturedVars,
11261 RequiresOuterTask, CS, OffloadingMandatory, CGF);
11262 return CGF.Builder.saveIP();
11263 };
11264
11265 bool IsBare = D.hasClausesOfKind<OMPXBareClause>();
11268 if (IsBare) {
11271 NumThreads);
11272 } else {
11273 NumTeams.push_back(OMPRuntime->emitNumTeamsForTargetDirective(CGF, D));
11274 NumThreads.push_back(
11275 OMPRuntime->emitNumThreadsForTargetDirective(CGF, D));
11276 }
11277
11278 llvm::Value *DeviceID = emitDeviceID(Device, CGF);
11279 llvm::Value *RTLoc = OMPRuntime->emitUpdateLocation(CGF, D.getBeginLoc());
11280 llvm::Value *NumIterations =
11281 OMPRuntime->emitTargetNumIterationsCall(CGF, D, SizeEmitter);
11282 auto [DynCGroupMem, DynCGroupMemFallback] = emitDynCGroupMem(D, CGF);
11283 llvm::OpenMPIRBuilder::InsertPointTy AllocaIP(
11284 CGF.AllocaInsertPt->getIterator());
11285
11286 llvm::OpenMPIRBuilder::TargetDataRTArgs RTArgs(
11287 BasePointersArray, PointersArray, SizesArray, MapTypesArray,
11288 nullptr /* MapTypesArrayEnd */, MappersArray, MapNamesArray);
11289
11290 llvm::OpenMPIRBuilder::TargetKernelArgs Args(
11291 NumTargetItems, RTArgs, NumIterations, NumTeams, NumThreads,
11292 DynCGroupMem, HasNoWait, /*StrictBlocks=*/IsBare,
11293 /*StrictThreads=*/IsBare, DynCGroupMemFallback);
11294
11295 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
11296 cantFail(OMPRuntime->getOMPBuilder().emitKernelLaunch(
11297 CGF.Builder, OutlinedFnID, EmitTargetCallFallbackCB, Args, DeviceID,
11298 RTLoc, AllocaIP));
11299 CGF.Builder.restoreIP(AfterIP);
11300 };
11301
11302 if (RequiresOuterTask)
11303 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
11304 else
11305 OMPRuntime->emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
11306}
11307
11308static void
11309emitTargetCallElse(CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn,
11310 const OMPExecutableDirective &D,
11312 bool RequiresOuterTask, const CapturedStmt &CS,
11313 bool OffloadingMandatory, CodeGenFunction &CGF) {
11314
11315 // Notify that the host version must be executed.
11316 auto &&ElseGen =
11317 [&OMPRuntime, OutlinedFn, &D, &CapturedVars, RequiresOuterTask, &CS,
11318 OffloadingMandatory](CodeGenFunction &CGF, PrePostActionTy &) {
11319 emitTargetCallFallback(OMPRuntime, OutlinedFn, D, CapturedVars,
11320 RequiresOuterTask, CS, OffloadingMandatory, CGF);
11321 };
11322
11323 if (RequiresOuterTask) {
11325 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo);
11326 } else {
11327 OMPRuntime->emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen);
11328 }
11329}
11330
11333 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
11334 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
11335 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
11336 const OMPLoopDirective &D)>
11337 SizeEmitter) {
11338 if (!CGF.HaveInsertPoint())
11339 return;
11340
11341 const bool OffloadingMandatory = !CGM.getLangOpts().OpenMPIsTargetDevice &&
11342 CGM.getLangOpts().OpenMPOffloadMandatory;
11343
11344 assert((OffloadingMandatory || OutlinedFn) && "Invalid outlined function!");
11345
11346 const bool RequiresOuterTask =
11347 D.hasClausesOfKind<OMPDependClause>() ||
11348 D.hasClausesOfKind<OMPNowaitClause>() ||
11349 D.hasClausesOfKind<OMPInReductionClause>() ||
11350 (CGM.getLangOpts().OpenMP >= 51 &&
11351 needsTaskBasedThreadLimit(D.getDirectiveKind()) &&
11352 D.hasClausesOfKind<OMPThreadLimitClause>());
11354 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
11355 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF,
11356 PrePostActionTy &) {
11357 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
11358 };
11359 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen);
11360
11362 llvm::Value *MapTypesArray = nullptr;
11363 llvm::Value *MapNamesArray = nullptr;
11364
11365 auto &&TargetThenGen = [this, OutlinedFn, &D, &CapturedVars,
11366 RequiresOuterTask, &CS, OffloadingMandatory, Device,
11367 OutlinedFnID, &InputInfo, &MapTypesArray,
11368 &MapNamesArray, SizeEmitter](CodeGenFunction &CGF,
11369 PrePostActionTy &) {
11370 emitTargetCallKernelLaunch(this, OutlinedFn, D, CapturedVars,
11371 RequiresOuterTask, CS, OffloadingMandatory,
11372 Device, OutlinedFnID, InputInfo, MapTypesArray,
11373 MapNamesArray, SizeEmitter, CGF, CGM);
11374 };
11375
11376 auto &&TargetElseGen =
11377 [this, OutlinedFn, &D, &CapturedVars, RequiresOuterTask, &CS,
11378 OffloadingMandatory](CodeGenFunction &CGF, PrePostActionTy &) {
11379 emitTargetCallElse(this, OutlinedFn, D, CapturedVars, RequiresOuterTask,
11380 CS, OffloadingMandatory, CGF);
11381 };
11382
11383 // If we have a target function ID it means that we need to support
11384 // offloading, otherwise, just execute on the host. We need to execute on host
11385 // regardless of the conditional in the if clause if, e.g., the user do not
11386 // specify target triples.
11387 if (OutlinedFnID) {
11388 if (IfCond) {
11389 emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen);
11390 } else {
11391 RegionCodeGenTy ThenRCG(TargetThenGen);
11392 ThenRCG(CGF);
11393 }
11394 } else {
11395 RegionCodeGenTy ElseRCG(TargetElseGen);
11396 ElseRCG(CGF);
11397 }
11398}
11399
11401 StringRef ParentName) {
11402 if (!S)
11403 return;
11404
11405 // Register vtable from device for target data and target directives.
11406 // Add this block here since scanForTargetRegionsFunctions ignores
11407 // target data by checking if S is a executable directive (target).
11408 if (auto *E = dyn_cast<OMPExecutableDirective>(S);
11409 E && isOpenMPTargetDataManagementDirective(E->getDirectiveKind())) {
11410 // Don't need to check if it's device compile
11411 // since scanForTargetRegionsFunctions currently only called
11412 // in device compilation.
11413 registerVTable(*E);
11414 }
11415
11416 // Codegen OMP target directives that offload compute to the device.
11417 bool RequiresDeviceCodegen =
11420 cast<OMPExecutableDirective>(S)->getDirectiveKind());
11421
11422 if (RequiresDeviceCodegen) {
11423 const auto &E = *cast<OMPExecutableDirective>(S);
11424
11425 llvm::TargetRegionEntryInfo EntryInfo = getEntryInfoFromPresumedLoc(
11426 CGM, OMPBuilder, E.getBeginLoc(), ParentName);
11427
11428 // Is this a target region that should not be emitted as an entry point? If
11429 // so just signal we are done with this target region.
11430 if (!OMPBuilder.OffloadInfoManager.hasTargetRegionEntryInfo(EntryInfo))
11431 return;
11432
11433 switch (E.getDirectiveKind()) {
11434 case OMPD_target:
11437 break;
11438 case OMPD_target_parallel:
11440 CGM, ParentName, cast<OMPTargetParallelDirective>(E));
11441 break;
11442 case OMPD_target_teams:
11444 CGM, ParentName, cast<OMPTargetTeamsDirective>(E));
11445 break;
11446 case OMPD_target_teams_distribute:
11449 break;
11450 case OMPD_target_teams_distribute_simd:
11453 break;
11454 case OMPD_target_parallel_for:
11457 break;
11458 case OMPD_target_parallel_for_simd:
11461 break;
11462 case OMPD_target_simd:
11464 CGM, ParentName, cast<OMPTargetSimdDirective>(E));
11465 break;
11466 case OMPD_target_teams_distribute_parallel_for:
11468 CGM, ParentName,
11470 break;
11471 case OMPD_target_teams_distribute_parallel_for_simd:
11474 CGM, ParentName,
11476 break;
11477 case OMPD_target_teams_loop:
11480 break;
11481 case OMPD_target_parallel_loop:
11484 break;
11485 case OMPD_parallel:
11486 case OMPD_for:
11487 case OMPD_parallel_for:
11488 case OMPD_parallel_master:
11489 case OMPD_parallel_sections:
11490 case OMPD_for_simd:
11491 case OMPD_parallel_for_simd:
11492 case OMPD_cancel:
11493 case OMPD_cancellation_point:
11494 case OMPD_ordered_standalone:
11495 case OMPD_ordered_blockassoc:
11496 case OMPD_threadprivate:
11497 case OMPD_allocate:
11498 case OMPD_task:
11499 case OMPD_simd:
11500 case OMPD_tile:
11501 case OMPD_unroll:
11502 case OMPD_sections:
11503 case OMPD_section:
11504 case OMPD_single:
11505 case OMPD_master:
11506 case OMPD_critical:
11507 case OMPD_taskyield:
11508 case OMPD_barrier:
11509 case OMPD_taskwait:
11510 case OMPD_taskgroup:
11511 case OMPD_atomic:
11512 case OMPD_flush:
11513 case OMPD_depobj:
11514 case OMPD_scan:
11515 case OMPD_teams:
11516 case OMPD_target_data:
11517 case OMPD_target_exit_data:
11518 case OMPD_target_enter_data:
11519 case OMPD_distribute:
11520 case OMPD_distribute_simd:
11521 case OMPD_distribute_parallel_for:
11522 case OMPD_distribute_parallel_for_simd:
11523 case OMPD_teams_distribute:
11524 case OMPD_teams_distribute_simd:
11525 case OMPD_teams_distribute_parallel_for:
11526 case OMPD_teams_distribute_parallel_for_simd:
11527 case OMPD_target_update:
11528 case OMPD_declare_simd:
11529 case OMPD_declare_variant:
11530 case OMPD_begin_declare_variant:
11531 case OMPD_end_declare_variant:
11532 case OMPD_declare_target:
11533 case OMPD_end_declare_target:
11534 case OMPD_declare_reduction:
11535 case OMPD_declare_mapper:
11536 case OMPD_taskloop:
11537 case OMPD_taskloop_simd:
11538 case OMPD_master_taskloop:
11539 case OMPD_master_taskloop_simd:
11540 case OMPD_parallel_master_taskloop:
11541 case OMPD_parallel_master_taskloop_simd:
11542 case OMPD_requires:
11543 case OMPD_metadirective:
11544 case OMPD_unknown:
11545 default:
11546 llvm_unreachable("Unknown target directive for OpenMP device codegen.");
11547 }
11548 return;
11549 }
11550
11551 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) {
11552 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt())
11553 return;
11554
11555 scanForTargetRegionsFunctions(E->getRawStmt(), ParentName);
11556 return;
11557 }
11558
11559 // If this is a lambda function, look into its body.
11560 if (const auto *L = dyn_cast<LambdaExpr>(S))
11561 S = L->getBody();
11562
11563 // Keep looking for target regions recursively.
11564 for (const Stmt *II : S->children())
11565 scanForTargetRegionsFunctions(II, ParentName);
11566}
11567
11568static bool isAssumedToBeNotEmitted(const ValueDecl *VD, bool IsDevice) {
11569 std::optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
11570 OMPDeclareTargetDeclAttr::getDeviceType(VD);
11571 if (!DevTy)
11572 return false;
11573 // Do not emit device_type(nohost) functions for the host.
11574 if (!IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_NoHost)
11575 return true;
11576 // Do not emit device_type(host) functions for the device.
11577 if (IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_Host)
11578 return true;
11579 return false;
11580}
11581
11583 // If emitting code for the host, we do not process FD here. Instead we do
11584 // the normal code generation.
11585 if (!CGM.getLangOpts().OpenMPIsTargetDevice) {
11586 if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl()))
11588 CGM.getLangOpts().OpenMPIsTargetDevice))
11589 return true;
11590 return false;
11591 }
11592
11593 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl());
11594 // Try to detect target regions in the function.
11595 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) {
11596 StringRef Name = CGM.getMangledName(GD);
11599 CGM.getLangOpts().OpenMPIsTargetDevice))
11600 return true;
11601 }
11602
11603 // Do not emit function if it is not marked as declare target.
11604 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) &&
11605 AlreadyEmittedTargetDecls.count(VD) == 0;
11606}
11607
11610 CGM.getLangOpts().OpenMPIsTargetDevice))
11611 return true;
11612
11613 if (!CGM.getLangOpts().OpenMPIsTargetDevice)
11614 return false;
11615
11616 // Check if there are Ctors/Dtors in this declaration and look for target
11617 // regions in it. We use the complete variant to produce the kernel name
11618 // mangling.
11619 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType();
11620 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) {
11621 for (const CXXConstructorDecl *Ctor : RD->ctors()) {
11622 StringRef ParentName =
11623 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete));
11624 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName);
11625 }
11626 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) {
11627 StringRef ParentName =
11628 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete));
11629 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName);
11630 }
11631 }
11632
11633 // Do not emit variable if it is not marked as declare target.
11634 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
11635 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(
11636 cast<VarDecl>(GD.getDecl()));
11637 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
11638 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
11639 *Res == OMPDeclareTargetDeclAttr::MT_Enter) &&
11642 return true;
11643 }
11644 return false;
11645}
11646
11648 llvm::Constant *Addr) {
11649 if (CGM.getLangOpts().OMPTargetTriples.empty() &&
11650 !CGM.getLangOpts().OpenMPIsTargetDevice)
11651 return;
11652
11653 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
11654 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
11655
11656 // If this is an 'extern' declaration we defer to the canonical definition and
11657 // do not emit an offloading entry.
11658 if (Res && *Res != OMPDeclareTargetDeclAttr::MT_Link &&
11659 VD->hasExternalStorage())
11660 return;
11661
11662 // MT_Local variables use direct access with no host-device mapping.
11663 // No offload entry needed — the device global keeps its own initializer.
11664 if (Res && *Res == OMPDeclareTargetDeclAttr::MT_Local)
11665 return;
11666
11667 if (!Res) {
11668 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
11669 // Register non-target variables being emitted in device code (debug info
11670 // may cause this).
11671 StringRef VarName = CGM.getMangledName(VD);
11672 EmittedNonTargetVariables.try_emplace(VarName, Addr);
11673 }
11674 return;
11675 }
11676
11677 auto AddrOfGlobal = [&VD, this]() { return CGM.GetAddrOfGlobal(VD); };
11678 auto LinkageForVariable = [&VD, this]() {
11679 return CGM.getLLVMLinkageVarDefinition(VD);
11680 };
11681
11682 std::vector<llvm::GlobalVariable *> GeneratedRefs;
11683 OMPBuilder.registerTargetGlobalVariable(
11685 VD->hasDefinition(CGM.getContext()) == VarDecl::DeclarationOnly,
11686 VD->isExternallyVisible(),
11688 VD->getCanonicalDecl()->getBeginLoc()),
11689 CGM.getMangledName(VD), GeneratedRefs, CGM.getLangOpts().OpenMPSimd,
11690 CGM.getLangOpts().OMPTargetTriples, AddrOfGlobal, LinkageForVariable,
11691 CGM.getTypes().ConvertTypeForMem(
11692 CGM.getContext().getPointerType(VD->getType())),
11693 Addr);
11694
11695 for (auto *ref : GeneratedRefs)
11696 CGM.addCompilerUsedGlobal(ref);
11697}
11698
11700 if (isa<FunctionDecl>(GD.getDecl()) ||
11702 return emitTargetFunctions(GD);
11703
11704 return emitTargetGlobalVariable(GD);
11705}
11706
11708 for (const VarDecl *VD : DeferredGlobalVariables) {
11709 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
11710 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
11711 if (!Res)
11712 continue;
11713 // MT_Local and MT_To/MT_Enter without USM are always emitted.
11714 if (*Res == OMPDeclareTargetDeclAttr::MT_Local ||
11715 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
11716 *Res == OMPDeclareTargetDeclAttr::MT_Enter) &&
11718 CGM.EmitGlobal(VD);
11719 } else {
11720 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link ||
11721 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
11722 *Res == OMPDeclareTargetDeclAttr::MT_Enter ||
11723 *Res == OMPDeclareTargetDeclAttr::MT_Local) &&
11725 "Expected link clause or to clause with unified memory.");
11726 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
11727 }
11728 }
11729}
11730
11732 CodeGenFunction &CGF, const OMPExecutableDirective &D) const {
11733 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) &&
11734 " Expected target-based directive.");
11735}
11736
11738 for (const OMPClause *Clause : D->clauselists()) {
11739 if (Clause->getClauseKind() == OMPC_unified_shared_memory) {
11741 OMPBuilder.Config.setHasRequiresUnifiedSharedMemory(true);
11742 } else if (const auto *AC =
11743 dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) {
11744 switch (AC->getAtomicDefaultMemOrderKind()) {
11745 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel:
11746 RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease;
11747 break;
11748 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst:
11749 RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent;
11750 break;
11751 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed:
11752 RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic;
11753 break;
11755 break;
11756 }
11757 }
11758 }
11759}
11760
11761llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const {
11763}
11764
11766 LangAS &AS) {
11767 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())
11768 return false;
11769 const auto *A = VD->getAttr<OMPAllocateDeclAttr>();
11770 switch(A->getAllocatorType()) {
11771 case OMPAllocateDeclAttr::OMPNullMemAlloc:
11772 case OMPAllocateDeclAttr::OMPDefaultMemAlloc:
11773 // Not supported, fallback to the default mem space.
11774 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc:
11775 case OMPAllocateDeclAttr::OMPCGroupMemAlloc:
11776 case OMPAllocateDeclAttr::OMPHighBWMemAlloc:
11777 case OMPAllocateDeclAttr::OMPLowLatMemAlloc:
11778 case OMPAllocateDeclAttr::OMPThreadMemAlloc:
11779 case OMPAllocateDeclAttr::OMPConstMemAlloc:
11780 case OMPAllocateDeclAttr::OMPPTeamMemAlloc:
11781 AS = LangAS::Default;
11782 return true;
11783 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc:
11784 llvm_unreachable("Expected predefined allocator for the variables with the "
11785 "static storage.");
11786 }
11787 return false;
11788}
11789
11793
11795 CodeGenModule &CGM)
11796 : CGM(CGM) {
11797 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
11798 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal;
11799 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false;
11800 }
11801}
11802
11804 if (CGM.getLangOpts().OpenMPIsTargetDevice)
11805 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal;
11806}
11807
11809 if (!CGM.getLangOpts().OpenMPIsTargetDevice || !ShouldMarkAsGlobal)
11810 return true;
11811
11812 const auto *D = cast<FunctionDecl>(GD.getDecl());
11813 // Do not emit function if it is marked as declare target as it was already
11814 // emitted.
11815 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) {
11816 if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) {
11817 if (auto *F = dyn_cast_or_null<llvm::Function>(
11818 CGM.GetGlobalValue(CGM.getMangledName(GD))))
11819 return !F->isDeclaration();
11820 return false;
11821 }
11822 return true;
11823 }
11824
11825 return !AlreadyEmittedTargetDecls.insert(D).second;
11826}
11827
11829 const OMPExecutableDirective &D,
11830 SourceLocation Loc,
11831 llvm::Function *OutlinedFn,
11832 ArrayRef<llvm::Value *> CapturedVars) {
11833 if (!CGF.HaveInsertPoint())
11834 return;
11835
11836 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11838
11839 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn);
11840 llvm::Value *Args[] = {
11841 RTLoc,
11842 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
11843 OutlinedFn};
11845 RealArgs.append(std::begin(Args), std::end(Args));
11846 RealArgs.append(CapturedVars.begin(), CapturedVars.end());
11847
11848 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction(
11849 CGM.getModule(), OMPRTL___kmpc_fork_teams);
11850 CGF.EmitRuntimeCall(RTLFn, RealArgs);
11851}
11852
11854 const Expr *NumTeams,
11855 const Expr *ThreadLimit,
11856 SourceLocation Loc) {
11857 if (!CGF.HaveInsertPoint())
11858 return;
11859
11860 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11861
11862 llvm::Value *NumTeamsVal =
11863 NumTeams
11864 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams),
11865 CGF.CGM.Int32Ty, /* isSigned = */ true)
11866 : CGF.Builder.getInt32(0);
11867
11868 llvm::Value *ThreadLimitVal =
11869 ThreadLimit
11870 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit),
11871 CGF.CGM.Int32Ty, /* isSigned = */ true)
11872 : CGF.Builder.getInt32(0);
11873
11874 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit)
11875 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal,
11876 ThreadLimitVal};
11877 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
11878 CGM.getModule(), OMPRTL___kmpc_push_num_teams),
11879 PushNumTeamsArgs);
11880}
11881
11883 const Expr *ThreadLimit,
11884 SourceLocation Loc) {
11885 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11886 llvm::Value *ThreadLimitVal =
11887 ThreadLimit
11888 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit),
11889 CGF.CGM.Int32Ty, /* isSigned = */ true)
11890 : CGF.Builder.getInt32(0);
11891
11892 // Build call __kmpc_set_thread_limit(&loc, global_tid, thread_limit)
11893 llvm::Value *ThreadLimitArgs[] = {RTLoc, getThreadID(CGF, Loc),
11894 ThreadLimitVal};
11895 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
11896 CGM.getModule(), OMPRTL___kmpc_set_thread_limit),
11897 ThreadLimitArgs);
11898}
11899
11901 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
11902 const Expr *Device, const RegionCodeGenTy &CodeGen,
11904 if (!CGF.HaveInsertPoint())
11905 return;
11906
11907 // Action used to replace the default codegen action and turn privatization
11908 // off.
11909 PrePostActionTy NoPrivAction;
11910
11911 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
11912
11913 llvm::Value *IfCondVal = nullptr;
11914 if (IfCond)
11915 IfCondVal = CGF.EvaluateExprAsBool(IfCond);
11916
11917 // Emit device ID if any.
11918 llvm::Value *DeviceID = nullptr;
11919 if (Device) {
11920 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
11921 CGF.Int64Ty, /*isSigned=*/true);
11922 } else {
11923 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
11924 }
11925
11926 // Fill up the arrays with all the mapped variables.
11927 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
11928 auto GenMapInfoCB =
11929 [&](InsertPointTy CodeGenIP) -> llvm::OpenMPIRBuilder::MapInfosTy & {
11930 CGF.Builder.restoreIP(CodeGenIP);
11931 // Get map clause information.
11932 MappableExprsHandler MEHandler(D, CGF);
11933 MEHandler.generateAllInfo(CombinedInfo, OMPBuilder);
11934
11935 auto FillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) {
11936 return emitMappingInformation(CGF, OMPBuilder, MapExpr);
11937 };
11938 if (CGM.getCodeGenOpts().getDebugInfo() !=
11939 llvm::codegenoptions::NoDebugInfo) {
11940 CombinedInfo.Names.resize(CombinedInfo.Exprs.size());
11941 llvm::transform(CombinedInfo.Exprs, CombinedInfo.Names.begin(),
11942 FillInfoMap);
11943 }
11944
11945 return CombinedInfo;
11946 };
11947 using BodyGenTy = llvm::OpenMPIRBuilder::BodyGenTy;
11948 auto BodyCB = [&](InsertPointTy CodeGenIP, BodyGenTy BodyGenType) {
11949 CGF.Builder.restoreIP(CodeGenIP);
11950 switch (BodyGenType) {
11951 case BodyGenTy::Priv:
11952 if (!Info.CaptureDeviceAddrMap.empty())
11953 CodeGen(CGF);
11954 break;
11955 case BodyGenTy::DupNoPriv:
11956 if (!Info.CaptureDeviceAddrMap.empty()) {
11957 CodeGen.setAction(NoPrivAction);
11958 CodeGen(CGF);
11959 }
11960 break;
11961 case BodyGenTy::NoPriv:
11962 if (Info.CaptureDeviceAddrMap.empty()) {
11963 CodeGen.setAction(NoPrivAction);
11964 CodeGen(CGF);
11965 }
11966 break;
11967 }
11968 return InsertPointTy(CGF.Builder.GetInsertPoint());
11969 };
11970
11971 auto DeviceAddrCB = [&](unsigned int I, llvm::Value *NewDecl) {
11972 if (const ValueDecl *DevVD = CombinedInfo.DevicePtrDecls[I]) {
11973 Info.CaptureDeviceAddrMap.try_emplace(DevVD, NewDecl);
11974 }
11975 };
11976
11977 auto CustomMapperCB = [&](unsigned int I) {
11978 llvm::Function *MFunc = nullptr;
11979 if (CombinedInfo.Mappers[I]) {
11980 Info.HasMapper = true;
11982 cast<OMPDeclareMapperDecl>(CombinedInfo.Mappers[I]));
11983 }
11984 return MFunc;
11985 };
11986
11987 // Source location for the ident struct
11988 llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc());
11989
11990 InsertPointTy AllocaIP(CGF.AllocaInsertPt->getIterator());
11991 InsertPointTy CodeGenIP(CGF.Builder.GetInsertPoint());
11992 llvm::OpenMPIRBuilder::LocationDescription OmpLoc(CGF.Builder);
11993 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
11994 cantFail(OMPBuilder.createTargetData(
11995 OmpLoc, AllocaIP, CodeGenIP, /*DeallocBlocks=*/{}, DeviceID,
11996 IfCondVal, Info, GenMapInfoCB, CustomMapperCB,
11997 /*MapperFunc=*/nullptr, BodyCB, DeviceAddrCB, RTLoc));
11998 CGF.Builder.restoreIP(AfterIP);
11999}
12000
12002 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12003 const Expr *Device) {
12004 if (!CGF.HaveInsertPoint())
12005 return;
12006
12010 "Expecting either target enter, exit data, or update directives.");
12011
12013 llvm::Value *MapTypesArray = nullptr;
12014 llvm::Value *MapNamesArray = nullptr;
12015 // Generate the code for the opening of the data environment.
12016 auto &&ThenGen = [this, &D, Device, &InputInfo, &MapTypesArray,
12017 &MapNamesArray](CodeGenFunction &CGF, PrePostActionTy &) {
12018 // Emit device ID if any.
12019 llvm::Value *DeviceID = nullptr;
12020 if (Device) {
12021 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
12022 CGF.Int64Ty, /*isSigned=*/true);
12023 } else {
12024 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
12025 }
12026
12027 // Emit the number of elements in the offloading arrays.
12028 llvm::Constant *PointerNum =
12029 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
12030
12031 // Source location for the ident struct
12032 llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc());
12033
12034 SmallVector<llvm::Value *, 13> OffloadingArgs(
12035 {RTLoc, DeviceID, PointerNum,
12036 InputInfo.BasePointersArray.emitRawPointer(CGF),
12037 InputInfo.PointersArray.emitRawPointer(CGF),
12038 InputInfo.SizesArray.emitRawPointer(CGF), MapTypesArray, MapNamesArray,
12039 InputInfo.MappersArray.emitRawPointer(CGF)});
12040
12041 // Select the right runtime function call for each standalone
12042 // directive.
12043 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
12044 RuntimeFunction RTLFn;
12045 switch (D.getDirectiveKind()) {
12046 case OMPD_target_enter_data:
12047 RTLFn = HasNowait ? OMPRTL___tgt_target_data_begin_nowait_mapper
12048 : OMPRTL___tgt_target_data_begin_mapper;
12049 break;
12050 case OMPD_target_exit_data:
12051 RTLFn = HasNowait ? OMPRTL___tgt_target_data_end_nowait_mapper
12052 : OMPRTL___tgt_target_data_end_mapper;
12053 break;
12054 case OMPD_target_update:
12055 RTLFn = HasNowait ? OMPRTL___tgt_target_data_update_nowait_mapper
12056 : OMPRTL___tgt_target_data_update_mapper;
12057 break;
12058 case OMPD_parallel:
12059 case OMPD_for:
12060 case OMPD_parallel_for:
12061 case OMPD_parallel_master:
12062 case OMPD_parallel_sections:
12063 case OMPD_for_simd:
12064 case OMPD_parallel_for_simd:
12065 case OMPD_cancel:
12066 case OMPD_cancellation_point:
12067 case OMPD_ordered_standalone:
12068 case OMPD_ordered_blockassoc:
12069 case OMPD_threadprivate:
12070 case OMPD_allocate:
12071 case OMPD_task:
12072 case OMPD_simd:
12073 case OMPD_tile:
12074 case OMPD_unroll:
12075 case OMPD_sections:
12076 case OMPD_section:
12077 case OMPD_single:
12078 case OMPD_master:
12079 case OMPD_critical:
12080 case OMPD_taskyield:
12081 case OMPD_barrier:
12082 case OMPD_taskwait:
12083 case OMPD_taskgroup:
12084 case OMPD_atomic:
12085 case OMPD_flush:
12086 case OMPD_depobj:
12087 case OMPD_scan:
12088 case OMPD_teams:
12089 case OMPD_target_data:
12090 case OMPD_distribute:
12091 case OMPD_distribute_simd:
12092 case OMPD_distribute_parallel_for:
12093 case OMPD_distribute_parallel_for_simd:
12094 case OMPD_teams_distribute:
12095 case OMPD_teams_distribute_simd:
12096 case OMPD_teams_distribute_parallel_for:
12097 case OMPD_teams_distribute_parallel_for_simd:
12098 case OMPD_declare_simd:
12099 case OMPD_declare_variant:
12100 case OMPD_begin_declare_variant:
12101 case OMPD_end_declare_variant:
12102 case OMPD_declare_target:
12103 case OMPD_end_declare_target:
12104 case OMPD_declare_reduction:
12105 case OMPD_declare_mapper:
12106 case OMPD_taskloop:
12107 case OMPD_taskloop_simd:
12108 case OMPD_master_taskloop:
12109 case OMPD_master_taskloop_simd:
12110 case OMPD_parallel_master_taskloop:
12111 case OMPD_parallel_master_taskloop_simd:
12112 case OMPD_target:
12113 case OMPD_target_simd:
12114 case OMPD_target_teams_distribute:
12115 case OMPD_target_teams_distribute_simd:
12116 case OMPD_target_teams_distribute_parallel_for:
12117 case OMPD_target_teams_distribute_parallel_for_simd:
12118 case OMPD_target_teams:
12119 case OMPD_target_parallel:
12120 case OMPD_target_parallel_for:
12121 case OMPD_target_parallel_for_simd:
12122 case OMPD_requires:
12123 case OMPD_metadirective:
12124 case OMPD_unknown:
12125 default:
12126 llvm_unreachable("Unexpected standalone target data directive.");
12127 break;
12128 }
12129 if (HasNowait) {
12130 OffloadingArgs.push_back(llvm::Constant::getNullValue(CGF.Int32Ty));
12131 OffloadingArgs.push_back(llvm::Constant::getNullValue(CGF.VoidPtrTy));
12132 OffloadingArgs.push_back(llvm::Constant::getNullValue(CGF.Int32Ty));
12133 OffloadingArgs.push_back(llvm::Constant::getNullValue(CGF.VoidPtrTy));
12134 }
12135 CGF.EmitRuntimeCall(
12136 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), RTLFn),
12137 OffloadingArgs);
12138 };
12139
12140 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
12141 &MapNamesArray](CodeGenFunction &CGF,
12142 PrePostActionTy &) {
12143 // Fill up the arrays with all the mapped variables.
12144 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
12146 MappableExprsHandler MEHandler(D, CGF);
12147 genMapInfo(MEHandler, CGF, CombinedInfo, OMPBuilder);
12148 emitOffloadingArraysAndArgs(CGF, CombinedInfo, Info, OMPBuilder,
12149 /*IsNonContiguous=*/true, /*ForEndCall=*/false);
12150
12151 bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() ||
12152 D.hasClausesOfKind<OMPNowaitClause>();
12153
12154 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
12155 InputInfo.BasePointersArray = Address(Info.RTArgs.BasePointersArray,
12156 CGF.VoidPtrTy, CGM.getPointerAlign());
12157 InputInfo.PointersArray = Address(Info.RTArgs.PointersArray, CGF.VoidPtrTy,
12158 CGM.getPointerAlign());
12159 InputInfo.SizesArray =
12160 Address(Info.RTArgs.SizesArray, CGF.Int64Ty, CGM.getPointerAlign());
12161 InputInfo.MappersArray =
12162 Address(Info.RTArgs.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
12163 MapTypesArray = Info.RTArgs.MapTypesArray;
12164 MapNamesArray = Info.RTArgs.MapNamesArray;
12165 if (RequiresOuterTask)
12166 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
12167 else
12168 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
12169 };
12170
12171 if (IfCond) {
12172 emitIfClause(CGF, IfCond, TargetThenGen,
12173 [](CodeGenFunction &CGF, PrePostActionTy &) {});
12174 } else {
12175 RegionCodeGenTy ThenRCG(TargetThenGen);
12176 ThenRCG(CGF);
12177 }
12178}
12179
12180static unsigned
12183 // Every vector variant of a SIMD-enabled function has a vector length (VLEN).
12184 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument
12185 // of that clause. The VLEN value must be power of 2.
12186 // In other case the notion of the function`s "characteristic data type" (CDT)
12187 // is used to compute the vector length.
12188 // CDT is defined in the following order:
12189 // a) For non-void function, the CDT is the return type.
12190 // b) If the function has any non-uniform, non-linear parameters, then the
12191 // CDT is the type of the first such parameter.
12192 // c) If the CDT determined by a) or b) above is struct, union, or class
12193 // type which is pass-by-value (except for the type that maps to the
12194 // built-in complex data type), the characteristic data type is int.
12195 // d) If none of the above three cases is applicable, the CDT is int.
12196 // The VLEN is then determined based on the CDT and the size of vector
12197 // register of that ISA for which current vector version is generated. The
12198 // VLEN is computed using the formula below:
12199 // VLEN = sizeof(vector_register) / sizeof(CDT),
12200 // where vector register size specified in section 3.2.1 Registers and the
12201 // Stack Frame of original AMD64 ABI document.
12202 QualType RetType = FD->getReturnType();
12203 if (RetType.isNull())
12204 return 0;
12205 ASTContext &C = FD->getASTContext();
12206 QualType CDT;
12207 if (!RetType.isNull() && !RetType->isVoidType()) {
12208 CDT = RetType;
12209 } else {
12210 unsigned Offset = 0;
12211 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) {
12212 if (ParamAttrs[Offset].Kind ==
12213 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Vector)
12214 CDT = C.getPointerType(C.getCanonicalTagType(MD->getParent()));
12215 ++Offset;
12216 }
12217 if (CDT.isNull()) {
12218 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
12219 if (ParamAttrs[I + Offset].Kind ==
12220 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Vector) {
12221 CDT = FD->getParamDecl(I)->getType();
12222 break;
12223 }
12224 }
12225 }
12226 }
12227 if (CDT.isNull())
12228 CDT = C.IntTy;
12229 CDT = CDT->getCanonicalTypeUnqualified();
12230 if (CDT->isRecordType() || CDT->isUnionType())
12231 CDT = C.IntTy;
12232 return C.getTypeSize(CDT);
12233}
12234
12235// This are the Functions that are needed to mangle the name of the
12236// vector functions generated by the compiler, according to the rules
12237// defined in the "Vector Function ABI specifications for AArch64",
12238// available at
12239// https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi.
12240
12241/// Maps To Vector (MTV), as defined in 4.1.1 of the AAVFABI (2021Q1).
12243 llvm::OpenMPIRBuilder::DeclareSimdKindTy Kind) {
12244 QT = QT.getCanonicalType();
12245
12246 if (QT->isVoidType())
12247 return false;
12248
12249 if (Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::Uniform)
12250 return false;
12251
12252 if (Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearUVal ||
12253 Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearRef)
12254 return false;
12255
12256 if ((Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::Linear ||
12257 Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearVal) &&
12258 !QT->isReferenceType())
12259 return false;
12260
12261 return true;
12262}
12263
12264/// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
12266 QT = QT.getCanonicalType();
12267 unsigned Size = C.getTypeSize(QT);
12268
12269 // Only scalars and complex within 16 bytes wide set PVB to true.
12270 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128)
12271 return false;
12272
12273 if (QT->isFloatingType())
12274 return true;
12275
12276 if (QT->isIntegerType())
12277 return true;
12278
12279 if (QT->isPointerType())
12280 return true;
12281
12282 // TODO: Add support for complex types (section 3.1.2, item 2).
12283
12284 return false;
12285}
12286
12287/// Computes the lane size (LS) of a return type or of an input parameter,
12288/// as defined by `LS(P)` in 3.2.1 of the AAVFABI.
12289/// TODO: Add support for references, section 3.2.1, item 1.
12290static unsigned getAArch64LS(QualType QT,
12291 llvm::OpenMPIRBuilder::DeclareSimdKindTy Kind,
12292 ASTContext &C) {
12293 if (!getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) {
12295 if (getAArch64PBV(PTy, C))
12296 return C.getTypeSize(PTy);
12297 }
12298 if (getAArch64PBV(QT, C))
12299 return C.getTypeSize(QT);
12300
12301 return C.getTypeSize(C.getUIntPtrType());
12302}
12303
12304// Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the
12305// signature of the scalar function, as defined in 3.2.2 of the
12306// AAVFABI.
12307static std::tuple<unsigned, unsigned, bool>
12310 QualType RetType = FD->getReturnType().getCanonicalType();
12311
12312 ASTContext &C = FD->getASTContext();
12313
12314 bool OutputBecomesInput = false;
12315
12317 if (!RetType->isVoidType()) {
12318 Sizes.push_back(getAArch64LS(
12319 RetType, llvm::OpenMPIRBuilder::DeclareSimdKindTy::Vector, C));
12320 if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {}))
12321 OutputBecomesInput = true;
12322 }
12323 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
12325 Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C));
12326 }
12327
12328 assert(!Sizes.empty() && "Unable to determine NDS and WDS.");
12329 // The LS of a function parameter / return value can only be a power
12330 // of 2, starting from 8 bits, up to 128.
12331 assert(llvm::all_of(Sizes,
12332 [](unsigned Size) {
12333 return Size == 8 || Size == 16 || Size == 32 ||
12334 Size == 64 || Size == 128;
12335 }) &&
12336 "Invalid size");
12337
12338 return std::make_tuple(*llvm::min_element(Sizes), *llvm::max_element(Sizes),
12339 OutputBecomesInput);
12340}
12341
12342static llvm::OpenMPIRBuilder::DeclareSimdBranch
12343convertDeclareSimdBranch(OMPDeclareSimdDeclAttr::BranchStateTy State) {
12344 switch (State) {
12345 case OMPDeclareSimdDeclAttr::BS_Undefined:
12346 return llvm::OpenMPIRBuilder::DeclareSimdBranch::Undefined;
12347 case OMPDeclareSimdDeclAttr::BS_Inbranch:
12348 return llvm::OpenMPIRBuilder::DeclareSimdBranch::Inbranch;
12349 case OMPDeclareSimdDeclAttr::BS_Notinbranch:
12350 return llvm::OpenMPIRBuilder::DeclareSimdBranch::Notinbranch;
12351 }
12352 llvm_unreachable("unexpected declare simd branch state");
12353}
12354
12355// Check the values provided via `simdlen` by the user.
12357 unsigned UserVLEN, unsigned WDS, char ISA) {
12358 // 1. A `simdlen(1)` doesn't produce vector signatures.
12359 if (UserVLEN == 1) {
12360 CGM.getDiags().Report(SLoc, diag::warn_simdlen_1_no_effect);
12361 return false;
12362 }
12363
12364 // 2. Section 3.3.1, item 1: user input must be a power of 2 for Advanced
12365 // SIMD.
12366 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) {
12367 CGM.getDiags().Report(SLoc, diag::warn_simdlen_requires_power_of_2);
12368 return false;
12369 }
12370
12371 // 3. Section 3.4.1: SVE fixed length must obey the architectural limits.
12372 if (ISA == 's' && UserVLEN != 0 &&
12373 ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0))) {
12374 CGM.getDiags().Report(SLoc, diag::warn_simdlen_must_fit_lanes) << WDS;
12375 return false;
12376 }
12377
12378 return true;
12379}
12380
12382 llvm::Function *Fn) {
12383 ASTContext &C = CGM.getContext();
12384 FD = FD->getMostRecentDecl();
12385 while (FD) {
12386 // Map params to their positions in function decl.
12387 llvm::DenseMap<const Decl *, unsigned> ParamPositions;
12388 if (isa<CXXMethodDecl>(FD))
12389 ParamPositions.try_emplace(FD, 0);
12390 unsigned ParamPos = ParamPositions.size();
12391 for (const ParmVarDecl *P : FD->parameters()) {
12392 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos);
12393 ++ParamPos;
12394 }
12395 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) {
12397 ParamPositions.size());
12398 // Mark uniform parameters.
12399 for (const Expr *E : Attr->uniforms()) {
12400 E = E->IgnoreParenImpCasts();
12401 unsigned Pos;
12402 if (isa<CXXThisExpr>(E)) {
12403 Pos = ParamPositions[FD];
12404 } else {
12405 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
12406 ->getCanonicalDecl();
12407 auto It = ParamPositions.find(PVD);
12408 assert(It != ParamPositions.end() && "Function parameter not found");
12409 Pos = It->second;
12410 }
12411 ParamAttrs[Pos].Kind =
12412 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Uniform;
12413 }
12414 // Get alignment info.
12415 auto *NI = Attr->alignments_begin();
12416 for (const Expr *E : Attr->aligneds()) {
12417 E = E->IgnoreParenImpCasts();
12418 unsigned Pos;
12419 QualType ParmTy;
12420 if (isa<CXXThisExpr>(E)) {
12421 Pos = ParamPositions[FD];
12422 ParmTy = E->getType();
12423 } else {
12424 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
12425 ->getCanonicalDecl();
12426 auto It = ParamPositions.find(PVD);
12427 assert(It != ParamPositions.end() && "Function parameter not found");
12428 Pos = It->second;
12429 ParmTy = PVD->getType();
12430 }
12431 ParamAttrs[Pos].Alignment =
12432 (*NI)
12433 ? (*NI)->EvaluateKnownConstInt(C)
12434 : llvm::APSInt::getUnsigned(
12435 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy))
12436 .getQuantity());
12437 ++NI;
12438 }
12439 // Mark linear parameters.
12440 auto *SI = Attr->steps_begin();
12441 auto *MI = Attr->modifiers_begin();
12442 for (const Expr *E : Attr->linears()) {
12443 E = E->IgnoreParenImpCasts();
12444 unsigned Pos;
12445 bool IsReferenceType = false;
12446 // Rescaling factor needed to compute the linear parameter
12447 // value in the mangled name.
12448 unsigned PtrRescalingFactor = 1;
12449 if (isa<CXXThisExpr>(E)) {
12450 Pos = ParamPositions[FD];
12451 auto *P = cast<PointerType>(E->getType());
12452 PtrRescalingFactor = CGM.getContext()
12453 .getTypeSizeInChars(P->getPointeeType())
12454 .getQuantity();
12455 } else {
12456 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
12457 ->getCanonicalDecl();
12458 auto It = ParamPositions.find(PVD);
12459 assert(It != ParamPositions.end() && "Function parameter not found");
12460 Pos = It->second;
12461 if (auto *P = dyn_cast<PointerType>(PVD->getType()))
12462 PtrRescalingFactor = CGM.getContext()
12463 .getTypeSizeInChars(P->getPointeeType())
12464 .getQuantity();
12465 else if (PVD->getType()->isReferenceType()) {
12466 IsReferenceType = true;
12467 PtrRescalingFactor =
12468 CGM.getContext()
12469 .getTypeSizeInChars(PVD->getType().getNonReferenceType())
12470 .getQuantity();
12471 }
12472 }
12473 llvm::OpenMPIRBuilder::DeclareSimdAttrTy &ParamAttr = ParamAttrs[Pos];
12474 if (*MI == OMPC_LINEAR_ref)
12475 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearRef;
12476 else if (*MI == OMPC_LINEAR_uval)
12477 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearUVal;
12478 else if (IsReferenceType)
12479 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearVal;
12480 else
12481 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::Linear;
12482 // Assuming a stride of 1, for `linear` without modifiers.
12483 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(1);
12484 if (*SI) {
12486 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) {
12487 if (const auto *DRE =
12488 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) {
12489 if (const auto *StridePVD =
12490 dyn_cast<ParmVarDecl>(DRE->getDecl())) {
12491 ParamAttr.HasVarStride = true;
12492 auto It = ParamPositions.find(StridePVD->getCanonicalDecl());
12493 assert(It != ParamPositions.end() &&
12494 "Function parameter not found");
12495 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(It->second);
12496 }
12497 }
12498 } else {
12499 ParamAttr.StrideOrArg = Result.Val.getInt();
12500 }
12501 }
12502 // If we are using a linear clause on a pointer, we need to
12503 // rescale the value of linear_step with the byte size of the
12504 // pointee type.
12505 if (!ParamAttr.HasVarStride &&
12506 (ParamAttr.Kind ==
12507 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Linear ||
12508 ParamAttr.Kind ==
12509 llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearRef))
12510 ParamAttr.StrideOrArg = ParamAttr.StrideOrArg * PtrRescalingFactor;
12511 ++SI;
12512 ++MI;
12513 }
12514 llvm::APSInt VLENVal;
12515 SourceLocation ExprLoc;
12516 const Expr *VLENExpr = Attr->getSimdlen();
12517 if (VLENExpr) {
12518 VLENVal = VLENExpr->EvaluateKnownConstInt(C);
12519 ExprLoc = VLENExpr->getExprLoc();
12520 }
12521 llvm::OpenMPIRBuilder::DeclareSimdBranch State =
12522 convertDeclareSimdBranch(Attr->getBranchState());
12523 if (CGM.getTriple().isX86()) {
12524 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs);
12525 assert(NumElts && "Non-zero simdlen/cdtsize expected");
12526 OMPBuilder.emitX86DeclareSimdFunction(Fn, NumElts, VLENVal, ParamAttrs,
12527 State);
12528 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) {
12529 unsigned VLEN = VLENVal.getExtValue();
12530 // Get basic data for building the vector signature.
12531 const auto Data = getNDSWDS(FD, ParamAttrs);
12532 const unsigned NDS = std::get<0>(Data);
12533 const unsigned WDS = std::get<1>(Data);
12534 const bool OutputBecomesInput = std::get<2>(Data);
12535 if (CGM.getTarget().hasFeature("sve")) {
12536 if (validateAArch64Simdlen(CGM, ExprLoc, VLEN, WDS, 's'))
12537 OMPBuilder.emitAArch64DeclareSimdFunction(
12538 Fn, VLEN, ParamAttrs, State, 's', NDS, OutputBecomesInput);
12539 } else if (CGM.getTarget().hasFeature("neon")) {
12540 if (validateAArch64Simdlen(CGM, ExprLoc, VLEN, WDS, 'n'))
12541 OMPBuilder.emitAArch64DeclareSimdFunction(
12542 Fn, VLEN, ParamAttrs, State, 'n', NDS, OutputBecomesInput);
12543 }
12544 }
12545 }
12546 FD = FD->getPreviousDecl();
12547 }
12548}
12549
12550namespace {
12551/// Cleanup action for doacross support.
12552class DoacrossCleanupTy final : public EHScopeStack::Cleanup {
12553public:
12554 static const int DoacrossFinArgs = 2;
12555
12556private:
12557 llvm::FunctionCallee RTLFn;
12558 llvm::Value *Args[DoacrossFinArgs];
12559
12560public:
12561 DoacrossCleanupTy(llvm::FunctionCallee RTLFn,
12562 ArrayRef<llvm::Value *> CallArgs)
12563 : RTLFn(RTLFn) {
12564 assert(CallArgs.size() == DoacrossFinArgs);
12565 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
12566 }
12567 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
12568 if (!CGF.HaveInsertPoint())
12569 return;
12570 CGF.EmitRuntimeCall(RTLFn, Args);
12571 }
12572};
12573} // namespace
12574
12576 const OMPLoopDirective &D,
12577 ArrayRef<Expr *> NumIterations) {
12578 if (!CGF.HaveInsertPoint())
12579 return;
12580
12581 ASTContext &C = CGM.getContext();
12582 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
12583 RecordDecl *RD;
12584 if (KmpDimTy.isNull()) {
12585 // Build struct kmp_dim { // loop bounds info casted to kmp_int64
12586 // kmp_int64 lo; // lower
12587 // kmp_int64 up; // upper
12588 // kmp_int64 st; // stride
12589 // };
12590 RD = C.buildImplicitRecord("kmp_dim");
12591 RD->startDefinition();
12592 addFieldToRecordDecl(C, RD, Int64Ty);
12593 addFieldToRecordDecl(C, RD, Int64Ty);
12594 addFieldToRecordDecl(C, RD, Int64Ty);
12595 RD->completeDefinition();
12596 KmpDimTy = C.getCanonicalTagType(RD);
12597 } else {
12598 RD = KmpDimTy->castAsRecordDecl();
12599 }
12600 llvm::APInt Size(/*numBits=*/32, NumIterations.size());
12601 QualType ArrayTy = C.getConstantArrayType(KmpDimTy, Size, nullptr,
12603
12604 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims");
12605 CGF.EmitNullInitialization(DimsAddr, ArrayTy);
12606 enum { LowerFD = 0, UpperFD, StrideFD };
12607 // Fill dims with data.
12608 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) {
12609 LValue DimsLVal = CGF.MakeAddrLValue(
12610 CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy);
12611 // dims.upper = num_iterations;
12612 LValue UpperLVal = CGF.EmitLValueForField(
12613 DimsLVal, *std::next(RD->field_begin(), UpperFD));
12614 llvm::Value *NumIterVal = CGF.EmitScalarConversion(
12615 CGF.EmitScalarExpr(NumIterations[I]), NumIterations[I]->getType(),
12616 Int64Ty, NumIterations[I]->getExprLoc());
12617 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal);
12618 // dims.stride = 1;
12619 LValue StrideLVal = CGF.EmitLValueForField(
12620 DimsLVal, *std::next(RD->field_begin(), StrideFD));
12621 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1),
12622 StrideLVal);
12623 }
12624
12625 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid,
12626 // kmp_int32 num_dims, struct kmp_dim * dims);
12627 llvm::Value *Args[] = {
12628 emitUpdateLocation(CGF, D.getBeginLoc()),
12629 getThreadID(CGF, D.getBeginLoc()),
12630 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()),
12632 CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).emitRawPointer(CGF),
12633 CGM.VoidPtrTy)};
12634
12635 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12636 CGM.getModule(), OMPRTL___kmpc_doacross_init);
12637 CGF.EmitRuntimeCall(RTLFn, Args);
12638 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = {
12639 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())};
12640 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12641 CGM.getModule(), OMPRTL___kmpc_doacross_fini);
12642 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
12643 llvm::ArrayRef(FiniArgs));
12644}
12645
12646template <typename T>
12648 const T *C, llvm::Value *ULoc,
12649 llvm::Value *ThreadID) {
12650 QualType Int64Ty =
12651 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
12652 llvm::APInt Size(/*numBits=*/32, C->getNumLoops());
12654 Int64Ty, Size, nullptr, ArraySizeModifier::Normal, 0);
12655 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr");
12656 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) {
12657 const Expr *CounterVal = C->getLoopData(I);
12658 assert(CounterVal);
12659 llvm::Value *CntVal = CGF.EmitScalarConversion(
12660 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty,
12661 CounterVal->getExprLoc());
12662 CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I),
12663 /*Volatile=*/false, Int64Ty);
12664 }
12665 llvm::Value *Args[] = {
12666 ULoc, ThreadID,
12667 CGF.Builder.CreateConstArrayGEP(CntAddr, 0).emitRawPointer(CGF)};
12668 llvm::FunctionCallee RTLFn;
12669 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
12670 OMPDoacrossKind<T> ODK;
12671 if (ODK.isSource(C)) {
12672 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
12673 OMPRTL___kmpc_doacross_post);
12674 } else {
12675 assert(ODK.isSink(C) && "Expect sink modifier.");
12676 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
12677 OMPRTL___kmpc_doacross_wait);
12678 }
12679 CGF.EmitRuntimeCall(RTLFn, Args);
12680}
12681
12683 const OMPDependClause *C) {
12685 CGF, CGM, C, emitUpdateLocation(CGF, C->getBeginLoc()),
12686 getThreadID(CGF, C->getBeginLoc()));
12687}
12688
12690 const OMPDoacrossClause *C) {
12692 CGF, CGM, C, emitUpdateLocation(CGF, C->getBeginLoc()),
12693 getThreadID(CGF, C->getBeginLoc()));
12694}
12695
12697 llvm::FunctionCallee Callee,
12698 ArrayRef<llvm::Value *> Args) const {
12699 assert(Loc.isValid() && "Outlined function call location must be valid.");
12701
12702 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) {
12703 if (Fn->doesNotThrow()) {
12704 CGF.EmitNounwindRuntimeCall(Fn, Args);
12705 return;
12706 }
12707 }
12708 CGF.EmitRuntimeCall(Callee, Args);
12709}
12710
12712 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn,
12713 ArrayRef<llvm::Value *> Args) const {
12714 emitCall(CGF, Loc, OutlinedFn, Args);
12715}
12716
12718 if (const auto *FD = dyn_cast<FunctionDecl>(D))
12719 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD))
12721}
12722
12724 const VarDecl *NativeParam,
12725 const VarDecl *TargetParam) const {
12726 return CGF.GetAddrOfLocalVar(NativeParam);
12727}
12728
12729/// Return allocator value from expression, or return a null allocator (default
12730/// when no allocator specified).
12731static llvm::Value *getAllocatorVal(CodeGenFunction &CGF,
12732 const Expr *Allocator) {
12733 llvm::Value *AllocVal;
12734 if (Allocator) {
12735 AllocVal = CGF.EmitScalarExpr(Allocator);
12736 // According to the standard, the original allocator type is a enum
12737 // (integer). Convert to pointer type, if required.
12738 AllocVal = CGF.EmitScalarConversion(AllocVal, Allocator->getType(),
12739 CGF.getContext().VoidPtrTy,
12740 Allocator->getExprLoc());
12741 } else {
12742 // If no allocator specified, it defaults to the null allocator.
12743 AllocVal = llvm::Constant::getNullValue(
12745 }
12746 return AllocVal;
12747}
12748
12749/// Return the alignment from an allocate directive if present.
12750static llvm::Value *getAlignmentValue(CodeGenModule &CGM, const VarDecl *VD) {
12751 std::optional<CharUnits> AllocateAlignment = CGM.getOMPAllocateAlignment(VD);
12752
12753 if (!AllocateAlignment)
12754 return nullptr;
12755
12756 return llvm::ConstantInt::get(CGM.SizeTy, AllocateAlignment->getQuantity());
12757}
12758
12760 const VarDecl *VD) {
12761 if (!VD)
12762 return Address::invalid();
12763 Address UntiedAddr = Address::invalid();
12764 Address UntiedRealAddr = Address::invalid();
12765 auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn);
12766 if (It != FunctionToUntiedTaskStackMap.end()) {
12767 const UntiedLocalVarsAddressesMap &UntiedData =
12768 UntiedLocalVarsStack[It->second];
12769 auto I = UntiedData.find(VD);
12770 if (I != UntiedData.end()) {
12771 UntiedAddr = I->second.first;
12772 UntiedRealAddr = I->second.second;
12773 }
12774 }
12775 const VarDecl *CVD = VD->getCanonicalDecl();
12776 if (CVD->hasAttr<OMPAllocateDeclAttr>()) {
12777 // Use the default allocation.
12778 if (!isAllocatableDecl(VD))
12779 return UntiedAddr;
12780 llvm::Value *Size;
12781 CharUnits Align = CGM.getContext().getDeclAlign(CVD);
12782 if (CVD->getType()->isVariablyModifiedType()) {
12783 Size = CGF.getTypeSize(CVD->getType());
12784 // Align the size: ((size + align - 1) / align) * align
12785 Size = CGF.Builder.CreateNUWAdd(
12786 Size, CGM.getSize(Align - CharUnits::fromQuantity(1)));
12787 Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align));
12788 Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align));
12789 } else {
12790 CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType());
12791 Size = CGM.getSize(Sz.alignTo(Align));
12792 }
12793 llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc());
12794 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
12795 const Expr *Allocator = AA->getAllocator();
12796 llvm::Value *AllocVal = getAllocatorVal(CGF, Allocator);
12797 llvm::Value *Alignment = getAlignmentValue(CGM, CVD);
12799 Args.push_back(ThreadID);
12800 if (Alignment)
12801 Args.push_back(Alignment);
12802 Args.push_back(Size);
12803 Args.push_back(AllocVal);
12804 llvm::omp::RuntimeFunction FnID =
12805 Alignment ? OMPRTL___kmpc_aligned_alloc : OMPRTL___kmpc_alloc;
12806 llvm::Value *Addr = CGF.EmitRuntimeCall(
12807 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), FnID), Args,
12808 getName({CVD->getName(), ".void.addr"}));
12809 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12810 CGM.getModule(), OMPRTL___kmpc_free);
12811 QualType Ty = CGM.getContext().getPointerType(CVD->getType());
12813 Addr, CGF.ConvertTypeForMem(Ty), getName({CVD->getName(), ".addr"}));
12814 if (UntiedAddr.isValid())
12815 CGF.EmitStoreOfScalar(Addr, UntiedAddr, /*Volatile=*/false, Ty);
12816
12817 // Cleanup action for allocate support.
12818 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup {
12819 llvm::FunctionCallee RTLFn;
12820 SourceLocation::UIntTy LocEncoding;
12821 Address Addr;
12822 const Expr *AllocExpr;
12823
12824 public:
12825 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn,
12826 SourceLocation::UIntTy LocEncoding, Address Addr,
12827 const Expr *AllocExpr)
12828 : RTLFn(RTLFn), LocEncoding(LocEncoding), Addr(Addr),
12829 AllocExpr(AllocExpr) {}
12830 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
12831 if (!CGF.HaveInsertPoint())
12832 return;
12833 llvm::Value *Args[3];
12834 Args[0] = CGF.CGM.getOpenMPRuntime().getThreadID(
12835 CGF, SourceLocation::getFromRawEncoding(LocEncoding));
12837 Addr.emitRawPointer(CGF), CGF.VoidPtrTy);
12838 llvm::Value *AllocVal = getAllocatorVal(CGF, AllocExpr);
12839 Args[2] = AllocVal;
12840 CGF.EmitRuntimeCall(RTLFn, Args);
12841 }
12842 };
12843 Address VDAddr =
12844 UntiedRealAddr.isValid()
12845 ? UntiedRealAddr
12846 : Address(Addr, CGF.ConvertTypeForMem(CVD->getType()), Align);
12847 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(
12848 NormalAndEHCleanup, FiniRTLFn, CVD->getLocation().getRawEncoding(),
12849 VDAddr, Allocator);
12850 if (UntiedRealAddr.isValid())
12851 if (auto *Region =
12852 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
12853 Region->emitUntiedSwitch(CGF);
12854 return VDAddr;
12855 }
12856 return UntiedAddr;
12857}
12858
12860 const VarDecl *VD) const {
12861 auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn);
12862 if (It == FunctionToUntiedTaskStackMap.end())
12863 return false;
12864 return UntiedLocalVarsStack[It->second].count(VD) > 0;
12865}
12866
12868 CodeGenModule &CGM, const OMPLoopDirective &S)
12869 : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) {
12870 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12871 if (!NeedToPush)
12872 return;
12874 CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back();
12875 for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) {
12876 for (const Stmt *Ref : C->private_refs()) {
12877 const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts();
12878 const ValueDecl *VD;
12879 if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) {
12880 VD = DRE->getDecl();
12881 } else {
12882 const auto *ME = cast<MemberExpr>(SimpleRefExpr);
12883 assert((ME->isImplicitCXXThis() ||
12884 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) &&
12885 "Expected member of current class.");
12886 VD = ME->getMemberDecl();
12887 }
12888 DS.insert(VD);
12889 }
12890 }
12891}
12892
12894 if (!NeedToPush)
12895 return;
12896 CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back();
12897}
12898
12900 CodeGenFunction &CGF,
12901 const llvm::MapVector<CanonicalDeclPtr<const VarDecl>,
12902 std::pair<Address, Address>> &LocalVars)
12903 : CGM(CGF.CGM), NeedToPush(!LocalVars.empty()) {
12904 if (!NeedToPush)
12905 return;
12906 CGM.getOpenMPRuntime().FunctionToUntiedTaskStackMap.try_emplace(
12907 CGF.CurFn, CGM.getOpenMPRuntime().UntiedLocalVarsStack.size());
12908 CGM.getOpenMPRuntime().UntiedLocalVarsStack.push_back(LocalVars);
12909}
12910
12912 if (!NeedToPush)
12913 return;
12914 CGM.getOpenMPRuntime().UntiedLocalVarsStack.pop_back();
12915}
12916
12918 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12919
12920 return llvm::any_of(
12921 CGM.getOpenMPRuntime().NontemporalDeclsStack,
12922 [VD](const NontemporalDeclsSet &Set) { return Set.contains(VD); });
12923}
12924
12925void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis(
12926 const OMPExecutableDirective &S,
12927 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled)
12928 const {
12929 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs;
12930 // Vars in target/task regions must be excluded completely.
12931 if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) ||
12932 isOpenMPTaskingDirective(S.getDirectiveKind())) {
12934 getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind());
12935 const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front());
12936 for (const CapturedStmt::Capture &Cap : CS->captures()) {
12937 if (Cap.capturesVariable() || Cap.capturesVariableByCopy())
12938 NeedToCheckForLPCs.insert(Cap.getCapturedVar());
12939 }
12940 }
12941 // Exclude vars in private clauses.
12942 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) {
12943 for (const Expr *Ref : C->varlist()) {
12944 if (!Ref->getType()->isScalarType())
12945 continue;
12946 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12947 if (!DRE)
12948 continue;
12949 NeedToCheckForLPCs.insert(DRE->getDecl());
12950 }
12951 }
12952 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
12953 for (const Expr *Ref : C->varlist()) {
12954 if (!Ref->getType()->isScalarType())
12955 continue;
12956 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12957 if (!DRE)
12958 continue;
12959 NeedToCheckForLPCs.insert(DRE->getDecl());
12960 }
12961 }
12962 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
12963 for (const Expr *Ref : C->varlist()) {
12964 if (!Ref->getType()->isScalarType())
12965 continue;
12966 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12967 if (!DRE)
12968 continue;
12969 NeedToCheckForLPCs.insert(DRE->getDecl());
12970 }
12971 }
12972 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
12973 for (const Expr *Ref : C->varlist()) {
12974 if (!Ref->getType()->isScalarType())
12975 continue;
12976 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12977 if (!DRE)
12978 continue;
12979 NeedToCheckForLPCs.insert(DRE->getDecl());
12980 }
12981 }
12982 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) {
12983 for (const Expr *Ref : C->varlist()) {
12984 if (!Ref->getType()->isScalarType())
12985 continue;
12986 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12987 if (!DRE)
12988 continue;
12989 NeedToCheckForLPCs.insert(DRE->getDecl());
12990 }
12991 }
12992 for (const Decl *VD : NeedToCheckForLPCs) {
12993 for (const LastprivateConditionalData &Data :
12994 llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) {
12995 if (Data.DeclToUniqueName.count(VD) > 0) {
12996 if (!Data.Disabled)
12997 NeedToAddForLPCsAsDisabled.insert(VD);
12998 break;
12999 }
13000 }
13001 }
13002}
13003
13004CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
13005 CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal)
13006 : CGM(CGF.CGM),
13007 Action((CGM.getLangOpts().OpenMP >= 50 &&
13008 llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(),
13009 [](const OMPLastprivateClause *C) {
13010 return C->getKind() ==
13011 OMPC_LASTPRIVATE_conditional;
13012 }))
13013 ? ActionToDo::PushAsLastprivateConditional
13014 : ActionToDo::DoNotPush) {
13015 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
13016 if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush)
13017 return;
13018 assert(Action == ActionToDo::PushAsLastprivateConditional &&
13019 "Expected a push action.");
13021 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
13022 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
13023 if (C->getKind() != OMPC_LASTPRIVATE_conditional)
13024 continue;
13025
13026 for (const Expr *Ref : C->varlist()) {
13027 Data.DeclToUniqueName.insert(std::make_pair(
13028 cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(),
13029 SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref))));
13030 }
13031 }
13032 Data.IVLVal = IVLVal;
13033 Data.Fn = CGF.CurFn;
13034}
13035
13036CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
13038 : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) {
13039 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
13040 if (CGM.getLangOpts().OpenMP < 50)
13041 return;
13042 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled;
13043 tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled);
13044 if (!NeedToAddForLPCsAsDisabled.empty()) {
13045 Action = ActionToDo::DisableLastprivateConditional;
13046 LastprivateConditionalData &Data =
13048 for (const Decl *VD : NeedToAddForLPCsAsDisabled)
13049 Data.DeclToUniqueName.try_emplace(VD);
13050 Data.Fn = CGF.CurFn;
13051 Data.Disabled = true;
13052 }
13053}
13054
13055CGOpenMPRuntime::LastprivateConditionalRAII
13057 CodeGenFunction &CGF, const OMPExecutableDirective &S) {
13058 return LastprivateConditionalRAII(CGF, S);
13059}
13060
13062 if (CGM.getLangOpts().OpenMP < 50)
13063 return;
13064 if (Action == ActionToDo::DisableLastprivateConditional) {
13065 assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
13066 "Expected list of disabled private vars.");
13067 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
13068 }
13069 if (Action == ActionToDo::PushAsLastprivateConditional) {
13070 assert(
13071 !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
13072 "Expected list of lastprivate conditional vars.");
13073 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
13074 }
13075}
13076
13078 const VarDecl *VD) {
13079 ASTContext &C = CGM.getContext();
13080 auto I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first;
13081 QualType NewType;
13082 const FieldDecl *VDField;
13083 const FieldDecl *FiredField;
13084 LValue BaseLVal;
13085 auto VI = I->getSecond().find(VD);
13086 if (VI == I->getSecond().end()) {
13087 RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional");
13088 RD->startDefinition();
13089 VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType());
13090 FiredField = addFieldToRecordDecl(C, RD, C.CharTy);
13091 RD->completeDefinition();
13092 NewType = C.getCanonicalTagType(RD);
13093 Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName());
13094 BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl);
13095 I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal);
13096 } else {
13097 NewType = std::get<0>(VI->getSecond());
13098 VDField = std::get<1>(VI->getSecond());
13099 FiredField = std::get<2>(VI->getSecond());
13100 BaseLVal = std::get<3>(VI->getSecond());
13101 }
13102 LValue FiredLVal =
13103 CGF.EmitLValueForField(BaseLVal, FiredField);
13105 llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)),
13106 FiredLVal);
13107 return CGF.EmitLValueForField(BaseLVal, VDField).getAddress();
13108}
13109
13110namespace {
13111/// Checks if the lastprivate conditional variable is referenced in LHS.
13112class LastprivateConditionalRefChecker final
13113 : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> {
13115 const Expr *FoundE = nullptr;
13116 const Decl *FoundD = nullptr;
13117 StringRef UniqueDeclName;
13118 LValue IVLVal;
13119 llvm::Function *FoundFn = nullptr;
13120 SourceLocation Loc;
13121
13122public:
13123 bool VisitDeclRefExpr(const DeclRefExpr *E) {
13125 llvm::reverse(LPM)) {
13126 auto It = D.DeclToUniqueName.find(E->getDecl());
13127 if (It == D.DeclToUniqueName.end())
13128 continue;
13129 if (D.Disabled)
13130 return false;
13131 FoundE = E;
13132 FoundD = E->getDecl()->getCanonicalDecl();
13133 UniqueDeclName = It->second;
13134 IVLVal = D.IVLVal;
13135 FoundFn = D.Fn;
13136 break;
13137 }
13138 return FoundE == E;
13139 }
13140 bool VisitMemberExpr(const MemberExpr *E) {
13142 return false;
13144 llvm::reverse(LPM)) {
13145 auto It = D.DeclToUniqueName.find(E->getMemberDecl());
13146 if (It == D.DeclToUniqueName.end())
13147 continue;
13148 if (D.Disabled)
13149 return false;
13150 FoundE = E;
13151 FoundD = E->getMemberDecl()->getCanonicalDecl();
13152 UniqueDeclName = It->second;
13153 IVLVal = D.IVLVal;
13154 FoundFn = D.Fn;
13155 break;
13156 }
13157 return FoundE == E;
13158 }
13159 bool VisitStmt(const Stmt *S) {
13160 for (const Stmt *Child : S->children()) {
13161 if (!Child)
13162 continue;
13163 if (const auto *E = dyn_cast<Expr>(Child))
13164 if (!E->isGLValue())
13165 continue;
13166 if (Visit(Child))
13167 return true;
13168 }
13169 return false;
13170 }
13171 explicit LastprivateConditionalRefChecker(
13172 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM)
13173 : LPM(LPM) {}
13174 std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *>
13175 getFoundData() const {
13176 return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn);
13177 }
13178};
13179} // namespace
13180
13182 LValue IVLVal,
13183 StringRef UniqueDeclName,
13184 LValue LVal,
13185 SourceLocation Loc) {
13186 // Last updated loop counter for the lastprivate conditional var.
13187 // int<xx> last_iv = 0;
13188 llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType());
13189 llvm::Constant *LastIV = OMPBuilder.getOrCreateInternalVariable(
13190 LLIVTy, getName({UniqueDeclName, "iv"}));
13191 cast<llvm::GlobalVariable>(LastIV)->setAlignment(
13192 IVLVal.getAlignment().getAsAlign());
13193 LValue LastIVLVal =
13194 CGF.MakeNaturalAlignRawAddrLValue(LastIV, IVLVal.getType());
13195
13196 // Last value of the lastprivate conditional.
13197 // decltype(priv_a) last_a;
13198 llvm::GlobalVariable *Last = OMPBuilder.getOrCreateInternalVariable(
13199 CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName);
13200 cast<llvm::GlobalVariable>(Last)->setAlignment(
13201 LVal.getAlignment().getAsAlign());
13202 LValue LastLVal =
13203 CGF.MakeRawAddrLValue(Last, LVal.getType(), LVal.getAlignment());
13204
13205 // Global loop counter. Required to handle inner parallel-for regions.
13206 // iv
13207 llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc);
13208
13209 // #pragma omp critical(a)
13210 // if (last_iv <= iv) {
13211 // last_iv = iv;
13212 // last_a = priv_a;
13213 // }
13214 auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal,
13215 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
13216 Action.Enter(CGF);
13217 llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc);
13218 // (last_iv <= iv) ? Check if the variable is updated and store new
13219 // value in global var.
13220 llvm::Value *CmpRes;
13221 if (IVLVal.getType()->isSignedIntegerType()) {
13222 CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal);
13223 } else {
13224 assert(IVLVal.getType()->isUnsignedIntegerType() &&
13225 "Loop iteration variable must be integer.");
13226 CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal);
13227 }
13228 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then");
13229 llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit");
13230 CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB);
13231 // {
13232 CGF.EmitBlock(ThenBB);
13233
13234 // last_iv = iv;
13235 CGF.EmitStoreOfScalar(IVVal, LastIVLVal);
13236
13237 // last_a = priv_a;
13238 switch (CGF.getEvaluationKind(LVal.getType())) {
13239 case TEK_Scalar: {
13240 llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc);
13241 CGF.EmitStoreOfScalar(PrivVal, LastLVal);
13242 break;
13243 }
13244 case TEK_Complex: {
13245 CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc);
13246 CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false);
13247 break;
13248 }
13249 case TEK_Aggregate:
13250 llvm_unreachable(
13251 "Aggregates are not supported in lastprivate conditional.");
13252 }
13253 // }
13254 CGF.EmitBranch(ExitBB);
13255 // There is no need to emit line number for unconditional branch.
13257 CGF.EmitBlock(ExitBB, /*IsFinished=*/true);
13258 };
13259
13260 if (CGM.getLangOpts().OpenMPSimd) {
13261 // Do not emit as a critical region as no parallel region could be emitted.
13262 RegionCodeGenTy ThenRCG(CodeGen);
13263 ThenRCG(CGF);
13264 } else {
13265 emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc);
13266 }
13267}
13268
13270 const Expr *LHS) {
13271 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
13272 return;
13273 LastprivateConditionalRefChecker Checker(LastprivateConditionalStack);
13274 if (!Checker.Visit(LHS))
13275 return;
13276 const Expr *FoundE;
13277 const Decl *FoundD;
13278 StringRef UniqueDeclName;
13279 LValue IVLVal;
13280 llvm::Function *FoundFn;
13281 std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) =
13282 Checker.getFoundData();
13283 if (FoundFn != CGF.CurFn) {
13284 // Special codegen for inner parallel regions.
13285 // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1;
13286 auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD);
13287 assert(It != LastprivateConditionalToTypes[FoundFn].end() &&
13288 "Lastprivate conditional is not found in outer region.");
13289 QualType StructTy = std::get<0>(It->getSecond());
13290 const FieldDecl* FiredDecl = std::get<2>(It->getSecond());
13291 LValue PrivLVal = CGF.EmitLValue(FoundE);
13293 PrivLVal.getAddress(),
13294 CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy)),
13295 CGF.ConvertTypeForMem(StructTy));
13296 LValue BaseLVal =
13297 CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl);
13298 LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl);
13299 CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get(
13300 CGF.ConvertTypeForMem(FiredDecl->getType()), 1)),
13301 FiredLVal, llvm::AtomicOrdering::Unordered,
13302 /*IsVolatile=*/true, /*isInit=*/false);
13303 return;
13304 }
13305
13306 // Private address of the lastprivate conditional in the current context.
13307 // priv_a
13308 LValue LVal = CGF.EmitLValue(FoundE);
13309 emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal,
13310 FoundE->getExprLoc());
13311}
13312
13315 const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) {
13316 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
13317 return;
13318 auto Range = llvm::reverse(LastprivateConditionalStack);
13319 auto It = llvm::find_if(
13320 Range, [](const LastprivateConditionalData &D) { return !D.Disabled; });
13321 if (It == Range.end() || It->Fn != CGF.CurFn)
13322 return;
13323 auto LPCI = LastprivateConditionalToTypes.find(It->Fn);
13324 assert(LPCI != LastprivateConditionalToTypes.end() &&
13325 "Lastprivates must be registered already.");
13327 getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind());
13328 const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back());
13329 for (const auto &Pair : It->DeclToUniqueName) {
13330 const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl());
13331 if (!CS->capturesVariable(VD) || IgnoredDecls.contains(VD))
13332 continue;
13333 auto I = LPCI->getSecond().find(Pair.first);
13334 assert(I != LPCI->getSecond().end() &&
13335 "Lastprivate must be rehistered already.");
13336 // bool Cmp = priv_a.Fired != 0;
13337 LValue BaseLVal = std::get<3>(I->getSecond());
13338 LValue FiredLVal =
13339 CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond()));
13340 llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc());
13341 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res);
13342 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then");
13343 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done");
13344 // if (Cmp) {
13345 CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB);
13346 CGF.EmitBlock(ThenBB);
13347 Address Addr = CGF.GetAddrOfLocalVar(VD);
13348 LValue LVal;
13349 if (VD->getType()->isReferenceType())
13350 LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(),
13352 else
13353 LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(),
13355 emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal,
13356 D.getBeginLoc());
13358 CGF.EmitBlock(DoneBB, /*IsFinal=*/true);
13359 // }
13360 }
13361}
13362
13364 CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD,
13365 SourceLocation Loc) {
13366 if (CGF.getLangOpts().OpenMP < 50)
13367 return;
13368 auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD);
13369 assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() &&
13370 "Unknown lastprivate conditional variable.");
13371 StringRef UniqueName = It->second;
13372 llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName);
13373 // The variable was not updated in the region - exit.
13374 if (!GV)
13375 return;
13376 LValue LPLVal = CGF.MakeRawAddrLValue(
13377 GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment());
13378 llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc);
13379 CGF.EmitStoreOfScalar(Res, PrivLVal);
13380}
13381
13384 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
13385 const RegionCodeGenTy &CodeGen) {
13386 llvm_unreachable("Not supported in SIMD-only mode");
13387}
13388
13391 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
13392 const RegionCodeGenTy &CodeGen) {
13393 llvm_unreachable("Not supported in SIMD-only mode");
13394}
13395
13397 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
13398 const VarDecl *PartIDVar, const VarDecl *TaskTVar,
13399 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
13400 bool Tied, unsigned &NumberOfParts) {
13401 llvm_unreachable("Not supported in SIMD-only mode");
13402}
13403
13405 CodeGenFunction &CGF, SourceLocation Loc, llvm::Function *OutlinedFn,
13406 ArrayRef<llvm::Value *> CapturedVars, const Expr *IfCond,
13407 llvm::Value *NumThreads, OpenMPNumThreadsClauseModifier NumThreadsModifier,
13408 OpenMPSeverityClauseKind Severity, const Expr *Message) {
13409 llvm_unreachable("Not supported in SIMD-only mode");
13410}
13411
13413 CodeGenFunction &CGF, StringRef CriticalName,
13414 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc,
13415 const Expr *Hint) {
13416 llvm_unreachable("Not supported in SIMD-only mode");
13417}
13418
13420 const RegionCodeGenTy &MasterOpGen,
13421 SourceLocation Loc) {
13422 llvm_unreachable("Not supported in SIMD-only mode");
13423}
13424
13426 const RegionCodeGenTy &MasterOpGen,
13427 SourceLocation Loc,
13428 const Expr *Filter) {
13429 llvm_unreachable("Not supported in SIMD-only mode");
13430}
13431
13433 SourceLocation Loc) {
13434 llvm_unreachable("Not supported in SIMD-only mode");
13435}
13436
13438 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen,
13439 SourceLocation Loc) {
13440 llvm_unreachable("Not supported in SIMD-only mode");
13441}
13442
13444 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen,
13445 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars,
13447 ArrayRef<const Expr *> AssignmentOps) {
13448 llvm_unreachable("Not supported in SIMD-only mode");
13449}
13450
13452 const RegionCodeGenTy &OrderedOpGen,
13453 SourceLocation Loc,
13454 bool IsThreads) {
13455 llvm_unreachable("Not supported in SIMD-only mode");
13456}
13457
13459 SourceLocation Loc,
13461 bool EmitChecks,
13462 bool ForceSimpleCall) {
13463 llvm_unreachable("Not supported in SIMD-only mode");
13464}
13465
13468 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
13469 bool Ordered, const DispatchRTInput &DispatchValues) {
13470 llvm_unreachable("Not supported in SIMD-only mode");
13471}
13472
13474 SourceLocation Loc) {
13475 llvm_unreachable("Not supported in SIMD-only mode");
13476}
13477
13480 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) {
13481 llvm_unreachable("Not supported in SIMD-only mode");
13482}
13483
13486 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) {
13487 llvm_unreachable("Not supported in SIMD-only mode");
13488}
13489
13491 SourceLocation Loc,
13492 unsigned IVSize,
13493 bool IVSigned) {
13494 llvm_unreachable("Not supported in SIMD-only mode");
13495}
13496
13498 SourceLocation Loc,
13499 OpenMPDirectiveKind DKind) {
13500 llvm_unreachable("Not supported in SIMD-only mode");
13501}
13502
13504 SourceLocation Loc,
13505 unsigned IVSize, bool IVSigned,
13506 Address IL, Address LB,
13507 Address UB, Address ST) {
13508 llvm_unreachable("Not supported in SIMD-only mode");
13509}
13510
13512 CodeGenFunction &CGF, llvm::Value *NumThreads, SourceLocation Loc,
13514 SourceLocation SeverityLoc, const Expr *Message,
13515 SourceLocation MessageLoc) {
13516 llvm_unreachable("Not supported in SIMD-only mode");
13517}
13518
13520 ProcBindKind ProcBind,
13521 SourceLocation Loc) {
13522 llvm_unreachable("Not supported in SIMD-only mode");
13523}
13524
13526 const VarDecl *VD,
13527 Address VDAddr,
13528 SourceLocation Loc) {
13529 llvm_unreachable("Not supported in SIMD-only mode");
13530}
13531
13533 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit,
13534 CodeGenFunction *CGF) {
13535 llvm_unreachable("Not supported in SIMD-only mode");
13536}
13537
13539 CodeGenFunction &CGF, QualType VarType, StringRef Name) {
13540 llvm_unreachable("Not supported in SIMD-only mode");
13541}
13542
13545 SourceLocation Loc,
13546 llvm::AtomicOrdering AO) {
13547 llvm_unreachable("Not supported in SIMD-only mode");
13548}
13549
13551 const OMPExecutableDirective &D,
13552 llvm::Function *TaskFunction,
13553 QualType SharedsTy, Address Shareds,
13554 const Expr *IfCond,
13555 const OMPTaskDataTy &Data) {
13556 llvm_unreachable("Not supported in SIMD-only mode");
13557}
13558
13561 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds,
13562 const Expr *IfCond, const OMPTaskDataTy &Data) {
13563 llvm_unreachable("Not supported in SIMD-only mode");
13564}
13565
13569 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) {
13570 assert(Options.SimpleReduction && "Only simple reduction is expected.");
13571 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs,
13572 ReductionOps, Options);
13573}
13574
13577 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
13578 llvm_unreachable("Not supported in SIMD-only mode");
13579}
13580
13582 SourceLocation Loc,
13583 bool IsWorksharingReduction) {
13584 llvm_unreachable("Not supported in SIMD-only mode");
13585}
13586
13588 SourceLocation Loc,
13589 ReductionCodeGen &RCG,
13590 unsigned N) {
13591 llvm_unreachable("Not supported in SIMD-only mode");
13592}
13593
13595 SourceLocation Loc,
13596 llvm::Value *ReductionsPtr,
13597 LValue SharedLVal) {
13598 llvm_unreachable("Not supported in SIMD-only mode");
13599}
13600
13602 SourceLocation Loc,
13603 const OMPTaskDataTy &Data) {
13604 llvm_unreachable("Not supported in SIMD-only mode");
13605}
13606
13609 OpenMPDirectiveKind CancelRegion) {
13610 llvm_unreachable("Not supported in SIMD-only mode");
13611}
13612
13614 SourceLocation Loc, const Expr *IfCond,
13615 OpenMPDirectiveKind CancelRegion) {
13616 llvm_unreachable("Not supported in SIMD-only mode");
13617}
13618
13620 const OMPExecutableDirective &D, StringRef ParentName,
13621 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
13622 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
13623 llvm_unreachable("Not supported in SIMD-only mode");
13624}
13625
13628 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
13629 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
13630 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
13631 const OMPLoopDirective &D)>
13632 SizeEmitter) {
13633 llvm_unreachable("Not supported in SIMD-only mode");
13634}
13635
13637 llvm_unreachable("Not supported in SIMD-only mode");
13638}
13639
13641 llvm_unreachable("Not supported in SIMD-only mode");
13642}
13643
13645 return false;
13646}
13647
13649 const OMPExecutableDirective &D,
13650 SourceLocation Loc,
13651 llvm::Function *OutlinedFn,
13652 ArrayRef<llvm::Value *> CapturedVars) {
13653 llvm_unreachable("Not supported in SIMD-only mode");
13654}
13655
13657 const Expr *NumTeams,
13658 const Expr *ThreadLimit,
13659 SourceLocation Loc) {
13660 llvm_unreachable("Not supported in SIMD-only mode");
13661}
13662
13664 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
13665 const Expr *Device, const RegionCodeGenTy &CodeGen,
13667 llvm_unreachable("Not supported in SIMD-only mode");
13668}
13669
13671 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
13672 const Expr *Device) {
13673 llvm_unreachable("Not supported in SIMD-only mode");
13674}
13675
13677 const OMPLoopDirective &D,
13678 ArrayRef<Expr *> NumIterations) {
13679 llvm_unreachable("Not supported in SIMD-only mode");
13680}
13681
13683 const OMPDependClause *C) {
13684 llvm_unreachable("Not supported in SIMD-only mode");
13685}
13686
13688 const OMPDoacrossClause *C) {
13689 llvm_unreachable("Not supported in SIMD-only mode");
13690}
13691
13692const VarDecl *
13694 const VarDecl *NativeParam) const {
13695 llvm_unreachable("Not supported in SIMD-only mode");
13696}
13697
13698Address
13700 const VarDecl *NativeParam,
13701 const VarDecl *TargetParam) const {
13702 llvm_unreachable("Not supported in SIMD-only mode");
13703}
#define V(N, I)
static llvm::Value * emitCopyprivateCopyFunction(CodeGenModule &CGM, llvm::Type *ArgsElemType, ArrayRef< const Expr * > CopyprivateVars, ArrayRef< const Expr * > DestExprs, ArrayRef< const Expr * > SrcExprs, ArrayRef< const Expr * > AssignmentOps, SourceLocation Loc)
static StringRef getIdentStringFromSourceLocation(CodeGenFunction &CGF, SourceLocation Loc, SmallString< 128 > &Buffer)
static void emitOffloadingArraysAndArgs(CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo, CGOpenMPRuntime::TargetDataInfo &Info, llvm::OpenMPIRBuilder &OMPBuilder, bool IsNonContiguous=false, bool ForEndCall=false)
Emit the arrays used to pass the captures and map information to the offloading runtime library.
static RecordDecl * createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, ArrayRef< PrivateDataTy > Privates)
static void emitInitWithReductionInitializer(CodeGenFunction &CGF, const OMPDeclareReductionDecl *DRD, const Expr *InitOp, Address Private, Address Original, QualType Ty)
static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, Address OriginalBaseAddress, llvm::Value *Addr)
static void emitPrivatesInit(CodeGenFunction &CGF, const OMPExecutableDirective &D, Address KmpTaskSharedsPtr, LValue TDBase, const RecordDecl *KmpTaskTWithPrivatesQTyRD, QualType SharedsTy, QualType SharedsPtrTy, const OMPTaskDataTy &Data, ArrayRef< PrivateDataTy > Privates, bool ForDup)
Emit initialization for private variables in task-based directives.
static void emitClauseForBareTargetDirective(CodeGenFunction &CGF, const OMPExecutableDirective &D, llvm::SmallVectorImpl< llvm::Value * > &Values)
static llvm::Value * emitDestructorsFunction(CodeGenModule &CGM, SourceLocation Loc, QualType KmpInt32Ty, QualType KmpTaskTWithPrivatesPtrQTy, QualType KmpTaskTWithPrivatesQTy)
static void EmitOMPAggregateReduction(CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, const VarDecl *RHSVar, const llvm::function_ref< void(CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *)> &RedOpGen, const Expr *XExpr=nullptr, const Expr *EExpr=nullptr, const Expr *UpExpr=nullptr)
Emit reduction operation for each element of array (required for array sections) LHS op = RHS.
static void emitTargetCallFallback(CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn, const OMPExecutableDirective &D, llvm::SmallVectorImpl< llvm::Value * > &CapturedVars, bool RequiresOuterTask, const CapturedStmt &CS, bool OffloadingMandatory, CodeGenFunction &CGF)
static llvm::Value * emitReduceInitFunction(CodeGenModule &CGM, SourceLocation Loc, ReductionCodeGen &RCG, unsigned N)
Emits reduction initializer function:
static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion)
static void emitDependData(CodeGenFunction &CGF, QualType &KmpDependInfoTy, llvm::PointerUnion< unsigned *, LValue * > Pos, const OMPTaskDataTy::DependData &Data, Address DependenciesArray)
static llvm::Value * emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, const OMPTaskDataTy &Data, QualType PrivatesQTy, ArrayRef< PrivateDataTy > Privates)
Emit a privates mapping function for correct handling of private and firstprivate variables.
static llvm::Value * emitReduceCombFunction(CodeGenModule &CGM, SourceLocation Loc, ReductionCodeGen &RCG, unsigned N, const Expr *ReductionOp, const Expr *LHS, const Expr *RHS, const Expr *PrivateRef)
Emits reduction combiner function:
static RecordDecl * createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef< PrivateDataTy > Privates)
static llvm::Value * getAllocatorVal(CodeGenFunction &CGF, const Expr *Allocator)
Return allocator value from expression, or return a null allocator (default when no allocator specifi...
static llvm::Function * emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, OpenMPDirectiveKind Kind, QualType KmpInt32Ty, QualType KmpTaskTWithPrivatesPtrQTy, QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, QualType SharedsPtrTy, llvm::Function *TaskFunction, llvm::Value *TaskPrivatesMap)
Emit a proxy function which accepts kmp_task_t as the second argument.
static bool isAllocatableDecl(const VarDecl *VD)
static llvm::Value * getAlignmentValue(CodeGenModule &CGM, const VarDecl *VD)
Return the alignment from an allocate directive if present.
static void emitTargetCallKernelLaunch(CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn, const OMPExecutableDirective &D, llvm::SmallVectorImpl< llvm::Value * > &CapturedVars, bool RequiresOuterTask, const CapturedStmt &CS, bool OffloadingMandatory, llvm::PointerIntPair< const Expr *, 2, OpenMPDeviceClauseModifier > Device, llvm::Value *OutlinedFnID, CodeGenFunction::OMPTargetDataInfo &InputInfo, llvm::Value *&MapTypesArray, llvm::Value *&MapNamesArray, llvm::function_ref< llvm::Value *(CodeGenFunction &CGF, const OMPLoopDirective &D)> SizeEmitter, CodeGenFunction &CGF, CodeGenModule &CGM)
static const OMPExecutableDirective * getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D)
Check for inner distribute directive.
static std::pair< llvm::Value *, llvm::Value * > getPointerAndSize(CodeGenFunction &CGF, const Expr *E)
static const VarDecl * getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE)
static bool getAArch64MTV(QualType QT, llvm::OpenMPIRBuilder::DeclareSimdKindTy Kind)
Maps To Vector (MTV), as defined in 4.1.1 of the AAVFABI (2021Q1).
static bool isTrivial(ASTContext &Ctx, const Expr *E)
Checks if the expression is constant or does not have non-trivial function calls.
static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, bool Chunked, bool Ordered)
Map the OpenMP loop schedule to the runtime enumeration.
static void getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS, const Expr **E, int32_t &UpperBound, bool UpperBoundOnly, llvm::Value **CondVal)
Check for a num threads constant value (stored in DefaultVal), or expression (stored in E).
static llvm::Value * emitDeviceID(llvm::PointerIntPair< const Expr *, 2, OpenMPDeviceClauseModifier > Device, CodeGenFunction &CGF)
static const OMPDeclareReductionDecl * getReductionInit(const Expr *ReductionOp)
Check if the combiner is a call to UDR combiner and if it is so return the UDR decl used for reductio...
static bool checkInitIsRequired(CodeGenFunction &CGF, ArrayRef< PrivateDataTy > Privates)
Check if duplication function is required for taskloops.
static bool validateAArch64Simdlen(CodeGenModule &CGM, SourceLocation SLoc, unsigned UserVLEN, unsigned WDS, char ISA)
static bool checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD, ArrayRef< PrivateDataTy > Privates)
Checks if destructor function is required to be generated.
static llvm::TargetRegionEntryInfo getEntryInfoFromPresumedLoc(CodeGenModule &CGM, llvm::OpenMPIRBuilder &OMPBuilder, SourceLocation BeginLoc, llvm::StringRef ParentName="")
static void genMapInfo(MappableExprsHandler &MEHandler, CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo, llvm::OpenMPIRBuilder &OMPBuilder, const llvm::DenseSet< CanonicalDeclPtr< const Decl > > &SkippedVarSet=llvm::DenseSet< CanonicalDeclPtr< const Decl > >())
static unsigned getAArch64LS(QualType QT, llvm::OpenMPIRBuilder::DeclareSimdKindTy Kind, ASTContext &C)
Computes the lane size (LS) of a return type or of an input parameter, as defined by LS(P) in 3....
static llvm::OpenMPIRBuilder::DeclareSimdBranch convertDeclareSimdBranch(OMPDeclareSimdDeclAttr::BranchStateTy State)
static void emitForStaticInitCall(CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, const CGOpenMPRuntime::StaticRTInput &Values)
static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, LValue BaseLV)
static void getKmpAffinityType(ASTContext &C, QualType &KmpTaskAffinityInfoTy)
Builds kmp_depend_info, if it is not built yet, and builds flags type.
static llvm::Constant * emitMappingInformation(CodeGenFunction &CGF, llvm::OpenMPIRBuilder &OMPBuilder, MappableExprsHandler::MappingExprInfo &MapExprs)
Emit a string constant containing the names of the values mapped to the offloading runtime library.
static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy, QualType &FlagsTy)
Builds kmp_depend_info, if it is not built yet, and builds flags type.
static llvm::Value * emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, const OMPExecutableDirective &D, QualType KmpTaskTWithPrivatesPtrQTy, const RecordDecl *KmpTaskTWithPrivatesQTyRD, const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, QualType SharedsPtrTy, const OMPTaskDataTy &Data, ArrayRef< PrivateDataTy > Privates, bool WithLastIter)
Emit task_dup function (for initialization of private/firstprivate/lastprivate vars and last_iter fla...
static std::pair< llvm::Value *, OMPDynGroupprivateFallbackType > emitDynCGroupMem(const OMPExecutableDirective &D, CodeGenFunction &CGF)
static llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseKind convertDeviceClause(const VarDecl *VD)
static llvm::Value * emitReduceFiniFunction(CodeGenModule &CGM, SourceLocation Loc, ReductionCodeGen &RCG, unsigned N)
Emits reduction finalizer function:
static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, QualType Type, bool EmitDeclareReductionInit, const Expr *Init, const OMPDeclareReductionDecl *DRD, Address SrcAddr=Address::invalid())
Emit initialization of arrays of complex types.
static bool getAArch64PBV(QualType QT, ASTContext &C)
Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
static void EmitDoacrossOrdered(CodeGenFunction &CGF, CodeGenModule &CGM, const T *C, llvm::Value *ULoc, llvm::Value *ThreadID)
static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K)
Translates internal dependency kind into the runtime kind.
static void emitTargetCallElse(CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn, const OMPExecutableDirective &D, llvm::SmallVectorImpl< llvm::Value * > &CapturedVars, bool RequiresOuterTask, const CapturedStmt &CS, bool OffloadingMandatory, CodeGenFunction &CGF)
static llvm::Function * emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, const Expr *CombinerInitializer, const VarDecl *In, const VarDecl *Out, bool IsCombiner)
static void emitReductionCombiner(CodeGenFunction &CGF, const Expr *ReductionOp)
Emit reduction combiner.
static std::tuple< unsigned, unsigned, bool > getNDSWDS(const FunctionDecl *FD, ArrayRef< llvm::OpenMPIRBuilder::DeclareSimdAttrTy > ParamAttrs)
static std::string generateUniqueName(CodeGenModule &CGM, llvm::StringRef Prefix, const Expr *Ref)
static llvm::Function * emitParallelOrTeamsOutlinedFunction(CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen)
static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, unsigned Index, const VarDecl *Var)
Given an array of pointers to variables, project the address of a given variable.
static const VarDecl * getOriginalVarDecl(const ValueDecl *Decl)
For BindingDecls, returns the DecomposedDecl as the original VarDecl.
static void mergeThreadCountUpperBound(int32_t &UpperBound, int32_t Val)
Merge the thread count upper bound Val into UpperBound.
static FieldDecl * addFieldToRecordDecl(ASTContext &C, DeclContext *DC, QualType FieldTy)
static unsigned evaluateCDTSize(const FunctionDecl *FD, ArrayRef< llvm::OpenMPIRBuilder::DeclareSimdAttrTy > ParamAttrs)
static ValueDecl * getDeclFromThisExpr(const Expr *E)
static void genMapInfoForCaptures(MappableExprsHandler &MEHandler, CodeGenFunction &CGF, const CapturedStmt &CS, llvm::SmallVectorImpl< llvm::Value * > &CapturedVars, llvm::OpenMPIRBuilder &OMPBuilder, llvm::DenseSet< CanonicalDeclPtr< const Decl > > &MappedVarSet, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo)
static RecordDecl * createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, QualType KmpInt32Ty, QualType KmpRoutineEntryPointerQTy)
static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule, OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2)
static mlir::omp::DeclareTargetCaptureClause convertCaptureClause(OMPDeclareTargetDeclAttr::MapTypeTy mapTy)
static bool isAssumedToBeNotEmitted(const ValueDecl *vd, bool isDevice)
Returns true if the declaration should be skipped based on its device_type attribute and the current ...
Expr::Classification Cl
TokenType getType() const
Returns the token's type, e.g.
FormatToken * Next
The next token in the unwrapped line.
Result
Implement __builtin_bit_cast and related operations.
#define X(type, name)
Definition Value.h:97
This file defines OpenMP AST classes for clauses.
Defines some OpenMP-specific enums and functions.
llvm::json::Array Array
Defines the SourceManager interface.
This file defines OpenMP AST classes for executable directives and clauses.
__DEVICE__ int max(int __a, int __b)
This represents clause 'affinity' in the 'pragma omp task'-based directives.
static std::pair< const Expr *, std::optional< size_t > > findAttachPtrExpr(MappableExprComponentListRef Components, OpenMPDirectiveKind CurDirKind)
Find the attach pointer expression from a list of mappable expression components.
static QualType getComponentExprElementType(const Expr *Exp)
Get the type of an element of a ComponentList Expr Exp.
ArrayRef< MappableComponent > MappableExprComponentListRef
This represents implicit clause 'depend' for the 'pragma omp task' directive.
This represents 'detach' clause in the 'pragma omp task' directive.
This represents 'device' clause in the 'pragma omp ...' directive.
This represents the 'doacross' clause for the 'pragma omp ordered' directive.
This represents 'dyn_groupprivate' clause in 'pragma omp target ...' and 'pragma omp teams ....
This is a common base class for loop directives ('omp simd', 'omp for', 'omp for simd' etc....
Expr * getLowerBoundVariable() const
Expr * getUpperBoundVariable() const
Expr * getStrideVariable() const
This represents clause 'map' in the 'pragma omp ...' directives.
This represents clause 'nontemporal' in the 'pragma omp ...' directives.
This represents 'num_teams' clause in the 'pragma omp ...' directive.
This represents 'thread_limit' clause in the 'pragma omp ...' directive.
This represents clause 'uses_allocators' in the 'pragma omp target'-based directives.
This represents 'ompx_attribute' clause in a directive that might generate an outlined function.
This represents 'ompx_bare' clause in the 'pragma omp target teams ...' directive.
This represents 'ompx_dyn_cgroup_mem' clause in the 'pragma omp target ...' directive.
Holds long-lived AST nodes (such as types and decls) that can be referred to throughout the semantic ...
Definition ASTContext.h:239
SourceManager & getSourceManager()
Definition ASTContext.h:911
const ConstantArrayType * getAsConstantArrayType(QualType T) const
CharUnits getTypeAlignInChars(QualType T) const
Return the ABI-specified alignment of a (complete) type T, in characters.
const ASTRecordLayout & getASTRecordLayout(const RecordDecl *D) const
Get or compute information about the layout of the specified record (struct/union/class) D,...
QualType getPointerType(QualType T) const
Return the uniqued reference to the type for a pointer to the specified type.
CanQualType VoidPtrTy
QualType getConstantArrayType(QualType EltTy, const llvm::APInt &ArySize, const Expr *SizeExpr, ArraySizeModifier ASM, unsigned IndexTypeQuals) const
Return the unique reference to the type for a constant array of the specified element type.
const LangOptions & getLangOpts() const
CanQualType BoolTy
QualType getIntTypeForBitwidth(unsigned DestWidth, unsigned Signed) const
getIntTypeForBitwidth - sets integer QualTy according to specified details: bitwidth,...
CharUnits getDeclAlign(const Decl *D, bool ForAlignof=false) const
Return a conservative estimate of the alignment of the specified decl D.
int64_t toBits(CharUnits CharSize) const
Convert a size in characters to a size in bits.
const ArrayType * getAsArrayType(QualType T) const
Type Query functions.
uint64_t getTypeSize(QualType T) const
Return the size of the specified (complete) type T, in bits.
CharUnits getTypeSizeInChars(QualType T) const
Return the size of the specified (complete) type T, in characters.
static bool hasSameType(QualType T1, QualType T2)
Determine whether the given types T1 and T2 are equivalent.
const VariableArrayType * getAsVariableArrayType(QualType T) const
QualType getSizeType() const
Return the unique type for "size_t" (C99 7.17), defined in <stddef.h>.
unsigned getTypeAlign(QualType T) const
Return the ABI-specified alignment of a (complete) type T, in bits.
CharUnits getSize() const
getSize - Get the record size in characters.
uint64_t getFieldOffset(unsigned FieldNo) const
getFieldOffset - Get the offset of the given field index, in bits.
CharUnits getNonVirtualSize() const
getNonVirtualSize - Get the non-virtual size (in chars) of an object, which is the size of the object...
static QualType getBaseOriginalType(const Expr *Base)
Return original type of the base expression for array section.
Definition Expr.cpp:5439
Represents an array type, per C99 6.7.5.2 - Array Declarators.
Definition TypeBase.h:3820
Attr - This represents one attribute.
Definition Attr.h:46
A binding in a decomposition declaration.
Definition DeclCXX.h:4215
Expr * getBinding() const
Get the expression to which this declaration is bound.
Definition DeclCXX.h:4241
DecompositionDecl * getDecomposedDecl() const
Get the decomposition declaration that this binding represents a decomposition of.
Definition DeclCXX.h:4248
Represents a base class of a C++ class.
Definition DeclCXX.h:146
Represents a C++ constructor within a class.
Definition DeclCXX.h:2642
Represents a C++ destructor within a class.
Definition DeclCXX.h:2907
const CXXRecordDecl * getParent() const
Return the parent of this method declaration, which is the class in which this method is defined.
Definition DeclCXX.h:2293
QualType getFunctionObjectParameterType() const
Definition DeclCXX.h:2317
Represents a C++ struct/union/class.
Definition DeclCXX.h:258
base_class_range bases()
Definition DeclCXX.h:609
bool isLambda() const
Determine whether this class describes a lambda function object.
Definition DeclCXX.h:1028
void getCaptureFields(llvm::DenseMap< const ValueDecl *, FieldDecl * > &Captures, FieldDecl *&ThisCapture) const
For a closure type, retrieve the mapping from captured variables and this to the non-static data memb...
Definition DeclCXX.cpp:1792
unsigned getNumBases() const
Retrieves the number of base classes of this class.
Definition DeclCXX.h:603
base_class_range vbases()
Definition DeclCXX.h:626
capture_const_range captures() const
Definition DeclCXX.h:1107
ctor_range ctors() const
Definition DeclCXX.h:671
CXXDestructorDecl * getDestructor() const
Returns the destructor decl for this class.
Definition DeclCXX.cpp:2129
CanProxy< U > castAs() const
A wrapper class around a pointer that always points to its canonical declaration.
Describes the capture of either a variable, or 'this', or variable-length array type.
Definition Stmt.h:3962
bool capturesVariableByCopy() const
Determine whether this capture handles a variable by copy.
Definition Stmt.h:3996
VarDecl * getCapturedVar() const
Retrieve the declaration of the variable being captured.
Definition Stmt.cpp:1391
bool capturesVariableArrayType() const
Determine whether this capture handles a variable-length array type.
Definition Stmt.h:4002
bool capturesThis() const
Determine whether this capture handles the C++ 'this' pointer.
Definition Stmt.h:3990
bool capturesVariable() const
Determine whether this capture handles a variable (by reference).
Definition Stmt.h:3993
This captures a statement into a function.
Definition Stmt.h:3949
const Capture * const_capture_iterator
Definition Stmt.h:4083
capture_iterator capture_end() const
Retrieve an iterator pointing past the end of the sequence of captures.
Definition Stmt.h:4100
const RecordDecl * getCapturedRecordDecl() const
Retrieve the record declaration for captured variables.
Definition Stmt.h:4070
Stmt * getCapturedStmt()
Retrieve the statement being captured.
Definition Stmt.h:4053
bool capturesVariable(const VarDecl *Var) const
True if this variable has been captured.
Definition Stmt.cpp:1517
capture_iterator capture_begin()
Retrieve an iterator pointing to the first capture.
Definition Stmt.h:4095
capture_range captures()
Definition Stmt.h:4087
This is an opaque type for sizes expressed in character units.
Definition CharUnits.h:38
bool isZero() const
Test whether the quantity equals zero.
Definition CharUnits.h:101
CharUnits alignTo(CharUnits Align) const
Returns the next integer (mod 2**64) that is greater than or equal to this quantity and is a multiple...
Definition CharUnits.h:169
llvm::Align getAsAlign() const
Returns Quantity as a valid llvm::Align, Beware llvm::Align assumes power of two 8-bit bytes.
Definition CharUnits.h:157
QuantityType getQuantity() const
Get the raw integer representation of this quantity.
Definition CharUnits.h:153
CharUnits alignmentOfArrayElement(CharUnits elementSize) const
Given that this is the alignment of the first element of an array, return the minimum alignment of an...
Definition CharUnits.h:182
static CharUnits fromQuantity(QuantityType Quantity)
Construct a CharUnits quantity from a raw integer type.
Definition CharUnits.h:58
std::string SampleProfileFile
Name of the profile file to use with -fprofile-sample-use.
Like RawAddress, an abstract representation of an aligned address, but the pointer contained in this ...
Definition Address.h:128
static Address invalid()
Definition Address.h:176
llvm::Value * emitRawPointer(CodeGenFunction &CGF) const
Return the pointer contained in this class after authenticating it and adding offset to it if necessa...
Definition Address.h:253
CharUnits getAlignment() const
Definition Address.h:194
llvm::Type * getElementType() const
Return the type of the values stored in this address.
Definition Address.h:209
Address withPointer(llvm::Value *NewPointer, KnownNonNull_t IsKnownNonNull) const
Return address with different pointer, but same element type and alignment.
Definition Address.h:261
Address withElementType(llvm::Type *ElemTy) const
Return address with different element type, but same pointer and alignment.
Definition Address.h:276
bool isValid() const
Definition Address.h:177
llvm::PointerType * getType() const
Return the type of the pointer value.
Definition Address.h:204
static ApplyDebugLocation CreateArtificial(CodeGenFunction &CGF)
Apply TemporaryLocation if it is valid.
static ApplyDebugLocation CreateDefaultArtificial(CodeGenFunction &CGF, SourceLocation TemporaryLocation)
Apply TemporaryLocation if it is valid.
static ApplyDebugLocation CreateEmpty(CodeGenFunction &CGF)
Set the IRBuilder to not attach debug locations.
llvm::StoreInst * CreateStore(llvm::Value *Val, Address Addr, bool IsVolatile=false)
Definition CGBuilder.h:146
Address CreateGEP(CodeGenFunction &CGF, Address Addr, llvm::Value *Index, const llvm::Twine &Name="")
Definition CGBuilder.h:302
Address CreatePointerBitCastOrAddrSpaceCast(Address Addr, llvm::Type *Ty, llvm::Type *ElementTy, const llvm::Twine &Name="")
Definition CGBuilder.h:213
Address CreateConstArrayGEP(Address Addr, uint64_t Index, const llvm::Twine &Name="")
Given addr = [n x T]* ... produce name = getelementptr inbounds addr, i64 0, i64 index where i64 is a...
Definition CGBuilder.h:251
llvm::LoadInst * CreateLoad(Address Addr, const llvm::Twine &Name="")
Definition CGBuilder.h:118
llvm::CallInst * CreateMemCpy(Address Dest, Address Src, llvm::Value *Size, bool IsVolatile=false)
Definition CGBuilder.h:397
Address CreateConstGEP(Address Addr, uint64_t Index, const llvm::Twine &Name="")
Given addr = T* ... produce name = getelementptr inbounds addr, i64 index where i64 is actually the t...
Definition CGBuilder.h:288
Address CreateAddrSpaceCast(Address Addr, llvm::Type *Ty, llvm::Type *ElementTy, const llvm::Twine &Name="")
Definition CGBuilder.h:199
CGFunctionInfo - Class to encapsulate the information about a function definition.
static LastprivateConditionalRAII disable(CodeGenFunction &CGF, const OMPExecutableDirective &S)
NontemporalDeclsRAII(CodeGenModule &CGM, const OMPLoopDirective &S)
Struct that keeps all the relevant information that should be kept throughout a 'target data' region.
llvm::DenseMap< const ValueDecl *, llvm::Value * > CaptureDeviceAddrMap
Map between the a declaration of a capture and the corresponding new llvm address where the runtime r...
UntiedTaskLocalDeclsRAII(CodeGenFunction &CGF, const llvm::MapVector< CanonicalDeclPtr< const VarDecl >, std::pair< Address, Address > > &LocalVars)
virtual Address emitThreadIDAddress(CodeGenFunction &CGF, SourceLocation Loc)
Emits address of the word in a memory where current thread id is stored.
llvm::StringSet ThreadPrivateWithDefinition
Set of threadprivate variables with the generated initializer.
void emitUpdateDependObjectsClause(CodeGenFunction &CGF, LValue DepobjLVal, OpenMPDependClauseKind NewDepKind, SourceLocation Loc)
Updates the dependency kind in the specified depobj object.
virtual void emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, const OMPExecutableDirective &D, llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, const Expr *IfCond, const OMPTaskDataTy &Data)
Emit task region for the task directive.
void createOffloadEntriesAndInfoMetadata()
Creates all the offload entries in the current compilation unit along with the associated metadata.
const Expr * getNumTeamsExprForTargetDirective(CodeGenFunction &CGF, const OMPExecutableDirective &D, int32_t &MinTeamsVal, int32_t &MaxTeamsVal)
Emit the number of teams for a target directive.
virtual Address getAddrOfThreadPrivate(CodeGenFunction &CGF, const VarDecl *VD, Address VDAddr, SourceLocation Loc)
Returns address of the threadprivate variable for the current thread.
void emitDeferredTargetDecls() const
Emit deferred declare target variables marked for deferred emission.
virtual llvm::Value * emitForNext(CodeGenFunction &CGF, SourceLocation Loc, unsigned IVSize, bool IVSigned, Address IL, Address LB, Address UB, Address ST)
Call __kmpc_dispatch_next( ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, kmp_int[32|64] *p_lowe...
bool markAsGlobalTarget(GlobalDecl GD)
Marks the declaration as already emitted for the device code and returns true, if it was marked alrea...
virtual void emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, llvm::Function *OutlinedFn, ArrayRef< llvm::Value * > CapturedVars, const Expr *IfCond, llvm::Value *NumThreads, OpenMPNumThreadsClauseModifier NumThreadsModifier=OMPC_NUMTHREADS_unknown, OpenMPSeverityClauseKind Severity=OMPC_SEVERITY_fatal, const Expr *Message=nullptr)
Emits code for parallel or serial call of the OutlinedFn with variables captured in a record which ad...
llvm::SmallDenseSet< CanonicalDeclPtr< const Decl > > NontemporalDeclsSet
virtual void emitTargetDataStandAloneCall(CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, const Expr *Device)
Emit the data mapping/movement code associated with the directive D that should be of the form 'targe...
virtual void emitNumThreadsClause(CodeGenFunction &CGF, llvm::Value *NumThreads, SourceLocation Loc, OpenMPNumThreadsClauseModifier Modifier=OMPC_NUMTHREADS_unknown, OpenMPSeverityClauseKind Severity=OMPC_SEVERITY_fatal, SourceLocation SeverityLoc=SourceLocation(), const Expr *Message=nullptr, SourceLocation MessageLoc=SourceLocation())
Emits call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32global_tid, kmp_int32 num_threads) ...
QualType SavedKmpTaskloopTQTy
Saved kmp_task_t for taskloop-based directive.
virtual void emitSingleRegion(CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, SourceLocation Loc, ArrayRef< const Expr * > CopyprivateVars, ArrayRef< const Expr * > DestExprs, ArrayRef< const Expr * > SrcExprs, ArrayRef< const Expr * > AssignmentOps)
Emits a single region.
virtual bool emitTargetGlobal(GlobalDecl GD)
Emit the global GD if it is meaningful for the target.
void setLocThreadIdInsertPt(CodeGenFunction &CGF, bool AtCurrentPoint=false)
std::string getOutlinedHelperName(StringRef Name) const
Get the function name of an outlined region.
bool HasEmittedDeclareTargetRegion
Flag for keeping track of weather a device routine has been emitted.
llvm::Constant * getOrCreateThreadPrivateCache(const VarDecl *VD)
If the specified mangled name is not in the module, create and return threadprivate cache object.
virtual Address getTaskReductionItem(CodeGenFunction &CGF, SourceLocation Loc, llvm::Value *ReductionsPtr, LValue SharedLVal)
Get the address of void * type of the privatue copy of the reduction item specified by the SharedLVal...
virtual void emitForDispatchDeinit(CodeGenFunction &CGF, SourceLocation Loc)
This is used for non static scheduled types and when the ordered clause is present on the loop constr...
void emitCall(CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee Callee, ArrayRef< llvm::Value * > Args={}) const
Emits Callee function call with arguments Args with location Loc.
virtual void getDefaultScheduleAndChunk(CodeGenFunction &CGF, const OMPLoopDirective &S, OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const
Choose default schedule type and chunk value for the schedule clause.
virtual std::pair< llvm::Function *, llvm::Function * > getUserDefinedReduction(const OMPDeclareReductionDecl *D)
Get combiner/initializer for the specified user-defined reduction, if any.
virtual bool isGPU() const
Returns true if the current target is a GPU.
static const Stmt * getSingleCompoundChild(ASTContext &Ctx, const Stmt *Body)
Checks if the Body is the CompoundStmt and returns its child statement iff there is only one that is ...
virtual void emitDeclareTargetFunction(const FunctionDecl *FD, llvm::GlobalValue *GV)
Emit code for handling declare target functions in the runtime.
bool HasRequiresUnifiedSharedMemory
Flag for keeping track of weather a requires unified_shared_memory directive is present.
llvm::Value * emitUpdateLocation(CodeGenFunction &CGF, SourceLocation Loc, unsigned Flags=0, bool EmitLoc=false)
Emits object of ident_t type with info for source location.
bool isLocalVarInUntiedTask(CodeGenFunction &CGF, const VarDecl *VD) const
Returns true if the variable is a local variable in untied task.
virtual void emitTeamsCall(CodeGenFunction &CGF, const OMPExecutableDirective &D, SourceLocation Loc, llvm::Function *OutlinedFn, ArrayRef< llvm::Value * > CapturedVars)
Emits code for teams call of the OutlinedFn with variables captured in a record which address is stor...
virtual void emitCancellationPointCall(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind CancelRegion)
Emit code for 'cancellation point' construct.
virtual llvm::Function * emitThreadPrivateVarDefinition(const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, CodeGenFunction *CGF=nullptr)
Emit a code for initialization of threadprivate variable.
virtual ConstantAddress getAddrOfDeclareTargetVar(const VarDecl *VD)
Returns the address of the variable marked as declare target with link clause OR as declare target wi...
llvm::Function * getOrCreateUserDefinedMapperFunc(const OMPDeclareMapperDecl *D)
Get the function for the specified user-defined mapper.
OpenMPLocThreadIDMapTy OpenMPLocThreadIDMap
virtual void functionFinished(CodeGenFunction &CGF)
Cleans up references to the objects in finished function.
virtual llvm::Function * emitTeamsOutlinedFunction(CodeGenFunction &CGF, const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen)
Emits outlined function for the specified OpenMP teams directive D.
QualType KmpTaskTQTy
Type typedef struct kmp_task { void * shareds; /‍**< pointer to block of pointers to shared vars ‍/ k...
llvm::OpenMPIRBuilder OMPBuilder
An OpenMP-IR-Builder instance.
virtual void emitDoacrossInit(CodeGenFunction &CGF, const OMPLoopDirective &D, ArrayRef< Expr * > NumIterations)
Emit initialization for doacross loop nesting support.
virtual void adjustTargetSpecificDataForLambdas(CodeGenFunction &CGF, const OMPExecutableDirective &D) const
Adjust some parameters for the target-based directives, like addresses of the variables captured by r...
virtual void emitTargetDataCalls(CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, const Expr *Device, const RegionCodeGenTy &CodeGen, CGOpenMPRuntime::TargetDataInfo &Info)
Emit the target data mapping code associated with D.
virtual unsigned getDefaultLocationReserved2Flags() const
Returns additional flags that can be stored in reserved_2 field of the default location.
virtual Address getParameterAddress(CodeGenFunction &CGF, const VarDecl *NativeParam, const VarDecl *TargetParam) const
Gets the address of the native argument basing on the address of the target-specific parameter.
void emitUsesAllocatorsFini(CodeGenFunction &CGF, const Expr *Allocator)
Destroys user defined allocators specified in the uses_allocators clause.
QualType KmpTaskAffinityInfoTy
Type typedef struct kmp_task_affinity_info { kmp_intptr_t base_addr; size_t len; struct { bool flag1 ...
void emitPrivateReduction(CodeGenFunction &CGF, SourceLocation Loc, const Expr *Privates, const Expr *LHSExprs, const Expr *RHSExprs, const Expr *ReductionOps)
Emits code for private variable reduction.
llvm::Value * emitNumTeamsForTargetDirective(CodeGenFunction &CGF, const OMPExecutableDirective &D)
virtual void emitTargetOutlinedFunctionHelper(const OMPExecutableDirective &D, StringRef ParentName, llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, bool IsOffloadEntry, const RegionCodeGenTy &CodeGen)
Helper to emit outlined function for 'target' directive.
void scanForTargetRegionsFunctions(const Stmt *S, StringRef ParentName)
Start scanning from statement S and emit all target regions found along the way.
SmallVector< llvm::Value *, 4 > emitDepobjElementsSizes(CodeGenFunction &CGF, QualType &KmpDependInfoTy, const OMPTaskDataTy::DependData &Data)
virtual llvm::Value * emitMessageClause(CodeGenFunction &CGF, const Expr *Message, SourceLocation Loc)
virtual void emitTaskgroupRegion(CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, SourceLocation Loc)
Emit a taskgroup region.
llvm::DenseMap< llvm::Function *, llvm::DenseMap< CanonicalDeclPtr< const Decl >, std::tuple< QualType, const FieldDecl *, const FieldDecl *, LValue > > > LastprivateConditionalToTypes
Maps local variables marked as lastprivate conditional to their internal types.
virtual bool emitTargetGlobalVariable(GlobalDecl GD)
Emit the global variable if it is a valid device global variable.
virtual void emitNumTeamsClause(CodeGenFunction &CGF, const Expr *NumTeams, const Expr *ThreadLimit, SourceLocation Loc)
Emits call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32global_tid, kmp_int32 num_teams,...
bool hasRequiresUnifiedSharedMemory() const
Return whether the unified_shared_memory has been specified.
virtual Address getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, QualType VarType, StringRef Name)
Creates artificial threadprivate variable with name Name and type VarType.
void emitUserDefinedMapper(const OMPDeclareMapperDecl *D, CodeGenFunction *CGF=nullptr)
Emit the function for the user defined mapper construct.
bool HasEmittedTargetRegion
Flag for keeping track of weather a target region has been emitted.
void emitDepobjElements(CodeGenFunction &CGF, QualType &KmpDependInfoTy, LValue PosLVal, const OMPTaskDataTy::DependData &Data, Address DependenciesArray)
std::string getReductionFuncName(StringRef Name) const
Get the function name of a reduction function.
virtual void processRequiresDirective(const OMPRequiresDecl *D)
Perform check on requires decl to ensure that target architecture supports unified addressing.
llvm::DenseSet< CanonicalDeclPtr< const Decl > > AlreadyEmittedTargetDecls
List of the emitted declarations.
virtual llvm::Value * emitTaskReductionInit(CodeGenFunction &CGF, SourceLocation Loc, ArrayRef< const Expr * > LHSExprs, ArrayRef< const Expr * > RHSExprs, const OMPTaskDataTy &Data)
Emit a code for initialization of task reduction clause.
llvm::Value * getThreadID(CodeGenFunction &CGF, SourceLocation Loc)
Gets thread id value for the current thread.
virtual void emitLastprivateConditionalFinalUpdate(CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD, SourceLocation Loc)
Gets the address of the global copy used for lastprivate conditional update, if any.
llvm::MapVector< CanonicalDeclPtr< const VarDecl >, std::pair< Address, Address > > UntiedLocalVarsAddressesMap
virtual void emitErrorCall(CodeGenFunction &CGF, SourceLocation Loc, Expr *ME, bool IsFatal)
Emit __kmpc_error call for error directive extern void __kmpc_error(ident_t *loc, int severity,...
void clearLocThreadIdInsertPt(CodeGenFunction &CGF)
virtual void emitTaskyieldCall(CodeGenFunction &CGF, SourceLocation Loc)
Emits code for a taskyield directive.
std::string getName(ArrayRef< StringRef > Parts) const
Get the platform-specific name separator.
void computeMinAndMaxThreadsAndTeams(const OMPExecutableDirective &D, CodeGenFunction &CGF, llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs &Attrs)
Helper to determine the min/max number of threads/teams for D.
virtual void emitFlush(CodeGenFunction &CGF, ArrayRef< const Expr * > Vars, SourceLocation Loc, llvm::AtomicOrdering AO)
Emit flush of the variables specified in 'omp flush' directive.
virtual void emitTaskwaitCall(CodeGenFunction &CGF, SourceLocation Loc, const OMPTaskDataTy &Data)
Emit code for 'taskwait' directive.
virtual void emitProcBindClause(CodeGenFunction &CGF, llvm::omp::ProcBindKind ProcBind, SourceLocation Loc)
Emit call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32global_tid, int proc_bind) to generate...
void emitLastprivateConditionalUpdate(CodeGenFunction &CGF, LValue IVLVal, StringRef UniqueDeclName, LValue LVal, SourceLocation Loc)
Emit update for lastprivate conditional data.
virtual void emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, const Expr *IfCond, const OMPTaskDataTy &Data)
Emit task region for the taskloop directive.
virtual void emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind Kind, bool EmitChecks=true, bool ForceSimpleCall=false)
Emit an implicit/explicit barrier for OpenMP threads.
static unsigned getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind)
Returns default flags for the barriers depending on the directive, for which this barier is going to ...
virtual bool emitTargetFunctions(GlobalDecl GD)
Emit the target regions enclosed in GD function definition or the function itself in case it is a val...
TaskResultTy emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, const OMPExecutableDirective &D, llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, const OMPTaskDataTy &Data)
Emit task region for the task directive.
llvm::Value * emitTargetNumIterationsCall(CodeGenFunction &CGF, const OMPExecutableDirective &D, llvm::function_ref< llvm::Value *(CodeGenFunction &CGF, const OMPLoopDirective &D)> SizeEmitter)
Return the trip count of loops associated with constructs / 'target teams distribute' and 'teams dist...
llvm::StringMap< llvm::AssertingVH< llvm::GlobalVariable >, llvm::BumpPtrAllocator > InternalVars
An ordered map of auto-generated variables to their unique names.
virtual void emitDistributeStaticInit(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values)
llvm::SmallVector< UntiedLocalVarsAddressesMap, 4 > UntiedLocalVarsStack
virtual void emitForStaticFinish(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind)
Call the appropriate runtime routine to notify that we finished all the work with current loop.
virtual void emitThreadLimitClause(CodeGenFunction &CGF, const Expr *ThreadLimit, SourceLocation Loc)
Emits call to void __kmpc_set_thread_limit(ident_t *loc, kmp_int32global_tid, kmp_int32 thread_limit)...
void emitIfClause(CodeGenFunction &CGF, const Expr *Cond, const RegionCodeGenTy &ThenGen, const RegionCodeGenTy &ElseGen)
Emits code for OpenMP 'if' clause using specified CodeGen function.
Address emitDepobjDependClause(CodeGenFunction &CGF, const OMPTaskDataTy::DependData &Dependencies, SourceLocation Loc)
Emits list of dependecies based on the provided data (array of dependence/expression pairs) for depob...
bool isNontemporalDecl(const ValueDecl *VD) const
Checks if the VD variable is marked as nontemporal declaration in current context.
virtual llvm::Function * emitParallelOutlinedFunction(CodeGenFunction &CGF, const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen)
Emits outlined function for the specified OpenMP parallel directive D.
const Expr * getNumThreadsExprForTargetDirective(CodeGenFunction &CGF, const OMPExecutableDirective &D, int32_t &UpperBound, bool UpperBoundOnly, llvm::Value **CondExpr=nullptr, const Expr **ThreadLimitExpr=nullptr)
Check for a number of threads upper bound constant value (stored in UpperBound), or expression (retur...
virtual void registerVTableOffloadEntry(llvm::GlobalVariable *VTable, const VarDecl *VD)
Register VTable to OpenMP offload entry.
virtual llvm::Value * emitSeverityClause(OpenMPSeverityClauseKind Severity, SourceLocation Loc)
llvm::SmallVector< LastprivateConditionalData, 4 > LastprivateConditionalStack
Stack for list of addresses of declarations in current context marked as lastprivate conditional.
virtual void emitForStaticInit(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values)
Call the appropriate runtime routine to initialize it before start of loop.
virtual void emitDeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn)
Marks function Fn with properly mangled versions of vector functions.
llvm::AtomicOrdering getDefaultMemoryOrdering() const
Gets default memory ordering as specified in requires directive.
virtual bool isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, bool Chunked) const
Check if the specified ScheduleKind is static non-chunked.
virtual void emitAndRegisterVTable(CodeGenModule &CGM, CXXRecordDecl *CXXRecord, const VarDecl *VD)
Emit and register VTable for the C++ class in OpenMP offload entry.
llvm::Value * getCriticalRegionLock(StringRef CriticalName)
Returns corresponding lock object for the specified critical region name.
virtual void emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, const Expr *IfCond, OpenMPDirectiveKind CancelRegion)
Emit code for 'cancel' construct.
QualType SavedKmpTaskTQTy
Saved kmp_task_t for task directive.
virtual void emitMasterRegion(CodeGenFunction &CGF, const RegionCodeGenTy &MasterOpGen, SourceLocation Loc)
Emits a master region.
virtual llvm::Function * emitTaskOutlinedFunction(const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, const VarDecl *PartIDVar, const VarDecl *TaskTVar, OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, bool Tied, unsigned &NumberOfParts)
Emits outlined function for the OpenMP task directive D.
llvm::DenseMap< llvm::Function *, unsigned > FunctionToUntiedTaskStackMap
Maps function to the position of the untied task locals stack.
void emitHostKernelEnvironment(const OMPExecutableDirective &D, CodeGenFunction &CGF)
Emit the '<kernel>_kernel_environment' global for a target region that is compiled for a non-GPU (hos...
void emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal, SourceLocation Loc)
Emits the code to destroy the dependency object provided in depobj directive.
virtual void emitTaskReductionFixups(CodeGenFunction &CGF, SourceLocation Loc, ReductionCodeGen &RCG, unsigned N)
Required to resolve existing problems in the runtime.
llvm::ArrayType * KmpCriticalNameTy
Type kmp_critical_name, originally defined as typedef kmp_int32 kmp_critical_name[8];.
virtual void emitDoacrossOrdered(CodeGenFunction &CGF, const OMPDependClause *C)
Emit code for doacross ordered directive with 'depend' clause.
llvm::DenseMap< const OMPDeclareMapperDecl *, llvm::Function * > UDMMap
Map from the user-defined mapper declaration to its corresponding functions.
virtual void checkAndEmitLastprivateConditional(CodeGenFunction &CGF, const Expr *LHS)
Checks if the provided LVal is lastprivate conditional and emits the code to update the value of the ...
std::pair< llvm::Value *, LValue > getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal, SourceLocation Loc)
Returns the number of the elements and the address of the depobj dependency array.
llvm::SmallDenseSet< const VarDecl * > DeferredGlobalVariables
List of variables that can become declare target implicitly and, thus, must be emitted.
void emitUsesAllocatorsInit(CodeGenFunction &CGF, const Expr *Allocator, const Expr *AllocatorTraits)
Initializes user defined allocators specified in the uses_allocators clauses.
virtual void registerVTable(const OMPExecutableDirective &D)
Emit code for registering vtable by scanning through map clause in OpenMP target region.
llvm::Type * KmpRoutineEntryPtrTy
Type typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *);.
llvm::Type * getIdentTyPointerTy()
Returns pointer to ident_t type.
void emitSingleReductionCombiner(CodeGenFunction &CGF, const Expr *ReductionOp, const Expr *PrivateRef, const DeclRefExpr *LHS, const DeclRefExpr *RHS)
Emits single reduction combiner.
llvm::OpenMPIRBuilder & getOMPBuilder()
virtual void emitTargetOutlinedFunction(const OMPExecutableDirective &D, StringRef ParentName, llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, bool IsOffloadEntry, const RegionCodeGenTy &CodeGen)
Emit outilined function for 'target' directive.
virtual void emitCriticalRegion(CodeGenFunction &CGF, StringRef CriticalName, const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, const Expr *Hint=nullptr)
Emits a critical region.
virtual void emitForOrderedIterationEnd(CodeGenFunction &CGF, SourceLocation Loc, unsigned IVSize, bool IVSigned)
Call the appropriate runtime routine to notify that we finished iteration of the ordered loop with th...
virtual void emitOutlinedFunctionCall(CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, ArrayRef< llvm::Value * > Args={}) const
Emits call of the outlined function with the provided arguments, translating these arguments to corre...
llvm::Value * emitNumThreadsForTargetDirective(CodeGenFunction &CGF, const OMPExecutableDirective &D)
Emit an expression that denotes the number of threads a target region shall use.
void emitThreadPrivateVarInit(CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc)
Emits initialization code for the threadprivate variables.
virtual void emitUserDefinedReduction(CodeGenFunction *CGF, const OMPDeclareReductionDecl *D)
Emit code for the specified user defined reduction construct.
virtual void checkAndEmitSharedLastprivateConditional(CodeGenFunction &CGF, const OMPExecutableDirective &D, const llvm::DenseSet< CanonicalDeclPtr< const VarDecl > > &IgnoredDecls)
Checks if the lastprivate conditional was updated in inner region and writes the value.
QualType KmpDimTy
struct kmp_dim { // loop bounds info casted to kmp_int64 kmp_int64 lo; // lower kmp_int64 up; // uppe...
virtual void emitInlinedDirective(CodeGenFunction &CGF, OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, bool HasCancel=false)
Emit code for the directive that does not require outlining.
virtual void registerTargetGlobalVariable(const VarDecl *VD, llvm::Constant *Addr)
Checks if the provided global decl GD is a declare target variable and registers it when emitting cod...
virtual void emitFunctionProlog(CodeGenFunction &CGF, const Decl *D)
Emits OpenMP-specific function prolog.
void emitKmpRoutineEntryT(QualType KmpInt32Ty)
Build type kmp_routine_entry_t (if not built yet).
virtual bool isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, bool Chunked) const
Check if the specified ScheduleKind is static chunked.
virtual void emitTargetCall(CodeGenFunction &CGF, const OMPExecutableDirective &D, llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, llvm::PointerIntPair< const Expr *, 2, OpenMPDeviceClauseModifier > Device, llvm::function_ref< llvm::Value *(CodeGenFunction &CGF, const OMPLoopDirective &D)> SizeEmitter)
Emit the target offloading code associated with D.
virtual bool hasAllocateAttributeForGlobalVar(const VarDecl *VD, LangAS &AS)
Checks if the variable has associated OMPAllocateDeclAttr attribute with the predefined allocator and...
llvm::AtomicOrdering RequiresAtomicOrdering
Atomic ordering from the omp requires directive.
virtual void emitReduction(CodeGenFunction &CGF, SourceLocation Loc, ArrayRef< const Expr * > Privates, ArrayRef< const Expr * > LHSExprs, ArrayRef< const Expr * > RHSExprs, ArrayRef< const Expr * > ReductionOps, ReductionOptionsTy Options)
Emit a code for reduction clause.
std::pair< llvm::Value *, Address > emitDependClause(CodeGenFunction &CGF, ArrayRef< OMPTaskDataTy::DependData > Dependencies, SourceLocation Loc)
Emits list of dependecies based on the provided data (array of dependence/expression pairs).
llvm::StringMap< llvm::WeakTrackingVH > EmittedNonTargetVariables
List of the global variables with their addresses that should not be emitted for the target.
virtual bool isDynamic(OpenMPScheduleClauseKind ScheduleKind) const
Check if the specified ScheduleKind is dynamic.
Address emitLastprivateConditionalInit(CodeGenFunction &CGF, const VarDecl *VD)
Create specialized alloca to handle lastprivate conditionals.
virtual void emitOrderedRegion(CodeGenFunction &CGF, const RegionCodeGenTy &OrderedOpGen, SourceLocation Loc, bool IsThreads)
Emit an ordered region.
virtual Address getAddressOfLocalVariable(CodeGenFunction &CGF, const VarDecl *VD)
Gets the OpenMP-specific address of the local variable.
virtual void emitTaskReductionFini(CodeGenFunction &CGF, SourceLocation Loc, bool IsWorksharingReduction)
Emits the following code for reduction clause with task modifier:
virtual void emitMaskedRegion(CodeGenFunction &CGF, const RegionCodeGenTy &MaskedOpGen, SourceLocation Loc, const Expr *Filter=nullptr)
Emits a masked region.
QualType KmpDependInfoTy
Type typedef struct kmp_depend_info { kmp_intptr_t base_addr; size_t len; struct { bool in:1; bool ou...
llvm::Function * emitReductionFunction(StringRef ReducerName, SourceLocation Loc, llvm::Type *ArgsElemType, ArrayRef< const Expr * > Privates, ArrayRef< const Expr * > LHSExprs, ArrayRef< const Expr * > RHSExprs, ArrayRef< const Expr * > ReductionOps)
Emits reduction function.
virtual void emitForDispatchInit(CodeGenFunction &CGF, SourceLocation Loc, const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, bool Ordered, const DispatchRTInput &DispatchValues)
Call the appropriate runtime routine to initialize it before start of loop.
Address getTaskReductionItem(CodeGenFunction &CGF, SourceLocation Loc, llvm::Value *ReductionsPtr, LValue SharedLVal) override
Get the address of void * type of the privatue copy of the reduction item specified by the SharedLVal...
void emitCriticalRegion(CodeGenFunction &CGF, StringRef CriticalName, const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, const Expr *Hint=nullptr) override
Emits a critical region.
void emitDistributeStaticInit(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) override
void emitForStaticInit(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) override
Call the appropriate runtime routine to initialize it before start of loop.
bool emitTargetGlobalVariable(GlobalDecl GD) override
Emit the global variable if it is a valid device global variable.
llvm::Value * emitForNext(CodeGenFunction &CGF, SourceLocation Loc, unsigned IVSize, bool IVSigned, Address IL, Address LB, Address UB, Address ST) override
Call __kmpc_dispatch_next( ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, kmp_int[32|64] *p_lowe...
llvm::Function * emitThreadPrivateVarDefinition(const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, CodeGenFunction *CGF=nullptr) override
Emit a code for initialization of threadprivate variable.
void emitTargetDataStandAloneCall(CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, const Expr *Device) override
Emit the data mapping/movement code associated with the directive D that should be of the form 'targe...
llvm::Function * emitTeamsOutlinedFunction(CodeGenFunction &CGF, const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) override
Emits outlined function for the specified OpenMP teams directive D.
void emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, llvm::Function *OutlinedFn, ArrayRef< llvm::Value * > CapturedVars, const Expr *IfCond, llvm::Value *NumThreads, OpenMPNumThreadsClauseModifier NumThreadsModifier=OMPC_NUMTHREADS_unknown, OpenMPSeverityClauseKind Severity=OMPC_SEVERITY_fatal, const Expr *Message=nullptr) override
Emits code for parallel or serial call of the OutlinedFn with variables captured in a record which ad...
void emitReduction(CodeGenFunction &CGF, SourceLocation Loc, ArrayRef< const Expr * > Privates, ArrayRef< const Expr * > LHSExprs, ArrayRef< const Expr * > RHSExprs, ArrayRef< const Expr * > ReductionOps, ReductionOptionsTy Options) override
Emit a code for reduction clause.
void emitFlush(CodeGenFunction &CGF, ArrayRef< const Expr * > Vars, SourceLocation Loc, llvm::AtomicOrdering AO) override
Emit flush of the variables specified in 'omp flush' directive.
void emitDoacrossOrdered(CodeGenFunction &CGF, const OMPDependClause *C) override
Emit code for doacross ordered directive with 'depend' clause.
void emitTaskyieldCall(CodeGenFunction &CGF, SourceLocation Loc) override
Emits a masked region.
Address getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, QualType VarType, StringRef Name) override
Creates artificial threadprivate variable with name Name and type VarType.
Address getAddrOfThreadPrivate(CodeGenFunction &CGF, const VarDecl *VD, Address VDAddr, SourceLocation Loc) override
Returns address of the threadprivate variable for the current thread.
void emitSingleRegion(CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, SourceLocation Loc, ArrayRef< const Expr * > CopyprivateVars, ArrayRef< const Expr * > DestExprs, ArrayRef< const Expr * > SrcExprs, ArrayRef< const Expr * > AssignmentOps) override
Emits a single region.
void emitTaskReductionFixups(CodeGenFunction &CGF, SourceLocation Loc, ReductionCodeGen &RCG, unsigned N) override
Required to resolve existing problems in the runtime.
llvm::Function * emitParallelOutlinedFunction(CodeGenFunction &CGF, const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) override
Emits outlined function for the specified OpenMP parallel directive D.
void emitCancellationPointCall(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind CancelRegion) override
Emit code for 'cancellation point' construct.
void emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind Kind, bool EmitChecks=true, bool ForceSimpleCall=false) override
Emit an implicit/explicit barrier for OpenMP threads.
Address getParameterAddress(CodeGenFunction &CGF, const VarDecl *NativeParam, const VarDecl *TargetParam) const override
Gets the address of the native argument basing on the address of the target-specific parameter.
void emitTeamsCall(CodeGenFunction &CGF, const OMPExecutableDirective &D, SourceLocation Loc, llvm::Function *OutlinedFn, ArrayRef< llvm::Value * > CapturedVars) override
Emits code for teams call of the OutlinedFn with variables captured in a record which address is stor...
void emitForOrderedIterationEnd(CodeGenFunction &CGF, SourceLocation Loc, unsigned IVSize, bool IVSigned) override
Call the appropriate runtime routine to notify that we finished iteration of the ordered loop with th...
bool emitTargetGlobal(GlobalDecl GD) override
Emit the global GD if it is meaningful for the target.
void emitTaskReductionFini(CodeGenFunction &CGF, SourceLocation Loc, bool IsWorksharingReduction) override
Emits the following code for reduction clause with task modifier:
void emitOrderedRegion(CodeGenFunction &CGF, const RegionCodeGenTy &OrderedOpGen, SourceLocation Loc, bool IsThreads) override
Emit an ordered region.
void emitForStaticFinish(CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind) override
Call the appropriate runtime routine to notify that we finished all the work with current loop.
llvm::Value * emitTaskReductionInit(CodeGenFunction &CGF, SourceLocation Loc, ArrayRef< const Expr * > LHSExprs, ArrayRef< const Expr * > RHSExprs, const OMPTaskDataTy &Data) override
Emit a code for initialization of task reduction clause.
void emitProcBindClause(CodeGenFunction &CGF, llvm::omp::ProcBindKind ProcBind, SourceLocation Loc) override
Emit call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32global_tid, int proc_bind) to generate...
void emitTargetOutlinedFunction(const OMPExecutableDirective &D, StringRef ParentName, llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) override
Emit outilined function for 'target' directive.
void emitMasterRegion(CodeGenFunction &CGF, const RegionCodeGenTy &MasterOpGen, SourceLocation Loc) override
Emits a master region.
void emitNumTeamsClause(CodeGenFunction &CGF, const Expr *NumTeams, const Expr *ThreadLimit, SourceLocation Loc) override
Emits call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32global_tid, kmp_int32 num_teams,...
void emitForDispatchDeinit(CodeGenFunction &CGF, SourceLocation Loc) override
This is used for non static scheduled types and when the ordered clause is present on the loop constr...
const VarDecl * translateParameter(const FieldDecl *FD, const VarDecl *NativeParam) const override
Translates the native parameter of outlined function if this is required for target.
void emitNumThreadsClause(CodeGenFunction &CGF, llvm::Value *NumThreads, SourceLocation Loc, OpenMPNumThreadsClauseModifier Modifier=OMPC_NUMTHREADS_unknown, OpenMPSeverityClauseKind Severity=OMPC_SEVERITY_fatal, SourceLocation SeverityLoc=SourceLocation(), const Expr *Message=nullptr, SourceLocation MessageLoc=SourceLocation()) override
Emits call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32global_tid, kmp_int32 num_threads) ...
void emitMaskedRegion(CodeGenFunction &CGF, const RegionCodeGenTy &MaskedOpGen, SourceLocation Loc, const Expr *Filter=nullptr) override
Emits a masked region.
void emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, const OMPExecutableDirective &D, llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, const Expr *IfCond, const OMPTaskDataTy &Data) override
Emit task region for the task directive.
void emitTargetCall(CodeGenFunction &CGF, const OMPExecutableDirective &D, llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, llvm::PointerIntPair< const Expr *, 2, OpenMPDeviceClauseModifier > Device, llvm::function_ref< llvm::Value *(CodeGenFunction &CGF, const OMPLoopDirective &D)> SizeEmitter) override
Emit the target offloading code associated with D.
bool emitTargetFunctions(GlobalDecl GD) override
Emit the target regions enclosed in GD function definition or the function itself in case it is a val...
void emitDoacrossInit(CodeGenFunction &CGF, const OMPLoopDirective &D, ArrayRef< Expr * > NumIterations) override
Emit initialization for doacross loop nesting support.
void emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, const Expr *IfCond, OpenMPDirectiveKind CancelRegion) override
Emit code for 'cancel' construct.
void emitTaskwaitCall(CodeGenFunction &CGF, SourceLocation Loc, const OMPTaskDataTy &Data) override
Emit code for 'taskwait' directive.
void emitTaskgroupRegion(CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, SourceLocation Loc) override
Emit a taskgroup region.
void emitTargetDataCalls(CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, const Expr *Device, const RegionCodeGenTy &CodeGen, CGOpenMPRuntime::TargetDataInfo &Info) override
Emit the target data mapping code associated with D.
void emitForDispatchInit(CodeGenFunction &CGF, SourceLocation Loc, const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, bool Ordered, const DispatchRTInput &DispatchValues) override
This is used for non static scheduled types and when the ordered clause is present on the loop constr...
llvm::Function * emitTaskOutlinedFunction(const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, const VarDecl *PartIDVar, const VarDecl *TaskTVar, OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, bool Tied, unsigned &NumberOfParts) override
Emits outlined function for the OpenMP task directive D.
void emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, const Expr *IfCond, const OMPTaskDataTy &Data) override
Emit task region for the taskloop directive.
unsigned getNonVirtualBaseLLVMFieldNo(const CXXRecordDecl *RD) const
llvm::StructType * getLLVMType() const
Return the "complete object" LLVM type associated with this record.
llvm::StructType * getBaseSubobjectLLVMType() const
Return the "base subobject" LLVM type associated with this record.
unsigned getLLVMFieldNo(const FieldDecl *FD) const
Return llvm::StructType element number that corresponds to the field FD.
unsigned getVirtualBaseIndex(const CXXRecordDecl *base) const
Return the LLVM field index corresponding to the given virtual base.
API for captured statement code generation.
virtual void EmitBody(CodeGenFunction &CGF, const Stmt *S)
Emit the captured statement body.
virtual const FieldDecl * lookup(const VarDecl *VD) const
Lookup the captured field decl for a variable.
RAII for correct setting/restoring of CapturedStmtInfo.
The scope used to remap some variables as private in the OpenMP loop body (or other captured region e...
bool Privatize()
Privatizes local variables previously registered as private.
bool addPrivate(const ValueDecl *LocalVD, Address Addr)
Registers LocalVD variable as a private with Addr as the address of the corresponding private variabl...
An RAII object to set (and then clear) a mapping for an OpaqueValueExpr.
Enters a new scope for capturing cleanups, all of which will be executed once the scope is exited.
CodeGenFunction - This class organizes the per-function state that is used while generating LLVM code...
LValue EmitLoadOfReferenceLValue(LValue RefLVal)
Definition CGExpr.cpp:3453
void EmitBranchOnBoolExpr(const Expr *Cond, llvm::BasicBlock *TrueBlock, llvm::BasicBlock *FalseBlock, uint64_t TrueCount, Stmt::Likelihood LH=Stmt::LH_None, const Expr *ConditionalOp=nullptr, const VarDecl *ConditionalDecl=nullptr)
EmitBranchOnBoolExpr - Emit a branch on a boolean condition (e.g.
void emitDestroy(Address addr, QualType type, Destroyer *destroyer, bool useEHCleanupForArray)
emitDestroy - Immediately perform the destruction of the given object.
Definition CGDecl.cpp:2483
JumpDest getJumpDestInCurrentScope(llvm::BasicBlock *Target)
The given basic block lies in the current EH scope, but may be a target of a potentially scope-crossi...
static void EmitOMPTargetParallelDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetParallelDirective &S)
void EmitNullInitialization(Address DestPtr, QualType Ty)
EmitNullInitialization - Generate code to set a value of the given type to null, If the type contains...
CGCapturedStmtInfo * CapturedStmtInfo
ComplexPairTy EmitLoadOfComplex(LValue src, SourceLocation loc)
EmitLoadOfComplex - Load a complex number from the specified l-value.
static void EmitOMPTargetDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetDirective &S)
Emit device code for the target directive.
static void EmitOMPTargetTeamsDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetTeamsDirective &S)
Emit device code for the target teams directive.
static void EmitOMPTargetTeamsDistributeDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetTeamsDistributeDirective &S)
Emit device code for the target teams distribute directive.
llvm::Function * GenerateOpenMPCapturedStmtFunctionAggregate(const CapturedStmt &S, const OMPExecutableDirective &D)
llvm::BasicBlock * createBasicBlock(const Twine &name="", llvm::Function *parent=nullptr, llvm::BasicBlock *before=nullptr)
createBasicBlock - Create an LLVM basic block.
const LangOptions & getLangOpts() const
AutoVarEmission EmitAutoVarAlloca(const VarDecl &var)
EmitAutoVarAlloca - Emit the alloca and debug information for a local variable.
Definition CGDecl.cpp:1494
void pushDestroy(QualType::DestructionKind dtorKind, Address addr, QualType type)
pushDestroy - Push the standard destructor for the given type as at least a normal cleanup.
Definition CGDecl.cpp:2366
Address EmitLoadOfPointer(Address Ptr, const PointerType *PtrTy, LValueBaseInfo *BaseInfo=nullptr, TBAAAccessInfo *TBAAInfo=nullptr)
Load a pointer with type PtrTy stored at address Ptr.
Definition CGExpr.cpp:3462
DeclMapTy::iterator localDeclMapEnd()
void EmitBranchThroughCleanup(JumpDest Dest)
EmitBranchThroughCleanup - Emit a branch from the current insert block through the normal cleanup han...
const Decl * CurCodeDecl
CurCodeDecl - This is the inner-most code context, which includes blocks.
Destroyer * getDestroyer(QualType::DestructionKind destructionKind)
Definition CGDecl.cpp:2339
llvm::AssertingVH< llvm::Instruction > AllocaInsertPt
AllocaInsertPoint - This is an instruction in the entry block before which we prefer to insert alloca...
void EmitAggregateAssign(LValue Dest, LValue Src, QualType EltTy)
Emit an aggregate assignment.
JumpDest ReturnBlock
ReturnBlock - Unified return block.
void EmitAggregateCopy(LValue Dest, LValue Src, QualType EltTy, AggValueSlot::Overlap_t MayOverlap, bool isVolatile=false)
EmitAggregateCopy - Emit an aggregate copy.
LValue EmitLValueForField(LValue Base, const FieldDecl *Field, bool IsInBounds=true)
Definition CGExpr.cpp:5981
RawAddress CreateDefaultAlignTempAlloca(llvm::Type *Ty, const Twine &Name="tmp")
CreateDefaultAlignedTempAlloca - This creates an alloca with the default ABI alignment of the given L...
Definition CGExpr.cpp:185
void GenerateOpenMPCapturedVars(const CapturedStmt &S, SmallVectorImpl< llvm::Value * > &CapturedVars)
DeclMapTy::iterator findLocalDecl(const Decl *D)
Accessors for LocalDeclMap.
void EmitIgnoredExpr(const Expr *E)
EmitIgnoredExpr - Emit an expression in a context which ignores the result.
Definition CGExpr.cpp:261
RValue EmitLoadOfLValue(LValue V, SourceLocation Loc)
EmitLoadOfLValue - Given an expression that represents a value lvalue, this method emits the address ...
Definition CGExpr.cpp:2558
std::pair< DeclMapTy::iterator, bool > insertLocalDecl(const Decl *D, Address Addr)
LValue EmitArraySectionExpr(const ArraySectionExpr *E, bool IsLowerBound=true)
Definition CGExpr.cpp:5490
LValue EmitOMPSharedLValue(const Expr *E)
Emits the lvalue for the expression with possibly captured variable.
void StartFunction(GlobalDecl GD, QualType RetTy, llvm::Function *Fn, const CGFunctionInfo &FnInfo, const FunctionArgList &Args, SourceLocation Loc=SourceLocation(), SourceLocation StartLoc=SourceLocation())
Emit code for the start of a function.
void EmitOMPCopy(QualType OriginalType, Address DestAddr, Address SrcAddr, const VarDecl *DestVD, const VarDecl *SrcVD, const Expr *Copy)
Emit proper copying of data from one variable to another.
llvm::Value * EvaluateExprAsBool(const Expr *E)
EvaluateExprAsBool - Perform the usual unary conversions on the specified expression and compare the ...
Definition CGExpr.cpp:242
JumpDest getOMPCancelDestination(OpenMPDirectiveKind Kind)
llvm::Value * emitArrayLength(const ArrayType *arrayType, QualType &baseType, Address &addr)
emitArrayLength - Compute the length of an array, even if it's a VLA, and drill down to the base elem...
void EmitOMPAggregateAssign(Address DestAddr, Address SrcAddr, QualType OriginalType, const llvm::function_ref< void(Address, Address)> CopyGen)
Perform element by element copying of arrays with type OriginalType from SrcAddr to DestAddr using co...
bool HaveInsertPoint() const
HaveInsertPoint - True if an insertion point is defined.
llvm::Value * getTypeSize(QualType Ty)
Returns calculated size of the specified type.
LValue MakeRawAddrLValue(llvm::Value *V, QualType T, CharUnits Alignment, AlignmentSource Source=AlignmentSource::Type)
Same as MakeAddrLValue above except that the pointer is known to be unsigned.
LValue EmitLValueForFieldInitialization(LValue Base, const FieldDecl *Field)
EmitLValueForFieldInitialization - Like EmitLValueForField, except that if the Field is a reference,...
Definition CGExpr.cpp:6143
RawAddress CreateMemTempWithoutCast(QualType T, const Twine &Name="tmp")
CreateMemTemp - Create a temporary memory object of the given type, with appropriate alignmen without...
Definition CGExpr.cpp:234
VlaSizePair getVLASize(const VariableArrayType *vla)
Returns an LLVM value that corresponds to the size, in non-variably-sized elements,...
llvm::CallInst * EmitNounwindRuntimeCall(llvm::FunctionCallee callee, const Twine &name="")
llvm::Value * EmitLoadOfScalar(Address Addr, bool Volatile, QualType Ty, SourceLocation Loc, AlignmentSource Source=AlignmentSource::Type, bool isNontemporal=false)
EmitLoadOfScalar - Load a scalar value from an address, taking care to appropriately convert from the...
void EmitStoreOfComplex(ComplexPairTy V, LValue dest, bool isInit)
EmitStoreOfComplex - Store a complex number into the specified l-value.
const Decl * CurFuncDecl
CurFuncDecl - Holds the Decl for the current outermost non-closure context.
void EmitAutoVarCleanups(const AutoVarEmission &emission)
Definition CGDecl.cpp:2285
void EmitStoreThroughLValue(RValue Src, LValue Dst, bool isInit=false)
EmitStoreThroughLValue - Store the specified rvalue into the specified lvalue, where both are guarant...
Definition CGExpr.cpp:2810
LValue EmitLoadOfPointerLValue(Address Ptr, const PointerType *PtrTy)
Definition CGExpr.cpp:3472
void EmitAnyExprToMem(const Expr *E, Address Location, Qualifiers Quals, bool IsInitializer)
EmitAnyExprToMem - Emits the code necessary to evaluate an arbitrary expression into the given memory...
Definition CGExpr.cpp:312
bool needsEHCleanup(QualType::DestructionKind kind)
Determines whether an EH cleanup is required to destroy a type with the given destruction kind.
llvm::DenseMap< const ValueDecl *, FieldDecl * > LambdaCaptureFields
llvm::CallInst * EmitRuntimeCall(llvm::FunctionCallee callee, const Twine &name="")
llvm::Type * ConvertTypeForMem(QualType T)
static void EmitOMPTargetTeamsDistributeParallelForDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetTeamsDistributeParallelForDirective &S)
static void EmitOMPTargetParallelForSimdDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetParallelForSimdDirective &S)
Emit device code for the target parallel for simd directive.
CodeGenTypes & getTypes() const
static TypeEvaluationKind getEvaluationKind(QualType T)
getEvaluationKind - Return the TypeEvaluationKind of QualType T.
void EmitOMPTargetTaskBasedDirective(const OMPExecutableDirective &S, const RegionCodeGenTy &BodyGen, OMPTargetDataInfo &InputInfo)
Address EmitPointerWithAlignment(const Expr *Addr, LValueBaseInfo *BaseInfo=nullptr, TBAAAccessInfo *TBAAInfo=nullptr, KnownNonNull_t IsKnownNonNull=NotKnownNonNull)
EmitPointerWithAlignment - Given an expression with a pointer type, emit the value and compute our be...
Definition CGExpr.cpp:1624
static void EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetTeamsDistributeParallelForSimdDirective &S)
Emit device code for the target teams distribute parallel for simd directive.
void EmitBranch(llvm::BasicBlock *Block)
EmitBranch - Emit a branch to the specified basic block from the current insert block,...
Definition CGStmt.cpp:675
llvm::Function * GenerateOpenMPCapturedStmtFunction(const CapturedStmt &S, const OMPExecutableDirective &D)
RawAddress CreateMemTemp(QualType T, const Twine &Name="tmp", RawAddress *Alloca=nullptr)
CreateMemTemp - Create a temporary memory object of the given type, with appropriate alignmen and cas...
Definition CGExpr.cpp:198
void EmitVarDecl(const VarDecl &D)
EmitVarDecl - Emit a local variable declaration.
Definition CGDecl.cpp:211
llvm::Value * EmitCheckedInBoundsGEP(llvm::Type *ElemTy, llvm::Value *Ptr, ArrayRef< llvm::Value * > IdxList, bool SignedIndices, bool IsSubtraction, SourceLocation Loc, const Twine &Name="")
Same as IRBuilder::CreateInBoundsGEP, but additionally emits a check to detect undefined behavior whe...
static void EmitOMPTargetParallelGenericLoopDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetParallelGenericLoopDirective &S)
Emit device code for the target parallel loop directive.
llvm::Value * EmitScalarExpr(const Expr *E, bool IgnoreResultAssign=false)
EmitScalarExpr - Emit the computation of the specified expression of LLVM scalar type,...
static bool IsWrappedCXXThis(const Expr *E)
Check if E is a C++ "this" pointer wrapped in value-preserving casts.
Definition CGExpr.cpp:1682
LValue MakeAddrLValue(Address Addr, QualType T, AlignmentSource Source=AlignmentSource::Type)
void FinishFunction(SourceLocation EndLoc=SourceLocation())
FinishFunction - Complete IR generation of the current function.
void EmitAtomicStore(RValue rvalue, LValue lvalue, bool isInit)
static void EmitOMPTargetSimdDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetSimdDirective &S)
Emit device code for the target simd directive.
static void EmitOMPTargetParallelForDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetParallelForDirective &S)
Emit device code for the target parallel for directive.
Address GetAddrOfLocalVar(const VarDecl *VD)
GetAddrOfLocalVar - Return the address of a local variable.
bool ConstantFoldsToSimpleInteger(const Expr *Cond, bool &Result, bool AllowLabels=false)
ConstantFoldsToSimpleInteger - If the specified expression does not fold to a constant,...
static void EmitOMPTargetTeamsGenericLoopDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetTeamsGenericLoopDirective &S)
Emit device code for the target teams loop directive.
LValue EmitMemberExpr(const MemberExpr *E)
Definition CGExpr.cpp:5759
std::pair< llvm::Value *, llvm::Value * > ComplexPairTy
Address ReturnValue
ReturnValue - The temporary alloca to hold the return value.
LValue EmitLValue(const Expr *E, KnownNonNull_t IsKnownNonNull=NotKnownNonNull)
EmitLValue - Emit code to compute a designator that specifies the location of the expression.
Definition CGExpr.cpp:1740
void incrementProfileCounter(const Stmt *S, llvm::Value *StepV=nullptr)
Increment the profiler's counter for the given statement by StepV.
static void EmitOMPTargetTeamsDistributeSimdDeviceFunction(CodeGenModule &CGM, StringRef ParentName, const OMPTargetTeamsDistributeSimdDirective &S)
Emit device code for the target teams distribute simd directive.
llvm::Value * EmitScalarConversion(llvm::Value *Src, QualType SrcTy, QualType DstTy, SourceLocation Loc)
Emit a conversion from the specified type to the specified destination type, both of which are LLVM s...
void EmitVariablyModifiedType(QualType Ty)
EmitVLASize - Capture all the sizes for the VLA expressions in the given variably-modified type and s...
bool isTrivialInitializer(const Expr *Init)
Determine whether the given initializer is trivial in the sense that it requires no code to be genera...
Definition CGDecl.cpp:1868
void EmitStoreOfScalar(llvm::Value *Value, Address Addr, bool Volatile, QualType Ty, AlignmentSource Source=AlignmentSource::Type, bool isInit=false, bool isNontemporal=false)
EmitStoreOfScalar - Store a scalar value to an address, taking care to appropriately convert from the...
void EmitBlock(llvm::BasicBlock *BB, bool IsFinished=false)
EmitBlock - Emit the given block.
Definition CGStmt.cpp:655
void EmitExprAsInit(const Expr *init, const ValueDecl *D, LValue lvalue, bool capturedByInit)
EmitExprAsInit - Emits the code necessary to initialize a location in memory with the given initializ...
Definition CGDecl.cpp:2175
LValue MakeNaturalAlignRawAddrLValue(llvm::Value *V, QualType T)
This class organizes the cross-function state that is used while generating LLVM code.
void SetInternalFunctionAttributes(GlobalDecl GD, llvm::Function *F, const CGFunctionInfo &FI)
Set the attributes on the LLVM function for the given decl and function info.
llvm::Module & getModule() const
const IntrusiveRefCntPtr< llvm::vfs::FileSystem > & getFileSystem() const
DiagnosticsEngine & getDiags() const
const LangOptions & getLangOpts() const
CharUnits getNaturalTypeAlignment(QualType T, LValueBaseInfo *BaseInfo=nullptr, TBAAAccessInfo *TBAAInfo=nullptr, bool forPointeeType=false)
const llvm::DataLayout & getDataLayout() const
CGOpenMPRuntime & getOpenMPRuntime()
Return a reference to the configured OpenMP runtime.
TBAAAccessInfo getTBAAInfoForSubobject(LValue Base, QualType AccessType)
getTBAAInfoForSubobject - Get TBAA information for an access with a given base lvalue.
ASTContext & getContext() const
const CodeGenOptions & getCodeGenOpts() const
StringRef getMangledName(GlobalDecl GD)
std::optional< CharUnits > getOMPAllocateAlignment(const VarDecl *VD)
Return the alignment specified in an allocate directive, if present.
Definition CGDecl.cpp:3035
llvm::Constant * EmitNullConstant(QualType T)
Return the result of value-initializing the given type, i.e.
llvm::Type * ConvertType(QualType T)
ConvertType - Convert type T into a llvm::Type.
llvm::FunctionType * GetFunctionType(const CGFunctionInfo &Info)
GetFunctionType - Get the LLVM function type for.
Definition CGCall.cpp:2064
const CGFunctionInfo & arrangeBuiltinFunctionDeclaration(QualType resultType, const FunctionArgList &args)
A builtin function is a freestanding function using the default C conventions.
Definition CGCall.cpp:774
const CGRecordLayout & getCGRecordLayout(const RecordDecl *)
getCGRecordLayout - Return record layout info for the given record decl.
llvm::GlobalVariable * GetAddrOfVTable(const CXXRecordDecl *RD)
GetAddrOfVTable - Get the address of the VTable for the given record decl.
Definition CGVTables.cpp:43
A specialization of Address that requires the address to be an LLVM Constant.
Definition Address.h:296
static ConstantAddress invalid()
Definition Address.h:304
void pushTerminate()
Push a terminate handler on the stack.
void popTerminate()
Pops a terminate handler off the stack.
Definition CGCleanup.h:658
FunctionArgList - Type for representing both the decl and type of parameters to a function.
Definition CGCall.h:378
LValue - This represents an lvalue references.
Definition CGValue.h:183
CharUnits getAlignment() const
Definition CGValue.h:355
llvm::Value * getPointer(CodeGenFunction &CGF) const
const Qualifiers & getQuals() const
Definition CGValue.h:350
Address getAddress() const
Definition CGValue.h:373
LValueBaseInfo getBaseInfo() const
Definition CGValue.h:358
QualType getType() const
Definition CGValue.h:303
TBAAAccessInfo getTBAAInfo() const
Definition CGValue.h:347
A basic class for pre|post-action for advanced codegen sequence for OpenMP region.
virtual void Enter(CodeGenFunction &CGF)
RValue - This trivial value class is used to represent the result of an expression that is evaluated.
Definition CGValue.h:42
static RValue get(llvm::Value *V)
Definition CGValue.h:99
static RValue getComplex(llvm::Value *V1, llvm::Value *V2)
Definition CGValue.h:109
llvm::Value * getScalarVal() const
getScalarVal() - Return the Value* of this scalar value.
Definition CGValue.h:72
An abstract representation of an aligned address.
Definition Address.h:42
llvm::Type * getElementType() const
Return the type of the values stored in this address.
Definition Address.h:77
llvm::Value * getPointer() const
Definition Address.h:66
static RawAddress invalid()
Definition Address.h:61
Class intended to support codegen of all kind of the reduction clauses.
LValue getSharedLValue(unsigned N) const
Returns LValue for the reduction item.
const Expr * getRefExpr(unsigned N) const
Returns the base declaration of the reduction item.
LValue getOrigLValue(unsigned N) const
Returns LValue for the original reduction item.
bool needCleanups(unsigned N)
Returns true if the private copy requires cleanups.
void emitAggregateType(CodeGenFunction &CGF, unsigned N)
Emits the code for the variable-modified type, if required.
const VarDecl * getBaseDecl(unsigned N) const
Returns the base declaration of the reduction item.
QualType getPrivateType(unsigned N) const
Return the type of the private item.
bool usesReductionInitializer(unsigned N) const
Returns true if the initialization of the reduction item uses initializer from declare reduction cons...
void emitSharedOrigLValue(CodeGenFunction &CGF, unsigned N)
Emits lvalue for the shared and original reduction item.
void emitInitialization(CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr, llvm::function_ref< bool(CodeGenFunction &)> DefaultInit)
Performs initialization of the private copy for the reduction item.
std::pair< llvm::Value *, llvm::Value * > getSizes(unsigned N) const
Returns the size of the reduction item (in chars and total number of elements in the item),...
ReductionCodeGen(ArrayRef< const Expr * > Shareds, ArrayRef< const Expr * > Origs, ArrayRef< const Expr * > Privates, ArrayRef< const Expr * > ReductionOps)
void emitCleanups(CodeGenFunction &CGF, unsigned N, Address PrivateAddr)
Emits cleanup code for the reduction item.
Address adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, Address PrivateAddr)
Adjusts PrivatedAddr for using instead of the original variable address in normal operations.
Class provides a way to call simple version of codegen for OpenMP region, or an advanced with possibl...
void operator()(CodeGenFunction &CGF) const
void setAction(PrePostActionTy &Action) const
ConstStmtVisitor - This class implements a simple visitor for Stmt subclasses.
DeclContext - This is used only as base class of specific decl types that can act as declaration cont...
Definition DeclBase.h:1466
void addDecl(Decl *D)
Add the declaration D into this context.
A reference to a declared variable, function, enum, etc.
Definition Expr.h:1290
ValueDecl * getDecl()
Definition Expr.h:1358
Decl - This represents one declaration (or definition), e.g.
Definition DeclBase.h:86
T * getAttr() const
Definition DeclBase.h:581
bool hasAttrs() const
Definition DeclBase.h:526
ASTContext & getASTContext() const LLVM_READONLY
Definition DeclBase.cpp:550
void addAttr(Attr *A)
virtual Stmt * getBody() const
getBody - If this Decl represents a declaration for a body of code, such as a function or method defi...
Definition DeclBase.h:1104
llvm::iterator_range< specific_attr_iterator< T > > specific_attrs() const
Definition DeclBase.h:567
SourceLocation getLocation() const
Definition DeclBase.h:447
DeclContext * getDeclContext()
Definition DeclBase.h:456
AttrVec & getAttrs()
Definition DeclBase.h:532
bool hasAttr() const
Definition DeclBase.h:585
virtual Decl * getCanonicalDecl()
Retrieves the "canonical" declaration of the given declaration.
Definition DeclBase.h:995
SourceLocation getBeginLoc() const LLVM_READONLY
Definition Decl.h:832
DiagnosticBuilder Report(SourceLocation Loc, unsigned DiagID)
Issue the message to the client.
This represents one expression.
Definition Expr.h:113
bool isIntegerConstantExpr(const ASTContext &Ctx) const
bool isGLValue() const
Definition Expr.h:288
Expr * IgnoreParenNoopCasts(const ASTContext &Ctx) LLVM_READONLY
Skip past any parentheses and casts which do not change the value (including ptr->int casts of the sa...
Definition Expr.cpp:3153
@ SE_AllowSideEffects
Allow any unmodeled side effect.
Definition Expr.h:695
@ SE_AllowUndefinedBehavior
Allow UB that we can give a value, but not arbitrary unmodeled side effects.
Definition Expr.h:693
Expr * IgnoreParenCasts() LLVM_READONLY
Skip past any parentheses and casts which might surround this expression until reaching a fixed point...
Definition Expr.cpp:3131
llvm::APSInt EvaluateKnownConstInt(const ASTContext &Ctx) const
EvaluateKnownConstInt - Call EvaluateAsRValue and return the folded integer.
Expr * IgnoreParenImpCasts() LLVM_READONLY
Skip past any parentheses and implicit casts which might surround this expression until reaching a fi...
Definition Expr.cpp:3126
bool isEvaluatable(const ASTContext &Ctx, SideEffectsKind AllowSideEffects=SE_NoSideEffects) const
isEvaluatable - Call EvaluateAsRValue to see if this expression can be constant folded without side-e...
bool HasSideEffects(const ASTContext &Ctx, bool IncludePossibleEffects=true) const
HasSideEffects - This routine returns true for all those expressions which have any effect other than...
Definition Expr.cpp:3725
std::optional< llvm::APSInt > getIntegerConstantExpr(const ASTContext &Ctx, bool AllowRelaxedEval=false) const
isIntegerConstantExpr - Return the value if this expression is a valid integer constant expression.
bool EvaluateAsBooleanCondition(bool &Result, const ASTContext &Ctx, bool InConstantContext=false) const
EvaluateAsBooleanCondition - Return true if this is a constant which we can fold and convert to a boo...
SourceLocation getExprLoc() const LLVM_READONLY
getExprLoc - Return the preferred location for the arrow when diagnosing a problem with a generic exp...
Definition Expr.cpp:283
static bool isSameComparisonOperand(const Expr *E1, const Expr *E2)
Checks that the two Expr's will refer to the same value as a comparison operand.
Definition Expr.cpp:4366
QualType getType() const
Definition Expr.h:145
bool hasNonTrivialCall(const ASTContext &Ctx) const
Determine whether this expression involves a call to any function that is not trivial.
Definition Expr.cpp:4095
Represents a member of a struct/union/class.
Definition Decl.h:3295
unsigned getFieldIndex() const
Returns the index of this field within its record, as appropriate for passing to ASTRecordLayout::get...
Definition Decl.h:3380
const RecordDecl * getParent() const
Returns the parent of this field declaration, which is the struct in which this field is defined.
Definition Decl.h:3531
static FieldDecl * Create(const ASTContext &C, DeclContext *DC, SourceLocation StartLoc, SourceLocation IdLoc, const IdentifierInfo *Id, QualType T, TypeSourceInfo *TInfo, Expr *BW, bool Mutable, InClassInitStyle InitStyle)
Definition Decl.cpp:4765
Represents a function declaration or definition.
Definition Decl.h:2059
const ParmVarDecl * getParamDecl(unsigned i) const
Definition Decl.h:2928
QualType getReturnType() const
Definition Decl.h:2976
ArrayRef< ParmVarDecl * > parameters() const
Definition Decl.h:2905
FunctionDecl * getCanonicalDecl() override
Retrieves the "canonical" declaration of the given declaration.
Definition Decl.cpp:3789
FunctionDecl * getMostRecentDecl()
Returns the most recent (re)declaration of this declaration.
unsigned getNumParams() const
Return the number of parameters this function must have based on its FunctionType.
Definition Decl.cpp:3868
FunctionDecl * getPreviousDecl()
Return the previous declaration of this declaration or NULL if this is the first declaration.
GlobalDecl - represents a global declaration.
Definition GlobalDecl.h:60
const Decl * getDecl() const
Definition GlobalDecl.h:115
static ImplicitParamDecl * Create(ASTContext &C, DeclContext *DC, SourceLocation IdLoc, const IdentifierInfo *Id, QualType T, ImplicitParamKind ParamKind)
Create implicit parameter.
Definition Decl.cpp:5673
static IntegerLiteral * Create(const ASTContext &C, const llvm::APInt &V, QualType type, SourceLocation l)
Returns a new integer literal with value 'V' and type 'type'.
Definition Expr.cpp:985
An lvalue reference type, per C++11 [dcl.ref].
Definition TypeBase.h:3715
MemberExpr - [C99 6.5.2.3] Structure and Union Members.
Definition Expr.h:3408
ValueDecl * getMemberDecl() const
Retrieve the member declaration to which this expression refers.
Definition Expr.h:3491
Expr * getBase() const
Definition Expr.h:3485
StringRef getName() const
Get the name of identifier for this declaration as a StringRef.
Definition Decl.h:302
bool isExternallyVisible() const
Definition Decl.h:434
const Stmt * getPreInitStmt() const
Get pre-initialization statement for the clause.
This is a basic class for representing single OpenMP clause.
ArrayRef< OMPClause * > clauses() const
Definition DeclOpenMP.h:91
This represents 'pragma omp declare mapper ...' directive.
Definition DeclOpenMP.h:349
Expr * getMapperVarRef()
Get the variable declared in the mapper.
Definition DeclOpenMP.h:411
This represents 'pragma omp declare reduction ...' directive.
Definition DeclOpenMP.h:239
Expr * getInitializer()
Get initializer expression (if specified) of the declare reduction construct.
Definition DeclOpenMP.h:300
Expr * getInitPriv()
Get Priv variable of the initializer.
Definition DeclOpenMP.h:311
Expr * getCombinerOut()
Get Out variable of the combiner.
Definition DeclOpenMP.h:288
Expr * getCombinerIn()
Get In variable of the combiner.
Definition DeclOpenMP.h:285
Expr * getCombiner()
Get combiner expression of the declare reduction construct.
Definition DeclOpenMP.h:282
Expr * getInitOrig()
Get Orig variable of the initializer.
Definition DeclOpenMP.h:308
OMPDeclareReductionInitKind getInitializerKind() const
Get initializer kind.
Definition DeclOpenMP.h:303
This represents 'if' clause in the 'pragma omp ...' directive.
Expr * getCondition() const
Returns condition.
OMPIteratorHelperData & getHelper(unsigned I)
Fetches helper data for the specified iteration space.
Definition Expr.cpp:5647
unsigned numOfIterators() const
Returns number of iterator definitions.
Definition ExprOpenMP.h:275
This represents 'num_threads' clause in the 'pragma omp ...' directive.
This represents 'pragma omp requires...' directive.
Definition DeclOpenMP.h:479
clauselist_range clauselists()
Definition DeclOpenMP.h:504
This represents 'threadset' clause in the 'pragma omp task ...' directive.
OpaqueValueExpr - An expression referring to an opaque object of a fixed type and value class.
Definition Expr.h:1198
Represents a parameter to a function.
Definition Decl.h:1820
PointerType - C99 6.7.5.1 - Pointer Declarators.
Definition TypeBase.h:3403
Represents an unpacked "presumed" location which can be presented to the user.
unsigned getColumn() const
Return the presumed column number of this location.
const char * getFilename() const
Return the presumed filename of this location.
unsigned getLine() const
Return the presumed line number of this location.
A (possibly-)qualified type.
Definition TypeBase.h:938
void addRestrict()
Add the restrict qualifier to this QualType.
Definition TypeBase.h:1188
QualType withRestrict() const
Definition TypeBase.h:1191
bool isNull() const
Return true if this QualType doesn't point to a type yet.
Definition TypeBase.h:1005
const Type * getTypePtr() const
Retrieves a pointer to the underlying (unqualified) type.
Definition TypeBase.h:8446
Qualifiers getQualifiers() const
Retrieve the set of qualifiers applied to this type.
Definition TypeBase.h:8486
QualType getNonReferenceType() const
If Type is a reference type (e.g., const int&), returns the type that the reference refers to ("const...
Definition TypeBase.h:8631
QualType getCanonicalType() const
Definition TypeBase.h:8498
DestructionKind isDestructedType() const
Returns a nonzero value if objects of this type require non-trivial work to clean up after.
Definition TypeBase.h:1561
Represents a struct/union/class.
Definition Decl.h:4460
field_iterator field_end() const
Definition Decl.h:4666
field_range fields() const
Definition Decl.h:4663
virtual void completeDefinition()
Note that the definition of this type is now complete.
Definition Decl.cpp:5362
bool field_empty() const
Definition Decl.h:4671
field_iterator field_begin() const
Definition Decl.cpp:5346
Scope - A scope is a transient data structure that is used while parsing the program.
Definition Scope.h:41
Encodes a location in the source.
static SourceLocation getFromRawEncoding(UIntTy Encoding)
Turn a raw encoding of a SourceLocation object into a real SourceLocation.
bool isValid() const
Return true if this is a valid SourceLocation object.
UIntTy getRawEncoding() const
When a SourceLocation itself cannot be used, this returns an (opaque) 32-bit integer encoding for it.
This class handles loading and caching of source files into memory.
PresumedLoc getPresumedLoc(SourceLocation Loc, bool UseLineDirectives=true) const
Returns the "presumed" location of a SourceLocation specifies.
Stmt - This represents one statement.
Definition Stmt.h:85
child_range children()
Definition Stmt.cpp:304
StmtClass getStmtClass() const
Definition Stmt.h:1505
SourceRange getSourceRange() const LLVM_READONLY
SourceLocation tokens are not useful in isolation - they are low level value objects created/interpre...
Definition Stmt.cpp:343
Stmt * IgnoreContainers(bool IgnoreCaptured=false)
Skip no-op (attributed, compound) container stmts and skip captured stmt at the top,...
Definition Stmt.cpp:210
SourceLocation getBeginLoc() const LLVM_READONLY
Definition Stmt.cpp:355
void startDefinition()
Starts the definition of this tag declaration.
Definition Decl.cpp:4977
bool isUnion() const
Definition Decl.h:4063
The base class of the type hierarchy.
Definition TypeBase.h:1879
bool isVoidType() const
Definition TypeBase.h:9068
const Type * getPointeeOrArrayElementType() const
If this is a pointer type, return the pointee type.
Definition TypeBase.h:9259
bool isSignedIntegerType() const
Return true if this is an integer type that is signed, according to C99 6.2.5p4 [char,...
Definition Type.cpp:2392
CXXRecordDecl * getAsCXXRecordDecl() const
Retrieves the CXXRecordDecl that this type refers to, either because the type is a RecordType or beca...
Definition Type.h:26
RecordDecl * getAsRecordDecl() const
Retrieves the RecordDecl this type refers to.
Definition Type.h:41
bool isArrayType() const
Definition TypeBase.h:8782
bool isPointerType() const
Definition TypeBase.h:8683
CanQualType getCanonicalTypeUnqualified() const
bool isIntegerType() const
isIntegerType() does not include complex integers (a GCC extension).
Definition TypeBase.h:9116
const T * castAs() const
Member-template castAs<specific type>.
Definition TypeBase.h:9366
bool isReferenceType() const
Definition TypeBase.h:8707
QualType getPointeeType() const
If this is a pointer, ObjC object pointer, or block pointer, this returns the respective pointee.
Definition Type.cpp:885
bool isLValueReferenceType() const
Definition TypeBase.h:8711
bool isAggregateType() const
Determines whether the type is a C++ aggregate type or C aggregate or union type.
Definition Type.cpp:2631
RecordDecl * castAsRecordDecl() const
Definition Type.h:48
QualType getCanonicalTypeInternal() const
Definition TypeBase.h:3200
const Type * getBaseElementTypeUnsafe() const
Get the base element type of this type, potentially discarding type qualifiers.
Definition TypeBase.h:9252
bool isVariablyModifiedType() const
Whether this type is a variably-modified type (C99 6.7.5).
Definition TypeBase.h:2881
const ArrayType * getAsArrayTypeUnsafe() const
A variant of getAs<> for array types which silently discards qualifiers from the outermost type.
Definition TypeBase.h:9352
bool isFloatingType() const
Definition Type.cpp:2517
bool isUnsignedIntegerType() const
Return true if this is an integer type that is unsigned, according to C99 6.2.5p6 [which returns true...
Definition Type.cpp:2460
bool isAnyPointerType() const
Definition TypeBase.h:8691
const T * getAs() const
Member-template getAs<specific type>'.
Definition TypeBase.h:9299
bool isRecordType() const
Definition TypeBase.h:8810
bool isUnionType() const
Definition Type.cpp:851
Represent the declaration of a variable (in which case it is an lvalue) a function (in which case it ...
Definition Decl.h:713
QualType getType() const
Definition Decl.h:724
Represents a variable declaration or definition.
Definition Decl.h:933
VarDecl * getCanonicalDecl() override
Retrieves the "canonical" declaration of the given declaration.
Definition Decl.cpp:2237
VarDecl * getDefinition(ASTContext &)
Get the real (not just tentative) definition for this declaration.
Definition Decl.cpp:2346
const Expr * getInit() const
Definition Decl.h:1392
bool hasExternalStorage() const
Returns true if a variable has extern or private_extern storage.
Definition Decl.h:1239
@ DeclarationOnly
This declaration is only a declaration.
Definition Decl.h:1319
DefinitionKind hasDefinition(ASTContext &) const
Check whether this variable is defined in this translation unit.
Definition Decl.cpp:2355
bool isLocalVarDeclOrParm() const
Similar to isLocalVarDecl but also includes parameters.
Definition Decl.h:1286
const Expr * getAnyInitializer() const
Get the initializer for this variable, no matter which declaration it is attached to.
Definition Decl.h:1382
Represents a C array with a specified size that is not an integer-constant-expression.
Definition TypeBase.h:4064
Expr * getSizeExpr() const
Definition TypeBase.h:4078
specific_attr_iterator - Iterates over a subrange of an AttrVec, only providing attributes that are o...
Definition SPIR.cpp:35
bool isEmptyRecordForLayout(const ASTContext &Context, QualType T)
isEmptyRecordForLayout - Return true iff a structure contains only empty base classes (per isEmptyRec...
@ Type
The l-value was considered opaque, so the alignment was determined from a type.
Definition CGValue.h:155
@ Decl
The l-value was an access to a declared entity or something equivalently strong, like the address of ...
Definition CGValue.h:146
bool isEmptyFieldForLayout(const ASTContext &Context, const FieldDecl *FD)
isEmptyFieldForLayout - Return true iff the field is "empty", that is, either a zero-width bit-field ...
ComparisonResult
Indicates the result of a tentative comparison.
@ Address
A pointer to a ValueDecl.
Definition Primitives.h:28
Top level wrappers for InstallAPI frontend operations.
bool isOpenMPWorksharingDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a worksharing directive.
CanQual< Type > CanQualType
Represents a canonical, potentially-qualified type.
bool needsTaskBasedThreadLimit(OpenMPDirectiveKind DKind)
Checks if the specified target directive, combined or not, needs task based thread_limit.
@ Match
This is not an overload because the signature exactly matches an existing declaration.
Definition Sema.h:824
@ Ctor_Complete
Complete object ctor.
Definition ABI.h:25
Privates[]
This class represents the 'transparent' clause in the 'pragma omp task' directive.
bool isa(CodeGen::Address addr)
Definition Address.h:330
if(T->getSizeExpr()) TRY_TO(TraverseStmt(const_cast< Expr * >(T -> getSizeExpr())))
bool isOpenMPTargetDataManagementDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a target data offload directive.
static bool classof(const OMPClause *T)
Stmt Stmt * Callback
Definition StmtOpenMP.h:919
@ Conditional
A conditional (?:) operator.
Definition Sema.h:663
@ ICIS_NoInit
No in-class initializer.
Definition Specifiers.h:276
bool isOpenMPDistributeDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a distribute directive.
@ LCK_ByRef
Capturing by reference.
Definition Lambda.h:37
LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE()
@ Private
'private' clause, allowed on 'parallel', 'serial', 'loop', 'parallel loop', and 'serial loop' constru...
@ Reduction
'reduction' clause, allowed on Parallel, Serial, Loop, and the combined constructs.
@ Present
'present' clause, allowed on Compute and Combined constructs, plus 'data' and 'declare'.
OpenMPScheduleClauseModifier
OpenMP modifiers for 'schedule' clause.
Definition OpenMPKinds.h:39
@ OMPC_SCHEDULE_MODIFIER_last
Definition OpenMPKinds.h:44
@ OMPC_SCHEDULE_MODIFIER_unknown
Definition OpenMPKinds.h:40
@ AS_public
Definition Specifiers.h:128
nullptr
This class represents a compute construct, representing a 'Kind' of ‘parallel’, 'serial',...
@ CR_OpenMP
bool isOpenMPParallelDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a parallel-kind directive.
OpenMPDistScheduleClauseKind
OpenMP attributes for 'dist_schedule' clause.
bool isOpenMPTaskingDirective(OpenMPDirectiveKind Kind)
Checks if the specified directive kind is one of tasking directives - task, taskloop,...
bool isOpenMPTargetExecutionDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a target code offload directive.
@ OMPC_DYN_GROUPPRIVATE_FALLBACK_unknown
@ Result
The result type of a method or function.
Definition TypeBase.h:906
bool isOpenMPTeamsDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a teams-kind directive.
const FunctionProtoType * T
OpenMPDependClauseKind
OpenMP attributes for 'depend' clause.
Definition OpenMPKinds.h:55
@ OMPC_DEPEND_unknown
Definition OpenMPKinds.h:59
@ Dtor_Complete
Complete object dtor.
Definition ABI.h:36
@ Union
The "union" keyword.
Definition TypeBase.h:6047
bool isOpenMPTargetMapEnteringDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a map-entering target directive.
bool isOpenMPLoopDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a directive with an associated loop construct.
OpenMPSeverityClauseKind
OpenMP attributes for 'severity' clause.
LangAS
Defines the address space values used by the address space qualifier of QualType.
llvm::omp::Directive OpenMPDirectiveKind
OpenMP directives.
Definition OpenMPKinds.h:25
bool isOpenMPSimdDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a simd directive.
@ VK_PRValue
A pr-value expression (in the C++11 taxonomy) produces a temporary value.
Definition Specifiers.h:139
@ VK_LValue
An l-value expression is a reference to an object with independent storage.
Definition Specifiers.h:143
for(const auto &A :T->param_types())
void getOpenMPCaptureRegions(llvm::SmallVectorImpl< OpenMPDirectiveKind > &CaptureRegions, OpenMPDirectiveKind DKind)
Return the captured regions of an OpenMP directive.
OpenMPNumThreadsClauseModifier
@ OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown
U cast(CodeGen::Address addr)
Definition Address.h:327
@ OMPC_DEVICE_unknown
Definition OpenMPKinds.h:51
OpenMPMapModifierKind
OpenMP modifier kind for 'map' clause.
Definition OpenMPKinds.h:79
@ OMPC_MAP_MODIFIER_unknown
Definition OpenMPKinds.h:80
@ Other
Other implicit parameter.
Definition Decl.h:1775
OpenMPScheduleClauseKind
OpenMP attributes for 'schedule' clause.
Definition OpenMPKinds.h:31
@ OMPC_SCHEDULE_unknown
Definition OpenMPKinds.h:35
bool isOpenMPTaskLoopDirective(OpenMPDirectiveKind DKind)
Checks if the specified directive is a taskloop directive.
OpenMPThreadsetKind
OpenMP modifiers for 'threadset' clause.
OpenMPMapClauseKind
OpenMP mapping kind for 'map' clause.
Definition OpenMPKinds.h:71
@ OMPC_MAP_unknown
Definition OpenMPKinds.h:75
unsigned long uint64_t
Diagnostic wrappers for TextAPI types for error reporting.
Definition Dominators.h:30
int32_t uint32_t
#define false
Definition stdbool.h:26
Data for list of allocators.
Expr * AllocatorTraits
Allocator traits.
struct with the values to be passed to the dispatch runtime function
llvm::Value * Chunk
Chunk size specified using 'schedule' clause (nullptr if chunk was not specified)
Maps the expression for the lastprivate variable to the global copy used to store new value because o...
Struct with the values to be passed to the static runtime function.
bool IVSigned
Sign of the iteration variable.
Address UB
Address of the output variable in which the upper iteration number is returned.
Address IL
Address of the output variable in which the flag of the last iteration is returned.
llvm::Value * Chunk
Value of the chunk for the static_chunked scheduled loop.
unsigned IVSize
Size of the iteration variable in bits.
Address ST
Address of the output variable in which the stride value is returned necessary to generated the stati...
bool Ordered
true if loop is ordered, false otherwise.
Address LB
Address of the output variable in which the lower iteration number is returned.
A jump destination is an abstract label, branching to which may require a jump out through normal cle...
llvm::IntegerType * Int8Ty
i8, i16, i32, and i64
llvm::CallingConv::ID getRuntimeCC() const
SmallVector< const Expr *, 4 > DepExprs
EvalResult is a struct with detailed info about an evaluated expression.
Definition Expr.h:666
Extra information about a function prototype.
Definition TypeBase.h:5501
Expr * CounterUpdate
Updater for the internal counter: ++CounterVD;.
Definition ExprOpenMP.h:121
Scheduling data for loop-based OpenMP directives.
bool UseFusedDistChunkSchedule
Request the fused distr_static_chunk + static_chunkone runtime schedule in for_static_init.
OpenMPScheduleClauseModifier M2
OpenMPScheduleClauseModifier M1
OpenMPScheduleClauseKind Schedule
Describes how types, statements, expressions, and declarations should be printed.