1//===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This provides a class for OpenMP runtime code generation.
10//
11//===----------------------------------------------------------------------===//
12
13#include "CGOpenMPRuntime.h"
14#include "ABIInfoImpl.h"
15#include "CGCXXABI.h"
16#include "CGCleanup.h"
17#include "CGDebugInfo.h"
18#include "CGRecordLayout.h"
19#include "CodeGenFunction.h"
20#include "TargetInfo.h"
21#include "clang/AST/APValue.h"
22#include "clang/AST/Attr.h"
23#include "clang/AST/Decl.h"
24#include "clang/AST/OpenMPClause.h"
25#include "clang/AST/StmtOpenMP.h"
26#include "clang/AST/StmtVisitor.h"
27#include "clang/Basic/DiagnosticFrontend.h"
28#include "clang/Basic/OpenMPKinds.h"
29#include "clang/Basic/SourceManager.h"
30#include "clang/CodeGen/ConstantInitBuilder.h"
31#include "clang/CodeGenUtils/RecordLayoutUtils.h"
32#include "llvm/ADT/ArrayRef.h"
33#include "llvm/ADT/SmallSet.h"
34#include "llvm/ADT/SmallVector.h"
35#include "llvm/ADT/StringExtras.h"
36#include "llvm/Bitcode/BitcodeReader.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DerivedTypes.h"
39#include "llvm/IR/GlobalValue.h"
40#include "llvm/IR/InstrTypes.h"
41#include "llvm/IR/Value.h"
42#include "llvm/Support/AtomicOrdering.h"
43#include "llvm/Support/VirtualFileSystem.h"
44#include "llvm/Support/raw_ostream.h"
45#include <cassert>
46#include <cstdint>
47#include <numeric>
48#include <optional>
49
50using namespace clang;
51using namespace CodeGen;
52using namespace llvm::omp;
53
54namespace {
55/// Base class for handling code generation inside OpenMP regions.
56class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
57public:
58 /// Kinds of OpenMP regions used in codegen.
59 enum CGOpenMPRegionKind {
60 /// Region with outlined function for standalone 'parallel'
61 /// directive.
62 ParallelOutlinedRegion,
63 /// Region with outlined function for standalone 'task' directive.
64 TaskOutlinedRegion,
65 /// Region for constructs that do not require function outlining,
66 /// like 'for', 'sections', 'atomic' etc. directives.
67 InlinedRegion,
68 /// Region with outlined function for standalone 'target' directive.
69 TargetRegion,
70 };
71
72 CGOpenMPRegionInfo(const CapturedStmt &CS,
73 const CGOpenMPRegionKind RegionKind,
74 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
75 bool HasCancel)
76 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
77 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {}
78
79 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
80 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
81 bool HasCancel)
82 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen),
83 Kind(Kind), HasCancel(HasCancel) {}
84
85 /// Get a variable or parameter for storing global thread id
86 /// inside OpenMP construct.
87 virtual const VarDecl *getThreadIDVariable() const = 0;
88
89 /// Emit the captured statement body.
90 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
91
92 /// Get an LValue for the current ThreadID variable.
93 /// \return LValue for thread id variable. This LValue always has type int32*.
94 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
95
96 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {}
97
98 CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
99
100 OpenMPDirectiveKind getDirectiveKind() const { return Kind; }
101
102 bool hasCancel() const { return HasCancel; }
103
104 static bool classof(const CGCapturedStmtInfo *Info) {
105 return Info->getKind() == CR_OpenMP;
106 }
107
108 ~CGOpenMPRegionInfo() override = default;
109
110protected:
111 CGOpenMPRegionKind RegionKind;
112 RegionCodeGenTy CodeGen;
113 OpenMPDirectiveKind Kind;
114 bool HasCancel;
115};
116
117/// API for captured statement code generation in OpenMP constructs.
118class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo {
119public:
120 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
121 const RegionCodeGenTy &CodeGen,
122 OpenMPDirectiveKind Kind, bool HasCancel,
123 StringRef HelperName)
124 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind,
125 HasCancel),
126 ThreadIDVar(ThreadIDVar), HelperName(HelperName) {
127 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
128 }
129
130 /// Get a variable or parameter for storing global thread id
131 /// inside OpenMP construct.
132 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
133
134 /// Get the name of the capture helper.
135 StringRef getHelperName() const override { return HelperName; }
136
137 static bool classof(const CGCapturedStmtInfo *Info) {
138 return CGOpenMPRegionInfo::classof(Info) &&
139 cast<CGOpenMPRegionInfo>(Val: Info)->getRegionKind() ==
140 ParallelOutlinedRegion;
141 }
142
143private:
144 /// A variable or parameter storing global thread id for OpenMP
145 /// constructs.
146 const VarDecl *ThreadIDVar;
147 StringRef HelperName;
148};
149
150/// API for captured statement code generation in OpenMP constructs.
151class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo {
152public:
153 class UntiedTaskActionTy final : public PrePostActionTy {
154 bool Untied;
155 const VarDecl *PartIDVar;
156 const RegionCodeGenTy UntiedCodeGen;
157 llvm::SwitchInst *UntiedSwitch = nullptr;
158
159 public:
160 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar,
161 const RegionCodeGenTy &UntiedCodeGen)
162 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {}
163 void Enter(CodeGenFunction &CGF) override {
164 if (Untied) {
165 // Emit task switching point.
166 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
167 Ptr: CGF.GetAddrOfLocalVar(VD: PartIDVar),
168 PtrTy: PartIDVar->getType()->castAs<PointerType>());
169 llvm::Value *Res =
170 CGF.EmitLoadOfScalar(lvalue: PartIdLVal, Loc: PartIDVar->getLocation());
171 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(name: ".untied.done.");
172 UntiedSwitch = CGF.Builder.CreateSwitch(V: Res, Dest: DoneBB);
173 CGF.EmitBlock(BB: DoneBB);
174 CGF.EmitBranchThroughCleanup(Dest: CGF.ReturnBlock);
175 CGF.EmitBlock(BB: CGF.createBasicBlock(name: ".untied.jmp."));
176 UntiedSwitch->addCase(OnVal: CGF.Builder.getInt32(C: 0),
177 Dest: CGF.Builder.GetInsertBlock());
178 emitUntiedSwitch(CGF);
179 }
180 }
181 void emitUntiedSwitch(CodeGenFunction &CGF) const {
182 if (Untied) {
183 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
184 Ptr: CGF.GetAddrOfLocalVar(VD: PartIDVar),
185 PtrTy: PartIDVar->getType()->castAs<PointerType>());
186 CGF.EmitStoreOfScalar(value: CGF.Builder.getInt32(C: UntiedSwitch->getNumCases()),
187 lvalue: PartIdLVal);
188 UntiedCodeGen(CGF);
189 CodeGenFunction::JumpDest CurPoint =
190 CGF.getJumpDestInCurrentScope(Name: ".untied.next.");
191 CGF.EmitBranch(Block: CGF.ReturnBlock.getBlock());
192 CGF.EmitBlock(BB: CGF.createBasicBlock(name: ".untied.jmp."));
193 UntiedSwitch->addCase(OnVal: CGF.Builder.getInt32(C: UntiedSwitch->getNumCases()),
194 Dest: CGF.Builder.GetInsertBlock());
195 CGF.EmitBranchThroughCleanup(Dest: CurPoint);
196 CGF.EmitBlock(BB: CurPoint.getBlock());
197 }
198 }
199 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); }
200 };
201 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
202 const VarDecl *ThreadIDVar,
203 const RegionCodeGenTy &CodeGen,
204 OpenMPDirectiveKind Kind, bool HasCancel,
205 const UntiedTaskActionTy &Action)
206 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel),
207 ThreadIDVar(ThreadIDVar), Action(Action) {
208 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
209 }
210
211 /// Get a variable or parameter for storing global thread id
212 /// inside OpenMP construct.
213 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
214
215 /// Get an LValue for the current ThreadID variable.
216 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
217
218 /// Get the name of the capture helper.
219 StringRef getHelperName() const override { return ".omp_outlined."; }
220
221 void emitUntiedSwitch(CodeGenFunction &CGF) override {
222 Action.emitUntiedSwitch(CGF);
223 }
224
225 static bool classof(const CGCapturedStmtInfo *Info) {
226 return CGOpenMPRegionInfo::classof(Info) &&
227 cast<CGOpenMPRegionInfo>(Val: Info)->getRegionKind() ==
228 TaskOutlinedRegion;
229 }
230
231private:
232 /// A variable or parameter storing global thread id for OpenMP
233 /// constructs.
234 const VarDecl *ThreadIDVar;
235 /// Action for emitting code for untied tasks.
236 const UntiedTaskActionTy &Action;
237};
238
239/// API for inlined captured statement code generation in OpenMP
240/// constructs.
241class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
242public:
243 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
244 const RegionCodeGenTy &CodeGen,
245 OpenMPDirectiveKind Kind, bool HasCancel)
246 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel),
247 OldCSI(OldCSI),
248 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(Val: OldCSI)) {}
249
250 // Retrieve the value of the context parameter.
251 llvm::Value *getContextValue() const override {
252 if (OuterRegionInfo)
253 return OuterRegionInfo->getContextValue();
254 llvm_unreachable("No context value for inlined OpenMP region");
255 }
256
257 void setContextValue(llvm::Value *V) override {
258 if (OuterRegionInfo) {
259 OuterRegionInfo->setContextValue(V);
260 return;
261 }
262 llvm_unreachable("No context value for inlined OpenMP region");
263 }
264
265 /// Lookup the captured field decl for a variable.
266 const FieldDecl *lookup(const VarDecl *VD) const override {
267 if (OuterRegionInfo)
268 return OuterRegionInfo->lookup(VD);
269 // If there is no outer outlined region,no need to lookup in a list of
270 // captured variables, we can use the original one.
271 return nullptr;
272 }
273
274 FieldDecl *getThisFieldDecl() const override {
275 if (OuterRegionInfo)
276 return OuterRegionInfo->getThisFieldDecl();
277 return nullptr;
278 }
279
280 /// Get a variable or parameter for storing global thread id
281 /// inside OpenMP construct.
282 const VarDecl *getThreadIDVariable() const override {
283 if (OuterRegionInfo)
284 return OuterRegionInfo->getThreadIDVariable();
285 return nullptr;
286 }
287
288 /// Get an LValue for the current ThreadID variable.
289 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override {
290 if (OuterRegionInfo)
291 return OuterRegionInfo->getThreadIDVariableLValue(CGF);
292 llvm_unreachable("No LValue for inlined OpenMP construct");
293 }
294
295 /// Get the name of the capture helper.
296 StringRef getHelperName() const override {
297 if (auto *OuterRegionInfo = getOldCSI())
298 return OuterRegionInfo->getHelperName();
299 llvm_unreachable("No helper name for inlined OpenMP construct");
300 }
301
302 void emitUntiedSwitch(CodeGenFunction &CGF) override {
303 if (OuterRegionInfo)
304 OuterRegionInfo->emitUntiedSwitch(CGF);
305 }
306
307 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
308
309 static bool classof(const CGCapturedStmtInfo *Info) {
310 return CGOpenMPRegionInfo::classof(Info) &&
311 cast<CGOpenMPRegionInfo>(Val: Info)->getRegionKind() == InlinedRegion;
312 }
313
314 ~CGOpenMPInlinedRegionInfo() override = default;
315
316private:
317 /// CodeGen info about outer OpenMP region.
318 CodeGenFunction::CGCapturedStmtInfo *OldCSI;
319 CGOpenMPRegionInfo *OuterRegionInfo;
320};
321
322/// API for captured statement code generation in OpenMP target
323/// constructs. For this captures, implicit parameters are used instead of the
324/// captured fields. The name of the target region has to be unique in a given
325/// application so it is provided by the client, because only the client has
326/// the information to generate that.
327class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo {
328public:
329 CGOpenMPTargetRegionInfo(const CapturedStmt &CS,
330 const RegionCodeGenTy &CodeGen, StringRef HelperName)
331 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target,
332 /*HasCancel=*/false),
333 HelperName(HelperName) {}
334
335 /// This is unused for target regions because each starts executing
336 /// with a single thread.
337 const VarDecl *getThreadIDVariable() const override { return nullptr; }
338
339 /// Get the name of the capture helper.
340 StringRef getHelperName() const override { return HelperName; }
341
342 static bool classof(const CGCapturedStmtInfo *Info) {
343 return CGOpenMPRegionInfo::classof(Info) &&
344 cast<CGOpenMPRegionInfo>(Val: Info)->getRegionKind() == TargetRegion;
345 }
346
347private:
348 StringRef HelperName;
349};
350
351static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) {
352 llvm_unreachable("No codegen for expressions");
353}
354/// API for generation of expressions captured in a innermost OpenMP
355/// region.
356class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo {
357public:
358 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS)
359 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen,
360 OMPD_unknown,
361 /*HasCancel=*/false),
362 PrivScope(CGF) {
363 // Make sure the globals captured in the provided statement are local by
364 // using the privatization logic. We assume the same variable is not
365 // captured more than once.
366 for (const auto &C : CS.captures()) {
367 if (!C.capturesVariable() && !C.capturesVariableByCopy())
368 continue;
369
370 const VarDecl *VD = C.getCapturedVar();
371 if (VD->isLocalVarDeclOrParm())
372 continue;
373
374 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
375 /*RefersToEnclosingVariableOrCapture=*/false,
376 VD->getType().getNonReferenceType(), VK_LValue,
377 C.getLocation());
378 PrivScope.addPrivate(LocalVD: VD, Addr: CGF.EmitLValue(E: &DRE).getAddress());
379 }
380 (void)PrivScope.Privatize();
381 }
382
383 /// Lookup the captured field decl for a variable.
384 const FieldDecl *lookup(const VarDecl *VD) const override {
385 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD))
386 return FD;
387 return nullptr;
388 }
389
390 /// Emit the captured statement body.
391 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override {
392 llvm_unreachable("No body for expressions");
393 }
394
395 /// Get a variable or parameter for storing global thread id
396 /// inside OpenMP construct.
397 const VarDecl *getThreadIDVariable() const override {
398 llvm_unreachable("No thread id for expressions");
399 }
400
401 /// Get the name of the capture helper.
402 StringRef getHelperName() const override {
403 llvm_unreachable("No helper name for expressions");
404 }
405
406 static bool classof(const CGCapturedStmtInfo *Info) { return false; }
407
408private:
409 /// Private scope to capture global variables.
410 CodeGenFunction::OMPPrivateScope PrivScope;
411};
412
413/// RAII for emitting code of OpenMP constructs.
414class InlinedOpenMPRegionRAII {
415 CodeGenFunction &CGF;
416 llvm::DenseMap<const ValueDecl *, FieldDecl *> LambdaCaptureFields;
417 FieldDecl *LambdaThisCaptureField = nullptr;
418 const CodeGen::CGBlockInfo *BlockInfo = nullptr;
419 bool NoInheritance = false;
420
421public:
422 /// Constructs region for combined constructs.
423 /// \param CodeGen Code generation sequence for combined directives. Includes
424 /// a list of functions used for code generation of implicitly inlined
425 /// regions.
426 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen,
427 OpenMPDirectiveKind Kind, bool HasCancel,
428 bool NoInheritance = true)
429 : CGF(CGF), NoInheritance(NoInheritance) {
430 // Start emission for the construct.
431 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(
432 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel);
433 if (NoInheritance) {
434 std::swap(a&: CGF.LambdaCaptureFields, b&: LambdaCaptureFields);
435 LambdaThisCaptureField = CGF.LambdaThisCaptureField;
436 CGF.LambdaThisCaptureField = nullptr;
437 BlockInfo = CGF.BlockInfo;
438 CGF.BlockInfo = nullptr;
439 }
440 }
441
442 ~InlinedOpenMPRegionRAII() {
443 // Restore original CapturedStmtInfo only if we're done with code emission.
444 auto *OldCSI =
445 cast<CGOpenMPInlinedRegionInfo>(Val: CGF.CapturedStmtInfo)->getOldCSI();
446 delete CGF.CapturedStmtInfo;
447 CGF.CapturedStmtInfo = OldCSI;
448 if (NoInheritance) {
449 std::swap(a&: CGF.LambdaCaptureFields, b&: LambdaCaptureFields);
450 CGF.LambdaThisCaptureField = LambdaThisCaptureField;
451 CGF.BlockInfo = BlockInfo;
452 }
453 }
454};
455
456/// Values for bit flags used in the ident_t to describe the fields.
457/// All enumeric elements are named and described in accordance with the code
458/// from https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h
459enum OpenMPLocationFlags : unsigned {
460 /// Use trampoline for internal microtask.
461 OMP_IDENT_IMD = 0x01,
462 /// Use c-style ident structure.
463 OMP_IDENT_KMPC = 0x02,
464 /// Atomic reduction option for kmpc_reduce.
465 OMP_ATOMIC_REDUCE = 0x10,
466 /// Explicit 'barrier' directive.
467 OMP_IDENT_BARRIER_EXPL = 0x20,
468 /// Implicit barrier in code.
469 OMP_IDENT_BARRIER_IMPL = 0x40,
470 /// Implicit barrier in 'for' directive.
471 OMP_IDENT_BARRIER_IMPL_FOR = 0x40,
472 /// Implicit barrier in 'sections' directive.
473 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0,
474 /// Implicit barrier in 'single' directive.
475 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140,
476 /// Call of __kmp_for_static_init for static loop.
477 OMP_IDENT_WORK_LOOP = 0x200,
478 /// Call of __kmp_for_static_init for sections.
479 OMP_IDENT_WORK_SECTIONS = 0x400,
480 /// Call of __kmp_for_static_init for distribute.
481 OMP_IDENT_WORK_DISTRIBUTE = 0x800,
482 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE)
483};
484
485/// Describes ident structure that describes a source location.
486/// All descriptions are taken from
487/// https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h
488/// Original structure:
489/// typedef struct ident {
490/// kmp_int32 reserved_1; /**< might be used in Fortran;
491/// see above */
492/// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags;
493/// KMP_IDENT_KMPC identifies this union
494/// member */
495/// kmp_int32 reserved_2; /**< not really used in Fortran any more;
496/// see above */
497///#if USE_ITT_BUILD
498/// /* but currently used for storing
499/// region-specific ITT */
500/// /* contextual information. */
501///#endif /* USE_ITT_BUILD */
502/// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for
503/// C++ */
504/// char const *psource; /**< String describing the source location.
505/// The string is composed of semi-colon separated
506// fields which describe the source file,
507/// the function and a pair of line numbers that
508/// delimit the construct.
509/// */
510/// } ident_t;
511enum IdentFieldIndex {
512 /// might be used in Fortran
513 IdentField_Reserved_1,
514 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member.
515 IdentField_Flags,
516 /// Not really used in Fortran any more
517 IdentField_Reserved_2,
518 /// Source[4] in Fortran, do not use for C++
519 IdentField_Reserved_3,
520 /// String describing the source location. The string is composed of
521 /// semi-colon separated fields which describe the source file, the function
522 /// and a pair of line numbers that delimit the construct.
523 IdentField_PSource
524};
525
526/// Schedule types for 'omp for' loops (these enumerators are taken from
527/// the enum sched_type in kmp.h).
528enum OpenMPSchedType {
529 /// Lower bound for default (unordered) versions.
530 OMP_sch_lower = 32,
531 OMP_sch_static_chunked = 33,
532 OMP_sch_static = 34,
533 OMP_sch_dynamic_chunked = 35,
534 OMP_sch_guided_chunked = 36,
535 OMP_sch_runtime = 37,
536 OMP_sch_auto = 38,
537 /// static with chunk adjustment (e.g., simd)
538 OMP_sch_static_balanced_chunked = 45,
539 /// Lower bound for 'ordered' versions.
540 OMP_ord_lower = 64,
541 OMP_ord_static_chunked = 65,
542 OMP_ord_static = 66,
543 OMP_ord_dynamic_chunked = 67,
544 OMP_ord_guided_chunked = 68,
545 OMP_ord_runtime = 69,
546 OMP_ord_auto = 70,
547 OMP_sch_default = OMP_sch_static,
548 /// dist_schedule types
549 OMP_dist_sch_static_chunked = 91,
550 OMP_dist_sch_static = 92,
551 /// Fused distribute+for static schedule (entityId = team*nthreads + tid,
552 /// num_entities = nteams*nthreads). One for_static_init call, no
553 /// surrounding distribute_static_init. Matches
554 /// kmp_sched_distr_static_chunk_sched_static_chunkone in the device RTL
555 /// (openmp/device/include/DeviceTypes.h).
556 OMP_dist_sch_static_chunked_sch_static_chunkone = 93,
557 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers.
558 /// Set if the monotonic schedule modifier was present.
559 OMP_sch_modifier_monotonic = (1 << 29),
560 /// Set if the nonmonotonic schedule modifier was present.
561 OMP_sch_modifier_nonmonotonic = (1 << 30),
562};
563
564/// A basic class for pre|post-action for advanced codegen sequence for OpenMP
565/// region.
566class CleanupTy final : public EHScopeStack::Cleanup {
567 PrePostActionTy *Action;
568
569public:
570 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {}
571 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
572 if (!CGF.HaveInsertPoint())
573 return;
574 Action->Exit(CGF);
575 }
576};
577
578} // anonymous namespace
579
580void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const {
581 CodeGenFunction::RunCleanupsScope Scope(CGF);
582 if (PrePostAction) {
583 CGF.EHStack.pushCleanup<CleanupTy>(Kind: NormalAndEHCleanup, A: PrePostAction);
584 Callback(CodeGen, CGF, *PrePostAction);
585 } else {
586 PrePostActionTy Action;
587 Callback(CodeGen, CGF, Action);
588 }
589}
590
591/// Check if the combiner is a call to UDR combiner and if it is so return the
592/// UDR decl used for reduction.
593static const OMPDeclareReductionDecl *
594getReductionInit(const Expr *ReductionOp) {
595 if (const auto *CE = dyn_cast<CallExpr>(Val: ReductionOp))
596 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(Val: CE->getCallee()))
597 if (const auto *DRE =
598 dyn_cast<DeclRefExpr>(Val: OVE->getSourceExpr()->IgnoreImpCasts()))
599 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(Val: DRE->getDecl()))
600 return DRD;
601 return nullptr;
602}
603
604static void emitInitWithReductionInitializer(CodeGenFunction &CGF,
605 const OMPDeclareReductionDecl *DRD,
606 const Expr *InitOp,
607 Address Private, Address Original,
608 QualType Ty) {
609 if (DRD->getInitializer()) {
610 std::pair<llvm::Function *, llvm::Function *> Reduction =
611 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(D: DRD);
612 const auto *CE = cast<CallExpr>(Val: InitOp);
613 const auto *OVE = cast<OpaqueValueExpr>(Val: CE->getCallee());
614 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts();
615 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts();
616 const auto *LHSDRE =
617 cast<DeclRefExpr>(Val: cast<UnaryOperator>(Val: LHS)->getSubExpr());
618 const auto *RHSDRE =
619 cast<DeclRefExpr>(Val: cast<UnaryOperator>(Val: RHS)->getSubExpr());
620 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
621 PrivateScope.addPrivate(LocalVD: cast<VarDecl>(Val: LHSDRE->getDecl()), Addr: Private);
622 PrivateScope.addPrivate(LocalVD: cast<VarDecl>(Val: RHSDRE->getDecl()), Addr: Original);
623 (void)PrivateScope.Privatize();
624 RValue Func = RValue::get(V: Reduction.second);
625 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
626 CGF.EmitIgnoredExpr(E: InitOp);
627 } else {
628 llvm::Constant *Init = CGF.CGM.EmitNullConstant(T: Ty);
629 std::string Name = CGF.CGM.getOpenMPRuntime().getName(Parts: {"init"});
630 auto *GV = new llvm::GlobalVariable(
631 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true,
632 llvm::GlobalValue::PrivateLinkage, Init, Name);
633 LValue LV = CGF.MakeNaturalAlignRawAddrLValue(V: GV, T: Ty);
634 RValue InitRVal;
635 switch (CGF.getEvaluationKind(T: Ty)) {
636 case TEK_Scalar:
637 InitRVal = CGF.EmitLoadOfLValue(V: LV, Loc: DRD->getLocation());
638 break;
639 case TEK_Complex:
640 InitRVal =
641 RValue::getComplex(C: CGF.EmitLoadOfComplex(src: LV, loc: DRD->getLocation()));
642 break;
643 case TEK_Aggregate: {
644 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_LValue);
645 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, LV);
646 CGF.EmitAnyExprToMem(E: &OVE, Location: Private, Quals: Ty.getQualifiers(),
647 /*IsInitializer=*/false);
648 return;
649 }
650 }
651 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_PRValue);
652 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal);
653 CGF.EmitAnyExprToMem(E: &OVE, Location: Private, Quals: Ty.getQualifiers(),
654 /*IsInitializer=*/false);
655 }
656}
657
658/// Emit initialization of arrays of complex types.
659/// \param DestAddr Address of the array.
660/// \param Type Type of array.
661/// \param Init Initial expression of array.
662/// \param SrcAddr Address of the original array.
663static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr,
664 QualType Type, bool EmitDeclareReductionInit,
665 const Expr *Init,
666 const OMPDeclareReductionDecl *DRD,
667 Address SrcAddr = Address::invalid()) {
668 // Perform element-by-element initialization.
669 QualType ElementTy;
670
671 // Drill down to the base element type on both arrays.
672 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
673 llvm::Value *NumElements = CGF.emitArrayLength(arrayType: ArrayTy, baseType&: ElementTy, addr&: DestAddr);
674 if (DRD)
675 SrcAddr = SrcAddr.withElementType(ElemTy: DestAddr.getElementType());
676
677 llvm::Value *SrcBegin = nullptr;
678 if (DRD)
679 SrcBegin = SrcAddr.emitRawPointer(CGF);
680 llvm::Value *DestBegin = DestAddr.emitRawPointer(CGF);
681 // Cast from pointer to array type to pointer to single element.
682 llvm::Value *DestEnd =
683 CGF.Builder.CreateGEP(Ty: DestAddr.getElementType(), Ptr: DestBegin, IdxList: NumElements);
684 // The basic structure here is a while-do loop.
685 llvm::BasicBlock *BodyBB = CGF.createBasicBlock(name: "omp.arrayinit.body");
686 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(name: "omp.arrayinit.done");
687 llvm::Value *IsEmpty =
688 CGF.Builder.CreateICmpEQ(LHS: DestBegin, RHS: DestEnd, Name: "omp.arrayinit.isempty");
689 CGF.Builder.CreateCondBr(Cond: IsEmpty, True: DoneBB, False: BodyBB);
690
691 // Enter the loop body, making that address the current address.
692 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
693 CGF.EmitBlock(BB: BodyBB);
694
695 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(T: ElementTy);
696
697 llvm::PHINode *SrcElementPHI = nullptr;
698 Address SrcElementCurrent = Address::invalid();
699 if (DRD) {
700 SrcElementPHI = CGF.Builder.CreatePHI(Ty: SrcBegin->getType(), NumReservedValues: 2,
701 Name: "omp.arraycpy.srcElementPast");
702 SrcElementPHI->addIncoming(V: SrcBegin, BB: EntryBB);
703 SrcElementCurrent =
704 Address(SrcElementPHI, SrcAddr.getElementType(),
705 SrcAddr.getAlignment().alignmentOfArrayElement(elementSize: ElementSize));
706 }
707 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI(
708 Ty: DestBegin->getType(), NumReservedValues: 2, Name: "omp.arraycpy.destElementPast");
709 DestElementPHI->addIncoming(V: DestBegin, BB: EntryBB);
710 Address DestElementCurrent =
711 Address(DestElementPHI, DestAddr.getElementType(),
712 DestAddr.getAlignment().alignmentOfArrayElement(elementSize: ElementSize));
713
714 // Emit copy.
715 {
716 CodeGenFunction::RunCleanupsScope InitScope(CGF);
717 if (EmitDeclareReductionInit) {
718 emitInitWithReductionInitializer(CGF, DRD, InitOp: Init, Private: DestElementCurrent,
719 Original: SrcElementCurrent, Ty: ElementTy);
720 } else
721 CGF.EmitAnyExprToMem(E: Init, Location: DestElementCurrent, Quals: ElementTy.getQualifiers(),
722 /*IsInitializer=*/false);
723 }
724
725 if (DRD) {
726 // Shift the address forward by one element.
727 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32(
728 Ty: SrcAddr.getElementType(), Ptr: SrcElementPHI, /*Idx0=*/1,
729 Name: "omp.arraycpy.dest.element");
730 SrcElementPHI->addIncoming(V: SrcElementNext, BB: CGF.Builder.GetInsertBlock());
731 }
732
733 // Shift the address forward by one element.
734 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32(
735 Ty: DestAddr.getElementType(), Ptr: DestElementPHI, /*Idx0=*/1,
736 Name: "omp.arraycpy.dest.element");
737 // Check whether we've reached the end.
738 llvm::Value *Done =
739 CGF.Builder.CreateICmpEQ(LHS: DestElementNext, RHS: DestEnd, Name: "omp.arraycpy.done");
740 CGF.Builder.CreateCondBr(Cond: Done, True: DoneBB, False: BodyBB);
741 DestElementPHI->addIncoming(V: DestElementNext, BB: CGF.Builder.GetInsertBlock());
742
743 // Done.
744 CGF.EmitBlock(BB: DoneBB, /*IsFinished=*/true);
745}
746
747LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) {
748 return CGF.EmitOMPSharedLValue(E);
749}
750
751LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF,
752 const Expr *E) {
753 if (const auto *OASE = dyn_cast<ArraySectionExpr>(Val: E))
754 return CGF.EmitArraySectionExpr(E: OASE, /*IsLowerBound=*/false);
755 return LValue();
756}
757
758void ReductionCodeGen::emitAggregateInitialization(
759 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr,
760 const OMPDeclareReductionDecl *DRD) {
761 // Emit VarDecl with copy init for arrays.
762 // Get the address of the original variable captured in current
763 // captured region.
764 const auto *PrivateVD =
765 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: ClausesData[N].Private)->getDecl());
766 bool EmitDeclareReductionInit =
767 DRD && (DRD->getInitializer() || !PrivateVD->hasInit());
768 EmitOMPAggregateInit(CGF, DestAddr: PrivateAddr, Type: PrivateVD->getType(),
769 EmitDeclareReductionInit,
770 Init: EmitDeclareReductionInit ? ClausesData[N].ReductionOp
771 : PrivateVD->getInit(),
772 DRD, SrcAddr: SharedAddr);
773}
774
775ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds,
776 ArrayRef<const Expr *> Origs,
777 ArrayRef<const Expr *> Privates,
778 ArrayRef<const Expr *> ReductionOps) {
779 ClausesData.reserve(N: Shareds.size());
780 SharedAddresses.reserve(N: Shareds.size());
781 Sizes.reserve(N: Shareds.size());
782 BaseDecls.reserve(N: Shareds.size());
783 const auto *IOrig = Origs.begin();
784 const auto *IPriv = Privates.begin();
785 const auto *IRed = ReductionOps.begin();
786 for (const Expr *Ref : Shareds) {
787 ClausesData.emplace_back(Args&: Ref, Args: *IOrig, Args: *IPriv, Args: *IRed);
788 std::advance(i&: IOrig, n: 1);
789 std::advance(i&: IPriv, n: 1);
790 std::advance(i&: IRed, n: 1);
791 }
792}
793
794void ReductionCodeGen::emitSharedOrigLValue(CodeGenFunction &CGF, unsigned N) {
795 assert(SharedAddresses.size() == N && OrigAddresses.size() == N &&
796 "Number of generated lvalues must be exactly N.");
797 LValue First = emitSharedLValue(CGF, E: ClausesData[N].Shared);
798 LValue Second = emitSharedLValueUB(CGF, E: ClausesData[N].Shared);
799 SharedAddresses.emplace_back(Args&: First, Args&: Second);
800 if (ClausesData[N].Shared == ClausesData[N].Ref) {
801 OrigAddresses.emplace_back(Args&: First, Args&: Second);
802 } else {
803 LValue First = emitSharedLValue(CGF, E: ClausesData[N].Ref);
804 LValue Second = emitSharedLValueUB(CGF, E: ClausesData[N].Ref);
805 OrigAddresses.emplace_back(Args&: First, Args&: Second);
806 }
807}
808
809void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) {
810 QualType PrivateType = getPrivateType(N);
811 bool AsArraySection = isa<ArraySectionExpr>(Val: ClausesData[N].Ref);
812 if (!PrivateType->isVariablyModifiedType()) {
813 Sizes.emplace_back(
814 Args: CGF.getTypeSize(Ty: OrigAddresses[N].first.getType().getNonReferenceType()),
815 Args: nullptr);
816 return;
817 }
818 llvm::Value *Size;
819 llvm::Value *SizeInChars;
820 auto *ElemType = OrigAddresses[N].first.getAddress().getElementType();
821 auto *ElemSizeOf = llvm::ConstantInt::get(
822 Ty: CGF.SizeTy, V: CGF.CGM.getDataLayout().getTypeAllocSize(Ty: ElemType));
823 if (AsArraySection) {
824 SizeInChars =
825 CGF.Builder.CreatePtrDiff(LHS: OrigAddresses[N].second.getPointer(CGF),
826 RHS: OrigAddresses[N].first.getPointer(CGF));
827 SizeInChars = CGF.Builder.CreateNUWAdd(LHS: SizeInChars, RHS: ElemSizeOf);
828 } else {
829 SizeInChars =
830 CGF.getTypeSize(Ty: OrigAddresses[N].first.getType().getNonReferenceType());
831 }
832 Size = ElemSizeOf->isOne()
833 ? SizeInChars
834 : CGF.Builder.CreateExactUDiv(LHS: SizeInChars, RHS: ElemSizeOf);
835 Sizes.emplace_back(Args&: SizeInChars, Args&: Size);
836 CodeGenFunction::OpaqueValueMapping OpaqueMap(
837 CGF,
838 cast<OpaqueValueExpr>(
839 Val: CGF.getContext().getAsVariableArrayType(T: PrivateType)->getSizeExpr()),
840 RValue::get(V: Size));
841 CGF.EmitVariablyModifiedType(Ty: PrivateType);
842}
843
844void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N,
845 llvm::Value *Size) {
846 QualType PrivateType = getPrivateType(N);
847 if (!PrivateType->isVariablyModifiedType()) {
848 assert(!Size && !Sizes[N].second &&
849 "Size should be nullptr for non-variably modified reduction "
850 "items.");
851 return;
852 }
853 CodeGenFunction::OpaqueValueMapping OpaqueMap(
854 CGF,
855 cast<OpaqueValueExpr>(
856 Val: CGF.getContext().getAsVariableArrayType(T: PrivateType)->getSizeExpr()),
857 RValue::get(V: Size));
858 CGF.EmitVariablyModifiedType(Ty: PrivateType);
859}
860
861void ReductionCodeGen::emitInitialization(
862 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr,
863 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) {
864 assert(SharedAddresses.size() > N && "No variable was generated");
865 const auto *PrivateVD =
866 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: ClausesData[N].Private)->getDecl());
867 const OMPDeclareReductionDecl *DRD =
868 getReductionInit(ReductionOp: ClausesData[N].ReductionOp);
869 if (CGF.getContext().getAsArrayType(T: PrivateVD->getType())) {
870 if (DRD && DRD->getInitializer())
871 (void)DefaultInit(CGF);
872 emitAggregateInitialization(CGF, N, PrivateAddr, SharedAddr, DRD);
873 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) {
874 (void)DefaultInit(CGF);
875 QualType SharedType = SharedAddresses[N].first.getType();
876 emitInitWithReductionInitializer(CGF, DRD, InitOp: ClausesData[N].ReductionOp,
877 Private: PrivateAddr, Original: SharedAddr, Ty: SharedType);
878 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() &&
879 !CGF.isTrivialInitializer(Init: PrivateVD->getInit())) {
880 CGF.EmitAnyExprToMem(E: PrivateVD->getInit(), Location: PrivateAddr,
881 Quals: PrivateVD->getType().getQualifiers(),
882 /*IsInitializer=*/false);
883 }
884}
885
886bool ReductionCodeGen::needCleanups(unsigned N) {
887 QualType PrivateType = getPrivateType(N);
888 QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
889 return DTorKind != QualType::DK_none;
890}
891
892void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N,
893 Address PrivateAddr) {
894 QualType PrivateType = getPrivateType(N);
895 QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
896 if (needCleanups(N)) {
897 PrivateAddr =
898 PrivateAddr.withElementType(ElemTy: CGF.ConvertTypeForMem(T: PrivateType));
899 CGF.pushDestroy(dtorKind: DTorKind, addr: PrivateAddr, type: PrivateType);
900 }
901}
902
903static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
904 LValue BaseLV) {
905 BaseTy = BaseTy.getNonReferenceType();
906 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
907 !CGF.getContext().hasSameType(T1: BaseTy, T2: ElTy)) {
908 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) {
909 BaseLV = CGF.EmitLoadOfPointerLValue(Ptr: BaseLV.getAddress(), PtrTy);
910 } else {
911 LValue RefLVal = CGF.MakeAddrLValue(Addr: BaseLV.getAddress(), T: BaseTy);
912 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal);
913 }
914 BaseTy = BaseTy->getPointeeType();
915 }
916 return CGF.MakeAddrLValue(
917 Addr: BaseLV.getAddress().withElementType(ElemTy: CGF.ConvertTypeForMem(T: ElTy)),
918 T: BaseLV.getType(), BaseInfo: BaseLV.getBaseInfo(),
919 TBAAInfo: CGF.CGM.getTBAAInfoForSubobject(Base: BaseLV, AccessType: BaseLV.getType()));
920}
921
922static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
923 Address OriginalBaseAddress, llvm::Value *Addr) {
924 RawAddress Tmp = RawAddress::invalid();
925 Address TopTmp = Address::invalid();
926 Address MostTopTmp = Address::invalid();
927 BaseTy = BaseTy.getNonReferenceType();
928 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
929 !CGF.getContext().hasSameType(T1: BaseTy, T2: ElTy)) {
930 Tmp = CGF.CreateMemTempWithoutCast(T: BaseTy);
931 if (TopTmp.isValid())
932 CGF.Builder.CreateStore(Val: Tmp.getPointer(), Addr: TopTmp);
933 else
934 MostTopTmp = Tmp;
935 TopTmp = Tmp;
936 BaseTy = BaseTy->getPointeeType();
937 }
938
939 if (Tmp.isValid()) {
940 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
941 V: Addr, DestTy: Tmp.getElementType());
942 CGF.Builder.CreateStore(Val: Addr, Addr: Tmp);
943 return MostTopTmp;
944 }
945
946 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
947 V: Addr, DestTy: OriginalBaseAddress.getType());
948 return OriginalBaseAddress.withPointer(NewPointer: Addr, IsKnownNonNull: NotKnownNonNull);
949}
950
951static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) {
952 const VarDecl *OrigVD = nullptr;
953 if (const auto *OASE = dyn_cast<ArraySectionExpr>(Val: Ref)) {
954 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
955 while (const auto *TempOASE = dyn_cast<ArraySectionExpr>(Val: Base))
956 Base = TempOASE->getBase()->IgnoreParenImpCasts();
957 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Val: Base))
958 Base = TempASE->getBase()->IgnoreParenImpCasts();
959 DE = cast<DeclRefExpr>(Val: Base);
960 OrigVD = cast<VarDecl>(Val: DE->getDecl());
961 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Val: Ref)) {
962 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts();
963 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Val: Base))
964 Base = TempASE->getBase()->IgnoreParenImpCasts();
965 DE = cast<DeclRefExpr>(Val: Base);
966 OrigVD = cast<VarDecl>(Val: DE->getDecl());
967 }
968 return OrigVD;
969}
970
971Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N,
972 Address PrivateAddr) {
973 const DeclRefExpr *DE;
974 if (const VarDecl *OrigVD = ::getBaseDecl(Ref: ClausesData[N].Ref, DE)) {
975 BaseDecls.emplace_back(Args&: OrigVD);
976 LValue OriginalBaseLValue = CGF.EmitLValue(E: DE);
977 LValue BaseLValue =
978 loadToBegin(CGF, BaseTy: OrigVD->getType(), ElTy: SharedAddresses[N].first.getType(),
979 BaseLV: OriginalBaseLValue);
980 Address SharedAddr = SharedAddresses[N].first.getAddress();
981 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff(
982 ElemTy: SharedAddr.getElementType(), LHS: BaseLValue.getPointer(CGF),
983 RHS: SharedAddr.emitRawPointer(CGF));
984 llvm::Value *PrivatePointer =
985 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
986 V: PrivateAddr.emitRawPointer(CGF), DestTy: SharedAddr.getType());
987 llvm::Value *Ptr = CGF.Builder.CreateGEP(
988 Ty: SharedAddr.getElementType(), Ptr: PrivatePointer, IdxList: Adjustment);
989 return castToBase(CGF, BaseTy: OrigVD->getType(),
990 ElTy: SharedAddresses[N].first.getType(),
991 OriginalBaseAddress: OriginalBaseLValue.getAddress(), Addr: Ptr);
992 }
993 BaseDecls.emplace_back(
994 Args: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: ClausesData[N].Ref)->getDecl()));
995 return PrivateAddr;
996}
997
998bool ReductionCodeGen::usesReductionInitializer(unsigned N) const {
999 const OMPDeclareReductionDecl *DRD =
1000 getReductionInit(ReductionOp: ClausesData[N].ReductionOp);
1001 return DRD && DRD->getInitializer();
1002}
1003
1004LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
1005 return CGF.EmitLoadOfPointerLValue(
1006 Ptr: CGF.GetAddrOfLocalVar(VD: getThreadIDVariable()),
1007 PtrTy: getThreadIDVariable()->getType()->castAs<PointerType>());
1008}
1009
1010void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt *S) {
1011 if (!CGF.HaveInsertPoint())
1012 return;
1013 // 1.2.2 OpenMP Language Terminology
1014 // Structured block - An executable statement with a single entry at the
1015 // top and a single exit at the bottom.
1016 // The point of exit cannot be a branch out of the structured block.
1017 // longjmp() and throw() must not violate the entry/exit criteria.
1018 CGF.EHStack.pushTerminate();
1019 if (S)
1020 CGF.incrementProfileCounter(S);
1021 CodeGen(CGF);
1022 CGF.EHStack.popTerminate();
1023}
1024
1025LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
1026 CodeGenFunction &CGF) {
1027 return CGF.MakeAddrLValue(Addr: CGF.GetAddrOfLocalVar(VD: getThreadIDVariable()),
1028 T: getThreadIDVariable()->getType(),
1029 Source: AlignmentSource::Decl);
1030}
1031
1032static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1033 QualType FieldTy) {
1034 auto *Field = FieldDecl::Create(
1035 C, DC, StartLoc: SourceLocation(), IdLoc: SourceLocation(), /*Id=*/nullptr, T: FieldTy,
1036 TInfo: C.getTrivialTypeSourceInfo(T: FieldTy, Loc: SourceLocation()),
1037 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1038 Field->setAccess(AS_public);
1039 DC->addDecl(D: Field);
1040 return Field;
1041}
1042
1043CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM)
1044 : CGM(CGM), OMPBuilder(CGM.getModule()) {
1045 KmpCriticalNameTy = llvm::ArrayType::get(ElementType: CGM.Int32Ty, /*NumElements*/ 8);
1046 llvm::OpenMPIRBuilderConfig Config(
1047 CGM.getLangOpts().OpenMPIsTargetDevice, isGPU(),
1048 CGM.getLangOpts().OpenMPOffloadMandatory,
1049 /*HasRequiresReverseOffload*/ false, /*HasRequiresUnifiedAddress*/ false,
1050 hasRequiresUnifiedSharedMemory(), /*HasRequiresDynamicAllocators*/ false);
1051 Config.setDefaultTargetAS(
1052 CGM.getContext().getTargetInfo().getTargetAddressSpace(AS: LangAS::Default));
1053 Config.setRuntimeCC(CGM.getRuntimeCC());
1054
1055 OMPBuilder.setConfig(Config);
1056 OMPBuilder.initialize();
1057 OMPBuilder.loadOffloadInfoMetadata(VFS&: *CGM.getFileSystem(),
1058 HostFilePath: CGM.getLangOpts().OpenMPIsTargetDevice
1059 ? CGM.getLangOpts().OMPHostIRFile
1060 : StringRef{});
1061
1062 // The user forces the compiler to behave as if omp requires
1063 // unified_shared_memory was given.
1064 if (CGM.getLangOpts().OpenMPForceUSM) {
1065 HasRequiresUnifiedSharedMemory = true;
1066 OMPBuilder.Config.setHasRequiresUnifiedSharedMemory(true);
1067 }
1068}
1069
1070void CGOpenMPRuntime::clear() {
1071 InternalVars.clear();
1072 // Clean non-target variable declarations possibly used only in debug info.
1073 for (const auto &Data : EmittedNonTargetVariables) {
1074 if (!Data.getValue().pointsToAliveValue())
1075 continue;
1076 auto *GV = dyn_cast<llvm::GlobalVariable>(Val: Data.getValue());
1077 if (!GV)
1078 continue;
1079 if (!GV->isDeclaration() || GV->getNumUses() > 0)
1080 continue;
1081 GV->eraseFromParent();
1082 }
1083}
1084
1085std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const {
1086 return OMPBuilder.createPlatformSpecificName(Parts);
1087}
1088
1089static llvm::Function *
1090emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty,
1091 const Expr *CombinerInitializer, const VarDecl *In,
1092 const VarDecl *Out, bool IsCombiner) {
1093 // void .omp_combiner.(Ty *in, Ty *out);
1094 ASTContext &C = CGM.getContext();
1095 QualType PtrTy = C.getPointerType(T: Ty).withRestrict();
1096 auto *OmpOutParm = ImplicitParamDecl::Create(
1097 C, /*DC=*/nullptr, IdLoc: Out->getLocation(),
1098 /*Id=*/nullptr, T: PtrTy, ParamKind: ImplicitParamKind::Other);
1099 auto *OmpInParm = ImplicitParamDecl::Create(
1100 C, /*DC=*/nullptr, IdLoc: In->getLocation(),
1101 /*Id=*/nullptr, T: PtrTy, ParamKind: ImplicitParamKind::Other);
1102 FunctionArgList Args{OmpOutParm, OmpInParm};
1103 const CGFunctionInfo &FnInfo =
1104 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
1105 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(Info: FnInfo);
1106 std::string Name = CGM.getOpenMPRuntime().getName(
1107 Parts: {IsCombiner ? "omp_combiner" : "omp_initializer", ""});
1108 auto *Fn = llvm::Function::Create(Ty: FnTy, Linkage: llvm::GlobalValue::InternalLinkage,
1109 N: Name, M: &CGM.getModule());
1110 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: Fn, FI: FnInfo);
1111 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
1112 Fn->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
1113 if (CGM.getCodeGenOpts().OptimizationLevel != 0) {
1114 Fn->removeFnAttr(Kind: llvm::Attribute::NoInline);
1115 Fn->removeFnAttr(Kind: llvm::Attribute::OptimizeNone);
1116 Fn->addFnAttr(Kind: llvm::Attribute::AlwaysInline);
1117 }
1118 CodeGenFunction CGF(CGM);
1119 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions.
1120 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions.
1121 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn, FnInfo, Args, Loc: In->getLocation(),
1122 StartLoc: Out->getLocation());
1123 CodeGenFunction::OMPPrivateScope Scope(CGF);
1124 Address AddrIn = CGF.GetAddrOfLocalVar(VD: OmpInParm);
1125 Scope.addPrivate(
1126 LocalVD: In, Addr: CGF.EmitLoadOfPointerLValue(Ptr: AddrIn, PtrTy: PtrTy->castAs<PointerType>())
1127 .getAddress());
1128 Address AddrOut = CGF.GetAddrOfLocalVar(VD: OmpOutParm);
1129 Scope.addPrivate(
1130 LocalVD: Out, Addr: CGF.EmitLoadOfPointerLValue(Ptr: AddrOut, PtrTy: PtrTy->castAs<PointerType>())
1131 .getAddress());
1132 (void)Scope.Privatize();
1133 if (!IsCombiner && Out->hasInit() &&
1134 !CGF.isTrivialInitializer(Init: Out->getInit())) {
1135 CGF.EmitAnyExprToMem(E: Out->getInit(), Location: CGF.GetAddrOfLocalVar(VD: Out),
1136 Quals: Out->getType().getQualifiers(),
1137 /*IsInitializer=*/true);
1138 }
1139 if (CombinerInitializer)
1140 CGF.EmitIgnoredExpr(E: CombinerInitializer);
1141 Scope.ForceCleanup();
1142 CGF.FinishFunction();
1143 return Fn;
1144}
1145
1146void CGOpenMPRuntime::emitUserDefinedReduction(
1147 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) {
1148 if (UDRMap.count(Val: D) > 0)
1149 return;
1150 llvm::Function *Combiner = emitCombinerOrInitializer(
1151 CGM, Ty: D->getType(), CombinerInitializer: D->getCombiner(),
1152 In: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D->getCombinerIn())->getDecl()),
1153 Out: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D->getCombinerOut())->getDecl()),
1154 /*IsCombiner=*/true);
1155 llvm::Function *Initializer = nullptr;
1156 if (const Expr *Init = D->getInitializer()) {
1157 Initializer = emitCombinerOrInitializer(
1158 CGM, Ty: D->getType(),
1159 CombinerInitializer: D->getInitializerKind() == OMPDeclareReductionInitKind::Call ? Init
1160 : nullptr,
1161 In: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D->getInitOrig())->getDecl()),
1162 Out: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D->getInitPriv())->getDecl()),
1163 /*IsCombiner=*/false);
1164 }
1165 UDRMap.try_emplace(Key: D, Args&: Combiner, Args&: Initializer);
1166 if (CGF)
1167 FunctionUDRMap[CGF->CurFn].push_back(Elt: D);
1168}
1169
1170std::pair<llvm::Function *, llvm::Function *>
1171CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) {
1172 auto I = UDRMap.find(Val: D);
1173 if (I != UDRMap.end())
1174 return I->second;
1175 emitUserDefinedReduction(/*CGF=*/nullptr, D);
1176 return UDRMap.lookup(Val: D);
1177}
1178
1179namespace {
1180// Temporary RAII solution to perform a push/pop stack event on the OpenMP IR
1181// Builder if one is present.
1182struct PushAndPopStackRAII {
1183 PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF,
1184 bool HasCancel, llvm::omp::Directive Kind)
1185 : OMPBuilder(OMPBuilder) {
1186 if (!OMPBuilder)
1187 return;
1188
1189 // The following callback is the crucial part of clangs cleanup process.
1190 //
1191 // NOTE:
1192 // Once the OpenMPIRBuilder is used to create parallel regions (and
1193 // similar), the cancellation destination (Dest below) is determined via
1194 // IP. That means if we have variables to finalize we split the block at IP,
1195 // use the new block (=BB) as destination to build a JumpDest (via
1196 // getJumpDestInCurrentScope(BB)) which then is fed to
1197 // EmitBranchThroughCleanup. Furthermore, there will not be the need
1198 // to push & pop an FinalizationInfo object.
1199 // The FiniCB will still be needed but at the point where the
1200 // OpenMPIRBuilder is asked to construct a parallel (or similar) construct.
1201 auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) {
1202 assert(IP == IP.getNodeParent()->end() &&
1203 "Clang CG should cause non-terminated block!");
1204 CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1205 CGF.Builder.restoreIP(IP);
1206 CodeGenFunction::JumpDest Dest =
1207 CGF.getOMPCancelDestination(Kind: OMPD_parallel);
1208 CGF.EmitBranchThroughCleanup(Dest);
1209 return llvm::Error::success();
1210 };
1211
1212 // TODO: Remove this once we emit parallel regions through the
1213 // OpenMPIRBuilder as it can do this setup internally.
1214 llvm::OpenMPIRBuilder::FinalizationInfo FI({FiniCB, Kind, HasCancel});
1215 OMPBuilder->pushFinalizationCB(FI: std::move(FI));
1216 }
1217 ~PushAndPopStackRAII() {
1218 if (OMPBuilder)
1219 OMPBuilder->popFinalizationCB();
1220 }
1221 llvm::OpenMPIRBuilder *OMPBuilder;
1222};
1223} // namespace
1224
1225static llvm::Function *emitParallelOrTeamsOutlinedFunction(
1226 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS,
1227 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1228 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) {
1229 assert(ThreadIDVar->getType()->isPointerType() &&
1230 "thread id variable must be of type kmp_int32 *");
1231 CodeGenFunction CGF(CGM, true);
1232 bool HasCancel = false;
1233 if (const auto *OPD = dyn_cast<OMPParallelDirective>(Val: &D))
1234 HasCancel = OPD->hasCancel();
1235 else if (const auto *OPD = dyn_cast<OMPTargetParallelDirective>(Val: &D))
1236 HasCancel = OPD->hasCancel();
1237 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(Val: &D))
1238 HasCancel = OPSD->hasCancel();
1239 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(Val: &D))
1240 HasCancel = OPFD->hasCancel();
1241 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(Val: &D))
1242 HasCancel = OPFD->hasCancel();
1243 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(Val: &D))
1244 HasCancel = OPFD->hasCancel();
1245 else if (const auto *OPFD =
1246 dyn_cast<OMPTeamsDistributeParallelForDirective>(Val: &D))
1247 HasCancel = OPFD->hasCancel();
1248 else if (const auto *OPFD =
1249 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(Val: &D))
1250 HasCancel = OPFD->hasCancel();
1251
1252 // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new
1253 // parallel region to make cancellation barriers work properly.
1254 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
1255 PushAndPopStackRAII PSR(&OMPBuilder, CGF, HasCancel, InnermostKind);
1256 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind,
1257 HasCancel, OutlinedHelperName);
1258 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1259 return CGF.GenerateOpenMPCapturedStmtFunction(S: *CS, D);
1260}
1261
1262std::string CGOpenMPRuntime::getOutlinedHelperName(StringRef Name) const {
1263 std::string Suffix = getName(Parts: {"omp_outlined"});
1264 return (Name + Suffix).str();
1265}
1266
1267std::string CGOpenMPRuntime::getOutlinedHelperName(CodeGenFunction &CGF) const {
1268 return getOutlinedHelperName(Name: CGF.CurFn->getName());
1269}
1270
1271std::string CGOpenMPRuntime::getReductionFuncName(StringRef Name) const {
1272 std::string Suffix = getName(Parts: {"omp", "reduction", "reduction_func"});
1273 return (Name + Suffix).str();
1274}
1275
1276llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction(
1277 CodeGenFunction &CGF, const OMPExecutableDirective &D,
1278 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1279 const RegionCodeGenTy &CodeGen) {
1280 const CapturedStmt *CS = D.getCapturedStmt(RegionKind: OMPD_parallel);
1281 return emitParallelOrTeamsOutlinedFunction(
1282 CGM, D, CS, ThreadIDVar, InnermostKind, OutlinedHelperName: getOutlinedHelperName(CGF),
1283 CodeGen);
1284}
1285
1286llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction(
1287 CodeGenFunction &CGF, const OMPExecutableDirective &D,
1288 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1289 const RegionCodeGenTy &CodeGen) {
1290 const CapturedStmt *CS = D.getCapturedStmt(RegionKind: OMPD_teams);
1291 llvm::Function *OutlinedFn = emitParallelOrTeamsOutlinedFunction(
1292 CGM, D, CS, ThreadIDVar, InnermostKind, OutlinedHelperName: getOutlinedHelperName(CGF),
1293 CodeGen);
1294 // A teams body is called once per team and is not handed back to the runtime
1295 // as a callback, so unlike a parallel body it cannot be re-entered while a
1296 // call to it is live.
1297 OutlinedFn->setDoesNotRecurse();
1298 return OutlinedFn;
1299}
1300
1301llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction(
1302 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1303 const VarDecl *PartIDVar, const VarDecl *TaskTVar,
1304 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
1305 bool Tied, unsigned &NumberOfParts) {
1306 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF,
1307 PrePostActionTy &) {
1308 llvm::Value *ThreadID = getThreadID(CGF, Loc: D.getBeginLoc());
1309 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc: D.getBeginLoc());
1310 llvm::Value *TaskArgs[] = {
1311 UpLoc, ThreadID,
1312 CGF.EmitLoadOfPointerLValue(Ptr: CGF.GetAddrOfLocalVar(VD: TaskTVar),
1313 PtrTy: TaskTVar->getType()->castAs<PointerType>())
1314 .getPointer(CGF)};
1315 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
1316 M&: CGM.getModule(), FnID: OMPRTL___kmpc_omp_task),
1317 args: TaskArgs);
1318 };
1319 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar,
1320 UntiedCodeGen);
1321 CodeGen.setAction(Action);
1322 assert(!ThreadIDVar->getType()->isPointerType() &&
1323 "thread id variable must be of type kmp_int32 for tasks");
1324 const OpenMPDirectiveKind Region =
1325 isOpenMPTaskLoopDirective(DKind: D.getDirectiveKind()) ? OMPD_taskloop
1326 : OMPD_task;
1327 const CapturedStmt *CS = D.getCapturedStmt(RegionKind: Region);
1328 bool HasCancel = false;
1329 if (const auto *TD = dyn_cast<OMPTaskDirective>(Val: &D))
1330 HasCancel = TD->hasCancel();
1331 else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(Val: &D))
1332 HasCancel = TD->hasCancel();
1333 else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(Val: &D))
1334 HasCancel = TD->hasCancel();
1335 else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(Val: &D))
1336 HasCancel = TD->hasCancel();
1337
1338 CodeGenFunction CGF(CGM, true);
1339 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen,
1340 InnermostKind, HasCancel, Action);
1341 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1342 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(S: *CS);
1343 if (!Tied)
1344 NumberOfParts = Action.getNumberOfParts();
1345 return Res;
1346}
1347
1348void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF,
1349 bool AtCurrentPoint) {
1350 auto &Elem = OpenMPLocThreadIDMap[CGF.CurFn];
1351 assert(!Elem.ServiceInsertPt && "Insert point is set already.");
1352
1353 llvm::Value *Undef = llvm::UndefValue::get(T: CGF.Int32Ty);
1354 if (AtCurrentPoint) {
1355 Elem.ServiceInsertPt = new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt",
1356 CGF.Builder.GetInsertBlock());
1357 } else {
1358 Elem.ServiceInsertPt = new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt");
1359 Elem.ServiceInsertPt->insertAfter(InsertPos: CGF.AllocaInsertPt->getIterator());
1360 }
1361}
1362
1363void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) {
1364 auto &Elem = OpenMPLocThreadIDMap[CGF.CurFn];
1365 if (Elem.ServiceInsertPt) {
1366 llvm::Instruction *Ptr = Elem.ServiceInsertPt;
1367 Elem.ServiceInsertPt = nullptr;
1368 Ptr->eraseFromParent();
1369 }
1370}
1371
1372static StringRef getIdentStringFromSourceLocation(CodeGenFunction &CGF,
1373 SourceLocation Loc,
1374 SmallString<128> &Buffer) {
1375 llvm::raw_svector_ostream OS(Buffer);
1376 // Build debug location
1377 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1378 OS << ";";
1379 if (auto *DbgInfo = CGF.getDebugInfo())
1380 OS << DbgInfo->remapDIPath(PLoc.getFilename());
1381 else
1382 OS << PLoc.getFilename();
1383 OS << ";";
1384 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(Val: CGF.CurFuncDecl))
1385 OS << FD->getQualifiedNameAsString();
1386 OS << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
1387 return OS.str();
1388}
1389
1390llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
1391 SourceLocation Loc,
1392 unsigned Flags, bool EmitLoc) {
1393 uint32_t SrcLocStrSize;
1394 llvm::Constant *SrcLocStr;
1395 if ((!EmitLoc && CGM.getCodeGenOpts().getDebugInfo() ==
1396 llvm::codegenoptions::NoDebugInfo) ||
1397 Loc.isInvalid()) {
1398 SrcLocStr = OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize);
1399 } else {
1400 std::string FunctionName;
1401 std::string FileName;
1402 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(Val: CGF.CurFuncDecl))
1403 FunctionName = FD->getQualifiedNameAsString();
1404 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1405 if (auto *DbgInfo = CGF.getDebugInfo())
1406 FileName = DbgInfo->remapDIPath(PLoc.getFilename());
1407 else
1408 FileName = PLoc.getFilename();
1409 unsigned Line = PLoc.getLine();
1410 unsigned Column = PLoc.getColumn();
1411 SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(FunctionName, FileName, Line,
1412 Column, SrcLocStrSize);
1413 }
1414 unsigned Reserved2Flags = getDefaultLocationReserved2Flags();
1415 return OMPBuilder.getOrCreateIdent(
1416 SrcLocStr, SrcLocStrSize, Flags: llvm::omp::IdentFlag(Flags), Reserve2Flags: Reserved2Flags);
1417}
1418
1419llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
1420 SourceLocation Loc) {
1421 assert(CGF.CurFn && "No function in current CodeGenFunction.");
1422 // If the OpenMPIRBuilder is used we need to use it for all thread id calls as
1423 // the clang invariants used below might be broken.
1424 if (CGM.getLangOpts().OpenMPIRBuilder) {
1425 SmallString<128> Buffer;
1426 OMPBuilder.updateToLocation(Loc: CGF.Builder);
1427 uint32_t SrcLocStrSize;
1428 auto *SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(
1429 LocStr: getIdentStringFromSourceLocation(CGF, Loc, Buffer), SrcLocStrSize);
1430 return OMPBuilder.getOrCreateThreadID(
1431 Ident: OMPBuilder.getOrCreateIdent(SrcLocStr, SrcLocStrSize));
1432 }
1433
1434 llvm::Value *ThreadID = nullptr;
1435 // Check whether we've already cached a load of the thread id in this
1436 // function.
1437 auto I = OpenMPLocThreadIDMap.find(Val: CGF.CurFn);
1438 if (I != OpenMPLocThreadIDMap.end()) {
1439 ThreadID = I->second.ThreadID;
1440 if (ThreadID != nullptr)
1441 return ThreadID;
1442 }
1443 // If exceptions are enabled, do not use parameter to avoid possible crash.
1444 if (auto *OMPRegionInfo =
1445 dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo)) {
1446 if (OMPRegionInfo->getThreadIDVariable()) {
1447 // Check if this an outlined function with thread id passed as argument.
1448 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
1449 llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent();
1450 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions ||
1451 !CGF.getLangOpts().CXXExceptions ||
1452 CGF.Builder.GetInsertBlock() == TopBlock ||
1453 !isa<llvm::Instruction>(Val: LVal.getPointer(CGF)) ||
1454 cast<llvm::Instruction>(Val: LVal.getPointer(CGF))->getParent() ==
1455 TopBlock ||
1456 cast<llvm::Instruction>(Val: LVal.getPointer(CGF))->getParent() ==
1457 CGF.Builder.GetInsertBlock()) {
1458 ThreadID = CGF.EmitLoadOfScalar(lvalue: LVal, Loc);
1459 // If value loaded in entry block, cache it and use it everywhere in
1460 // function.
1461 if (CGF.Builder.GetInsertBlock() == TopBlock)
1462 OpenMPLocThreadIDMap[CGF.CurFn].ThreadID = ThreadID;
1463 return ThreadID;
1464 }
1465 }
1466 }
1467
1468 // This is not an outlined function region - need to call __kmpc_int32
1469 // kmpc_global_thread_num(ident_t *loc).
1470 // Generate thread id value and cache this value for use across the
1471 // function.
1472 auto &Elem = OpenMPLocThreadIDMap[CGF.CurFn];
1473 if (!Elem.ServiceInsertPt)
1474 setLocThreadIdInsertPt(CGF);
1475 CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1476 CGF.Builder.SetInsertPoint(Elem.ServiceInsertPt);
1477 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, TemporaryLocation: Loc);
1478 llvm::CallInst *Call = CGF.Builder.CreateCall(
1479 Callee: OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(),
1480 FnID: OMPRTL___kmpc_global_thread_num),
1481 Args: emitUpdateLocation(CGF, Loc));
1482 Call->setCallingConv(CGF.getRuntimeCC());
1483 Elem.ThreadID = Call;
1484 return Call;
1485}
1486
1487void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
1488 assert(CGF.CurFn && "No function in current CodeGenFunction.");
1489 if (OpenMPLocThreadIDMap.count(Val: CGF.CurFn)) {
1490 clearLocThreadIdInsertPt(CGF);
1491 OpenMPLocThreadIDMap.erase(Val: CGF.CurFn);
1492 }
1493 if (auto I = FunctionUDRMap.find(Val: CGF.CurFn); I != FunctionUDRMap.end()) {
1494 for (const auto *D : I->second)
1495 UDRMap.erase(Val: D);
1496 FunctionUDRMap.erase(I);
1497 }
1498 if (auto I = FunctionUDMMap.find(Val: CGF.CurFn); I != FunctionUDMMap.end()) {
1499 for (const auto *D : I->second)
1500 UDMMap.erase(Val: D);
1501 FunctionUDMMap.erase(I);
1502 }
1503 LastprivateConditionalToTypes.erase(Val: CGF.CurFn);
1504 FunctionToUntiedTaskStackMap.erase(Val: CGF.CurFn);
1505}
1506
1507llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
1508 return OMPBuilder.IdentPtr;
1509}
1510
1511static llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseKind
1512convertDeviceClause(const VarDecl *VD) {
1513 std::optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
1514 OMPDeclareTargetDeclAttr::getDeviceType(VD);
1515 if (!DevTy)
1516 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseNone;
1517
1518 switch ((int)*DevTy) { // Avoid -Wcovered-switch-default
1519 case OMPDeclareTargetDeclAttr::DT_Host:
1520 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseHost;
1521 break;
1522 case OMPDeclareTargetDeclAttr::DT_NoHost:
1523 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseNoHost;
1524 break;
1525 case OMPDeclareTargetDeclAttr::DT_Any:
1526 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseAny;
1527 break;
1528 default:
1529 return llvm::OffloadEntriesInfoManager::OMPTargetDeviceClauseNone;
1530 break;
1531 }
1532}
1533
1534static llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind
1535convertCaptureClause(const VarDecl *VD) {
1536 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> MapType =
1537 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
1538 if (!MapType)
1539 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryNone;
1540 switch ((int)*MapType) { // Avoid -Wcovered-switch-default
1541 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_To:
1542 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryTo;
1543 break;
1544 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_Enter:
1545 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryEnter;
1546 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_Link:
1547 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryLink;
1548 break;
1549 case OMPDeclareTargetDeclAttr::MapTypeTy::MT_Local:
1550 // MT_Local variables don't need offload entry (device-local).
1551 llvm_unreachable("MT_Local should not reach convertCaptureClause");
1552 break;
1553 default:
1554 return llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryNone;
1555 break;
1556 }
1557}
1558
1559static llvm::TargetRegionEntryInfo getEntryInfoFromPresumedLoc(
1560 CodeGenModule &CGM, llvm::OpenMPIRBuilder &OMPBuilder,
1561 SourceLocation BeginLoc, llvm::StringRef ParentName = "") {
1562
1563 auto FileInfoCallBack = [&]() {
1564 SourceManager &SM = CGM.getContext().getSourceManager();
1565 PresumedLoc PLoc = SM.getPresumedLoc(Loc: BeginLoc);
1566
1567 if (!CGM.getFileSystem()->exists(Path: PLoc.getFilename()))
1568 PLoc = SM.getPresumedLoc(Loc: BeginLoc, /*UseLineDirectives=*/false);
1569
1570 return std::pair<std::string, uint64_t>(PLoc.getFilename(), PLoc.getLine());
1571 };
1572
1573 return OMPBuilder.getTargetEntryUniqueInfo(CallBack: FileInfoCallBack,
1574 VFS&: *CGM.getFileSystem(), ParentName);
1575}
1576
1577ConstantAddress CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) {
1578 auto AddrOfGlobal = [&VD, this]() { return CGM.GetAddrOfGlobal(GD: VD); };
1579
1580 auto LinkageForVariable = [&VD, this]() {
1581 return CGM.getLLVMLinkageVarDefinition(VD);
1582 };
1583
1584 std::vector<llvm::GlobalVariable *> GeneratedRefs;
1585
1586 llvm::Type *LlvmPtrTy = CGM.getTypes().ConvertTypeForMem(
1587 T: CGM.getContext().getPointerType(T: VD->getType()));
1588 llvm::Constant *addr = OMPBuilder.getAddrOfDeclareTargetVar(
1589 CaptureClause: convertCaptureClause(VD), DeviceClause: convertDeviceClause(VD),
1590 IsDeclaration: VD->hasDefinition(CGM.getContext()) == VarDecl::DeclarationOnly,
1591 IsExternallyVisible: VD->isExternallyVisible(),
1592 EntryInfo: getEntryInfoFromPresumedLoc(CGM, OMPBuilder,
1593 BeginLoc: VD->getCanonicalDecl()->getBeginLoc()),
1594 MangledName: CGM.getMangledName(GD: VD), GeneratedRefs, OpenMPSIMD: CGM.getLangOpts().OpenMPSimd,
1595 TargetTriple: CGM.getLangOpts().OMPTargetTriples, LlvmPtrTy, GlobalInitializer: AddrOfGlobal,
1596 VariableLinkage: LinkageForVariable);
1597
1598 if (!addr)
1599 return ConstantAddress::invalid();
1600 return ConstantAddress(addr, LlvmPtrTy, CGM.getContext().getDeclAlign(D: VD));
1601}
1602
1603llvm::Constant *
1604CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
1605 assert(!CGM.getLangOpts().OpenMPUseTLS ||
1606 !CGM.getContext().getTargetInfo().isTLSSupported());
1607 // Lookup the entry, lazily creating it if necessary.
1608 std::string Suffix = getName(Parts: {"cache", ""});
1609 return OMPBuilder.getOrCreateInternalVariable(
1610 Ty: CGM.Int8PtrPtrTy, Name: Twine(CGM.getMangledName(GD: VD)).concat(Suffix).str());
1611}
1612
1613Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
1614 const VarDecl *VD,
1615 Address VDAddr,
1616 SourceLocation Loc) {
1617 if (CGM.getLangOpts().OpenMPUseTLS &&
1618 CGM.getContext().getTargetInfo().isTLSSupported())
1619 return VDAddr;
1620
1621 llvm::Type *VarTy = VDAddr.getElementType();
1622 llvm::Value *Args[] = {
1623 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
1624 CGF.Builder.CreatePointerCast(V: VDAddr.emitRawPointer(CGF), DestTy: CGM.Int8PtrTy),
1625 CGM.getSize(numChars: CGM.GetTargetTypeStoreSize(Ty: VarTy)),
1626 getOrCreateThreadPrivateCache(VD)};
1627 return Address(
1628 CGF.EmitRuntimeCall(
1629 callee: OMPBuilder.getOrCreateRuntimeFunction(
1630 M&: CGM.getModule(), FnID: OMPRTL___kmpc_threadprivate_cached),
1631 args: Args),
1632 CGF.Int8Ty, VDAddr.getAlignment());
1633}
1634
1635void CGOpenMPRuntime::emitThreadPrivateVarInit(
1636 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor,
1637 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
1638 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
1639 // library.
1640 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc);
1641 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
1642 M&: CGM.getModule(), FnID: OMPRTL___kmpc_global_thread_num),
1643 args: OMPLoc);
1644 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
1645 // to register constructor/destructor for variable.
1646 llvm::Value *Args[] = {
1647 OMPLoc,
1648 CGF.Builder.CreatePointerCast(V: VDAddr.emitRawPointer(CGF), DestTy: CGM.VoidPtrTy),
1649 Ctor, CopyCtor, Dtor};
1650 CGF.EmitRuntimeCall(
1651 callee: OMPBuilder.getOrCreateRuntimeFunction(
1652 M&: CGM.getModule(), FnID: OMPRTL___kmpc_threadprivate_register),
1653 args: Args);
1654}
1655
1656llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
1657 const VarDecl *VD, Address VDAddr, SourceLocation Loc,
1658 bool PerformInit, CodeGenFunction *CGF) {
1659 if (CGM.getLangOpts().OpenMPUseTLS &&
1660 CGM.getContext().getTargetInfo().isTLSSupported())
1661 return nullptr;
1662
1663 VD = VD->getDefinition(C&: CGM.getContext());
1664 if (VD && ThreadPrivateWithDefinition.insert(key: CGM.getMangledName(GD: VD)).second) {
1665 QualType ASTTy = VD->getType();
1666
1667 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
1668 const Expr *Init = VD->getAnyInitializer();
1669 if (CGM.getLangOpts().CPlusPlus && PerformInit) {
1670 // Generate function that re-emits the declaration's initializer into the
1671 // threadprivate copy of the variable VD
1672 CodeGenFunction CtorCGF(CGM);
1673 auto *Dst = ImplicitParamDecl::Create(
1674 C&: CGM.getContext(), /*DC=*/nullptr, IdLoc: Loc,
1675 /*Id=*/nullptr, T: CGM.getContext().VoidPtrTy, ParamKind: ImplicitParamKind::Other);
1676
1677 FunctionArgList Args{Dst};
1678 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
1679 resultType: CGM.getContext().VoidPtrTy, args: Args);
1680 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(Info: FI);
1681 std::string Name = getName(Parts: {"__kmpc_global_ctor_", ""});
1682 llvm::Function *Fn =
1683 CGM.CreateGlobalInitOrCleanUpFunction(ty: FTy, name: Name, FI, Loc);
1684 CtorCGF.StartFunction(GD: GlobalDecl(), RetTy: CGM.getContext().VoidPtrTy, Fn, FnInfo: FI,
1685 Args, Loc, StartLoc: Loc);
1686 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar(
1687 Addr: CtorCGF.GetAddrOfLocalVar(VD: Dst), /*Volatile=*/false,
1688 Ty: CGM.getContext().VoidPtrTy, Loc: Dst->getLocation());
1689 Address Arg(ArgVal, CtorCGF.ConvertTypeForMem(T: ASTTy),
1690 VDAddr.getAlignment());
1691 CtorCGF.EmitAnyExprToMem(E: Init, Location: Arg, Quals: Init->getType().getQualifiers(),
1692 /*IsInitializer=*/true);
1693 ArgVal = CtorCGF.EmitLoadOfScalar(
1694 Addr: CtorCGF.GetAddrOfLocalVar(VD: Dst), /*Volatile=*/false,
1695 Ty: CGM.getContext().VoidPtrTy, Loc: Dst->getLocation());
1696 CtorCGF.Builder.CreateStore(Val: ArgVal, Addr: CtorCGF.ReturnValue);
1697 CtorCGF.FinishFunction();
1698 Ctor = Fn;
1699 }
1700 if (VD->getType().isDestructedType() != QualType::DK_none) {
1701 // Generate function that emits destructor call for the threadprivate copy
1702 // of the variable VD
1703 CodeGenFunction DtorCGF(CGM);
1704 auto *Dst = ImplicitParamDecl::Create(
1705 C&: CGM.getContext(), /*DC=*/nullptr, IdLoc: Loc,
1706 /*Id=*/nullptr, T: CGM.getContext().VoidPtrTy, ParamKind: ImplicitParamKind::Other);
1707
1708 FunctionArgList Args{Dst};
1709 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
1710 resultType: CGM.getContext().VoidTy, args: Args);
1711 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(Info: FI);
1712 std::string Name = getName(Parts: {"__kmpc_global_dtor_", ""});
1713 llvm::Function *Fn =
1714 CGM.CreateGlobalInitOrCleanUpFunction(ty: FTy, name: Name, FI, Loc);
1715 auto NL = ApplyDebugLocation::CreateEmpty(CGF&: DtorCGF);
1716 DtorCGF.StartFunction(GD: GlobalDecl(), RetTy: CGM.getContext().VoidTy, Fn, FnInfo: FI, Args,
1717 Loc, StartLoc: Loc);
1718 // Create a scope with an artificial location for the body of this function.
1719 auto AL = ApplyDebugLocation::CreateArtificial(CGF&: DtorCGF);
1720 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar(
1721 Addr: DtorCGF.GetAddrOfLocalVar(VD: Dst),
1722 /*Volatile=*/false, Ty: CGM.getContext().VoidPtrTy, Loc: Dst->getLocation());
1723 DtorCGF.emitDestroy(
1724 addr: Address(ArgVal, DtorCGF.Int8Ty, VDAddr.getAlignment()), type: ASTTy,
1725 destroyer: DtorCGF.getDestroyer(destructionKind: ASTTy.isDestructedType()),
1726 useEHCleanupForArray: DtorCGF.needsEHCleanup(kind: ASTTy.isDestructedType()));
1727 DtorCGF.FinishFunction();
1728 Dtor = Fn;
1729 }
1730 // Do not emit init function if it is not required.
1731 if (!Ctor && !Dtor)
1732 return nullptr;
1733
1734 // Copying constructor for the threadprivate variable.
1735 // Must be NULL - reserved by runtime, but currently it requires that this
1736 // parameter is always NULL. Otherwise it fires assertion.
1737 CopyCtor = llvm::Constant::getNullValue(Ty: CGM.DefaultPtrTy);
1738 if (Ctor == nullptr) {
1739 Ctor = llvm::Constant::getNullValue(Ty: CGM.DefaultPtrTy);
1740 }
1741 if (Dtor == nullptr) {
1742 Dtor = llvm::Constant::getNullValue(Ty: CGM.DefaultPtrTy);
1743 }
1744 if (!CGF) {
1745 auto *InitFunctionTy =
1746 llvm::FunctionType::get(Result: CGM.VoidTy, /*isVarArg*/ false);
1747 std::string Name = getName(Parts: {"__omp_threadprivate_init_", ""});
1748 llvm::Function *InitFunction = CGM.CreateGlobalInitOrCleanUpFunction(
1749 ty: InitFunctionTy, name: Name, FI: CGM.getTypes().arrangeNullaryFunction());
1750 CodeGenFunction InitCGF(CGM);
1751 FunctionArgList ArgList;
1752 InitCGF.StartFunction(GD: GlobalDecl(), RetTy: CGM.getContext().VoidTy, Fn: InitFunction,
1753 FnInfo: CGM.getTypes().arrangeNullaryFunction(), Args: ArgList,
1754 Loc, StartLoc: Loc);
1755 emitThreadPrivateVarInit(CGF&: InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
1756 InitCGF.FinishFunction();
1757 return InitFunction;
1758 }
1759 emitThreadPrivateVarInit(CGF&: *CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
1760 }
1761 return nullptr;
1762}
1763
1764void CGOpenMPRuntime::emitDeclareTargetFunction(const FunctionDecl *FD,
1765 llvm::GlobalValue *GV) {
1766 std::optional<OMPDeclareTargetDeclAttr *> ActiveAttr =
1767 OMPDeclareTargetDeclAttr::getActiveAttr(VD: FD);
1768
1769 // We only need to handle active 'indirect' declare target functions.
1770 if (!ActiveAttr || !(*ActiveAttr)->getIndirect())
1771 return;
1772
1773 // Get a mangled name to store the new device global in.
1774 llvm::TargetRegionEntryInfo EntryInfo = getEntryInfoFromPresumedLoc(
1775 CGM, OMPBuilder, BeginLoc: FD->getCanonicalDecl()->getBeginLoc(), ParentName: FD->getName());
1776 SmallString<128> Name;
1777 OMPBuilder.OffloadInfoManager.getTargetRegionEntryFnName(Name, EntryInfo);
1778
1779 // We need to generate a new global to hold the address of the indirectly
1780 // called device function. Doing this allows us to keep the visibility and
1781 // linkage of the associated function unchanged while allowing the runtime to
1782 // access its value.
1783 llvm::GlobalValue *Addr = GV;
1784 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
1785 llvm::PointerType *FnPtrTy = llvm::PointerType::get(
1786 C&: CGM.getLLVMContext(),
1787 AddressSpace: CGM.getModule().getDataLayout().getProgramAddressSpace());
1788 Addr = new llvm::GlobalVariable(
1789 CGM.getModule(), FnPtrTy,
1790 /*isConstant=*/true, llvm::GlobalValue::ExternalLinkage, GV, Name,
1791 nullptr, llvm::GlobalValue::NotThreadLocal,
1792 CGM.getModule().getDataLayout().getDefaultGlobalsAddressSpace());
1793 Addr->setVisibility(llvm::GlobalValue::ProtectedVisibility);
1794 }
1795
1796 // Register the indirect Vtable:
1797 // This is similar to OMPTargetGlobalVarEntryIndirect, except that the
1798 // size field refers to the size of memory pointed to, not the size of
1799 // the pointer symbol itself (which is implicitly the size of a pointer).
1800 OMPBuilder.OffloadInfoManager.registerDeviceGlobalVarEntryInfo(
1801 VarName: Name, Addr, VarSize: CGM.GetTargetTypeStoreSize(Ty: CGM.VoidPtrTy).getQuantity(),
1802 Flags: llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryIndirect,
1803 Linkage: llvm::GlobalValue::WeakODRLinkage);
1804}
1805
1806void CGOpenMPRuntime::registerVTableOffloadEntry(llvm::GlobalVariable *VTable,
1807 const VarDecl *VD) {
1808 // TODO: add logic to avoid duplicate vtable registrations per
1809 // translation unit; though for external linkage, this should no
1810 // longer be an issue - or at least we can avoid the issue by
1811 // checking for an existing offloading entry. But, perhaps the
1812 // better approach is to defer emission of the vtables and offload
1813 // entries until later (by tracking a list of items that need to be
1814 // emitted).
1815
1816 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
1817
1818 // Generate a new externally visible global to point to the
1819 // internally visible vtable. Doing this allows us to keep the
1820 // visibility and linkage of the associated vtable unchanged while
1821 // allowing the runtime to access its value. The externally
1822 // visible global var needs to be emitted with a unique mangled
1823 // name that won't conflict with similarly named (internal)
1824 // vtables in other translation units.
1825
1826 // Register vtable with source location of dynamic object in map
1827 // clause.
1828 llvm::TargetRegionEntryInfo EntryInfo = getEntryInfoFromPresumedLoc(
1829 CGM, OMPBuilder, BeginLoc: VD->getCanonicalDecl()->getBeginLoc(),
1830 ParentName: VTable->getName());
1831
1832 llvm::GlobalVariable *Addr = VTable;
1833 SmallString<128> AddrName;
1834 OMPBuilder.OffloadInfoManager.getTargetRegionEntryFnName(Name&: AddrName, EntryInfo);
1835 AddrName.append(RHS: "addr");
1836
1837 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
1838 Addr = new llvm::GlobalVariable(
1839 CGM.getModule(), VTable->getType(),
1840 /*isConstant=*/true, llvm::GlobalValue::ExternalLinkage, VTable,
1841 AddrName,
1842 /*InsertBefore=*/nullptr, llvm::GlobalValue::NotThreadLocal,
1843 CGM.getModule().getDataLayout().getDefaultGlobalsAddressSpace());
1844 Addr->setVisibility(llvm::GlobalValue::ProtectedVisibility);
1845 }
1846 OMPBuilder.OffloadInfoManager.registerDeviceGlobalVarEntryInfo(
1847 VarName: AddrName, Addr: VTable,
1848 VarSize: CGM.getDataLayout().getTypeAllocSize(Ty: VTable->getInitializer()->getType()),
1849 Flags: llvm::OffloadEntriesInfoManager::OMPTargetGlobalVarEntryIndirectVTable,
1850 Linkage: llvm::GlobalValue::WeakODRLinkage);
1851}
1852
1853void CGOpenMPRuntime::emitAndRegisterVTable(CodeGenModule &CGM,
1854 CXXRecordDecl *CXXRecord,
1855 const VarDecl *VD) {
1856 // Register C++ VTable to OpenMP Offload Entry if it's a new
1857 // CXXRecordDecl.
1858 if (CXXRecord && CXXRecord->hasDefinition() && CXXRecord->isDynamicClass() &&
1859 !CGM.getOpenMPRuntime().VTableDeclMap.contains(Val: CXXRecord)) {
1860 auto Res = CGM.getOpenMPRuntime().VTableDeclMap.try_emplace(Key: CXXRecord, Args&: VD);
1861 if (Res.second) {
1862 CGM.EmitVTable(Class: CXXRecord);
1863 CodeGenVTables VTables = CGM.getVTables();
1864 llvm::GlobalVariable *VTablesAddr = VTables.GetAddrOfVTable(RD: CXXRecord);
1865 assert(VTablesAddr && "Expected non-null VTable address");
1866 // Must set VTables to weak since we're emitting them in multiple TUs now
1867 if (VTablesAddr->hasExternalLinkage())
1868 VTablesAddr->setLinkage(llvm::GlobalValue::WeakODRLinkage);
1869 CGM.getOpenMPRuntime().registerVTableOffloadEntry(VTable: VTablesAddr, VD);
1870 // Emit VTable for all the fields containing dynamic CXXRecord
1871 for (const FieldDecl *Field : CXXRecord->fields()) {
1872 if (CXXRecordDecl *RecordDecl = Field->getType()->getAsCXXRecordDecl())
1873 emitAndRegisterVTable(CGM, CXXRecord: RecordDecl, VD);
1874 }
1875 // Emit VTable for all dynamic parent class
1876 for (CXXBaseSpecifier &Base : CXXRecord->bases()) {
1877 if (CXXRecordDecl *BaseDecl = Base.getType()->getAsCXXRecordDecl())
1878 emitAndRegisterVTable(CGM, CXXRecord: BaseDecl, VD);
1879 }
1880 }
1881 }
1882}
1883
1884void CGOpenMPRuntime::registerVTable(const OMPExecutableDirective &D) {
1885 // Register VTable by scanning through the map clause of OpenMP target region.
1886 // Get CXXRecordDecl and VarDecl from Expr.
1887 auto GetVTableDecl = [](const Expr *E) {
1888 QualType VDTy = E->getType();
1889 CXXRecordDecl *CXXRecord = nullptr;
1890 if (const auto *RefType = VDTy->getAs<LValueReferenceType>())
1891 VDTy = RefType->getPointeeType();
1892 if (VDTy->isPointerType())
1893 CXXRecord = VDTy->getPointeeType()->getAsCXXRecordDecl();
1894 else
1895 CXXRecord = VDTy->getAsCXXRecordDecl();
1896
1897 const VarDecl *VD = nullptr;
1898 if (auto *DRE = dyn_cast<DeclRefExpr>(Val: E)) {
1899 // Handle BindingDecls by redirecting to their DecompositionDecl.
1900 if (auto *BD = dyn_cast<BindingDecl>(Val: DRE->getDecl()))
1901 VD = cast<VarDecl>(Val: BD->getDecomposedDecl());
1902 else
1903 VD = cast<VarDecl>(Val: DRE->getDecl());
1904 } else if (auto *MRE = dyn_cast<MemberExpr>(Val: E)) {
1905 if (auto *BaseDRE = dyn_cast<DeclRefExpr>(Val: MRE->getBase())) {
1906 if (auto *BaseVD = dyn_cast<VarDecl>(Val: BaseDRE->getDecl()))
1907 VD = BaseVD;
1908 }
1909 }
1910 return std::pair<CXXRecordDecl *, const VarDecl *>(CXXRecord, VD);
1911 };
1912 // Collect VTable from OpenMP map clause.
1913 for (const auto *C : D.getClausesOfKind<OMPMapClause>()) {
1914 for (const auto *E : C->varlist()) {
1915 auto DeclPair = GetVTableDecl(E);
1916 // Ensure VD is not null
1917 if (DeclPair.second)
1918 emitAndRegisterVTable(CGM, CXXRecord: DeclPair.first, VD: DeclPair.second);
1919 }
1920 }
1921}
1922
1923Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF,
1924 QualType VarType,
1925 StringRef Name) {
1926 std::string Suffix = getName(Parts: {"artificial", ""});
1927 llvm::Type *VarLVType = CGF.ConvertTypeForMem(T: VarType);
1928 llvm::GlobalVariable *GAddr = OMPBuilder.getOrCreateInternalVariable(
1929 Ty: VarLVType, Name: Twine(Name).concat(Suffix).str());
1930 if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS &&
1931 CGM.getTarget().isTLSSupported()) {
1932 GAddr->setThreadLocal(/*Val=*/true);
1933 return Address(GAddr, GAddr->getValueType(),
1934 CGM.getContext().getTypeAlignInChars(T: VarType));
1935 }
1936 std::string CacheSuffix = getName(Parts: {"cache", ""});
1937 llvm::Value *Args[] = {
1938 emitUpdateLocation(CGF, Loc: SourceLocation()),
1939 getThreadID(CGF, Loc: SourceLocation()),
1940 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(V: GAddr, DestTy: CGM.VoidPtrTy),
1941 CGF.Builder.CreateIntCast(V: CGF.getTypeSize(Ty: VarType), DestTy: CGM.SizeTy,
1942 /*isSigned=*/false),
1943 OMPBuilder.getOrCreateInternalVariable(
1944 Ty: CGM.VoidPtrPtrTy,
1945 Name: Twine(Name).concat(Suffix).concat(Suffix: CacheSuffix).str())};
1946 return Address(
1947 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1948 V: CGF.EmitRuntimeCall(
1949 callee: OMPBuilder.getOrCreateRuntimeFunction(
1950 M&: CGM.getModule(), FnID: OMPRTL___kmpc_threadprivate_cached),
1951 args: Args),
1952 DestTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
1953 VarLVType, CGM.getContext().getTypeAlignInChars(T: VarType));
1954}
1955
1956void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond,
1957 const RegionCodeGenTy &ThenGen,
1958 const RegionCodeGenTy &ElseGen) {
1959 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
1960
1961 // If the condition constant folds and can be elided, try to avoid emitting
1962 // the condition and the dead arm of the if/else.
1963 bool CondConstant;
1964 if (CGF.ConstantFoldsToSimpleInteger(Cond, Result&: CondConstant)) {
1965 if (CondConstant)
1966 ThenGen(CGF);
1967 else
1968 ElseGen(CGF);
1969 return;
1970 }
1971
1972 // Otherwise, the condition did not fold, or we couldn't elide it. Just
1973 // emit the conditional branch.
1974 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock(name: "omp_if.then");
1975 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock(name: "omp_if.else");
1976 llvm::BasicBlock *ContBlock = CGF.createBasicBlock(name: "omp_if.end");
1977 CGF.EmitBranchOnBoolExpr(Cond, TrueBlock: ThenBlock, FalseBlock: ElseBlock, /*TrueCount=*/0);
1978
1979 // Emit the 'then' code.
1980 CGF.EmitBlock(BB: ThenBlock);
1981 ThenGen(CGF);
1982 CGF.EmitBranch(Block: ContBlock);
1983 // Emit the 'else' code if present.
1984 // There is no need to emit line number for unconditional branch.
1985 (void)ApplyDebugLocation::CreateEmpty(CGF);
1986 CGF.EmitBlock(BB: ElseBlock);
1987 ElseGen(CGF);
1988 // There is no need to emit line number for unconditional branch.
1989 (void)ApplyDebugLocation::CreateEmpty(CGF);
1990 CGF.EmitBranch(Block: ContBlock);
1991 // Emit the continuation block for code after the if.
1992 CGF.EmitBlock(BB: ContBlock, /*IsFinished=*/true);
1993}
1994
1995void CGOpenMPRuntime::emitParallelCall(
1996 CodeGenFunction &CGF, SourceLocation Loc, llvm::Function *OutlinedFn,
1997 ArrayRef<llvm::Value *> CapturedVars, const Expr *IfCond,
1998 llvm::Value *NumThreads, OpenMPNumThreadsClauseModifier NumThreadsModifier,
1999 OpenMPSeverityClauseKind Severity, const Expr *Message) {
2000 if (!CGF.HaveInsertPoint())
2001 return;
2002 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
2003 auto &M = CGM.getModule();
2004 auto &&ThenGen = [&M, OutlinedFn, CapturedVars, RTLoc,
2005 this](CodeGenFunction &CGF, PrePostActionTy &) {
2006 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn);
2007 llvm::Value *Args[] = {
2008 RTLoc,
2009 CGF.Builder.getInt32(C: CapturedVars.size()), // Number of captured vars
2010 OutlinedFn};
2011 llvm::SmallVector<llvm::Value *, 16> RealArgs;
2012 RealArgs.append(in_start: std::begin(arr&: Args), in_end: std::end(arr&: Args));
2013 RealArgs.append(in_start: CapturedVars.begin(), in_end: CapturedVars.end());
2014
2015 llvm::FunctionCallee RTLFn =
2016 OMPBuilder.getOrCreateRuntimeFunction(M, FnID: OMPRTL___kmpc_fork_call);
2017 CGF.EmitRuntimeCall(callee: RTLFn, args: RealArgs);
2018 };
2019 auto &&ElseGen = [&M, OutlinedFn, CapturedVars, RTLoc, Loc,
2020 this](CodeGenFunction &CGF, PrePostActionTy &) {
2021 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
2022 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc);
2023 // Build calls:
2024 // __kmpc_serialized_parallel(&Loc, GTid);
2025 llvm::Value *Args[] = {RTLoc, ThreadID};
2026 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2027 M, FnID: OMPRTL___kmpc_serialized_parallel),
2028 args: Args);
2029
2030 // OutlinedFn(&GTid, &zero_bound, CapturedStruct);
2031 Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc);
2032 RawAddress ZeroAddrBound =
2033 CGF.CreateDefaultAlignTempAlloca(Ty: CGF.Int32Ty,
2034 /*Name=*/".bound.zero.addr");
2035 CGF.Builder.CreateStore(Val: CGF.Builder.getInt32(/*C*/ 0), Addr: ZeroAddrBound);
2036 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs;
2037 // ThreadId for serialized parallels is 0.
2038 OutlinedFnArgs.push_back(Elt: ThreadIDAddr.emitRawPointer(CGF));
2039 OutlinedFnArgs.push_back(Elt: ZeroAddrBound.getPointer());
2040 OutlinedFnArgs.append(in_start: CapturedVars.begin(), in_end: CapturedVars.end());
2041
2042 // Ensure we do not inline the function. This is trivially true for the ones
2043 // passed to __kmpc_fork_call but the ones called in serialized regions
2044 // could be inlined. This is not a perfect but it is closer to the invariant
2045 // we want, namely, every data environment starts with a new function.
2046 // TODO: We should pass the if condition to the runtime function and do the
2047 // handling there. Much cleaner code.
2048 OutlinedFn->removeFnAttr(Kind: llvm::Attribute::AlwaysInline);
2049 OutlinedFn->addFnAttr(Kind: llvm::Attribute::NoInline);
2050 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, Args: OutlinedFnArgs);
2051
2052 // __kmpc_end_serialized_parallel(&Loc, GTid);
2053 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID};
2054 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2055 M, FnID: OMPRTL___kmpc_end_serialized_parallel),
2056 args: EndArgs);
2057 };
2058 if (IfCond) {
2059 emitIfClause(CGF, Cond: IfCond, ThenGen, ElseGen);
2060 } else {
2061 RegionCodeGenTy ThenRCG(ThenGen);
2062 ThenRCG(CGF);
2063 }
2064}
2065
2066// If we're inside an (outlined) parallel region, use the region info's
2067// thread-ID variable (it is passed in a first argument of the outlined function
2068// as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
2069// regular serial code region, get thread ID by calling kmp_int32
2070// kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
2071// return the address of that temp.
2072Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
2073 SourceLocation Loc) {
2074 if (auto *OMPRegionInfo =
2075 dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo))
2076 if (OMPRegionInfo->getThreadIDVariable())
2077 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress();
2078
2079 llvm::Value *ThreadID = getThreadID(CGF, Loc);
2080 QualType Int32Ty =
2081 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
2082 Address ThreadIDTemp =
2083 CGF.CreateMemTempWithoutCast(T: Int32Ty, /*Name*/ ".threadid_temp.");
2084 CGF.EmitStoreOfScalar(value: ThreadID,
2085 lvalue: CGF.MakeAddrLValue(Addr: ThreadIDTemp, T: Int32Ty));
2086
2087 return ThreadIDTemp;
2088}
2089
2090llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
2091 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str();
2092 std::string Name = getName(Parts: {Prefix, "var"});
2093 llvm::GlobalVariable *GV =
2094 OMPBuilder.getOrCreateInternalVariable(Ty: KmpCriticalNameTy, Name);
2095 CGM.setDSOLocal(GV);
2096 return GV;
2097}
2098
2099namespace {
2100/// Common pre(post)-action for different OpenMP constructs.
2101class CommonActionTy final : public PrePostActionTy {
2102 llvm::FunctionCallee EnterCallee;
2103 ArrayRef<llvm::Value *> EnterArgs;
2104 llvm::FunctionCallee ExitCallee;
2105 ArrayRef<llvm::Value *> ExitArgs;
2106 bool Conditional;
2107 llvm::BasicBlock *ContBlock = nullptr;
2108
2109public:
2110 CommonActionTy(llvm::FunctionCallee EnterCallee,
2111 ArrayRef<llvm::Value *> EnterArgs,
2112 llvm::FunctionCallee ExitCallee,
2113 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false)
2114 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee),
2115 ExitArgs(ExitArgs), Conditional(Conditional) {}
2116 void Enter(CodeGenFunction &CGF) override {
2117 llvm::Value *EnterRes = CGF.EmitRuntimeCall(callee: EnterCallee, args: EnterArgs);
2118 if (Conditional) {
2119 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(Arg: EnterRes);
2120 auto *ThenBlock = CGF.createBasicBlock(name: "omp_if.then");
2121 ContBlock = CGF.createBasicBlock(name: "omp_if.end");
2122 // Generate the branch (If-stmt)
2123 CGF.Builder.CreateCondBr(Cond: CallBool, True: ThenBlock, False: ContBlock);
2124 CGF.EmitBlock(BB: ThenBlock);
2125 }
2126 }
2127 void Done(CodeGenFunction &CGF) {
2128 // Emit the rest of blocks/branches
2129 CGF.EmitBranch(Block: ContBlock);
2130 CGF.EmitBlock(BB: ContBlock, IsFinished: true);
2131 }
2132 void Exit(CodeGenFunction &CGF) override {
2133 CGF.EmitRuntimeCall(callee: ExitCallee, args: ExitArgs);
2134 }
2135};
2136} // anonymous namespace
2137
2138void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF,
2139 StringRef CriticalName,
2140 const RegionCodeGenTy &CriticalOpGen,
2141 SourceLocation Loc, const Expr *Hint) {
2142 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]);
2143 // CriticalOpGen();
2144 // __kmpc_end_critical(ident_t *, gtid, Lock);
2145 // Prepare arguments and build a call to __kmpc_critical
2146 if (!CGF.HaveInsertPoint())
2147 return;
2148 llvm::FunctionCallee RuntimeFcn = OMPBuilder.getOrCreateRuntimeFunction(
2149 M&: CGM.getModule(),
2150 FnID: Hint ? OMPRTL___kmpc_critical_with_hint : OMPRTL___kmpc_critical);
2151 llvm::Value *LockVar = getCriticalRegionLock(CriticalName);
2152 unsigned LockVarArgIdx = 2;
2153 if (cast<llvm::GlobalVariable>(Val: LockVar)->getAddressSpace() !=
2154 RuntimeFcn.getFunctionType()
2155 ->getParamType(i: LockVarArgIdx)
2156 ->getPointerAddressSpace())
2157 LockVar = CGF.Builder.CreateAddrSpaceCast(
2158 V: LockVar, DestTy: RuntimeFcn.getFunctionType()->getParamType(i: LockVarArgIdx));
2159 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2160 LockVar};
2161 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(arr&: Args),
2162 std::end(arr&: Args));
2163 if (Hint) {
2164 EnterArgs.push_back(Elt: CGF.Builder.CreateIntCast(
2165 V: CGF.EmitScalarExpr(E: Hint), DestTy: CGM.Int32Ty, /*isSigned=*/false));
2166 }
2167 CommonActionTy Action(RuntimeFcn, EnterArgs,
2168 OMPBuilder.getOrCreateRuntimeFunction(
2169 M&: CGM.getModule(), FnID: OMPRTL___kmpc_end_critical),
2170 Args);
2171 CriticalOpGen.setAction(Action);
2172 emitInlinedDirective(CGF, InnermostKind: OMPD_critical, CodeGen: CriticalOpGen);
2173}
2174
2175void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
2176 const RegionCodeGenTy &MasterOpGen,
2177 SourceLocation Loc) {
2178 if (!CGF.HaveInsertPoint())
2179 return;
2180 // if(__kmpc_master(ident_t *, gtid)) {
2181 // MasterOpGen();
2182 // __kmpc_end_master(ident_t *, gtid);
2183 // }
2184 // Prepare arguments and build a call to __kmpc_master
2185 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2186 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2187 M&: CGM.getModule(), FnID: OMPRTL___kmpc_master),
2188 Args,
2189 OMPBuilder.getOrCreateRuntimeFunction(
2190 M&: CGM.getModule(), FnID: OMPRTL___kmpc_end_master),
2191 Args,
2192 /*Conditional=*/true);
2193 MasterOpGen.setAction(Action);
2194 emitInlinedDirective(CGF, InnermostKind: OMPD_master, CodeGen: MasterOpGen);
2195 Action.Done(CGF);
2196}
2197
2198void CGOpenMPRuntime::emitMaskedRegion(CodeGenFunction &CGF,
2199 const RegionCodeGenTy &MaskedOpGen,
2200 SourceLocation Loc, const Expr *Filter) {
2201 if (!CGF.HaveInsertPoint())
2202 return;
2203 // if(__kmpc_masked(ident_t *, gtid, filter)) {
2204 // MaskedOpGen();
2205 // __kmpc_end_masked(iden_t *, gtid);
2206 // }
2207 // Prepare arguments and build a call to __kmpc_masked
2208 llvm::Value *FilterVal = Filter
2209 ? CGF.EmitScalarExpr(E: Filter, IgnoreResultAssign: CGF.Int32Ty)
2210 : llvm::ConstantInt::get(Ty: CGM.Int32Ty, /*V=*/0);
2211 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2212 FilterVal};
2213 llvm::Value *ArgsEnd[] = {emitUpdateLocation(CGF, Loc),
2214 getThreadID(CGF, Loc)};
2215 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2216 M&: CGM.getModule(), FnID: OMPRTL___kmpc_masked),
2217 Args,
2218 OMPBuilder.getOrCreateRuntimeFunction(
2219 M&: CGM.getModule(), FnID: OMPRTL___kmpc_end_masked),
2220 ArgsEnd,
2221 /*Conditional=*/true);
2222 MaskedOpGen.setAction(Action);
2223 emitInlinedDirective(CGF, InnermostKind: OMPD_masked, CodeGen: MaskedOpGen);
2224 Action.Done(CGF);
2225}
2226
2227void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
2228 SourceLocation Loc) {
2229 if (!CGF.HaveInsertPoint())
2230 return;
2231 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2232 OMPBuilder.createTaskyield(Loc: CGF.Builder);
2233 } else {
2234 // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
2235 llvm::Value *Args[] = {
2236 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2237 llvm::ConstantInt::get(Ty: CGM.IntTy, /*V=*/0, /*isSigned=*/IsSigned: true)};
2238 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2239 M&: CGM.getModule(), FnID: OMPRTL___kmpc_omp_taskyield),
2240 args: Args);
2241 }
2242
2243 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo))
2244 Region->emitUntiedSwitch(CGF);
2245}
2246
2247void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF,
2248 const RegionCodeGenTy &TaskgroupOpGen,
2249 SourceLocation Loc) {
2250 if (!CGF.HaveInsertPoint())
2251 return;
2252 // __kmpc_taskgroup(ident_t *, gtid);
2253 // TaskgroupOpGen();
2254 // __kmpc_end_taskgroup(ident_t *, gtid);
2255 // Prepare arguments and build a call to __kmpc_taskgroup
2256 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2257 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2258 M&: CGM.getModule(), FnID: OMPRTL___kmpc_taskgroup),
2259 Args,
2260 OMPBuilder.getOrCreateRuntimeFunction(
2261 M&: CGM.getModule(), FnID: OMPRTL___kmpc_end_taskgroup),
2262 Args);
2263 TaskgroupOpGen.setAction(Action);
2264 emitInlinedDirective(CGF, InnermostKind: OMPD_taskgroup, CodeGen: TaskgroupOpGen);
2265}
2266
2267/// Given an array of pointers to variables, project the address of a
2268/// given variable.
2269static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array,
2270 unsigned Index, const VarDecl *Var) {
2271 // Pull out the pointer to the variable.
2272 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Addr: Array, Index);
2273 llvm::Value *Ptr = CGF.Builder.CreateLoad(Addr: PtrAddr);
2274
2275 llvm::Type *ElemTy = CGF.ConvertTypeForMem(T: Var->getType());
2276 return Address(Ptr, ElemTy, CGF.getContext().getDeclAlign(D: Var));
2277}
2278
2279static llvm::Value *emitCopyprivateCopyFunction(
2280 CodeGenModule &CGM, llvm::Type *ArgsElemType,
2281 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
2282 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps,
2283 SourceLocation Loc) {
2284 ASTContext &C = CGM.getContext();
2285 // void copy_func(void *LHSArg, void *RHSArg);
2286
2287 auto *LHSArg =
2288 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
2289 T: C.VoidPtrTy, ParamKind: ImplicitParamKind::Other);
2290 auto *RHSArg =
2291 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
2292 T: C.VoidPtrTy, ParamKind: ImplicitParamKind::Other);
2293 FunctionArgList Args{LHSArg, RHSArg};
2294 const auto &CGFI =
2295 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
2296 std::string Name =
2297 CGM.getOpenMPRuntime().getName(Parts: {"omp", "copyprivate", "copy_func"});
2298 auto *Fn = llvm::Function::Create(Ty: CGM.getTypes().GetFunctionType(Info: CGFI),
2299 Linkage: llvm::GlobalValue::InternalLinkage, N: Name,
2300 M: &CGM.getModule());
2301 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: Fn, FI: CGFI);
2302 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
2303 Fn->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
2304 Fn->setDoesNotRecurse();
2305 CodeGenFunction CGF(CGM);
2306 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn, FnInfo: CGFI, Args, Loc, StartLoc: Loc);
2307 // Dest = (void*[n])(LHSArg);
2308 // Src = (void*[n])(RHSArg);
2309 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2310 V: CGF.Builder.CreateLoad(Addr: CGF.GetAddrOfLocalVar(VD: LHSArg)),
2311 DestTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
2312 ArgsElemType, CGF.getPointerAlign());
2313 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2314 V: CGF.Builder.CreateLoad(Addr: CGF.GetAddrOfLocalVar(VD: RHSArg)),
2315 DestTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
2316 ArgsElemType, CGF.getPointerAlign());
2317 // *(Type0*)Dst[0] = *(Type0*)Src[0];
2318 // *(Type1*)Dst[1] = *(Type1*)Src[1];
2319 // ...
2320 // *(Typen*)Dst[n] = *(Typen*)Src[n];
2321 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
2322 const auto *DestVar =
2323 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: DestExprs[I])->getDecl());
2324 Address DestAddr = emitAddrOfVarFromArray(CGF, Array: LHS, Index: I, Var: DestVar);
2325
2326 const auto *SrcVar =
2327 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: SrcExprs[I])->getDecl());
2328 Address SrcAddr = emitAddrOfVarFromArray(CGF, Array: RHS, Index: I, Var: SrcVar);
2329
2330 const auto *VD = cast<DeclRefExpr>(Val: CopyprivateVars[I])->getDecl();
2331 QualType Type = VD->getType();
2332 CGF.EmitOMPCopy(OriginalType: Type, DestAddr, SrcAddr, DestVD: DestVar, SrcVD: SrcVar, Copy: AssignmentOps[I]);
2333 }
2334 CGF.FinishFunction();
2335 return Fn;
2336}
2337
2338void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
2339 const RegionCodeGenTy &SingleOpGen,
2340 SourceLocation Loc,
2341 ArrayRef<const Expr *> CopyprivateVars,
2342 ArrayRef<const Expr *> SrcExprs,
2343 ArrayRef<const Expr *> DstExprs,
2344 ArrayRef<const Expr *> AssignmentOps) {
2345 if (!CGF.HaveInsertPoint())
2346 return;
2347 assert(CopyprivateVars.size() == SrcExprs.size() &&
2348 CopyprivateVars.size() == DstExprs.size() &&
2349 CopyprivateVars.size() == AssignmentOps.size());
2350 ASTContext &C = CGM.getContext();
2351 // int32 did_it = 0;
2352 // if(__kmpc_single(ident_t *, gtid)) {
2353 // SingleOpGen();
2354 // __kmpc_end_single(ident_t *, gtid);
2355 // did_it = 1;
2356 // }
2357 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
2358 // <copy_func>, did_it);
2359
2360 Address DidIt = Address::invalid();
2361 if (!CopyprivateVars.empty()) {
2362 // int32 did_it = 0;
2363 QualType KmpInt32Ty =
2364 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
2365 DidIt = CGF.CreateMemTempWithoutCast(T: KmpInt32Ty, Name: ".omp.copyprivate.did_it");
2366 CGF.Builder.CreateStore(Val: CGF.Builder.getInt32(C: 0), Addr: DidIt);
2367 }
2368 // Prepare arguments and build a call to __kmpc_single
2369 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2370 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2371 M&: CGM.getModule(), FnID: OMPRTL___kmpc_single),
2372 Args,
2373 OMPBuilder.getOrCreateRuntimeFunction(
2374 M&: CGM.getModule(), FnID: OMPRTL___kmpc_end_single),
2375 Args,
2376 /*Conditional=*/true);
2377 SingleOpGen.setAction(Action);
2378 emitInlinedDirective(CGF, InnermostKind: OMPD_single, CodeGen: SingleOpGen);
2379 if (DidIt.isValid()) {
2380 // did_it = 1;
2381 CGF.Builder.CreateStore(Val: CGF.Builder.getInt32(C: 1), Addr: DidIt);
2382 }
2383 Action.Done(CGF);
2384 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
2385 // <copy_func>, did_it);
2386 if (DidIt.isValid()) {
2387 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
2388 QualType CopyprivateArrayTy = C.getConstantArrayType(
2389 EltTy: C.VoidPtrTy, ArySize: ArraySize, SizeExpr: nullptr, ASM: ArraySizeModifier::Normal,
2390 /*IndexTypeQuals=*/0);
2391 // Create a list of all private variables for copyprivate.
2392 Address CopyprivateList = CGF.CreateMemTempWithoutCast(
2393 T: CopyprivateArrayTy, Name: ".omp.copyprivate.cpr_list");
2394 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
2395 Address Elem = CGF.Builder.CreateConstArrayGEP(Addr: CopyprivateList, Index: I);
2396 CGF.Builder.CreateStore(
2397 Val: CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2398 V: CGF.EmitLValue(E: CopyprivateVars[I]).getPointer(CGF),
2399 DestTy: CGF.VoidPtrTy),
2400 Addr: Elem);
2401 }
2402 // Build function that copies private values from single region to all other
2403 // threads in the corresponding parallel region.
2404 llvm::Value *CpyFn = emitCopyprivateCopyFunction(
2405 CGM, ArgsElemType: CGF.ConvertTypeForMem(T: CopyprivateArrayTy), CopyprivateVars,
2406 DestExprs: SrcExprs, SrcExprs: DstExprs, AssignmentOps, Loc);
2407 llvm::Value *BufSize = CGF.getTypeSize(Ty: CopyprivateArrayTy);
2408 Address CL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2409 Addr: CopyprivateList, Ty: CGF.VoidPtrTy, ElementTy: CGF.Int8Ty);
2410 llvm::Value *DidItVal = CGF.Builder.CreateLoad(Addr: DidIt);
2411 llvm::Value *Args[] = {
2412 emitUpdateLocation(CGF, Loc), // ident_t *<loc>
2413 getThreadID(CGF, Loc), // i32 <gtid>
2414 BufSize, // size_t <buf_size>
2415 CL.emitRawPointer(CGF), // void *<copyprivate list>
2416 CpyFn, // void (*) (void *, void *) <copy_func>
2417 DidItVal // i32 did_it
2418 };
2419 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2420 M&: CGM.getModule(), FnID: OMPRTL___kmpc_copyprivate),
2421 args: Args);
2422 }
2423}
2424
2425void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF,
2426 const RegionCodeGenTy &OrderedOpGen,
2427 SourceLocation Loc, bool IsThreads) {
2428 if (!CGF.HaveInsertPoint())
2429 return;
2430 // __kmpc_ordered(ident_t *, gtid);
2431 // OrderedOpGen();
2432 // __kmpc_end_ordered(ident_t *, gtid);
2433 // Prepare arguments and build a call to __kmpc_ordered
2434 if (IsThreads) {
2435 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2436 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2437 M&: CGM.getModule(), FnID: OMPRTL___kmpc_ordered),
2438 Args,
2439 OMPBuilder.getOrCreateRuntimeFunction(
2440 M&: CGM.getModule(), FnID: OMPRTL___kmpc_end_ordered),
2441 Args);
2442 OrderedOpGen.setAction(Action);
2443 emitInlinedDirective(CGF, InnermostKind: OMPD_ordered_blockassoc, CodeGen: OrderedOpGen);
2444 return;
2445 }
2446 emitInlinedDirective(CGF, InnermostKind: OMPD_ordered_blockassoc, CodeGen: OrderedOpGen);
2447}
2448
2449unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) {
2450 unsigned Flags;
2451 if (Kind == OMPD_for)
2452 Flags = OMP_IDENT_BARRIER_IMPL_FOR;
2453 else if (Kind == OMPD_sections)
2454 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS;
2455 else if (Kind == OMPD_single)
2456 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE;
2457 else if (Kind == OMPD_barrier)
2458 Flags = OMP_IDENT_BARRIER_EXPL;
2459 else
2460 Flags = OMP_IDENT_BARRIER_IMPL;
2461 return Flags;
2462}
2463
2464void CGOpenMPRuntime::getDefaultScheduleAndChunk(
2465 CodeGenFunction &CGF, const OMPLoopDirective &S,
2466 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const {
2467 // Check if the loop directive is actually a doacross loop directive. In this
2468 // case choose static, 1 schedule.
2469 if (llvm::any_of(
2470 Range: S.getClausesOfKind<OMPOrderedClause>(),
2471 P: [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) {
2472 ScheduleKind = OMPC_SCHEDULE_static;
2473 // Chunk size is 1 in this case.
2474 llvm::APInt ChunkSize(32, 1);
2475 ChunkExpr = IntegerLiteral::Create(
2476 C: CGF.getContext(), V: ChunkSize,
2477 type: CGF.getContext().getIntTypeForBitwidth(DestWidth: 32, /*Signed=*/0),
2478 l: SourceLocation());
2479 }
2480}
2481
2482void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
2483 OpenMPDirectiveKind Kind, bool EmitChecks,
2484 bool ForceSimpleCall) {
2485 // Check if we should use the OMPBuilder
2486 auto *OMPRegionInfo =
2487 dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo);
2488 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2489 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
2490 cantFail(ValOrErr: OMPBuilder.createBarrier(Loc: CGF.Builder, Kind, ForceSimpleCall,
2491 CheckCancelFlag: EmitChecks));
2492 CGF.Builder.restoreIP(IP: AfterIP);
2493 return;
2494 }
2495
2496 if (!CGF.HaveInsertPoint())
2497 return;
2498 // Build call __kmpc_cancel_barrier(loc, thread_id);
2499 // Build call __kmpc_barrier(loc, thread_id);
2500 unsigned Flags = getDefaultFlagsForBarriers(Kind);
2501 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc,
2502 // thread_id);
2503 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
2504 getThreadID(CGF, Loc)};
2505 if (OMPRegionInfo) {
2506 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) {
2507 llvm::Value *Result = CGF.EmitRuntimeCall(
2508 callee: OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(),
2509 FnID: OMPRTL___kmpc_cancel_barrier),
2510 args: Args);
2511 if (EmitChecks) {
2512 // if (__kmpc_cancel_barrier()) {
2513 // exit from construct;
2514 // }
2515 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(name: ".cancel.exit");
2516 llvm::BasicBlock *ContBB = CGF.createBasicBlock(name: ".cancel.continue");
2517 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Arg: Result);
2518 CGF.Builder.CreateCondBr(Cond: Cmp, True: ExitBB, False: ContBB);
2519 CGF.EmitBlock(BB: ExitBB);
2520 // exit from construct;
2521 CodeGenFunction::JumpDest CancelDestination =
2522 CGF.getOMPCancelDestination(Kind: OMPRegionInfo->getDirectiveKind());
2523 CGF.EmitBranchThroughCleanup(Dest: CancelDestination);
2524 CGF.EmitBlock(BB: ContBB, /*IsFinished=*/true);
2525 }
2526 return;
2527 }
2528 }
2529 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2530 M&: CGM.getModule(), FnID: OMPRTL___kmpc_barrier),
2531 args: Args);
2532}
2533
2534void CGOpenMPRuntime::emitErrorCall(CodeGenFunction &CGF, SourceLocation Loc,
2535 Expr *ME, bool IsFatal) {
2536 llvm::Value *MVL = ME ? CGF.EmitScalarExpr(E: ME)
2537 : llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
2538 // Build call void __kmpc_error(ident_t *loc, int severity, const char
2539 // *message)
2540 llvm::Value *Args[] = {
2541 emitUpdateLocation(CGF, Loc, /*Flags=*/0, /*GenLoc=*/EmitLoc: true),
2542 llvm::ConstantInt::get(Ty: CGM.Int32Ty, V: IsFatal ? 2 : 1),
2543 CGF.Builder.CreatePointerCast(V: MVL, DestTy: CGM.Int8PtrTy)};
2544 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2545 M&: CGM.getModule(), FnID: OMPRTL___kmpc_error),
2546 args: Args);
2547}
2548
2549/// Map the OpenMP loop schedule to the runtime enumeration.
2550static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
2551 bool Chunked, bool Ordered) {
2552 switch (ScheduleKind) {
2553 case OMPC_SCHEDULE_static:
2554 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked)
2555 : (Ordered ? OMP_ord_static : OMP_sch_static);
2556 case OMPC_SCHEDULE_dynamic:
2557 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked;
2558 case OMPC_SCHEDULE_guided:
2559 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked;
2560 case OMPC_SCHEDULE_runtime:
2561 return Ordered ? OMP_ord_runtime : OMP_sch_runtime;
2562 case OMPC_SCHEDULE_auto:
2563 return Ordered ? OMP_ord_auto : OMP_sch_auto;
2564 case OMPC_SCHEDULE_unknown:
2565 assert(!Chunked && "chunk was specified but schedule kind not known");
2566 return Ordered ? OMP_ord_static : OMP_sch_static;
2567 }
2568 llvm_unreachable("Unexpected runtime schedule");
2569}
2570
2571/// Map the OpenMP distribute schedule to the runtime enumeration.
2572static OpenMPSchedType
2573getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) {
2574 // only static is allowed for dist_schedule
2575 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static;
2576}
2577
2578bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
2579 bool Chunked) const {
2580 OpenMPSchedType Schedule =
2581 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
2582 return Schedule == OMP_sch_static;
2583}
2584
2585bool CGOpenMPRuntime::isStaticNonchunked(
2586 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
2587 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
2588 return Schedule == OMP_dist_sch_static;
2589}
2590
2591bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind,
2592 bool Chunked) const {
2593 OpenMPSchedType Schedule =
2594 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
2595 return Schedule == OMP_sch_static_chunked;
2596}
2597
2598bool CGOpenMPRuntime::isStaticChunked(
2599 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
2600 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
2601 return Schedule == OMP_dist_sch_static_chunked;
2602}
2603
2604bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
2605 OpenMPSchedType Schedule =
2606 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false);
2607 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
2608 return Schedule != OMP_sch_static;
2609}
2610
2611static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule,
2612 OpenMPScheduleClauseModifier M1,
2613 OpenMPScheduleClauseModifier M2) {
2614 int Modifier = 0;
2615 switch (M1) {
2616 case OMPC_SCHEDULE_MODIFIER_monotonic:
2617 Modifier = OMP_sch_modifier_monotonic;
2618 break;
2619 case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
2620 Modifier = OMP_sch_modifier_nonmonotonic;
2621 break;
2622 case OMPC_SCHEDULE_MODIFIER_simd:
2623 if (Schedule == OMP_sch_static_chunked)
2624 Schedule = OMP_sch_static_balanced_chunked;
2625 break;
2626 case OMPC_SCHEDULE_MODIFIER_last:
2627 case OMPC_SCHEDULE_MODIFIER_unknown:
2628 break;
2629 }
2630 switch (M2) {
2631 case OMPC_SCHEDULE_MODIFIER_monotonic:
2632 Modifier = OMP_sch_modifier_monotonic;
2633 break;
2634 case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
2635 Modifier = OMP_sch_modifier_nonmonotonic;
2636 break;
2637 case OMPC_SCHEDULE_MODIFIER_simd:
2638 if (Schedule == OMP_sch_static_chunked)
2639 Schedule = OMP_sch_static_balanced_chunked;
2640 break;
2641 case OMPC_SCHEDULE_MODIFIER_last:
2642 case OMPC_SCHEDULE_MODIFIER_unknown:
2643 break;
2644 }
2645 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription.
2646 // If the static schedule kind is specified or if the ordered clause is
2647 // specified, and if the nonmonotonic modifier is not specified, the effect is
2648 // as if the monotonic modifier is specified. Otherwise, unless the monotonic
2649 // modifier is specified, the effect is as if the nonmonotonic modifier is
2650 // specified.
2651 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) {
2652 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static ||
2653 Schedule == OMP_sch_static_balanced_chunked ||
2654 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static ||
2655 Schedule == OMP_dist_sch_static_chunked ||
2656 Schedule == OMP_dist_sch_static ||
2657 Schedule == OMP_dist_sch_static_chunked_sch_static_chunkone))
2658 Modifier = OMP_sch_modifier_nonmonotonic;
2659 }
2660 return Schedule | Modifier;
2661}
2662
2663void CGOpenMPRuntime::emitForDispatchInit(
2664 CodeGenFunction &CGF, SourceLocation Loc,
2665 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
2666 bool Ordered, const DispatchRTInput &DispatchValues) {
2667 if (!CGF.HaveInsertPoint())
2668 return;
2669 OpenMPSchedType Schedule = getRuntimeSchedule(
2670 ScheduleKind: ScheduleKind.Schedule, Chunked: DispatchValues.Chunk != nullptr, Ordered);
2671 assert(Ordered ||
2672 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked &&
2673 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked &&
2674 Schedule != OMP_sch_static_balanced_chunked));
2675 // Call __kmpc_dispatch_init(
2676 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
2677 // kmp_int[32|64] lower, kmp_int[32|64] upper,
2678 // kmp_int[32|64] stride, kmp_int[32|64] chunk);
2679
2680 // If the Chunk was not specified in the clause - use default value 1.
2681 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk
2682 : CGF.Builder.getIntN(N: IVSize, C: 1);
2683 llvm::Value *Args[] = {
2684 emitUpdateLocation(CGF, Loc),
2685 getThreadID(CGF, Loc),
2686 CGF.Builder.getInt32(C: addMonoNonMonoModifier(
2687 CGM, Schedule, M1: ScheduleKind.M1, M2: ScheduleKind.M2)), // Schedule type
2688 DispatchValues.LB, // Lower
2689 DispatchValues.UB, // Upper
2690 CGF.Builder.getIntN(N: IVSize, C: 1), // Stride
2691 Chunk // Chunk
2692 };
2693 CGF.EmitRuntimeCall(callee: OMPBuilder.createDispatchInitFunction(IVSize, IVSigned),
2694 args: Args);
2695}
2696
2697void CGOpenMPRuntime::emitForDispatchDeinit(CodeGenFunction &CGF,
2698 SourceLocation Loc) {
2699 if (!CGF.HaveInsertPoint())
2700 return;
2701 // Call __kmpc_dispatch_deinit(ident_t *loc, kmp_int32 tid);
2702 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2703 CGF.EmitRuntimeCall(callee: OMPBuilder.createDispatchDeinitFunction(), args: Args);
2704}
2705
2706static void emitForStaticInitCall(
2707 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId,
2708 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule,
2709 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2,
2710 const CGOpenMPRuntime::StaticRTInput &Values) {
2711 if (!CGF.HaveInsertPoint())
2712 return;
2713
2714 assert(!Values.Ordered);
2715 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked ||
2716 Schedule == OMP_sch_static_balanced_chunked ||
2717 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked ||
2718 Schedule == OMP_dist_sch_static ||
2719 Schedule == OMP_dist_sch_static_chunked ||
2720 Schedule == OMP_dist_sch_static_chunked_sch_static_chunkone);
2721
2722 // Call __kmpc_for_static_init(
2723 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
2724 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
2725 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
2726 // kmp_int[32|64] incr, kmp_int[32|64] chunk);
2727 llvm::Value *Chunk = Values.Chunk;
2728 if (Chunk == nullptr) {
2729 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static ||
2730 Schedule == OMP_dist_sch_static) &&
2731 "expected static non-chunked schedule");
2732 // If the Chunk was not specified in the clause - use default value 1.
2733 Chunk = CGF.Builder.getIntN(N: Values.IVSize, C: 1);
2734 } else {
2735 assert((Schedule == OMP_sch_static_chunked ||
2736 Schedule == OMP_sch_static_balanced_chunked ||
2737 Schedule == OMP_ord_static_chunked ||
2738 Schedule == OMP_dist_sch_static_chunked ||
2739 Schedule == OMP_dist_sch_static_chunked_sch_static_chunkone) &&
2740 "expected static chunked schedule");
2741 }
2742 llvm::Value *Args[] = {
2743 UpdateLocation,
2744 ThreadId,
2745 CGF.Builder.getInt32(C: addMonoNonMonoModifier(CGM&: CGF.CGM, Schedule, M1,
2746 M2)), // Schedule type
2747 Values.IL.emitRawPointer(CGF), // &isLastIter
2748 Values.LB.emitRawPointer(CGF), // &LB
2749 Values.UB.emitRawPointer(CGF), // &UB
2750 Values.ST.emitRawPointer(CGF), // &Stride
2751 CGF.Builder.getIntN(N: Values.IVSize, C: 1), // Incr
2752 Chunk // Chunk
2753 };
2754 CGF.EmitRuntimeCall(callee: ForStaticInitFunction, args: Args);
2755}
2756
2757void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF,
2758 SourceLocation Loc,
2759 OpenMPDirectiveKind DKind,
2760 const OpenMPScheduleTy &ScheduleKind,
2761 const StaticRTInput &Values) {
2762 OpenMPSchedType ScheduleNum =
2763 ScheduleKind.UseFusedDistChunkSchedule
2764 ? OMP_dist_sch_static_chunked_sch_static_chunkone
2765 : getRuntimeSchedule(ScheduleKind: ScheduleKind.Schedule, Chunked: Values.Chunk != nullptr,
2766 Ordered: Values.Ordered);
2767 assert((isOpenMPWorksharingDirective(DKind) || (DKind == OMPD_loop)) &&
2768 "Expected loop-based or sections-based directive.");
2769 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc,
2770 Flags: isOpenMPLoopDirective(DKind)
2771 ? OMP_IDENT_WORK_LOOP
2772 : OMP_IDENT_WORK_SECTIONS);
2773 llvm::Value *ThreadId = getThreadID(CGF, Loc);
2774 llvm::FunctionCallee StaticInitFunction =
2775 OMPBuilder.createForStaticInitFunction(IVSize: Values.IVSize, IVSigned: Values.IVSigned,
2776 IsGPUDistribute: false);
2777 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, TemporaryLocation: Loc);
2778 emitForStaticInitCall(CGF, UpdateLocation: UpdatedLocation, ThreadId, ForStaticInitFunction: StaticInitFunction,
2779 Schedule: ScheduleNum, M1: ScheduleKind.M1, M2: ScheduleKind.M2, Values);
2780}
2781
2782void CGOpenMPRuntime::emitDistributeStaticInit(
2783 CodeGenFunction &CGF, SourceLocation Loc,
2784 OpenMPDistScheduleClauseKind SchedKind,
2785 const CGOpenMPRuntime::StaticRTInput &Values) {
2786 OpenMPSchedType ScheduleNum =
2787 getRuntimeSchedule(ScheduleKind: SchedKind, Chunked: Values.Chunk != nullptr);
2788 llvm::Value *UpdatedLocation =
2789 emitUpdateLocation(CGF, Loc, Flags: OMP_IDENT_WORK_DISTRIBUTE);
2790 llvm::Value *ThreadId = getThreadID(CGF, Loc);
2791 llvm::FunctionCallee StaticInitFunction;
2792 bool isGPUDistribute =
2793 CGM.getLangOpts().OpenMPIsTargetDevice && CGM.getTriple().isGPU();
2794 StaticInitFunction = OMPBuilder.createForStaticInitFunction(
2795 IVSize: Values.IVSize, IVSigned: Values.IVSigned, IsGPUDistribute: isGPUDistribute);
2796
2797 emitForStaticInitCall(CGF, UpdateLocation: UpdatedLocation, ThreadId, ForStaticInitFunction: StaticInitFunction,
2798 Schedule: ScheduleNum, M1: OMPC_SCHEDULE_MODIFIER_unknown,
2799 M2: OMPC_SCHEDULE_MODIFIER_unknown, Values);
2800}
2801
2802void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF,
2803 SourceLocation Loc,
2804 OpenMPDirectiveKind DKind) {
2805 assert((DKind == OMPD_distribute || DKind == OMPD_for ||
2806 DKind == OMPD_sections) &&
2807 "Expected distribute, for, or sections directive kind");
2808 if (!CGF.HaveInsertPoint())
2809 return;
2810 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
2811 llvm::Value *Args[] = {
2812 emitUpdateLocation(CGF, Loc,
2813 Flags: isOpenMPDistributeDirective(DKind) ||
2814 (DKind == OMPD_target_teams_loop)
2815 ? OMP_IDENT_WORK_DISTRIBUTE
2816 : isOpenMPLoopDirective(DKind)
2817 ? OMP_IDENT_WORK_LOOP
2818 : OMP_IDENT_WORK_SECTIONS),
2819 getThreadID(CGF, Loc)};
2820 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, TemporaryLocation: Loc);
2821 if (isOpenMPDistributeDirective(DKind) &&
2822 CGM.getLangOpts().OpenMPIsTargetDevice && CGM.getTriple().isGPU())
2823 CGF.EmitRuntimeCall(
2824 callee: OMPBuilder.getOrCreateRuntimeFunction(
2825 M&: CGM.getModule(), FnID: OMPRTL___kmpc_distribute_static_fini),
2826 args: Args);
2827 else
2828 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2829 M&: CGM.getModule(), FnID: OMPRTL___kmpc_for_static_fini),
2830 args: Args);
2831}
2832
2833void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
2834 SourceLocation Loc,
2835 unsigned IVSize,
2836 bool IVSigned) {
2837 if (!CGF.HaveInsertPoint())
2838 return;
2839 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
2840 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2841 CGF.EmitRuntimeCall(callee: OMPBuilder.createDispatchFiniFunction(IVSize, IVSigned),
2842 args: Args);
2843}
2844
2845llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
2846 SourceLocation Loc, unsigned IVSize,
2847 bool IVSigned, Address IL,
2848 Address LB, Address UB,
2849 Address ST) {
2850 // Call __kmpc_dispatch_next(
2851 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
2852 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
2853 // kmp_int[32|64] *p_stride);
2854 llvm::Value *Args[] = {
2855 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2856 IL.emitRawPointer(CGF), // &isLastIter
2857 LB.emitRawPointer(CGF), // &Lower
2858 UB.emitRawPointer(CGF), // &Upper
2859 ST.emitRawPointer(CGF) // &Stride
2860 };
2861 llvm::Value *Call = CGF.EmitRuntimeCall(
2862 callee: OMPBuilder.createDispatchNextFunction(IVSize, IVSigned), args: Args);
2863 return CGF.EmitScalarConversion(
2864 Src: Call, SrcTy: CGF.getContext().getIntTypeForBitwidth(DestWidth: 32, /*Signed=*/1),
2865 DstTy: CGF.getContext().BoolTy, Loc);
2866}
2867
2868llvm::Value *CGOpenMPRuntime::emitMessageClause(CodeGenFunction &CGF,
2869 const Expr *Message,
2870 SourceLocation Loc) {
2871 if (!Message)
2872 return llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
2873 return CGF.EmitScalarExpr(E: Message);
2874}
2875
2876llvm::Value *
2877CGOpenMPRuntime::emitSeverityClause(OpenMPSeverityClauseKind Severity,
2878 SourceLocation Loc) {
2879 // OpenMP 6.0, 10.4: "If no severity clause is specified then the effect is
2880 // as if sev-level is fatal."
2881 return llvm::ConstantInt::get(Ty: CGM.Int32Ty,
2882 V: Severity == OMPC_SEVERITY_warning ? 1 : 2);
2883}
2884
2885void CGOpenMPRuntime::emitNumThreadsClause(
2886 CodeGenFunction &CGF, llvm::Value *NumThreads, SourceLocation Loc,
2887 OpenMPNumThreadsClauseModifier Modifier, OpenMPSeverityClauseKind Severity,
2888 SourceLocation SeverityLoc, const Expr *Message,
2889 SourceLocation MessageLoc) {
2890 if (!CGF.HaveInsertPoint())
2891 return;
2892 llvm::SmallVector<llvm::Value *, 4> Args(
2893 {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2894 CGF.Builder.CreateIntCast(V: NumThreads, DestTy: CGF.Int32Ty, /*isSigned*/ true)});
2895 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
2896 // or __kmpc_push_num_threads_strict(&loc, global_tid, num_threads, severity,
2897 // messsage) if strict modifier is used.
2898 RuntimeFunction FnID = OMPRTL___kmpc_push_num_threads;
2899 if (Modifier == OMPC_NUMTHREADS_strict) {
2900 FnID = OMPRTL___kmpc_push_num_threads_strict;
2901 Args.push_back(Elt: emitSeverityClause(Severity, Loc: SeverityLoc));
2902 Args.push_back(Elt: emitMessageClause(CGF, Message, Loc: MessageLoc));
2903 }
2904 CGF.EmitRuntimeCall(
2905 callee: OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(), FnID), args: Args);
2906}
2907
2908void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF,
2909 ProcBindKind ProcBind,
2910 SourceLocation Loc) {
2911 if (!CGF.HaveInsertPoint())
2912 return;
2913 assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value.");
2914 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind)
2915 llvm::Value *Args[] = {
2916 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2917 llvm::ConstantInt::get(Ty: CGM.IntTy, V: unsigned(ProcBind), /*isSigned=*/IsSigned: true)};
2918 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2919 M&: CGM.getModule(), FnID: OMPRTL___kmpc_push_proc_bind),
2920 args: Args);
2921}
2922
2923void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
2924 SourceLocation Loc, llvm::AtomicOrdering AO) {
2925 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2926 OMPBuilder.createFlush(Loc: CGF.Builder);
2927 } else {
2928 if (!CGF.HaveInsertPoint())
2929 return;
2930 // Build call void __kmpc_flush(ident_t *loc)
2931 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
2932 M&: CGM.getModule(), FnID: OMPRTL___kmpc_flush),
2933 args: emitUpdateLocation(CGF, Loc));
2934 }
2935}
2936
2937namespace {
2938/// Indexes of fields for type kmp_task_t.
2939enum KmpTaskTFields {
2940 /// List of shared variables.
2941 KmpTaskTShareds,
2942 /// Task routine.
2943 KmpTaskTRoutine,
2944 /// Partition id for the untied tasks.
2945 KmpTaskTPartId,
2946 /// Function with call of destructors for private variables.
2947 Data1,
2948 /// Task priority.
2949 Data2,
2950 /// (Taskloops only) Lower bound.
2951 KmpTaskTLowerBound,
2952 /// (Taskloops only) Upper bound.
2953 KmpTaskTUpperBound,
2954 /// (Taskloops only) Stride.
2955 KmpTaskTStride,
2956 /// (Taskloops only) Is last iteration flag.
2957 KmpTaskTLastIter,
2958 /// (Taskloops only) Reduction data.
2959 KmpTaskTReductions,
2960};
2961} // anonymous namespace
2962
2963void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() {
2964 // If we are in simd mode or there are no entries, we don't need to do
2965 // anything.
2966 if (CGM.getLangOpts().OpenMPSimd || OMPBuilder.OffloadInfoManager.empty())
2967 return;
2968
2969 llvm::OpenMPIRBuilder::EmitMetadataErrorReportFunctionTy &&ErrorReportFn =
2970 [this](llvm::OpenMPIRBuilder::EmitMetadataErrorKind Kind,
2971 const llvm::TargetRegionEntryInfo &EntryInfo) -> void {
2972 SourceLocation Loc;
2973 if (Kind != llvm::OpenMPIRBuilder::EMIT_MD_GLOBAL_VAR_LINK_ERROR) {
2974 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(),
2975 E = CGM.getContext().getSourceManager().fileinfo_end();
2976 I != E; ++I) {
2977 if (I->getFirst().getUniqueID().getDevice() == EntryInfo.DeviceID &&
2978 I->getFirst().getUniqueID().getFile() == EntryInfo.FileID) {
2979 Loc = CGM.getContext().getSourceManager().translateFileLineCol(
2980 SourceFile: I->getFirst(), Line: EntryInfo.Line, Col: 1);
2981 break;
2982 }
2983 }
2984 }
2985 switch (Kind) {
2986 case llvm::OpenMPIRBuilder::EMIT_MD_TARGET_REGION_ERROR: {
2987 CGM.getDiags().Report(Loc,
2988 DiagID: diag::err_target_region_offloading_entry_incorrect)
2989 << EntryInfo.ParentName;
2990 } break;
2991 case llvm::OpenMPIRBuilder::EMIT_MD_DECLARE_TARGET_ERROR: {
2992 CGM.getDiags().Report(
2993 Loc, DiagID: diag::err_target_var_offloading_entry_incorrect_with_parent)
2994 << EntryInfo.ParentName;
2995 } break;
2996 case llvm::OpenMPIRBuilder::EMIT_MD_GLOBAL_VAR_LINK_ERROR: {
2997 CGM.getDiags().Report(DiagID: diag::err_target_var_offloading_entry_incorrect);
2998 } break;
2999 case llvm::OpenMPIRBuilder::EMIT_MD_GLOBAL_VAR_INDIRECT_ERROR: {
3000 unsigned DiagID = CGM.getDiags().getCustomDiagID(
3001 L: DiagnosticsEngine::Error, FormatString: "Offloading entry for indirect declare "
3002 "target variable is incorrect: the "
3003 "address is invalid.");
3004 CGM.getDiags().Report(DiagID);
3005 } break;
3006 }
3007 };
3008
3009 OMPBuilder.createOffloadEntriesAndInfoMetadata(ErrorReportFunction&: ErrorReportFn);
3010}
3011
3012void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
3013 if (!KmpRoutineEntryPtrTy) {
3014 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
3015 ASTContext &C = CGM.getContext();
3016 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
3017 FunctionProtoType::ExtProtoInfo EPI;
3018 KmpRoutineEntryPtrQTy = C.getPointerType(
3019 T: C.getFunctionType(ResultTy: KmpInt32Ty, Args: KmpRoutineEntryTyArgs, EPI));
3020 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(T: KmpRoutineEntryPtrQTy);
3021 }
3022}
3023
3024namespace {
3025struct PrivateHelpersTy {
3026 PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original,
3027 const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit)
3028 : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy),
3029 PrivateElemInit(PrivateElemInit) {}
3030 PrivateHelpersTy(const VarDecl *Original) : Original(Original) {}
3031 const Expr *OriginalRef = nullptr;
3032 const VarDecl *Original = nullptr;
3033 const VarDecl *PrivateCopy = nullptr;
3034 const VarDecl *PrivateElemInit = nullptr;
3035 bool isLocalPrivate() const {
3036 return !OriginalRef && !PrivateCopy && !PrivateElemInit;
3037 }
3038};
3039typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
3040} // anonymous namespace
3041
3042/// For BindingDecls, returns the DecomposedDecl as the original VarDecl.
3043/// For regular VarDecls, returns the VarDecl itself.
3044static const VarDecl *getOriginalVarDecl(const ValueDecl *Decl) {
3045 if (const auto *BD = dyn_cast<BindingDecl>(Val: Decl))
3046 return cast<VarDecl>(Val: BD->getDecomposedDecl());
3047 return cast<VarDecl>(Val: Decl);
3048}
3049
3050static bool isAllocatableDecl(const VarDecl *VD) {
3051 const VarDecl *CVD = VD->getCanonicalDecl();
3052 if (!CVD->hasAttr<OMPAllocateDeclAttr>())
3053 return false;
3054 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
3055 // Use the default allocation.
3056 return !(AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc &&
3057 !AA->getAllocator());
3058}
3059
3060static RecordDecl *
3061createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) {
3062 if (!Privates.empty()) {
3063 ASTContext &C = CGM.getContext();
3064 // Build struct .kmp_privates_t. {
3065 // /* private vars */
3066 // };
3067 RecordDecl *RD = C.buildImplicitRecord(Name: ".kmp_privates.t");
3068 RD->startDefinition();
3069 for (const auto &Pair : Privates) {
3070 const VarDecl *VD = Pair.second.Original;
3071 const VarDecl *PrivateCopy = Pair.second.PrivateCopy;
3072 // For BindingDecls, use PrivateCopy type (binding's actual type).
3073 // For regular variables, use Original type to preserve qualifiers.
3074 // Check OriginalRef to detect BindingDecls since Original may be the
3075 // DecompositionDecl.
3076 bool IsBinding =
3077 Pair.second.OriginalRef &&
3078 isa<BindingDecl>(
3079 Val: cast<DeclRefExpr>(Val: Pair.second.OriginalRef)->getDecl());
3080 QualType Type = IsBinding ? PrivateCopy->getType().getNonReferenceType()
3081 : VD->getType().getNonReferenceType();
3082
3083 // If the private variable is a local variable with lvalue ref type,
3084 // allocate the pointer instead of the pointee type.
3085 if (Pair.second.isLocalPrivate()) {
3086 if (VD->getType()->isLValueReferenceType())
3087 Type = C.getPointerType(T: Type);
3088 if (isAllocatableDecl(VD))
3089 Type = C.getPointerType(T: Type);
3090 }
3091 FieldDecl *FD = addFieldToRecordDecl(C, DC: RD, FieldTy: Type);
3092 if (VD->hasAttrs()) {
3093 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()),
3094 E(VD->getAttrs().end());
3095 I != E; ++I)
3096 FD->addAttr(A: *I);
3097 }
3098 }
3099 RD->completeDefinition();
3100 return RD;
3101 }
3102 return nullptr;
3103}
3104
3105static RecordDecl *
3106createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind,
3107 QualType KmpInt32Ty,
3108 QualType KmpRoutineEntryPointerQTy) {
3109 ASTContext &C = CGM.getContext();
3110 // Build struct kmp_task_t {
3111 // void * shareds;
3112 // kmp_routine_entry_t routine;
3113 // kmp_int32 part_id;
3114 // kmp_cmplrdata_t data1;
3115 // kmp_cmplrdata_t data2;
3116 // For taskloops additional fields:
3117 // kmp_uint64 lb;
3118 // kmp_uint64 ub;
3119 // kmp_int64 st;
3120 // kmp_int32 liter;
3121 // void * reductions;
3122 // };
3123 RecordDecl *UD = C.buildImplicitRecord(Name: "kmp_cmplrdata_t", TK: TagTypeKind::Union);
3124 UD->startDefinition();
3125 addFieldToRecordDecl(C, DC: UD, FieldTy: KmpInt32Ty);
3126 addFieldToRecordDecl(C, DC: UD, FieldTy: KmpRoutineEntryPointerQTy);
3127 UD->completeDefinition();
3128 CanQualType KmpCmplrdataTy = C.getCanonicalTagType(TD: UD);
3129 RecordDecl *RD = C.buildImplicitRecord(Name: "kmp_task_t");
3130 RD->startDefinition();
3131 addFieldToRecordDecl(C, DC: RD, FieldTy: C.VoidPtrTy);
3132 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpRoutineEntryPointerQTy);
3133 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpInt32Ty);
3134 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpCmplrdataTy);
3135 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpCmplrdataTy);
3136 if (isOpenMPTaskLoopDirective(DKind: Kind)) {
3137 QualType KmpUInt64Ty =
3138 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
3139 QualType KmpInt64Ty =
3140 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
3141 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpUInt64Ty);
3142 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpUInt64Ty);
3143 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpInt64Ty);
3144 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpInt32Ty);
3145 addFieldToRecordDecl(C, DC: RD, FieldTy: C.VoidPtrTy);
3146 }
3147 RD->completeDefinition();
3148 return RD;
3149}
3150
3151static RecordDecl *
3152createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy,
3153 ArrayRef<PrivateDataTy> Privates) {
3154 ASTContext &C = CGM.getContext();
3155 // Build struct kmp_task_t_with_privates {
3156 // kmp_task_t task_data;
3157 // .kmp_privates_t. privates;
3158 // };
3159 RecordDecl *RD = C.buildImplicitRecord(Name: "kmp_task_t_with_privates");
3160 RD->startDefinition();
3161 addFieldToRecordDecl(C, DC: RD, FieldTy: KmpTaskTQTy);
3162 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates))
3163 addFieldToRecordDecl(C, DC: RD, FieldTy: C.getCanonicalTagType(TD: PrivateRD));
3164 RD->completeDefinition();
3165 return RD;
3166}
3167
3168/// Emit a proxy function which accepts kmp_task_t as the second
3169/// argument.
3170/// \code
3171/// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
3172/// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt,
3173/// For taskloops:
3174/// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
3175/// tt->reductions, tt->shareds);
3176/// return 0;
3177/// }
3178/// \endcode
3179static llvm::Function *
3180emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
3181 OpenMPDirectiveKind Kind, QualType KmpInt32Ty,
3182 QualType KmpTaskTWithPrivatesPtrQTy,
3183 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
3184 QualType SharedsPtrTy, llvm::Function *TaskFunction,
3185 llvm::Value *TaskPrivatesMap) {
3186 ASTContext &C = CGM.getContext();
3187 auto *GtidArg =
3188 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3189 T: KmpInt32Ty, ParamKind: ImplicitParamKind::Other);
3190 auto *TaskTypeArg = ImplicitParamDecl::Create(
3191 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3192 T: KmpTaskTWithPrivatesPtrQTy.withRestrict(), ParamKind: ImplicitParamKind::Other);
3193 FunctionArgList Args{GtidArg, TaskTypeArg};
3194 const auto &TaskEntryFnInfo =
3195 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: KmpInt32Ty, args: Args);
3196 llvm::FunctionType *TaskEntryTy =
3197 CGM.getTypes().GetFunctionType(Info: TaskEntryFnInfo);
3198 std::string Name = CGM.getOpenMPRuntime().getName(Parts: {"omp_task_entry", ""});
3199 auto *TaskEntry = llvm::Function::Create(
3200 Ty: TaskEntryTy, Linkage: llvm::GlobalValue::InternalLinkage, N: Name, M: &CGM.getModule());
3201 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: TaskEntry, FI: TaskEntryFnInfo);
3202 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3203 TaskEntry->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
3204 TaskEntry->setDoesNotRecurse();
3205 CodeGenFunction CGF(CGM);
3206 CGF.StartFunction(GD: GlobalDecl(), RetTy: KmpInt32Ty, Fn: TaskEntry, FnInfo: TaskEntryFnInfo, Args,
3207 Loc, StartLoc: Loc);
3208
3209 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map,
3210 // tt,
3211 // For taskloops:
3212 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
3213 // tt->task_data.shareds);
3214 llvm::Value *GtidParam = CGF.EmitLoadOfScalar(
3215 Addr: CGF.GetAddrOfLocalVar(VD: GtidArg), /*Volatile=*/false, Ty: KmpInt32Ty, Loc);
3216 LValue TDBase = CGF.EmitLoadOfPointerLValue(
3217 Ptr: CGF.GetAddrOfLocalVar(VD: TaskTypeArg),
3218 PtrTy: KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3219 const auto *KmpTaskTWithPrivatesQTyRD =
3220 KmpTaskTWithPrivatesQTy->castAsRecordDecl();
3221 LValue Base =
3222 CGF.EmitLValueForField(Base: TDBase, Field: *KmpTaskTWithPrivatesQTyRD->field_begin());
3223 const auto *KmpTaskTQTyRD = KmpTaskTQTy->castAsRecordDecl();
3224 auto PartIdFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTPartId);
3225 LValue PartIdLVal = CGF.EmitLValueForField(Base, Field: *PartIdFI);
3226 llvm::Value *PartidParam = PartIdLVal.getPointer(CGF);
3227
3228 auto SharedsFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTShareds);
3229 LValue SharedsLVal = CGF.EmitLValueForField(Base, Field: *SharedsFI);
3230 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3231 V: CGF.EmitLoadOfScalar(lvalue: SharedsLVal, Loc),
3232 DestTy: CGF.ConvertTypeForMem(T: SharedsPtrTy));
3233
3234 auto PrivatesFI = std::next(x: KmpTaskTWithPrivatesQTyRD->field_begin(), n: 1);
3235 llvm::Value *PrivatesParam;
3236 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) {
3237 LValue PrivatesLVal = CGF.EmitLValueForField(Base: TDBase, Field: *PrivatesFI);
3238 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3239 V: PrivatesLVal.getPointer(CGF), DestTy: CGF.VoidPtrTy);
3240 } else {
3241 PrivatesParam = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
3242 }
3243
3244 llvm::Value *CommonArgs[] = {
3245 GtidParam, PartidParam, PrivatesParam, TaskPrivatesMap,
3246 CGF.Builder
3247 .CreatePointerBitCastOrAddrSpaceCast(Addr: TDBase.getAddress(),
3248 Ty: CGF.VoidPtrTy, ElementTy: CGF.Int8Ty)
3249 .emitRawPointer(CGF)};
3250 SmallVector<llvm::Value *, 16> CallArgs(std::begin(arr&: CommonArgs),
3251 std::end(arr&: CommonArgs));
3252 if (isOpenMPTaskLoopDirective(DKind: Kind)) {
3253 auto LBFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTLowerBound);
3254 LValue LBLVal = CGF.EmitLValueForField(Base, Field: *LBFI);
3255 llvm::Value *LBParam = CGF.EmitLoadOfScalar(lvalue: LBLVal, Loc);
3256 auto UBFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTUpperBound);
3257 LValue UBLVal = CGF.EmitLValueForField(Base, Field: *UBFI);
3258 llvm::Value *UBParam = CGF.EmitLoadOfScalar(lvalue: UBLVal, Loc);
3259 auto StFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTStride);
3260 LValue StLVal = CGF.EmitLValueForField(Base, Field: *StFI);
3261 llvm::Value *StParam = CGF.EmitLoadOfScalar(lvalue: StLVal, Loc);
3262 auto LIFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTLastIter);
3263 LValue LILVal = CGF.EmitLValueForField(Base, Field: *LIFI);
3264 llvm::Value *LIParam = CGF.EmitLoadOfScalar(lvalue: LILVal, Loc);
3265 auto RFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTReductions);
3266 LValue RLVal = CGF.EmitLValueForField(Base, Field: *RFI);
3267 llvm::Value *RParam = CGF.EmitLoadOfScalar(lvalue: RLVal, Loc);
3268 CallArgs.push_back(Elt: LBParam);
3269 CallArgs.push_back(Elt: UBParam);
3270 CallArgs.push_back(Elt: StParam);
3271 CallArgs.push_back(Elt: LIParam);
3272 CallArgs.push_back(Elt: RParam);
3273 }
3274 CallArgs.push_back(Elt: SharedsParam);
3275
3276 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, OutlinedFn: TaskFunction,
3277 Args: CallArgs);
3278 CGF.EmitStoreThroughLValue(Src: RValue::get(V: CGF.Builder.getInt32(/*C=*/0)),
3279 Dst: CGF.MakeAddrLValue(Addr: CGF.ReturnValue, T: KmpInt32Ty));
3280 CGF.FinishFunction();
3281 return TaskEntry;
3282}
3283
3284static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM,
3285 SourceLocation Loc,
3286 QualType KmpInt32Ty,
3287 QualType KmpTaskTWithPrivatesPtrQTy,
3288 QualType KmpTaskTWithPrivatesQTy) {
3289 ASTContext &C = CGM.getContext();
3290 auto *GtidArg =
3291 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3292 T: KmpInt32Ty, ParamKind: ImplicitParamKind::Other);
3293 auto *TaskTypeArg = ImplicitParamDecl::Create(
3294 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3295 T: KmpTaskTWithPrivatesPtrQTy.withRestrict(), ParamKind: ImplicitParamKind::Other);
3296 FunctionArgList Args{GtidArg, TaskTypeArg};
3297 const auto &DestructorFnInfo =
3298 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: KmpInt32Ty, args: Args);
3299 llvm::FunctionType *DestructorFnTy =
3300 CGM.getTypes().GetFunctionType(Info: DestructorFnInfo);
3301 std::string Name =
3302 CGM.getOpenMPRuntime().getName(Parts: {"omp_task_destructor", ""});
3303 auto *DestructorFn =
3304 llvm::Function::Create(Ty: DestructorFnTy, Linkage: llvm::GlobalValue::InternalLinkage,
3305 N: Name, M: &CGM.getModule());
3306 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: DestructorFn,
3307 FI: DestructorFnInfo);
3308 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3309 DestructorFn->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
3310 DestructorFn->setDoesNotRecurse();
3311 CodeGenFunction CGF(CGM);
3312 CGF.StartFunction(GD: GlobalDecl(), RetTy: KmpInt32Ty, Fn: DestructorFn, FnInfo: DestructorFnInfo,
3313 Args, Loc, StartLoc: Loc);
3314
3315 LValue Base = CGF.EmitLoadOfPointerLValue(
3316 Ptr: CGF.GetAddrOfLocalVar(VD: TaskTypeArg),
3317 PtrTy: KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3318 const auto *KmpTaskTWithPrivatesQTyRD =
3319 KmpTaskTWithPrivatesQTy->castAsRecordDecl();
3320 auto FI = std::next(x: KmpTaskTWithPrivatesQTyRD->field_begin());
3321 Base = CGF.EmitLValueForField(Base, Field: *FI);
3322 for (const auto *Field : FI->getType()->castAsRecordDecl()->fields()) {
3323 if (QualType::DestructionKind DtorKind =
3324 Field->getType().isDestructedType()) {
3325 LValue FieldLValue = CGF.EmitLValueForField(Base, Field);
3326 CGF.pushDestroy(dtorKind: DtorKind, addr: FieldLValue.getAddress(), type: Field->getType());
3327 }
3328 }
3329 CGF.FinishFunction();
3330 return DestructorFn;
3331}
3332
3333/// Emit a privates mapping function for correct handling of private and
3334/// firstprivate variables.
3335/// \code
3336/// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1>
3337/// **noalias priv1,..., <tyn> **noalias privn) {
3338/// *priv1 = &.privates.priv1;
3339/// ...;
3340/// *privn = &.privates.privn;
3341/// }
3342/// \endcode
3343static llvm::Value *
3344emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc,
3345 const OMPTaskDataTy &Data, QualType PrivatesQTy,
3346 ArrayRef<PrivateDataTy> Privates) {
3347 ASTContext &C = CGM.getContext();
3348 FunctionArgList Args;
3349 auto *TaskPrivatesArg = ImplicitParamDecl::Create(
3350 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3351 T: C.getPointerType(T: PrivatesQTy).withConst().withRestrict(),
3352 ParamKind: ImplicitParamKind::Other);
3353 Args.push_back(Elt: TaskPrivatesArg);
3354 llvm::SmallDenseMap<CanonicalDeclPtr<const VarDecl>, unsigned> PrivateVarsPos;
3355 // Track BindingDecl positions separately since BindingDecl is not a VarDecl.
3356 llvm::SmallDenseMap<const BindingDecl *, unsigned> BindingDeclPos;
3357 unsigned Counter = 1;
3358 for (const Expr *E : Data.PrivateVars) {
3359 Args.push_back(Elt: ImplicitParamDecl::Create(
3360 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3361 T: C.getPointerType(T: C.getPointerType(T: E->getType()))
3362 .withConst()
3363 .withRestrict(),
3364 ParamKind: ImplicitParamKind::Other));
3365 const ValueDecl *VD = cast<DeclRefExpr>(Val: E)->getDecl();
3366 if (const auto *BD = dyn_cast<BindingDecl>(Val: VD))
3367 BindingDeclPos[cast<BindingDecl>(Val: BD->getCanonicalDecl())] = Counter;
3368 else
3369 PrivateVarsPos[cast<VarDecl>(Val: VD)] = Counter;
3370 ++Counter;
3371 }
3372 for (const Expr *E : Data.FirstprivateVars) {
3373 Args.push_back(Elt: ImplicitParamDecl::Create(
3374 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3375 T: C.getPointerType(T: C.getPointerType(T: E->getType()))
3376 .withConst()
3377 .withRestrict(),
3378 ParamKind: ImplicitParamKind::Other));
3379 const ValueDecl *VD = cast<DeclRefExpr>(Val: E)->getDecl();
3380 if (const auto *BD = dyn_cast<BindingDecl>(Val: VD))
3381 BindingDeclPos[cast<BindingDecl>(Val: BD->getCanonicalDecl())] = Counter;
3382 else
3383 PrivateVarsPos[cast<VarDecl>(Val: VD)] = Counter;
3384 ++Counter;
3385 }
3386 for (const Expr *E : Data.LastprivateVars) {
3387 Args.push_back(Elt: ImplicitParamDecl::Create(
3388 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3389 T: C.getPointerType(T: C.getPointerType(T: E->getType()))
3390 .withConst()
3391 .withRestrict(),
3392 ParamKind: ImplicitParamKind::Other));
3393 const ValueDecl *VD = cast<DeclRefExpr>(Val: E)->getDecl();
3394 if (const auto *BD = dyn_cast<BindingDecl>(Val: VD))
3395 BindingDeclPos[cast<BindingDecl>(Val: BD->getCanonicalDecl())] = Counter;
3396 else
3397 PrivateVarsPos[cast<VarDecl>(Val: VD)] = Counter;
3398 ++Counter;
3399 }
3400 for (const VarDecl *VD : Data.PrivateLocals) {
3401 QualType Ty = VD->getType().getNonReferenceType();
3402 if (VD->getType()->isLValueReferenceType())
3403 Ty = C.getPointerType(T: Ty);
3404 if (isAllocatableDecl(VD))
3405 Ty = C.getPointerType(T: Ty);
3406 Args.push_back(Elt: ImplicitParamDecl::Create(
3407 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
3408 T: C.getPointerType(T: C.getPointerType(T: Ty)).withConst().withRestrict(),
3409 ParamKind: ImplicitParamKind::Other));
3410 PrivateVarsPos[VD] = Counter;
3411 ++Counter;
3412 }
3413 const auto &TaskPrivatesMapFnInfo =
3414 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
3415 llvm::FunctionType *TaskPrivatesMapTy =
3416 CGM.getTypes().GetFunctionType(Info: TaskPrivatesMapFnInfo);
3417 std::string Name =
3418 CGM.getOpenMPRuntime().getName(Parts: {"omp_task_privates_map", ""});
3419 auto *TaskPrivatesMap = llvm::Function::Create(
3420 Ty: TaskPrivatesMapTy, Linkage: llvm::GlobalValue::InternalLinkage, N: Name,
3421 M: &CGM.getModule());
3422 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: TaskPrivatesMap,
3423 FI: TaskPrivatesMapFnInfo);
3424 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3425 TaskPrivatesMap->addFnAttr(Kind: "sample-profile-suffix-elision-policy",
3426 Val: "selected");
3427 if (CGM.getCodeGenOpts().OptimizationLevel != 0) {
3428 TaskPrivatesMap->removeFnAttr(Kind: llvm::Attribute::NoInline);
3429 TaskPrivatesMap->removeFnAttr(Kind: llvm::Attribute::OptimizeNone);
3430 TaskPrivatesMap->addFnAttr(Kind: llvm::Attribute::AlwaysInline);
3431 }
3432 CodeGenFunction CGF(CGM);
3433 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn: TaskPrivatesMap,
3434 FnInfo: TaskPrivatesMapFnInfo, Args, Loc, StartLoc: Loc);
3435
3436 // *privi = &.privates.privi;
3437 LValue Base = CGF.EmitLoadOfPointerLValue(
3438 Ptr: CGF.GetAddrOfLocalVar(VD: TaskPrivatesArg),
3439 PtrTy: TaskPrivatesArg->getType()->castAs<PointerType>());
3440 const auto *PrivatesQTyRD = PrivatesQTy->castAsRecordDecl();
3441 Counter = 0;
3442 for (const FieldDecl *Field : PrivatesQTyRD->fields()) {
3443 LValue FieldLVal = CGF.EmitLValueForField(Base, Field);
3444 // Lookup by the original declaration (BindingDecl or VarDecl).
3445 const ValueDecl *LookupVD;
3446 if (Privates[Counter].second.OriginalRef) {
3447 LookupVD =
3448 cast<DeclRefExpr>(Val: Privates[Counter].second.OriginalRef)->getDecl();
3449 } else {
3450 LookupVD = Privates[Counter].second.Original;
3451 }
3452
3453 // For BindingDecls, the privates record now stores each binding's type
3454 // directly (not the full DecompositionDecl), so FieldLVal is already
3455 // correct.
3456
3457 unsigned Position;
3458 if (const auto *BD = dyn_cast<BindingDecl>(Val: LookupVD)) {
3459 Position =
3460 BindingDeclPos.lookup(Val: cast<BindingDecl>(Val: BD->getCanonicalDecl()));
3461 assert(Position && "binding not in privates mapping");
3462 } else {
3463 Position = PrivateVarsPos[cast<VarDecl>(Val: LookupVD)];
3464 }
3465 const VarDecl *VD = Args[Position];
3466 LValue RefLVal =
3467 CGF.MakeAddrLValue(Addr: CGF.GetAddrOfLocalVar(VD), T: VD->getType());
3468 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue(
3469 Ptr: RefLVal.getAddress(), PtrTy: RefLVal.getType()->castAs<PointerType>());
3470 CGF.EmitStoreOfScalar(value: FieldLVal.getPointer(CGF), lvalue: RefLoadLVal);
3471 ++Counter;
3472 }
3473 CGF.FinishFunction();
3474 return TaskPrivatesMap;
3475}
3476
3477/// Emit initialization for private variables in task-based directives.
3478static void emitPrivatesInit(CodeGenFunction &CGF,
3479 const OMPExecutableDirective &D,
3480 Address KmpTaskSharedsPtr, LValue TDBase,
3481 const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3482 QualType SharedsTy, QualType SharedsPtrTy,
3483 const OMPTaskDataTy &Data,
3484 ArrayRef<PrivateDataTy> Privates, bool ForDup) {
3485 ASTContext &C = CGF.getContext();
3486 auto FI = std::next(x: KmpTaskTWithPrivatesQTyRD->field_begin());
3487 LValue PrivatesBase = CGF.EmitLValueForField(Base: TDBase, Field: *FI);
3488 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(DKind: D.getDirectiveKind())
3489 ? OMPD_taskloop
3490 : OMPD_task;
3491 const CapturedStmt &CS = *D.getCapturedStmt(RegionKind: Kind);
3492 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS);
3493 LValue SrcBase;
3494 bool IsTargetTask =
3495 isOpenMPTargetDataManagementDirective(DKind: D.getDirectiveKind()) ||
3496 isOpenMPTargetExecutionDirective(DKind: D.getDirectiveKind());
3497 // For target-based directives skip 4 firstprivate arrays BasePointersArray,
3498 // PointersArray, SizesArray, and MappersArray. The original variables for
3499 // these arrays are not captured and we get their addresses explicitly.
3500 if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) ||
3501 (IsTargetTask && KmpTaskSharedsPtr.isValid())) {
3502 SrcBase = CGF.MakeAddrLValue(
3503 Addr: CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3504 Addr: KmpTaskSharedsPtr, Ty: CGF.ConvertTypeForMem(T: SharedsPtrTy),
3505 ElementTy: CGF.ConvertTypeForMem(T: SharedsTy)),
3506 T: SharedsTy);
3507 }
3508 FI = FI->getType()->castAsRecordDecl()->field_begin();
3509 for (const PrivateDataTy &Pair : Privates) {
3510 // Do not initialize private locals.
3511 if (Pair.second.isLocalPrivate()) {
3512 ++FI;
3513 continue;
3514 }
3515 const VarDecl *VD = Pair.second.PrivateCopy;
3516 const Expr *Init = VD->getAnyInitializer();
3517 if (Init && (!ForDup || (isa<CXXConstructExpr>(Val: Init) &&
3518 !CGF.isTrivialInitializer(Init)))) {
3519 LValue PrivateLValue = CGF.EmitLValueForField(Base: PrivatesBase, Field: *FI);
3520 if (const VarDecl *Elem = Pair.second.PrivateElemInit) {
3521 const VarDecl *OriginalVD = Pair.second.Original;
3522 // Check if the variable is the target-based BasePointersArray,
3523 // PointersArray, SizesArray, or MappersArray.
3524 LValue SharedRefLValue;
3525 QualType Type = PrivateLValue.getType();
3526 const FieldDecl *SharedField = CapturesInfo.lookup(VD: OriginalVD);
3527 if (IsTargetTask && !SharedField) {
3528 assert(isa<ImplicitParamDecl>(OriginalVD) &&
3529 isa<CapturedDecl>(OriginalVD->getDeclContext()) &&
3530 cast<CapturedDecl>(OriginalVD->getDeclContext())
3531 ->getNumParams() == 0 &&
3532 isa<TranslationUnitDecl>(
3533 cast<CapturedDecl>(OriginalVD->getDeclContext())
3534 ->getDeclContext()) &&
3535 "Expected artificial target data variable.");
3536 SharedRefLValue =
3537 CGF.MakeAddrLValue(Addr: CGF.GetAddrOfLocalVar(VD: OriginalVD), T: Type);
3538 } else if (ForDup) {
3539 SharedRefLValue = CGF.EmitLValueForField(Base: SrcBase, Field: SharedField);
3540 // For BindingDecls, access the specific binding field within the
3541 // captured DecompositionDecl.
3542 if (Pair.second.OriginalRef) {
3543 if (const auto *DRE =
3544 dyn_cast<DeclRefExpr>(Val: Pair.second.OriginalRef)) {
3545 if (const auto *BD = dyn_cast<BindingDecl>(Val: DRE->getDecl())) {
3546 // Emit the binding subobject (member or array element) with
3547 // the decomposed decl temporarily mapped to the capture.
3548 const VarDecl *DD = cast<VarDecl>(Val: BD->getDecomposedDecl());
3549 auto It = CGF.findLocalDecl(D: DD);
3550 bool WasMapped = It != CGF.localDeclMapEnd();
3551 Address Saved = WasMapped ? It->second : Address::invalid();
3552 if (WasMapped)
3553 It->second = SharedRefLValue.getAddress();
3554 else
3555 CGF.insertLocalDecl(D: DD, Addr: SharedRefLValue.getAddress());
3556 SharedRefLValue = CGF.EmitLValue(E: BD->getBinding());
3557 if (WasMapped) {
3558 auto RestoreIt = CGF.findLocalDecl(D: DD);
3559 RestoreIt->second = Saved;
3560 } else {
3561 CGF.eraseLocalDecl(D: DD);
3562 }
3563 }
3564 }
3565 }
3566 bool IsBinding =
3567 Pair.second.OriginalRef &&
3568 isa<BindingDecl>(
3569 Val: cast<DeclRefExpr>(Val: Pair.second.OriginalRef)->getDecl());
3570 SharedRefLValue = CGF.MakeAddrLValue(
3571 Addr: SharedRefLValue.getAddress().withAlignment(
3572 NewAlignment: IsBinding ? C.toCharUnitsFromBits(
3573 BitSize: C.getTypeAlign(T: SharedRefLValue.getType()))
3574 : C.getDeclAlign(D: OriginalVD)),
3575 T: SharedRefLValue.getType(), BaseInfo: LValueBaseInfo(AlignmentSource::Decl),
3576 TBAAInfo: SharedRefLValue.getTBAAInfo());
3577 } else if (CGF.LambdaCaptureFields.count(
3578 Val: Pair.second.Original->getCanonicalDecl()) > 0 ||
3579 isa_and_nonnull<BlockDecl>(Val: CGF.CurCodeDecl)) {
3580 SharedRefLValue = CGF.EmitLValue(E: Pair.second.OriginalRef);
3581 } else {
3582 // Processing for implicitly captured variables.
3583 InlinedOpenMPRegionRAII Region(
3584 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown,
3585 /*HasCancel=*/false, /*NoInheritance=*/true);
3586 SharedRefLValue = CGF.EmitLValue(E: Pair.second.OriginalRef);
3587 }
3588 if (Type->isArrayType()) {
3589 // Initialize firstprivate array.
3590 if (!isa<CXXConstructExpr>(Val: Init) || CGF.isTrivialInitializer(Init)) {
3591 // Perform simple memcpy.
3592 CGF.EmitAggregateAssign(Dest: PrivateLValue, Src: SharedRefLValue, EltTy: Type);
3593 } else {
3594 // Initialize firstprivate array using element-by-element
3595 // initialization.
3596 CGF.EmitOMPAggregateAssign(
3597 DestAddr: PrivateLValue.getAddress(), SrcAddr: SharedRefLValue.getAddress(), OriginalType: Type,
3598 CopyGen: [&CGF, Elem, Init, &CapturesInfo](Address DestElement,
3599 Address SrcElement) {
3600 // Clean up any temporaries needed by the initialization.
3601 CodeGenFunction::OMPPrivateScope InitScope(CGF);
3602 InitScope.addPrivate(LocalVD: Elem, Addr: SrcElement);
3603 (void)InitScope.Privatize();
3604 // Emit initialization for single element.
3605 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(
3606 CGF, &CapturesInfo);
3607 CGF.EmitAnyExprToMem(E: Init, Location: DestElement,
3608 Quals: Init->getType().getQualifiers(),
3609 /*IsInitializer=*/false);
3610 });
3611 }
3612 } else {
3613 CodeGenFunction::OMPPrivateScope InitScope(CGF);
3614 InitScope.addPrivate(LocalVD: Elem, Addr: SharedRefLValue.getAddress());
3615 (void)InitScope.Privatize();
3616 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo);
3617 CGF.EmitExprAsInit(init: Init, D: VD, lvalue: PrivateLValue,
3618 /*capturedByInit=*/false);
3619 }
3620 } else {
3621 CGF.EmitExprAsInit(init: Init, D: VD, lvalue: PrivateLValue, /*capturedByInit=*/false);
3622 }
3623 } else if (const VarDecl *OriginalVD = Pair.second.Original) {
3624 // Handle array bindings without initializers (firstprivate only).
3625 // For private clause (PrivateElemInit is nullptr), skip initialization.
3626 // Check if OriginalRef is a BindingDecl with array type.
3627 const BindingDecl *BD = nullptr;
3628 if (Pair.second.OriginalRef) {
3629 if (const auto *DRE = dyn_cast<DeclRefExpr>(Val: Pair.second.OriginalRef)) {
3630 BD = dyn_cast<BindingDecl>(Val: DRE->getDecl());
3631 }
3632 }
3633 if (BD && BD->getType()->isArrayType() && Pair.second.PrivateElemInit) {
3634 LValue PrivateLValue = CGF.EmitLValueForField(Base: PrivatesBase, Field: *FI);
3635 QualType Type = PrivateLValue.getType();
3636 const FieldDecl *SharedField = CapturesInfo.lookup(VD: OriginalVD);
3637
3638 // Lambda to temporarily map DecompositionDecl and emit the binding
3639 // expression.
3640 auto EmitBindingWithTempMap = [&CGF](const BindingDecl *BD,
3641 Address DDAddr) -> LValue {
3642 const VarDecl *DD = cast<VarDecl>(Val: BD->getDecomposedDecl());
3643 auto It = CGF.findLocalDecl(D: DD);
3644 bool WasMapped = It != CGF.localDeclMapEnd();
3645 Address Saved = WasMapped ? It->second : Address::invalid();
3646 if (WasMapped)
3647 It->second = DDAddr;
3648 else
3649 CGF.insertLocalDecl(D: DD, Addr: DDAddr);
3650
3651 LValue Result = CGF.EmitLValue(E: BD->getBinding());
3652
3653 // Restore the mapping
3654 if (WasMapped) {
3655 auto RestoreIt = CGF.findLocalDecl(D: DD);
3656 RestoreIt->second = Saved;
3657 } else {
3658 CGF.eraseLocalDecl(D: DD);
3659 }
3660
3661 return Result;
3662 };
3663
3664 LValue SharedRefLValue;
3665 if (ForDup) {
3666 SharedRefLValue = CGF.EmitLValueForField(Base: SrcBase, Field: SharedField);
3667 SharedRefLValue =
3668 EmitBindingWithTempMap(BD, SharedRefLValue.getAddress());
3669 SharedRefLValue = CGF.MakeAddrLValue(
3670 Addr: SharedRefLValue.getAddress().withAlignment(
3671 NewAlignment: C.getDeclAlign(D: OriginalVD)),
3672 T: SharedRefLValue.getType(), BaseInfo: LValueBaseInfo(AlignmentSource::Decl),
3673 TBAAInfo: SharedRefLValue.getTBAAInfo());
3674 } else {
3675 // For !ForDup (first task), emit binding from parent scope using
3676 // InlinedOpenMPRegionRAII to access the correct scope
3677 InlinedOpenMPRegionRAII Region(
3678 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown,
3679 /*HasCancel=*/false, /*NoInheritance=*/true);
3680
3681 // Get the decomposed decl address from parent scope
3682 const VarDecl *DD = cast<VarDecl>(Val: BD->getDecomposedDecl());
3683 Address DDAddr = CGF.GetAddrOfLocalVar(VD: DD);
3684 SharedRefLValue = EmitBindingWithTempMap(BD, DDAddr);
3685 }
3686 // Perform simple memcpy for array binding
3687 CGF.EmitAggregateAssign(Dest: PrivateLValue, Src: SharedRefLValue, EltTy: Type);
3688 }
3689 }
3690 ++FI;
3691 }
3692}
3693
3694/// Check if duplication function is required for taskloops.
3695static bool checkInitIsRequired(CodeGenFunction &CGF,
3696 ArrayRef<PrivateDataTy> Privates) {
3697 bool InitRequired = false;
3698 for (const PrivateDataTy &Pair : Privates) {
3699 if (Pair.second.isLocalPrivate())
3700 continue;
3701 const VarDecl *VD = Pair.second.PrivateCopy;
3702 const Expr *Init = VD->getAnyInitializer();
3703 InitRequired = InitRequired || (isa_and_nonnull<CXXConstructExpr>(Val: Init) &&
3704 !CGF.isTrivialInitializer(Init));
3705 if (InitRequired)
3706 break;
3707 }
3708 return InitRequired;
3709}
3710
3711
3712/// Emit task_dup function (for initialization of
3713/// private/firstprivate/lastprivate vars and last_iter flag)
3714/// \code
3715/// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int
3716/// lastpriv) {
3717/// // setup lastprivate flag
3718/// task_dst->last = lastpriv;
3719/// // could be constructor calls here...
3720/// }
3721/// \endcode
3722static llvm::Value *
3723emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc,
3724 const OMPExecutableDirective &D,
3725 QualType KmpTaskTWithPrivatesPtrQTy,
3726 const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3727 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy,
3728 QualType SharedsPtrTy, const OMPTaskDataTy &Data,
3729 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) {
3730 ASTContext &C = CGM.getContext();
3731 auto *DstArg = ImplicitParamDecl::Create(
3732 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr, T: KmpTaskTWithPrivatesPtrQTy,
3733 ParamKind: ImplicitParamKind::Other);
3734 auto *SrcArg = ImplicitParamDecl::Create(
3735 C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr, T: KmpTaskTWithPrivatesPtrQTy,
3736 ParamKind: ImplicitParamKind::Other);
3737 auto *LastprivArg =
3738 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr, T: C.IntTy,
3739 ParamKind: ImplicitParamKind::Other);
3740 FunctionArgList Args{DstArg, SrcArg, LastprivArg};
3741 const auto &TaskDupFnInfo =
3742 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
3743 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(Info: TaskDupFnInfo);
3744 std::string Name = CGM.getOpenMPRuntime().getName(Parts: {"omp_task_dup", ""});
3745 auto *TaskDup = llvm::Function::Create(
3746 Ty: TaskDupTy, Linkage: llvm::GlobalValue::InternalLinkage, N: Name, M: &CGM.getModule());
3747 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: TaskDup, FI: TaskDupFnInfo);
3748 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
3749 TaskDup->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
3750 TaskDup->setDoesNotRecurse();
3751 CodeGenFunction CGF(CGM);
3752 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn: TaskDup, FnInfo: TaskDupFnInfo, Args, Loc,
3753 StartLoc: Loc);
3754
3755 LValue TDBase = CGF.EmitLoadOfPointerLValue(
3756 Ptr: CGF.GetAddrOfLocalVar(VD: DstArg),
3757 PtrTy: KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3758 // task_dst->liter = lastpriv;
3759 if (WithLastIter) {
3760 auto LIFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTLastIter);
3761 LValue Base = CGF.EmitLValueForField(
3762 Base: TDBase, Field: *KmpTaskTWithPrivatesQTyRD->field_begin());
3763 LValue LILVal = CGF.EmitLValueForField(Base, Field: *LIFI);
3764 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar(
3765 Addr: CGF.GetAddrOfLocalVar(VD: LastprivArg), /*Volatile=*/false, Ty: C.IntTy, Loc);
3766 CGF.EmitStoreOfScalar(value: Lastpriv, lvalue: LILVal);
3767 }
3768
3769 // Emit initial values for private copies (if any).
3770 assert(!Privates.empty());
3771 Address KmpTaskSharedsPtr = Address::invalid();
3772 if (!Data.FirstprivateVars.empty()) {
3773 LValue TDBase = CGF.EmitLoadOfPointerLValue(
3774 Ptr: CGF.GetAddrOfLocalVar(VD: SrcArg),
3775 PtrTy: KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3776 LValue Base = CGF.EmitLValueForField(
3777 Base: TDBase, Field: *KmpTaskTWithPrivatesQTyRD->field_begin());
3778 KmpTaskSharedsPtr = Address(
3779 CGF.EmitLoadOfScalar(lvalue: CGF.EmitLValueForField(
3780 Base, Field: *std::next(x: KmpTaskTQTyRD->field_begin(),
3781 n: KmpTaskTShareds)),
3782 Loc),
3783 CGF.Int8Ty, CGM.getNaturalTypeAlignment(T: SharedsTy));
3784 }
3785 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD,
3786 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true);
3787 CGF.FinishFunction();
3788 return TaskDup;
3789}
3790
3791/// Checks if destructor function is required to be generated.
3792/// \return true if cleanups are required, false otherwise.
3793static bool
3794checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3795 ArrayRef<PrivateDataTy> Privates) {
3796 for (const PrivateDataTy &P : Privates) {
3797 if (P.second.isLocalPrivate())
3798 continue;
3799 QualType Ty = P.second.Original->getType().getNonReferenceType();
3800 if (Ty.isDestructedType())
3801 return true;
3802 }
3803 return false;
3804}
3805
3806namespace {
3807/// Loop generator for OpenMP iterator expression.
3808class OMPIteratorGeneratorScope final
3809 : public CodeGenFunction::OMPPrivateScope {
3810 CodeGenFunction &CGF;
3811 const OMPIteratorExpr *E = nullptr;
3812 SmallVector<CodeGenFunction::JumpDest, 4> ContDests;
3813 SmallVector<CodeGenFunction::JumpDest, 4> ExitDests;
3814 OMPIteratorGeneratorScope() = delete;
3815 OMPIteratorGeneratorScope(OMPIteratorGeneratorScope &) = delete;
3816
3817public:
3818 OMPIteratorGeneratorScope(CodeGenFunction &CGF, const OMPIteratorExpr *E)
3819 : CodeGenFunction::OMPPrivateScope(CGF), CGF(CGF), E(E) {
3820 if (!E)
3821 return;
3822 SmallVector<llvm::Value *, 4> Uppers;
3823 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) {
3824 Uppers.push_back(Elt: CGF.EmitScalarExpr(E: E->getHelper(I).Upper));
3825 const auto *VD = cast<VarDecl>(Val: E->getIteratorDecl(I));
3826 addPrivate(LocalVD: VD, Addr: CGF.CreateMemTemp(T: VD->getType(), Name: VD->getName()));
3827 const OMPIteratorHelperData &HelperData = E->getHelper(I);
3828 addPrivate(
3829 LocalVD: HelperData.CounterVD,
3830 Addr: CGF.CreateMemTemp(T: HelperData.CounterVD->getType(), Name: "counter.addr"));
3831 }
3832 Privatize();
3833
3834 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) {
3835 const OMPIteratorHelperData &HelperData = E->getHelper(I);
3836 LValue CLVal =
3837 CGF.MakeAddrLValue(Addr: CGF.GetAddrOfLocalVar(VD: HelperData.CounterVD),
3838 T: HelperData.CounterVD->getType());
3839 // Counter = 0;
3840 CGF.EmitStoreOfScalar(
3841 value: llvm::ConstantInt::get(Ty: CLVal.getAddress().getElementType(), V: 0),
3842 lvalue: CLVal);
3843 CodeGenFunction::JumpDest &ContDest =
3844 ContDests.emplace_back(Args: CGF.getJumpDestInCurrentScope(Name: "iter.cont"));
3845 CodeGenFunction::JumpDest &ExitDest =
3846 ExitDests.emplace_back(Args: CGF.getJumpDestInCurrentScope(Name: "iter.exit"));
3847 // N = <number-of_iterations>;
3848 llvm::Value *N = Uppers[I];
3849 // cont:
3850 // if (Counter < N) goto body; else goto exit;
3851 CGF.EmitBlock(BB: ContDest.getBlock());
3852 auto *CVal =
3853 CGF.EmitLoadOfScalar(lvalue: CLVal, Loc: HelperData.CounterVD->getLocation());
3854 llvm::Value *Cmp =
3855 HelperData.CounterVD->getType()->isSignedIntegerOrEnumerationType()
3856 ? CGF.Builder.CreateICmpSLT(LHS: CVal, RHS: N)
3857 : CGF.Builder.CreateICmpULT(LHS: CVal, RHS: N);
3858 llvm::BasicBlock *BodyBB = CGF.createBasicBlock(name: "iter.body");
3859 CGF.Builder.CreateCondBr(Cond: Cmp, True: BodyBB, False: ExitDest.getBlock());
3860 // body:
3861 CGF.EmitBlock(BB: BodyBB);
3862 // Iteri = Begini + Counter * Stepi;
3863 CGF.EmitIgnoredExpr(E: HelperData.Update);
3864 }
3865 }
3866 ~OMPIteratorGeneratorScope() {
3867 if (!E)
3868 return;
3869 for (unsigned I = E->numOfIterators(); I > 0; --I) {
3870 // Counter = Counter + 1;
3871 const OMPIteratorHelperData &HelperData = E->getHelper(I: I - 1);
3872 CGF.EmitIgnoredExpr(E: HelperData.CounterUpdate);
3873 // goto cont;
3874 CGF.EmitBranchThroughCleanup(Dest: ContDests[I - 1]);
3875 // exit:
3876 CGF.EmitBlock(BB: ExitDests[I - 1].getBlock(), /*IsFinished=*/I == 1);
3877 }
3878 }
3879};
3880} // namespace
3881
3882static std::pair<llvm::Value *, llvm::Value *>
3883getPointerAndSize(CodeGenFunction &CGF, const Expr *E) {
3884 const auto *OASE = dyn_cast<OMPArrayShapingExpr>(Val: E);
3885 llvm::Value *Addr;
3886 if (OASE) {
3887 const Expr *Base = OASE->getBase();
3888 Addr = CGF.EmitScalarExpr(E: Base);
3889 } else {
3890 Addr = CGF.EmitLValue(E).getPointer(CGF);
3891 }
3892 llvm::Value *SizeVal;
3893 QualType Ty = E->getType();
3894 if (OASE) {
3895 SizeVal = CGF.getTypeSize(Ty: OASE->getBase()->getType()->getPointeeType());
3896 for (const Expr *SE : OASE->getDimensions()) {
3897 llvm::Value *Sz = CGF.EmitScalarExpr(E: SE);
3898 Sz = CGF.EmitScalarConversion(
3899 Src: Sz, SrcTy: SE->getType(), DstTy: CGF.getContext().getSizeType(), Loc: SE->getExprLoc());
3900 SizeVal = CGF.Builder.CreateNUWMul(LHS: SizeVal, RHS: Sz);
3901 }
3902 } else if (const auto *ASE =
3903 dyn_cast<ArraySectionExpr>(Val: E->IgnoreParenImpCasts())) {
3904 LValue UpAddrLVal = CGF.EmitArraySectionExpr(E: ASE, /*IsLowerBound=*/false);
3905 Address UpAddrAddress = UpAddrLVal.getAddress();
3906 llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32(
3907 Ty: UpAddrAddress.getElementType(), Ptr: UpAddrAddress.emitRawPointer(CGF),
3908 /*Idx0=*/1);
3909 SizeVal = CGF.Builder.CreatePtrDiff(LHS: UpAddr, RHS: Addr, Name: "", /*IsNUW=*/true);
3910 } else {
3911 SizeVal = CGF.getTypeSize(Ty);
3912 }
3913 return std::make_pair(x&: Addr, y&: SizeVal);
3914}
3915
3916/// Builds kmp_depend_info, if it is not built yet, and builds flags type.
3917static void getKmpAffinityType(ASTContext &C, QualType &KmpTaskAffinityInfoTy) {
3918 QualType FlagsTy = C.getIntTypeForBitwidth(DestWidth: 32, /*Signed=*/false);
3919 if (KmpTaskAffinityInfoTy.isNull()) {
3920 RecordDecl *KmpAffinityInfoRD =
3921 C.buildImplicitRecord(Name: "kmp_task_affinity_info_t");
3922 KmpAffinityInfoRD->startDefinition();
3923 addFieldToRecordDecl(C, DC: KmpAffinityInfoRD, FieldTy: C.getIntPtrType());
3924 addFieldToRecordDecl(C, DC: KmpAffinityInfoRD, FieldTy: C.getSizeType());
3925 addFieldToRecordDecl(C, DC: KmpAffinityInfoRD, FieldTy: FlagsTy);
3926 KmpAffinityInfoRD->completeDefinition();
3927 KmpTaskAffinityInfoTy = C.getCanonicalTagType(TD: KmpAffinityInfoRD);
3928 }
3929}
3930
3931CGOpenMPRuntime::TaskResultTy
3932CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc,
3933 const OMPExecutableDirective &D,
3934 llvm::Function *TaskFunction, QualType SharedsTy,
3935 Address Shareds, const OMPTaskDataTy &Data) {
3936 ASTContext &C = CGM.getContext();
3937 llvm::SmallVector<PrivateDataTy, 4> Privates;
3938 // Aggregate privates and sort them by the alignment.
3939 const auto *I = Data.PrivateCopies.begin();
3940 for (const Expr *E : Data.PrivateVars) {
3941 const auto *Decl = cast<DeclRefExpr>(Val: E)->getDecl();
3942 const auto *VD = getOriginalVarDecl(Decl);
3943 const auto *CopyVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *I)->getDecl());
3944 Privates.emplace_back(Args: C.getDeclAlign(D: VD),
3945 Args: PrivateHelpersTy(E, VD, CopyVD,
3946 /*PrivateElemInit=*/nullptr));
3947 ++I;
3948 }
3949 I = Data.FirstprivateCopies.begin();
3950 const auto *IElemInitRef = Data.FirstprivateInits.begin();
3951 for (const Expr *E : Data.FirstprivateVars) {
3952 const auto *Decl = cast<DeclRefExpr>(Val: E)->getDecl();
3953 const auto *VD = getOriginalVarDecl(Decl);
3954 const auto *CopyVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *I)->getDecl());
3955 const auto *InitVD =
3956 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IElemInitRef)->getDecl());
3957 Privates.emplace_back(Args: C.getDeclAlign(D: VD),
3958 Args: PrivateHelpersTy(E, VD, CopyVD, InitVD));
3959 ++I;
3960 ++IElemInitRef;
3961 }
3962 I = Data.LastprivateCopies.begin();
3963 for (const Expr *E : Data.LastprivateVars) {
3964 const auto *Decl = cast<DeclRefExpr>(Val: E)->getDecl();
3965 const auto *VD = getOriginalVarDecl(Decl);
3966 const auto *CopyVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *I)->getDecl());
3967 Privates.emplace_back(Args: C.getDeclAlign(D: VD),
3968 Args: PrivateHelpersTy(E, VD, CopyVD,
3969 /*PrivateElemInit=*/nullptr));
3970 ++I;
3971 }
3972 for (const VarDecl *VD : Data.PrivateLocals) {
3973 if (isAllocatableDecl(VD))
3974 Privates.emplace_back(Args: CGM.getPointerAlign(), Args: PrivateHelpersTy(VD));
3975 else
3976 Privates.emplace_back(Args: C.getDeclAlign(D: VD), Args: PrivateHelpersTy(VD));
3977 }
3978 llvm::stable_sort(Range&: Privates,
3979 C: [](const PrivateDataTy &L, const PrivateDataTy &R) {
3980 return L.first > R.first;
3981 });
3982 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
3983 // Build type kmp_routine_entry_t (if not built yet).
3984 emitKmpRoutineEntryT(KmpInt32Ty);
3985 // Build type kmp_task_t (if not built yet).
3986 if (isOpenMPTaskLoopDirective(DKind: D.getDirectiveKind())) {
3987 if (SavedKmpTaskloopTQTy.isNull()) {
3988 SavedKmpTaskloopTQTy = C.getCanonicalTagType(TD: createKmpTaskTRecordDecl(
3989 CGM, Kind: D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPointerQTy: KmpRoutineEntryPtrQTy));
3990 }
3991 KmpTaskTQTy = SavedKmpTaskloopTQTy;
3992 } else {
3993 assert((D.getDirectiveKind() == OMPD_task ||
3994 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) ||
3995 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) &&
3996 "Expected taskloop, task or target directive");
3997 if (SavedKmpTaskTQTy.isNull()) {
3998 SavedKmpTaskTQTy = C.getCanonicalTagType(TD: createKmpTaskTRecordDecl(
3999 CGM, Kind: D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPointerQTy: KmpRoutineEntryPtrQTy));
4000 }
4001 KmpTaskTQTy = SavedKmpTaskTQTy;
4002 }
4003 const auto *KmpTaskTQTyRD = KmpTaskTQTy->castAsRecordDecl();
4004 // Build particular struct kmp_task_t for the given task.
4005 const RecordDecl *KmpTaskTWithPrivatesQTyRD =
4006 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates);
4007 CanQualType KmpTaskTWithPrivatesQTy =
4008 C.getCanonicalTagType(TD: KmpTaskTWithPrivatesQTyRD);
4009 QualType KmpTaskTWithPrivatesPtrQTy =
4010 C.getPointerType(T: KmpTaskTWithPrivatesQTy);
4011 llvm::Type *KmpTaskTWithPrivatesPtrTy = CGF.Builder.getPtrTy(AddrSpace: 0);
4012 llvm::Value *KmpTaskTWithPrivatesTySize =
4013 CGF.getTypeSize(Ty: KmpTaskTWithPrivatesQTy);
4014 QualType SharedsPtrTy = C.getPointerType(T: SharedsTy);
4015
4016 // Emit initial values for private copies (if any).
4017 llvm::Value *TaskPrivatesMap = nullptr;
4018 llvm::Type *TaskPrivatesMapTy =
4019 std::next(x: TaskFunction->arg_begin(), n: 3)->getType();
4020 if (!Privates.empty()) {
4021 auto FI = std::next(x: KmpTaskTWithPrivatesQTyRD->field_begin());
4022 TaskPrivatesMap =
4023 emitTaskPrivateMappingFunction(CGM, Loc, Data, PrivatesQTy: FI->getType(), Privates);
4024 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4025 V: TaskPrivatesMap, DestTy: TaskPrivatesMapTy);
4026 } else {
4027 TaskPrivatesMap = llvm::ConstantPointerNull::get(
4028 T: cast<llvm::PointerType>(Val: TaskPrivatesMapTy));
4029 }
4030 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
4031 // kmp_task_t *tt);
4032 llvm::Function *TaskEntry = emitProxyTaskFunction(
4033 CGM, Loc, Kind: D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
4034 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction,
4035 TaskPrivatesMap);
4036
4037 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
4038 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
4039 // kmp_routine_entry_t *task_entry);
4040 // Task flags. Format is taken from
4041 // https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h,
4042 // description of kmp_tasking_flags struct.
4043 enum {
4044 TiedFlag = 0x1,
4045 FinalFlag = 0x2,
4046 DestructorsFlag = 0x8,
4047 PriorityFlag = 0x20,
4048 DetachableFlag = 0x40,
4049 FreeAgentFlag = 0x80,
4050 TransparentFlag = 0x100,
4051 };
4052 unsigned Flags = Data.Tied ? TiedFlag : 0;
4053 bool NeedsCleanup = false;
4054 if (!Privates.empty()) {
4055 NeedsCleanup =
4056 checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD, Privates);
4057 if (NeedsCleanup)
4058 Flags = Flags | DestructorsFlag;
4059 }
4060 if (const auto *Clause = D.getSingleClause<OMPThreadsetClause>()) {
4061 OpenMPThreadsetKind Kind = Clause->getThreadsetKind();
4062 if (Kind == OMPC_THREADSET_omp_pool)
4063 Flags = Flags | FreeAgentFlag;
4064 }
4065 if (D.getSingleClause<OMPTransparentClause>())
4066 Flags |= TransparentFlag;
4067
4068 if (Data.Priority.getInt())
4069 Flags = Flags | PriorityFlag;
4070 if (D.hasClausesOfKind<OMPDetachClause>())
4071 Flags = Flags | DetachableFlag;
4072 llvm::Value *TaskFlags =
4073 Data.Final.getPointer()
4074 ? CGF.Builder.CreateSelect(C: Data.Final.getPointer(),
4075 True: CGF.Builder.getInt32(C: FinalFlag),
4076 False: CGF.Builder.getInt32(/*C=*/0))
4077 : CGF.Builder.getInt32(C: Data.Final.getInt() ? FinalFlag : 0);
4078 TaskFlags = CGF.Builder.CreateOr(LHS: TaskFlags, RHS: CGF.Builder.getInt32(C: Flags));
4079 llvm::Value *SharedsSize = CGM.getSize(numChars: C.getTypeSizeInChars(T: SharedsTy));
4080 SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc),
4081 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize,
4082 SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4083 V: TaskEntry, DestTy: KmpRoutineEntryPtrTy)};
4084 llvm::Value *NewTask;
4085 if (D.hasClausesOfKind<OMPNowaitClause>()) {
4086 // Check if we have any device clause associated with the directive.
4087 const Expr *Device = nullptr;
4088 if (auto *C = D.getSingleClause<OMPDeviceClause>())
4089 Device = C->getDevice();
4090 // Emit device ID if any otherwise use default value.
4091 llvm::Value *DeviceID;
4092 if (Device)
4093 DeviceID = CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: Device),
4094 DestTy: CGF.Int64Ty, /*isSigned=*/true);
4095 else
4096 DeviceID = CGF.Builder.getInt64(C: OMP_DEVICEID_UNDEF);
4097 AllocArgs.push_back(Elt: DeviceID);
4098 NewTask = CGF.EmitRuntimeCall(
4099 callee: OMPBuilder.getOrCreateRuntimeFunction(
4100 M&: CGM.getModule(), FnID: OMPRTL___kmpc_omp_target_task_alloc),
4101 args: AllocArgs);
4102 } else {
4103 NewTask =
4104 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
4105 M&: CGM.getModule(), FnID: OMPRTL___kmpc_omp_task_alloc),
4106 args: AllocArgs);
4107 }
4108 // Emit detach clause initialization.
4109 // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid,
4110 // task_descriptor);
4111 if (const auto *DC = D.getSingleClause<OMPDetachClause>()) {
4112 const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts();
4113 LValue EvtLVal = CGF.EmitLValue(E: Evt);
4114
4115 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
4116 // int gtid, kmp_task_t *task);
4117 llvm::Value *Loc = emitUpdateLocation(CGF, Loc: DC->getBeginLoc());
4118 llvm::Value *Tid = getThreadID(CGF, Loc: DC->getBeginLoc());
4119 Tid = CGF.Builder.CreateIntCast(V: Tid, DestTy: CGF.IntTy, /*isSigned=*/false);
4120 llvm::Value *EvtVal = CGF.EmitRuntimeCall(
4121 callee: OMPBuilder.getOrCreateRuntimeFunction(
4122 M&: CGM.getModule(), FnID: OMPRTL___kmpc_task_allow_completion_event),
4123 args: {Loc, Tid, NewTask});
4124 EvtVal = CGF.EmitScalarConversion(Src: EvtVal, SrcTy: C.VoidPtrTy, DstTy: Evt->getType(),
4125 Loc: Evt->getExprLoc());
4126 CGF.EmitStoreOfScalar(value: EvtVal, lvalue: EvtLVal);
4127 }
4128 // Process affinity clauses.
4129 if (D.hasClausesOfKind<OMPAffinityClause>()) {
4130 // Process list of affinity data.
4131 ASTContext &C = CGM.getContext();
4132 Address AffinitiesArray = Address::invalid();
4133 // Calculate number of elements to form the array of affinity data.
4134 llvm::Value *NumOfElements = nullptr;
4135 unsigned NumAffinities = 0;
4136 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4137 if (const Expr *Modifier = C->getModifier()) {
4138 const auto *IE = cast<OMPIteratorExpr>(Val: Modifier->IgnoreParenImpCasts());
4139 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4140 llvm::Value *Sz = CGF.EmitScalarExpr(E: IE->getHelper(I).Upper);
4141 Sz = CGF.Builder.CreateIntCast(V: Sz, DestTy: CGF.SizeTy, /*isSigned=*/false);
4142 NumOfElements =
4143 NumOfElements ? CGF.Builder.CreateNUWMul(LHS: NumOfElements, RHS: Sz) : Sz;
4144 }
4145 } else {
4146 NumAffinities += C->varlist_size();
4147 }
4148 }
4149 getKmpAffinityType(C&: CGM.getContext(), KmpTaskAffinityInfoTy);
4150 // Fields ids in kmp_task_affinity_info record.
4151 enum RTLAffinityInfoFieldsTy { BaseAddr, Len, Flags };
4152
4153 QualType KmpTaskAffinityInfoArrayTy;
4154 if (NumOfElements) {
4155 NumOfElements = CGF.Builder.CreateNUWAdd(
4156 LHS: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: NumAffinities), RHS: NumOfElements);
4157 auto *OVE = new (C) OpaqueValueExpr(
4158 Loc,
4159 C.getIntTypeForBitwidth(DestWidth: C.getTypeSize(T: C.getSizeType()), /*Signed=*/0),
4160 VK_PRValue);
4161 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE,
4162 RValue::get(V: NumOfElements));
4163 KmpTaskAffinityInfoArrayTy = C.getVariableArrayType(
4164 EltTy: KmpTaskAffinityInfoTy, NumElts: OVE, ASM: ArraySizeModifier::Normal,
4165 /*IndexTypeQuals=*/0);
4166 // Properly emit variable-sized array.
4167 auto *PD = ImplicitParamDecl::Create(C, T: KmpTaskAffinityInfoArrayTy,
4168 ParamKind: ImplicitParamKind::Other);
4169 CGF.EmitVarDecl(D: *PD);
4170 AffinitiesArray = CGF.GetAddrOfLocalVar(VD: PD);
4171 NumOfElements = CGF.Builder.CreateIntCast(V: NumOfElements, DestTy: CGF.Int32Ty,
4172 /*isSigned=*/false);
4173 } else {
4174 KmpTaskAffinityInfoArrayTy = C.getConstantArrayType(
4175 EltTy: KmpTaskAffinityInfoTy,
4176 ArySize: llvm::APInt(C.getTypeSize(T: C.getSizeType()), NumAffinities), SizeExpr: nullptr,
4177 ASM: ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
4178 AffinitiesArray = CGF.CreateMemTempWithoutCast(T: KmpTaskAffinityInfoArrayTy,
4179 Name: ".affs.arr.addr");
4180 AffinitiesArray = CGF.Builder.CreateConstArrayGEP(Addr: AffinitiesArray, Index: 0);
4181 NumOfElements = llvm::ConstantInt::get(Ty: CGM.Int32Ty, V: NumAffinities,
4182 /*isSigned=*/IsSigned: false);
4183 }
4184
4185 const auto *KmpAffinityInfoRD = KmpTaskAffinityInfoTy->getAsRecordDecl();
4186 // Fill array by elements without iterators.
4187 unsigned Pos = 0;
4188 bool HasIterator = false;
4189 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4190 if (C->getModifier()) {
4191 HasIterator = true;
4192 continue;
4193 }
4194 for (const Expr *E : C->varlist()) {
4195 llvm::Value *Addr;
4196 llvm::Value *Size;
4197 std::tie(args&: Addr, args&: Size) = getPointerAndSize(CGF, E);
4198 LValue Base =
4199 CGF.MakeAddrLValue(Addr: CGF.Builder.CreateConstGEP(Addr: AffinitiesArray, Index: Pos),
4200 T: KmpTaskAffinityInfoTy);
4201 // affs[i].base_addr = &<Affinities[i].second>;
4202 LValue BaseAddrLVal = CGF.EmitLValueForField(
4203 Base, Field: *std::next(x: KmpAffinityInfoRD->field_begin(), n: BaseAddr));
4204 CGF.EmitStoreOfScalar(value: CGF.Builder.CreatePtrToInt(V: Addr, DestTy: CGF.IntPtrTy),
4205 lvalue: BaseAddrLVal);
4206 // affs[i].len = sizeof(<Affinities[i].second>);
4207 LValue LenLVal = CGF.EmitLValueForField(
4208 Base, Field: *std::next(x: KmpAffinityInfoRD->field_begin(), n: Len));
4209 CGF.EmitStoreOfScalar(value: Size, lvalue: LenLVal);
4210 ++Pos;
4211 }
4212 }
4213 LValue PosLVal;
4214 if (HasIterator) {
4215 PosLVal = CGF.MakeAddrLValue(
4216 Addr: CGF.CreateMemTempWithoutCast(T: C.getSizeType(), Name: "affs.counter.addr"),
4217 T: C.getSizeType());
4218 CGF.EmitStoreOfScalar(value: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: Pos), lvalue: PosLVal);
4219 }
4220 // Process elements with iterators.
4221 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4222 const Expr *Modifier = C->getModifier();
4223 if (!Modifier)
4224 continue;
4225 OMPIteratorGeneratorScope IteratorScope(
4226 CGF, cast_or_null<OMPIteratorExpr>(Val: Modifier->IgnoreParenImpCasts()));
4227 for (const Expr *E : C->varlist()) {
4228 llvm::Value *Addr;
4229 llvm::Value *Size;
4230 std::tie(args&: Addr, args&: Size) = getPointerAndSize(CGF, E);
4231 llvm::Value *Idx = CGF.EmitLoadOfScalar(lvalue: PosLVal, Loc: E->getExprLoc());
4232 LValue Base =
4233 CGF.MakeAddrLValue(Addr: CGF.Builder.CreateGEP(CGF, Addr: AffinitiesArray, Index: Idx),
4234 T: KmpTaskAffinityInfoTy);
4235 // affs[i].base_addr = &<Affinities[i].second>;
4236 LValue BaseAddrLVal = CGF.EmitLValueForField(
4237 Base, Field: *std::next(x: KmpAffinityInfoRD->field_begin(), n: BaseAddr));
4238 CGF.EmitStoreOfScalar(value: CGF.Builder.CreatePtrToInt(V: Addr, DestTy: CGF.IntPtrTy),
4239 lvalue: BaseAddrLVal);
4240 // affs[i].len = sizeof(<Affinities[i].second>);
4241 LValue LenLVal = CGF.EmitLValueForField(
4242 Base, Field: *std::next(x: KmpAffinityInfoRD->field_begin(), n: Len));
4243 CGF.EmitStoreOfScalar(value: Size, lvalue: LenLVal);
4244 Idx = CGF.Builder.CreateNUWAdd(
4245 LHS: Idx, RHS: llvm::ConstantInt::get(Ty: Idx->getType(), V: 1));
4246 CGF.EmitStoreOfScalar(value: Idx, lvalue: PosLVal);
4247 }
4248 }
4249 // Call to kmp_int32 __kmpc_omp_reg_task_with_affinity(ident_t *loc_ref,
4250 // kmp_int32 gtid, kmp_task_t *new_task, kmp_int32
4251 // naffins, kmp_task_affinity_info_t *affin_list);
4252 llvm::Value *LocRef = emitUpdateLocation(CGF, Loc);
4253 llvm::Value *GTid = getThreadID(CGF, Loc);
4254 llvm::Value *AffinListPtr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4255 V: AffinitiesArray.emitRawPointer(CGF), DestTy: CGM.VoidPtrTy);
4256 // FIXME: Emit the function and ignore its result for now unless the
4257 // runtime function is properly implemented.
4258 (void)CGF.EmitRuntimeCall(
4259 callee: OMPBuilder.getOrCreateRuntimeFunction(
4260 M&: CGM.getModule(), FnID: OMPRTL___kmpc_omp_reg_task_with_affinity),
4261 args: {LocRef, GTid, NewTask, NumOfElements, AffinListPtr});
4262 }
4263 llvm::Value *NewTaskNewTaskTTy =
4264 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4265 V: NewTask, DestTy: KmpTaskTWithPrivatesPtrTy);
4266 LValue Base = CGF.MakeNaturalAlignRawAddrLValue(V: NewTaskNewTaskTTy,
4267 T: KmpTaskTWithPrivatesQTy);
4268 LValue TDBase =
4269 CGF.EmitLValueForField(Base, Field: *KmpTaskTWithPrivatesQTyRD->field_begin());
4270 // Fill the data in the resulting kmp_task_t record.
4271 // Copy shareds if there are any.
4272 Address KmpTaskSharedsPtr = Address::invalid();
4273 if (!SharedsTy->castAsRecordDecl()->field_empty()) {
4274 KmpTaskSharedsPtr = Address(
4275 CGF.EmitLoadOfScalar(
4276 lvalue: CGF.EmitLValueForField(
4277 Base: TDBase,
4278 Field: *std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTShareds)),
4279 Loc),
4280 CGF.Int8Ty, CGM.getNaturalTypeAlignment(T: SharedsTy));
4281 LValue Dest = CGF.MakeAddrLValue(Addr: KmpTaskSharedsPtr, T: SharedsTy);
4282 LValue Src = CGF.MakeAddrLValue(Addr: Shareds, T: SharedsTy);
4283 CGF.EmitAggregateCopy(Dest, Src, EltTy: SharedsTy, MayOverlap: AggValueSlot::DoesNotOverlap);
4284 }
4285 // Emit initial values for private copies (if any).
4286 TaskResultTy Result;
4287 if (!Privates.empty()) {
4288 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase: Base, KmpTaskTWithPrivatesQTyRD,
4289 SharedsTy, SharedsPtrTy, Data, Privates,
4290 /*ForDup=*/false);
4291 if (isOpenMPTaskLoopDirective(DKind: D.getDirectiveKind()) &&
4292 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) {
4293 Result.TaskDupFn = emitTaskDupFunction(
4294 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD,
4295 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates,
4296 /*WithLastIter=*/!Data.LastprivateVars.empty());
4297 }
4298 }
4299 // Fields of union "kmp_cmplrdata_t" for destructors and priority.
4300 enum { Priority = 0, Destructors = 1 };
4301 // Provide pointer to function with destructors for privates.
4302 auto FI = std::next(x: KmpTaskTQTyRD->field_begin(), n: Data1);
4303 const auto *KmpCmplrdataUD = (*FI)->getType()->castAsRecordDecl();
4304 assert(KmpCmplrdataUD->isUnion());
4305 if (NeedsCleanup) {
4306 llvm::Value *DestructorFn = emitDestructorsFunction(
4307 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
4308 KmpTaskTWithPrivatesQTy);
4309 LValue Data1LV = CGF.EmitLValueForField(Base: TDBase, Field: *FI);
4310 LValue DestructorsLV = CGF.EmitLValueForField(
4311 Base: Data1LV, Field: *std::next(x: KmpCmplrdataUD->field_begin(), n: Destructors));
4312 CGF.EmitStoreOfScalar(value: CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4313 V: DestructorFn, DestTy: KmpRoutineEntryPtrTy),
4314 lvalue: DestructorsLV);
4315 }
4316 // Set priority.
4317 if (Data.Priority.getInt()) {
4318 LValue Data2LV = CGF.EmitLValueForField(
4319 Base: TDBase, Field: *std::next(x: KmpTaskTQTyRD->field_begin(), n: Data2));
4320 LValue PriorityLV = CGF.EmitLValueForField(
4321 Base: Data2LV, Field: *std::next(x: KmpCmplrdataUD->field_begin(), n: Priority));
4322 CGF.EmitStoreOfScalar(value: Data.Priority.getPointer(), lvalue: PriorityLV);
4323 }
4324 Result.NewTask = NewTask;
4325 Result.TaskEntry = TaskEntry;
4326 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy;
4327 Result.TDBase = TDBase;
4328 Result.KmpTaskTQTyRD = KmpTaskTQTyRD;
4329 return Result;
4330}
4331
4332/// Translates internal dependency kind into the runtime kind.
4333static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) {
4334 RTLDependenceKindTy DepKind;
4335 switch (K) {
4336 case OMPC_DEPEND_in:
4337 DepKind = RTLDependenceKindTy::DepIn;
4338 break;
4339 // Out and InOut dependencies must use the same code.
4340 case OMPC_DEPEND_out:
4341 case OMPC_DEPEND_inout:
4342 DepKind = RTLDependenceKindTy::DepInOut;
4343 break;
4344 case OMPC_DEPEND_mutexinoutset:
4345 DepKind = RTLDependenceKindTy::DepMutexInOutSet;
4346 break;
4347 case OMPC_DEPEND_inoutset:
4348 DepKind = RTLDependenceKindTy::DepInOutSet;
4349 break;
4350 case OMPC_DEPEND_outallmemory:
4351 DepKind = RTLDependenceKindTy::DepOmpAllMem;
4352 break;
4353 case OMPC_DEPEND_source:
4354 case OMPC_DEPEND_sink:
4355 case OMPC_DEPEND_depobj:
4356 case OMPC_DEPEND_inoutallmemory:
4357 case OMPC_DEPEND_unknown:
4358 llvm_unreachable("Unknown task dependence type");
4359 }
4360 return DepKind;
4361}
4362
4363/// Builds kmp_depend_info, if it is not built yet, and builds flags type.
4364static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy,
4365 QualType &FlagsTy) {
4366 FlagsTy = C.getIntTypeForBitwidth(DestWidth: C.getTypeSize(T: C.BoolTy), /*Signed=*/false);
4367 if (KmpDependInfoTy.isNull()) {
4368 RecordDecl *KmpDependInfoRD = C.buildImplicitRecord(Name: "kmp_depend_info");
4369 KmpDependInfoRD->startDefinition();
4370 addFieldToRecordDecl(C, DC: KmpDependInfoRD, FieldTy: C.getIntPtrType());
4371 addFieldToRecordDecl(C, DC: KmpDependInfoRD, FieldTy: C.getSizeType());
4372 addFieldToRecordDecl(C, DC: KmpDependInfoRD, FieldTy: FlagsTy);
4373 KmpDependInfoRD->completeDefinition();
4374 KmpDependInfoTy = C.getCanonicalTagType(TD: KmpDependInfoRD);
4375 }
4376}
4377
4378std::pair<llvm::Value *, LValue>
4379CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal,
4380 SourceLocation Loc) {
4381 ASTContext &C = CGM.getContext();
4382 QualType FlagsTy;
4383 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4384 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4385 QualType KmpDependInfoPtrTy = C.getPointerType(T: KmpDependInfoTy);
4386 LValue Base = CGF.EmitLoadOfPointerLValue(
4387 Ptr: DepobjLVal.getAddress().withElementType(
4388 ElemTy: CGF.ConvertTypeForMem(T: KmpDependInfoPtrTy)),
4389 PtrTy: KmpDependInfoPtrTy->castAs<PointerType>());
4390 Address DepObjAddr = CGF.Builder.CreateGEP(
4391 CGF, Addr: Base.getAddress(),
4392 Index: llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: -1, /*isSigned=*/IsSigned: true));
4393 LValue NumDepsBase = CGF.MakeAddrLValue(
4394 Addr: DepObjAddr, T: KmpDependInfoTy, BaseInfo: Base.getBaseInfo(), TBAAInfo: Base.getTBAAInfo());
4395 // NumDeps = deps[i].base_addr;
4396 LValue BaseAddrLVal = CGF.EmitLValueForField(
4397 Base: NumDepsBase,
4398 Field: *std::next(x: KmpDependInfoRD->field_begin(),
4399 n: static_cast<unsigned int>(RTLDependInfoFields::BaseAddr)));
4400 llvm::Value *NumDeps = CGF.EmitLoadOfScalar(lvalue: BaseAddrLVal, Loc);
4401 return std::make_pair(x&: NumDeps, y&: Base);
4402}
4403
4404static void emitDependData(CodeGenFunction &CGF, QualType &KmpDependInfoTy,
4405 llvm::PointerUnion<unsigned *, LValue *> Pos,
4406 const OMPTaskDataTy::DependData &Data,
4407 Address DependenciesArray) {
4408 CodeGenModule &CGM = CGF.CGM;
4409 ASTContext &C = CGM.getContext();
4410 QualType FlagsTy;
4411 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4412 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4413 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(T: FlagsTy);
4414
4415 OMPIteratorGeneratorScope IteratorScope(
4416 CGF, cast_or_null<OMPIteratorExpr>(
4417 Val: Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4418 : nullptr));
4419 for (const Expr *E : Data.DepExprs) {
4420 llvm::Value *Addr;
4421 llvm::Value *Size;
4422
4423 // The expression will be a nullptr in the 'omp_all_memory' case.
4424 if (E) {
4425 std::tie(args&: Addr, args&: Size) = getPointerAndSize(CGF, E);
4426 Addr = CGF.Builder.CreatePtrToInt(V: Addr, DestTy: CGF.IntPtrTy);
4427 } else {
4428 Addr = llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: 0);
4429 Size = llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 0);
4430 }
4431 LValue Base;
4432 if (unsigned *P = dyn_cast<unsigned *>(Val&: Pos)) {
4433 Base = CGF.MakeAddrLValue(
4434 Addr: CGF.Builder.CreateConstGEP(Addr: DependenciesArray, Index: *P), T: KmpDependInfoTy);
4435 } else {
4436 assert(E && "Expected a non-null expression");
4437 LValue &PosLVal = *cast<LValue *>(Val&: Pos);
4438 llvm::Value *Idx = CGF.EmitLoadOfScalar(lvalue: PosLVal, Loc: E->getExprLoc());
4439 Base = CGF.MakeAddrLValue(
4440 Addr: CGF.Builder.CreateGEP(CGF, Addr: DependenciesArray, Index: Idx), T: KmpDependInfoTy);
4441 }
4442 // deps[i].base_addr = &<Dependencies[i].second>;
4443 LValue BaseAddrLVal = CGF.EmitLValueForField(
4444 Base,
4445 Field: *std::next(x: KmpDependInfoRD->field_begin(),
4446 n: static_cast<unsigned int>(RTLDependInfoFields::BaseAddr)));
4447 CGF.EmitStoreOfScalar(value: Addr, lvalue: BaseAddrLVal);
4448 // deps[i].len = sizeof(<Dependencies[i].second>);
4449 LValue LenLVal = CGF.EmitLValueForField(
4450 Base, Field: *std::next(x: KmpDependInfoRD->field_begin(),
4451 n: static_cast<unsigned int>(RTLDependInfoFields::Len)));
4452 CGF.EmitStoreOfScalar(value: Size, lvalue: LenLVal);
4453 // deps[i].flags = <Dependencies[i].first>;
4454 RTLDependenceKindTy DepKind = translateDependencyKind(K: Data.DepKind);
4455 LValue FlagsLVal = CGF.EmitLValueForField(
4456 Base,
4457 Field: *std::next(x: KmpDependInfoRD->field_begin(),
4458 n: static_cast<unsigned int>(RTLDependInfoFields::Flags)));
4459 CGF.EmitStoreOfScalar(
4460 value: llvm::ConstantInt::get(Ty: LLVMFlagsTy, V: static_cast<unsigned int>(DepKind)),
4461 lvalue: FlagsLVal);
4462 if (unsigned *P = dyn_cast<unsigned *>(Val&: Pos)) {
4463 ++(*P);
4464 } else {
4465 LValue &PosLVal = *cast<LValue *>(Val&: Pos);
4466 llvm::Value *Idx = CGF.EmitLoadOfScalar(lvalue: PosLVal, Loc: E->getExprLoc());
4467 Idx = CGF.Builder.CreateNUWAdd(LHS: Idx,
4468 RHS: llvm::ConstantInt::get(Ty: Idx->getType(), V: 1));
4469 CGF.EmitStoreOfScalar(value: Idx, lvalue: PosLVal);
4470 }
4471 }
4472}
4473
4474SmallVector<llvm::Value *, 4> CGOpenMPRuntime::emitDepobjElementsSizes(
4475 CodeGenFunction &CGF, QualType &KmpDependInfoTy,
4476 const OMPTaskDataTy::DependData &Data) {
4477 assert(Data.DepKind == OMPC_DEPEND_depobj &&
4478 "Expected depobj dependency kind.");
4479 SmallVector<llvm::Value *, 4> Sizes;
4480 SmallVector<LValue, 4> SizeLVals;
4481 ASTContext &C = CGF.getContext();
4482 {
4483 OMPIteratorGeneratorScope IteratorScope(
4484 CGF, cast_or_null<OMPIteratorExpr>(
4485 Val: Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4486 : nullptr));
4487 for (const Expr *E : Data.DepExprs) {
4488 llvm::Value *NumDeps;
4489 LValue Base;
4490 LValue DepobjLVal = CGF.EmitLValue(E: E->IgnoreParenImpCasts());
4491 std::tie(args&: NumDeps, args&: Base) =
4492 getDepobjElements(CGF, DepobjLVal, Loc: E->getExprLoc());
4493 LValue NumLVal = CGF.MakeAddrLValue(
4494 Addr: CGF.CreateMemTempWithoutCast(T: C.getUIntPtrType(), Name: "depobj.size.addr"),
4495 T: C.getUIntPtrType());
4496 CGF.Builder.CreateStore(Val: llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: 0),
4497 Addr: NumLVal.getAddress());
4498 llvm::Value *PrevVal = CGF.EmitLoadOfScalar(lvalue: NumLVal, Loc: E->getExprLoc());
4499 llvm::Value *Add = CGF.Builder.CreateNUWAdd(LHS: PrevVal, RHS: NumDeps);
4500 CGF.EmitStoreOfScalar(value: Add, lvalue: NumLVal);
4501 SizeLVals.push_back(Elt: NumLVal);
4502 }
4503 }
4504 for (unsigned I = 0, E = SizeLVals.size(); I < E; ++I) {
4505 llvm::Value *Size =
4506 CGF.EmitLoadOfScalar(lvalue: SizeLVals[I], Loc: Data.DepExprs[I]->getExprLoc());
4507 Sizes.push_back(Elt: Size);
4508 }
4509 return Sizes;
4510}
4511
4512void CGOpenMPRuntime::emitDepobjElements(CodeGenFunction &CGF,
4513 QualType &KmpDependInfoTy,
4514 LValue PosLVal,
4515 const OMPTaskDataTy::DependData &Data,
4516 Address DependenciesArray) {
4517 assert(Data.DepKind == OMPC_DEPEND_depobj &&
4518 "Expected depobj dependency kind.");
4519 llvm::Value *ElSize = CGF.getTypeSize(Ty: KmpDependInfoTy);
4520 {
4521 OMPIteratorGeneratorScope IteratorScope(
4522 CGF, cast_or_null<OMPIteratorExpr>(
4523 Val: Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4524 : nullptr));
4525 for (const Expr *E : Data.DepExprs) {
4526 llvm::Value *NumDeps;
4527 LValue Base;
4528 LValue DepobjLVal = CGF.EmitLValue(E: E->IgnoreParenImpCasts());
4529 std::tie(args&: NumDeps, args&: Base) =
4530 getDepobjElements(CGF, DepobjLVal, Loc: E->getExprLoc());
4531
4532 // memcopy dependency data.
4533 llvm::Value *Size = CGF.Builder.CreateNUWMul(
4534 LHS: ElSize,
4535 RHS: CGF.Builder.CreateIntCast(V: NumDeps, DestTy: CGF.SizeTy, /*isSigned=*/false));
4536 llvm::Value *Pos = CGF.EmitLoadOfScalar(lvalue: PosLVal, Loc: E->getExprLoc());
4537 Address DepAddr = CGF.Builder.CreateGEP(CGF, Addr: DependenciesArray, Index: Pos);
4538 CGF.Builder.CreateMemCpy(Dest: DepAddr, Src: Base.getAddress(), Size);
4539
4540 // Increase pos.
4541 // pos += size;
4542 llvm::Value *Add = CGF.Builder.CreateNUWAdd(LHS: Pos, RHS: NumDeps);
4543 CGF.EmitStoreOfScalar(value: Add, lvalue: PosLVal);
4544 }
4545 }
4546}
4547
4548std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause(
4549 CodeGenFunction &CGF, ArrayRef<OMPTaskDataTy::DependData> Dependencies,
4550 SourceLocation Loc) {
4551 if (llvm::all_of(Range&: Dependencies, P: [](const OMPTaskDataTy::DependData &D) {
4552 return D.DepExprs.empty();
4553 }))
4554 return std::make_pair(x: nullptr, y: Address::invalid());
4555 // Process list of dependencies.
4556 ASTContext &C = CGM.getContext();
4557 Address DependenciesArray = Address::invalid();
4558 llvm::Value *NumOfElements = nullptr;
4559 unsigned NumDependencies = std::accumulate(
4560 first: Dependencies.begin(), last: Dependencies.end(), init: 0,
4561 binary_op: [](unsigned V, const OMPTaskDataTy::DependData &D) {
4562 return D.DepKind == OMPC_DEPEND_depobj
4563 ? V
4564 : (V + (D.IteratorExpr ? 0 : D.DepExprs.size()));
4565 });
4566 QualType FlagsTy;
4567 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4568 bool HasDepobjDeps = false;
4569 bool HasRegularWithIterators = false;
4570 llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: 0);
4571 llvm::Value *NumOfRegularWithIterators =
4572 llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: 0);
4573 // Calculate number of depobj dependencies and regular deps with the
4574 // iterators.
4575 for (const OMPTaskDataTy::DependData &D : Dependencies) {
4576 if (D.DepKind == OMPC_DEPEND_depobj) {
4577 SmallVector<llvm::Value *, 4> Sizes =
4578 emitDepobjElementsSizes(CGF, KmpDependInfoTy, Data: D);
4579 for (llvm::Value *Size : Sizes) {
4580 NumOfDepobjElements =
4581 CGF.Builder.CreateNUWAdd(LHS: NumOfDepobjElements, RHS: Size);
4582 }
4583 HasDepobjDeps = true;
4584 continue;
4585 }
4586 // Include number of iterations, if any.
4587
4588 if (const auto *IE = cast_or_null<OMPIteratorExpr>(Val: D.IteratorExpr)) {
4589 llvm::Value *ClauseIteratorSpace =
4590 llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: 1);
4591 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4592 llvm::Value *Sz = CGF.EmitScalarExpr(E: IE->getHelper(I).Upper);
4593 Sz = CGF.Builder.CreateIntCast(V: Sz, DestTy: CGF.IntPtrTy, /*isSigned=*/false);
4594 ClauseIteratorSpace = CGF.Builder.CreateNUWMul(LHS: Sz, RHS: ClauseIteratorSpace);
4595 }
4596 llvm::Value *NumClauseDeps = CGF.Builder.CreateNUWMul(
4597 LHS: ClauseIteratorSpace,
4598 RHS: llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: D.DepExprs.size()));
4599 NumOfRegularWithIterators =
4600 CGF.Builder.CreateNUWAdd(LHS: NumOfRegularWithIterators, RHS: NumClauseDeps);
4601 HasRegularWithIterators = true;
4602 continue;
4603 }
4604 }
4605
4606 QualType KmpDependInfoArrayTy;
4607 if (HasDepobjDeps || HasRegularWithIterators) {
4608 NumOfElements = llvm::ConstantInt::get(Ty: CGM.IntPtrTy, V: NumDependencies,
4609 /*isSigned=*/IsSigned: false);
4610 if (HasDepobjDeps) {
4611 NumOfElements =
4612 CGF.Builder.CreateNUWAdd(LHS: NumOfDepobjElements, RHS: NumOfElements);
4613 }
4614 if (HasRegularWithIterators) {
4615 NumOfElements =
4616 CGF.Builder.CreateNUWAdd(LHS: NumOfRegularWithIterators, RHS: NumOfElements);
4617 }
4618 auto *OVE = new (C) OpaqueValueExpr(
4619 Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0),
4620 VK_PRValue);
4621 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE,
4622 RValue::get(V: NumOfElements));
4623 KmpDependInfoArrayTy =
4624 C.getVariableArrayType(EltTy: KmpDependInfoTy, NumElts: OVE, ASM: ArraySizeModifier::Normal,
4625 /*IndexTypeQuals=*/0);
4626 // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy);
4627 // Properly emit variable-sized array.
4628 auto *PD = ImplicitParamDecl::Create(C, T: KmpDependInfoArrayTy,
4629 ParamKind: ImplicitParamKind::Other);
4630 CGF.EmitVarDecl(D: *PD);
4631 DependenciesArray = CGF.GetAddrOfLocalVar(VD: PD);
4632 NumOfElements = CGF.Builder.CreateIntCast(V: NumOfElements, DestTy: CGF.Int32Ty,
4633 /*isSigned=*/false);
4634 } else {
4635 KmpDependInfoArrayTy = C.getConstantArrayType(
4636 EltTy: KmpDependInfoTy, ArySize: llvm::APInt(/*numBits=*/64, NumDependencies), SizeExpr: nullptr,
4637 ASM: ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
4638 DependenciesArray =
4639 CGF.CreateMemTempWithoutCast(T: KmpDependInfoArrayTy, Name: ".dep.arr.addr");
4640 DependenciesArray = CGF.Builder.CreateConstArrayGEP(Addr: DependenciesArray, Index: 0);
4641 NumOfElements = llvm::ConstantInt::get(Ty: CGM.Int32Ty, V: NumDependencies,
4642 /*isSigned=*/IsSigned: false);
4643 }
4644 unsigned Pos = 0;
4645 for (const OMPTaskDataTy::DependData &Dep : Dependencies) {
4646 if (Dep.DepKind == OMPC_DEPEND_depobj || Dep.IteratorExpr)
4647 continue;
4648 emitDependData(CGF, KmpDependInfoTy, Pos: &Pos, Data: Dep, DependenciesArray);
4649 }
4650 // Copy regular dependencies with iterators.
4651 LValue PosLVal = CGF.MakeAddrLValue(
4652 Addr: CGF.CreateMemTempWithoutCast(T: C.getSizeType(), Name: "dep.counter.addr"),
4653 T: C.getSizeType());
4654 CGF.EmitStoreOfScalar(value: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: Pos), lvalue: PosLVal);
4655 for (const OMPTaskDataTy::DependData &Dep : Dependencies) {
4656 if (Dep.DepKind == OMPC_DEPEND_depobj || !Dep.IteratorExpr)
4657 continue;
4658 emitDependData(CGF, KmpDependInfoTy, Pos: &PosLVal, Data: Dep, DependenciesArray);
4659 }
4660 // Copy final depobj arrays without iterators.
4661 if (HasDepobjDeps) {
4662 for (const OMPTaskDataTy::DependData &Dep : Dependencies) {
4663 if (Dep.DepKind != OMPC_DEPEND_depobj)
4664 continue;
4665 emitDepobjElements(CGF, KmpDependInfoTy, PosLVal, Data: Dep, DependenciesArray);
4666 }
4667 }
4668 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4669 Addr: DependenciesArray, Ty: CGF.VoidPtrTy, ElementTy: CGF.Int8Ty);
4670 return std::make_pair(x&: NumOfElements, y&: DependenciesArray);
4671}
4672
4673Address CGOpenMPRuntime::emitDepobjDependClause(
4674 CodeGenFunction &CGF, const OMPTaskDataTy::DependData &Dependencies,
4675 SourceLocation Loc) {
4676 if (Dependencies.DepExprs.empty())
4677 return Address::invalid();
4678 // Process list of dependencies.
4679 ASTContext &C = CGM.getContext();
4680 Address DependenciesArray = Address::invalid();
4681 unsigned NumDependencies = Dependencies.DepExprs.size();
4682 QualType FlagsTy;
4683 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4684 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4685
4686 llvm::Value *Size;
4687 // Define type kmp_depend_info[<Dependencies.size()>];
4688 // For depobj reserve one extra element to store the number of elements.
4689 // It is required to handle depobj(x) update(in) construct.
4690 // kmp_depend_info[<Dependencies.size()>] deps;
4691 llvm::Value *NumDepsVal;
4692 CharUnits Align = C.getTypeAlignInChars(T: KmpDependInfoTy);
4693 if (const auto *IE =
4694 cast_or_null<OMPIteratorExpr>(Val: Dependencies.IteratorExpr)) {
4695 NumDepsVal = llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 1);
4696 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4697 llvm::Value *Sz = CGF.EmitScalarExpr(E: IE->getHelper(I).Upper);
4698 Sz = CGF.Builder.CreateIntCast(V: Sz, DestTy: CGF.SizeTy, /*isSigned=*/false);
4699 NumDepsVal = CGF.Builder.CreateNUWMul(LHS: NumDepsVal, RHS: Sz);
4700 }
4701 Size = CGF.Builder.CreateNUWAdd(LHS: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 1),
4702 RHS: NumDepsVal);
4703 CharUnits SizeInBytes =
4704 C.getTypeSizeInChars(T: KmpDependInfoTy).alignTo(Align);
4705 llvm::Value *RecSize = CGM.getSize(numChars: SizeInBytes);
4706 Size = CGF.Builder.CreateNUWMul(LHS: Size, RHS: RecSize);
4707 NumDepsVal =
4708 CGF.Builder.CreateIntCast(V: NumDepsVal, DestTy: CGF.IntPtrTy, /*isSigned=*/false);
4709 } else {
4710 QualType KmpDependInfoArrayTy = C.getConstantArrayType(
4711 EltTy: KmpDependInfoTy, ArySize: llvm::APInt(/*numBits=*/64, NumDependencies + 1),
4712 SizeExpr: nullptr, ASM: ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
4713 CharUnits Sz = C.getTypeSizeInChars(T: KmpDependInfoArrayTy);
4714 Size = CGM.getSize(numChars: Sz.alignTo(Align));
4715 NumDepsVal = llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: NumDependencies);
4716 }
4717 // Need to allocate on the dynamic memory.
4718 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4719 // Use default allocator.
4720 llvm::Value *Allocator = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
4721 llvm::Value *Args[] = {ThreadID, Size, Allocator};
4722
4723 llvm::Value *Addr =
4724 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
4725 M&: CGM.getModule(), FnID: OMPRTL___kmpc_alloc),
4726 args: Args, name: ".dep.arr.addr");
4727 llvm::Type *KmpDependInfoLlvmTy = CGF.ConvertTypeForMem(T: KmpDependInfoTy);
4728 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4729 V: Addr, DestTy: CGF.Builder.getPtrTy(AddrSpace: 0));
4730 DependenciesArray = Address(Addr, KmpDependInfoLlvmTy, Align);
4731 // Write number of elements in the first element of array for depobj.
4732 LValue Base = CGF.MakeAddrLValue(Addr: DependenciesArray, T: KmpDependInfoTy);
4733 // deps[i].base_addr = NumDependencies;
4734 LValue BaseAddrLVal = CGF.EmitLValueForField(
4735 Base,
4736 Field: *std::next(x: KmpDependInfoRD->field_begin(),
4737 n: static_cast<unsigned int>(RTLDependInfoFields::BaseAddr)));
4738 CGF.EmitStoreOfScalar(value: NumDepsVal, lvalue: BaseAddrLVal);
4739 llvm::PointerUnion<unsigned *, LValue *> Pos;
4740 unsigned Idx = 1;
4741 LValue PosLVal;
4742 if (Dependencies.IteratorExpr) {
4743 PosLVal = CGF.MakeAddrLValue(
4744 Addr: CGF.CreateMemTempWithoutCast(T: C.getSizeType(), Name: "iterator.counter.addr"),
4745 T: C.getSizeType());
4746 CGF.EmitStoreOfScalar(value: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: Idx), lvalue: PosLVal,
4747 /*IsInit=*/isInit: true);
4748 Pos = &PosLVal;
4749 } else {
4750 Pos = &Idx;
4751 }
4752 emitDependData(CGF, KmpDependInfoTy, Pos, Data: Dependencies, DependenciesArray);
4753 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4754 Addr: CGF.Builder.CreateConstGEP(Addr: DependenciesArray, Index: 1), Ty: CGF.VoidPtrTy,
4755 ElementTy: CGF.Int8Ty);
4756 return DependenciesArray;
4757}
4758
4759void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal,
4760 SourceLocation Loc) {
4761 ASTContext &C = CGM.getContext();
4762 QualType FlagsTy;
4763 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4764 LValue Base = CGF.EmitLoadOfPointerLValue(Ptr: DepobjLVal.getAddress(),
4765 PtrTy: C.VoidPtrTy.castAs<PointerType>());
4766 QualType KmpDependInfoPtrTy = C.getPointerType(T: KmpDependInfoTy);
4767 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4768 Addr: Base.getAddress(), Ty: CGF.ConvertTypeForMem(T: KmpDependInfoPtrTy),
4769 ElementTy: CGF.ConvertTypeForMem(T: KmpDependInfoTy));
4770 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP(
4771 Ty: Addr.getElementType(), Ptr: Addr.emitRawPointer(CGF),
4772 IdxList: llvm::ConstantInt::get(Ty: CGF.IntPtrTy, V: -1, /*isSigned=*/IsSigned: true));
4773 DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(V: DepObjAddr,
4774 DestTy: CGF.VoidPtrTy);
4775 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4776 // Use default allocator.
4777 llvm::Value *Allocator = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
4778 llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator};
4779
4780 // _kmpc_free(gtid, addr, nullptr);
4781 (void)CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
4782 M&: CGM.getModule(), FnID: OMPRTL___kmpc_free),
4783 args: Args);
4784}
4785
4786void CGOpenMPRuntime::emitUpdateDependObjectsClause(
4787 CodeGenFunction &CGF, LValue DepobjLVal, OpenMPDependClauseKind NewDepKind,
4788 SourceLocation Loc) {
4789 ASTContext &C = CGM.getContext();
4790 QualType FlagsTy;
4791 getDependTypes(C, KmpDependInfoTy, FlagsTy);
4792 auto *KmpDependInfoRD = KmpDependInfoTy->castAsRecordDecl();
4793 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(T: FlagsTy);
4794 llvm::Value *NumDeps;
4795 LValue Base;
4796 std::tie(args&: NumDeps, args&: Base) = getDepobjElements(CGF, DepobjLVal, Loc);
4797
4798 Address Begin = Base.getAddress();
4799 // Cast from pointer to array type to pointer to single element.
4800 llvm::Value *End = CGF.Builder.CreateGEP(Ty: Begin.getElementType(),
4801 Ptr: Begin.emitRawPointer(CGF), IdxList: NumDeps);
4802 // The basic structure here is a while-do loop.
4803 llvm::BasicBlock *BodyBB = CGF.createBasicBlock(name: "omp.body");
4804 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(name: "omp.done");
4805 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
4806 CGF.EmitBlock(BB: BodyBB);
4807 llvm::PHINode *ElementPHI =
4808 CGF.Builder.CreatePHI(Ty: Begin.getType(), NumReservedValues: 2, Name: "omp.elementPast");
4809 ElementPHI->addIncoming(V: Begin.emitRawPointer(CGF), BB: EntryBB);
4810 Begin = Begin.withPointer(NewPointer: ElementPHI, IsKnownNonNull: KnownNonNull);
4811 Base = CGF.MakeAddrLValue(Addr: Begin, T: KmpDependInfoTy, BaseInfo: Base.getBaseInfo(),
4812 TBAAInfo: Base.getTBAAInfo());
4813 // deps[i].flags = NewDepKind;
4814 RTLDependenceKindTy DepKind = translateDependencyKind(K: NewDepKind);
4815 LValue FlagsLVal = CGF.EmitLValueForField(
4816 Base, Field: *std::next(x: KmpDependInfoRD->field_begin(),
4817 n: static_cast<unsigned int>(RTLDependInfoFields::Flags)));
4818 CGF.EmitStoreOfScalar(
4819 value: llvm::ConstantInt::get(Ty: LLVMFlagsTy, V: static_cast<unsigned int>(DepKind)),
4820 lvalue: FlagsLVal);
4821
4822 // Shift the address forward by one element.
4823 llvm::Value *ElementNext =
4824 CGF.Builder.CreateConstGEP(Addr: Begin, /*Index=*/1, Name: "omp.elementNext")
4825 .emitRawPointer(CGF);
4826 ElementPHI->addIncoming(V: ElementNext, BB: CGF.Builder.GetInsertBlock());
4827 llvm::Value *IsEmpty =
4828 CGF.Builder.CreateICmpEQ(LHS: ElementNext, RHS: End, Name: "omp.isempty");
4829 CGF.Builder.CreateCondBr(Cond: IsEmpty, True: DoneBB, False: BodyBB);
4830 // Done.
4831 CGF.EmitBlock(BB: DoneBB, /*IsFinished=*/true);
4832}
4833
4834void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
4835 const OMPExecutableDirective &D,
4836 llvm::Function *TaskFunction,
4837 QualType SharedsTy, Address Shareds,
4838 const Expr *IfCond,
4839 const OMPTaskDataTy &Data) {
4840 if (!CGF.HaveInsertPoint())
4841 return;
4842
4843 TaskResultTy Result =
4844 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
4845 llvm::Value *NewTask = Result.NewTask;
4846 llvm::Function *TaskEntry = Result.TaskEntry;
4847 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy;
4848 LValue TDBase = Result.TDBase;
4849 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD;
4850 // Process list of dependences.
4851 Address DependenciesArray = Address::invalid();
4852 llvm::Value *NumOfElements;
4853 std::tie(args&: NumOfElements, args&: DependenciesArray) =
4854 emitDependClause(CGF, Dependencies: Data.Dependences, Loc);
4855
4856 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
4857 // libcall.
4858 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
4859 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
4860 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence
4861 // list is not empty
4862 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4863 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
4864 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask };
4865 llvm::Value *DepTaskArgs[7];
4866 if (!Data.Dependences.empty()) {
4867 DepTaskArgs[0] = UpLoc;
4868 DepTaskArgs[1] = ThreadID;
4869 DepTaskArgs[2] = NewTask;
4870 DepTaskArgs[3] = NumOfElements;
4871 DepTaskArgs[4] = DependenciesArray.emitRawPointer(CGF);
4872 DepTaskArgs[5] = CGF.Builder.getInt32(C: 0);
4873 DepTaskArgs[6] = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
4874 }
4875 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs,
4876 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) {
4877 if (!Data.Tied) {
4878 auto PartIdFI = std::next(x: KmpTaskTQTyRD->field_begin(), n: KmpTaskTPartId);
4879 LValue PartIdLVal = CGF.EmitLValueForField(Base: TDBase, Field: *PartIdFI);
4880 CGF.EmitStoreOfScalar(value: CGF.Builder.getInt32(C: 0), lvalue: PartIdLVal);
4881 }
4882 if (!Data.Dependences.empty()) {
4883 CGF.EmitRuntimeCall(
4884 callee: OMPBuilder.getOrCreateRuntimeFunction(
4885 M&: CGM.getModule(), FnID: OMPRTL___kmpc_omp_task_with_deps),
4886 args: DepTaskArgs);
4887 } else {
4888 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
4889 M&: CGM.getModule(), FnID: OMPRTL___kmpc_omp_task),
4890 args: TaskArgs);
4891 }
4892 // Check if parent region is untied and build return for untied task;
4893 if (auto *Region =
4894 dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo))
4895 Region->emitUntiedSwitch(CGF);
4896 };
4897
4898 llvm::Value *DepWaitTaskArgs[7];
4899 if (!Data.Dependences.empty()) {
4900 DepWaitTaskArgs[0] = UpLoc;
4901 DepWaitTaskArgs[1] = ThreadID;
4902 DepWaitTaskArgs[2] = NumOfElements;
4903 DepWaitTaskArgs[3] = DependenciesArray.emitRawPointer(CGF);
4904 DepWaitTaskArgs[4] = CGF.Builder.getInt32(C: 0);
4905 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
4906 DepWaitTaskArgs[6] =
4907 llvm::ConstantInt::get(Ty: CGF.Int32Ty, V: Data.HasNowaitClause);
4908 }
4909 auto &M = CGM.getModule();
4910 auto &&ElseCodeGen = [this, &M, &TaskArgs, ThreadID, NewTaskNewTaskTTy,
4911 TaskEntry, &Data, &DepWaitTaskArgs,
4912 Loc](CodeGenFunction &CGF, PrePostActionTy &) {
4913 CodeGenFunction::RunCleanupsScope LocalScope(CGF);
4914 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
4915 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
4916 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
4917 // is specified.
4918 if (!Data.Dependences.empty())
4919 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
4920 M, FnID: OMPRTL___kmpc_omp_taskwait_deps_51),
4921 args: DepWaitTaskArgs);
4922 // Call proxy_task_entry(gtid, new_task);
4923 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy,
4924 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
4925 Action.Enter(CGF);
4926 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
4927 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, OutlinedFn: TaskEntry,
4928 Args: OutlinedFnArgs);
4929 };
4930
4931 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
4932 // kmp_task_t *new_task);
4933 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
4934 // kmp_task_t *new_task);
4935 RegionCodeGenTy RCG(CodeGen);
4936 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
4937 M, FnID: OMPRTL___kmpc_omp_task_begin_if0),
4938 TaskArgs,
4939 OMPBuilder.getOrCreateRuntimeFunction(
4940 M, FnID: OMPRTL___kmpc_omp_task_complete_if0),
4941 TaskArgs);
4942 RCG.setAction(Action);
4943 RCG(CGF);
4944 };
4945
4946 if (IfCond) {
4947 emitIfClause(CGF, Cond: IfCond, ThenGen: ThenCodeGen, ElseGen: ElseCodeGen);
4948 } else {
4949 RegionCodeGenTy ThenRCG(ThenCodeGen);
4950 ThenRCG(CGF);
4951 }
4952}
4953
4954void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc,
4955 const OMPLoopDirective &D,
4956 llvm::Function *TaskFunction,
4957 QualType SharedsTy, Address Shareds,
4958 const Expr *IfCond,
4959 const OMPTaskDataTy &Data) {
4960 if (!CGF.HaveInsertPoint())
4961 return;
4962 TaskResultTy Result =
4963 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
4964 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
4965 // libcall.
4966 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
4967 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
4968 // sched, kmp_uint64 grainsize, void *task_dup);
4969 llvm::Value *ThreadID = getThreadID(CGF, Loc);
4970 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
4971 llvm::Value *IfVal;
4972 if (IfCond) {
4973 IfVal = CGF.Builder.CreateIntCast(V: CGF.EvaluateExprAsBool(E: IfCond), DestTy: CGF.IntTy,
4974 /*isSigned=*/true);
4975 } else {
4976 IfVal = llvm::ConstantInt::getSigned(Ty: CGF.IntTy, /*V=*/1);
4977 }
4978
4979 LValue LBLVal = CGF.EmitLValueForField(
4980 Base: Result.TDBase,
4981 Field: *std::next(x: Result.KmpTaskTQTyRD->field_begin(), n: KmpTaskTLowerBound));
4982 const auto *LBVar =
4983 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D.getLowerBoundVariable())->getDecl());
4984 CGF.EmitAnyExprToMem(E: LBVar->getInit(), Location: LBLVal.getAddress(), Quals: LBLVal.getQuals(),
4985 /*IsInitializer=*/true);
4986 LValue UBLVal = CGF.EmitLValueForField(
4987 Base: Result.TDBase,
4988 Field: *std::next(x: Result.KmpTaskTQTyRD->field_begin(), n: KmpTaskTUpperBound));
4989 const auto *UBVar =
4990 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D.getUpperBoundVariable())->getDecl());
4991 CGF.EmitAnyExprToMem(E: UBVar->getInit(), Location: UBLVal.getAddress(), Quals: UBLVal.getQuals(),
4992 /*IsInitializer=*/true);
4993 LValue StLVal = CGF.EmitLValueForField(
4994 Base: Result.TDBase,
4995 Field: *std::next(x: Result.KmpTaskTQTyRD->field_begin(), n: KmpTaskTStride));
4996 const auto *StVar =
4997 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D.getStrideVariable())->getDecl());
4998 CGF.EmitAnyExprToMem(E: StVar->getInit(), Location: StLVal.getAddress(), Quals: StLVal.getQuals(),
4999 /*IsInitializer=*/true);
5000 // Store reductions address.
5001 LValue RedLVal = CGF.EmitLValueForField(
5002 Base: Result.TDBase,
5003 Field: *std::next(x: Result.KmpTaskTQTyRD->field_begin(), n: KmpTaskTReductions));
5004 if (Data.Reductions) {
5005 CGF.EmitStoreOfScalar(value: Data.Reductions, lvalue: RedLVal);
5006 } else {
5007 CGF.EmitNullInitialization(DestPtr: RedLVal.getAddress(),
5008 Ty: CGF.getContext().VoidPtrTy);
5009 }
5010 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 };
5011 llvm::SmallVector<llvm::Value *, 12> TaskArgs{
5012 UpLoc,
5013 ThreadID,
5014 Result.NewTask,
5015 IfVal,
5016 LBLVal.getPointer(CGF),
5017 UBLVal.getPointer(CGF),
5018 CGF.EmitLoadOfScalar(lvalue: StLVal, Loc),
5019 llvm::ConstantInt::getSigned(
5020 Ty: CGF.IntTy, V: 1), // Always 1 because taskgroup emitted by the compiler
5021 llvm::ConstantInt::getSigned(
5022 Ty: CGF.IntTy, V: Data.Schedule.getPointer()
5023 ? Data.Schedule.getInt() ? NumTasks : Grainsize
5024 : NoSchedule),
5025 Data.Schedule.getPointer()
5026 ? CGF.Builder.CreateIntCast(V: Data.Schedule.getPointer(), DestTy: CGF.Int64Ty,
5027 /*isSigned=*/false)
5028 : llvm::ConstantInt::get(Ty: CGF.Int64Ty, /*V=*/0)};
5029 if (Data.HasModifier)
5030 TaskArgs.push_back(Elt: llvm::ConstantInt::get(Ty: CGF.Int32Ty, V: 1));
5031
5032 TaskArgs.push_back(Elt: Result.TaskDupFn
5033 ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5034 V: Result.TaskDupFn, DestTy: CGF.VoidPtrTy)
5035 : llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy));
5036 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
5037 M&: CGM.getModule(), FnID: Data.HasModifier
5038 ? OMPRTL___kmpc_taskloop_5
5039 : OMPRTL___kmpc_taskloop),
5040 args: TaskArgs);
5041}
5042
5043/// Emit reduction operation for each element of array (required for
5044/// array sections) LHS op = RHS.
5045/// \param Type Type of array.
5046/// \param LHSVar Variable on the left side of the reduction operation
5047/// (references element of array in original variable).
5048/// \param RHSVar Variable on the right side of the reduction operation
5049/// (references element of array in original variable).
5050/// \param RedOpGen Generator of reduction operation with use of LHSVar and
5051/// RHSVar.
5052static void EmitOMPAggregateReduction(
5053 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar,
5054 const VarDecl *RHSVar,
5055 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *,
5056 const Expr *, const Expr *)> &RedOpGen,
5057 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr,
5058 const Expr *UpExpr = nullptr) {
5059 // Perform element-by-element initialization.
5060 QualType ElementTy;
5061 Address LHSAddr = CGF.GetAddrOfLocalVar(VD: LHSVar);
5062 Address RHSAddr = CGF.GetAddrOfLocalVar(VD: RHSVar);
5063
5064 // Drill down to the base element type on both arrays.
5065 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
5066 llvm::Value *NumElements = CGF.emitArrayLength(arrayType: ArrayTy, baseType&: ElementTy, addr&: LHSAddr);
5067
5068 llvm::Value *RHSBegin = RHSAddr.emitRawPointer(CGF);
5069 llvm::Value *LHSBegin = LHSAddr.emitRawPointer(CGF);
5070 // Cast from pointer to array type to pointer to single element.
5071 llvm::Value *LHSEnd =
5072 CGF.Builder.CreateGEP(Ty: LHSAddr.getElementType(), Ptr: LHSBegin, IdxList: NumElements);
5073 // The basic structure here is a while-do loop.
5074 llvm::BasicBlock *BodyBB = CGF.createBasicBlock(name: "omp.arraycpy.body");
5075 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(name: "omp.arraycpy.done");
5076 llvm::Value *IsEmpty =
5077 CGF.Builder.CreateICmpEQ(LHS: LHSBegin, RHS: LHSEnd, Name: "omp.arraycpy.isempty");
5078 CGF.Builder.CreateCondBr(Cond: IsEmpty, True: DoneBB, False: BodyBB);
5079
5080 // Enter the loop body, making that address the current address.
5081 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5082 CGF.EmitBlock(BB: BodyBB);
5083
5084 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(T: ElementTy);
5085
5086 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI(
5087 Ty: RHSBegin->getType(), NumReservedValues: 2, Name: "omp.arraycpy.srcElementPast");
5088 RHSElementPHI->addIncoming(V: RHSBegin, BB: EntryBB);
5089 Address RHSElementCurrent(
5090 RHSElementPHI, RHSAddr.getElementType(),
5091 RHSAddr.getAlignment().alignmentOfArrayElement(elementSize: ElementSize));
5092
5093 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI(
5094 Ty: LHSBegin->getType(), NumReservedValues: 2, Name: "omp.arraycpy.destElementPast");
5095 LHSElementPHI->addIncoming(V: LHSBegin, BB: EntryBB);
5096 Address LHSElementCurrent(
5097 LHSElementPHI, LHSAddr.getElementType(),
5098 LHSAddr.getAlignment().alignmentOfArrayElement(elementSize: ElementSize));
5099
5100 // Emit copy.
5101 CodeGenFunction::OMPPrivateScope Scope(CGF);
5102 Scope.addPrivate(LocalVD: LHSVar, Addr: LHSElementCurrent);
5103 Scope.addPrivate(LocalVD: RHSVar, Addr: RHSElementCurrent);
5104 Scope.Privatize();
5105 RedOpGen(CGF, XExpr, EExpr, UpExpr);
5106 Scope.ForceCleanup();
5107
5108 // Shift the address forward by one element.
5109 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32(
5110 Ty: LHSAddr.getElementType(), Ptr: LHSElementPHI, /*Idx0=*/1,
5111 Name: "omp.arraycpy.dest.element");
5112 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32(
5113 Ty: RHSAddr.getElementType(), Ptr: RHSElementPHI, /*Idx0=*/1,
5114 Name: "omp.arraycpy.src.element");
5115 // Check whether we've reached the end.
5116 llvm::Value *Done =
5117 CGF.Builder.CreateICmpEQ(LHS: LHSElementNext, RHS: LHSEnd, Name: "omp.arraycpy.done");
5118 CGF.Builder.CreateCondBr(Cond: Done, True: DoneBB, False: BodyBB);
5119 LHSElementPHI->addIncoming(V: LHSElementNext, BB: CGF.Builder.GetInsertBlock());
5120 RHSElementPHI->addIncoming(V: RHSElementNext, BB: CGF.Builder.GetInsertBlock());
5121
5122 // Done.
5123 CGF.EmitBlock(BB: DoneBB, /*IsFinished=*/true);
5124}
5125
5126/// Emit reduction combiner. If the combiner is a simple expression emit it as
5127/// is, otherwise consider it as combiner of UDR decl and emit it as a call of
5128/// UDR combiner function.
5129static void emitReductionCombiner(CodeGenFunction &CGF,
5130 const Expr *ReductionOp) {
5131 if (const auto *CE = dyn_cast<CallExpr>(Val: ReductionOp))
5132 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(Val: CE->getCallee()))
5133 if (const auto *DRE =
5134 dyn_cast<DeclRefExpr>(Val: OVE->getSourceExpr()->IgnoreImpCasts()))
5135 if (const auto *DRD =
5136 dyn_cast<OMPDeclareReductionDecl>(Val: DRE->getDecl())) {
5137 std::pair<llvm::Function *, llvm::Function *> Reduction =
5138 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(D: DRD);
5139 RValue Func = RValue::get(V: Reduction.first);
5140 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
5141 CGF.EmitIgnoredExpr(E: ReductionOp);
5142 return;
5143 }
5144 CGF.EmitIgnoredExpr(E: ReductionOp);
5145}
5146
5147llvm::Function *CGOpenMPRuntime::emitReductionFunction(
5148 StringRef ReducerName, SourceLocation Loc, llvm::Type *ArgsElemType,
5149 ArrayRef<const Expr *> Privates, ArrayRef<const Expr *> LHSExprs,
5150 ArrayRef<const Expr *> RHSExprs, ArrayRef<const Expr *> ReductionOps) {
5151 ASTContext &C = CGM.getContext();
5152
5153 // void reduction_func(void *LHSArg, void *RHSArg);
5154 auto *LHSArg =
5155 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
5156 T: C.VoidPtrTy, ParamKind: ImplicitParamKind::Other);
5157 auto *RHSArg =
5158 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
5159 T: C.VoidPtrTy, ParamKind: ImplicitParamKind::Other);
5160 FunctionArgList Args{LHSArg, RHSArg};
5161 const auto &CGFI =
5162 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
5163 std::string Name = getReductionFuncName(Name: ReducerName);
5164 auto *Fn = llvm::Function::Create(Ty: CGM.getTypes().GetFunctionType(Info: CGFI),
5165 Linkage: llvm::GlobalValue::InternalLinkage, N: Name,
5166 M: &CGM.getModule());
5167 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: Fn, FI: CGFI);
5168 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
5169 Fn->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
5170 Fn->setDoesNotRecurse();
5171 CodeGenFunction CGF(CGM);
5172 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn, FnInfo: CGFI, Args, Loc, StartLoc: Loc);
5173
5174 // Dst = (void*[n])(LHSArg);
5175 // Src = (void*[n])(RHSArg);
5176 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5177 V: CGF.Builder.CreateLoad(Addr: CGF.GetAddrOfLocalVar(VD: LHSArg)),
5178 DestTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
5179 ArgsElemType, CGF.getPointerAlign());
5180 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5181 V: CGF.Builder.CreateLoad(Addr: CGF.GetAddrOfLocalVar(VD: RHSArg)),
5182 DestTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
5183 ArgsElemType, CGF.getPointerAlign());
5184
5185 // ...
5186 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
5187 // ...
5188 CodeGenFunction::OMPPrivateScope Scope(CGF);
5189 const auto *IPriv = Privates.begin();
5190 unsigned Idx = 0;
5191 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) {
5192 const auto *RHSVar =
5193 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: RHSExprs[I])->getDecl());
5194 Scope.addPrivate(LocalVD: RHSVar, Addr: emitAddrOfVarFromArray(CGF, Array: RHS, Index: Idx, Var: RHSVar));
5195 const auto *LHSVar =
5196 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: LHSExprs[I])->getDecl());
5197 Scope.addPrivate(LocalVD: LHSVar, Addr: emitAddrOfVarFromArray(CGF, Array: LHS, Index: Idx, Var: LHSVar));
5198 QualType PrivTy = (*IPriv)->getType();
5199 if (PrivTy->isVariablyModifiedType()) {
5200 // Get array size and emit VLA type.
5201 ++Idx;
5202 Address Elem = CGF.Builder.CreateConstArrayGEP(Addr: LHS, Index: Idx);
5203 llvm::Value *Ptr = CGF.Builder.CreateLoad(Addr: Elem);
5204 const VariableArrayType *VLA =
5205 CGF.getContext().getAsVariableArrayType(T: PrivTy);
5206 const auto *OVE = cast<OpaqueValueExpr>(Val: VLA->getSizeExpr());
5207 CodeGenFunction::OpaqueValueMapping OpaqueMap(
5208 CGF, OVE, RValue::get(V: CGF.Builder.CreatePtrToInt(V: Ptr, DestTy: CGF.SizeTy)));
5209 CGF.EmitVariablyModifiedType(Ty: PrivTy);
5210 }
5211 }
5212 Scope.Privatize();
5213 IPriv = Privates.begin();
5214 const auto *ILHS = LHSExprs.begin();
5215 const auto *IRHS = RHSExprs.begin();
5216 for (const Expr *E : ReductionOps) {
5217 if ((*IPriv)->getType()->isArrayType()) {
5218 // Emit reduction for array section.
5219 const auto *LHSVar = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ILHS)->getDecl());
5220 const auto *RHSVar = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IRHS)->getDecl());
5221 EmitOMPAggregateReduction(
5222 CGF, Type: (*IPriv)->getType(), LHSVar, RHSVar,
5223 RedOpGen: [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5224 emitReductionCombiner(CGF, ReductionOp: E);
5225 });
5226 } else {
5227 // Emit reduction for array subscript or single variable.
5228 emitReductionCombiner(CGF, ReductionOp: E);
5229 }
5230 ++IPriv;
5231 ++ILHS;
5232 ++IRHS;
5233 }
5234 Scope.ForceCleanup();
5235 CGF.FinishFunction();
5236 return Fn;
5237}
5238
5239void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF,
5240 const Expr *ReductionOp,
5241 const Expr *PrivateRef,
5242 const DeclRefExpr *LHS,
5243 const DeclRefExpr *RHS) {
5244 if (PrivateRef->getType()->isArrayType()) {
5245 // Emit reduction for array section.
5246 const auto *LHSVar = cast<VarDecl>(Val: LHS->getDecl());
5247 const auto *RHSVar = cast<VarDecl>(Val: RHS->getDecl());
5248 EmitOMPAggregateReduction(
5249 CGF, Type: PrivateRef->getType(), LHSVar, RHSVar,
5250 RedOpGen: [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5251 emitReductionCombiner(CGF, ReductionOp);
5252 });
5253 } else {
5254 // Emit reduction for array subscript or single variable.
5255 emitReductionCombiner(CGF, ReductionOp);
5256 }
5257}
5258
5259static std::string generateUniqueName(CodeGenModule &CGM,
5260 llvm::StringRef Prefix, const Expr *Ref);
5261
5262void CGOpenMPRuntime::emitPrivateReduction(
5263 CodeGenFunction &CGF, SourceLocation Loc, const Expr *Privates,
5264 const Expr *LHSExprs, const Expr *RHSExprs, const Expr *ReductionOps) {
5265
5266 // Create a shared global variable (__shared_reduction_var) to accumulate the
5267 // final result.
5268 //
5269 // Call __kmpc_barrier to synchronize threads before initialization.
5270 //
5271 // The master thread (thread_id == 0) initializes __shared_reduction_var
5272 // with the identity value or initializer.
5273 //
5274 // Call __kmpc_barrier to synchronize before combining.
5275 // For each i:
5276 // - Thread enters critical section.
5277 // - Reads its private value from LHSExprs[i].
5278 // - Updates __shared_reduction_var[i] = RedOp_i(__shared_reduction_var[i],
5279 // Privates[i]).
5280 // - Exits critical section.
5281 //
5282 // Call __kmpc_barrier after combining.
5283 //
5284 // Each thread copies __shared_reduction_var[i] back to RHSExprs[i].
5285 //
5286 // Final __kmpc_barrier to synchronize after broadcasting
5287 QualType PrivateType = Privates->getType();
5288 llvm::Type *LLVMType = CGF.ConvertTypeForMem(T: PrivateType);
5289
5290 const OMPDeclareReductionDecl *UDR = getReductionInit(ReductionOp: ReductionOps);
5291 std::string ReductionVarNameStr;
5292 if (const auto *DRE = dyn_cast<DeclRefExpr>(Val: Privates->IgnoreParenCasts()))
5293 ReductionVarNameStr =
5294 generateUniqueName(CGM, Prefix: DRE->getDecl()->getNameAsString(), Ref: Privates);
5295 else
5296 ReductionVarNameStr = "unnamed_priv_var";
5297
5298 // Create an internal shared variable
5299 std::string SharedName =
5300 CGM.getOpenMPRuntime().getName(Parts: {"internal_pivate_", ReductionVarNameStr});
5301 llvm::GlobalVariable *SharedVar = OMPBuilder.getOrCreateInternalVariable(
5302 Ty: LLVMType, Name: ".omp.reduction." + SharedName);
5303
5304 SharedVar->setAlignment(
5305 llvm::MaybeAlign(CGF.getContext().getTypeAlign(T: PrivateType) / 8));
5306
5307 Address SharedResult =
5308 CGF.MakeNaturalAlignRawAddrLValue(V: SharedVar, T: PrivateType).getAddress();
5309
5310 llvm::Value *ThreadId = getThreadID(CGF, Loc);
5311 llvm::Value *BarrierLoc = emitUpdateLocation(CGF, Loc, Flags: OMP_ATOMIC_REDUCE);
5312 llvm::Value *BarrierArgs[] = {BarrierLoc, ThreadId};
5313
5314 llvm::BasicBlock *InitBB = CGF.createBasicBlock(name: "init");
5315 llvm::BasicBlock *InitEndBB = CGF.createBasicBlock(name: "init.end");
5316
5317 llvm::Value *IsWorker = CGF.Builder.CreateICmpEQ(
5318 LHS: ThreadId, RHS: llvm::ConstantInt::get(Ty: ThreadId->getType(), V: 0));
5319 CGF.Builder.CreateCondBr(Cond: IsWorker, True: InitBB, False: InitEndBB);
5320
5321 CGF.EmitBlock(BB: InitBB);
5322
5323 auto EmitSharedInit = [&]() {
5324 if (UDR) { // Check if it's a User-Defined Reduction
5325 if (const Expr *UDRInitExpr = UDR->getInitializer()) {
5326 std::pair<llvm::Function *, llvm::Function *> FnPair =
5327 getUserDefinedReduction(D: UDR);
5328 llvm::Function *InitializerFn = FnPair.second;
5329 if (InitializerFn) {
5330 if (const auto *CE =
5331 dyn_cast<CallExpr>(Val: UDRInitExpr->IgnoreParenImpCasts())) {
5332 const auto *OutDRE = cast<DeclRefExpr>(
5333 Val: cast<UnaryOperator>(Val: CE->getArg(Arg: 0)->IgnoreParenImpCasts())
5334 ->getSubExpr());
5335 const VarDecl *OutVD = cast<VarDecl>(Val: OutDRE->getDecl());
5336
5337 CodeGenFunction::OMPPrivateScope LocalScope(CGF);
5338 LocalScope.addPrivate(LocalVD: OutVD, Addr: SharedResult);
5339
5340 (void)LocalScope.Privatize();
5341 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(
5342 Val: CE->getCallee()->IgnoreParenImpCasts())) {
5343 CodeGenFunction::OpaqueValueMapping OpaqueMap(
5344 CGF, OVE, RValue::get(V: InitializerFn));
5345 CGF.EmitIgnoredExpr(E: CE);
5346 } else {
5347 CGF.EmitAnyExprToMem(E: UDRInitExpr, Location: SharedResult,
5348 Quals: PrivateType.getQualifiers(),
5349 /*IsInitializer=*/true);
5350 }
5351 } else {
5352 CGF.EmitAnyExprToMem(E: UDRInitExpr, Location: SharedResult,
5353 Quals: PrivateType.getQualifiers(),
5354 /*IsInitializer=*/true);
5355 }
5356 } else {
5357 CGF.EmitAnyExprToMem(E: UDRInitExpr, Location: SharedResult,
5358 Quals: PrivateType.getQualifiers(),
5359 /*IsInitializer=*/true);
5360 }
5361 } else {
5362 // EmitNullInitialization handles default construction for C++ classes
5363 // and zeroing for scalars, which is a reasonable default.
5364 CGF.EmitNullInitialization(DestPtr: SharedResult, Ty: PrivateType);
5365 }
5366 return; // UDR initialization handled
5367 }
5368 if (const auto *DRE = dyn_cast<DeclRefExpr>(Val: Privates)) {
5369 if (const auto *VD = dyn_cast<VarDecl>(Val: DRE->getDecl())) {
5370 if (const Expr *InitExpr = VD->getInit()) {
5371 CGF.EmitAnyExprToMem(E: InitExpr, Location: SharedResult,
5372 Quals: PrivateType.getQualifiers(), IsInitializer: true);
5373 return;
5374 }
5375 }
5376 }
5377 CGF.EmitNullInitialization(DestPtr: SharedResult, Ty: PrivateType);
5378 };
5379 EmitSharedInit();
5380 CGF.Builder.CreateBr(Dest: InitEndBB);
5381 CGF.EmitBlock(BB: InitEndBB);
5382
5383 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
5384 M&: CGM.getModule(), FnID: OMPRTL___kmpc_barrier),
5385 args: BarrierArgs);
5386
5387 const Expr *ReductionOp = ReductionOps;
5388 const OMPDeclareReductionDecl *CurrentUDR = getReductionInit(ReductionOp);
5389 LValue SharedLV = CGF.MakeAddrLValue(Addr: SharedResult, T: PrivateType);
5390 LValue LHSLV = CGF.EmitLValue(E: Privates);
5391
5392 auto EmitCriticalReduction = [&](auto ReductionGen) {
5393 std::string CriticalName = getName(Parts: {"reduction_critical"});
5394 emitCriticalRegion(CGF, CriticalName, CriticalOpGen: ReductionGen, Loc);
5395 };
5396
5397 if (CurrentUDR) {
5398 // Handle user-defined reduction.
5399 auto ReductionGen = [&](CodeGenFunction &CGF, PrePostActionTy &Action) {
5400 Action.Enter(CGF);
5401 std::pair<llvm::Function *, llvm::Function *> FnPair =
5402 getUserDefinedReduction(D: CurrentUDR);
5403 if (FnPair.first) {
5404 if (const auto *CE = dyn_cast<CallExpr>(Val: ReductionOp)) {
5405 const auto *OutDRE = cast<DeclRefExpr>(
5406 Val: cast<UnaryOperator>(Val: CE->getArg(Arg: 0)->IgnoreParenImpCasts())
5407 ->getSubExpr());
5408 const auto *InDRE = cast<DeclRefExpr>(
5409 Val: cast<UnaryOperator>(Val: CE->getArg(Arg: 1)->IgnoreParenImpCasts())
5410 ->getSubExpr());
5411 CodeGenFunction::OMPPrivateScope LocalScope(CGF);
5412 LocalScope.addPrivate(LocalVD: cast<VarDecl>(Val: OutDRE->getDecl()),
5413 Addr: SharedLV.getAddress());
5414 LocalScope.addPrivate(LocalVD: cast<VarDecl>(Val: InDRE->getDecl()),
5415 Addr: LHSLV.getAddress());
5416 (void)LocalScope.Privatize();
5417 emitReductionCombiner(CGF, ReductionOp);
5418 }
5419 }
5420 };
5421 EmitCriticalReduction(ReductionGen);
5422 } else {
5423 // Handle built-in reduction operations.
5424#ifndef NDEBUG
5425 const Expr *ReductionClauseExpr = ReductionOp->IgnoreParenCasts();
5426 if (const auto *Cleanup = dyn_cast<ExprWithCleanups>(ReductionClauseExpr))
5427 ReductionClauseExpr = Cleanup->getSubExpr()->IgnoreParenCasts();
5428
5429 const Expr *AssignRHS = nullptr;
5430 if (const auto *BinOp = dyn_cast<BinaryOperator>(ReductionClauseExpr)) {
5431 if (BinOp->getOpcode() == BO_Assign)
5432 AssignRHS = BinOp->getRHS();
5433 } else if (const auto *OpCall =
5434 dyn_cast<CXXOperatorCallExpr>(ReductionClauseExpr)) {
5435 if (OpCall->getOperator() == OO_Equal)
5436 AssignRHS = OpCall->getArg(1);
5437 }
5438
5439 assert(AssignRHS &&
5440 "Private Variable Reduction : Invalid ReductionOp expression");
5441#endif
5442
5443 auto ReductionGen = [&](CodeGenFunction &CGF, PrePostActionTy &Action) {
5444 Action.Enter(CGF);
5445 const auto *OmpOutDRE =
5446 dyn_cast<DeclRefExpr>(Val: LHSExprs->IgnoreParenImpCasts());
5447 const auto *OmpInDRE =
5448 dyn_cast<DeclRefExpr>(Val: RHSExprs->IgnoreParenImpCasts());
5449 assert(
5450 OmpOutDRE && OmpInDRE &&
5451 "Private Variable Reduction : LHSExpr/RHSExpr must be DeclRefExprs");
5452 const VarDecl *OmpOutVD = cast<VarDecl>(Val: OmpOutDRE->getDecl());
5453 const VarDecl *OmpInVD = cast<VarDecl>(Val: OmpInDRE->getDecl());
5454 CodeGenFunction::OMPPrivateScope LocalScope(CGF);
5455 LocalScope.addPrivate(LocalVD: OmpOutVD, Addr: SharedLV.getAddress());
5456 LocalScope.addPrivate(LocalVD: OmpInVD, Addr: LHSLV.getAddress());
5457 (void)LocalScope.Privatize();
5458 // Emit the actual reduction operation
5459 CGF.EmitIgnoredExpr(E: ReductionOp);
5460 };
5461 EmitCriticalReduction(ReductionGen);
5462 }
5463
5464 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
5465 M&: CGM.getModule(), FnID: OMPRTL___kmpc_barrier),
5466 args: BarrierArgs);
5467
5468 // Broadcast final result
5469 bool IsAggregate = PrivateType->isAggregateType();
5470 LValue SharedLV1 = CGF.MakeAddrLValue(Addr: SharedResult, T: PrivateType);
5471 llvm::Value *FinalResultVal = nullptr;
5472 Address FinalResultAddr = Address::invalid();
5473
5474 if (IsAggregate)
5475 FinalResultAddr = SharedResult;
5476 else
5477 FinalResultVal = CGF.EmitLoadOfScalar(lvalue: SharedLV1, Loc);
5478
5479 LValue TargetLHSLV = CGF.EmitLValue(E: RHSExprs);
5480 if (IsAggregate) {
5481 CGF.EmitAggregateCopy(Dest: TargetLHSLV,
5482 Src: CGF.MakeAddrLValue(Addr: FinalResultAddr, T: PrivateType),
5483 EltTy: PrivateType, MayOverlap: AggValueSlot::DoesNotOverlap, isVolatile: false);
5484 } else {
5485 CGF.EmitStoreOfScalar(value: FinalResultVal, lvalue: TargetLHSLV);
5486 }
5487 // Final synchronization barrier
5488 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
5489 M&: CGM.getModule(), FnID: OMPRTL___kmpc_barrier),
5490 args: BarrierArgs);
5491
5492 // Combiner with original list item
5493 auto OriginalListCombiner = [&](CodeGenFunction &CGF,
5494 PrePostActionTy &Action) {
5495 Action.Enter(CGF);
5496 emitSingleReductionCombiner(CGF, ReductionOp: ReductionOps, PrivateRef: Privates,
5497 LHS: cast<DeclRefExpr>(Val: LHSExprs),
5498 RHS: cast<DeclRefExpr>(Val: RHSExprs));
5499 };
5500 EmitCriticalReduction(OriginalListCombiner);
5501}
5502
5503void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc,
5504 ArrayRef<const Expr *> OrgPrivates,
5505 ArrayRef<const Expr *> OrgLHSExprs,
5506 ArrayRef<const Expr *> OrgRHSExprs,
5507 ArrayRef<const Expr *> OrgReductionOps,
5508 ReductionOptionsTy Options) {
5509 if (!CGF.HaveInsertPoint())
5510 return;
5511
5512 bool WithNowait = Options.WithNowait;
5513 bool SimpleReduction = Options.SimpleReduction;
5514
5515 // Next code should be emitted for reduction:
5516 //
5517 // static kmp_critical_name lock = { 0 };
5518 //
5519 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
5520 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
5521 // ...
5522 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
5523 // *(Type<n>-1*)rhs[<n>-1]);
5524 // }
5525 //
5526 // ...
5527 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
5528 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5529 // RedList, reduce_func, &<lock>)) {
5530 // case 1:
5531 // ...
5532 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5533 // ...
5534 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5535 // break;
5536 // case 2:
5537 // ...
5538 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5539 // ...
5540 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
5541 // break;
5542 // default:;
5543 // }
5544 //
5545 // if SimpleReduction is true, only the next code is generated:
5546 // ...
5547 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5548 // ...
5549
5550 ASTContext &C = CGM.getContext();
5551
5552 if (SimpleReduction) {
5553 CodeGenFunction::RunCleanupsScope Scope(CGF);
5554 const auto *IPriv = OrgPrivates.begin();
5555 const auto *ILHS = OrgLHSExprs.begin();
5556 const auto *IRHS = OrgRHSExprs.begin();
5557 for (const Expr *E : OrgReductionOps) {
5558 emitSingleReductionCombiner(CGF, ReductionOp: E, PrivateRef: *IPriv, LHS: cast<DeclRefExpr>(Val: *ILHS),
5559 RHS: cast<DeclRefExpr>(Val: *IRHS));
5560 ++IPriv;
5561 ++ILHS;
5562 ++IRHS;
5563 }
5564 return;
5565 }
5566
5567 // Filter out shared reduction variables based on IsPrivateVarReduction flag.
5568 // Only keep entries where the corresponding variable is not private.
5569 SmallVector<const Expr *> FilteredPrivates, FilteredLHSExprs,
5570 FilteredRHSExprs, FilteredReductionOps;
5571 for (unsigned I : llvm::seq<unsigned>(
5572 Size: std::min(a: OrgReductionOps.size(), b: OrgLHSExprs.size()))) {
5573 if (!Options.IsPrivateVarReduction[I]) {
5574 FilteredPrivates.emplace_back(Args: OrgPrivates[I]);
5575 FilteredLHSExprs.emplace_back(Args: OrgLHSExprs[I]);
5576 FilteredRHSExprs.emplace_back(Args: OrgRHSExprs[I]);
5577 FilteredReductionOps.emplace_back(Args: OrgReductionOps[I]);
5578 }
5579 }
5580 // Wrap filtered vectors in ArrayRef for downstream shared reduction
5581 // processing.
5582 ArrayRef<const Expr *> Privates = FilteredPrivates;
5583 ArrayRef<const Expr *> LHSExprs = FilteredLHSExprs;
5584 ArrayRef<const Expr *> RHSExprs = FilteredRHSExprs;
5585 ArrayRef<const Expr *> ReductionOps = FilteredReductionOps;
5586
5587 // 1. Build a list of reduction variables.
5588 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
5589 auto Size = RHSExprs.size();
5590 for (const Expr *E : Privates) {
5591 if (E->getType()->isVariablyModifiedType())
5592 // Reserve place for array size.
5593 ++Size;
5594 }
5595 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size);
5596 QualType ReductionArrayTy = C.getConstantArrayType(
5597 EltTy: C.VoidPtrTy, ArySize: ArraySize, SizeExpr: nullptr, ASM: ArraySizeModifier::Normal,
5598 /*IndexTypeQuals=*/0);
5599 RawAddress ReductionList =
5600 CGF.CreateMemTemp(T: ReductionArrayTy, Name: ".omp.reduction.red_list");
5601 const auto *IPriv = Privates.begin();
5602 unsigned Idx = 0;
5603 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) {
5604 Address Elem = CGF.Builder.CreateConstArrayGEP(Addr: ReductionList, Index: Idx);
5605 CGF.Builder.CreateStore(
5606 Val: CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5607 V: CGF.EmitLValue(E: RHSExprs[I]).getPointer(CGF), DestTy: CGF.VoidPtrTy),
5608 Addr: Elem);
5609 if ((*IPriv)->getType()->isVariablyModifiedType()) {
5610 // Store array size.
5611 ++Idx;
5612 Elem = CGF.Builder.CreateConstArrayGEP(Addr: ReductionList, Index: Idx);
5613 llvm::Value *Size = CGF.Builder.CreateIntCast(
5614 V: CGF.getVLASize(
5615 vla: CGF.getContext().getAsVariableArrayType(T: (*IPriv)->getType()))
5616 .NumElts,
5617 DestTy: CGF.SizeTy, /*isSigned=*/false);
5618 CGF.Builder.CreateStore(Val: CGF.Builder.CreateIntToPtr(V: Size, DestTy: CGF.VoidPtrTy),
5619 Addr: Elem);
5620 }
5621 }
5622
5623 // 2. Emit reduce_func().
5624 llvm::Function *ReductionFn = emitReductionFunction(
5625 ReducerName: CGF.CurFn->getName(), Loc, ArgsElemType: CGF.ConvertTypeForMem(T: ReductionArrayTy),
5626 Privates, LHSExprs, RHSExprs, ReductionOps);
5627
5628 // 3. Create static kmp_critical_name lock = { 0 };
5629 std::string Name = getName(Parts: {"reduction"});
5630 llvm::Value *Lock = getCriticalRegionLock(CriticalName: Name);
5631
5632 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5633 // RedList, reduce_func, &<lock>);
5634 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, Flags: OMP_ATOMIC_REDUCE);
5635 llvm::Value *ThreadId = getThreadID(CGF, Loc);
5636 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(Ty: ReductionArrayTy);
5637 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5638 V: ReductionList.getPointer(), DestTy: CGF.VoidPtrTy);
5639 llvm::Value *Args[] = {
5640 IdentTLoc, // ident_t *<loc>
5641 ThreadId, // i32 <gtid>
5642 CGF.Builder.getInt32(C: RHSExprs.size()), // i32 <n>
5643 ReductionArrayTySize, // size_type sizeof(RedList)
5644 RL, // void *RedList
5645 ReductionFn, // void (*) (void *, void *) <reduce_func>
5646 Lock // kmp_critical_name *&<lock>
5647 };
5648 llvm::Value *Res = CGF.EmitRuntimeCall(
5649 callee: OMPBuilder.getOrCreateRuntimeFunction(
5650 M&: CGM.getModule(),
5651 FnID: WithNowait ? OMPRTL___kmpc_reduce_nowait : OMPRTL___kmpc_reduce),
5652 args: Args);
5653
5654 // 5. Build switch(res)
5655 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(name: ".omp.reduction.default");
5656 llvm::SwitchInst *SwInst =
5657 CGF.Builder.CreateSwitch(V: Res, Dest: DefaultBB, /*NumCases=*/2);
5658
5659 // 6. Build case 1:
5660 // ...
5661 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5662 // ...
5663 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5664 // break;
5665 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(name: ".omp.reduction.case1");
5666 SwInst->addCase(OnVal: CGF.Builder.getInt32(C: 1), Dest: Case1BB);
5667 CGF.EmitBlock(BB: Case1BB);
5668
5669 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5670 llvm::Value *EndArgs[] = {
5671 IdentTLoc, // ident_t *<loc>
5672 ThreadId, // i32 <gtid>
5673 Lock // kmp_critical_name *&<lock>
5674 };
5675 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps](
5676 CodeGenFunction &CGF, PrePostActionTy &Action) {
5677 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5678 const auto *IPriv = Privates.begin();
5679 const auto *ILHS = LHSExprs.begin();
5680 const auto *IRHS = RHSExprs.begin();
5681 for (const Expr *E : ReductionOps) {
5682 RT.emitSingleReductionCombiner(CGF, ReductionOp: E, PrivateRef: *IPriv, LHS: cast<DeclRefExpr>(Val: *ILHS),
5683 RHS: cast<DeclRefExpr>(Val: *IRHS));
5684 ++IPriv;
5685 ++ILHS;
5686 ++IRHS;
5687 }
5688 };
5689 RegionCodeGenTy RCG(CodeGen);
5690 CommonActionTy Action(
5691 nullptr, {},
5692 OMPBuilder.getOrCreateRuntimeFunction(
5693 M&: CGM.getModule(), FnID: WithNowait ? OMPRTL___kmpc_end_reduce_nowait
5694 : OMPRTL___kmpc_end_reduce),
5695 EndArgs);
5696 RCG.setAction(Action);
5697 RCG(CGF);
5698
5699 CGF.EmitBranch(Block: DefaultBB);
5700
5701 // 7. Build case 2:
5702 // ...
5703 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5704 // ...
5705 // break;
5706 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(name: ".omp.reduction.case2");
5707 SwInst->addCase(OnVal: CGF.Builder.getInt32(C: 2), Dest: Case2BB);
5708 CGF.EmitBlock(BB: Case2BB);
5709
5710 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps](
5711 CodeGenFunction &CGF, PrePostActionTy &Action) {
5712 const auto *ILHS = LHSExprs.begin();
5713 const auto *IRHS = RHSExprs.begin();
5714 const auto *IPriv = Privates.begin();
5715 for (const Expr *E : ReductionOps) {
5716 const Expr *XExpr = nullptr;
5717 const Expr *EExpr = nullptr;
5718 const Expr *UpExpr = nullptr;
5719 BinaryOperatorKind BO = BO_Comma;
5720 if (const auto *BO = dyn_cast<BinaryOperator>(Val: E)) {
5721 if (BO->getOpcode() == BO_Assign) {
5722 XExpr = BO->getLHS();
5723 UpExpr = BO->getRHS();
5724 }
5725 }
5726 // Try to emit update expression as a simple atomic.
5727 const Expr *RHSExpr = UpExpr;
5728 if (RHSExpr) {
5729 // Analyze RHS part of the whole expression.
5730 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>(
5731 Val: RHSExpr->IgnoreParenImpCasts())) {
5732 // If this is a conditional operator, analyze its condition for
5733 // min/max reduction operator.
5734 RHSExpr = ACO->getCond();
5735 }
5736 if (const auto *BORHS =
5737 dyn_cast<BinaryOperator>(Val: RHSExpr->IgnoreParenImpCasts())) {
5738 EExpr = BORHS->getRHS();
5739 BO = BORHS->getOpcode();
5740 }
5741 }
5742 if (XExpr) {
5743 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ILHS)->getDecl());
5744 auto &&AtomicRedGen = [BO, VD,
5745 Loc](CodeGenFunction &CGF, const Expr *XExpr,
5746 const Expr *EExpr, const Expr *UpExpr) {
5747 LValue X = CGF.EmitLValue(E: XExpr);
5748 RValue E;
5749 if (EExpr)
5750 E = CGF.EmitAnyExpr(E: EExpr);
5751 CGF.EmitOMPAtomicSimpleUpdateExpr(
5752 X, E, BO, /*IsXLHSInRHSPart=*/true,
5753 AO: llvm::AtomicOrdering::Monotonic, Loc,
5754 CommonGen: [&CGF, UpExpr, VD, Loc](RValue XRValue) {
5755 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
5756 Address LHSTemp = CGF.CreateMemTemp(T: VD->getType());
5757 CGF.emitOMPSimpleStore(
5758 LVal: CGF.MakeAddrLValue(Addr: LHSTemp, T: VD->getType()), RVal: XRValue,
5759 RValTy: VD->getType().getNonReferenceType(), Loc);
5760 PrivateScope.addPrivate(LocalVD: VD, Addr: LHSTemp);
5761 (void)PrivateScope.Privatize();
5762 return CGF.EmitAnyExpr(E: UpExpr);
5763 });
5764 };
5765 if ((*IPriv)->getType()->isArrayType()) {
5766 // Emit atomic reduction for array section.
5767 const auto *RHSVar =
5768 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IRHS)->getDecl());
5769 EmitOMPAggregateReduction(CGF, Type: (*IPriv)->getType(), LHSVar: VD, RHSVar,
5770 RedOpGen: AtomicRedGen, XExpr, EExpr, UpExpr);
5771 } else {
5772 // Emit atomic reduction for array subscript or single variable.
5773 AtomicRedGen(CGF, XExpr, EExpr, UpExpr);
5774 }
5775 } else {
5776 // Emit as a critical region.
5777 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *,
5778 const Expr *, const Expr *) {
5779 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5780 std::string Name = RT.getName(Parts: {"atomic_reduction"});
5781 RT.emitCriticalRegion(
5782 CGF, CriticalName: Name,
5783 CriticalOpGen: [=](CodeGenFunction &CGF, PrePostActionTy &Action) {
5784 Action.Enter(CGF);
5785 emitReductionCombiner(CGF, ReductionOp: E);
5786 },
5787 Loc);
5788 };
5789 if ((*IPriv)->getType()->isArrayType()) {
5790 const auto *LHSVar =
5791 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ILHS)->getDecl());
5792 const auto *RHSVar =
5793 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IRHS)->getDecl());
5794 EmitOMPAggregateReduction(CGF, Type: (*IPriv)->getType(), LHSVar, RHSVar,
5795 RedOpGen: CritRedGen);
5796 } else {
5797 CritRedGen(CGF, nullptr, nullptr, nullptr);
5798 }
5799 }
5800 ++ILHS;
5801 ++IRHS;
5802 ++IPriv;
5803 }
5804 };
5805 RegionCodeGenTy AtomicRCG(AtomicCodeGen);
5806 if (!WithNowait) {
5807 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
5808 llvm::Value *EndArgs[] = {
5809 IdentTLoc, // ident_t *<loc>
5810 ThreadId, // i32 <gtid>
5811 Lock // kmp_critical_name *&<lock>
5812 };
5813 CommonActionTy Action(nullptr, {},
5814 OMPBuilder.getOrCreateRuntimeFunction(
5815 M&: CGM.getModule(), FnID: OMPRTL___kmpc_end_reduce),
5816 EndArgs);
5817 AtomicRCG.setAction(Action);
5818 AtomicRCG(CGF);
5819 } else {
5820 AtomicRCG(CGF);
5821 }
5822
5823 CGF.EmitBranch(Block: DefaultBB);
5824 CGF.EmitBlock(BB: DefaultBB, /*IsFinished=*/true);
5825 assert(OrgLHSExprs.size() == OrgPrivates.size() &&
5826 "PrivateVarReduction: Privates size mismatch");
5827 assert(OrgLHSExprs.size() == OrgReductionOps.size() &&
5828 "PrivateVarReduction: ReductionOps size mismatch");
5829 for (unsigned I : llvm::seq<unsigned>(
5830 Size: std::min(a: OrgReductionOps.size(), b: OrgLHSExprs.size()))) {
5831 if (Options.IsPrivateVarReduction[I])
5832 emitPrivateReduction(CGF, Loc, Privates: OrgPrivates[I], LHSExprs: OrgLHSExprs[I],
5833 RHSExprs: OrgRHSExprs[I], ReductionOps: OrgReductionOps[I]);
5834 }
5835}
5836
5837/// Generates unique name for artificial threadprivate variables.
5838/// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>"
5839static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix,
5840 const Expr *Ref) {
5841 SmallString<256> Buffer;
5842 llvm::raw_svector_ostream Out(Buffer);
5843 const clang::DeclRefExpr *DE;
5844 const VarDecl *D = ::getBaseDecl(Ref, DE);
5845 if (!D) {
5846 auto *DRE = cast<DeclRefExpr>(Val: Ref);
5847 if (const auto *BD = dyn_cast<BindingDecl>(Val: DRE->getDecl())) {
5848 // For BindingDecls, use the decomposed declaration as the base.
5849 D = cast<VarDecl>(Val: BD->getDecomposedDecl());
5850 } else {
5851 D = cast<VarDecl>(Val: DRE->getDecl());
5852 }
5853 }
5854 D = D->getCanonicalDecl();
5855 std::string Name = CGM.getOpenMPRuntime().getName(
5856 Parts: {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(GD: D)});
5857 Out << Prefix << Name << "_"
5858 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding();
5859 return std::string(Out.str());
5860}
5861
5862/// Emits reduction initializer function:
5863/// \code
5864/// void @.red_init(void* %arg, void* %orig) {
5865/// %0 = bitcast void* %arg to <type>*
5866/// store <type> <init>, <type>* %0
5867/// ret void
5868/// }
5869/// \endcode
5870static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM,
5871 SourceLocation Loc,
5872 ReductionCodeGen &RCG, unsigned N) {
5873 ASTContext &C = CGM.getContext();
5874 QualType VoidPtrTy = C.VoidPtrTy;
5875 VoidPtrTy.addRestrict();
5876 FunctionArgList Args;
5877 auto *Param =
5878 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
5879 T: VoidPtrTy, ParamKind: ImplicitParamKind::Other);
5880 auto *ParamOrig =
5881 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
5882 T: VoidPtrTy, ParamKind: ImplicitParamKind::Other);
5883 Args.emplace_back(Args&: Param);
5884 Args.emplace_back(Args&: ParamOrig);
5885 const auto &FnInfo =
5886 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
5887 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(Info: FnInfo);
5888 std::string Name = CGM.getOpenMPRuntime().getName(Parts: {"red_init", ""});
5889 auto *Fn = llvm::Function::Create(Ty: FnTy, Linkage: llvm::GlobalValue::InternalLinkage,
5890 N: Name, M: &CGM.getModule());
5891 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: Fn, FI: FnInfo);
5892 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
5893 Fn->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
5894 Fn->setDoesNotRecurse();
5895 CodeGenFunction CGF(CGM);
5896 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn, FnInfo, Args, Loc, StartLoc: Loc);
5897 QualType PrivateType = RCG.getPrivateType(N);
5898 Address PrivateAddr = CGF.EmitLoadOfPointer(
5899 Ptr: CGF.GetAddrOfLocalVar(VD: Param).withElementType(ElemTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
5900 PtrTy: C.getPointerType(T: PrivateType)->castAs<PointerType>());
5901 llvm::Value *Size = nullptr;
5902 // If the size of the reduction item is non-constant, load it from global
5903 // threadprivate variable.
5904 if (RCG.getSizes(N).second) {
5905 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
5906 CGF, VarType: CGM.getContext().getSizeType(),
5907 Name: generateUniqueName(CGM, Prefix: "reduction_size", Ref: RCG.getRefExpr(N)));
5908 Size = CGF.EmitLoadOfScalar(Addr: SizeAddr, /*Volatile=*/false,
5909 Ty: CGM.getContext().getSizeType(), Loc);
5910 }
5911 RCG.emitAggregateType(CGF, N, Size);
5912 Address OrigAddr = Address::invalid();
5913 // If initializer uses initializer from declare reduction construct, emit a
5914 // pointer to the address of the original reduction item (reuired by reduction
5915 // initializer)
5916 if (RCG.usesReductionInitializer(N)) {
5917 Address SharedAddr = CGF.GetAddrOfLocalVar(VD: ParamOrig);
5918 OrigAddr = CGF.EmitLoadOfPointer(
5919 Ptr: SharedAddr,
5920 PtrTy: CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr());
5921 }
5922 // Emit the initializer:
5923 // %0 = bitcast void* %arg to <type>*
5924 // store <type> <init>, <type>* %0
5925 RCG.emitInitialization(CGF, N, PrivateAddr, SharedAddr: OrigAddr,
5926 DefaultInit: [](CodeGenFunction &) { return false; });
5927 CGF.FinishFunction();
5928 return Fn;
5929}
5930
5931/// Emits reduction combiner function:
5932/// \code
5933/// void @.red_comb(void* %arg0, void* %arg1) {
5934/// %lhs = bitcast void* %arg0 to <type>*
5935/// %rhs = bitcast void* %arg1 to <type>*
5936/// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs)
5937/// store <type> %2, <type>* %lhs
5938/// ret void
5939/// }
5940/// \endcode
5941static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM,
5942 SourceLocation Loc,
5943 ReductionCodeGen &RCG, unsigned N,
5944 const Expr *ReductionOp,
5945 const Expr *LHS, const Expr *RHS,
5946 const Expr *PrivateRef) {
5947 ASTContext &C = CGM.getContext();
5948 const auto *LHSVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: LHS)->getDecl());
5949 const auto *RHSVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: RHS)->getDecl());
5950 FunctionArgList Args;
5951 auto *ParamInOut =
5952 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
5953 T: C.VoidPtrTy, ParamKind: ImplicitParamKind::Other);
5954 auto *ParamIn =
5955 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
5956 T: C.VoidPtrTy, ParamKind: ImplicitParamKind::Other);
5957 Args.emplace_back(Args&: ParamInOut);
5958 Args.emplace_back(Args&: ParamIn);
5959 const auto &FnInfo =
5960 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
5961 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(Info: FnInfo);
5962 std::string Name = CGM.getOpenMPRuntime().getName(Parts: {"red_comb", ""});
5963 auto *Fn = llvm::Function::Create(Ty: FnTy, Linkage: llvm::GlobalValue::InternalLinkage,
5964 N: Name, M: &CGM.getModule());
5965 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: Fn, FI: FnInfo);
5966 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
5967 Fn->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
5968 Fn->setDoesNotRecurse();
5969 CodeGenFunction CGF(CGM);
5970 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn, FnInfo, Args, Loc, StartLoc: Loc);
5971 llvm::Value *Size = nullptr;
5972 // If the size of the reduction item is non-constant, load it from global
5973 // threadprivate variable.
5974 if (RCG.getSizes(N).second) {
5975 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
5976 CGF, VarType: CGM.getContext().getSizeType(),
5977 Name: generateUniqueName(CGM, Prefix: "reduction_size", Ref: RCG.getRefExpr(N)));
5978 Size = CGF.EmitLoadOfScalar(Addr: SizeAddr, /*Volatile=*/false,
5979 Ty: CGM.getContext().getSizeType(), Loc);
5980 }
5981 RCG.emitAggregateType(CGF, N, Size);
5982 // Remap lhs and rhs variables to the addresses of the function arguments.
5983 // %lhs = bitcast void* %arg0 to <type>*
5984 // %rhs = bitcast void* %arg1 to <type>*
5985 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
5986 PrivateScope.addPrivate(
5987 LocalVD: LHSVD,
5988 // Pull out the pointer to the variable.
5989 Addr: CGF.EmitLoadOfPointer(
5990 Ptr: CGF.GetAddrOfLocalVar(VD: ParamInOut)
5991 .withElementType(ElemTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
5992 PtrTy: C.getPointerType(T: LHSVD->getType())->castAs<PointerType>()));
5993 PrivateScope.addPrivate(
5994 LocalVD: RHSVD,
5995 // Pull out the pointer to the variable.
5996 Addr: CGF.EmitLoadOfPointer(
5997 Ptr: CGF.GetAddrOfLocalVar(VD: ParamIn).withElementType(
5998 ElemTy: CGF.Builder.getPtrTy(AddrSpace: 0)),
5999 PtrTy: C.getPointerType(T: RHSVD->getType())->castAs<PointerType>()));
6000 PrivateScope.Privatize();
6001 // Emit the combiner body:
6002 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs)
6003 // store <type> %2, <type>* %lhs
6004 CGM.getOpenMPRuntime().emitSingleReductionCombiner(
6005 CGF, ReductionOp, PrivateRef, LHS: cast<DeclRefExpr>(Val: LHS),
6006 RHS: cast<DeclRefExpr>(Val: RHS));
6007 CGF.FinishFunction();
6008 return Fn;
6009}
6010
6011/// Emits reduction finalizer function:
6012/// \code
6013/// void @.red_fini(void* %arg) {
6014/// %0 = bitcast void* %arg to <type>*
6015/// <destroy>(<type>* %0)
6016/// ret void
6017/// }
6018/// \endcode
6019static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM,
6020 SourceLocation Loc,
6021 ReductionCodeGen &RCG, unsigned N) {
6022 if (!RCG.needCleanups(N))
6023 return nullptr;
6024 ASTContext &C = CGM.getContext();
6025 FunctionArgList Args;
6026 auto *Param =
6027 ImplicitParamDecl::Create(C, /*DC=*/nullptr, IdLoc: Loc, /*Id=*/nullptr,
6028 T: C.VoidPtrTy, ParamKind: ImplicitParamKind::Other);
6029 Args.emplace_back(Args&: Param);
6030 const auto &FnInfo =
6031 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: C.VoidTy, args: Args);
6032 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(Info: FnInfo);
6033 std::string Name = CGM.getOpenMPRuntime().getName(Parts: {"red_fini", ""});
6034 auto *Fn = llvm::Function::Create(Ty: FnTy, Linkage: llvm::GlobalValue::InternalLinkage,
6035 N: Name, M: &CGM.getModule());
6036 CGM.SetInternalFunctionAttributes(GD: GlobalDecl(), F: Fn, FI: FnInfo);
6037 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
6038 Fn->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
6039 Fn->setDoesNotRecurse();
6040 CodeGenFunction CGF(CGM);
6041 CGF.StartFunction(GD: GlobalDecl(), RetTy: C.VoidTy, Fn, FnInfo, Args, Loc, StartLoc: Loc);
6042 Address PrivateAddr = CGF.EmitLoadOfPointer(
6043 Ptr: CGF.GetAddrOfLocalVar(VD: Param), PtrTy: C.VoidPtrTy.castAs<PointerType>());
6044 llvm::Value *Size = nullptr;
6045 // If the size of the reduction item is non-constant, load it from global
6046 // threadprivate variable.
6047 if (RCG.getSizes(N).second) {
6048 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6049 CGF, VarType: CGM.getContext().getSizeType(),
6050 Name: generateUniqueName(CGM, Prefix: "reduction_size", Ref: RCG.getRefExpr(N)));
6051 Size = CGF.EmitLoadOfScalar(Addr: SizeAddr, /*Volatile=*/false,
6052 Ty: CGM.getContext().getSizeType(), Loc);
6053 }
6054 RCG.emitAggregateType(CGF, N, Size);
6055 // Emit the finalizer body:
6056 // <destroy>(<type>* %0)
6057 RCG.emitCleanups(CGF, N, PrivateAddr);
6058 CGF.FinishFunction(EndLoc: Loc);
6059 return Fn;
6060}
6061
6062llvm::Value *CGOpenMPRuntime::emitTaskReductionInit(
6063 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
6064 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
6065 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty())
6066 return nullptr;
6067
6068 // Build typedef struct:
6069 // kmp_taskred_input {
6070 // void *reduce_shar; // shared reduction item
6071 // void *reduce_orig; // original reduction item used for initialization
6072 // size_t reduce_size; // size of data item
6073 // void *reduce_init; // data initialization routine
6074 // void *reduce_fini; // data finalization routine
6075 // void *reduce_comb; // data combiner routine
6076 // kmp_task_red_flags_t flags; // flags for additional info from compiler
6077 // } kmp_taskred_input_t;
6078 ASTContext &C = CGM.getContext();
6079 RecordDecl *RD = C.buildImplicitRecord(Name: "kmp_taskred_input_t");
6080 RD->startDefinition();
6081 const FieldDecl *SharedFD = addFieldToRecordDecl(C, DC: RD, FieldTy: C.VoidPtrTy);
6082 const FieldDecl *OrigFD = addFieldToRecordDecl(C, DC: RD, FieldTy: C.VoidPtrTy);
6083 const FieldDecl *SizeFD = addFieldToRecordDecl(C, DC: RD, FieldTy: C.getSizeType());
6084 const FieldDecl *InitFD = addFieldToRecordDecl(C, DC: RD, FieldTy: C.VoidPtrTy);
6085 const FieldDecl *FiniFD = addFieldToRecordDecl(C, DC: RD, FieldTy: C.VoidPtrTy);
6086 const FieldDecl *CombFD = addFieldToRecordDecl(C, DC: RD, FieldTy: C.VoidPtrTy);
6087 const FieldDecl *FlagsFD = addFieldToRecordDecl(
6088 C, DC: RD, FieldTy: C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false));
6089 RD->completeDefinition();
6090 CanQualType RDType = C.getCanonicalTagType(TD: RD);
6091 unsigned Size = Data.ReductionVars.size();
6092 llvm::APInt ArraySize(/*numBits=*/64, Size);
6093 QualType ArrayRDType =
6094 C.getConstantArrayType(EltTy: RDType, ArySize: ArraySize, SizeExpr: nullptr,
6095 ASM: ArraySizeModifier::Normal, /*IndexTypeQuals=*/0);
6096 // kmp_task_red_input_t .rd_input.[Size];
6097 RawAddress TaskRedInput = CGF.CreateMemTemp(T: ArrayRDType, Name: ".rd_input.");
6098 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionOrigs,
6099 Data.ReductionCopies, Data.ReductionOps);
6100 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) {
6101 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt];
6102 llvm::Value *Idxs[] = {llvm::ConstantInt::get(Ty: CGM.SizeTy, /*V=*/0),
6103 llvm::ConstantInt::get(Ty: CGM.SizeTy, V: Cnt)};
6104 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP(
6105 ElemTy: TaskRedInput.getElementType(), Ptr: TaskRedInput.getPointer(), IdxList: Idxs,
6106 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc,
6107 Name: ".rd_input.gep.");
6108 LValue ElemLVal = CGF.MakeNaturalAlignRawAddrLValue(V: GEP, T: RDType);
6109 // ElemLVal.reduce_shar = &Shareds[Cnt];
6110 LValue SharedLVal = CGF.EmitLValueForField(Base: ElemLVal, Field: SharedFD);
6111 RCG.emitSharedOrigLValue(CGF, N: Cnt);
6112 llvm::Value *Shared = RCG.getSharedLValue(N: Cnt).getPointer(CGF);
6113 CGF.EmitStoreOfScalar(value: Shared, lvalue: SharedLVal);
6114 // ElemLVal.reduce_orig = &Origs[Cnt];
6115 LValue OrigLVal = CGF.EmitLValueForField(Base: ElemLVal, Field: OrigFD);
6116 llvm::Value *Orig = RCG.getOrigLValue(N: Cnt).getPointer(CGF);
6117 CGF.EmitStoreOfScalar(value: Orig, lvalue: OrigLVal);
6118 RCG.emitAggregateType(CGF, N: Cnt);
6119 llvm::Value *SizeValInChars;
6120 llvm::Value *SizeVal;
6121 std::tie(args&: SizeValInChars, args&: SizeVal) = RCG.getSizes(N: Cnt);
6122 // We use delayed creation/initialization for VLAs and array sections. It is
6123 // required because runtime does not provide the way to pass the sizes of
6124 // VLAs/array sections to initializer/combiner/finalizer functions. Instead
6125 // threadprivate global variables are used to store these values and use
6126 // them in the functions.
6127 bool DelayedCreation = !!SizeVal;
6128 SizeValInChars = CGF.Builder.CreateIntCast(V: SizeValInChars, DestTy: CGM.SizeTy,
6129 /*isSigned=*/false);
6130 LValue SizeLVal = CGF.EmitLValueForField(Base: ElemLVal, Field: SizeFD);
6131 CGF.EmitStoreOfScalar(value: SizeValInChars, lvalue: SizeLVal);
6132 // ElemLVal.reduce_init = init;
6133 LValue InitLVal = CGF.EmitLValueForField(Base: ElemLVal, Field: InitFD);
6134 llvm::Value *InitAddr = emitReduceInitFunction(CGM, Loc, RCG, N: Cnt);
6135 CGF.EmitStoreOfScalar(value: InitAddr, lvalue: InitLVal);
6136 // ElemLVal.reduce_fini = fini;
6137 LValue FiniLVal = CGF.EmitLValueForField(Base: ElemLVal, Field: FiniFD);
6138 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, N: Cnt);
6139 llvm::Value *FiniAddr =
6140 Fini ? Fini : llvm::ConstantPointerNull::get(T: CGM.VoidPtrTy);
6141 CGF.EmitStoreOfScalar(value: FiniAddr, lvalue: FiniLVal);
6142 // ElemLVal.reduce_comb = comb;
6143 LValue CombLVal = CGF.EmitLValueForField(Base: ElemLVal, Field: CombFD);
6144 llvm::Value *CombAddr = emitReduceCombFunction(
6145 CGM, Loc, RCG, N: Cnt, ReductionOp: Data.ReductionOps[Cnt], LHS: LHSExprs[Cnt],
6146 RHS: RHSExprs[Cnt], PrivateRef: Data.ReductionCopies[Cnt]);
6147 CGF.EmitStoreOfScalar(value: CombAddr, lvalue: CombLVal);
6148 // ElemLVal.flags = 0;
6149 LValue FlagsLVal = CGF.EmitLValueForField(Base: ElemLVal, Field: FlagsFD);
6150 if (DelayedCreation) {
6151 CGF.EmitStoreOfScalar(
6152 value: llvm::ConstantInt::get(Ty: CGM.Int32Ty, /*V=*/1, /*isSigned=*/IsSigned: true),
6153 lvalue: FlagsLVal);
6154 } else
6155 CGF.EmitNullInitialization(DestPtr: FlagsLVal.getAddress(), Ty: FlagsLVal.getType());
6156 }
6157 if (Data.IsReductionWithTaskMod) {
6158 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int
6159 // is_ws, int num, void *data);
6160 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc);
6161 llvm::Value *GTid = CGF.Builder.CreateIntCast(V: getThreadID(CGF, Loc),
6162 DestTy: CGM.IntTy, /*isSigned=*/true);
6163 llvm::Value *Args[] = {
6164 IdentTLoc, GTid,
6165 llvm::ConstantInt::get(Ty: CGM.IntTy, V: Data.IsWorksharingReduction ? 1 : 0,
6166 /*isSigned=*/IsSigned: true),
6167 llvm::ConstantInt::get(Ty: CGM.IntTy, V: Size, /*isSigned=*/IsSigned: true),
6168 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6169 V: TaskRedInput.getPointer(), DestTy: CGM.VoidPtrTy)};
6170 return CGF.EmitRuntimeCall(
6171 callee: OMPBuilder.getOrCreateRuntimeFunction(
6172 M&: CGM.getModule(), FnID: OMPRTL___kmpc_taskred_modifier_init),
6173 args: Args);
6174 }
6175 // Build call void *__kmpc_taskred_init(int gtid, int num_data, void *data);
6176 llvm::Value *Args[] = {
6177 CGF.Builder.CreateIntCast(V: getThreadID(CGF, Loc), DestTy: CGM.IntTy,
6178 /*isSigned=*/true),
6179 llvm::ConstantInt::get(Ty: CGM.IntTy, V: Size, /*isSigned=*/IsSigned: true),
6180 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(V: TaskRedInput.getPointer(),
6181 DestTy: CGM.VoidPtrTy)};
6182 return CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
6183 M&: CGM.getModule(), FnID: OMPRTL___kmpc_taskred_init),
6184 args: Args);
6185}
6186
6187void CGOpenMPRuntime::emitTaskReductionFini(CodeGenFunction &CGF,
6188 SourceLocation Loc,
6189 bool IsWorksharingReduction) {
6190 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int
6191 // is_ws, int num, void *data);
6192 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc);
6193 llvm::Value *GTid = CGF.Builder.CreateIntCast(V: getThreadID(CGF, Loc),
6194 DestTy: CGM.IntTy, /*isSigned=*/true);
6195 llvm::Value *Args[] = {IdentTLoc, GTid,
6196 llvm::ConstantInt::get(Ty: CGM.IntTy,
6197 V: IsWorksharingReduction ? 1 : 0,
6198 /*isSigned=*/IsSigned: true)};
6199 (void)CGF.EmitRuntimeCall(
6200 callee: OMPBuilder.getOrCreateRuntimeFunction(
6201 M&: CGM.getModule(), FnID: OMPRTL___kmpc_task_reduction_modifier_fini),
6202 args: Args);
6203}
6204
6205void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
6206 SourceLocation Loc,
6207 ReductionCodeGen &RCG,
6208 unsigned N) {
6209 auto Sizes = RCG.getSizes(N);
6210 // Emit threadprivate global variable if the type is non-constant
6211 // (Sizes.second = nullptr).
6212 if (Sizes.second) {
6213 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(V: Sizes.second, DestTy: CGM.SizeTy,
6214 /*isSigned=*/false);
6215 Address SizeAddr = getAddrOfArtificialThreadPrivate(
6216 CGF, VarType: CGM.getContext().getSizeType(),
6217 Name: generateUniqueName(CGM, Prefix: "reduction_size", Ref: RCG.getRefExpr(N)));
6218 CGF.Builder.CreateStore(Val: SizeVal, Addr: SizeAddr, /*IsVolatile=*/false);
6219 }
6220}
6221
6222Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF,
6223 SourceLocation Loc,
6224 llvm::Value *ReductionsPtr,
6225 LValue SharedLVal) {
6226 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
6227 // *d);
6228 llvm::Value *Args[] = {CGF.Builder.CreateIntCast(V: getThreadID(CGF, Loc),
6229 DestTy: CGM.IntTy,
6230 /*isSigned=*/true),
6231 ReductionsPtr,
6232 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6233 V: SharedLVal.getPointer(CGF), DestTy: CGM.VoidPtrTy)};
6234 return Address(
6235 CGF.EmitRuntimeCall(
6236 callee: OMPBuilder.getOrCreateRuntimeFunction(
6237 M&: CGM.getModule(), FnID: OMPRTL___kmpc_task_reduction_get_th_data),
6238 args: Args),
6239 CGF.Int8Ty, SharedLVal.getAlignment());
6240}
6241
6242void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, SourceLocation Loc,
6243 const OMPTaskDataTy &Data) {
6244 if (!CGF.HaveInsertPoint())
6245 return;
6246
6247 if (CGF.CGM.getLangOpts().OpenMPIRBuilder && Data.Dependences.empty()) {
6248 // TODO: Need to support taskwait with dependences in the OpenMPIRBuilder.
6249 OMPBuilder.createTaskwait(Loc: CGF.Builder);
6250 } else {
6251 llvm::Value *ThreadID = getThreadID(CGF, Loc);
6252 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
6253 auto &M = CGM.getModule();
6254 Address DependenciesArray = Address::invalid();
6255 llvm::Value *NumOfElements;
6256 std::tie(args&: NumOfElements, args&: DependenciesArray) =
6257 emitDependClause(CGF, Dependencies: Data.Dependences, Loc);
6258 if (!Data.Dependences.empty()) {
6259 llvm::Value *DepWaitTaskArgs[7];
6260 DepWaitTaskArgs[0] = UpLoc;
6261 DepWaitTaskArgs[1] = ThreadID;
6262 DepWaitTaskArgs[2] = NumOfElements;
6263 DepWaitTaskArgs[3] = DependenciesArray.emitRawPointer(CGF);
6264 DepWaitTaskArgs[4] = CGF.Builder.getInt32(C: 0);
6265 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
6266 DepWaitTaskArgs[6] =
6267 llvm::ConstantInt::get(Ty: CGF.Int32Ty, V: Data.HasNowaitClause);
6268
6269 CodeGenFunction::RunCleanupsScope LocalScope(CGF);
6270
6271 // Build void __kmpc_omp_taskwait_deps_51(ident_t *, kmp_int32 gtid,
6272 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
6273 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list,
6274 // kmp_int32 has_no_wait); if dependence info is specified.
6275 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
6276 M, FnID: OMPRTL___kmpc_omp_taskwait_deps_51),
6277 args: DepWaitTaskArgs);
6278
6279 } else {
6280
6281 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
6282 // global_tid);
6283 llvm::Value *Args[] = {UpLoc, ThreadID};
6284 // Ignore return result until untied tasks are supported.
6285 CGF.EmitRuntimeCall(
6286 callee: OMPBuilder.getOrCreateRuntimeFunction(M, FnID: OMPRTL___kmpc_omp_taskwait),
6287 args: Args);
6288 }
6289 }
6290
6291 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo))
6292 Region->emitUntiedSwitch(CGF);
6293}
6294
6295void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF,
6296 OpenMPDirectiveKind InnerKind,
6297 const RegionCodeGenTy &CodeGen,
6298 bool HasCancel) {
6299 if (!CGF.HaveInsertPoint())
6300 return;
6301 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel,
6302 InnerKind != OMPD_critical &&
6303 InnerKind != OMPD_master &&
6304 InnerKind != OMPD_masked);
6305 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
6306}
6307
6308namespace {
6309enum RTCancelKind {
6310 CancelNoreq = 0,
6311 CancelParallel = 1,
6312 CancelLoop = 2,
6313 CancelSections = 3,
6314 CancelTaskgroup = 4
6315};
6316} // anonymous namespace
6317
6318static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) {
6319 RTCancelKind CancelKind = CancelNoreq;
6320 if (CancelRegion == OMPD_parallel)
6321 CancelKind = CancelParallel;
6322 else if (CancelRegion == OMPD_for)
6323 CancelKind = CancelLoop;
6324 else if (CancelRegion == OMPD_sections)
6325 CancelKind = CancelSections;
6326 else {
6327 assert(CancelRegion == OMPD_taskgroup);
6328 CancelKind = CancelTaskgroup;
6329 }
6330 return CancelKind;
6331}
6332
6333void CGOpenMPRuntime::emitCancellationPointCall(
6334 CodeGenFunction &CGF, SourceLocation Loc,
6335 OpenMPDirectiveKind CancelRegion) {
6336 if (!CGF.HaveInsertPoint())
6337 return;
6338 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
6339 // global_tid, kmp_int32 cncl_kind);
6340 if (auto *OMPRegionInfo =
6341 dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo)) {
6342 // For 'cancellation point taskgroup', the task region info may not have a
6343 // cancel. This may instead happen in another adjacent task.
6344 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) {
6345 llvm::Value *Args[] = {
6346 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
6347 CGF.Builder.getInt32(C: getCancellationKind(CancelRegion))};
6348 // Ignore return result until untied tasks are supported.
6349 llvm::Value *Result = CGF.EmitRuntimeCall(
6350 callee: OMPBuilder.getOrCreateRuntimeFunction(
6351 M&: CGM.getModule(), FnID: OMPRTL___kmpc_cancellationpoint),
6352 args: Args);
6353 // if (__kmpc_cancellationpoint()) {
6354 // call i32 @__kmpc_cancel_barrier( // for parallel cancellation only
6355 // exit from construct;
6356 // }
6357 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(name: ".cancel.exit");
6358 llvm::BasicBlock *ContBB = CGF.createBasicBlock(name: ".cancel.continue");
6359 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Arg: Result);
6360 CGF.Builder.CreateCondBr(Cond: Cmp, True: ExitBB, False: ContBB);
6361 CGF.EmitBlock(BB: ExitBB);
6362 if (CancelRegion == OMPD_parallel)
6363 emitBarrierCall(CGF, Loc, Kind: OMPD_unknown, /*EmitChecks=*/false);
6364 // exit from construct;
6365 CodeGenFunction::JumpDest CancelDest =
6366 CGF.getOMPCancelDestination(Kind: OMPRegionInfo->getDirectiveKind());
6367 CGF.EmitBranchThroughCleanup(Dest: CancelDest);
6368 CGF.EmitBlock(BB: ContBB, /*IsFinished=*/true);
6369 }
6370 }
6371}
6372
6373void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc,
6374 const Expr *IfCond,
6375 OpenMPDirectiveKind CancelRegion) {
6376 if (!CGF.HaveInsertPoint())
6377 return;
6378 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
6379 // kmp_int32 cncl_kind);
6380 auto &M = CGM.getModule();
6381 if (auto *OMPRegionInfo =
6382 dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo)) {
6383 auto &&ThenGen = [this, &M, Loc, CancelRegion,
6384 OMPRegionInfo](CodeGenFunction &CGF, PrePostActionTy &) {
6385 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6386 llvm::Value *Args[] = {
6387 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc),
6388 CGF.Builder.getInt32(C: getCancellationKind(CancelRegion))};
6389 // Ignore return result until untied tasks are supported.
6390 llvm::Value *Result = CGF.EmitRuntimeCall(
6391 callee: OMPBuilder.getOrCreateRuntimeFunction(M, FnID: OMPRTL___kmpc_cancel), args: Args);
6392 // if (__kmpc_cancel()) {
6393 // call i32 @__kmpc_cancel_barrier( // for parallel cancellation only
6394 // exit from construct;
6395 // }
6396 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(name: ".cancel.exit");
6397 llvm::BasicBlock *ContBB = CGF.createBasicBlock(name: ".cancel.continue");
6398 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Arg: Result);
6399 CGF.Builder.CreateCondBr(Cond: Cmp, True: ExitBB, False: ContBB);
6400 CGF.EmitBlock(BB: ExitBB);
6401 if (CancelRegion == OMPD_parallel)
6402 RT.emitBarrierCall(CGF, Loc, Kind: OMPD_unknown, /*EmitChecks=*/false);
6403 // exit from construct;
6404 CodeGenFunction::JumpDest CancelDest =
6405 CGF.getOMPCancelDestination(Kind: OMPRegionInfo->getDirectiveKind());
6406 CGF.EmitBranchThroughCleanup(Dest: CancelDest);
6407 CGF.EmitBlock(BB: ContBB, /*IsFinished=*/true);
6408 };
6409 if (IfCond) {
6410 emitIfClause(CGF, Cond: IfCond, ThenGen,
6411 ElseGen: [](CodeGenFunction &, PrePostActionTy &) {});
6412 } else {
6413 RegionCodeGenTy ThenRCG(ThenGen);
6414 ThenRCG(CGF);
6415 }
6416 }
6417}
6418
6419namespace {
6420/// Cleanup action for uses_allocators support.
6421class OMPUsesAllocatorsActionTy final : public PrePostActionTy {
6422 ArrayRef<std::pair<const Expr *, const Expr *>> Allocators;
6423 const OMPExecutableDirective &D;
6424 bool IsOffloadEntry;
6425
6426public:
6427 OMPUsesAllocatorsActionTy(
6428 ArrayRef<std::pair<const Expr *, const Expr *>> Allocators,
6429 const OMPExecutableDirective &D, bool IsOffloadEntry)
6430 : Allocators(Allocators), D(D), IsOffloadEntry(IsOffloadEntry) {}
6431 void Enter(CodeGenFunction &CGF) override {
6432 if (!CGF.HaveInsertPoint())
6433 return;
6434 for (const auto &AllocatorData : Allocators) {
6435 CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsInit(
6436 CGF, Allocator: AllocatorData.first, AllocatorTraits: AllocatorData.second);
6437 }
6438 // This kernel does not go through the device-side runtime
6439 // init/deinit sequence (that is GPU-only), but the runtime still
6440 // needs a '<kernel>_kernel_environment' global to know how the
6441 // kernel was configured, so emit it directly here.
6442 if (IsOffloadEntry)
6443 CGF.CGM.getOpenMPRuntime().emitHostKernelEnvironment(D, CGF);
6444 }
6445 void Exit(CodeGenFunction &CGF) override {
6446 if (!CGF.HaveInsertPoint())
6447 return;
6448 for (const auto &AllocatorData : Allocators) {
6449 CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsFini(CGF,
6450 Allocator: AllocatorData.first);
6451 }
6452 }
6453};
6454} // namespace
6455
6456void CGOpenMPRuntime::emitHostKernelEnvironment(const OMPExecutableDirective &D,
6457 CodeGenFunction &CGF) {
6458 llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs Attrs;
6459 Attrs.ExecFlags = llvm::omp::OMPTgtExecModeFlags::OMP_TGT_EXEC_MODE_GENERIC;
6460 computeMinAndMaxThreadsAndTeams(D, CGF, Attrs);
6461 OMPBuilder.emitKernelEnvironment(Loc: CGF.Builder, Attrs);
6462}
6463
6464void CGOpenMPRuntime::emitTargetOutlinedFunction(
6465 const OMPExecutableDirective &D, StringRef ParentName,
6466 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6467 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6468 assert(!ParentName.empty() && "Invalid target entry parent name!");
6469 HasEmittedTargetRegion = true;
6470 SmallVector<std::pair<const Expr *, const Expr *>, 4> Allocators;
6471 for (const auto *C : D.getClausesOfKind<OMPUsesAllocatorsClause>()) {
6472 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) {
6473 const OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I);
6474 if (!D.AllocatorTraits)
6475 continue;
6476 Allocators.emplace_back(Args: D.Allocator, Args: D.AllocatorTraits);
6477 }
6478 }
6479 OMPUsesAllocatorsActionTy UsesAllocatorAction(Allocators, D, IsOffloadEntry);
6480 CodeGen.setAction(UsesAllocatorAction);
6481 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID,
6482 IsOffloadEntry, CodeGen);
6483}
6484
6485void CGOpenMPRuntime::emitUsesAllocatorsInit(CodeGenFunction &CGF,
6486 const Expr *Allocator,
6487 const Expr *AllocatorTraits) {
6488 llvm::Value *ThreadId = getThreadID(CGF, Loc: Allocator->getExprLoc());
6489 ThreadId = CGF.Builder.CreateIntCast(V: ThreadId, DestTy: CGF.IntTy, /*isSigned=*/true);
6490 // Use default memspace handle.
6491 llvm::Value *MemSpaceHandle = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
6492 llvm::Value *NumTraits = llvm::ConstantInt::get(
6493 Ty: CGF.IntTy, V: cast<ConstantArrayType>(
6494 Val: AllocatorTraits->getType()->getAsArrayTypeUnsafe())
6495 ->getSize()
6496 .getLimitedValue());
6497 LValue AllocatorTraitsLVal = CGF.EmitLValue(E: AllocatorTraits);
6498 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6499 Addr: AllocatorTraitsLVal.getAddress(), Ty: CGF.VoidPtrPtrTy, ElementTy: CGF.VoidPtrTy);
6500 AllocatorTraitsLVal = CGF.MakeAddrLValue(Addr, T: CGF.getContext().VoidPtrTy,
6501 BaseInfo: AllocatorTraitsLVal.getBaseInfo(),
6502 TBAAInfo: AllocatorTraitsLVal.getTBAAInfo());
6503 llvm::Value *Traits = Addr.emitRawPointer(CGF);
6504
6505 llvm::Value *AllocatorVal =
6506 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
6507 M&: CGM.getModule(), FnID: OMPRTL___kmpc_init_allocator),
6508 args: {ThreadId, MemSpaceHandle, NumTraits, Traits});
6509 // Store to allocator.
6510 CGF.EmitAutoVarAlloca(var: *cast<VarDecl>(
6511 Val: cast<DeclRefExpr>(Val: Allocator->IgnoreParenImpCasts())->getDecl()));
6512 LValue AllocatorLVal = CGF.EmitLValue(E: Allocator->IgnoreParenImpCasts());
6513 AllocatorVal =
6514 CGF.EmitScalarConversion(Src: AllocatorVal, SrcTy: CGF.getContext().VoidPtrTy,
6515 DstTy: Allocator->getType(), Loc: Allocator->getExprLoc());
6516 CGF.EmitStoreOfScalar(value: AllocatorVal, lvalue: AllocatorLVal);
6517}
6518
6519void CGOpenMPRuntime::emitUsesAllocatorsFini(CodeGenFunction &CGF,
6520 const Expr *Allocator) {
6521 llvm::Value *ThreadId = getThreadID(CGF, Loc: Allocator->getExprLoc());
6522 ThreadId = CGF.Builder.CreateIntCast(V: ThreadId, DestTy: CGF.IntTy, /*isSigned=*/true);
6523 LValue AllocatorLVal = CGF.EmitLValue(E: Allocator->IgnoreParenImpCasts());
6524 llvm::Value *AllocatorVal =
6525 CGF.EmitLoadOfScalar(lvalue: AllocatorLVal, Loc: Allocator->getExprLoc());
6526 AllocatorVal = CGF.EmitScalarConversion(Src: AllocatorVal, SrcTy: Allocator->getType(),
6527 DstTy: CGF.getContext().VoidPtrTy,
6528 Loc: Allocator->getExprLoc());
6529 (void)CGF.EmitRuntimeCall(
6530 callee: OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(),
6531 FnID: OMPRTL___kmpc_destroy_allocator),
6532 args: {ThreadId, AllocatorVal});
6533}
6534
6535void CGOpenMPRuntime::computeMinAndMaxThreadsAndTeams(
6536 const OMPExecutableDirective &D, CodeGenFunction &CGF,
6537 llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs &Attrs) {
6538 assert(Attrs.MaxTeams.size() == 1 && Attrs.MaxThreads.size() == 1 &&
6539 "invalid default attrs structure");
6540 int32_t &MaxTeamsVal = Attrs.MaxTeams.front();
6541 int32_t &MaxThreadsVal = Attrs.MaxThreads.front();
6542
6543 getNumTeamsExprForTargetDirective(CGF, D, MinTeamsVal&: Attrs.MinTeams.front(),
6544 MaxTeamsVal);
6545 getNumThreadsExprForTargetDirective(CGF, D, UpperBound&: MaxThreadsVal,
6546 /*UpperBoundOnly=*/true);
6547
6548 for (auto *C : D.getClausesOfKind<OMPXAttributeClause>()) {
6549 for (auto *A : C->getAttrs()) {
6550 int32_t AttrMinThreadsVal = 1, AttrMaxThreadsVal = -1;
6551 int32_t AttrMinBlocksVal = 1, AttrMaxBlocksVal = -1;
6552 if (auto *Attr = dyn_cast<CUDALaunchBoundsAttr>(Val: A))
6553 CGM.handleCUDALaunchBoundsAttr(F: nullptr, A: Attr, MaxThreadsVal: &AttrMaxThreadsVal,
6554 MinBlocksVal: &AttrMinBlocksVal, MaxClusterRankVal: &AttrMaxBlocksVal);
6555 else if (auto *Attr = dyn_cast<AMDGPUFlatWorkGroupSizeAttr>(Val: A))
6556 CGM.handleAMDGPUFlatWorkGroupSizeAttr(
6557 F: nullptr, A: Attr, /*ReqdWGS=*/nullptr, MinThreadsVal: &AttrMinThreadsVal,
6558 MaxThreadsVal: &AttrMaxThreadsVal);
6559 else
6560 continue;
6561
6562 Attrs.MinThreads.front() =
6563 std::max(a: Attrs.MinThreads.front(), b: AttrMinThreadsVal);
6564 if (AttrMaxThreadsVal > 0)
6565 MaxThreadsVal = MaxThreadsVal > 0
6566 ? std::min(a: MaxThreadsVal, b: AttrMaxThreadsVal)
6567 : AttrMaxThreadsVal;
6568 Attrs.MinTeams.front() =
6569 std::max(a: Attrs.MinTeams.front(), b: AttrMinBlocksVal);
6570 if (AttrMaxBlocksVal > 0)
6571 MaxTeamsVal = MaxTeamsVal > 0 ? std::min(a: MaxTeamsVal, b: AttrMaxBlocksVal)
6572 : AttrMaxBlocksVal;
6573 }
6574 }
6575}
6576
6577void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper(
6578 const OMPExecutableDirective &D, StringRef ParentName,
6579 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6580 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6581
6582 llvm::TargetRegionEntryInfo EntryInfo =
6583 getEntryInfoFromPresumedLoc(CGM, OMPBuilder, BeginLoc: D.getBeginLoc(), ParentName);
6584
6585 CodeGenFunction CGF(CGM, true);
6586 llvm::OpenMPIRBuilder::FunctionGenCallback &&GenerateOutlinedFunction =
6587 [&CGF, &D, &CodeGen, this](StringRef EntryFnName) {
6588 const CapturedStmt &CS = *D.getCapturedStmt(RegionKind: OMPD_target);
6589
6590 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName);
6591 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6592 if (CGM.getLangOpts().OpenMPIsTargetDevice && !isGPU())
6593 return CGF.GenerateOpenMPCapturedStmtFunctionAggregate(S: CS, D);
6594 return CGF.GenerateOpenMPCapturedStmtFunction(S: CS, D);
6595 };
6596
6597 cantFail(Err: OMPBuilder.emitTargetRegionFunction(
6598 EntryInfo, GenerateFunctionCallback&: GenerateOutlinedFunction, IsOffloadEntry, OutlinedFn,
6599 OutlinedFnID));
6600
6601 if (!OutlinedFn)
6602 return;
6603
6604 // A target body is entered once, from the kernel, and never re-entered by
6605 // the runtime, so it cannot occur in a cycle.
6606 OutlinedFn->setDoesNotRecurse();
6607
6608 CGM.getTargetCodeGenInfo().setTargetAttributes(D: nullptr, GV: OutlinedFn, M&: CGM);
6609
6610 for (auto *C : D.getClausesOfKind<OMPXAttributeClause>()) {
6611 for (auto *A : C->getAttrs()) {
6612 if (auto *Attr = dyn_cast<AMDGPUWavesPerEUAttr>(Val: A))
6613 CGM.handleAMDGPUWavesPerEUAttr(F: OutlinedFn, A: Attr);
6614 }
6615 }
6616 registerVTable(D);
6617}
6618
6619/// Checks if the expression is constant or does not have non-trivial function
6620/// calls.
6621static bool isTrivial(ASTContext &Ctx, const Expr * E) {
6622 // We can skip constant expressions.
6623 // We can skip expressions with trivial calls or simple expressions.
6624 return (E->isEvaluatable(Ctx, AllowSideEffects: Expr::SE_AllowUndefinedBehavior) ||
6625 !E->hasNonTrivialCall(Ctx)) &&
6626 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true);
6627}
6628
6629const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx,
6630 const Stmt *Body) {
6631 const Stmt *Child = Body->IgnoreContainers();
6632 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Val: Child)) {
6633 Child = nullptr;
6634 for (const Stmt *S : C->body()) {
6635 if (const auto *E = dyn_cast<Expr>(Val: S)) {
6636 if (isTrivial(Ctx, E))
6637 continue;
6638 }
6639 // Some of the statements can be ignored.
6640 if (isa<AsmStmt>(Val: S) || isa<NullStmt>(Val: S) || isa<OMPFlushDirective>(Val: S) ||
6641 isa<OMPBarrierDirective>(Val: S) || isa<OMPTaskyieldDirective>(Val: S))
6642 continue;
6643 // Analyze declarations.
6644 if (const auto *DS = dyn_cast<DeclStmt>(Val: S)) {
6645 if (llvm::all_of(Range: DS->decls(), P: [](const Decl *D) {
6646 if (isa<EmptyDecl>(Val: D) || isa<DeclContext>(Val: D) ||
6647 isa<TypeDecl>(Val: D) || isa<PragmaCommentDecl>(Val: D) ||
6648 isa<PragmaDetectMismatchDecl>(Val: D) || isa<UsingDecl>(Val: D) ||
6649 isa<UsingDirectiveDecl>(Val: D) ||
6650 isa<OMPDeclareReductionDecl>(Val: D) ||
6651 isa<OMPThreadPrivateDecl>(Val: D) || isa<OMPAllocateDecl>(Val: D))
6652 return true;
6653 const auto *VD = dyn_cast<VarDecl>(Val: D);
6654 if (!VD)
6655 return false;
6656 return VD->hasGlobalStorage() || !VD->isUsed();
6657 }))
6658 continue;
6659 }
6660 // Found multiple children - cannot get the one child only.
6661 if (Child)
6662 return nullptr;
6663 Child = S;
6664 }
6665 if (Child)
6666 Child = Child->IgnoreContainers();
6667 }
6668 return Child;
6669}
6670
6671const Expr *CGOpenMPRuntime::getNumTeamsExprForTargetDirective(
6672 CodeGenFunction &CGF, const OMPExecutableDirective &D, int32_t &MinTeamsVal,
6673 int32_t &MaxTeamsVal) {
6674
6675 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6676 assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6677 "Expected target-based executable directive.");
6678 switch (DirectiveKind) {
6679 case OMPD_target: {
6680 const auto *CS = D.getInnermostCapturedStmt();
6681 const auto *Body =
6682 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
6683 const Stmt *ChildStmt =
6684 CGOpenMPRuntime::getSingleCompoundChild(Ctx&: CGF.getContext(), Body);
6685 if (const auto *NestedDir =
6686 dyn_cast_or_null<OMPExecutableDirective>(Val: ChildStmt)) {
6687 if (isOpenMPTeamsDirective(DKind: NestedDir->getDirectiveKind())) {
6688 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) {
6689 const Expr *NumTeams = NestedDir->getSingleClause<OMPNumTeamsClause>()
6690 ->getNumTeams()
6691 .front();
6692 if (NumTeams->isIntegerConstantExpr(Ctx: CGF.getContext()))
6693 if (auto Constant =
6694 NumTeams->getIntegerConstantExpr(Ctx: CGF.getContext()))
6695 MinTeamsVal = MaxTeamsVal = Constant->getExtValue();
6696 return NumTeams;
6697 }
6698 MinTeamsVal = MaxTeamsVal = 0;
6699 return nullptr;
6700 }
6701 MinTeamsVal = MaxTeamsVal = 1;
6702 return nullptr;
6703 }
6704 // A value of -1 is used to check if we need to emit no teams region
6705 MinTeamsVal = MaxTeamsVal = -1;
6706 return nullptr;
6707 }
6708 case OMPD_target_teams_loop:
6709 case OMPD_target_teams:
6710 case OMPD_target_teams_distribute:
6711 case OMPD_target_teams_distribute_simd:
6712 case OMPD_target_teams_distribute_parallel_for:
6713 case OMPD_target_teams_distribute_parallel_for_simd: {
6714 if (D.hasClausesOfKind<OMPNumTeamsClause>()) {
6715 const Expr *NumTeams =
6716 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams().front();
6717 if (NumTeams->isIntegerConstantExpr(Ctx: CGF.getContext()))
6718 if (auto Constant = NumTeams->getIntegerConstantExpr(Ctx: CGF.getContext()))
6719 MinTeamsVal = MaxTeamsVal = Constant->getExtValue();
6720 return NumTeams;
6721 }
6722 MinTeamsVal = MaxTeamsVal = 0;
6723 return nullptr;
6724 }
6725 case OMPD_target_parallel:
6726 case OMPD_target_parallel_for:
6727 case OMPD_target_parallel_for_simd:
6728 case OMPD_target_parallel_loop:
6729 case OMPD_target_simd:
6730 MinTeamsVal = MaxTeamsVal = 1;
6731 return nullptr;
6732 case OMPD_parallel:
6733 case OMPD_for:
6734 case OMPD_parallel_for:
6735 case OMPD_parallel_loop:
6736 case OMPD_parallel_master:
6737 case OMPD_parallel_sections:
6738 case OMPD_for_simd:
6739 case OMPD_parallel_for_simd:
6740 case OMPD_cancel:
6741 case OMPD_cancellation_point:
6742 case OMPD_ordered_standalone:
6743 case OMPD_ordered_blockassoc:
6744 case OMPD_threadprivate:
6745 case OMPD_allocate:
6746 case OMPD_task:
6747 case OMPD_simd:
6748 case OMPD_tile:
6749 case OMPD_unroll:
6750 case OMPD_sections:
6751 case OMPD_section:
6752 case OMPD_single:
6753 case OMPD_master:
6754 case OMPD_critical:
6755 case OMPD_taskyield:
6756 case OMPD_barrier:
6757 case OMPD_taskwait:
6758 case OMPD_taskgroup:
6759 case OMPD_atomic:
6760 case OMPD_flush:
6761 case OMPD_depobj:
6762 case OMPD_scan:
6763 case OMPD_teams:
6764 case OMPD_target_data:
6765 case OMPD_target_exit_data:
6766 case OMPD_target_enter_data:
6767 case OMPD_distribute:
6768 case OMPD_distribute_simd:
6769 case OMPD_distribute_parallel_for:
6770 case OMPD_distribute_parallel_for_simd:
6771 case OMPD_teams_distribute:
6772 case OMPD_teams_distribute_simd:
6773 case OMPD_teams_distribute_parallel_for:
6774 case OMPD_teams_distribute_parallel_for_simd:
6775 case OMPD_target_update:
6776 case OMPD_declare_simd:
6777 case OMPD_declare_variant:
6778 case OMPD_begin_declare_variant:
6779 case OMPD_end_declare_variant:
6780 case OMPD_declare_target:
6781 case OMPD_end_declare_target:
6782 case OMPD_declare_reduction:
6783 case OMPD_declare_mapper:
6784 case OMPD_taskloop:
6785 case OMPD_taskloop_simd:
6786 case OMPD_master_taskloop:
6787 case OMPD_master_taskloop_simd:
6788 case OMPD_parallel_master_taskloop:
6789 case OMPD_parallel_master_taskloop_simd:
6790 case OMPD_requires:
6791 case OMPD_metadirective:
6792 case OMPD_unknown:
6793 break;
6794 default:
6795 break;
6796 }
6797 llvm_unreachable("Unexpected directive kind.");
6798}
6799
6800llvm::Value *CGOpenMPRuntime::emitNumTeamsForTargetDirective(
6801 CodeGenFunction &CGF, const OMPExecutableDirective &D) {
6802 assert(!CGF.getLangOpts().OpenMPIsTargetDevice &&
6803 "Clauses associated with the teams directive expected to be emitted "
6804 "only for the host!");
6805 CGBuilderTy &Bld = CGF.Builder;
6806 int32_t MinNT = -1, MaxNT = -1;
6807 const Expr *NumTeams =
6808 getNumTeamsExprForTargetDirective(CGF, D, MinTeamsVal&: MinNT, MaxTeamsVal&: MaxNT);
6809 if (NumTeams != nullptr) {
6810 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6811
6812 switch (DirectiveKind) {
6813 case OMPD_target: {
6814 const auto *CS = D.getInnermostCapturedStmt();
6815 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6816 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6817 llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(E: NumTeams,
6818 /*IgnoreResultAssign*/ true);
6819 return Bld.CreateIntCast(V: NumTeamsVal, DestTy: CGF.Int32Ty,
6820 /*isSigned=*/true);
6821 }
6822 case OMPD_target_teams:
6823 case OMPD_target_teams_distribute:
6824 case OMPD_target_teams_distribute_simd:
6825 case OMPD_target_teams_distribute_parallel_for:
6826 case OMPD_target_teams_distribute_parallel_for_simd: {
6827 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF);
6828 llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(E: NumTeams,
6829 /*IgnoreResultAssign*/ true);
6830 return Bld.CreateIntCast(V: NumTeamsVal, DestTy: CGF.Int32Ty,
6831 /*isSigned=*/true);
6832 }
6833 default:
6834 break;
6835 }
6836 }
6837
6838 assert(MinNT == MaxNT && "Num threads ranges require handling here.");
6839 return llvm::ConstantInt::getSigned(Ty: CGF.Int32Ty, V: MinNT);
6840}
6841
6842/// Merge the thread count upper bound \p Val into \p UpperBound.
6843///
6844/// \p UpperBound is -1 while no thread limiting clause has been seen, 0 once
6845/// one has been seen whose value is not known at compile time, and otherwise
6846/// the smallest constant bound found so far.
6847///
6848/// Thread limiting clauses compose by taking the minimum, so a constant bound
6849/// stays valid whatever the clauses that are not compile time constants
6850/// evaluate to. That makes it correct to replace the 0 marker with \p Val, and
6851/// necessary to keep a clause from raising a smaller bound found earlier.
6852static void mergeThreadCountUpperBound(int32_t &UpperBound, int32_t Val) {
6853 UpperBound = UpperBound > 0 ? std::min(a: UpperBound, b: Val) : Val;
6854}
6855
6856/// Check for a num threads constant value (stored in \p DefaultVal), or
6857/// expression (stored in \p E). If the value is conditional (via an if-clause),
6858/// store the condition in \p CondVal. If \p E, and \p CondVal respectively, are
6859/// nullptr, no expression evaluation is perfomed.
6860static void getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS,
6861 const Expr **E, int32_t &UpperBound,
6862 bool UpperBoundOnly, llvm::Value **CondVal) {
6863 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6864 Ctx&: CGF.getContext(), Body: CS->getCapturedStmt());
6865 const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Val: Child);
6866 if (!Dir)
6867 return;
6868
6869 if (isOpenMPParallelDirective(DKind: Dir->getDirectiveKind())) {
6870 // Handle if clause. If if clause present, the number of threads is
6871 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
6872 if (CondVal && Dir->hasClausesOfKind<OMPIfClause>()) {
6873 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6874 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6875 const OMPIfClause *IfClause = nullptr;
6876 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) {
6877 if (C->getNameModifier() == OMPD_unknown ||
6878 C->getNameModifier() == OMPD_parallel) {
6879 IfClause = C;
6880 break;
6881 }
6882 }
6883 if (IfClause) {
6884 const Expr *CondExpr = IfClause->getCondition();
6885 bool Result;
6886 if (CondExpr->EvaluateAsBooleanCondition(Result, Ctx: CGF.getContext())) {
6887 if (!Result) {
6888 UpperBound = 1;
6889 return;
6890 }
6891 } else {
6892 CodeGenFunction::LexicalScope Scope(CGF, CondExpr->getSourceRange());
6893 if (const auto *PreInit =
6894 cast_or_null<DeclStmt>(Val: IfClause->getPreInitStmt())) {
6895 for (const auto *I : PreInit->decls()) {
6896 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6897 CGF.EmitVarDecl(D: cast<VarDecl>(Val: *I));
6898 } else {
6899 CodeGenFunction::AutoVarEmission Emission =
6900 CGF.EmitAutoVarAlloca(var: cast<VarDecl>(Val: *I));
6901 CGF.EmitAutoVarCleanups(emission: Emission);
6902 }
6903 }
6904 *CondVal = CGF.EvaluateExprAsBool(E: CondExpr);
6905 }
6906 }
6907 }
6908 }
6909 // Check the value of num_threads clause iff if clause was not specified
6910 // or is not evaluated to false.
6911 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) {
6912 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6913 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6914 const auto *NumThreadsClause =
6915 Dir->getSingleClause<OMPNumThreadsClause>();
6916 const Expr *NTExpr = NumThreadsClause->getNumThreads().front();
6917 if (NTExpr->isIntegerConstantExpr(Ctx: CGF.getContext()))
6918 if (auto Constant = NTExpr->getIntegerConstantExpr(Ctx: CGF.getContext()))
6919 mergeThreadCountUpperBound(
6920 UpperBound, Val: static_cast<int32_t>(Constant->getZExtValue()));
6921 // If we haven't found a upper bound, remember we saw a thread limiting
6922 // clause.
6923 if (UpperBound == -1)
6924 UpperBound = 0;
6925 if (!E)
6926 return;
6927 CodeGenFunction::LexicalScope Scope(CGF, NTExpr->getSourceRange());
6928 if (const auto *PreInit =
6929 cast_or_null<DeclStmt>(Val: NumThreadsClause->getPreInitStmt())) {
6930 for (const auto *I : PreInit->decls()) {
6931 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6932 CGF.EmitVarDecl(D: cast<VarDecl>(Val: *I));
6933 } else {
6934 CodeGenFunction::AutoVarEmission Emission =
6935 CGF.EmitAutoVarAlloca(var: cast<VarDecl>(Val: *I));
6936 CGF.EmitAutoVarCleanups(emission: Emission);
6937 }
6938 }
6939 }
6940 *E = NTExpr;
6941 }
6942 return;
6943 }
6944 if (isOpenMPSimdDirective(DKind: Dir->getDirectiveKind()))
6945 UpperBound = 1;
6946}
6947
6948const Expr *CGOpenMPRuntime::getNumThreadsExprForTargetDirective(
6949 CodeGenFunction &CGF, const OMPExecutableDirective &D, int32_t &UpperBound,
6950 bool UpperBoundOnly, llvm::Value **CondVal, const Expr **ThreadLimitExpr) {
6951 assert((!CGF.getLangOpts().OpenMPIsTargetDevice || UpperBoundOnly) &&
6952 "Clauses associated with the teams directive expected to be emitted "
6953 "only for the host!");
6954 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6955 assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6956 "Expected target-based executable directive.");
6957
6958 const Expr *NT = nullptr;
6959 const Expr **NTPtr = UpperBoundOnly ? nullptr : &NT;
6960
6961 auto CheckForConstExpr = [&](const Expr *E, const Expr **EPtr) {
6962 if (E->isIntegerConstantExpr(Ctx: CGF.getContext())) {
6963 if (auto Constant = E->getIntegerConstantExpr(Ctx: CGF.getContext()))
6964 mergeThreadCountUpperBound(
6965 UpperBound, Val: static_cast<int32_t>(Constant->getZExtValue()));
6966 }
6967 // If we haven't found a upper bound, remember we saw a thread limiting
6968 // clause.
6969 if (UpperBound == -1)
6970 UpperBound = 0;
6971 if (EPtr)
6972 *EPtr = E;
6973 };
6974
6975 auto ReturnSequential = [&]() {
6976 UpperBound = 1;
6977 return NT;
6978 };
6979
6980 switch (DirectiveKind) {
6981 case OMPD_target: {
6982 const CapturedStmt *CS = D.getInnermostCapturedStmt();
6983 getNumThreads(CGF, CS, E: NTPtr, UpperBound, UpperBoundOnly, CondVal);
6984 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6985 Ctx&: CGF.getContext(), Body: CS->getCapturedStmt());
6986 // TODO: The standard is not clear how to resolve two thread limit clauses,
6987 // let's pick the teams one if it's present, otherwise the target one.
6988 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
6989 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Val: Child)) {
6990 if (const auto *TLC = Dir->getSingleClause<OMPThreadLimitClause>()) {
6991 ThreadLimitClause = TLC;
6992 if (ThreadLimitExpr) {
6993 CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6994 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6995 CodeGenFunction::LexicalScope Scope(
6996 CGF,
6997 ThreadLimitClause->getThreadLimit().front()->getSourceRange());
6998 if (const auto *PreInit =
6999 cast_or_null<DeclStmt>(Val: ThreadLimitClause->getPreInitStmt())) {
7000 for (const auto *I : PreInit->decls()) {
7001 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7002 CGF.EmitVarDecl(D: cast<VarDecl>(Val: *I));
7003 } else {
7004 CodeGenFunction::AutoVarEmission Emission =
7005 CGF.EmitAutoVarAlloca(var: cast<VarDecl>(Val: *I));
7006 CGF.EmitAutoVarCleanups(emission: Emission);
7007 }
7008 }
7009 }
7010 }
7011 }
7012 }
7013 if (ThreadLimitClause)
7014 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7015 ThreadLimitExpr);
7016 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Val: Child)) {
7017 if (isOpenMPTeamsDirective(DKind: Dir->getDirectiveKind()) &&
7018 !isOpenMPDistributeDirective(DKind: Dir->getDirectiveKind())) {
7019 CS = Dir->getInnermostCapturedStmt();
7020 // Now that the 'teams' level has been peeled off, the remainder is
7021 // shaped like a 'target teams' region, so pick up the num_threads of
7022 // the directive nested in it the same way the OMPD_target_teams case
7023 // below does. Without this the upper bound of a construct written as
7024 // 'target' / 'teams' / 'distribute parallel for' would stay at the
7025 // default, while every combined spelling of the same construct honors
7026 // the clause. Only the bound is taken here: passing null for the
7027 // expression and the condition keeps this from emitting anything, so
7028 // the value the host passes to the kernel launch is left as it was.
7029 getNumThreads(CGF, CS, /*E=*/nullptr, UpperBound, UpperBoundOnly,
7030 /*CondVal=*/nullptr);
7031 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7032 Ctx&: CGF.getContext(), Body: CS->getCapturedStmt());
7033 Dir = dyn_cast_or_null<OMPExecutableDirective>(Val: Child);
7034 }
7035 if (Dir && isOpenMPParallelDirective(DKind: Dir->getDirectiveKind())) {
7036 CS = Dir->getInnermostCapturedStmt();
7037 getNumThreads(CGF, CS, E: NTPtr, UpperBound, UpperBoundOnly, CondVal);
7038 } else if (Dir && isOpenMPSimdDirective(DKind: Dir->getDirectiveKind()))
7039 return ReturnSequential();
7040 }
7041 return NT;
7042 }
7043 case OMPD_target_teams: {
7044 if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7045 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7046 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7047 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7048 ThreadLimitExpr);
7049 }
7050 const CapturedStmt *CS = D.getInnermostCapturedStmt();
7051 getNumThreads(CGF, CS, E: NTPtr, UpperBound, UpperBoundOnly, CondVal);
7052 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7053 Ctx&: CGF.getContext(), Body: CS->getCapturedStmt());
7054 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Val: Child)) {
7055 if (Dir->getDirectiveKind() == OMPD_distribute) {
7056 CS = Dir->getInnermostCapturedStmt();
7057 getNumThreads(CGF, CS, E: NTPtr, UpperBound, UpperBoundOnly, CondVal);
7058 }
7059 }
7060 return NT;
7061 }
7062 case OMPD_target_teams_distribute:
7063 if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7064 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7065 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7066 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7067 ThreadLimitExpr);
7068 }
7069 getNumThreads(CGF, CS: D.getInnermostCapturedStmt(), E: NTPtr, UpperBound,
7070 UpperBoundOnly, CondVal);
7071 return NT;
7072 case OMPD_target_teams_loop:
7073 case OMPD_target_parallel_loop:
7074 case OMPD_target_parallel:
7075 case OMPD_target_parallel_for:
7076 case OMPD_target_parallel_for_simd:
7077 case OMPD_target_teams_distribute_parallel_for:
7078 case OMPD_target_teams_distribute_parallel_for_simd: {
7079 if (CondVal && D.hasClausesOfKind<OMPIfClause>()) {
7080 const OMPIfClause *IfClause = nullptr;
7081 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) {
7082 if (C->getNameModifier() == OMPD_unknown ||
7083 C->getNameModifier() == OMPD_parallel) {
7084 IfClause = C;
7085 break;
7086 }
7087 }
7088 if (IfClause) {
7089 const Expr *Cond = IfClause->getCondition();
7090 bool Result;
7091 if (Cond->EvaluateAsBooleanCondition(Result, Ctx: CGF.getContext())) {
7092 if (!Result)
7093 return ReturnSequential();
7094 } else {
7095 CodeGenFunction::RunCleanupsScope Scope(CGF);
7096 *CondVal = CGF.EvaluateExprAsBool(E: Cond);
7097 }
7098 }
7099 }
7100 if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7101 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7102 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7103 CheckForConstExpr(ThreadLimitClause->getThreadLimit().front(),
7104 ThreadLimitExpr);
7105 }
7106 if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
7107 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
7108 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
7109 CheckForConstExpr(NumThreadsClause->getNumThreads().front(), nullptr);
7110 return NumThreadsClause->getNumThreads().front();
7111 }
7112 return NT;
7113 }
7114 case OMPD_target_teams_distribute_simd:
7115 case OMPD_target_simd:
7116 return ReturnSequential();
7117 default:
7118 break;
7119 }
7120 llvm_unreachable("Unsupported directive kind.");
7121}
7122
7123llvm::Value *CGOpenMPRuntime::emitNumThreadsForTargetDirective(
7124 CodeGenFunction &CGF, const OMPExecutableDirective &D) {
7125 llvm::Value *NumThreadsVal = nullptr;
7126 llvm::Value *CondVal = nullptr;
7127 llvm::Value *ThreadLimitVal = nullptr;
7128 const Expr *ThreadLimitExpr = nullptr;
7129 int32_t UpperBound = -1;
7130
7131 const Expr *NT = getNumThreadsExprForTargetDirective(
7132 CGF, D, UpperBound, /* UpperBoundOnly */ false, CondVal: &CondVal,
7133 ThreadLimitExpr: &ThreadLimitExpr);
7134
7135 // Thread limit expressions are used below, emit them.
7136 if (ThreadLimitExpr) {
7137 ThreadLimitVal =
7138 CGF.EmitScalarExpr(E: ThreadLimitExpr, /*IgnoreResultAssign=*/true);
7139 ThreadLimitVal = CGF.Builder.CreateIntCast(V: ThreadLimitVal, DestTy: CGF.Int32Ty,
7140 /*isSigned=*/false);
7141 }
7142
7143 // Generate the num teams expression.
7144 if (UpperBound == 1) {
7145 NumThreadsVal = CGF.Builder.getInt32(C: UpperBound);
7146 } else if (NT) {
7147 NumThreadsVal = CGF.EmitScalarExpr(E: NT, /*IgnoreResultAssign=*/true);
7148 NumThreadsVal = CGF.Builder.CreateIntCast(V: NumThreadsVal, DestTy: CGF.Int32Ty,
7149 /*isSigned=*/false);
7150 } else if (ThreadLimitVal) {
7151 // If we do not have a num threads value but a thread limit, replace the
7152 // former with the latter. We know handled the thread limit expression.
7153 NumThreadsVal = ThreadLimitVal;
7154 ThreadLimitVal = nullptr;
7155 } else {
7156 // Default to "0" which means runtime choice.
7157 assert(!ThreadLimitVal && "Default not applicable with thread limit value");
7158 NumThreadsVal = CGF.Builder.getInt32(C: 0);
7159 }
7160
7161 // Handle if clause. If if clause present, the number of threads is
7162 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
7163 if (CondVal) {
7164 CodeGenFunction::RunCleanupsScope Scope(CGF);
7165 NumThreadsVal = CGF.Builder.CreateSelect(C: CondVal, True: NumThreadsVal,
7166 False: CGF.Builder.getInt32(C: 1));
7167 }
7168
7169 // If the thread limit and num teams expression were present, take the
7170 // minimum.
7171 if (ThreadLimitVal) {
7172 NumThreadsVal = CGF.Builder.CreateSelect(
7173 C: CGF.Builder.CreateICmpULT(LHS: ThreadLimitVal, RHS: NumThreadsVal),
7174 True: ThreadLimitVal, False: NumThreadsVal);
7175 }
7176
7177 return NumThreadsVal;
7178}
7179
7180namespace {
7181LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
7182
7183// Utility to handle information from clauses associated with a given
7184// construct that use mappable expressions (e.g. 'map' clause, 'to' clause).
7185// It provides a convenient interface to obtain the information and generate
7186// code for that information.
7187class MappableExprsHandler {
7188public:
7189 /// Custom comparator for attach-pointer expressions that compares them by
7190 /// complexity (i.e. their component-depth) first, then by the order in which
7191 /// they were computed by collectAttachPtrExprInfo(), if they are semantically
7192 /// different.
7193 struct AttachPtrExprComparator {
7194 const MappableExprsHandler &Handler;
7195 // Cache of previous equality comparison results.
7196 mutable llvm::DenseMap<std::pair<const Expr *, const Expr *>, bool>
7197 CachedEqualityComparisons;
7198
7199 AttachPtrExprComparator(const MappableExprsHandler &H) : Handler(H) {}
7200 AttachPtrExprComparator() = delete;
7201
7202 // Return true iff LHS is "less than" RHS.
7203 bool operator()(const Expr *LHS, const Expr *RHS) const {
7204 if (LHS == RHS)
7205 return false;
7206
7207 // First, compare by complexity (depth)
7208 const auto ItLHS = Handler.AttachPtrComponentDepthMap.find(Val: LHS);
7209 const auto ItRHS = Handler.AttachPtrComponentDepthMap.find(Val: RHS);
7210
7211 std::optional<size_t> DepthLHS =
7212 (ItLHS != Handler.AttachPtrComponentDepthMap.end()) ? ItLHS->second
7213 : std::nullopt;
7214 std::optional<size_t> DepthRHS =
7215 (ItRHS != Handler.AttachPtrComponentDepthMap.end()) ? ItRHS->second
7216 : std::nullopt;
7217
7218 // std::nullopt (no attach pointer) has lowest complexity
7219 if (!DepthLHS.has_value() && !DepthRHS.has_value()) {
7220 // Both have same complexity, now check semantic equality
7221 if (areEqual(LHS, RHS))
7222 return false;
7223 // Different semantically, compare by computation order
7224 return wasComputedBefore(LHS, RHS);
7225 }
7226 if (!DepthLHS.has_value())
7227 return true; // LHS has lower complexity
7228 if (!DepthRHS.has_value())
7229 return false; // RHS has lower complexity
7230
7231 // Both have values, compare by depth (lower depth = lower complexity)
7232 if (DepthLHS.value() != DepthRHS.value())
7233 return DepthLHS.value() < DepthRHS.value();
7234
7235 // Same complexity, now check semantic equality
7236 if (areEqual(LHS, RHS))
7237 return false;
7238 // Different semantically, compare by computation order
7239 return wasComputedBefore(LHS, RHS);
7240 }
7241
7242 public:
7243 /// Return true if \p LHS and \p RHS are semantically equal. Uses pre-cached
7244 /// results, if available, otherwise does a recursive semantic comparison.
7245 bool areEqual(const Expr *LHS, const Expr *RHS) const {
7246 // Check cache first for faster lookup
7247 const auto CachedResultIt = CachedEqualityComparisons.find(Val: {LHS, RHS});
7248 if (CachedResultIt != CachedEqualityComparisons.end())
7249 return CachedResultIt->second;
7250
7251 bool ComparisonResult = areSemanticallyEqual(LHS, RHS);
7252
7253 // Cache the result for future lookups (both orders since semantic
7254 // equality is commutative)
7255 CachedEqualityComparisons[{LHS, RHS}] = ComparisonResult;
7256 CachedEqualityComparisons[{RHS, LHS}] = ComparisonResult;
7257 return ComparisonResult;
7258 }
7259
7260 /// Compare the two attach-ptr expressions by their computation order.
7261 /// Returns true iff LHS was computed before RHS by
7262 /// collectAttachPtrExprInfo().
7263 bool wasComputedBefore(const Expr *LHS, const Expr *RHS) const {
7264 const size_t &OrderLHS = Handler.AttachPtrComputationOrderMap.at(Val: LHS);
7265 const size_t &OrderRHS = Handler.AttachPtrComputationOrderMap.at(Val: RHS);
7266
7267 return OrderLHS < OrderRHS;
7268 }
7269
7270 private:
7271 /// Helper function to compare attach-pointer expressions semantically.
7272 /// This function handles various expression types that can be part of an
7273 /// attach-pointer.
7274 /// TODO: Not urgent, but we should ideally return true when comparing
7275 /// `p[10]`, `*(p + 10)`, `*(p + 5 + 5)`, `p[10:1]` etc.
7276 bool areSemanticallyEqual(const Expr *LHS, const Expr *RHS) const {
7277 if (LHS == RHS)
7278 return true;
7279
7280 // If only one is null, they aren't equal
7281 if (!LHS || !RHS)
7282 return false;
7283
7284 ASTContext &Ctx = Handler.CGF.getContext();
7285 // Strip away parentheses and no-op casts to get to the core expression
7286 LHS = LHS->IgnoreParenNoopCasts(Ctx);
7287 RHS = RHS->IgnoreParenNoopCasts(Ctx);
7288
7289 // Direct pointer comparison of the underlying expressions
7290 if (LHS == RHS)
7291 return true;
7292
7293 // Check if the expression classes match
7294 if (LHS->getStmtClass() != RHS->getStmtClass())
7295 return false;
7296
7297 // Handle DeclRefExpr (variable references)
7298 if (const auto *LD = dyn_cast<DeclRefExpr>(Val: LHS)) {
7299 const auto *RD = dyn_cast<DeclRefExpr>(Val: RHS);
7300 if (!RD)
7301 return false;
7302 return LD->getDecl()->getCanonicalDecl() ==
7303 RD->getDecl()->getCanonicalDecl();
7304 }
7305
7306 // Handle ArraySubscriptExpr (array indexing like a[i])
7307 if (const auto *LA = dyn_cast<ArraySubscriptExpr>(Val: LHS)) {
7308 const auto *RA = dyn_cast<ArraySubscriptExpr>(Val: RHS);
7309 if (!RA)
7310 return false;
7311 return areSemanticallyEqual(LHS: LA->getBase(), RHS: RA->getBase()) &&
7312 areSemanticallyEqual(LHS: LA->getIdx(), RHS: RA->getIdx());
7313 }
7314
7315 // Handle MemberExpr (member access like s.m or p->m)
7316 if (const auto *LM = dyn_cast<MemberExpr>(Val: LHS)) {
7317 const auto *RM = dyn_cast<MemberExpr>(Val: RHS);
7318 if (!RM)
7319 return false;
7320 if (LM->getMemberDecl()->getCanonicalDecl() !=
7321 RM->getMemberDecl()->getCanonicalDecl())
7322 return false;
7323 return areSemanticallyEqual(LHS: LM->getBase(), RHS: RM->getBase());
7324 }
7325
7326 // Handle UnaryOperator (unary operations like *p, &x, etc.)
7327 if (const auto *LU = dyn_cast<UnaryOperator>(Val: LHS)) {
7328 const auto *RU = dyn_cast<UnaryOperator>(Val: RHS);
7329 if (!RU)
7330 return false;
7331 if (LU->getOpcode() != RU->getOpcode())
7332 return false;
7333 return areSemanticallyEqual(LHS: LU->getSubExpr(), RHS: RU->getSubExpr());
7334 }
7335
7336 // Handle BinaryOperator (binary operations like p + offset)
7337 if (const auto *LB = dyn_cast<BinaryOperator>(Val: LHS)) {
7338 const auto *RB = dyn_cast<BinaryOperator>(Val: RHS);
7339 if (!RB)
7340 return false;
7341 if (LB->getOpcode() != RB->getOpcode())
7342 return false;
7343 return areSemanticallyEqual(LHS: LB->getLHS(), RHS: RB->getLHS()) &&
7344 areSemanticallyEqual(LHS: LB->getRHS(), RHS: RB->getRHS());
7345 }
7346
7347 // Handle ArraySectionExpr (array sections like a[0:1])
7348 // Attach pointers should not contain array-sections, but currently we
7349 // don't emit an error.
7350 if (const auto *LAS = dyn_cast<ArraySectionExpr>(Val: LHS)) {
7351 const auto *RAS = dyn_cast<ArraySectionExpr>(Val: RHS);
7352 if (!RAS)
7353 return false;
7354 return areSemanticallyEqual(LHS: LAS->getBase(), RHS: RAS->getBase()) &&
7355 areSemanticallyEqual(LHS: LAS->getLowerBound(),
7356 RHS: RAS->getLowerBound()) &&
7357 areSemanticallyEqual(LHS: LAS->getLength(), RHS: RAS->getLength());
7358 }
7359
7360 // Handle CastExpr (explicit casts)
7361 if (const auto *LC = dyn_cast<CastExpr>(Val: LHS)) {
7362 const auto *RC = dyn_cast<CastExpr>(Val: RHS);
7363 if (!RC)
7364 return false;
7365 if (LC->getCastKind() != RC->getCastKind())
7366 return false;
7367 return areSemanticallyEqual(LHS: LC->getSubExpr(), RHS: RC->getSubExpr());
7368 }
7369
7370 // Handle CXXThisExpr (this pointer)
7371 if (isa<CXXThisExpr>(Val: LHS) && isa<CXXThisExpr>(Val: RHS))
7372 return true;
7373
7374 // Handle IntegerLiteral (integer constants)
7375 if (const auto *LI = dyn_cast<IntegerLiteral>(Val: LHS)) {
7376 const auto *RI = dyn_cast<IntegerLiteral>(Val: RHS);
7377 if (!RI)
7378 return false;
7379 return LI->getValue() == RI->getValue();
7380 }
7381
7382 // Handle CharacterLiteral (character constants)
7383 if (const auto *LC = dyn_cast<CharacterLiteral>(Val: LHS)) {
7384 const auto *RC = dyn_cast<CharacterLiteral>(Val: RHS);
7385 if (!RC)
7386 return false;
7387 return LC->getValue() == RC->getValue();
7388 }
7389
7390 // Handle FloatingLiteral (floating point constants)
7391 if (const auto *LF = dyn_cast<FloatingLiteral>(Val: LHS)) {
7392 const auto *RF = dyn_cast<FloatingLiteral>(Val: RHS);
7393 if (!RF)
7394 return false;
7395 // Use bitwise comparison for floating point literals
7396 return LF->getValue().bitwiseIsEqual(RHS: RF->getValue());
7397 }
7398
7399 // Handle StringLiteral (string constants)
7400 if (const auto *LS = dyn_cast<StringLiteral>(Val: LHS)) {
7401 const auto *RS = dyn_cast<StringLiteral>(Val: RHS);
7402 if (!RS)
7403 return false;
7404 return LS->getString() == RS->getString();
7405 }
7406
7407 // Handle CXXNullPtrLiteralExpr (nullptr)
7408 if (isa<CXXNullPtrLiteralExpr>(Val: LHS) && isa<CXXNullPtrLiteralExpr>(Val: RHS))
7409 return true;
7410
7411 // Handle CXXBoolLiteralExpr (true/false)
7412 if (const auto *LB = dyn_cast<CXXBoolLiteralExpr>(Val: LHS)) {
7413 const auto *RB = dyn_cast<CXXBoolLiteralExpr>(Val: RHS);
7414 if (!RB)
7415 return false;
7416 return LB->getValue() == RB->getValue();
7417 }
7418
7419 // Fallback for other forms - use the existing comparison method
7420 return Expr::isSameComparisonOperand(E1: LHS, E2: RHS);
7421 }
7422 };
7423
7424 /// Get the offset of the OMP_MAP_MEMBER_OF field.
7425 static unsigned getFlagMemberOffset() {
7426 unsigned Offset = 0;
7427 for (uint64_t Remain =
7428 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
7429 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF);
7430 !(Remain & 1); Remain = Remain >> 1)
7431 Offset++;
7432 return Offset;
7433 }
7434
7435 /// Class that holds debugging information for a data mapping to be passed to
7436 /// the runtime library.
7437 class MappingExprInfo {
7438 /// The variable declaration used for the data mapping.
7439 const ValueDecl *MapDecl = nullptr;
7440 /// The original expression used in the map clause, or null if there is
7441 /// none.
7442 const Expr *MapExpr = nullptr;
7443
7444 public:
7445 MappingExprInfo(const ValueDecl *MapDecl, const Expr *MapExpr = nullptr)
7446 : MapDecl(MapDecl), MapExpr(MapExpr) {}
7447
7448 const ValueDecl *getMapDecl() const { return MapDecl; }
7449 const Expr *getMapExpr() const { return MapExpr; }
7450 };
7451
7452 using DeviceInfoTy = llvm::OpenMPIRBuilder::DeviceInfoTy;
7453 using MapBaseValuesArrayTy = llvm::OpenMPIRBuilder::MapValuesArrayTy;
7454 using MapValuesArrayTy = llvm::OpenMPIRBuilder::MapValuesArrayTy;
7455 using MapFlagsArrayTy = llvm::OpenMPIRBuilder::MapFlagsArrayTy;
7456 using MapDimArrayTy = llvm::OpenMPIRBuilder::MapDimArrayTy;
7457 using MapNonContiguousArrayTy =
7458 llvm::OpenMPIRBuilder::MapNonContiguousArrayTy;
7459 using MapExprsArrayTy = SmallVector<MappingExprInfo, 4>;
7460 using MapValueDeclsArrayTy = SmallVector<const ValueDecl *, 4>;
7461 using MapData =
7462 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef,
7463 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>,
7464 bool /*IsImplicit*/, const ValueDecl *, const Expr *>;
7465 using MapDataArrayTy = SmallVector<MapData, 4>;
7466
7467 /// This structure contains combined information generated for mappable
7468 /// clauses, including base pointers, pointers, sizes, map types, user-defined
7469 /// mappers, and non-contiguous information.
7470 struct MapCombinedInfoTy : llvm::OpenMPIRBuilder::MapInfosTy {
7471 MapExprsArrayTy Exprs;
7472 MapValueDeclsArrayTy Mappers;
7473 MapValueDeclsArrayTy DevicePtrDecls;
7474
7475 /// Append arrays in \a CurInfo.
7476 void append(MapCombinedInfoTy &CurInfo) {
7477 Exprs.append(in_start: CurInfo.Exprs.begin(), in_end: CurInfo.Exprs.end());
7478 DevicePtrDecls.append(in_start: CurInfo.DevicePtrDecls.begin(),
7479 in_end: CurInfo.DevicePtrDecls.end());
7480 Mappers.append(in_start: CurInfo.Mappers.begin(), in_end: CurInfo.Mappers.end());
7481 llvm::OpenMPIRBuilder::MapInfosTy::append(CurInfo);
7482 }
7483 };
7484
7485 /// Map between a struct and the its lowest & highest elements which have been
7486 /// mapped.
7487 /// [ValueDecl *] --> {LE(FieldIndex, Pointer),
7488 /// HE(FieldIndex, Pointer)}
7489 struct StructRangeInfoTy {
7490 MapCombinedInfoTy PreliminaryMapData;
7491 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = {
7492 0, Address::invalid()};
7493 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = {
7494 0, Address::invalid()};
7495 Address Base = Address::invalid();
7496 Address LB = Address::invalid();
7497 bool IsArraySection = false;
7498 bool HasCompleteRecord = false;
7499 };
7500
7501 /// A struct to store the attach pointer and pointee information, to be used
7502 /// when emitting an attach entry.
7503 struct AttachInfoTy {
7504 Address AttachPtrAddr = Address::invalid();
7505 Address AttachPteeAddr = Address::invalid();
7506 const ValueDecl *AttachPtrDecl = nullptr;
7507 const Expr *AttachMapExpr = nullptr;
7508
7509 bool isValid() const {
7510 return AttachPtrAddr.isValid() && AttachPteeAddr.isValid();
7511 }
7512 };
7513
7514 /// Check if there's any component list where the attach pointer expression
7515 /// matches the given captured variable.
7516 bool hasAttachEntryForCapturedVar(const ValueDecl *VD) const {
7517 for (const auto &AttachEntry : AttachPtrExprMap) {
7518 if (AttachEntry.second) {
7519 // Check if the attach pointer expression is a DeclRefExpr that
7520 // references the captured variable
7521 if (const auto *DRE = dyn_cast<DeclRefExpr>(Val: AttachEntry.second))
7522 if (DRE->getDecl() == VD)
7523 return true;
7524 }
7525 }
7526 return false;
7527 }
7528
7529 /// Get the previously-cached attach pointer for a component list, if-any.
7530 const Expr *getAttachPtrExpr(
7531 OMPClauseMappableExprCommon::MappableExprComponentListRef Components)
7532 const {
7533 const auto It = AttachPtrExprMap.find(Val: Components);
7534 if (It != AttachPtrExprMap.end())
7535 return It->second;
7536
7537 return nullptr;
7538 }
7539
7540private:
7541 /// Kind that defines how a device pointer has to be returned.
7542 struct MapInfo {
7543 OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
7544 OpenMPMapClauseKind MapType = OMPC_MAP_unknown;
7545 ArrayRef<OpenMPMapModifierKind> MapModifiers;
7546 ArrayRef<OpenMPMotionModifierKind> MotionModifiers;
7547 bool ReturnDevicePointer = false;
7548 bool IsImplicit = false;
7549 const ValueDecl *Mapper = nullptr;
7550 const Expr *VarRef = nullptr;
7551 bool ForDeviceAddr = false;
7552 bool HasUdpFbNullify = false;
7553
7554 MapInfo() = default;
7555 MapInfo(
7556 OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7557 OpenMPMapClauseKind MapType,
7558 ArrayRef<OpenMPMapModifierKind> MapModifiers,
7559 ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
7560 bool ReturnDevicePointer, bool IsImplicit,
7561 const ValueDecl *Mapper = nullptr, const Expr *VarRef = nullptr,
7562 bool ForDeviceAddr = false, bool HasUdpFbNullify = false)
7563 : Components(Components), MapType(MapType), MapModifiers(MapModifiers),
7564 MotionModifiers(MotionModifiers),
7565 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit),
7566 Mapper(Mapper), VarRef(VarRef), ForDeviceAddr(ForDeviceAddr),
7567 HasUdpFbNullify(HasUdpFbNullify) {}
7568 };
7569
7570 /// The target directive from where the mappable clauses were extracted. It
7571 /// is either a executable directive or a user-defined mapper directive.
7572 llvm::PointerUnion<const OMPExecutableDirective *,
7573 const OMPDeclareMapperDecl *>
7574 CurDir;
7575
7576 /// Function the directive is being generated for.
7577 CodeGenFunction &CGF;
7578
7579 /// Set of all first private variables in the current directive.
7580 /// bool data is set to true if the variable is implicitly marked as
7581 /// firstprivate, false otherwise.
7582 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls;
7583
7584 /// Set of defaultmap clause kinds that use firstprivate behavior.
7585 llvm::SmallSet<OpenMPDefaultmapClauseKind, 4> DefaultmapFirstprivateKinds;
7586
7587 /// Map between device pointer declarations and their expression components.
7588 /// The key value for declarations in 'this' is null.
7589 llvm::DenseMap<
7590 const ValueDecl *,
7591 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7592 DevPointersMap;
7593
7594 /// Map between device addr declarations and their expression components.
7595 /// The key value for declarations in 'this' is null.
7596 llvm::DenseMap<
7597 const ValueDecl *,
7598 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7599 HasDevAddrsMap;
7600
7601 /// Map between lambda declarations and their map type.
7602 llvm::DenseMap<const ValueDecl *, const OMPMapClause *> LambdasMap;
7603
7604 /// Map from component lists to their attach pointer expressions.
7605 llvm::DenseMap<OMPClauseMappableExprCommon::MappableExprComponentListRef,
7606 const Expr *>
7607 AttachPtrExprMap;
7608
7609 /// Map from attach pointer expressions to their component depth.
7610 /// nullptr key has std::nullopt depth. This can be used to order attach-ptr
7611 /// expressions with increasing/decreasing depth.
7612 /// The component-depth of `nullptr` (i.e. no attach-ptr) is `std::nullopt`.
7613 /// TODO: Not urgent, but we should ideally use the number of pointer
7614 /// dereferences in an expr as an indicator of its complexity, instead of the
7615 /// component-depth. That would be needed for us to treat `p[1]`, `*(p + 10)`,
7616 /// `*(p + 5 + 5)` together.
7617 llvm::DenseMap<const Expr *, std::optional<size_t>>
7618 AttachPtrComponentDepthMap = {{nullptr, std::nullopt}};
7619
7620 /// Map from attach pointer expressions to the order they were computed in, in
7621 /// collectAttachPtrExprInfo().
7622 llvm::DenseMap<const Expr *, size_t> AttachPtrComputationOrderMap = {
7623 {nullptr, 0}};
7624
7625 /// An instance of attach-ptr-expr comparator that can be used throughout the
7626 /// lifetime of this handler.
7627 AttachPtrExprComparator AttachPtrComparator;
7628
7629 llvm::Value *getExprTypeSize(const Expr *E) const {
7630 QualType ExprTy = E->getType().getCanonicalType();
7631
7632 // Calculate the size for array shaping expression.
7633 if (const auto *OAE = dyn_cast<OMPArrayShapingExpr>(Val: E)) {
7634 llvm::Value *Size =
7635 CGF.getTypeSize(Ty: OAE->getBase()->getType()->getPointeeType());
7636 for (const Expr *SE : OAE->getDimensions()) {
7637 llvm::Value *Sz = CGF.EmitScalarExpr(E: SE);
7638 Sz = CGF.EmitScalarConversion(Src: Sz, SrcTy: SE->getType(),
7639 DstTy: CGF.getContext().getSizeType(),
7640 Loc: SE->getExprLoc());
7641 Size = CGF.Builder.CreateNUWMul(LHS: Size, RHS: Sz);
7642 }
7643 return Size;
7644 }
7645
7646 // Reference types are ignored for mapping purposes.
7647 if (const auto *RefTy = ExprTy->getAs<ReferenceType>())
7648 ExprTy = RefTy->getPointeeType().getCanonicalType();
7649
7650 // Given that an array section is considered a built-in type, we need to
7651 // do the calculation based on the length of the section instead of relying
7652 // on CGF.getTypeSize(E->getType()).
7653 if (const auto *OAE = dyn_cast<ArraySectionExpr>(Val: E)) {
7654 QualType BaseTy = ArraySectionExpr::getBaseOriginalType(
7655 Base: OAE->getBase()->IgnoreParenImpCasts())
7656 .getCanonicalType();
7657
7658 // If there is no length associated with the expression and lower bound is
7659 // not specified too, that means we are using the whole length of the
7660 // base.
7661 if (!OAE->getLength() && OAE->getColonLocFirst().isValid() &&
7662 !OAE->getLowerBound())
7663 return CGF.getTypeSize(Ty: BaseTy);
7664
7665 llvm::Value *ElemSize;
7666 if (const auto *PTy = BaseTy->getAs<PointerType>()) {
7667 ElemSize = CGF.getTypeSize(Ty: PTy->getPointeeType().getCanonicalType());
7668 } else {
7669 const auto *ATy = cast<ArrayType>(Val: BaseTy.getTypePtr());
7670 assert(ATy && "Expecting array type if not a pointer type.");
7671 ElemSize = CGF.getTypeSize(Ty: ATy->getElementType().getCanonicalType());
7672 }
7673
7674 // If we don't have a length at this point, that is because we have an
7675 // array section with a single element.
7676 if (!OAE->getLength() && OAE->getColonLocFirst().isInvalid())
7677 return ElemSize;
7678
7679 if (const Expr *LenExpr = OAE->getLength()) {
7680 llvm::Value *LengthVal = CGF.EmitScalarExpr(E: LenExpr);
7681 LengthVal = CGF.EmitScalarConversion(Src: LengthVal, SrcTy: LenExpr->getType(),
7682 DstTy: CGF.getContext().getSizeType(),
7683 Loc: LenExpr->getExprLoc());
7684 return CGF.Builder.CreateNUWMul(LHS: LengthVal, RHS: ElemSize);
7685 }
7686 assert(!OAE->getLength() && OAE->getColonLocFirst().isValid() &&
7687 OAE->getLowerBound() && "expected array_section[lb:].");
7688 // Size = sizetype - lb * elemtype;
7689 llvm::Value *LengthVal = CGF.getTypeSize(Ty: BaseTy);
7690 llvm::Value *LBVal = CGF.EmitScalarExpr(E: OAE->getLowerBound());
7691 LBVal = CGF.EmitScalarConversion(Src: LBVal, SrcTy: OAE->getLowerBound()->getType(),
7692 DstTy: CGF.getContext().getSizeType(),
7693 Loc: OAE->getLowerBound()->getExprLoc());
7694 LBVal = CGF.Builder.CreateNUWMul(LHS: LBVal, RHS: ElemSize);
7695 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LHS: LengthVal, RHS: LBVal);
7696 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LHS: LengthVal, RHS: LBVal);
7697 LengthVal = CGF.Builder.CreateSelect(
7698 C: Cmp, True: TrueVal, False: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 0));
7699 return LengthVal;
7700 }
7701 return CGF.getTypeSize(Ty: ExprTy);
7702 }
7703
7704 /// Return the corresponding bits for a given map clause modifier. Add
7705 /// a flag marking the map as a pointer if requested. Add a flag marking the
7706 /// map as the first one of a series of maps that relate to the same map
7707 /// expression.
7708 OpenMPOffloadMappingFlags getMapTypeBits(
7709 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7710 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, bool IsImplicit,
7711 bool AddPtrFlag, bool AddIsTargetParamFlag, bool IsNonContiguous) const {
7712 OpenMPOffloadMappingFlags Bits =
7713 IsImplicit ? OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT
7714 : OpenMPOffloadMappingFlags::OMP_MAP_NONE;
7715 switch (MapType) {
7716 case OMPC_MAP_alloc:
7717 case OMPC_MAP_release:
7718 // alloc and release is the default behavior in the runtime library, i.e.
7719 // if we don't pass any bits alloc/release that is what the runtime is
7720 // going to do. Therefore, we don't need to signal anything for these two
7721 // type modifiers.
7722 break;
7723 case OMPC_MAP_to:
7724 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_TO;
7725 break;
7726 case OMPC_MAP_from:
7727 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_FROM;
7728 break;
7729 case OMPC_MAP_tofrom:
7730 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_TO |
7731 OpenMPOffloadMappingFlags::OMP_MAP_FROM;
7732 break;
7733 case OMPC_MAP_delete:
7734 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_DELETE;
7735 break;
7736 case OMPC_MAP_unknown:
7737 llvm_unreachable("Unexpected map type!");
7738 }
7739 if (AddPtrFlag)
7740 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ;
7741 if (AddIsTargetParamFlag)
7742 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM;
7743 if (llvm::is_contained(Range&: MapModifiers, Element: OMPC_MAP_MODIFIER_always))
7744 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_ALWAYS;
7745 if (llvm::is_contained(Range&: MapModifiers, Element: OMPC_MAP_MODIFIER_close))
7746 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_CLOSE;
7747 if (llvm::is_contained(Range&: MapModifiers, Element: OMPC_MAP_MODIFIER_present) ||
7748 llvm::is_contained(Range&: MotionModifiers, Element: OMPC_MOTION_MODIFIER_present))
7749 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_PRESENT;
7750 if (llvm::is_contained(Range&: MapModifiers, Element: OMPC_MAP_MODIFIER_ompx_hold))
7751 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD;
7752 if (IsNonContiguous)
7753 Bits |= OpenMPOffloadMappingFlags::OMP_MAP_NON_CONTIG;
7754 return Bits;
7755 }
7756
7757 /// Return true if the provided expression is a final array section. A
7758 /// final array section, is one whose length can't be proved to be one.
7759 bool isFinalArraySectionExpression(const Expr *E) const {
7760 const auto *OASE = dyn_cast<ArraySectionExpr>(Val: E);
7761
7762 // It is not an array section and therefore not a unity-size one.
7763 if (!OASE)
7764 return false;
7765
7766 // An array section with no colon always refer to a single element.
7767 if (OASE->getColonLocFirst().isInvalid())
7768 return false;
7769
7770 const Expr *Length = OASE->getLength();
7771
7772 // If we don't have a length we have to check if the array has size 1
7773 // for this dimension. Also, we should always expect a length if the
7774 // base type is pointer.
7775 if (!Length) {
7776 QualType BaseQTy = ArraySectionExpr::getBaseOriginalType(
7777 Base: OASE->getBase()->IgnoreParenImpCasts())
7778 .getCanonicalType();
7779 if (const auto *ATy = dyn_cast<ConstantArrayType>(Val: BaseQTy.getTypePtr()))
7780 return ATy->getSExtSize() != 1;
7781 // If we don't have a constant dimension length, we have to consider
7782 // the current section as having any size, so it is not necessarily
7783 // unitary. If it happen to be unity size, that's user fault.
7784 return true;
7785 }
7786
7787 // Check if the length evaluates to 1.
7788 Expr::EvalResult Result;
7789 if (!Length->EvaluateAsInt(Result, Ctx: CGF.getContext()))
7790 return true; // Can have more that size 1.
7791
7792 llvm::APSInt ConstLength = Result.Val.getInt();
7793 return ConstLength.getSExtValue() != 1;
7794 }
7795
7796 /// Emit an attach entry into \p CombinedInfo, using the information from \p
7797 /// AttachInfo. For example, for a map of form `int *p; ... map(p[1:10])`,
7798 /// an attach entry has the following form:
7799 /// &p, &p[1], sizeof(void*), ATTACH
7800 void emitAttachEntry(CodeGenFunction &CGF, MapCombinedInfoTy &CombinedInfo,
7801 const AttachInfoTy &AttachInfo) const {
7802 assert(AttachInfo.isValid() &&
7803 "Expected valid attach pointer/pointee information!");
7804
7805 // Size is the size of the pointer itself - use pointer size, not BaseDecl
7806 // size
7807 llvm::Value *PointerSize = CGF.Builder.CreateIntCast(
7808 V: llvm::ConstantInt::get(
7809 Ty: CGF.CGM.SizeTy, V: CGF.getContext()
7810 .getTypeSizeInChars(T: CGF.getContext().VoidPtrTy)
7811 .getQuantity()),
7812 DestTy: CGF.Int64Ty, /*isSigned=*/true);
7813
7814 CombinedInfo.Exprs.emplace_back(Args: AttachInfo.AttachPtrDecl,
7815 Args: AttachInfo.AttachMapExpr);
7816 CombinedInfo.BasePointers.push_back(
7817 Elt: AttachInfo.AttachPtrAddr.emitRawPointer(CGF));
7818 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
7819 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
7820 CombinedInfo.Pointers.push_back(
7821 Elt: AttachInfo.AttachPteeAddr.emitRawPointer(CGF));
7822 CombinedInfo.Sizes.push_back(Elt: PointerSize);
7823 CombinedInfo.Types.push_back(Elt: OpenMPOffloadMappingFlags::OMP_MAP_ATTACH);
7824 // ATTACH entries themselves don't "have" a base attach-ptr.
7825 CombinedInfo.HasAttachPtr.push_back(Elt: false);
7826 CombinedInfo.Mappers.push_back(Elt: nullptr);
7827 CombinedInfo.NonContigInfo.Dims.push_back(Elt: 1);
7828 }
7829
7830 /// A helper class to copy structures with overlapped elements, i.e. those
7831 /// which have mappings of both "s" and "s.mem". Consecutive elements that
7832 /// are not explicitly copied have mapping nodes synthesized for them,
7833 /// taking care to avoid generating zero-sized copies.
7834 class CopyOverlappedEntryGaps {
7835 CodeGenFunction &CGF;
7836 MapCombinedInfoTy &CombinedInfo;
7837 OpenMPOffloadMappingFlags Flags = OpenMPOffloadMappingFlags::OMP_MAP_NONE;
7838 const ValueDecl *MapDecl = nullptr;
7839 const Expr *MapExpr = nullptr;
7840 Address BP = Address::invalid();
7841 bool IsNonContiguous = false;
7842 uint64_t DimSize = 0;
7843 // These elements track the position as the struct is iterated over
7844 // (in order of increasing element address).
7845 const RecordDecl *LastParent = nullptr;
7846 uint64_t Cursor = 0;
7847 unsigned LastIndex = -1u;
7848 Address LB = Address::invalid();
7849
7850 public:
7851 CopyOverlappedEntryGaps(CodeGenFunction &CGF,
7852 MapCombinedInfoTy &CombinedInfo,
7853 OpenMPOffloadMappingFlags Flags,
7854 const ValueDecl *MapDecl, const Expr *MapExpr,
7855 Address BP, Address LB, bool IsNonContiguous,
7856 uint64_t DimSize)
7857 : CGF(CGF), CombinedInfo(CombinedInfo), Flags(Flags), MapDecl(MapDecl),
7858 MapExpr(MapExpr), BP(BP), IsNonContiguous(IsNonContiguous),
7859 DimSize(DimSize), LB(LB) {}
7860
7861 void processField(
7862 const OMPClauseMappableExprCommon::MappableComponent &MC,
7863 const FieldDecl *FD,
7864 llvm::function_ref<LValue(CodeGenFunction &, const MemberExpr *)>
7865 EmitMemberExprBase) {
7866 const RecordDecl *RD = FD->getParent();
7867 const ASTRecordLayout &RL = CGF.getContext().getASTRecordLayout(D: RD);
7868 uint64_t FieldOffset = RL.getFieldOffset(FieldNo: FD->getFieldIndex());
7869 uint64_t FieldSize =
7870 CGF.getContext().getTypeSize(T: FD->getType().getCanonicalType());
7871 Address ComponentLB = Address::invalid();
7872
7873 if (FD->getType()->isLValueReferenceType()) {
7874 const auto *ME = cast<MemberExpr>(Val: MC.getAssociatedExpression());
7875 LValue BaseLVal = EmitMemberExprBase(CGF, ME);
7876 ComponentLB =
7877 CGF.EmitLValueForFieldInitialization(Base: BaseLVal, Field: FD).getAddress();
7878 } else {
7879 ComponentLB =
7880 CGF.EmitOMPSharedLValue(E: MC.getAssociatedExpression()).getAddress();
7881 }
7882
7883 if (!LastParent)
7884 LastParent = RD;
7885 if (FD->getParent() == LastParent) {
7886 if (FD->getFieldIndex() != LastIndex + 1)
7887 copyUntilField(FD, ComponentLB);
7888 } else {
7889 LastParent = FD->getParent();
7890 if (((int64_t)FieldOffset - (int64_t)Cursor) > 0)
7891 copyUntilField(FD, ComponentLB);
7892 }
7893 Cursor = FieldOffset + FieldSize;
7894 LastIndex = FD->getFieldIndex();
7895 LB = CGF.Builder.CreateConstGEP(Addr: ComponentLB, Index: 1);
7896 }
7897
7898 void copyUntilField(const FieldDecl *FD, Address ComponentLB) {
7899 llvm::Value *ComponentLBPtr = ComponentLB.emitRawPointer(CGF);
7900 llvm::Value *LBPtr = LB.emitRawPointer(CGF);
7901 llvm::Value *Size = CGF.Builder.CreatePtrDiff(LHS: ComponentLBPtr, RHS: LBPtr);
7902 copySizedChunk(Base: LBPtr, Size);
7903 }
7904
7905 void copyUntilEnd(Address HB) {
7906 if (LastParent) {
7907 const ASTRecordLayout &RL =
7908 CGF.getContext().getASTRecordLayout(D: LastParent);
7909 if ((uint64_t)CGF.getContext().toBits(CharSize: RL.getSize()) <= Cursor)
7910 return;
7911 }
7912 llvm::Value *LBPtr = LB.emitRawPointer(CGF);
7913 llvm::Value *Size = CGF.Builder.CreatePtrDiff(
7914 LHS: CGF.Builder.CreateConstGEP(Addr: HB, Index: 1).emitRawPointer(CGF), RHS: LBPtr);
7915 copySizedChunk(Base: LBPtr, Size);
7916 }
7917
7918 void copySizedChunk(llvm::Value *Base, llvm::Value *Size) {
7919 CombinedInfo.Exprs.emplace_back(Args&: MapDecl, Args&: MapExpr);
7920 CombinedInfo.BasePointers.push_back(Elt: BP.emitRawPointer(CGF));
7921 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
7922 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
7923 CombinedInfo.Pointers.push_back(Elt: Base);
7924 CombinedInfo.Sizes.push_back(
7925 Elt: CGF.Builder.CreateIntCast(V: Size, DestTy: CGF.Int64Ty, /*isSigned=*/false));
7926 CombinedInfo.Types.push_back(Elt: Flags);
7927 CombinedInfo.HasAttachPtr.push_back(Elt: false);
7928 CombinedInfo.Mappers.push_back(Elt: nullptr);
7929 CombinedInfo.NonContigInfo.Dims.push_back(Elt: IsNonContiguous ? DimSize : 1);
7930 }
7931 };
7932
7933 /// Generate the base pointers, section pointers, sizes, map type bits, and
7934 /// user-defined mappers (all included in \a CombinedInfo) for the provided
7935 /// map type, map or motion modifiers, and expression components.
7936 /// \a IsFirstComponent should be set to true if the provided set of
7937 /// components is the first associated with a capture.
7938 void generateInfoForComponentList(
7939 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7940 ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
7941 OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7942 MapCombinedInfoTy &CombinedInfo,
7943 MapCombinedInfoTy &StructBaseCombinedInfo,
7944 StructRangeInfoTy &PartialStruct, AttachInfoTy &AttachInfo,
7945 bool IsFirstComponentList, bool IsImplicit,
7946 bool GenerateAllInfoForClauses, const ValueDecl *Mapper = nullptr,
7947 bool ForDeviceAddr = false, const ValueDecl *BaseDecl = nullptr,
7948 const Expr *MapExpr = nullptr,
7949 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
7950 OverlappedElements = {}) const {
7951
7952 // The following summarizes what has to be generated for each map and the
7953 // types below. The generated information is expressed in this order:
7954 // base pointer, section pointer, size, flags
7955 // (to add to the ones that come from the map type and modifier).
7956 // Entries annotated with (+) are only generated for "target" constructs,
7957 // and only if the variable at the beginning of the expression is used in
7958 // the region.
7959 //
7960 // double d;
7961 // int i[100];
7962 // float *p;
7963 // int **a = &i;
7964 //
7965 // struct S1 {
7966 // int i;
7967 // float f[50];
7968 // }
7969 // struct S2 {
7970 // int i;
7971 // float f[50];
7972 // S1 s;
7973 // double *p;
7974 // double *&pref;
7975 // struct S2 *ps;
7976 // int &ref;
7977 // }
7978 // S2 s;
7979 // S2 *ps;
7980 //
7981 // map(d)
7982 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM
7983 //
7984 // map(i)
7985 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM
7986 //
7987 // map(i[1:23])
7988 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM
7989 //
7990 // map(p)
7991 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
7992 //
7993 // map(p[1:24])
7994 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM // map pointee
7995 // &p, &p[1], sizeof(void*), ATTACH // attach pointer/pointee, if both
7996 // // are present, and either is new
7997 //
7998 // map(([22])p)
7999 // p, p, 22*sizeof(float), TARGET_PARAM | TO | FROM
8000 // &p, p, sizeof(void*), ATTACH
8001 //
8002 // map((*a)[0:3])
8003 // a, a, 0, TARGET_PARAM | IMPLICIT // (+)
8004 // (*a)[0], &(*a)[0], 3 * sizeof(int), TO | FROM
8005 // &(*a), &(*a)[0], sizeof(void*), ATTACH
8006 // (+) Only on target, if a is used in the region
8007 // Note: Since the attach base-pointer is `*a`, which is not a scalar
8008 // variable, it doesn't determine the clause on `a`. `a` is mapped using
8009 // a zero-length-array-section map by generateDefaultMapInfo, if it is
8010 // referenced in the target region, because it is a pointer.
8011 //
8012 // map(**a)
8013 // a, a, 0, TARGET_PARAM | IMPLICIT // (+)
8014 // &(*a)[0], &(*a)[0], sizeof(int), TO | FROM
8015 // &(*a), &(*a)[0], sizeof(void*), ATTACH
8016 // (+) Only on target, if a is used in the region
8017 //
8018 // map(s)
8019 // FIXME: This needs to also imply map(ref_ptr_ptee: s.ref), since the
8020 // effect is supposed to be same as if the user had a map for every element
8021 // of the struct. We currently do a shallow-map of s.
8022 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM
8023 //
8024 // map(s.i)
8025 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM
8026 //
8027 // map(s.s.f)
8028 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
8029 //
8030 // map(s.p)
8031 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM
8032 //
8033 // map(to: s.p[:22])
8034 // &s, &(s.p), sizeof(double*), TARGET_PARAM | IMPLICIT // (+)
8035 // &(s.p[0]), &(s.p[0]), 22 * sizeof(double*), TO | FROM
8036 // &(s.p), &(s.p[0]), sizeof(void*), ATTACH
8037 //
8038 // map(to: s.ref)
8039 // &s, &(ptr(s.ref)), sizeof(int*), TARGET_PARAM (*)
8040 // &s, &(ptee(s.ref)), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
8041 // (*) alloc space for struct members, only this is a target parameter.
8042 // (**) map the pointer (nothing to be mapped in this example) (the compiler
8043 // optimizes this entry out, same in the examples below)
8044 // (***) map the pointee (map: to)
8045 // Note: ptr(s.ref) represents the referring pointer of s.ref
8046 // ptee(s.ref) represents the referenced pointee of s.ref
8047 //
8048 // map(to: s.pref)
8049 // &s, &(ptr(s.pref)), sizeof(double**), TARGET_PARAM
8050 // &s, &(ptee(s.pref)), sizeof(double*), MEMBER_OF(1) | PTR_AND_OBJ | TO
8051 //
8052 // map(to: s.pref[:22])
8053 // &s, &(ptr(s.pref)), sizeof(double**), TARGET_PARAM | IMPLICIT // (+)
8054 // &s, &(ptee(s.pref)), sizeof(double*), MEMBER_OF(1) | PTR_AND_OBJ | TO |
8055 // FROM | IMPLICIT // (+)
8056 // &(ptee(s.pref)[0]), &(ptee(s.pref)[0]), 22 * sizeof(double), TO
8057 // &(ptee(s.pref)), &(ptee(s.pref)[0]), sizeof(void*), ATTACH
8058 //
8059 // map(s.ps)
8060 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM
8061 //
8062 // map(from: s.ps->s.i)
8063 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8064 // &(s.ps[0]), &(s.ps->s.i), sizeof(int), FROM
8065 // &(s.ps), &(s.ps->s.i), sizeof(void*), ATTACH
8066 //
8067 // map(to: s.ps->ps)
8068 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8069 // &(s.ps[0]), &(s.ps->ps), sizeof(S2*), TO
8070 // &(s.ps), &(s.ps->ps), sizeof(void*), ATTACH
8071 //
8072 // map(s.ps->ps->ps)
8073 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8074 // &(s.ps->ps[0]), &(s.ps->ps->ps), sizeof(S2*), TO
8075 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(void*), ATTACH
8076 //
8077 // map(to: s.ps->ps->s.f[:22])
8078 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM | IMPLICIT // (+)
8079 // &(s.ps->ps[0]), &(s.ps->ps->s.f[0]), 22*sizeof(float), TO
8080 // &(s.ps->ps), &(s.ps->ps->s.f[0]), sizeof(void*), ATTACH
8081 //
8082 // map(ps)
8083 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM
8084 //
8085 // map(ps->i)
8086 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM
8087 // &ps, &(ps->i), sizeof(void*), ATTACH
8088 //
8089 // map(ps->s.f)
8090 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
8091 // &ps, &(ps->s.f[0]), sizeof(ps), ATTACH
8092 //
8093 // map(from: ps->p)
8094 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM
8095 // &ps, &(ps->p), sizeof(ps), ATTACH
8096 //
8097 // map(to: ps->p[:22])
8098 // ps, &(ps[0]), 0, TARGET_PARAM | IMPLICIT // (+)
8099 // &(ps->p[0]), &(ps->p[0]), 22*sizeof(double), TO
8100 // &(ps->p), &(ps->p[0]), sizeof(void*), ATTACH
8101 //
8102 // map(ps->ps)
8103 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM
8104 // &ps, &(ps->ps), sizeof(ps), ATTACH
8105 //
8106 // map(from: ps->ps->s.i)
8107 // ps, &(ps[0]), 0, TARGET_PARAM | IMPLICIT // (+)
8108 // &(ps->ps[0]), &(ps->ps->s.i), sizeof(int), FROM
8109 // &(ps->ps), &(ps->ps->s.i), sizeof(void*), ATTACH
8110 //
8111 // map(from: ps->ps->ps)
8112 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8113 // &(ps->ps[0]), &(ps->ps->ps), sizeof(S2*), FROM
8114 // &(ps->ps), &(ps->ps->ps), sizeof(void*), ATTACH
8115 //
8116 // map(ps->ps->ps->ps)
8117 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8118 // &(ps->ps->ps[0]), &(ps->ps->ps->ps), sizeof(S2*), FROM
8119 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(void*), ATTACH
8120 //
8121 // map(to: ps->ps->ps->s.f[:22])
8122 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8123 // &(ps->ps->ps[0]), &(ps->ps->ps->s.f[0]), 22*sizeof(float), TO
8124 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), sizeof(void*), ATTACH
8125 //
8126 // map(to: s.f[:22]) map(from: s.p[:33])
8127 // On target, and if s is used in the region:
8128 //
8129 // &s, &(s.f[0]), 50*sizeof(float) +
8130 // sizeof(struct S1) +
8131 // sizeof(double*) (**), TARGET_PARAM
8132 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO
8133 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) | TO |
8134 // FROM | IMPLICIT
8135 // &(s.p[0]), &(s.p[0]), 33*sizeof(double), FROM
8136 // &(s.p), &(s.p[0]), sizeof(void*), ATTACH
8137 // (**) allocate contiguous space needed to fit all mapped members even if
8138 // we allocate space for members not mapped (in this example,
8139 // s.f[22..49] and s.s are not mapped, yet we must allocate space for
8140 // them as well because they fall between &s.f[0] and &s.p)
8141 //
8142 // On other constructs, and, if s is not used in the region, on target:
8143 // &s, &(s.f[0]), 22*sizeof(float), TO
8144 // &(s.p[0]), &(s.p[0]), 33*sizeof(double), FROM
8145 // &(s.p), &(s.p[0]), sizeof(void*), ATTACH
8146 //
8147 // map(from: s.f[:22]) map(to: ps->p[:33])
8148 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM
8149 // &ps[0], &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8150 // &(ps->p[0]), &(ps->p[0]), 33*sizeof(double), TO
8151 // &(ps->p), &(ps->p[0]), sizeof(void*), ATTACH
8152 //
8153 // map(from: s.f[:22], s.s) map(to: ps->p[:33])
8154 // &s, &(s.f[0]), 50*sizeof(float) +
8155 // sizeof(struct S1), TARGET_PARAM
8156 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM
8157 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM
8158 // ps, &ps[0], 0, TARGET_PARAM | IMPLICIT // (+)
8159 // &(ps->p[0]), &(ps->p[0]), 33*sizeof(double), TO
8160 // &(ps->p), &(ps->p[0]), sizeof(void*), ATTACH
8161 //
8162 // map(p[:100], p)
8163 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
8164 // p, &p[0], 100*sizeof(float), TO | FROM
8165 // &p, &p[0], sizeof(float*), ATTACH
8166
8167 // Track if the map information being generated is the first for a capture.
8168 bool IsCaptureFirstInfo = IsFirstComponentList;
8169 // When the variable is on a declare target link or in a to clause with
8170 // unified memory, a reference is needed to hold the host/device address
8171 // of the variable.
8172 bool RequiresReference = false;
8173
8174 // Scan the components from the base to the complete expression.
8175 auto CI = Components.rbegin();
8176 auto CE = Components.rend();
8177 auto I = CI;
8178
8179 // Track if the map information being generated is the first for a list of
8180 // components.
8181 bool IsExpressionFirstInfo = true;
8182 bool FirstPointerInComplexData = false;
8183 Address BP = Address::invalid();
8184 Address FinalLowestElem = Address::invalid();
8185 const Expr *AssocExpr = I->getAssociatedExpression();
8186 const auto *AE = dyn_cast<ArraySubscriptExpr>(Val: AssocExpr);
8187 const auto *OASE = dyn_cast<ArraySectionExpr>(Val: AssocExpr);
8188 const auto *OAShE = dyn_cast<OMPArrayShapingExpr>(Val: AssocExpr);
8189
8190 // Get the pointer-attachment base-pointer for the given list, if any.
8191 const Expr *AttachPtrExpr = getAttachPtrExpr(Components);
8192 auto [AttachPtrAddr, AttachPteeBaseAddr] =
8193 getAttachPtrAddrAndPteeBaseAddr(AttachPtrExpr, CGF);
8194
8195 bool HasAttachPtr = AttachPtrExpr != nullptr;
8196 bool FirstComponentIsForAttachPtr = AssocExpr == AttachPtrExpr;
8197 bool SeenAttachPtr = FirstComponentIsForAttachPtr;
8198
8199 if (FirstComponentIsForAttachPtr) {
8200 // No need to process AttachPtr here. It will be processed at the end
8201 // after we have computed the pointee's address.
8202 ++I;
8203 } else if (isa<MemberExpr>(Val: AssocExpr)) {
8204 // The base is the 'this' pointer. The content of the pointer is going
8205 // to be the base of the field being mapped.
8206 BP = CGF.LoadCXXThisAddress();
8207 } else if ((AE && isa<CXXThisExpr>(Val: AE->getBase()->IgnoreParenImpCasts())) ||
8208 (OASE &&
8209 isa<CXXThisExpr>(Val: OASE->getBase()->IgnoreParenImpCasts()))) {
8210 BP = CGF.EmitOMPSharedLValue(E: AssocExpr).getAddress();
8211 } else if (OAShE &&
8212 isa<CXXThisExpr>(Val: OAShE->getBase()->IgnoreParenCasts())) {
8213 BP = Address(
8214 CGF.EmitScalarExpr(E: OAShE->getBase()),
8215 CGF.ConvertTypeForMem(T: OAShE->getBase()->getType()->getPointeeType()),
8216 CGF.getContext().getTypeAlignInChars(T: OAShE->getBase()->getType()));
8217 } else {
8218 // The base is the reference to the variable.
8219 // BP = &Var.
8220 BP = CGF.EmitOMPSharedLValue(E: AssocExpr).getAddress();
8221 if (const auto *VD =
8222 dyn_cast_or_null<VarDecl>(Val: I->getAssociatedDeclaration())) {
8223 if (std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
8224 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) {
8225 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
8226 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
8227 *Res == OMPDeclareTargetDeclAttr::MT_Enter) &&
8228 CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) {
8229 RequiresReference = true;
8230 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
8231 }
8232 }
8233 }
8234
8235 // If the variable is a pointer and is being dereferenced (i.e. is not
8236 // the last component), the base has to be the pointer itself, not its
8237 // reference. References are ignored for mapping purposes.
8238 QualType Ty =
8239 I->getAssociatedDeclaration()->getType().getNonReferenceType();
8240 if (Ty->isAnyPointerType() && std::next(x: I) != CE) {
8241 // No need to generate individual map information for the pointer, it
8242 // can be associated with the combined storage if shared memory mode is
8243 // active or the base declaration is not global variable.
8244 const auto *VD = dyn_cast<VarDecl>(Val: I->getAssociatedDeclaration());
8245 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
8246 !VD || VD->hasLocalStorage() || HasAttachPtr)
8247 BP = CGF.EmitLoadOfPointer(Ptr: BP, PtrTy: Ty->castAs<PointerType>());
8248 else
8249 FirstPointerInComplexData = true;
8250 ++I;
8251 }
8252 }
8253
8254 // Track whether a component of the list should be marked as MEMBER_OF some
8255 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry
8256 // in a component list should be marked as MEMBER_OF, all subsequent entries
8257 // do not belong to the base struct. E.g.
8258 // struct S2 s;
8259 // s.ps->ps->ps->f[:]
8260 // (1) (2) (3) (4)
8261 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a
8262 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3)
8263 // is the pointee of ps(2) which is not member of struct s, so it should not
8264 // be marked as such (it is still PTR_AND_OBJ).
8265 // The variable is initialized to false so that PTR_AND_OBJ entries which
8266 // are not struct members are not considered (e.g. array of pointers to
8267 // data).
8268 bool ShouldBeMemberOf = false;
8269
8270 // Variable keeping track of whether or not we have encountered a component
8271 // in the component list which is a member expression. Useful when we have a
8272 // pointer or a final array section, in which case it is the previous
8273 // component in the list which tells us whether we have a member expression.
8274 // E.g. X.f[:]
8275 // While processing the final array section "[:]" it is "f" which tells us
8276 // whether we are dealing with a member of a declared struct.
8277 const MemberExpr *EncounteredME = nullptr;
8278
8279 // Track for the total number of dimension. Start from one for the dummy
8280 // dimension.
8281 uint64_t DimSize = 1;
8282
8283 // Detects non-contiguous updates due to strided accesses.
8284 // Sets the 'IsNonContiguous' flag so that the 'MapType' bits are set
8285 // correctly when generating information to be passed to the runtime. The
8286 // flag is set to true if any array section has a stride not equal to 1, or
8287 // if the stride is not a constant expression (conservatively assumed
8288 // non-contiguous).
8289 bool IsNonContiguous =
8290 CombinedInfo.NonContigInfo.IsNonContiguous ||
8291 any_of(Range&: Components, P: [&](const auto &Component) {
8292 const auto *OASE =
8293 dyn_cast<ArraySectionExpr>(Component.getAssociatedExpression());
8294 if (!OASE)
8295 return false;
8296
8297 const Expr *StrideExpr = OASE->getStride();
8298 if (!StrideExpr)
8299 return false;
8300
8301 assert(StrideExpr->getType()->isIntegerType() &&
8302 "Stride expression must be of integer type");
8303
8304 // If stride is not evaluatable as a constant, treat as
8305 // non-contiguous.
8306 const auto Constant =
8307 StrideExpr->getIntegerConstantExpr(Ctx: CGF.getContext());
8308 if (!Constant)
8309 return true;
8310
8311 // Treat non-unitary strides as non-contiguous.
8312 return !Constant->isOne();
8313 });
8314
8315 bool IsPrevMemberReference = false;
8316
8317 bool IsPartialMapped =
8318 !PartialStruct.PreliminaryMapData.BasePointers.empty();
8319
8320 // We need to check if we will be encountering any MEs. If we do not
8321 // encounter any ME expression it means we will be mapping the whole struct.
8322 // In that case we need to skip adding an entry for the struct to the
8323 // CombinedInfo list and instead add an entry to the StructBaseCombinedInfo
8324 // list only when generating all info for clauses.
8325 bool IsMappingWholeStruct = true;
8326 if (!GenerateAllInfoForClauses) {
8327 IsMappingWholeStruct = false;
8328 } else {
8329 for (auto TempI = I; TempI != CE; ++TempI) {
8330 const MemberExpr *PossibleME =
8331 dyn_cast<MemberExpr>(Val: TempI->getAssociatedExpression());
8332 if (PossibleME) {
8333 IsMappingWholeStruct = false;
8334 break;
8335 }
8336 }
8337 }
8338
8339 bool SeenFirstNonBinOpExprAfterAttachPtr = false;
8340 for (; I != CE; ++I) {
8341 // If we have a valid attach-ptr, we skip processing all components until
8342 // after the attach-ptr.
8343 if (HasAttachPtr && !SeenAttachPtr) {
8344 SeenAttachPtr = I->getAssociatedExpression() == AttachPtrExpr;
8345 continue;
8346 }
8347
8348 // After finding the attach pointer, skip binary-ops, to skip past
8349 // expressions like (p + 10), for a map like map(*(p + 10)), where p is
8350 // the attach-ptr.
8351 if (HasAttachPtr && !SeenFirstNonBinOpExprAfterAttachPtr) {
8352 const auto *BO = dyn_cast<BinaryOperator>(Val: I->getAssociatedExpression());
8353 if (BO)
8354 continue;
8355
8356 // Found the first non-binary-operator component after attach
8357 SeenFirstNonBinOpExprAfterAttachPtr = true;
8358 BP = AttachPteeBaseAddr;
8359 }
8360
8361 // If the current component is member of a struct (parent struct) mark it.
8362 if (!EncounteredME) {
8363 EncounteredME = dyn_cast<MemberExpr>(Val: I->getAssociatedExpression());
8364 // If we encounter a PTR_AND_OBJ entry from now on it should be marked
8365 // as MEMBER_OF the parent struct.
8366 if (EncounteredME) {
8367 ShouldBeMemberOf = true;
8368 // Do not emit as complex pointer if this is actually not array-like
8369 // expression.
8370 if (FirstPointerInComplexData) {
8371 QualType Ty = std::prev(x: I)
8372 ->getAssociatedDeclaration()
8373 ->getType()
8374 .getNonReferenceType();
8375 BP = CGF.EmitLoadOfPointer(Ptr: BP, PtrTy: Ty->castAs<PointerType>());
8376 FirstPointerInComplexData = false;
8377 }
8378 }
8379 }
8380
8381 auto Next = std::next(x: I);
8382
8383 // We need to generate the addresses and sizes if this is the last
8384 // component, if the component is a pointer or if it is an array section
8385 // whose length can't be proved to be one. If this is a pointer, it
8386 // becomes the base address for the following components.
8387
8388 // A final array section, is one whose length can't be proved to be one.
8389 // If the map item is non-contiguous then we don't treat any array section
8390 // as final array section.
8391 bool IsFinalArraySection =
8392 !IsNonContiguous &&
8393 isFinalArraySectionExpression(E: I->getAssociatedExpression());
8394
8395 // If we have a declaration for the mapping use that, otherwise use
8396 // the base declaration of the map clause.
8397 const ValueDecl *MapDecl = (I->getAssociatedDeclaration())
8398 ? I->getAssociatedDeclaration()
8399 : BaseDecl;
8400 MapExpr = (I->getAssociatedExpression()) ? I->getAssociatedExpression()
8401 : MapExpr;
8402
8403 // Get information on whether the element is a pointer. Have to do a
8404 // special treatment for array sections given that they are built-in
8405 // types.
8406 const auto *OASE =
8407 dyn_cast<ArraySectionExpr>(Val: I->getAssociatedExpression());
8408 const auto *OAShE =
8409 dyn_cast<OMPArrayShapingExpr>(Val: I->getAssociatedExpression());
8410 const auto *UO = dyn_cast<UnaryOperator>(Val: I->getAssociatedExpression());
8411 const auto *BO = dyn_cast<BinaryOperator>(Val: I->getAssociatedExpression());
8412 bool IsPointer =
8413 OAShE ||
8414 (OASE && ArraySectionExpr::getBaseOriginalType(Base: OASE)
8415 .getCanonicalType()
8416 ->isAnyPointerType()) ||
8417 I->getAssociatedExpression()->getType()->isAnyPointerType();
8418 bool IsMemberReference = isa<MemberExpr>(Val: I->getAssociatedExpression()) &&
8419 MapDecl &&
8420 MapDecl->getType()->isLValueReferenceType();
8421 bool IsNonDerefPointer = IsPointer &&
8422 !(UO && UO->getOpcode() != UO_Deref) && !BO &&
8423 !IsNonContiguous;
8424
8425 if (OASE)
8426 ++DimSize;
8427
8428 if (Next == CE || IsMemberReference || IsNonDerefPointer ||
8429 IsFinalArraySection) {
8430 // If this is not the last component, we expect the pointer to be
8431 // associated with an array expression or member expression.
8432 assert((Next == CE ||
8433 isa<MemberExpr>(Next->getAssociatedExpression()) ||
8434 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) ||
8435 isa<ArraySectionExpr>(Next->getAssociatedExpression()) ||
8436 isa<OMPArrayShapingExpr>(Next->getAssociatedExpression()) ||
8437 isa<UnaryOperator>(Next->getAssociatedExpression()) ||
8438 isa<BinaryOperator>(Next->getAssociatedExpression())) &&
8439 "Unexpected expression");
8440
8441 Address LB = Address::invalid();
8442 Address LowestElem = Address::invalid();
8443 auto &&EmitMemberExprBase = [](CodeGenFunction &CGF,
8444 const MemberExpr *E) {
8445 const Expr *BaseExpr = E->getBase();
8446 // If this is s.x, emit s as an lvalue. If it is s->x, emit s as a
8447 // scalar.
8448 LValue BaseLV;
8449 if (E->isArrow()) {
8450 LValueBaseInfo BaseInfo;
8451 TBAAAccessInfo TBAAInfo;
8452 Address Addr =
8453 CGF.EmitPointerWithAlignment(Addr: BaseExpr, BaseInfo: &BaseInfo, TBAAInfo: &TBAAInfo);
8454 QualType PtrTy = BaseExpr->getType()->getPointeeType();
8455 BaseLV = CGF.MakeAddrLValue(Addr, T: PtrTy, BaseInfo, TBAAInfo);
8456 } else {
8457 BaseLV = CGF.EmitOMPSharedLValue(E: BaseExpr);
8458 }
8459 return BaseLV;
8460 };
8461 if (OAShE) {
8462 LowestElem = LB =
8463 Address(CGF.EmitScalarExpr(E: OAShE->getBase()),
8464 CGF.ConvertTypeForMem(
8465 T: OAShE->getBase()->getType()->getPointeeType()),
8466 CGF.getContext().getTypeAlignInChars(
8467 T: OAShE->getBase()->getType()));
8468 } else if (IsMemberReference) {
8469 const auto *ME = cast<MemberExpr>(Val: I->getAssociatedExpression());
8470 LValue BaseLVal = EmitMemberExprBase(CGF, ME);
8471 LowestElem = CGF.EmitLValueForFieldInitialization(
8472 Base: BaseLVal, Field: cast<FieldDecl>(Val: MapDecl))
8473 .getAddress();
8474 LB = CGF.EmitLoadOfReferenceLValue(RefAddr: LowestElem, RefTy: MapDecl->getType())
8475 .getAddress();
8476 } else {
8477 LowestElem = LB =
8478 CGF.EmitOMPSharedLValue(E: I->getAssociatedExpression())
8479 .getAddress();
8480 }
8481
8482 // Save the final LowestElem, to use it as the pointee in attach maps,
8483 // if emitted.
8484 if (Next == CE)
8485 FinalLowestElem = LowestElem;
8486
8487 // If this component is a pointer inside the base struct then we don't
8488 // need to create any entry for it - it will be combined with the object
8489 // it is pointing to into a single PTR_AND_OBJ entry.
8490 bool IsMemberPointerOrAddr =
8491 EncounteredME &&
8492 (((IsPointer || ForDeviceAddr) &&
8493 I->getAssociatedExpression() == EncounteredME) ||
8494 (IsPrevMemberReference && !IsPointer) ||
8495 (IsMemberReference && Next != CE &&
8496 !Next->getAssociatedExpression()->getType()->isPointerType()));
8497 if (!OverlappedElements.empty() && Next == CE) {
8498 // Handle base element with the info for overlapped elements.
8499 assert(!PartialStruct.Base.isValid() && "The base element is set.");
8500 assert(!IsPointer &&
8501 "Unexpected base element with the pointer type.");
8502 // Mark the whole struct as the struct that requires allocation on the
8503 // device.
8504 PartialStruct.LowestElem = {0, LowestElem};
8505 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars(
8506 T: I->getAssociatedExpression()->getType());
8507 Address HB = CGF.Builder.CreateConstGEP(
8508 Addr: CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8509 Addr: LowestElem, Ty: CGF.VoidPtrTy, ElementTy: CGF.Int8Ty),
8510 Index: TypeSize.getQuantity() - 1);
8511 PartialStruct.HighestElem = {
8512 std::numeric_limits<decltype(
8513 PartialStruct.HighestElem.first)>::max(),
8514 HB};
8515 PartialStruct.Base = BP;
8516 PartialStruct.LB = LB;
8517 assert(
8518 PartialStruct.PreliminaryMapData.BasePointers.empty() &&
8519 "Overlapped elements must be used only once for the variable.");
8520 std::swap(a&: PartialStruct.PreliminaryMapData, b&: CombinedInfo);
8521 // Emit data for non-overlapped data.
8522 OpenMPOffloadMappingFlags Flags =
8523 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
8524 getMapTypeBits(MapType, MapModifiers, MotionModifiers, IsImplicit,
8525 /*AddPtrFlag=*/false,
8526 /*AddIsTargetParamFlag=*/false, IsNonContiguous);
8527 CopyOverlappedEntryGaps CopyGaps(CGF, CombinedInfo, Flags, MapDecl,
8528 MapExpr, BP, LB, IsNonContiguous,
8529 DimSize);
8530 // Do bitcopy of all non-overlapped structure elements.
8531 for (OMPClauseMappableExprCommon::MappableExprComponentListRef
8532 Component : OverlappedElements) {
8533 for (const OMPClauseMappableExprCommon::MappableComponent &MC :
8534 Component) {
8535 if (const ValueDecl *VD = MC.getAssociatedDeclaration()) {
8536 if (const auto *FD = dyn_cast<FieldDecl>(Val: VD)) {
8537 CopyGaps.processField(MC, FD, EmitMemberExprBase);
8538 }
8539 }
8540 }
8541 }
8542 CopyGaps.copyUntilEnd(HB);
8543 break;
8544 }
8545 llvm::Value *Size = getExprTypeSize(E: I->getAssociatedExpression());
8546 // Skip adding an entry in the CurInfo of this combined entry if the
8547 // whole struct is currently being mapped. The struct needs to be added
8548 // in the first position before any data internal to the struct is being
8549 // mapped.
8550 // Skip adding an entry in the CurInfo of this combined entry if the
8551 // PartialStruct.PreliminaryMapData.BasePointers has been mapped.
8552 if ((!IsMemberPointerOrAddr && !IsPartialMapped) ||
8553 (Next == CE && MapType != OMPC_MAP_unknown)) {
8554 if (!IsMappingWholeStruct) {
8555 CombinedInfo.Exprs.emplace_back(Args&: MapDecl, Args&: MapExpr);
8556 CombinedInfo.BasePointers.push_back(Elt: BP.emitRawPointer(CGF));
8557 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
8558 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
8559 CombinedInfo.Pointers.push_back(Elt: LB.emitRawPointer(CGF));
8560 CombinedInfo.Sizes.push_back(Elt: CGF.Builder.CreateIntCast(
8561 V: Size, DestTy: CGF.Int64Ty, /*isSigned=*/true));
8562 CombinedInfo.NonContigInfo.Dims.push_back(Elt: IsNonContiguous ? DimSize
8563 : 1);
8564 } else {
8565 StructBaseCombinedInfo.Exprs.emplace_back(Args&: MapDecl, Args&: MapExpr);
8566 StructBaseCombinedInfo.BasePointers.push_back(
8567 Elt: BP.emitRawPointer(CGF));
8568 StructBaseCombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
8569 StructBaseCombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
8570 StructBaseCombinedInfo.Pointers.push_back(Elt: LB.emitRawPointer(CGF));
8571 StructBaseCombinedInfo.Sizes.push_back(Elt: CGF.Builder.CreateIntCast(
8572 V: Size, DestTy: CGF.Int64Ty, /*isSigned=*/true));
8573 StructBaseCombinedInfo.NonContigInfo.Dims.push_back(
8574 Elt: IsNonContiguous ? DimSize : 1);
8575 }
8576
8577 // If Mapper is valid, the last component inherits the mapper.
8578 bool HasMapper = Mapper && Next == CE;
8579 if (!IsMappingWholeStruct)
8580 CombinedInfo.Mappers.push_back(Elt: HasMapper ? Mapper : nullptr);
8581 else
8582 StructBaseCombinedInfo.Mappers.push_back(Elt: HasMapper ? Mapper
8583 : nullptr);
8584
8585 // We need to add a pointer flag for each map that comes from the
8586 // same expression except for the first one. We also need to signal
8587 // this map is the first one that relates with the current capture
8588 // (there is a set of entries for each capture).
8589 OpenMPOffloadMappingFlags Flags = getMapTypeBits(
8590 MapType, MapModifiers, MotionModifiers, IsImplicit,
8591 AddPtrFlag: !IsExpressionFirstInfo || RequiresReference ||
8592 FirstPointerInComplexData || IsMemberReference,
8593 AddIsTargetParamFlag: IsCaptureFirstInfo && !RequiresReference, IsNonContiguous);
8594
8595 if (!IsExpressionFirstInfo || IsMemberReference) {
8596 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well,
8597 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags.
8598 if (IsPointer || (IsMemberReference && Next != CE))
8599 Flags &= ~(OpenMPOffloadMappingFlags::OMP_MAP_TO |
8600 OpenMPOffloadMappingFlags::OMP_MAP_FROM |
8601 OpenMPOffloadMappingFlags::OMP_MAP_ALWAYS |
8602 OpenMPOffloadMappingFlags::OMP_MAP_DELETE |
8603 OpenMPOffloadMappingFlags::OMP_MAP_CLOSE);
8604
8605 if (ShouldBeMemberOf) {
8606 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag
8607 // should be later updated with the correct value of MEMBER_OF.
8608 Flags |= OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF;
8609 // From now on, all subsequent PTR_AND_OBJ entries should not be
8610 // marked as MEMBER_OF.
8611 ShouldBeMemberOf = false;
8612 }
8613 }
8614
8615 if (!IsMappingWholeStruct) {
8616 CombinedInfo.Types.push_back(Elt: Flags);
8617 // HasAttachPtr marks pointee entries, which have a base attach-ptr.
8618 CombinedInfo.HasAttachPtr.push_back(Elt: HasAttachPtr);
8619 } else {
8620 StructBaseCombinedInfo.Types.push_back(Elt: Flags);
8621 StructBaseCombinedInfo.HasAttachPtr.push_back(Elt: HasAttachPtr);
8622 }
8623 }
8624
8625 // If we have encountered a member expression so far, keep track of the
8626 // mapped member. If the parent is "*this", then the value declaration
8627 // is nullptr.
8628 if (EncounteredME) {
8629 const auto *FD = cast<FieldDecl>(Val: EncounteredME->getMemberDecl());
8630 unsigned FieldIndex = FD->getFieldIndex();
8631
8632 // Update info about the lowest and highest elements for this struct
8633 if (!PartialStruct.Base.isValid()) {
8634 PartialStruct.LowestElem = {FieldIndex, LowestElem};
8635 if (IsFinalArraySection && OASE) {
8636 Address HB =
8637 CGF.EmitArraySectionExpr(E: OASE, /*IsLowerBound=*/false)
8638 .getAddress();
8639 PartialStruct.HighestElem = {FieldIndex, HB};
8640 } else {
8641 PartialStruct.HighestElem = {FieldIndex, LowestElem};
8642 }
8643 PartialStruct.Base = BP;
8644 PartialStruct.LB = BP;
8645 } else if (FieldIndex < PartialStruct.LowestElem.first) {
8646 PartialStruct.LowestElem = {FieldIndex, LowestElem};
8647 } else if (FieldIndex > PartialStruct.HighestElem.first) {
8648 if (IsFinalArraySection && OASE) {
8649 Address HB =
8650 CGF.EmitArraySectionExpr(E: OASE, /*IsLowerBound=*/false)
8651 .getAddress();
8652 PartialStruct.HighestElem = {FieldIndex, HB};
8653 } else {
8654 PartialStruct.HighestElem = {FieldIndex, LowestElem};
8655 }
8656 }
8657 }
8658
8659 // Need to emit combined struct for array sections.
8660 if (IsFinalArraySection || IsNonContiguous)
8661 PartialStruct.IsArraySection = true;
8662
8663 // If we have a final array section, we are done with this expression.
8664 if (IsFinalArraySection)
8665 break;
8666
8667 // The pointer becomes the base for the next element.
8668 if (Next != CE)
8669 BP = IsMemberReference ? LowestElem : LB;
8670 if (!IsPartialMapped)
8671 IsExpressionFirstInfo = false;
8672 IsCaptureFirstInfo = false;
8673 FirstPointerInComplexData = false;
8674 IsPrevMemberReference = IsMemberReference;
8675 } else if (FirstPointerInComplexData) {
8676 QualType Ty = Components.rbegin()
8677 ->getAssociatedDeclaration()
8678 ->getType()
8679 .getNonReferenceType();
8680 BP = CGF.EmitLoadOfPointer(Ptr: BP, PtrTy: Ty->castAs<PointerType>());
8681 FirstPointerInComplexData = false;
8682 }
8683 }
8684 // If ran into the whole component - allocate the space for the whole
8685 // record.
8686 if (!EncounteredME)
8687 PartialStruct.HasCompleteRecord = true;
8688
8689 // Populate ATTACH information for later processing by emitAttachEntry.
8690 if (shouldEmitAttachEntry(PointerExpr: AttachPtrExpr, MapBaseDecl: BaseDecl, CGF, CurDir)) {
8691 AttachInfo.AttachPtrAddr = AttachPtrAddr;
8692 AttachInfo.AttachPteeAddr = FinalLowestElem;
8693 AttachInfo.AttachPtrDecl = BaseDecl;
8694 AttachInfo.AttachMapExpr = MapExpr;
8695 }
8696
8697 if (!IsNonContiguous)
8698 return;
8699
8700 const ASTContext &Context = CGF.getContext();
8701
8702 // For supporting stride in array section, we need to initialize the first
8703 // dimension size as 1, first offset as 0, and first count as 1
8704 MapValuesArrayTy CurOffsets = {llvm::ConstantInt::get(Ty: CGF.CGM.Int64Ty, V: 0)};
8705 MapValuesArrayTy CurCounts;
8706 MapValuesArrayTy CurStrides = {llvm::ConstantInt::get(Ty: CGF.CGM.Int64Ty, V: 1)};
8707 MapValuesArrayTy DimSizes{llvm::ConstantInt::get(Ty: CGF.CGM.Int64Ty, V: 1)};
8708 uint64_t ElementTypeSize;
8709
8710 // Collect Size information for each dimension and get the element size as
8711 // the first Stride. For example, for `int arr[10][10]`, the DimSizes
8712 // should be [10, 10] and the first stride is 4 btyes.
8713 for (const OMPClauseMappableExprCommon::MappableComponent &Component :
8714 Components) {
8715 const Expr *AssocExpr = Component.getAssociatedExpression();
8716 const auto *OASE = dyn_cast<ArraySectionExpr>(Val: AssocExpr);
8717
8718 if (!OASE)
8719 continue;
8720
8721 QualType Ty = ArraySectionExpr::getBaseOriginalType(Base: OASE->getBase());
8722 auto *CAT = Context.getAsConstantArrayType(T: Ty);
8723 auto *VAT = Context.getAsVariableArrayType(T: Ty);
8724
8725 // We need all the dimension size except for the last dimension.
8726 assert((VAT || CAT || &Component == &*Components.begin()) &&
8727 "Should be either ConstantArray or VariableArray if not the "
8728 "first Component");
8729
8730 // Get element size if CurCounts is empty.
8731 if (CurCounts.empty()) {
8732 const Type *ElementType = nullptr;
8733 if (CAT)
8734 ElementType = CAT->getElementType().getTypePtr();
8735 else if (VAT)
8736 ElementType = VAT->getElementType().getTypePtr();
8737 else if (&Component == &*Components.begin()) {
8738 // If the base is a raw pointer (e.g. T *data with data[a:b:c]),
8739 // there was no earlier CAT/VAT/array handling to establish
8740 // ElementType. Capture the pointee type now so that subsequent
8741 // components (offset/length/stride) have a concrete element type to
8742 // work with. This makes pointer-backed sections behave consistently
8743 // with CAT/VAT/array bases.
8744 if (const auto *PtrType = Ty->getAs<PointerType>())
8745 ElementType = PtrType->getPointeeType().getTypePtr();
8746 } else {
8747 // Any component after the first should never have a raw pointer type;
8748 // by this point. ElementType must already be known (set above or in
8749 // prior array / CAT / VAT handling).
8750 assert(!Ty->isPointerType() &&
8751 "Non-first components should not be raw pointers");
8752 }
8753
8754 // At this stage, if ElementType was a base pointer and we are in the
8755 // first iteration, it has been computed.
8756 if (ElementType) {
8757 // For the case that having pointer as base, we need to remove one
8758 // level of indirection.
8759 if (&Component != &*Components.begin())
8760 ElementType = ElementType->getPointeeOrArrayElementType();
8761 ElementTypeSize =
8762 Context.getTypeSizeInChars(T: ElementType).getQuantity();
8763 CurCounts.push_back(
8764 Elt: llvm::ConstantInt::get(Ty: CGF.Int64Ty, V: ElementTypeSize));
8765 }
8766 }
8767 // Get dimension value except for the last dimension since we don't need
8768 // it.
8769 if (DimSizes.size() < Components.size() - 1) {
8770 if (CAT)
8771 DimSizes.push_back(
8772 Elt: llvm::ConstantInt::get(Ty: CGF.Int64Ty, V: CAT->getZExtSize()));
8773 else if (VAT)
8774 DimSizes.push_back(Elt: CGF.Builder.CreateIntCast(
8775 V: CGF.EmitScalarExpr(E: VAT->getSizeExpr()), DestTy: CGF.Int64Ty,
8776 /*IsSigned=*/isSigned: false));
8777 }
8778 }
8779
8780 // Skip the dummy dimension since we have already have its information.
8781 auto *DI = DimSizes.begin() + 1;
8782 // Product of dimension.
8783 llvm::Value *DimProd =
8784 llvm::ConstantInt::get(Ty: CGF.CGM.Int64Ty, V: ElementTypeSize);
8785
8786 // Collect info for non-contiguous. Notice that offset, count, and stride
8787 // are only meaningful for array-section, so we insert a null for anything
8788 // other than array-section.
8789 // Also, the size of offset, count, and stride are not the same as
8790 // pointers, base_pointers, sizes, or dims. Instead, the size of offset,
8791 // count, and stride are the same as the number of non-contiguous
8792 // declaration in target update to/from clause.
8793 for (const OMPClauseMappableExprCommon::MappableComponent &Component :
8794 Components) {
8795 const Expr *AssocExpr = Component.getAssociatedExpression();
8796
8797 if (const auto *AE = dyn_cast<ArraySubscriptExpr>(Val: AssocExpr)) {
8798 llvm::Value *Offset = CGF.Builder.CreateIntCast(
8799 V: CGF.EmitScalarExpr(E: AE->getIdx()), DestTy: CGF.Int64Ty,
8800 /*isSigned=*/false);
8801 CurOffsets.push_back(Elt: Offset);
8802 CurCounts.push_back(Elt: llvm::ConstantInt::get(Ty: CGF.Int64Ty, /*V=*/1));
8803 CurStrides.push_back(Elt: CurStrides.back());
8804 continue;
8805 }
8806
8807 const auto *OASE = dyn_cast<ArraySectionExpr>(Val: AssocExpr);
8808
8809 if (!OASE)
8810 continue;
8811
8812 // Offset
8813 const Expr *OffsetExpr = OASE->getLowerBound();
8814 llvm::Value *Offset = nullptr;
8815 if (!OffsetExpr) {
8816 // If offset is absent, then we just set it to zero.
8817 Offset = llvm::ConstantInt::get(Ty: CGF.Int64Ty, V: 0);
8818 } else {
8819 Offset = CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: OffsetExpr),
8820 DestTy: CGF.Int64Ty,
8821 /*isSigned=*/false);
8822 }
8823
8824 // Count
8825 const Expr *CountExpr = OASE->getLength();
8826 llvm::Value *Count = nullptr;
8827 if (!CountExpr) {
8828 // In Clang, once a high dimension is an array section, we construct all
8829 // the lower dimension as array section, however, for case like
8830 // arr[0:2][2], Clang construct the inner dimension as an array section
8831 // but it actually is not in an array section form according to spec.
8832 if (!OASE->getColonLocFirst().isValid() &&
8833 !OASE->getColonLocSecond().isValid()) {
8834 Count = llvm::ConstantInt::get(Ty: CGF.Int64Ty, V: 1);
8835 } else {
8836 // OpenMP 5.0, 2.1.5 Array Sections, Description.
8837 // When the length is absent it defaults to ⌈(size −
8838 // lower-bound)/stride⌉, where size is the size of the array
8839 // dimension.
8840 const Expr *StrideExpr = OASE->getStride();
8841 llvm::Value *Stride =
8842 StrideExpr
8843 ? CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: StrideExpr),
8844 DestTy: CGF.Int64Ty, /*isSigned=*/false)
8845 : nullptr;
8846 if (Stride)
8847 Count = CGF.Builder.CreateUDiv(
8848 LHS: CGF.Builder.CreateNUWSub(LHS: *DI, RHS: Offset), RHS: Stride);
8849 else
8850 Count = CGF.Builder.CreateNUWSub(LHS: *DI, RHS: Offset);
8851 }
8852 } else {
8853 Count = CGF.EmitScalarExpr(E: CountExpr);
8854 }
8855 Count = CGF.Builder.CreateIntCast(V: Count, DestTy: CGF.Int64Ty, /*isSigned=*/false);
8856 CurCounts.push_back(Elt: Count);
8857
8858 // Stride_n' = Stride_n * (D_0 * D_1 ... * D_n-1) * Unit size
8859 // Offset_n' = Offset_n * (D_0 * D_1 ... * D_n-1) * Unit size
8860 // Take `int arr[5][5][5]` and `arr[0:2:2][1:2:1][0:2:2]` as an example:
8861 // Offset Count Stride
8862 // D0 0 4 1 (int) <- dummy dimension
8863 // D1 0 2 8 (2 * (1) * 4)
8864 // D2 100 2 20 (1 * (1 * 5) * 4)
8865 // D3 0 2 200 (2 * (1 * 5 * 4) * 4)
8866 const Expr *StrideExpr = OASE->getStride();
8867 llvm::Value *Stride =
8868 StrideExpr
8869 ? CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: StrideExpr),
8870 DestTy: CGF.Int64Ty, /*isSigned=*/false)
8871 : nullptr;
8872 DimProd = CGF.Builder.CreateNUWMul(LHS: DimProd, RHS: *(DI - 1));
8873 if (Stride)
8874 CurStrides.push_back(Elt: CGF.Builder.CreateNUWMul(LHS: DimProd, RHS: Stride));
8875 else
8876 CurStrides.push_back(Elt: DimProd);
8877
8878 Offset = CGF.Builder.CreateNUWMul(LHS: DimProd, RHS: Offset);
8879 CurOffsets.push_back(Elt: Offset);
8880
8881 if (DI != DimSizes.end())
8882 ++DI;
8883 }
8884
8885 CombinedInfo.NonContigInfo.Offsets.push_back(Elt: CurOffsets);
8886 CombinedInfo.NonContigInfo.Counts.push_back(Elt: CurCounts);
8887 CombinedInfo.NonContigInfo.Strides.push_back(Elt: CurStrides);
8888 }
8889
8890 /// Return the adjusted map modifiers if the declaration a capture refers to
8891 /// appears in a first-private clause. This is expected to be used only with
8892 /// directives that start with 'target'.
8893 OpenMPOffloadMappingFlags
8894 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const {
8895 assert(Cap.capturesVariable() && "Expected capture by reference only!");
8896
8897 // A first private variable captured by reference will use only the
8898 // 'private ptr' and 'map to' flag. Return the right flags if the captured
8899 // declaration is known as first-private in this handler.
8900 if (FirstPrivateDecls.count(Val: Cap.getCapturedVar())) {
8901 if (Cap.getCapturedVar()->getType()->isAnyPointerType())
8902 return OpenMPOffloadMappingFlags::OMP_MAP_TO |
8903 OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ;
8904 return OpenMPOffloadMappingFlags::OMP_MAP_PRIVATE |
8905 OpenMPOffloadMappingFlags::OMP_MAP_TO;
8906 }
8907 auto I = LambdasMap.find(Val: Cap.getCapturedVar()->getCanonicalDecl());
8908 if (I != LambdasMap.end())
8909 // for map(to: lambda): using user specified map type.
8910 return getMapTypeBits(
8911 MapType: I->getSecond()->getMapType(), MapModifiers: I->getSecond()->getMapTypeModifiers(),
8912 /*MotionModifiers=*/{}, IsImplicit: I->getSecond()->isImplicit(),
8913 /*AddPtrFlag=*/false,
8914 /*AddIsTargetParamFlag=*/false,
8915 /*isNonContiguous=*/IsNonContiguous: false);
8916 return OpenMPOffloadMappingFlags::OMP_MAP_TO |
8917 OpenMPOffloadMappingFlags::OMP_MAP_FROM;
8918 }
8919
8920 void getPlainLayout(const CXXRecordDecl *RD,
8921 llvm::SmallVectorImpl<const FieldDecl *> &Layout,
8922 bool AsBase) const {
8923 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD);
8924
8925 llvm::StructType *St =
8926 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType();
8927
8928 unsigned NumElements = St->getNumElements();
8929 llvm::SmallVector<
8930 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4>
8931 RecordLayout(NumElements);
8932
8933 // Fill bases.
8934 for (const auto &I : RD->bases()) {
8935 if (I.isVirtual())
8936 continue;
8937
8938 QualType BaseTy = I.getType();
8939 const auto *Base = BaseTy->getAsCXXRecordDecl();
8940 // Ignore empty bases.
8941 if (CodeGenUtils::isEmptyRecordForLayout(Ctx: CGF.getContext(), T: BaseTy) ||
8942 CGF.getContext()
8943 .getASTRecordLayout(D: Base)
8944 .getNonVirtualSize()
8945 .isZero())
8946 continue;
8947
8948 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(RD: Base);
8949 RecordLayout[FieldIndex] = Base;
8950 }
8951 // Fill in virtual bases.
8952 for (const auto &I : RD->vbases()) {
8953 QualType BaseTy = I.getType();
8954 // Ignore empty bases.
8955 if (CodeGenUtils::isEmptyRecordForLayout(Ctx: CGF.getContext(), T: BaseTy))
8956 continue;
8957
8958 const auto *Base = BaseTy->getAsCXXRecordDecl();
8959 unsigned FieldIndex = RL.getVirtualBaseIndex(base: Base);
8960 if (RecordLayout[FieldIndex])
8961 continue;
8962 RecordLayout[FieldIndex] = Base;
8963 }
8964 // Fill in all the fields.
8965 assert(!RD->isUnion() && "Unexpected union.");
8966 for (const auto *Field : RD->fields()) {
8967 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we
8968 // will fill in later.)
8969 if (!Field->isBitField() &&
8970 !CodeGenUtils::isEmptyFieldForLayout(Ctx: CGF.getContext(), FD: Field)) {
8971 unsigned FieldIndex = RL.getLLVMFieldNo(FD: Field);
8972 RecordLayout[FieldIndex] = Field;
8973 }
8974 }
8975 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>
8976 &Data : RecordLayout) {
8977 if (Data.isNull())
8978 continue;
8979 if (const auto *Base = dyn_cast<const CXXRecordDecl *>(Val: Data))
8980 getPlainLayout(RD: Base, Layout, /*AsBase=*/true);
8981 else
8982 Layout.push_back(Elt: cast<const FieldDecl *>(Val: Data));
8983 }
8984 }
8985
8986 /// Returns the address corresponding to \p PointerExpr.
8987 static Address getAttachPtrAddr(const Expr *PointerExpr,
8988 CodeGenFunction &CGF) {
8989 assert(PointerExpr && "Cannot get addr from null attach-ptr expr");
8990 Address AttachPtrAddr = Address::invalid();
8991
8992 if (auto *DRE = dyn_cast<DeclRefExpr>(Val: PointerExpr)) {
8993 // If the pointer is a variable, we can use its address directly.
8994 AttachPtrAddr = CGF.EmitLValue(E: DRE).getAddress();
8995 } else if (auto *OASE = dyn_cast<ArraySectionExpr>(Val: PointerExpr)) {
8996 AttachPtrAddr =
8997 CGF.EmitArraySectionExpr(E: OASE, /*IsLowerBound=*/true).getAddress();
8998 } else if (auto *ASE = dyn_cast<ArraySubscriptExpr>(Val: PointerExpr)) {
8999 AttachPtrAddr = CGF.EmitLValue(E: ASE).getAddress();
9000 } else if (auto *ME = dyn_cast<MemberExpr>(Val: PointerExpr)) {
9001 AttachPtrAddr = CGF.EmitMemberExpr(E: ME).getAddress();
9002 } else if (auto *UO = dyn_cast<UnaryOperator>(Val: PointerExpr)) {
9003 assert(UO->getOpcode() == UO_Deref &&
9004 "Unexpected unary-operator on attach-ptr-expr");
9005 AttachPtrAddr = CGF.EmitLValue(E: UO).getAddress();
9006 }
9007 assert(AttachPtrAddr.isValid() &&
9008 "Failed to get address for attach pointer expression");
9009 return AttachPtrAddr;
9010 }
9011
9012 /// Get the address of the attach pointer, and a load from it, to get the
9013 /// pointee base address.
9014 /// \return A pair containing AttachPtrAddr and AttachPteeBaseAddr. The pair
9015 /// contains invalid addresses if \p AttachPtrExpr is null.
9016 static std::pair<Address, Address>
9017 getAttachPtrAddrAndPteeBaseAddr(const Expr *AttachPtrExpr,
9018 CodeGenFunction &CGF) {
9019
9020 if (!AttachPtrExpr)
9021 return {Address::invalid(), Address::invalid()};
9022
9023 Address AttachPtrAddr = getAttachPtrAddr(PointerExpr: AttachPtrExpr, CGF);
9024 assert(AttachPtrAddr.isValid() && "Invalid attach pointer addr");
9025
9026 QualType AttachPtrType =
9027 OMPClauseMappableExprCommon::getComponentExprElementType(Exp: AttachPtrExpr)
9028 .getCanonicalType();
9029
9030 Address AttachPteeBaseAddr = CGF.EmitLoadOfPointer(
9031 Ptr: AttachPtrAddr, PtrTy: AttachPtrType->castAs<PointerType>());
9032 assert(AttachPteeBaseAddr.isValid() && "Invalid attach pointee base addr");
9033
9034 return {AttachPtrAddr, AttachPteeBaseAddr};
9035 }
9036
9037 /// Returns whether an attach entry should be emitted for a map on
9038 /// \p MapBaseDecl on the directive \p CurDir.
9039 static bool
9040 shouldEmitAttachEntry(const Expr *PointerExpr, const ValueDecl *MapBaseDecl,
9041 CodeGenFunction &CGF,
9042 llvm::PointerUnion<const OMPExecutableDirective *,
9043 const OMPDeclareMapperDecl *>
9044 CurDir) {
9045 if (!PointerExpr)
9046 return false;
9047
9048 // Pointer attachment is needed at map-entering time or for declare
9049 // mappers.
9050 return isa<const OMPDeclareMapperDecl *>(Val: CurDir) ||
9051 isOpenMPTargetMapEnteringDirective(
9052 DKind: cast<const OMPExecutableDirective *>(Val&: CurDir)
9053 ->getDirectiveKind());
9054 }
9055
9056 /// Computes the attach-ptr expr for \p Components, and updates various maps
9057 /// with the information.
9058 /// It internally calls OMPClauseMappableExprCommon::findAttachPtrExpr()
9059 /// with the OpenMPDirectiveKind extracted from \p CurDir.
9060 /// It updates AttachPtrComputationOrderMap, AttachPtrComponentDepthMap, and
9061 /// AttachPtrExprMap.
9062 void collectAttachPtrExprInfo(
9063 OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
9064 llvm::PointerUnion<const OMPExecutableDirective *,
9065 const OMPDeclareMapperDecl *>
9066 CurDir) {
9067
9068 OpenMPDirectiveKind CurDirectiveID =
9069 isa<const OMPDeclareMapperDecl *>(Val: CurDir)
9070 ? OMPD_declare_mapper
9071 : cast<const OMPExecutableDirective *>(Val&: CurDir)->getDirectiveKind();
9072
9073 const auto &[AttachPtrExpr, Depth] =
9074 OMPClauseMappableExprCommon::findAttachPtrExpr(Components,
9075 CurDirKind: CurDirectiveID);
9076
9077 AttachPtrComputationOrderMap.try_emplace(
9078 Key: AttachPtrExpr, Args: AttachPtrComputationOrderMap.size());
9079 AttachPtrComponentDepthMap.try_emplace(Key: AttachPtrExpr, Args: Depth);
9080 AttachPtrExprMap.try_emplace(Key: Components, Args: AttachPtrExpr);
9081 }
9082
9083 /// Generate all the base pointers, section pointers, sizes, map types, and
9084 /// mappers for the extracted mappable expressions (all included in \a
9085 /// CombinedInfo). Also, for each item that relates with a device pointer, a
9086 /// pair of the relevant declaration and index where it occurs is appended to
9087 /// the device pointers info array.
9088 void generateAllInfoForClauses(
9089 ArrayRef<const OMPClause *> Clauses, MapCombinedInfoTy &CombinedInfo,
9090 llvm::OpenMPIRBuilder &OMPBuilder,
9091 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet =
9092 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const {
9093 // We have to process the component lists that relate with the same
9094 // declaration in a single chunk so that we can generate the map flags
9095 // correctly. Therefore, we organize all lists in a map.
9096 enum MapKind { Present, Allocs, Other, Total };
9097 llvm::MapVector<CanonicalDeclPtr<const Decl>,
9098 SmallVector<SmallVector<MapInfo, 8>, 4>>
9099 Info;
9100
9101 // Helper function to fill the information map for the different supported
9102 // clauses.
9103 auto &&InfoGen =
9104 [&Info, &SkipVarSet](
9105 const ValueDecl *D, MapKind Kind,
9106 OMPClauseMappableExprCommon::MappableExprComponentListRef L,
9107 OpenMPMapClauseKind MapType,
9108 ArrayRef<OpenMPMapModifierKind> MapModifiers,
9109 ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
9110 bool ReturnDevicePointer, bool IsImplicit, const ValueDecl *Mapper,
9111 const Expr *VarRef = nullptr, bool ForDeviceAddr = false) {
9112 if (SkipVarSet.contains(V: D))
9113 return;
9114 auto It = Info.try_emplace(Key: D, Args: Total).first;
9115 It->second[Kind].emplace_back(
9116 Args&: L, Args&: MapType, Args&: MapModifiers, Args&: MotionModifiers, Args&: ReturnDevicePointer,
9117 Args&: IsImplicit, Args&: Mapper, Args&: VarRef, Args&: ForDeviceAddr);
9118 };
9119
9120 for (const auto *Cl : Clauses) {
9121 const auto *C = dyn_cast<OMPMapClause>(Val: Cl);
9122 if (!C)
9123 continue;
9124 MapKind Kind = Other;
9125 if (llvm::is_contained(Range: C->getMapTypeModifiers(),
9126 Element: OMPC_MAP_MODIFIER_present))
9127 Kind = Present;
9128 else if (C->getMapType() == OMPC_MAP_alloc)
9129 Kind = Allocs;
9130 const auto *EI = C->getVarRefs().begin();
9131 for (const auto L : C->component_lists()) {
9132 const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr;
9133 InfoGen(std::get<0>(t: L), Kind, std::get<1>(t: L), C->getMapType(),
9134 C->getMapTypeModifiers(), {},
9135 /*ReturnDevicePointer=*/false, C->isImplicit(), std::get<2>(t: L),
9136 E);
9137 ++EI;
9138 }
9139 }
9140 for (const auto *Cl : Clauses) {
9141 const auto *C = dyn_cast<OMPToClause>(Val: Cl);
9142 if (!C)
9143 continue;
9144 MapKind Kind = Other;
9145 if (llvm::is_contained(Range: C->getMotionModifiers(),
9146 Element: OMPC_MOTION_MODIFIER_present))
9147 Kind = Present;
9148 if (llvm::is_contained(Range: C->getMotionModifiers(),
9149 Element: OMPC_MOTION_MODIFIER_iterator)) {
9150 if (auto *IteratorExpr = dyn_cast<OMPIteratorExpr>(
9151 Val: C->getIteratorModifier()->IgnoreParenImpCasts())) {
9152 const auto *VD = cast<VarDecl>(Val: IteratorExpr->getIteratorDecl(I: 0));
9153 CGF.EmitVarDecl(D: *VD);
9154 }
9155 }
9156
9157 const auto *EI = C->getVarRefs().begin();
9158 for (const auto L : C->component_lists()) {
9159 InfoGen(std::get<0>(t: L), Kind, std::get<1>(t: L), OMPC_MAP_to, {},
9160 C->getMotionModifiers(), /*ReturnDevicePointer=*/false,
9161 C->isImplicit(), std::get<2>(t: L), *EI);
9162 ++EI;
9163 }
9164 }
9165 for (const auto *Cl : Clauses) {
9166 const auto *C = dyn_cast<OMPFromClause>(Val: Cl);
9167 if (!C)
9168 continue;
9169 MapKind Kind = Other;
9170 if (llvm::is_contained(Range: C->getMotionModifiers(),
9171 Element: OMPC_MOTION_MODIFIER_present))
9172 Kind = Present;
9173 if (llvm::is_contained(Range: C->getMotionModifiers(),
9174 Element: OMPC_MOTION_MODIFIER_iterator)) {
9175 if (auto *IteratorExpr = dyn_cast<OMPIteratorExpr>(
9176 Val: C->getIteratorModifier()->IgnoreParenImpCasts())) {
9177 const auto *VD = cast<VarDecl>(Val: IteratorExpr->getIteratorDecl(I: 0));
9178 CGF.EmitVarDecl(D: *VD);
9179 }
9180 }
9181
9182 const auto *EI = C->getVarRefs().begin();
9183 for (const auto L : C->component_lists()) {
9184 InfoGen(std::get<0>(t: L), Kind, std::get<1>(t: L), OMPC_MAP_from, {},
9185 C->getMotionModifiers(),
9186 /*ReturnDevicePointer=*/false, C->isImplicit(), std::get<2>(t: L),
9187 *EI);
9188 ++EI;
9189 }
9190 }
9191
9192 // Look at the use_device_ptr and use_device_addr clauses information and
9193 // mark the existing map entries as such. If there is no map information for
9194 // an entry in the use_device_ptr and use_device_addr list, we create one
9195 // with map type 'return_param' and zero size section. It is the user's
9196 // fault if that was not mapped before. If there is no map information, then
9197 // we defer the emission of that entry until all the maps for the same VD
9198 // have been handled.
9199 MapCombinedInfoTy UseDeviceDataCombinedInfo;
9200
9201 auto &&UseDeviceDataCombinedInfoGen =
9202 [&UseDeviceDataCombinedInfo](const ValueDecl *VD, llvm::Value *Ptr,
9203 CodeGenFunction &CGF, bool IsDevAddr,
9204 bool HasUdpFbNullify = false) {
9205 UseDeviceDataCombinedInfo.Exprs.push_back(Elt: VD);
9206 UseDeviceDataCombinedInfo.BasePointers.emplace_back(Args&: Ptr);
9207 UseDeviceDataCombinedInfo.DevicePtrDecls.emplace_back(Args&: VD);
9208 UseDeviceDataCombinedInfo.DevicePointers.emplace_back(
9209 Args: IsDevAddr ? DeviceInfoTy::Address : DeviceInfoTy::Pointer);
9210 // FIXME: For use_device_addr on array-sections, this should
9211 // be the starting address of the section.
9212 // e.g. int *p;
9213 // ... use_device_addr(p[3])
9214 // &p[0], &p[3], /*size=*/0, RETURN_PARAM
9215 UseDeviceDataCombinedInfo.Pointers.push_back(Elt: Ptr);
9216 UseDeviceDataCombinedInfo.Sizes.push_back(
9217 Elt: llvm::Constant::getNullValue(Ty: CGF.Int64Ty));
9218 OpenMPOffloadMappingFlags Flags =
9219 OpenMPOffloadMappingFlags::OMP_MAP_RETURN_PARAM;
9220 if (HasUdpFbNullify)
9221 Flags |= OpenMPOffloadMappingFlags::OMP_MAP_FB_NULLIFY;
9222 UseDeviceDataCombinedInfo.Types.push_back(Elt: Flags);
9223 UseDeviceDataCombinedInfo.HasAttachPtr.push_back(Elt: false);
9224 UseDeviceDataCombinedInfo.Mappers.push_back(Elt: nullptr);
9225 };
9226
9227 auto &&MapInfoGen =
9228 [&UseDeviceDataCombinedInfoGen](
9229 CodeGenFunction &CGF, const Expr *IE, const ValueDecl *VD,
9230 OMPClauseMappableExprCommon::MappableExprComponentListRef
9231 Components,
9232 bool IsDevAddr, bool IEIsAttachPtrForDevAddr = false,
9233 bool HasUdpFbNullify = false) {
9234 // We didn't find any match in our map information - generate a zero
9235 // size array section.
9236 llvm::Value *Ptr;
9237 if (IsDevAddr && !IEIsAttachPtrForDevAddr) {
9238 if (IE->isGLValue())
9239 Ptr = CGF.EmitLValue(E: IE).getPointer(CGF);
9240 else
9241 Ptr = CGF.EmitScalarExpr(E: IE);
9242 } else {
9243 Ptr = CGF.EmitLoadOfScalar(lvalue: CGF.EmitLValue(E: IE), Loc: IE->getExprLoc());
9244 }
9245 bool TreatDevAddrAsDevPtr = IEIsAttachPtrForDevAddr;
9246 // For the purpose of address-translation, treat something like the
9247 // following:
9248 // int *p;
9249 // ... use_device_addr(p[1])
9250 // equivalent to
9251 // ... use_device_ptr(p)
9252 UseDeviceDataCombinedInfoGen(VD, Ptr, CGF, /*IsDevAddr=*/IsDevAddr &&
9253 !TreatDevAddrAsDevPtr,
9254 HasUdpFbNullify);
9255 };
9256
9257 auto &&IsMapInfoExist =
9258 [&Info, this](CodeGenFunction &CGF, const ValueDecl *VD, const Expr *IE,
9259 const Expr *DesiredAttachPtrExpr, bool IsDevAddr,
9260 bool HasUdpFbNullify = false) -> bool {
9261 // We potentially have map information for this declaration already.
9262 // Look for the first set of components that refer to it. If found,
9263 // return true.
9264 // If the first component is a member expression, we have to look into
9265 // 'this', which maps to null in the map of map information. Otherwise
9266 // look directly for the information.
9267 auto It = Info.find(Key: isa<MemberExpr>(Val: IE) ? nullptr : VD);
9268 if (It != Info.end()) {
9269 bool Found = false;
9270 for (auto &Data : It->second) {
9271 MapInfo *CI = nullptr;
9272 // We potentially have multiple maps for the same decl. We need to
9273 // only consider those for which the attach-ptr matches the desired
9274 // attach-ptr.
9275 auto *It = llvm::find_if(Range&: Data, P: [&](const MapInfo &MI) {
9276 if (MI.Components.back().getAssociatedDeclaration() != VD)
9277 return false;
9278
9279 const Expr *MapAttachPtr = getAttachPtrExpr(Components: MI.Components);
9280 bool Match = AttachPtrComparator.areEqual(LHS: MapAttachPtr,
9281 RHS: DesiredAttachPtrExpr);
9282 return Match;
9283 });
9284
9285 if (It != Data.end())
9286 CI = &*It;
9287
9288 if (CI) {
9289 if (IsDevAddr) {
9290 CI->ForDeviceAddr = true;
9291 CI->ReturnDevicePointer = true;
9292 CI->HasUdpFbNullify = HasUdpFbNullify;
9293 Found = true;
9294 break;
9295 } else {
9296 auto PrevCI = std::next(x: CI->Components.rbegin());
9297 const auto *VarD = dyn_cast<VarDecl>(Val: VD);
9298 const Expr *AttachPtrExpr = getAttachPtrExpr(Components: CI->Components);
9299 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
9300 isa<MemberExpr>(Val: IE) ||
9301 !VD->getType().getNonReferenceType()->isPointerType() ||
9302 PrevCI == CI->Components.rend() ||
9303 isa<MemberExpr>(Val: PrevCI->getAssociatedExpression()) || !VarD ||
9304 VarD->hasLocalStorage() ||
9305 (isa_and_nonnull<DeclRefExpr>(Val: AttachPtrExpr) &&
9306 VD == cast<DeclRefExpr>(Val: AttachPtrExpr)->getDecl())) {
9307 CI->ForDeviceAddr = IsDevAddr;
9308 CI->ReturnDevicePointer = true;
9309 CI->HasUdpFbNullify = HasUdpFbNullify;
9310 Found = true;
9311 break;
9312 }
9313 }
9314 }
9315 }
9316 return Found;
9317 }
9318 return false;
9319 };
9320
9321 // Look at the use_device_ptr clause information and mark the existing map
9322 // entries as such. If there is no map information for an entry in the
9323 // use_device_ptr list, we create one with map type 'alloc' and zero size
9324 // section. It is the user fault if that was not mapped before. If there is
9325 // no map information and the pointer is a struct member, then we defer the
9326 // emission of that entry until the whole struct has been processed.
9327 for (const auto *Cl : Clauses) {
9328 const auto *C = dyn_cast<OMPUseDevicePtrClause>(Val: Cl);
9329 if (!C)
9330 continue;
9331 bool HasUdpFbNullify =
9332 C->getFallbackModifier() == OMPC_USE_DEVICE_PTR_FALLBACK_fb_nullify;
9333 for (const auto L : C->component_lists()) {
9334 OMPClauseMappableExprCommon::MappableExprComponentListRef Components =
9335 std::get<1>(t: L);
9336 assert(!Components.empty() &&
9337 "Not expecting empty list of components!");
9338 const ValueDecl *VD = Components.back().getAssociatedDeclaration();
9339 VD = cast<ValueDecl>(Val: VD->getCanonicalDecl());
9340 const Expr *IE = Components.back().getAssociatedExpression();
9341 // For use_device_ptr, we match an existing map clause if its attach-ptr
9342 // is same as the use_device_ptr operand. e.g.
9343 // map expr | use_device_ptr expr | current behavior
9344 // ---------|---------------------|-----------------
9345 // p[1] | p | match
9346 // ps->a | ps | match
9347 // p | p | no match
9348 const Expr *UDPOperandExpr =
9349 Components.front().getAssociatedExpression();
9350 if (IsMapInfoExist(CGF, VD, IE,
9351 /*DesiredAttachPtrExpr=*/UDPOperandExpr,
9352 /*IsDevAddr=*/false, HasUdpFbNullify))
9353 continue;
9354 MapInfoGen(CGF, IE, VD, Components, /*IsDevAddr=*/false,
9355 /*IEIsAttachPtrForDevAddr=*/false, HasUdpFbNullify);
9356 }
9357 }
9358
9359 llvm::SmallDenseSet<CanonicalDeclPtr<const Decl>, 4> Processed;
9360 for (const auto *Cl : Clauses) {
9361 const auto *C = dyn_cast<OMPUseDeviceAddrClause>(Val: Cl);
9362 if (!C)
9363 continue;
9364 for (const auto L : C->component_lists()) {
9365 OMPClauseMappableExprCommon::MappableExprComponentListRef Components =
9366 std::get<1>(t: L);
9367 assert(!std::get<1>(L).empty() &&
9368 "Not expecting empty list of components!");
9369 const ValueDecl *VD = std::get<1>(t: L).back().getAssociatedDeclaration();
9370 if (!Processed.insert(V: VD).second)
9371 continue;
9372 VD = cast<ValueDecl>(Val: VD->getCanonicalDecl());
9373 // For use_device_addr, we match an existing map clause if the
9374 // use_device_addr operand's attach-ptr matches the map operand's
9375 // attach-ptr.
9376 // We chould also restrict to only match cases when there is a full
9377 // match between the map/use_device_addr clause exprs, but that may be
9378 // unnecessary.
9379 //
9380 // map expr | use_device_addr expr | current | possible restrictive/
9381 // | | behavior | safer behavior
9382 // ---------|----------------------|-----------|-----------------------
9383 // p | p | match | match
9384 // p[0] | p[0] | match | match
9385 // p[0:1] | p[0] | match | no match
9386 // p[0:1] | p[2:1] | match | no match
9387 // p[1] | p[0] | match | no match
9388 // ps->a | ps->b | match | no match
9389 // p | p[0] | no match | no match
9390 // pp | pp[0][0] | no match | no match
9391 const Expr *UDAAttachPtrExpr = getAttachPtrExpr(Components);
9392 const Expr *IE = std::get<1>(t: L).back().getAssociatedExpression();
9393 assert((!UDAAttachPtrExpr || UDAAttachPtrExpr == IE) &&
9394 "use_device_addr operand has an attach-ptr, but does not match "
9395 "last component's expr.");
9396 if (IsMapInfoExist(CGF, VD, IE,
9397 /*DesiredAttachPtrExpr=*/UDAAttachPtrExpr,
9398 /*IsDevAddr=*/true))
9399 continue;
9400 MapInfoGen(CGF, IE, VD, Components,
9401 /*IsDevAddr=*/true,
9402 /*IEIsAttachPtrForDevAddr=*/UDAAttachPtrExpr != nullptr);
9403 }
9404 }
9405
9406 for (const auto &Data : Info) {
9407 MapCombinedInfoTy CurInfo;
9408 const Decl *D = Data.first;
9409 const ValueDecl *VD = cast_or_null<ValueDecl>(Val: D);
9410 // Group component lists by their AttachPtrExpr and process them in order
9411 // of increasing complexity (nullptr first, then simple expressions like
9412 // p, then more complex ones like p[0], etc.)
9413 //
9414 // This is similar to how generateInfoForCaptureFromClauseInfo handles
9415 // grouping for target constructs.
9416 SmallVector<std::pair<const Expr *, MapInfo>, 16> AttachPtrMapInfoPairs;
9417
9418 // First, collect all MapData entries with their attach-ptr exprs.
9419 for (const auto &M : Data.second) {
9420 for (const MapInfo &L : M) {
9421 assert(!L.Components.empty() &&
9422 "Not expecting declaration with no component lists.");
9423
9424 const Expr *AttachPtrExpr = getAttachPtrExpr(Components: L.Components);
9425 AttachPtrMapInfoPairs.emplace_back(Args&: AttachPtrExpr, Args: L);
9426 }
9427 }
9428
9429 // Next, sort by increasing order of their complexity.
9430 llvm::stable_sort(Range&: AttachPtrMapInfoPairs,
9431 C: [this](const auto &LHS, const auto &RHS) {
9432 return AttachPtrComparator(LHS.first, RHS.first);
9433 });
9434
9435 // And finally, process them all in order, grouping those with
9436 // equivalent attach-ptr exprs together.
9437 auto *It = AttachPtrMapInfoPairs.begin();
9438 while (It != AttachPtrMapInfoPairs.end()) {
9439 const Expr *AttachPtrExpr = It->first;
9440
9441 SmallVector<MapInfo, 8> GroupLists;
9442 while (It != AttachPtrMapInfoPairs.end() &&
9443 (It->first == AttachPtrExpr ||
9444 AttachPtrComparator.areEqual(LHS: It->first, RHS: AttachPtrExpr))) {
9445 GroupLists.push_back(Elt: It->second);
9446 ++It;
9447 }
9448 assert(!GroupLists.empty() && "GroupLists should not be empty");
9449
9450 StructRangeInfoTy PartialStruct;
9451 AttachInfoTy AttachInfo;
9452 MapCombinedInfoTy GroupCurInfo;
9453 // Current group's struct base information:
9454 MapCombinedInfoTy GroupStructBaseCurInfo;
9455 for (const MapInfo &L : GroupLists) {
9456 // Remember the current base pointer index.
9457 unsigned CurrentBasePointersIdx = GroupCurInfo.BasePointers.size();
9458 unsigned StructBasePointersIdx =
9459 GroupStructBaseCurInfo.BasePointers.size();
9460
9461 GroupCurInfo.NonContigInfo.IsNonContiguous =
9462 L.Components.back().isNonContiguous();
9463 generateInfoForComponentList(
9464 MapType: L.MapType, MapModifiers: L.MapModifiers, MotionModifiers: L.MotionModifiers, Components: L.Components,
9465 CombinedInfo&: GroupCurInfo, StructBaseCombinedInfo&: GroupStructBaseCurInfo, PartialStruct, AttachInfo,
9466 /*IsFirstComponentList=*/false, IsImplicit: L.IsImplicit,
9467 /*GenerateAllInfoForClauses*/ true, Mapper: L.Mapper, ForDeviceAddr: L.ForDeviceAddr, BaseDecl: VD,
9468 MapExpr: L.VarRef, /*OverlappedElements*/ {});
9469
9470 // If this entry relates to a device pointer, set the relevant
9471 // declaration and add the 'return pointer' flag.
9472 if (L.ReturnDevicePointer) {
9473 // Check whether a value was added to either GroupCurInfo or
9474 // GroupStructBaseCurInfo and error if no value was added to either
9475 // of them:
9476 assert((CurrentBasePointersIdx < GroupCurInfo.BasePointers.size() ||
9477 StructBasePointersIdx <
9478 GroupStructBaseCurInfo.BasePointers.size()) &&
9479 "Unexpected number of mapped base pointers.");
9480
9481 // Choose a base pointer index which is always valid:
9482 const ValueDecl *RelevantVD =
9483 L.Components.back().getAssociatedDeclaration();
9484 assert(RelevantVD &&
9485 "No relevant declaration related with device pointer??");
9486
9487 // If GroupStructBaseCurInfo has been updated this iteration then
9488 // work on the first new entry added to it i.e. make sure that when
9489 // multiple values are added to any of the lists, the first value
9490 // added is being modified by the assignments below (not the last
9491 // value added).
9492 auto SetDevicePointerInfo = [&](MapCombinedInfoTy &Info,
9493 unsigned Idx) {
9494 Info.DevicePtrDecls[Idx] = RelevantVD;
9495 Info.DevicePointers[Idx] = L.ForDeviceAddr
9496 ? DeviceInfoTy::Address
9497 : DeviceInfoTy::Pointer;
9498 Info.Types[Idx] |=
9499 OpenMPOffloadMappingFlags::OMP_MAP_RETURN_PARAM;
9500 if (L.HasUdpFbNullify)
9501 Info.Types[Idx] |=
9502 OpenMPOffloadMappingFlags::OMP_MAP_FB_NULLIFY;
9503 };
9504
9505 if (StructBasePointersIdx <
9506 GroupStructBaseCurInfo.BasePointers.size())
9507 SetDevicePointerInfo(GroupStructBaseCurInfo,
9508 StructBasePointersIdx);
9509 else
9510 SetDevicePointerInfo(GroupCurInfo, CurrentBasePointersIdx);
9511 }
9512 }
9513
9514 // Unify entries in one list making sure the struct mapping precedes the
9515 // individual fields:
9516 MapCombinedInfoTy GroupUnionCurInfo;
9517 GroupUnionCurInfo.append(CurInfo&: GroupStructBaseCurInfo);
9518 GroupUnionCurInfo.append(CurInfo&: GroupCurInfo);
9519
9520 // If there is an entry in PartialStruct it means we have a struct with
9521 // individual members mapped. Emit an extra combined entry.
9522 if (PartialStruct.Base.isValid()) {
9523 // Prepend a synthetic dimension of length 1 to represent the
9524 // aggregated struct object. Using 1 (not 0, as 0 produced an
9525 // incorrect non-contiguous descriptor (DimSize==1), causing the
9526 // non-contiguous motion clause path to be skipped.) is important:
9527 // * It preserves the correct rank so targetDataUpdate() computes
9528 // DimSize == 2 for cases like strided array sections originating
9529 // from user-defined mappers (e.g. test with s.data[0:8:2]).
9530 GroupUnionCurInfo.NonContigInfo.Dims.insert(
9531 I: GroupUnionCurInfo.NonContigInfo.Dims.begin(), Elt: 1);
9532 emitCombinedEntry(
9533 CombinedInfo&: CurInfo, CurTypes&: GroupUnionCurInfo.Types, PartialStruct, AttachInfo,
9534 /*IsMapThis=*/!VD, OMPBuilder, VD,
9535 /*OffsetForMemberOfFlag=*/CombinedInfo.BasePointers.size(),
9536 /*NotTargetParams=*/true);
9537 }
9538
9539 // Append this group's results to the overall CurInfo in the correct
9540 // order: combined-entry -> original-field-entries -> attach-entry
9541 CurInfo.append(CurInfo&: GroupUnionCurInfo);
9542 if (AttachInfo.isValid())
9543 emitAttachEntry(CGF, CombinedInfo&: CurInfo, AttachInfo);
9544 }
9545
9546 // We need to append the results of this capture to what we already have.
9547 CombinedInfo.append(CurInfo);
9548 }
9549 // Append data for use_device_ptr/addr clauses.
9550 CombinedInfo.append(CurInfo&: UseDeviceDataCombinedInfo);
9551 }
9552
9553public:
9554 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF)
9555 : CurDir(&Dir), CGF(CGF), AttachPtrComparator(*this) {
9556 // Extract firstprivate clause information.
9557 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>())
9558 for (const auto *D : C->varlist()) {
9559 const ValueDecl *VD = cast<DeclRefExpr>(Val: D)->getDecl();
9560 if (const auto *BD = dyn_cast<BindingDecl>(Val: VD))
9561 VD = cast<VarDecl>(Val: BD->getDecomposedDecl());
9562 FirstPrivateDecls.try_emplace(Key: cast<VarDecl>(Val: VD), Args: C->isImplicit());
9563 }
9564 // Extract implicit firstprivates from uses_allocators clauses.
9565 for (const auto *C : Dir.getClausesOfKind<OMPUsesAllocatorsClause>()) {
9566 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) {
9567 OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I);
9568 if (const auto *DRE = dyn_cast_or_null<DeclRefExpr>(Val: D.AllocatorTraits))
9569 FirstPrivateDecls.try_emplace(Key: cast<VarDecl>(Val: DRE->getDecl()),
9570 /*Implicit=*/Args: true);
9571 else if (const auto *VD = dyn_cast<VarDecl>(
9572 Val: cast<DeclRefExpr>(Val: D.Allocator->IgnoreParenImpCasts())
9573 ->getDecl()))
9574 FirstPrivateDecls.try_emplace(Key: VD, /*Implicit=*/Args: true);
9575 }
9576 }
9577 // Extract defaultmap clause information.
9578 for (const auto *C : Dir.getClausesOfKind<OMPDefaultmapClause>())
9579 if (C->getDefaultmapModifier() == OMPC_DEFAULTMAP_MODIFIER_firstprivate)
9580 DefaultmapFirstprivateKinds.insert(V: C->getDefaultmapKind());
9581 // Extract device pointer clause information.
9582 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
9583 for (auto L : C->component_lists())
9584 DevPointersMap[std::get<0>(t&: L)].push_back(Elt: std::get<1>(t&: L));
9585 // Extract device addr clause information.
9586 for (const auto *C : Dir.getClausesOfKind<OMPHasDeviceAddrClause>())
9587 for (auto L : C->component_lists())
9588 HasDevAddrsMap[std::get<0>(t&: L)].push_back(Elt: std::get<1>(t&: L));
9589 // Extract map information.
9590 for (const auto *C : Dir.getClausesOfKind<OMPMapClause>()) {
9591 if (C->getMapType() != OMPC_MAP_to)
9592 continue;
9593 for (auto L : C->component_lists()) {
9594 const ValueDecl *VD = std::get<0>(t&: L);
9595 const auto *RD = VD ? VD->getType()
9596 .getCanonicalType()
9597 .getNonReferenceType()
9598 ->getAsCXXRecordDecl()
9599 : nullptr;
9600 if (RD && RD->isLambda())
9601 LambdasMap.try_emplace(Key: std::get<0>(t&: L), Args&: C);
9602 }
9603 }
9604
9605 auto CollectAttachPtrExprsForClauseComponents = [this](const auto *C) {
9606 for (auto L : C->component_lists()) {
9607 OMPClauseMappableExprCommon::MappableExprComponentListRef Components =
9608 std::get<1>(L);
9609 if (!Components.empty())
9610 collectAttachPtrExprInfo(Components, CurDir);
9611 }
9612 };
9613
9614 // Populate the AttachPtrExprMap for all component lists from map-related
9615 // clauses.
9616 for (const auto *C : Dir.getClausesOfKind<OMPMapClause>())
9617 CollectAttachPtrExprsForClauseComponents(C);
9618 for (const auto *C : Dir.getClausesOfKind<OMPToClause>())
9619 CollectAttachPtrExprsForClauseComponents(C);
9620 for (const auto *C : Dir.getClausesOfKind<OMPFromClause>())
9621 CollectAttachPtrExprsForClauseComponents(C);
9622 for (const auto *C : Dir.getClausesOfKind<OMPUseDevicePtrClause>())
9623 CollectAttachPtrExprsForClauseComponents(C);
9624 for (const auto *C : Dir.getClausesOfKind<OMPUseDeviceAddrClause>())
9625 CollectAttachPtrExprsForClauseComponents(C);
9626 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
9627 CollectAttachPtrExprsForClauseComponents(C);
9628 for (const auto *C : Dir.getClausesOfKind<OMPHasDeviceAddrClause>())
9629 CollectAttachPtrExprsForClauseComponents(C);
9630 }
9631
9632 /// Constructor for the declare mapper directive.
9633 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF)
9634 : CurDir(&Dir), CGF(CGF), AttachPtrComparator(*this) {
9635 auto CollectAttachPtrExprsForClauseComponents = [this](const auto *C) {
9636 for (auto L : C->component_lists()) {
9637 OMPClauseMappableExprCommon::MappableExprComponentListRef Components =
9638 std::get<1>(L);
9639 if (!Components.empty())
9640 collectAttachPtrExprInfo(Components, CurDir);
9641 }
9642 };
9643
9644 // Populate the AttachPtrExprMap for all component lists from map-related
9645 // clauses in the declare mapper directive, to enable attach-style mapping
9646 // for mappers.
9647 for (const auto *Cl : Dir.clauses()) {
9648 if (const auto *C = dyn_cast<OMPMapClause>(Val: Cl))
9649 CollectAttachPtrExprsForClauseComponents(C);
9650 else if (const auto *C = dyn_cast<OMPToClause>(Val: Cl))
9651 CollectAttachPtrExprsForClauseComponents(C);
9652 else if (const auto *C = dyn_cast<OMPFromClause>(Val: Cl))
9653 CollectAttachPtrExprsForClauseComponents(C);
9654 }
9655 }
9656
9657 /// Generate code for the combined entry if we have a partially mapped struct
9658 /// and take care of the mapping flags of the arguments corresponding to
9659 /// individual struct members.
9660 /// If a valid \p AttachInfo exists, its pointee addr will be updated to point
9661 /// to the combined-entry's begin address, if emitted.
9662 /// \p PartialStruct contains attach base-pointer information.
9663 /// \returns The index of the combined entry if one was added, std::nullopt
9664 /// otherwise.
9665 void emitCombinedEntry(MapCombinedInfoTy &CombinedInfo,
9666 MapFlagsArrayTy &CurTypes,
9667 const StructRangeInfoTy &PartialStruct,
9668 AttachInfoTy &AttachInfo, bool IsMapThis,
9669 llvm::OpenMPIRBuilder &OMPBuilder, const ValueDecl *VD,
9670 unsigned OffsetForMemberOfFlag,
9671 bool NotTargetParams) const {
9672 if (CurTypes.size() == 1 &&
9673 ((CurTypes.back() & OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF) !=
9674 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF) &&
9675 !PartialStruct.IsArraySection)
9676 return;
9677 Address LBAddr = PartialStruct.LowestElem.second;
9678 Address HBAddr = PartialStruct.HighestElem.second;
9679 if (PartialStruct.HasCompleteRecord) {
9680 LBAddr = PartialStruct.LB;
9681 HBAddr = PartialStruct.LB;
9682 }
9683 CombinedInfo.Exprs.push_back(Elt: VD);
9684 // Base is the base of the struct
9685 CombinedInfo.BasePointers.push_back(Elt: PartialStruct.Base.emitRawPointer(CGF));
9686 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
9687 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
9688 // Pointer is the address of the lowest element
9689 llvm::Value *LB = LBAddr.emitRawPointer(CGF);
9690 const CXXMethodDecl *MD =
9691 CGF.CurFuncDecl ? dyn_cast<CXXMethodDecl>(Val: CGF.CurFuncDecl) : nullptr;
9692 const CXXRecordDecl *RD = MD ? MD->getParent() : nullptr;
9693 bool HasBaseClass = RD && IsMapThis ? RD->getNumBases() > 0 : false;
9694 // There should not be a mapper for a combined entry.
9695 if (HasBaseClass) {
9696 // OpenMP 5.2 148:21:
9697 // If the target construct is within a class non-static member function,
9698 // and a variable is an accessible data member of the object for which the
9699 // non-static data member function is invoked, the variable is treated as
9700 // if the this[:1] expression had appeared in a map clause with a map-type
9701 // of tofrom.
9702 // Emit this[:1]
9703 CombinedInfo.Pointers.push_back(Elt: PartialStruct.Base.emitRawPointer(CGF));
9704 QualType Ty = MD->getFunctionObjectParameterType();
9705 llvm::Value *Size =
9706 CGF.Builder.CreateIntCast(V: CGF.getTypeSize(Ty), DestTy: CGF.Int64Ty,
9707 /*isSigned=*/true);
9708 CombinedInfo.Sizes.push_back(Elt: Size);
9709 } else {
9710 CombinedInfo.Pointers.push_back(Elt: LB);
9711 // Size is (addr of {highest+1} element) - (addr of lowest element)
9712 llvm::Value *HB = HBAddr.emitRawPointer(CGF);
9713 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(
9714 Ty: HBAddr.getElementType(), Ptr: HB, /*Idx0=*/1);
9715 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(V: LB, DestTy: CGF.VoidPtrTy);
9716 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(V: HAddr, DestTy: CGF.VoidPtrTy);
9717 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(LHS: CHAddr, RHS: CLAddr);
9718 llvm::Value *Size = CGF.Builder.CreateIntCast(V: Diff, DestTy: CGF.Int64Ty,
9719 /*isSigned=*/false);
9720 CombinedInfo.Sizes.push_back(Elt: Size);
9721 }
9722 CombinedInfo.Mappers.push_back(Elt: nullptr);
9723 // Map type is always TARGET_PARAM, if generate info for captures.
9724 CombinedInfo.Types.push_back(
9725 Elt: NotTargetParams ? OpenMPOffloadMappingFlags::OMP_MAP_NONE
9726 : !PartialStruct.PreliminaryMapData.BasePointers.empty()
9727 ? OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ
9728 : OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM);
9729 // A combined entry has a base attach-ptr if its constituents do. e.g.:
9730 // map(s2.s1p->x, s2.s1p->y)
9731 // combined entry:
9732 // s2.s1p[0], s2.s1p->x, sizeof(s1p->x..y), ALLOC
9733 // here s2.s1p is the attach-ptr for the combined entry.
9734 // See the inline comments in emitUserDefinedMapper's definition for how
9735 // entries with an attach-ptr are treated.
9736 CombinedInfo.HasAttachPtr.push_back(Elt: AttachInfo.isValid());
9737 // If any element has the present modifier, then make sure the runtime
9738 // doesn't attempt to allocate the struct.
9739 if (CurTypes.end() !=
9740 llvm::find_if(Range&: CurTypes, P: [](OpenMPOffloadMappingFlags Type) {
9741 return static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
9742 Type & OpenMPOffloadMappingFlags::OMP_MAP_PRESENT);
9743 }))
9744 CombinedInfo.Types.back() |= OpenMPOffloadMappingFlags::OMP_MAP_PRESENT;
9745 // Remove TARGET_PARAM flag from the first element
9746 (*CurTypes.begin()) &= ~OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM;
9747 // If any element has the ompx_hold modifier, then make sure the runtime
9748 // uses the hold reference count for the struct as a whole so that it won't
9749 // be unmapped by an extra dynamic reference count decrement. Add it to all
9750 // elements as well so the runtime knows which reference count to check
9751 // when determining whether it's time for device-to-host transfers of
9752 // individual elements.
9753 if (CurTypes.end() !=
9754 llvm::find_if(Range&: CurTypes, P: [](OpenMPOffloadMappingFlags Type) {
9755 return static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
9756 Type & OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD);
9757 })) {
9758 CombinedInfo.Types.back() |= OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD;
9759 for (auto &M : CurTypes)
9760 M |= OpenMPOffloadMappingFlags::OMP_MAP_OMPX_HOLD;
9761 }
9762
9763 // All other current entries will be MEMBER_OF the combined entry
9764 // (except for PTR_AND_OBJ entries which do not have a placeholder value
9765 // 0xFFFF in the MEMBER_OF field, or ATTACH entries since they are expected
9766 // to be handled by themselves, after all other maps).
9767 OpenMPOffloadMappingFlags MemberOfFlag = OMPBuilder.getMemberOfFlag(
9768 Position: OffsetForMemberOfFlag + CombinedInfo.BasePointers.size() - 1);
9769 for (auto &M : CurTypes)
9770 OMPBuilder.setCorrectMemberOfFlag(Flags&: M, MemberOfFlag);
9771
9772 // When we are emitting a combined entry. If there were any pending
9773 // attachments to be done, we do them to the begin address of the combined
9774 // entry. Note that this means only one attachment per combined-entry will
9775 // be done. So, for instance, if we have:
9776 // S *ps;
9777 // ... map(ps->a, ps->b)
9778 // When we are emitting a combined entry. If AttachInfo is valid,
9779 // update the pointee address to point to the begin address of the combined
9780 // entry. This ensures that if we have multiple maps like:
9781 // `map(ps->a, ps->b)`, we still get a single ATTACH entry, like:
9782 //
9783 // &ps[0], &ps->a, sizeof(ps->a to ps->b), ALLOC // combined-entry
9784 // &ps[0], &ps->a, sizeof(ps->a), TO | FROM
9785 // &ps[0], &ps->b, sizeof(ps->b), TO | FROM
9786 // &ps, &ps->a, sizeof(void*), ATTACH // Use combined-entry's LB
9787 if (AttachInfo.isValid())
9788 AttachInfo.AttachPteeAddr = LBAddr;
9789 }
9790
9791 /// Generate all the base pointers, section pointers, sizes, map types, and
9792 /// mappers for the extracted mappable expressions (all included in \a
9793 /// CombinedInfo). Also, for each item that relates with a device pointer, a
9794 /// pair of the relevant declaration and index where it occurs is appended to
9795 /// the device pointers info array.
9796 void generateAllInfo(
9797 MapCombinedInfoTy &CombinedInfo, llvm::OpenMPIRBuilder &OMPBuilder,
9798 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet =
9799 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const {
9800 assert(isa<const OMPExecutableDirective *>(CurDir) &&
9801 "Expect a executable directive");
9802 const auto *CurExecDir = cast<const OMPExecutableDirective *>(Val: CurDir);
9803 generateAllInfoForClauses(Clauses: CurExecDir->clauses(), CombinedInfo, OMPBuilder,
9804 SkipVarSet);
9805 }
9806
9807 /// Generate all the base pointers, section pointers, sizes, map types, and
9808 /// mappers for the extracted map clauses of user-defined mapper (all included
9809 /// in \a CombinedInfo).
9810 void generateAllInfoForMapper(MapCombinedInfoTy &CombinedInfo,
9811 llvm::OpenMPIRBuilder &OMPBuilder) const {
9812 assert(isa<const OMPDeclareMapperDecl *>(CurDir) &&
9813 "Expect a declare mapper directive");
9814 const auto *CurMapperDir = cast<const OMPDeclareMapperDecl *>(Val: CurDir);
9815 generateAllInfoForClauses(Clauses: CurMapperDir->clauses(), CombinedInfo,
9816 OMPBuilder);
9817 }
9818
9819 /// Emit capture info for lambdas for variables captured by reference.
9820 void generateInfoForLambdaCaptures(
9821 const ValueDecl *VD, llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo,
9822 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const {
9823 QualType VDType = VD->getType().getCanonicalType().getNonReferenceType();
9824 const auto *RD = VDType->getAsCXXRecordDecl();
9825 if (!RD || !RD->isLambda())
9826 return;
9827 Address VDAddr(Arg, CGF.ConvertTypeForMem(T: VDType),
9828 CGF.getContext().getDeclAlign(D: VD));
9829 LValue VDLVal = CGF.MakeAddrLValue(Addr: VDAddr, T: VDType);
9830 llvm::DenseMap<const ValueDecl *, FieldDecl *> Captures;
9831 FieldDecl *ThisCapture = nullptr;
9832 RD->getCaptureFields(Captures, ThisCapture);
9833 if (ThisCapture) {
9834 LValue ThisLVal =
9835 CGF.EmitLValueForFieldInitialization(Base: VDLVal, Field: ThisCapture);
9836 LValue ThisLValVal = CGF.EmitLValueForField(Base: VDLVal, Field: ThisCapture);
9837 LambdaPointers.try_emplace(Key: ThisLVal.getPointer(CGF),
9838 Args: VDLVal.getPointer(CGF));
9839 CombinedInfo.Exprs.push_back(Elt: VD);
9840 CombinedInfo.BasePointers.push_back(Elt: ThisLVal.getPointer(CGF));
9841 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
9842 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
9843 CombinedInfo.Pointers.push_back(Elt: ThisLValVal.getPointer(CGF));
9844 CombinedInfo.Sizes.push_back(
9845 Elt: CGF.Builder.CreateIntCast(V: CGF.getTypeSize(Ty: CGF.getContext().VoidPtrTy),
9846 DestTy: CGF.Int64Ty, /*isSigned=*/true));
9847 CombinedInfo.Types.push_back(
9848 Elt: OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ |
9849 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
9850 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
9851 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT);
9852 CombinedInfo.HasAttachPtr.push_back(Elt: false);
9853 CombinedInfo.Mappers.push_back(Elt: nullptr);
9854 }
9855 for (const LambdaCapture &LC : RD->captures()) {
9856 if (!LC.capturesVariable())
9857 continue;
9858 const VarDecl *VD = cast<VarDecl>(Val: LC.getCapturedVar());
9859 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType())
9860 continue;
9861 auto It = Captures.find(Val: VD);
9862 assert(It != Captures.end() && "Found lambda capture without field.");
9863 LValue VarLVal = CGF.EmitLValueForFieldInitialization(Base: VDLVal, Field: It->second);
9864 if (LC.getCaptureKind() == LCK_ByRef) {
9865 LValue VarLValVal = CGF.EmitLValueForField(Base: VDLVal, Field: It->second);
9866 LambdaPointers.try_emplace(Key: VarLVal.getPointer(CGF),
9867 Args: VDLVal.getPointer(CGF));
9868 CombinedInfo.Exprs.push_back(Elt: VD);
9869 CombinedInfo.BasePointers.push_back(Elt: VarLVal.getPointer(CGF));
9870 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
9871 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
9872 CombinedInfo.Pointers.push_back(Elt: VarLValVal.getPointer(CGF));
9873 CombinedInfo.Sizes.push_back(Elt: CGF.Builder.CreateIntCast(
9874 V: CGF.getTypeSize(
9875 Ty: VD->getType().getCanonicalType().getNonReferenceType()),
9876 DestTy: CGF.Int64Ty, /*isSigned=*/true));
9877 } else {
9878 RValue VarRVal = CGF.EmitLoadOfLValue(V: VarLVal, Loc: RD->getLocation());
9879 LambdaPointers.try_emplace(Key: VarLVal.getPointer(CGF),
9880 Args: VDLVal.getPointer(CGF));
9881 CombinedInfo.Exprs.push_back(Elt: VD);
9882 CombinedInfo.BasePointers.push_back(Elt: VarLVal.getPointer(CGF));
9883 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
9884 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
9885 CombinedInfo.Pointers.push_back(Elt: VarRVal.getScalarVal());
9886 CombinedInfo.Sizes.push_back(Elt: llvm::ConstantInt::get(Ty: CGF.Int64Ty, V: 0));
9887 }
9888 CombinedInfo.Types.push_back(
9889 Elt: OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ |
9890 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
9891 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
9892 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT);
9893 CombinedInfo.HasAttachPtr.push_back(Elt: false);
9894 CombinedInfo.Mappers.push_back(Elt: nullptr);
9895 }
9896 }
9897
9898 /// Set correct indices for lambdas captures.
9899 void adjustMemberOfForLambdaCaptures(
9900 llvm::OpenMPIRBuilder &OMPBuilder,
9901 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers,
9902 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
9903 MapFlagsArrayTy &Types) const {
9904 for (unsigned I = 0, E = Types.size(); I < E; ++I) {
9905 // Set correct member_of idx for all implicit lambda captures.
9906 if (Types[I] != (OpenMPOffloadMappingFlags::OMP_MAP_PTR_AND_OBJ |
9907 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
9908 OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF |
9909 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT))
9910 continue;
9911 llvm::Value *BasePtr = LambdaPointers.lookup(Val: BasePointers[I]);
9912 assert(BasePtr && "Unable to find base lambda address.");
9913 int TgtIdx = -1;
9914 for (unsigned J = I; J > 0; --J) {
9915 unsigned Idx = J - 1;
9916 if (Pointers[Idx] != BasePtr)
9917 continue;
9918 TgtIdx = Idx;
9919 break;
9920 }
9921 assert(TgtIdx != -1 && "Unable to find parent lambda.");
9922 // All other current entries will be MEMBER_OF the combined entry
9923 // (except for PTR_AND_OBJ entries which do not have a placeholder value
9924 // 0xFFFF in the MEMBER_OF field).
9925 OpenMPOffloadMappingFlags MemberOfFlag =
9926 OMPBuilder.getMemberOfFlag(Position: TgtIdx);
9927 OMPBuilder.setCorrectMemberOfFlag(Flags&: Types[I], MemberOfFlag);
9928 }
9929 }
9930
9931 /// Populate component lists for non-lambda captured variables from map,
9932 /// is_device_ptr and has_device_addr clause info.
9933 void populateComponentListsForNonLambdaCaptureFromClauses(
9934 const ValueDecl *VD, MapDataArrayTy &DeclComponentLists,
9935 SmallVectorImpl<
9936 SmallVector<OMPClauseMappableExprCommon::MappableComponent, 8>>
9937 &StorageForImplicitlyAddedComponentLists) const {
9938 if (VD && LambdasMap.count(Val: VD))
9939 return;
9940
9941 // For member fields list in is_device_ptr, store it in
9942 // DeclComponentLists for generating components info.
9943 static const OpenMPMapModifierKind Unknown = OMPC_MAP_MODIFIER_unknown;
9944 auto It = DevPointersMap.find(Val: VD);
9945 if (It != DevPointersMap.end())
9946 for (const auto &MCL : It->second)
9947 DeclComponentLists.emplace_back(Args: MCL, Args: OMPC_MAP_to, Args: Unknown,
9948 /*IsImpicit = */ Args: true, Args: nullptr,
9949 Args: nullptr);
9950 auto I = HasDevAddrsMap.find(Val: VD);
9951 if (I != HasDevAddrsMap.end())
9952 for (const auto &MCL : I->second)
9953 DeclComponentLists.emplace_back(Args: MCL, Args: OMPC_MAP_tofrom, Args: Unknown,
9954 /*IsImpicit = */ Args: true, Args: nullptr,
9955 Args: nullptr);
9956 assert(isa<const OMPExecutableDirective *>(CurDir) &&
9957 "Expect a executable directive");
9958 const auto *CurExecDir = cast<const OMPExecutableDirective *>(Val: CurDir);
9959 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
9960 const auto *EI = C->getVarRefs().begin();
9961 for (const auto L : C->decl_component_lists(VD)) {
9962 const ValueDecl *VDecl, *Mapper;
9963 // The Expression is not correct if the mapping is implicit
9964 const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr;
9965 OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
9966 std::tie(args&: VDecl, args&: Components, args&: Mapper) = L;
9967 assert(VDecl == VD && "We got information for the wrong declaration??");
9968 assert(!Components.empty() &&
9969 "Not expecting declaration with no component lists.");
9970 DeclComponentLists.emplace_back(Args&: Components, Args: C->getMapType(),
9971 Args: C->getMapTypeModifiers(),
9972 Args: C->isImplicit(), Args&: Mapper, Args&: E);
9973 ++EI;
9974 }
9975 }
9976
9977 // For the target construct, if there's a map with a base-pointer that's
9978 // a member of an implicitly captured struct, of the current class,
9979 // we need to emit an implicit map on the pointer.
9980 if (isOpenMPTargetExecutionDirective(DKind: CurExecDir->getDirectiveKind()))
9981 addImplicitMapForAttachPtrBaseIfMemberOfCapturedVD(
9982 CapturedVD: VD, DeclComponentLists, ComponentVectorStorage&: StorageForImplicitlyAddedComponentLists);
9983
9984 llvm::stable_sort(Range&: DeclComponentLists, C: [](const MapData &LHS,
9985 const MapData &RHS) {
9986 ArrayRef<OpenMPMapModifierKind> MapModifiers = std::get<2>(t: LHS);
9987 OpenMPMapClauseKind MapType = std::get<1>(t: RHS);
9988 bool HasPresent =
9989 llvm::is_contained(Range&: MapModifiers, Element: clang::OMPC_MAP_MODIFIER_present);
9990 bool HasAllocs = MapType == OMPC_MAP_alloc;
9991 MapModifiers = std::get<2>(t: RHS);
9992 MapType = std::get<1>(t: LHS);
9993 bool HasPresentR =
9994 llvm::is_contained(Range&: MapModifiers, Element: clang::OMPC_MAP_MODIFIER_present);
9995 bool HasAllocsR = MapType == OMPC_MAP_alloc;
9996 return (HasPresent && !HasPresentR) || (HasAllocs && !HasAllocsR);
9997 });
9998 }
9999
10000 /// On a target construct, if there's an implicit map on a struct, or that of
10001 /// this[:], and an explicit map with a member of that struct/class as the
10002 /// base-pointer, we need to make sure that base-pointer is implicitly mapped,
10003 /// to make sure we don't map the full struct/class. For example:
10004 ///
10005 /// \code
10006 /// struct S {
10007 /// int dummy[10000];
10008 /// int *p;
10009 /// void f1() {
10010 /// #pragma omp target map(p[0:1])
10011 /// (void)this;
10012 /// }
10013 /// }; S s;
10014 ///
10015 /// void f2() {
10016 /// #pragma omp target map(s.p[0:10])
10017 /// (void)s;
10018 /// }
10019 /// \endcode
10020 ///
10021 /// Only `this-p` and `s.p` should be mapped in the two cases above.
10022 //
10023 // OpenMP 6.0: 7.9.6 map clause, pg 285
10024 // If a list item with an implicitly determined data-mapping attribute does
10025 // not have any corresponding storage in the device data environment prior to
10026 // a task encountering the construct associated with the map clause, and one
10027 // or more contiguous parts of the original storage are either list items or
10028 // base pointers to list items that are explicitly mapped on the construct,
10029 // only those parts of the original storage will have corresponding storage in
10030 // the device data environment as a result of the map clauses on the
10031 // construct.
10032 void addImplicitMapForAttachPtrBaseIfMemberOfCapturedVD(
10033 const ValueDecl *CapturedVD, MapDataArrayTy &DeclComponentLists,
10034 SmallVectorImpl<
10035 SmallVector<OMPClauseMappableExprCommon::MappableComponent, 8>>
10036 &ComponentVectorStorage) const {
10037 bool IsThisCapture = CapturedVD == nullptr;
10038
10039 for (const auto &ComponentsAndAttachPtr : AttachPtrExprMap) {
10040 OMPClauseMappableExprCommon::MappableExprComponentListRef
10041 ComponentsWithAttachPtr = ComponentsAndAttachPtr.first;
10042 const Expr *AttachPtrExpr = ComponentsAndAttachPtr.second;
10043 if (!AttachPtrExpr)
10044 continue;
10045
10046 const auto *ME = dyn_cast<MemberExpr>(Val: AttachPtrExpr);
10047 if (!ME)
10048 continue;
10049
10050 const Expr *Base = ME->getBase()->IgnoreParenImpCasts();
10051
10052 // If we are handling a "this" capture, then we are looking for
10053 // attach-ptrs of form `this->p`, either explicitly or implicitly.
10054 if (IsThisCapture && !ME->isImplicitCXXThis() && !isa<CXXThisExpr>(Val: Base))
10055 continue;
10056
10057 if (!IsThisCapture && (!isa<DeclRefExpr>(Val: Base) ||
10058 cast<DeclRefExpr>(Val: Base)->getDecl() != CapturedVD))
10059 continue;
10060
10061 // For non-this captures, we are looking for attach-ptrs of form
10062 // `s.p`.
10063 // For non-this captures, we are looking for attach-ptrs like `s.p`.
10064 if (!IsThisCapture && (ME->isArrow() || !isa<DeclRefExpr>(Val: Base) ||
10065 cast<DeclRefExpr>(Val: Base)->getDecl() != CapturedVD))
10066 continue;
10067
10068 // Check if we have an existing map on either:
10069 // this[:], s, this->p, or s.p, in which case, we don't need to add
10070 // an implicit one for the attach-ptr s.p/this->p.
10071 bool FoundExistingMap = false;
10072 for (const MapData &ExistingL : DeclComponentLists) {
10073 OMPClauseMappableExprCommon::MappableExprComponentListRef
10074 ExistingComponents = std::get<0>(t: ExistingL);
10075
10076 if (ExistingComponents.empty())
10077 continue;
10078
10079 // First check if we have a map like map(this->p) or map(s.p).
10080 const auto &FirstComponent = ExistingComponents.front();
10081 const Expr *FirstExpr = FirstComponent.getAssociatedExpression();
10082
10083 if (!FirstExpr)
10084 continue;
10085
10086 // First check if we have a map like map(this->p) or map(s.p).
10087 if (AttachPtrComparator.areEqual(LHS: FirstExpr, RHS: AttachPtrExpr)) {
10088 FoundExistingMap = true;
10089 break;
10090 }
10091
10092 // Check if we have a map like this[0:1]
10093 if (IsThisCapture) {
10094 if (const auto *OASE = dyn_cast<ArraySectionExpr>(Val: FirstExpr)) {
10095 if (isa<CXXThisExpr>(Val: OASE->getBase()->IgnoreParenImpCasts())) {
10096 FoundExistingMap = true;
10097 break;
10098 }
10099 }
10100 continue;
10101 }
10102
10103 // When the attach-ptr is something like `s.p`, check if
10104 // `s` itself is mapped explicitly.
10105 if (const auto *DRE = dyn_cast<DeclRefExpr>(Val: FirstExpr)) {
10106 if (DRE->getDecl() == CapturedVD) {
10107 FoundExistingMap = true;
10108 break;
10109 }
10110 }
10111 }
10112
10113 if (FoundExistingMap)
10114 continue;
10115
10116 // If no base map is found, we need to create an implicit map for the
10117 // attach-pointer expr.
10118
10119 ComponentVectorStorage.emplace_back();
10120 auto &AttachPtrComponents = ComponentVectorStorage.back();
10121
10122 static const OpenMPMapModifierKind Unknown = OMPC_MAP_MODIFIER_unknown;
10123 bool SeenAttachPtrComponent = false;
10124 // For creating a map on the attach-ptr `s.p/this->p`, we copy all
10125 // components from the component-list which has `s.p/this->p`
10126 // as the attach-ptr, starting from the component which matches
10127 // `s.p/this->p`. This way, we'll have component-lists of
10128 // `s.p` -> `s`, and `this->p` -> `this`.
10129 for (size_t i = 0; i < ComponentsWithAttachPtr.size(); ++i) {
10130 const auto &Component = ComponentsWithAttachPtr[i];
10131 const Expr *ComponentExpr = Component.getAssociatedExpression();
10132
10133 if (!SeenAttachPtrComponent && ComponentExpr != AttachPtrExpr)
10134 continue;
10135 SeenAttachPtrComponent = true;
10136
10137 AttachPtrComponents.emplace_back(Args: Component.getAssociatedExpression(),
10138 Args: Component.getAssociatedDeclaration(),
10139 Args: Component.isNonContiguous());
10140 }
10141 assert(!AttachPtrComponents.empty() &&
10142 "Could not populate component-lists for mapping attach-ptr");
10143
10144 DeclComponentLists.emplace_back(
10145 Args&: AttachPtrComponents, Args: OMPC_MAP_tofrom, Args: Unknown,
10146 /*IsImplicit=*/Args: true, /*mapper=*/Args: nullptr, Args&: AttachPtrExpr);
10147 }
10148 }
10149
10150 /// For a capture that has an associated clause, generate the base pointers,
10151 /// section pointers, sizes, map types, and mappers (all included in
10152 /// \a CurCaptureVarInfo).
10153 void generateInfoForCaptureFromClauseInfo(
10154 const MapDataArrayTy &DeclComponentListsFromClauses,
10155 const CapturedStmt::Capture *Cap, llvm::Value *Arg,
10156 MapCombinedInfoTy &CurCaptureVarInfo, llvm::OpenMPIRBuilder &OMPBuilder,
10157 unsigned OffsetForMemberOfFlag) const {
10158 assert(!Cap->capturesVariableArrayType() &&
10159 "Not expecting to generate map info for a variable array type!");
10160
10161 // We need to know when we generating information for the first component
10162 const ValueDecl *VD = Cap->capturesThis()
10163 ? nullptr
10164 : Cap->getCapturedVar()->getCanonicalDecl();
10165
10166 // for map(to: lambda): skip here, processing it in
10167 // generateDefaultMapInfo
10168 if (LambdasMap.count(Val: VD))
10169 return;
10170
10171 // If this declaration appears in a is_device_ptr clause we just have to
10172 // pass the pointer by value. If it is a reference to a declaration, we just
10173 // pass its value.
10174 if (VD && (DevPointersMap.count(Val: VD) || HasDevAddrsMap.count(Val: VD))) {
10175 CurCaptureVarInfo.Exprs.push_back(Elt: VD);
10176 CurCaptureVarInfo.BasePointers.emplace_back(Args&: Arg);
10177 CurCaptureVarInfo.DevicePtrDecls.emplace_back(Args&: VD);
10178 CurCaptureVarInfo.DevicePointers.emplace_back(Args: DeviceInfoTy::Pointer);
10179 CurCaptureVarInfo.Pointers.push_back(Elt: Arg);
10180 CurCaptureVarInfo.Sizes.push_back(Elt: CGF.Builder.CreateIntCast(
10181 V: CGF.getTypeSize(Ty: CGF.getContext().VoidPtrTy), DestTy: CGF.Int64Ty,
10182 /*isSigned=*/true));
10183 CurCaptureVarInfo.Types.push_back(
10184 Elt: OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
10185 OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM);
10186 CurCaptureVarInfo.HasAttachPtr.push_back(Elt: false);
10187 CurCaptureVarInfo.Mappers.push_back(Elt: nullptr);
10188 return;
10189 }
10190
10191 auto GenerateInfoForComponentLists =
10192 [&](ArrayRef<MapData> DeclComponentListsFromClauses,
10193 bool IsEligibleForTargetParamFlag) {
10194 MapCombinedInfoTy CurInfoForComponentLists;
10195 StructRangeInfoTy PartialStruct;
10196 AttachInfoTy AttachInfo;
10197
10198 if (DeclComponentListsFromClauses.empty())
10199 return;
10200
10201 generateInfoForCaptureFromComponentLists(
10202 VD, DeclComponentLists: DeclComponentListsFromClauses, CurComponentListInfo&: CurInfoForComponentLists,
10203 PartialStruct, AttachInfo, IsListEligibleForTargetParamFlag: IsEligibleForTargetParamFlag);
10204
10205 // If there is an entry in PartialStruct it means we have a
10206 // struct with individual members mapped. Emit an extra combined
10207 // entry.
10208 if (PartialStruct.Base.isValid()) {
10209 CurCaptureVarInfo.append(CurInfo&: PartialStruct.PreliminaryMapData);
10210 emitCombinedEntry(
10211 CombinedInfo&: CurCaptureVarInfo, CurTypes&: CurInfoForComponentLists.Types,
10212 PartialStruct, AttachInfo, IsMapThis: Cap->capturesThis(), OMPBuilder,
10213 /*VD=*/nullptr, OffsetForMemberOfFlag,
10214 /*NotTargetParams*/ !IsEligibleForTargetParamFlag);
10215 }
10216
10217 // We do the appends to get the entries in the following order:
10218 // combined-entry -> individual-field-entries -> attach-entry,
10219 CurCaptureVarInfo.append(CurInfo&: CurInfoForComponentLists);
10220 if (AttachInfo.isValid())
10221 emitAttachEntry(CGF, CombinedInfo&: CurCaptureVarInfo, AttachInfo);
10222 };
10223
10224 // Group component lists by their AttachPtrExpr and process them in order
10225 // of increasing complexity (nullptr first, then simple expressions like p,
10226 // then more complex ones like p[0], etc.)
10227 //
10228 // This ensure that we:
10229 // * handle maps that can contribute towards setting the kernel argument,
10230 // (e.g. map(ps), or map(ps[0])), before any that cannot (e.g. ps->pt->d).
10231 // * allocate a single contiguous storage for all exprs with the same
10232 // captured var and having the same attach-ptr.
10233 //
10234 // Example: The map clauses below should be handled grouped together based
10235 // on their attachable-base-pointers:
10236 // map-clause | attachable-base-pointer
10237 // --------------------------+------------------------
10238 // map(p, ps) | nullptr
10239 // map(p[0]) | p
10240 // map(p[0]->b, p[0]->c) | p[0]
10241 // map(ps->d, ps->e, ps->pt) | ps
10242 // map(ps->pt->d, ps->pt->e) | ps->pt
10243
10244 // First, collect all MapData entries with their attach-ptr exprs.
10245 SmallVector<std::pair<const Expr *, MapData>, 16> AttachPtrMapDataPairs;
10246
10247 for (const MapData &L : DeclComponentListsFromClauses) {
10248 OMPClauseMappableExprCommon::MappableExprComponentListRef Components =
10249 std::get<0>(t: L);
10250 const Expr *AttachPtrExpr = getAttachPtrExpr(Components);
10251 AttachPtrMapDataPairs.emplace_back(Args&: AttachPtrExpr, Args: L);
10252 }
10253
10254 // Next, sort by increasing order of their complexity.
10255 llvm::stable_sort(Range&: AttachPtrMapDataPairs,
10256 C: [this](const auto &LHS, const auto &RHS) {
10257 return AttachPtrComparator(LHS.first, RHS.first);
10258 });
10259
10260 bool NoDefaultMappingDoneForVD = CurCaptureVarInfo.BasePointers.empty();
10261 bool IsFirstGroup = true;
10262
10263 // And finally, process them all in order, grouping those with
10264 // equivalent attach-ptr exprs together.
10265 auto *It = AttachPtrMapDataPairs.begin();
10266 while (It != AttachPtrMapDataPairs.end()) {
10267 const Expr *AttachPtrExpr = It->first;
10268
10269 MapDataArrayTy GroupLists;
10270 while (It != AttachPtrMapDataPairs.end() &&
10271 (It->first == AttachPtrExpr ||
10272 AttachPtrComparator.areEqual(LHS: It->first, RHS: AttachPtrExpr))) {
10273 GroupLists.push_back(Elt: It->second);
10274 ++It;
10275 }
10276 assert(!GroupLists.empty() && "GroupLists should not be empty");
10277
10278 // Determine if this group of component-lists is eligible for TARGET_PARAM
10279 // flag. Only the first group processed should be eligible, and only if no
10280 // default mapping was done.
10281 bool IsEligibleForTargetParamFlag =
10282 IsFirstGroup && NoDefaultMappingDoneForVD;
10283
10284 GenerateInfoForComponentLists(GroupLists, IsEligibleForTargetParamFlag);
10285 IsFirstGroup = false;
10286 }
10287 }
10288
10289 /// Generate the base pointers, section pointers, sizes, map types, and
10290 /// mappers associated to \a DeclComponentLists for a given capture
10291 /// \a VD (all included in \a CurComponentListInfo).
10292 void generateInfoForCaptureFromComponentLists(
10293 const ValueDecl *VD, ArrayRef<MapData> DeclComponentLists,
10294 MapCombinedInfoTy &CurComponentListInfo, StructRangeInfoTy &PartialStruct,
10295 AttachInfoTy &AttachInfo, bool IsListEligibleForTargetParamFlag) const {
10296 // Find overlapping elements (including the offset from the base element).
10297 llvm::SmallDenseMap<
10298 const MapData *,
10299 llvm::SmallVector<
10300 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>,
10301 4>
10302 OverlappedData;
10303 size_t Count = 0;
10304 for (const MapData &L : DeclComponentLists) {
10305 OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
10306 OpenMPMapClauseKind MapType;
10307 ArrayRef<OpenMPMapModifierKind> MapModifiers;
10308 bool IsImplicit;
10309 const ValueDecl *Mapper;
10310 const Expr *VarRef;
10311 std::tie(args&: Components, args&: MapType, args&: MapModifiers, args&: IsImplicit, args&: Mapper, args&: VarRef) =
10312 L;
10313 ++Count;
10314 for (const MapData &L1 : ArrayRef(DeclComponentLists).slice(N: Count)) {
10315 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1;
10316 std::tie(args&: Components1, args&: MapType, args&: MapModifiers, args&: IsImplicit, args&: Mapper,
10317 args&: VarRef) = L1;
10318 auto CI = Components.rbegin();
10319 auto CE = Components.rend();
10320 auto SI = Components1.rbegin();
10321 auto SE = Components1.rend();
10322 for (; CI != CE && SI != SE; ++CI, ++SI) {
10323 if (CI->getAssociatedExpression()->getStmtClass() !=
10324 SI->getAssociatedExpression()->getStmtClass())
10325 break;
10326 // Are we dealing with different variables/fields?
10327 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration())
10328 break;
10329 }
10330 // Found overlapping if, at least for one component, reached the head
10331 // of the components list.
10332 if (CI == CE || SI == SE) {
10333 // Ignore it if it is the same component.
10334 if (CI == CE && SI == SE)
10335 continue;
10336 const auto It = (SI == SE) ? CI : SI;
10337 // If one component is a pointer and another one is a kind of
10338 // dereference of this pointer (array subscript, section, dereference,
10339 // etc.), it is not an overlapping.
10340 // Same, if one component is a base and another component is a
10341 // dereferenced pointer memberexpr with the same base.
10342 if (!isa<MemberExpr>(Val: It->getAssociatedExpression()) ||
10343 (std::prev(x: It)->getAssociatedDeclaration() &&
10344 std::prev(x: It)
10345 ->getAssociatedDeclaration()
10346 ->getType()
10347 ->isPointerType()) ||
10348 (It->getAssociatedDeclaration() &&
10349 It->getAssociatedDeclaration()->getType()->isPointerType() &&
10350 std::next(x: It) != CE && std::next(x: It) != SE))
10351 continue;
10352 const MapData &BaseData = CI == CE ? L : L1;
10353 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData =
10354 SI == SE ? Components : Components1;
10355 OverlappedData[&BaseData].push_back(Elt: SubData);
10356 }
10357 }
10358 }
10359 // Sort the overlapped elements for each item.
10360 llvm::SmallVector<const FieldDecl *, 4> Layout;
10361 if (!OverlappedData.empty()) {
10362 const Type *BaseType = VD->getType().getCanonicalType().getTypePtr();
10363 const Type *OrigType = BaseType->getPointeeOrArrayElementType();
10364 while (BaseType != OrigType) {
10365 BaseType = OrigType->getCanonicalTypeInternal().getTypePtr();
10366 OrigType = BaseType->getPointeeOrArrayElementType();
10367 }
10368
10369 if (const auto *CRD = BaseType->getAsCXXRecordDecl())
10370 getPlainLayout(RD: CRD, Layout, /*AsBase=*/false);
10371 else {
10372 const auto *RD = BaseType->getAsRecordDecl();
10373 Layout.append(in_start: RD->field_begin(), in_end: RD->field_end());
10374 }
10375 }
10376 for (auto &Pair : OverlappedData) {
10377 llvm::stable_sort(
10378 Range&: Pair.getSecond(),
10379 C: [&Layout](
10380 OMPClauseMappableExprCommon::MappableExprComponentListRef First,
10381 OMPClauseMappableExprCommon::MappableExprComponentListRef
10382 Second) {
10383 auto CI = First.rbegin();
10384 auto CE = First.rend();
10385 auto SI = Second.rbegin();
10386 auto SE = Second.rend();
10387 for (; CI != CE && SI != SE; ++CI, ++SI) {
10388 if (CI->getAssociatedExpression()->getStmtClass() !=
10389 SI->getAssociatedExpression()->getStmtClass())
10390 break;
10391 // Are we dealing with different variables/fields?
10392 if (CI->getAssociatedDeclaration() !=
10393 SI->getAssociatedDeclaration())
10394 break;
10395 }
10396
10397 // Lists contain the same elements.
10398 if (CI == CE && SI == SE)
10399 return false;
10400
10401 // List with less elements is less than list with more elements.
10402 if (CI == CE || SI == SE)
10403 return CI == CE;
10404
10405 const auto *FD1 = cast<FieldDecl>(Val: CI->getAssociatedDeclaration());
10406 const auto *FD2 = cast<FieldDecl>(Val: SI->getAssociatedDeclaration());
10407 if (FD1->getParent() == FD2->getParent())
10408 return FD1->getFieldIndex() < FD2->getFieldIndex();
10409 const auto *It =
10410 llvm::find_if(Range&: Layout, P: [FD1, FD2](const FieldDecl *FD) {
10411 return FD == FD1 || FD == FD2;
10412 });
10413 return *It == FD1;
10414 });
10415 }
10416
10417 // Associated with a capture, because the mapping flags depend on it.
10418 // Go through all of the elements with the overlapped elements.
10419 bool AddTargetParamFlag = IsListEligibleForTargetParamFlag;
10420 MapCombinedInfoTy StructBaseCombinedInfo;
10421 for (const auto &Pair : OverlappedData) {
10422 const MapData &L = *Pair.getFirst();
10423 OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
10424 OpenMPMapClauseKind MapType;
10425 ArrayRef<OpenMPMapModifierKind> MapModifiers;
10426 bool IsImplicit;
10427 const ValueDecl *Mapper;
10428 const Expr *VarRef;
10429 std::tie(args&: Components, args&: MapType, args&: MapModifiers, args&: IsImplicit, args&: Mapper, args&: VarRef) =
10430 L;
10431 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
10432 OverlappedComponents = Pair.getSecond();
10433 generateInfoForComponentList(
10434 MapType, MapModifiers, MotionModifiers: {}, Components, CombinedInfo&: CurComponentListInfo,
10435 StructBaseCombinedInfo, PartialStruct, AttachInfo, IsFirstComponentList: AddTargetParamFlag,
10436 IsImplicit, /*GenerateAllInfoForClauses*/ false, Mapper,
10437 /*ForDeviceAddr=*/false, BaseDecl: VD, MapExpr: VarRef, OverlappedElements: OverlappedComponents);
10438 AddTargetParamFlag = false;
10439 }
10440 // Go through other elements without overlapped elements.
10441 for (const MapData &L : DeclComponentLists) {
10442 OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
10443 OpenMPMapClauseKind MapType;
10444 ArrayRef<OpenMPMapModifierKind> MapModifiers;
10445 bool IsImplicit;
10446 const ValueDecl *Mapper;
10447 const Expr *VarRef;
10448 std::tie(args&: Components, args&: MapType, args&: MapModifiers, args&: IsImplicit, args&: Mapper, args&: VarRef) =
10449 L;
10450 auto It = OverlappedData.find(Val: &L);
10451 if (It == OverlappedData.end())
10452 generateInfoForComponentList(
10453 MapType, MapModifiers, MotionModifiers: {}, Components, CombinedInfo&: CurComponentListInfo,
10454 StructBaseCombinedInfo, PartialStruct, AttachInfo,
10455 IsFirstComponentList: AddTargetParamFlag, IsImplicit, /*GenerateAllInfoForClauses*/ false,
10456 Mapper, /*ForDeviceAddr=*/false, BaseDecl: VD, MapExpr: VarRef,
10457 /*OverlappedElements*/ {});
10458 AddTargetParamFlag = false;
10459 }
10460 }
10461
10462 /// Check if a variable should be treated as firstprivate due to explicit
10463 /// firstprivate clause or defaultmap(firstprivate:...).
10464 bool isEffectivelyFirstprivate(const VarDecl *VD, QualType Type) const {
10465 // Check explicit firstprivate clauses (not implicit from defaultmap)
10466 auto I = FirstPrivateDecls.find(Val: VD);
10467 if (I != FirstPrivateDecls.end() && !I->getSecond())
10468 return true; // Explicit firstprivate only
10469
10470 // Check defaultmap(firstprivate:scalar) for scalar types
10471 if (DefaultmapFirstprivateKinds.count(V: OMPC_DEFAULTMAP_scalar)) {
10472 if (Type->isScalarType())
10473 return true;
10474 }
10475
10476 // Check defaultmap(firstprivate:pointer) for pointer types
10477 if (DefaultmapFirstprivateKinds.count(V: OMPC_DEFAULTMAP_pointer)) {
10478 if (Type->isAnyPointerType())
10479 return true;
10480 }
10481
10482 // Check defaultmap(firstprivate:aggregate) for aggregate types
10483 if (DefaultmapFirstprivateKinds.count(V: OMPC_DEFAULTMAP_aggregate)) {
10484 if (Type->isAggregateType())
10485 return true;
10486 }
10487
10488 // Check defaultmap(firstprivate:all) for all types
10489 return DefaultmapFirstprivateKinds.count(V: OMPC_DEFAULTMAP_all);
10490 }
10491
10492 /// Generate the default map information for a given capture \a CI,
10493 /// record field declaration \a RI and captured value \a CV.
10494 void generateDefaultMapInfo(const CapturedStmt::Capture &CI,
10495 const FieldDecl &RI, llvm::Value *CV,
10496 MapCombinedInfoTy &CombinedInfo) const {
10497 bool IsImplicit = true;
10498 // Do the default mapping.
10499 if (CI.capturesThis()) {
10500 CombinedInfo.Exprs.push_back(Elt: nullptr);
10501 CombinedInfo.BasePointers.push_back(Elt: CV);
10502 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
10503 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
10504 CombinedInfo.Pointers.push_back(Elt: CV);
10505 const auto *PtrTy = cast<PointerType>(Val: RI.getType().getTypePtr());
10506 CombinedInfo.Sizes.push_back(
10507 Elt: CGF.Builder.CreateIntCast(V: CGF.getTypeSize(Ty: PtrTy->getPointeeType()),
10508 DestTy: CGF.Int64Ty, /*isSigned=*/true));
10509 // Default map type.
10510 CombinedInfo.Types.push_back(Elt: OpenMPOffloadMappingFlags::OMP_MAP_TO |
10511 OpenMPOffloadMappingFlags::OMP_MAP_FROM);
10512 } else if (CI.capturesVariableByCopy()) {
10513 const VarDecl *VD = CI.getCapturedVar();
10514 CombinedInfo.Exprs.push_back(Elt: VD->getCanonicalDecl());
10515 CombinedInfo.BasePointers.push_back(Elt: CV);
10516 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
10517 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
10518 CombinedInfo.Pointers.push_back(Elt: CV);
10519 bool IsFirstprivate =
10520 isEffectivelyFirstprivate(VD, Type: RI.getType().getNonReferenceType());
10521
10522 if (!RI.getType()->isAnyPointerType()) {
10523 // We have to signal to the runtime captures passed by value that are
10524 // not pointers.
10525 CombinedInfo.Types.push_back(
10526 Elt: OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
10527 CombinedInfo.Sizes.push_back(Elt: CGF.Builder.CreateIntCast(
10528 V: CGF.getTypeSize(Ty: RI.getType()), DestTy: CGF.Int64Ty, /*isSigned=*/true));
10529 } else if (IsFirstprivate) {
10530 // Firstprivate pointers should be passed by value (as literals)
10531 // without performing a present table lookup at runtime.
10532 CombinedInfo.Types.push_back(
10533 Elt: OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
10534 // Use zero size for pointer literals (just passing the pointer value)
10535 CombinedInfo.Sizes.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.Int64Ty));
10536 } else {
10537 // Pointers are implicitly mapped with a zero size and no flags
10538 // (other than first map that is added for all implicit maps).
10539 CombinedInfo.Types.push_back(Elt: OpenMPOffloadMappingFlags::OMP_MAP_NONE);
10540 CombinedInfo.Sizes.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.Int64Ty));
10541 }
10542 auto I = FirstPrivateDecls.find(Val: VD);
10543 if (I != FirstPrivateDecls.end())
10544 IsImplicit = I->getSecond();
10545 } else {
10546 assert(CI.capturesVariable() && "Expected captured reference.");
10547 const auto *PtrTy = cast<ReferenceType>(Val: RI.getType().getTypePtr());
10548 QualType ElementType = PtrTy->getPointeeType();
10549 const VarDecl *VD = CI.getCapturedVar();
10550 bool IsFirstprivate = isEffectivelyFirstprivate(VD, Type: ElementType);
10551 CombinedInfo.Exprs.push_back(Elt: VD->getCanonicalDecl());
10552 CombinedInfo.BasePointers.push_back(Elt: CV);
10553 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
10554 CombinedInfo.DevicePointers.push_back(Elt: DeviceInfoTy::None);
10555
10556 // For firstprivate pointers, pass by value instead of dereferencing
10557 if (IsFirstprivate && ElementType->isAnyPointerType()) {
10558 // Treat as a literal value (pass the pointer value itself)
10559 CombinedInfo.Pointers.push_back(Elt: CV);
10560 // Use zero size for pointer literals
10561 CombinedInfo.Sizes.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.Int64Ty));
10562 CombinedInfo.Types.push_back(
10563 Elt: OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
10564 } else {
10565 CombinedInfo.Sizes.push_back(Elt: CGF.Builder.CreateIntCast(
10566 V: CGF.getTypeSize(Ty: ElementType), DestTy: CGF.Int64Ty, /*isSigned=*/true));
10567 // The default map type for a scalar/complex type is 'to' because by
10568 // default the value doesn't have to be retrieved. For an aggregate
10569 // type, the default is 'tofrom'.
10570 CombinedInfo.Types.push_back(Elt: getMapModifiersForPrivateClauses(Cap: CI));
10571 CombinedInfo.Pointers.push_back(Elt: CV);
10572 }
10573 auto I = FirstPrivateDecls.find(Val: VD);
10574 if (I != FirstPrivateDecls.end())
10575 IsImplicit = I->getSecond();
10576 }
10577 // Every default map produces a single argument which is a target parameter.
10578 CombinedInfo.Types.back() |=
10579 OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM;
10580
10581 // Add flag stating this is an implicit map.
10582 if (IsImplicit)
10583 CombinedInfo.Types.back() |= OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT;
10584
10585 CombinedInfo.HasAttachPtr.push_back(Elt: false);
10586 // No user-defined mapper for default mapping.
10587 CombinedInfo.Mappers.push_back(Elt: nullptr);
10588 }
10589};
10590} // anonymous namespace
10591
10592// Try to extract the base declaration from a `this->x` expression if possible.
10593static ValueDecl *getDeclFromThisExpr(const Expr *E) {
10594 if (!E)
10595 return nullptr;
10596
10597 if (const auto *OASE = dyn_cast<ArraySectionExpr>(Val: E->IgnoreParenCasts()))
10598 if (const MemberExpr *ME =
10599 dyn_cast<MemberExpr>(Val: OASE->getBase()->IgnoreParenImpCasts()))
10600 return ME->getMemberDecl();
10601 return nullptr;
10602}
10603
10604/// Emit a string constant containing the names of the values mapped to the
10605/// offloading runtime library.
10606static llvm::Constant *
10607emitMappingInformation(CodeGenFunction &CGF, llvm::OpenMPIRBuilder &OMPBuilder,
10608 MappableExprsHandler::MappingExprInfo &MapExprs) {
10609
10610 uint32_t SrcLocStrSize;
10611 if (!MapExprs.getMapDecl() && !MapExprs.getMapExpr())
10612 return OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize);
10613
10614 SourceLocation Loc;
10615 if (!MapExprs.getMapDecl() && MapExprs.getMapExpr()) {
10616 if (const ValueDecl *VD = getDeclFromThisExpr(E: MapExprs.getMapExpr()))
10617 Loc = VD->getLocation();
10618 else
10619 Loc = MapExprs.getMapExpr()->getExprLoc();
10620 } else {
10621 Loc = MapExprs.getMapDecl()->getLocation();
10622 }
10623
10624 std::string ExprName;
10625 if (MapExprs.getMapExpr()) {
10626 PrintingPolicy P(CGF.getContext().getLangOpts());
10627 llvm::raw_string_ostream OS(ExprName);
10628 MapExprs.getMapExpr()->printPretty(OS, Helper: nullptr, Policy: P);
10629 } else {
10630 ExprName = MapExprs.getMapDecl()->getNameAsString();
10631 }
10632
10633 std::string FileName;
10634 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
10635 if (auto *DbgInfo = CGF.getDebugInfo())
10636 FileName = DbgInfo->remapDIPath(PLoc.getFilename());
10637 else
10638 FileName = PLoc.getFilename();
10639 return OMPBuilder.getOrCreateSrcLocStr(FunctionName: FileName, FileName: ExprName, Line: PLoc.getLine(),
10640 Column: PLoc.getColumn(), SrcLocStrSize);
10641}
10642/// Emit the arrays used to pass the captures and map information to the
10643/// offloading runtime library. If there is no map or capture information,
10644/// return nullptr by reference.
10645static void emitOffloadingArraysAndArgs(
10646 CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo,
10647 CGOpenMPRuntime::TargetDataInfo &Info, llvm::OpenMPIRBuilder &OMPBuilder,
10648 bool IsNonContiguous = false, bool ForEndCall = false) {
10649 CodeGenModule &CGM = CGF.CGM;
10650
10651 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
10652 InsertPointTy AllocaIP(CGF.AllocaInsertPt->getIterator());
10653 InsertPointTy CodeGenIP(CGF.Builder.GetInsertPoint());
10654
10655 auto DeviceAddrCB = [&](unsigned int I, llvm::Value *NewDecl) {
10656 if (const ValueDecl *DevVD = CombinedInfo.DevicePtrDecls[I]) {
10657 Info.CaptureDeviceAddrMap.try_emplace(Key: DevVD, Args&: NewDecl);
10658 }
10659 };
10660
10661 auto CustomMapperCB = [&](unsigned int I) {
10662 llvm::Function *MFunc = nullptr;
10663 if (CombinedInfo.Mappers[I]) {
10664 Info.HasMapper = true;
10665 MFunc = CGM.getOpenMPRuntime().getOrCreateUserDefinedMapperFunc(
10666 D: cast<OMPDeclareMapperDecl>(Val: CombinedInfo.Mappers[I]));
10667 }
10668 return MFunc;
10669 };
10670 cantFail(Err: OMPBuilder.emitOffloadingArraysAndArgs(
10671 AllocaIP, CodeGenIP, Info, RTArgs&: Info.RTArgs, CombinedInfo, CustomMapperCB,
10672 IsNonContiguous, ForEndCall, DeviceAddrCB));
10673}
10674
10675/// Check for inner distribute directive.
10676static const OMPExecutableDirective *
10677getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) {
10678 const auto *CS = D.getInnermostCapturedStmt();
10679 const auto *Body =
10680 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
10681 const Stmt *ChildStmt =
10682 CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
10683
10684 if (const auto *NestedDir =
10685 dyn_cast_or_null<OMPExecutableDirective>(Val: ChildStmt)) {
10686 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind();
10687 switch (D.getDirectiveKind()) {
10688 case OMPD_target:
10689 // For now, treat 'target' with nested 'teams loop' as if it's
10690 // distributed (target teams distribute).
10691 if (isOpenMPDistributeDirective(DKind) || DKind == OMPD_teams_loop)
10692 return NestedDir;
10693 if (DKind == OMPD_teams) {
10694 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers(
10695 /*IgnoreCaptured=*/true);
10696 if (!Body)
10697 return nullptr;
10698 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
10699 if (const auto *NND =
10700 dyn_cast_or_null<OMPExecutableDirective>(Val: ChildStmt)) {
10701 DKind = NND->getDirectiveKind();
10702 if (isOpenMPDistributeDirective(DKind))
10703 return NND;
10704 }
10705 }
10706 return nullptr;
10707 case OMPD_target_teams:
10708 if (isOpenMPDistributeDirective(DKind))
10709 return NestedDir;
10710 return nullptr;
10711 case OMPD_target_parallel:
10712 case OMPD_target_simd:
10713 case OMPD_target_parallel_for:
10714 case OMPD_target_parallel_for_simd:
10715 return nullptr;
10716 case OMPD_target_teams_distribute:
10717 case OMPD_target_teams_distribute_simd:
10718 case OMPD_target_teams_distribute_parallel_for:
10719 case OMPD_target_teams_distribute_parallel_for_simd:
10720 case OMPD_parallel:
10721 case OMPD_for:
10722 case OMPD_parallel_for:
10723 case OMPD_parallel_master:
10724 case OMPD_parallel_sections:
10725 case OMPD_for_simd:
10726 case OMPD_parallel_for_simd:
10727 case OMPD_cancel:
10728 case OMPD_cancellation_point:
10729 case OMPD_ordered_standalone:
10730 case OMPD_ordered_blockassoc:
10731 case OMPD_threadprivate:
10732 case OMPD_allocate:
10733 case OMPD_task:
10734 case OMPD_simd:
10735 case OMPD_tile:
10736 case OMPD_unroll:
10737 case OMPD_sections:
10738 case OMPD_section:
10739 case OMPD_single:
10740 case OMPD_master:
10741 case OMPD_critical:
10742 case OMPD_taskyield:
10743 case OMPD_barrier:
10744 case OMPD_taskwait:
10745 case OMPD_taskgroup:
10746 case OMPD_atomic:
10747 case OMPD_flush:
10748 case OMPD_depobj:
10749 case OMPD_scan:
10750 case OMPD_teams:
10751 case OMPD_target_data:
10752 case OMPD_target_exit_data:
10753 case OMPD_target_enter_data:
10754 case OMPD_distribute:
10755 case OMPD_distribute_simd:
10756 case OMPD_distribute_parallel_for:
10757 case OMPD_distribute_parallel_for_simd:
10758 case OMPD_teams_distribute:
10759 case OMPD_teams_distribute_simd:
10760 case OMPD_teams_distribute_parallel_for:
10761 case OMPD_teams_distribute_parallel_for_simd:
10762 case OMPD_target_update:
10763 case OMPD_declare_simd:
10764 case OMPD_declare_variant:
10765 case OMPD_begin_declare_variant:
10766 case OMPD_end_declare_variant:
10767 case OMPD_declare_target:
10768 case OMPD_end_declare_target:
10769 case OMPD_declare_reduction:
10770 case OMPD_declare_mapper:
10771 case OMPD_taskloop:
10772 case OMPD_taskloop_simd:
10773 case OMPD_master_taskloop:
10774 case OMPD_master_taskloop_simd:
10775 case OMPD_parallel_master_taskloop:
10776 case OMPD_parallel_master_taskloop_simd:
10777 case OMPD_requires:
10778 case OMPD_metadirective:
10779 case OMPD_unknown:
10780 default:
10781 llvm_unreachable("Unexpected directive.");
10782 }
10783 }
10784
10785 return nullptr;
10786}
10787
10788/// Emit the user-defined mapper function. The code generation follows the
10789/// pattern in the example below.
10790/// \code
10791/// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle,
10792/// void *base, void *begin,
10793/// int64_t size, int64_t type,
10794/// void *name = nullptr) {
10795/// // Allocate space for an array section first.
10796/// if ((size > 1 || (base != begin)) && !maptype.IsDelete)
10797/// __tgt_push_mapper_component(rt_mapper_handle, base, begin,
10798/// size*sizeof(Ty), clearToFromMember(type));
10799/// // Map members.
10800/// for (unsigned i = 0; i < size; i++) {
10801/// N = __tgt_mapper_num_components(rt_mapper_handle);
10802/// // For each component specified by this mapper:
10803/// for (auto c : begin[i]->all_components) {
10804/// // MEMBER_OF grouping: tie this component to the current array element
10805/// // (component N) by adding N<<48. Exceptions:
10806/// // - ATTACH entries are not members of any struct storage range.
10807/// // - Pointee entries (reached via a pointer member) occupy separate
10808/// // storage; their inner MEMBER_OF bits are shifted by N instead.
10809/// if (c.isAttach() || c.isPointee())
10810/// member_type = c.arg_type + (c.hasInnerMemberOf() ? N<<48 : 0);
10811/// else
10812/// member_type = c.arg_type + N<<48;
10813/// // Map-type-modifying bits (ALWAYS, DELETE, CLOSE) from the outer map
10814/// // clause are propagated to each component, except ATTACH entries
10815/// // (ATTACH|ALWAYS is reserved for attach(always), and other modifier
10816/// // bits have no meaning for ATTACH). PRESENT is additionally
10817/// // propagated to components with HasAttachPtr (the pointee data) at
10818/// // OpenMP >= 6.0.
10819/// present_bit = (v60 && c.hasAttachPtr()) ? PRESENT : 0;
10820/// imported_modifier_bits =
10821/// type & (ALWAYS | DELETE | CLOSE | present_bit);
10822/// effective_type = c.isAttach() ? member_type
10823/// : member_type | imported_modifier_bits;
10824/// if (c.hasMapper())
10825/// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size,
10826/// effective_type, c.arg_name);
10827/// else
10828/// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base,
10829/// c.arg_begin, c.arg_size, effective_type,
10830/// c.arg_name);
10831/// }
10832/// }
10833/// // Delete the array section.
10834/// if (size > 1 && maptype.IsDelete)
10835/// __tgt_push_mapper_component(rt_mapper_handle, base, begin,
10836/// size*sizeof(Ty), clearToFromMember(type));
10837/// }
10838/// \endcode
10839void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D,
10840 CodeGenFunction *CGF) {
10841 if (UDMMap.count(Val: D) > 0)
10842 return;
10843 ASTContext &C = CGM.getContext();
10844 QualType Ty = D->getType();
10845 auto *MapperVarDecl =
10846 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: D->getMapperVarRef())->getDecl());
10847 CharUnits ElementSize = C.getTypeSizeInChars(T: Ty);
10848 llvm::Type *ElemTy = CGM.getTypes().ConvertTypeForMem(T: Ty);
10849
10850 CodeGenFunction MapperCGF(CGM);
10851 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
10852 auto PrivatizeAndGenMapInfoCB =
10853 [&](llvm::OpenMPIRBuilder::InsertPointTy CodeGenIP, llvm::Value *PtrPHI,
10854 llvm::Value *BeginArg) -> llvm::OpenMPIRBuilder::MapInfosTy & {
10855 MapperCGF.Builder.restoreIP(IP: CodeGenIP);
10856
10857 // Privatize the declared variable of mapper to be the current array
10858 // element.
10859 Address PtrCurrent(
10860 PtrPHI, ElemTy,
10861 Address(BeginArg, MapperCGF.VoidPtrTy, CGM.getPointerAlign())
10862 .getAlignment()
10863 .alignmentOfArrayElement(elementSize: ElementSize));
10864 CodeGenFunction::OMPPrivateScope Scope(MapperCGF);
10865 Scope.addPrivate(LocalVD: MapperVarDecl, Addr: PtrCurrent);
10866 (void)Scope.Privatize();
10867
10868 // Get map clause information.
10869 MappableExprsHandler MEHandler(*D, MapperCGF);
10870 MEHandler.generateAllInfoForMapper(CombinedInfo, OMPBuilder);
10871
10872 auto FillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) {
10873 return emitMappingInformation(CGF&: MapperCGF, OMPBuilder, MapExprs&: MapExpr);
10874 };
10875 if (CGM.getCodeGenOpts().getDebugInfo() !=
10876 llvm::codegenoptions::NoDebugInfo) {
10877 CombinedInfo.Names.resize(N: CombinedInfo.Exprs.size());
10878 llvm::transform(Range&: CombinedInfo.Exprs, d_first: CombinedInfo.Names.begin(),
10879 F: FillInfoMap);
10880 }
10881
10882 return CombinedInfo;
10883 };
10884
10885 auto CustomMapperCB = [&](unsigned I) {
10886 llvm::Function *MapperFunc = nullptr;
10887 if (CombinedInfo.Mappers[I]) {
10888 // Call the corresponding mapper function.
10889 MapperFunc = getOrCreateUserDefinedMapperFunc(
10890 D: cast<OMPDeclareMapperDecl>(Val: CombinedInfo.Mappers[I]));
10891 assert(MapperFunc && "Expect a valid mapper function is available.");
10892 }
10893 return MapperFunc;
10894 };
10895
10896 SmallString<64> TyStr;
10897 llvm::raw_svector_ostream Out(TyStr);
10898 CGM.getCXXABI().getMangleContext().mangleCanonicalTypeName(T: Ty, Out);
10899 std::string Name = getName(Parts: {"omp_mapper", TyStr, D->getName()});
10900
10901 // Propagate the PRESENT modifier to the pointee entries (those with
10902 // HasAttachPtr) only for OpenMP >= 6.0; before 6.0 the present modifier does
10903 // not apply to the pointee (see the OpenMP 6.0 erratum on the present motion
10904 // vs. map-type modifier divergence).
10905 bool PropagatePresentToPointee = CGM.getLangOpts().OpenMP >= 60;
10906 llvm::Function *NewFn = cantFail(ValOrErr: OMPBuilder.emitUserDefinedMapper(
10907 PrivAndGenMapInfoCB: PrivatizeAndGenMapInfoCB, ElemTy, FuncName: Name, CustomMapperCB,
10908 /*PreserveMemberOfFlags=*/false, PropagatePresentToPointee));
10909 UDMMap.try_emplace(Key: D, Args&: NewFn);
10910 if (CGF)
10911 FunctionUDMMap[CGF->CurFn].push_back(Elt: D);
10912}
10913
10914llvm::Function *CGOpenMPRuntime::getOrCreateUserDefinedMapperFunc(
10915 const OMPDeclareMapperDecl *D) {
10916 auto I = UDMMap.find(Val: D);
10917 if (I != UDMMap.end())
10918 return I->second;
10919 emitUserDefinedMapper(D);
10920 return UDMMap.lookup(Val: D);
10921}
10922
10923llvm::Value *CGOpenMPRuntime::emitTargetNumIterationsCall(
10924 CodeGenFunction &CGF, const OMPExecutableDirective &D,
10925 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
10926 const OMPLoopDirective &D)>
10927 SizeEmitter) {
10928 OpenMPDirectiveKind Kind = D.getDirectiveKind();
10929 const OMPExecutableDirective *TD = &D;
10930 // Get nested teams distribute kind directive, if any. For now, treat
10931 // 'target_teams_loop' as if it's really a target_teams_distribute.
10932 if ((!isOpenMPDistributeDirective(DKind: Kind) || !isOpenMPTeamsDirective(DKind: Kind)) &&
10933 Kind != OMPD_target_teams_loop)
10934 TD = getNestedDistributeDirective(Ctx&: CGM.getContext(), D);
10935 if (!TD)
10936 return llvm::ConstantInt::get(Ty: CGF.Int64Ty, V: 0);
10937
10938 const auto *LD = cast<OMPLoopDirective>(Val: TD);
10939 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD))
10940 return NumIterations;
10941 return llvm::ConstantInt::get(Ty: CGF.Int64Ty, V: 0);
10942}
10943
10944static void
10945emitTargetCallFallback(CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn,
10946 const OMPExecutableDirective &D,
10947 llvm::SmallVectorImpl<llvm::Value *> &CapturedVars,
10948 bool RequiresOuterTask, const CapturedStmt &CS,
10949 bool OffloadingMandatory, CodeGenFunction &CGF) {
10950 if (OffloadingMandatory) {
10951 CGF.Builder.CreateUnreachable();
10952 } else {
10953 if (RequiresOuterTask) {
10954 CapturedVars.clear();
10955 CGF.GenerateOpenMPCapturedVars(S: CS, CapturedVars);
10956 }
10957 llvm::SmallVector<llvm::Value *, 16> Args(CapturedVars.begin(),
10958 CapturedVars.end());
10959 Args.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.Builder.getPtrTy()));
10960 OMPRuntime->emitOutlinedFunctionCall(CGF, Loc: D.getBeginLoc(), OutlinedFn,
10961 Args);
10962 }
10963}
10964
10965static llvm::Value *emitDeviceID(
10966 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
10967 CodeGenFunction &CGF) {
10968 // Emit device ID if any.
10969 llvm::Value *DeviceID;
10970 if (Device.getPointer()) {
10971 assert((Device.getInt() == OMPC_DEVICE_unknown ||
10972 Device.getInt() == OMPC_DEVICE_device_num) &&
10973 "Expected device_num modifier.");
10974 llvm::Value *DevVal = CGF.EmitScalarExpr(E: Device.getPointer());
10975 DeviceID =
10976 CGF.Builder.CreateIntCast(V: DevVal, DestTy: CGF.Int64Ty, /*isSigned=*/true);
10977 } else {
10978 DeviceID = CGF.Builder.getInt64(C: OMP_DEVICEID_UNDEF);
10979 }
10980 return DeviceID;
10981}
10982
10983static std::pair<llvm::Value *, OMPDynGroupprivateFallbackType>
10984emitDynCGroupMem(const OMPExecutableDirective &D, CodeGenFunction &CGF) {
10985 llvm::Value *DynGP = CGF.Builder.getInt32(C: 0);
10986 auto DynGPFallback = OMPDynGroupprivateFallbackType::Abort;
10987
10988 if (auto *DynGPClause = D.getSingleClause<OMPDynGroupprivateClause>()) {
10989 CodeGenFunction::RunCleanupsScope DynGPScope(CGF);
10990 llvm::Value *DynGPVal =
10991 CGF.EmitScalarExpr(E: DynGPClause->getSize(), /*IgnoreResultAssign=*/true);
10992 DynGP = CGF.Builder.CreateIntCast(V: DynGPVal, DestTy: CGF.Int32Ty,
10993 /*isSigned=*/false);
10994 auto FallbackModifier = DynGPClause->getDynGroupprivateFallbackModifier();
10995 switch (FallbackModifier) {
10996 case OMPC_DYN_GROUPPRIVATE_FALLBACK_abort:
10997 DynGPFallback = OMPDynGroupprivateFallbackType::Abort;
10998 break;
10999 case OMPC_DYN_GROUPPRIVATE_FALLBACK_null:
11000 DynGPFallback = OMPDynGroupprivateFallbackType::Null;
11001 break;
11002 case OMPC_DYN_GROUPPRIVATE_FALLBACK_default_mem:
11003 case OMPC_DYN_GROUPPRIVATE_FALLBACK_unknown:
11004 // This is the default for dyn_groupprivate.
11005 DynGPFallback = OMPDynGroupprivateFallbackType::DefaultMem;
11006 break;
11007 default:
11008 llvm_unreachable("Unknown fallback modifier for OpenMP dyn_groupprivate");
11009 }
11010 } else if (auto *OMPXDynCGClause =
11011 D.getSingleClause<OMPXDynCGroupMemClause>()) {
11012 CodeGenFunction::RunCleanupsScope DynCGMemScope(CGF);
11013 llvm::Value *DynCGMemVal = CGF.EmitScalarExpr(E: OMPXDynCGClause->getSize(),
11014 /*IgnoreResultAssign=*/true);
11015 DynGP = CGF.Builder.CreateIntCast(V: DynCGMemVal, DestTy: CGF.Int32Ty,
11016 /*isSigned=*/false);
11017 }
11018 return {DynGP, DynGPFallback};
11019}
11020
11021static void genMapInfoForCaptures(
11022 MappableExprsHandler &MEHandler, CodeGenFunction &CGF,
11023 const CapturedStmt &CS, llvm::SmallVectorImpl<llvm::Value *> &CapturedVars,
11024 llvm::OpenMPIRBuilder &OMPBuilder,
11025 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &MappedVarSet,
11026 MappableExprsHandler::MapCombinedInfoTy &CombinedInfo) {
11027
11028 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers;
11029 auto RI = CS.getCapturedRecordDecl()->field_begin();
11030 auto *CV = CapturedVars.begin();
11031 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(),
11032 CE = CS.capture_end();
11033 CI != CE; ++CI, ++RI, ++CV) {
11034 MappableExprsHandler::MapCombinedInfoTy CurInfo;
11035
11036 // VLA sizes are passed to the outlined region by copy and do not have map
11037 // information associated.
11038 if (CI->capturesVariableArrayType()) {
11039 CurInfo.Exprs.push_back(Elt: nullptr);
11040 CurInfo.BasePointers.push_back(Elt: *CV);
11041 CurInfo.DevicePtrDecls.push_back(Elt: nullptr);
11042 CurInfo.DevicePointers.push_back(
11043 Elt: MappableExprsHandler::DeviceInfoTy::None);
11044 CurInfo.Pointers.push_back(Elt: *CV);
11045 CurInfo.Sizes.push_back(Elt: CGF.Builder.CreateIntCast(
11046 V: CGF.getTypeSize(Ty: RI->getType()), DestTy: CGF.Int64Ty, /*isSigned=*/true));
11047 // Copy to the device as an argument. No need to retrieve it.
11048 CurInfo.Types.push_back(Elt: OpenMPOffloadMappingFlags::OMP_MAP_LITERAL |
11049 OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM |
11050 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT);
11051 CurInfo.HasAttachPtr.push_back(Elt: false);
11052 CurInfo.Mappers.push_back(Elt: nullptr);
11053 } else {
11054 const ValueDecl *CapturedVD =
11055 CI->capturesThis() ? nullptr
11056 : CI->getCapturedVar()->getCanonicalDecl();
11057 bool HasEntryWithCVAsAttachPtr = false;
11058 if (CapturedVD)
11059 HasEntryWithCVAsAttachPtr =
11060 MEHandler.hasAttachEntryForCapturedVar(VD: CapturedVD);
11061
11062 // Populate component lists for the captured variable from clauses.
11063 MappableExprsHandler::MapDataArrayTy DeclComponentLists;
11064 SmallVector<
11065 SmallVector<OMPClauseMappableExprCommon::MappableComponent, 8>, 4>
11066 StorageForImplicitlyAddedComponentLists;
11067 MEHandler.populateComponentListsForNonLambdaCaptureFromClauses(
11068 VD: CapturedVD, DeclComponentLists,
11069 StorageForImplicitlyAddedComponentLists);
11070
11071 // OpenMP 6.0, 15.8, target construct, restrictions:
11072 // * A list item in a map clause that is specified on a target construct
11073 // must have a base variable or base pointer.
11074 //
11075 // Map clauses on a target construct must either have a base pointer, or a
11076 // base-variable. So, if we don't have a base-pointer, that means that it
11077 // must have a base-variable, i.e. we have a map like `map(s)`, `map(s.x)`
11078 // etc. In such cases, we do not need to handle default map generation
11079 // for `s`.
11080 bool HasEntryWithoutAttachPtr =
11081 llvm::any_of(Range&: DeclComponentLists, P: [&](const auto &MapData) {
11082 OMPClauseMappableExprCommon::MappableExprComponentListRef
11083 Components = std::get<0>(MapData);
11084 return !MEHandler.getAttachPtrExpr(Components);
11085 });
11086
11087 // Generate default map info first if there's no direct map with CV as
11088 // the base-variable, or attach pointer.
11089 if (DeclComponentLists.empty() ||
11090 (!HasEntryWithCVAsAttachPtr && !HasEntryWithoutAttachPtr))
11091 MEHandler.generateDefaultMapInfo(CI: *CI, RI: **RI, CV: *CV, CombinedInfo&: CurInfo);
11092
11093 // If we have any information in the map clause, we use it, otherwise we
11094 // just do a default mapping.
11095 MEHandler.generateInfoForCaptureFromClauseInfo(
11096 DeclComponentListsFromClauses: DeclComponentLists, Cap: CI, Arg: *CV, CurCaptureVarInfo&: CurInfo, OMPBuilder,
11097 /*OffsetForMemberOfFlag=*/CombinedInfo.BasePointers.size());
11098
11099 if (!CI->capturesThis())
11100 MappedVarSet.insert(V: CI->getCapturedVar());
11101 else
11102 MappedVarSet.insert(V: nullptr);
11103
11104 // Generate correct mapping for variables captured by reference in
11105 // lambdas.
11106 if (CI->capturesVariable())
11107 MEHandler.generateInfoForLambdaCaptures(VD: CI->getCapturedVar(), Arg: *CV,
11108 CombinedInfo&: CurInfo, LambdaPointers);
11109 }
11110 // We expect to have at least an element of information for this capture.
11111 assert(!CurInfo.BasePointers.empty() &&
11112 "Non-existing map pointer for capture!");
11113 assert(CurInfo.BasePointers.size() == CurInfo.Pointers.size() &&
11114 CurInfo.BasePointers.size() == CurInfo.Sizes.size() &&
11115 CurInfo.BasePointers.size() == CurInfo.Types.size() &&
11116 CurInfo.BasePointers.size() == CurInfo.Mappers.size() &&
11117 "Inconsistent map information sizes!");
11118
11119 // We need to append the results of this capture to what we already have.
11120 CombinedInfo.append(CurInfo);
11121 }
11122 // Adjust MEMBER_OF flags for the lambdas captures.
11123 MEHandler.adjustMemberOfForLambdaCaptures(
11124 OMPBuilder, LambdaPointers, BasePointers&: CombinedInfo.BasePointers,
11125 Pointers&: CombinedInfo.Pointers, Types&: CombinedInfo.Types);
11126}
11127static void
11128genMapInfo(MappableExprsHandler &MEHandler, CodeGenFunction &CGF,
11129 MappableExprsHandler::MapCombinedInfoTy &CombinedInfo,
11130 llvm::OpenMPIRBuilder &OMPBuilder,
11131 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkippedVarSet =
11132 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) {
11133
11134 CodeGenModule &CGM = CGF.CGM;
11135 // Map any list items in a map clause that were not captures because they
11136 // weren't referenced within the construct.
11137 MEHandler.generateAllInfo(CombinedInfo, OMPBuilder, SkipVarSet: SkippedVarSet);
11138
11139 auto FillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) {
11140 return emitMappingInformation(CGF, OMPBuilder, MapExprs&: MapExpr);
11141 };
11142 if (CGM.getCodeGenOpts().getDebugInfo() !=
11143 llvm::codegenoptions::NoDebugInfo) {
11144 CombinedInfo.Names.resize(N: CombinedInfo.Exprs.size());
11145 llvm::transform(Range&: CombinedInfo.Exprs, d_first: CombinedInfo.Names.begin(),
11146 F: FillInfoMap);
11147 }
11148}
11149
11150static void genMapInfo(const OMPExecutableDirective &D, CodeGenFunction &CGF,
11151 const CapturedStmt &CS,
11152 llvm::SmallVectorImpl<llvm::Value *> &CapturedVars,
11153 llvm::OpenMPIRBuilder &OMPBuilder,
11154 MappableExprsHandler::MapCombinedInfoTy &CombinedInfo) {
11155 // Get mappable expression information.
11156 MappableExprsHandler MEHandler(D, CGF);
11157 llvm::DenseSet<CanonicalDeclPtr<const Decl>> MappedVarSet;
11158
11159 genMapInfoForCaptures(MEHandler, CGF, CS, CapturedVars, OMPBuilder,
11160 MappedVarSet, CombinedInfo);
11161 genMapInfo(MEHandler, CGF, CombinedInfo, OMPBuilder, SkippedVarSet: MappedVarSet);
11162}
11163
11164template <typename ClauseTy>
11165static void
11166emitClauseForBareTargetDirective(CodeGenFunction &CGF,
11167 const OMPExecutableDirective &D,
11168 llvm::SmallVectorImpl<llvm::Value *> &Values) {
11169 const auto *C = D.getSingleClause<ClauseTy>();
11170 assert(!C->varlist_empty() &&
11171 "ompx_bare requires explicit num_teams and thread_limit");
11172 CodeGenFunction::RunCleanupsScope Scope(CGF);
11173 for (auto *E : C->varlist()) {
11174 llvm::Value *V = CGF.EmitScalarExpr(E);
11175 Values.push_back(
11176 Elt: CGF.Builder.CreateIntCast(V, DestTy: CGF.Int32Ty, /*isSigned=*/true));
11177 }
11178}
11179
11180static void emitTargetCallKernelLaunch(
11181 CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn,
11182 const OMPExecutableDirective &D,
11183 llvm::SmallVectorImpl<llvm::Value *> &CapturedVars, bool RequiresOuterTask,
11184 const CapturedStmt &CS, bool OffloadingMandatory,
11185 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
11186 llvm::Value *OutlinedFnID, CodeGenFunction::OMPTargetDataInfo &InputInfo,
11187 llvm::Value *&MapTypesArray, llvm::Value *&MapNamesArray,
11188 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
11189 const OMPLoopDirective &D)>
11190 SizeEmitter,
11191 CodeGenFunction &CGF, CodeGenModule &CGM) {
11192 llvm::OpenMPIRBuilder &OMPBuilder = OMPRuntime->getOMPBuilder();
11193
11194 // Fill up the arrays with all the captured variables.
11195 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
11196 CGOpenMPRuntime::TargetDataInfo Info;
11197 genMapInfo(D, CGF, CS, CapturedVars, OMPBuilder, CombinedInfo);
11198
11199 // Append a null entry for the implicit dyn_ptr argument.
11200 using OpenMPOffloadMappingFlags = llvm::omp::OpenMPOffloadMappingFlags;
11201 auto *NullPtr = llvm::Constant::getNullValue(Ty: CGF.Builder.getPtrTy());
11202 CombinedInfo.BasePointers.push_back(Elt: NullPtr);
11203 CombinedInfo.Pointers.push_back(Elt: NullPtr);
11204 CombinedInfo.DevicePointers.push_back(
11205 Elt: llvm::OpenMPIRBuilder::DeviceInfoTy::None);
11206 CombinedInfo.Sizes.push_back(Elt: CGF.Builder.getInt64(C: 0));
11207 CombinedInfo.Types.push_back(Elt: OpenMPOffloadMappingFlags::OMP_MAP_TARGET_PARAM |
11208 OpenMPOffloadMappingFlags::OMP_MAP_LITERAL);
11209 CombinedInfo.HasAttachPtr.push_back(Elt: false);
11210 if (!CombinedInfo.Names.empty())
11211 CombinedInfo.Names.push_back(Elt: NullPtr);
11212 CombinedInfo.Exprs.push_back(Elt: nullptr);
11213 CombinedInfo.Mappers.push_back(Elt: nullptr);
11214 CombinedInfo.DevicePtrDecls.push_back(Elt: nullptr);
11215
11216 emitOffloadingArraysAndArgs(CGF, CombinedInfo, Info, OMPBuilder,
11217 /*IsNonContiguous=*/true, /*ForEndCall=*/false);
11218
11219 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
11220 InputInfo.BasePointersArray = Address(Info.RTArgs.BasePointersArray,
11221 CGF.VoidPtrTy, CGM.getPointerAlign());
11222 InputInfo.PointersArray =
11223 Address(Info.RTArgs.PointersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
11224 InputInfo.SizesArray =
11225 Address(Info.RTArgs.SizesArray, CGF.Int64Ty, CGM.getPointerAlign());
11226 InputInfo.MappersArray =
11227 Address(Info.RTArgs.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
11228 MapTypesArray = Info.RTArgs.MapTypesArray;
11229 MapNamesArray = Info.RTArgs.MapNamesArray;
11230
11231 auto &&ThenGen = [&OMPRuntime, OutlinedFn, &D, &CapturedVars,
11232 RequiresOuterTask, &CS, OffloadingMandatory, Device,
11233 OutlinedFnID, &InputInfo, &MapTypesArray, &MapNamesArray,
11234 SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) {
11235 bool IsReverseOffloading = Device.getInt() == OMPC_DEVICE_ancestor;
11236
11237 if (IsReverseOffloading) {
11238 // Reverse offloading is not supported, so just execute on the host.
11239 // FIXME: This fallback solution is incorrect since it ignores the
11240 // OMP_TARGET_OFFLOAD environment variable. Instead it would be better to
11241 // assert here and ensure SEMA emits an error.
11242 emitTargetCallFallback(OMPRuntime, OutlinedFn, D, CapturedVars,
11243 RequiresOuterTask, CS, OffloadingMandatory, CGF);
11244 return;
11245 }
11246
11247 bool HasNoWait = D.hasClausesOfKind<OMPNowaitClause>();
11248 unsigned NumTargetItems = InputInfo.NumberOfTargetItems;
11249
11250 llvm::Value *BasePointersArray =
11251 InputInfo.BasePointersArray.emitRawPointer(CGF);
11252 llvm::Value *PointersArray = InputInfo.PointersArray.emitRawPointer(CGF);
11253 llvm::Value *SizesArray = InputInfo.SizesArray.emitRawPointer(CGF);
11254 llvm::Value *MappersArray = InputInfo.MappersArray.emitRawPointer(CGF);
11255
11256 auto &&EmitTargetCallFallbackCB =
11257 [&OMPRuntime, OutlinedFn, &D, &CapturedVars, RequiresOuterTask, &CS,
11258 OffloadingMandatory, &CGF](llvm::OpenMPIRBuilder::InsertPointTy IP)
11259 -> llvm::OpenMPIRBuilder::InsertPointTy {
11260 CGF.Builder.restoreIP(IP);
11261 emitTargetCallFallback(OMPRuntime, OutlinedFn, D, CapturedVars,
11262 RequiresOuterTask, CS, OffloadingMandatory, CGF);
11263 return CGF.Builder.saveIP();
11264 };
11265
11266 bool IsBare = D.hasClausesOfKind<OMPXBareClause>();
11267 SmallVector<llvm::Value *, 3> NumTeams;
11268 SmallVector<llvm::Value *, 3> NumThreads;
11269 if (IsBare) {
11270 emitClauseForBareTargetDirective<OMPNumTeamsClause>(CGF, D, Values&: NumTeams);
11271 emitClauseForBareTargetDirective<OMPThreadLimitClause>(CGF, D,
11272 Values&: NumThreads);
11273 } else {
11274 NumTeams.push_back(Elt: OMPRuntime->emitNumTeamsForTargetDirective(CGF, D));
11275 NumThreads.push_back(
11276 Elt: OMPRuntime->emitNumThreadsForTargetDirective(CGF, D));
11277 }
11278
11279 llvm::Value *DeviceID = emitDeviceID(Device, CGF);
11280 llvm::Value *RTLoc = OMPRuntime->emitUpdateLocation(CGF, Loc: D.getBeginLoc());
11281 llvm::Value *NumIterations =
11282 OMPRuntime->emitTargetNumIterationsCall(CGF, D, SizeEmitter);
11283 auto [DynCGroupMem, DynCGroupMemFallback] = emitDynCGroupMem(D, CGF);
11284 llvm::OpenMPIRBuilder::InsertPointTy AllocaIP(
11285 CGF.AllocaInsertPt->getIterator());
11286
11287 llvm::OpenMPIRBuilder::TargetDataRTArgs RTArgs(
11288 BasePointersArray, PointersArray, SizesArray, MapTypesArray,
11289 nullptr /* MapTypesArrayEnd */, MappersArray, MapNamesArray);
11290
11291 llvm::OpenMPIRBuilder::TargetKernelArgs Args(
11292 NumTargetItems, RTArgs, NumIterations, NumTeams, NumThreads,
11293 DynCGroupMem, HasNoWait, /*StrictBlocks=*/IsBare,
11294 /*StrictThreads=*/IsBare, DynCGroupMemFallback);
11295
11296 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
11297 cantFail(ValOrErr: OMPRuntime->getOMPBuilder().emitKernelLaunch(
11298 Loc: CGF.Builder, OutlinedFnID, EmitTargetCallFallbackCB, Args, DeviceID,
11299 RTLoc, AllocaIP));
11300 CGF.Builder.restoreIP(IP: AfterIP);
11301 };
11302
11303 if (RequiresOuterTask)
11304 CGF.EmitOMPTargetTaskBasedDirective(S: D, BodyGen: ThenGen, InputInfo);
11305 else
11306 OMPRuntime->emitInlinedDirective(CGF, InnerKind: D.getDirectiveKind(), CodeGen: ThenGen);
11307}
11308
11309static void
11310emitTargetCallElse(CGOpenMPRuntime *OMPRuntime, llvm::Function *OutlinedFn,
11311 const OMPExecutableDirective &D,
11312 llvm::SmallVectorImpl<llvm::Value *> &CapturedVars,
11313 bool RequiresOuterTask, const CapturedStmt &CS,
11314 bool OffloadingMandatory, CodeGenFunction &CGF) {
11315
11316 // Notify that the host version must be executed.
11317 auto &&ElseGen =
11318 [&OMPRuntime, OutlinedFn, &D, &CapturedVars, RequiresOuterTask, &CS,
11319 OffloadingMandatory](CodeGenFunction &CGF, PrePostActionTy &) {
11320 emitTargetCallFallback(OMPRuntime, OutlinedFn, D, CapturedVars,
11321 RequiresOuterTask, CS, OffloadingMandatory, CGF);
11322 };
11323
11324 if (RequiresOuterTask) {
11325 CodeGenFunction::OMPTargetDataInfo InputInfo;
11326 CGF.EmitOMPTargetTaskBasedDirective(S: D, BodyGen: ElseGen, InputInfo);
11327 } else {
11328 OMPRuntime->emitInlinedDirective(CGF, InnerKind: D.getDirectiveKind(), CodeGen: ElseGen);
11329 }
11330}
11331
11332void CGOpenMPRuntime::emitTargetCall(
11333 CodeGenFunction &CGF, const OMPExecutableDirective &D,
11334 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
11335 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
11336 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
11337 const OMPLoopDirective &D)>
11338 SizeEmitter) {
11339 if (!CGF.HaveInsertPoint())
11340 return;
11341
11342 const bool OffloadingMandatory = !CGM.getLangOpts().OpenMPIsTargetDevice &&
11343 CGM.getLangOpts().OpenMPOffloadMandatory;
11344
11345 assert((OffloadingMandatory || OutlinedFn) && "Invalid outlined function!");
11346
11347 const bool RequiresOuterTask =
11348 D.hasClausesOfKind<OMPDependClause>() ||
11349 D.hasClausesOfKind<OMPNowaitClause>() ||
11350 D.hasClausesOfKind<OMPInReductionClause>() ||
11351 (CGM.getLangOpts().OpenMP >= 51 &&
11352 needsTaskBasedThreadLimit(DKind: D.getDirectiveKind()) &&
11353 D.hasClausesOfKind<OMPThreadLimitClause>());
11354 llvm::SmallVector<llvm::Value *, 16> CapturedVars;
11355 const CapturedStmt &CS = *D.getCapturedStmt(RegionKind: OMPD_target);
11356 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF,
11357 PrePostActionTy &) {
11358 CGF.GenerateOpenMPCapturedVars(S: CS, CapturedVars);
11359 };
11360 emitInlinedDirective(CGF, InnerKind: OMPD_unknown, CodeGen: ArgsCodegen);
11361
11362 CodeGenFunction::OMPTargetDataInfo InputInfo;
11363 llvm::Value *MapTypesArray = nullptr;
11364 llvm::Value *MapNamesArray = nullptr;
11365
11366 auto &&TargetThenGen = [this, OutlinedFn, &D, &CapturedVars,
11367 RequiresOuterTask, &CS, OffloadingMandatory, Device,
11368 OutlinedFnID, &InputInfo, &MapTypesArray,
11369 &MapNamesArray, SizeEmitter](CodeGenFunction &CGF,
11370 PrePostActionTy &) {
11371 emitTargetCallKernelLaunch(OMPRuntime: this, OutlinedFn, D, CapturedVars,
11372 RequiresOuterTask, CS, OffloadingMandatory,
11373 Device, OutlinedFnID, InputInfo, MapTypesArray,
11374 MapNamesArray, SizeEmitter, CGF, CGM);
11375 };
11376
11377 auto &&TargetElseGen =
11378 [this, OutlinedFn, &D, &CapturedVars, RequiresOuterTask, &CS,
11379 OffloadingMandatory](CodeGenFunction &CGF, PrePostActionTy &) {
11380 emitTargetCallElse(OMPRuntime: this, OutlinedFn, D, CapturedVars, RequiresOuterTask,
11381 CS, OffloadingMandatory, CGF);
11382 };
11383
11384 // If we have a target function ID it means that we need to support
11385 // offloading, otherwise, just execute on the host. We need to execute on host
11386 // regardless of the conditional in the if clause if, e.g., the user do not
11387 // specify target triples.
11388 if (OutlinedFnID) {
11389 if (IfCond) {
11390 emitIfClause(CGF, Cond: IfCond, ThenGen: TargetThenGen, ElseGen: TargetElseGen);
11391 } else {
11392 RegionCodeGenTy ThenRCG(TargetThenGen);
11393 ThenRCG(CGF);
11394 }
11395 } else {
11396 RegionCodeGenTy ElseRCG(TargetElseGen);
11397 ElseRCG(CGF);
11398 }
11399}
11400
11401void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S,
11402 StringRef ParentName) {
11403 if (!S)
11404 return;
11405
11406 // Register vtable from device for target data and target directives.
11407 // Add this block here since scanForTargetRegionsFunctions ignores
11408 // target data by checking if S is a executable directive (target).
11409 if (auto *E = dyn_cast<OMPExecutableDirective>(Val: S);
11410 E && isOpenMPTargetDataManagementDirective(DKind: E->getDirectiveKind())) {
11411 // Don't need to check if it's device compile
11412 // since scanForTargetRegionsFunctions currently only called
11413 // in device compilation.
11414 registerVTable(D: *E);
11415 }
11416
11417 // Codegen OMP target directives that offload compute to the device.
11418 bool RequiresDeviceCodegen =
11419 isa<OMPExecutableDirective>(Val: S) &&
11420 isOpenMPTargetExecutionDirective(
11421 DKind: cast<OMPExecutableDirective>(Val: S)->getDirectiveKind());
11422
11423 if (RequiresDeviceCodegen) {
11424 const auto &E = *cast<OMPExecutableDirective>(Val: S);
11425
11426 llvm::TargetRegionEntryInfo EntryInfo = getEntryInfoFromPresumedLoc(
11427 CGM, OMPBuilder, BeginLoc: E.getBeginLoc(), ParentName);
11428
11429 // Is this a target region that should not be emitted as an entry point? If
11430 // so just signal we are done with this target region.
11431 if (!OMPBuilder.OffloadInfoManager.hasTargetRegionEntryInfo(EntryInfo))
11432 return;
11433
11434 switch (E.getDirectiveKind()) {
11435 case OMPD_target:
11436 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName,
11437 S: cast<OMPTargetDirective>(Val: E));
11438 break;
11439 case OMPD_target_parallel:
11440 CodeGenFunction::EmitOMPTargetParallelDeviceFunction(
11441 CGM, ParentName, S: cast<OMPTargetParallelDirective>(Val: E));
11442 break;
11443 case OMPD_target_teams:
11444 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction(
11445 CGM, ParentName, S: cast<OMPTargetTeamsDirective>(Val: E));
11446 break;
11447 case OMPD_target_teams_distribute:
11448 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction(
11449 CGM, ParentName, S: cast<OMPTargetTeamsDistributeDirective>(Val: E));
11450 break;
11451 case OMPD_target_teams_distribute_simd:
11452 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction(
11453 CGM, ParentName, S: cast<OMPTargetTeamsDistributeSimdDirective>(Val: E));
11454 break;
11455 case OMPD_target_parallel_for:
11456 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction(
11457 CGM, ParentName, S: cast<OMPTargetParallelForDirective>(Val: E));
11458 break;
11459 case OMPD_target_parallel_for_simd:
11460 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction(
11461 CGM, ParentName, S: cast<OMPTargetParallelForSimdDirective>(Val: E));
11462 break;
11463 case OMPD_target_simd:
11464 CodeGenFunction::EmitOMPTargetSimdDeviceFunction(
11465 CGM, ParentName, S: cast<OMPTargetSimdDirective>(Val: E));
11466 break;
11467 case OMPD_target_teams_distribute_parallel_for:
11468 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction(
11469 CGM, ParentName,
11470 S: cast<OMPTargetTeamsDistributeParallelForDirective>(Val: E));
11471 break;
11472 case OMPD_target_teams_distribute_parallel_for_simd:
11473 CodeGenFunction::
11474 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(
11475 CGM, ParentName,
11476 S: cast<OMPTargetTeamsDistributeParallelForSimdDirective>(Val: E));
11477 break;
11478 case OMPD_target_teams_loop:
11479 CodeGenFunction::EmitOMPTargetTeamsGenericLoopDeviceFunction(
11480 CGM, ParentName, S: cast<OMPTargetTeamsGenericLoopDirective>(Val: E));
11481 break;
11482 case OMPD_target_parallel_loop:
11483 CodeGenFunction::EmitOMPTargetParallelGenericLoopDeviceFunction(
11484 CGM, ParentName, S: cast<OMPTargetParallelGenericLoopDirective>(Val: E));
11485 break;
11486 case OMPD_parallel:
11487 case OMPD_for:
11488 case OMPD_parallel_for:
11489 case OMPD_parallel_master:
11490 case OMPD_parallel_sections:
11491 case OMPD_for_simd:
11492 case OMPD_parallel_for_simd:
11493 case OMPD_cancel:
11494 case OMPD_cancellation_point:
11495 case OMPD_ordered_standalone:
11496 case OMPD_ordered_blockassoc:
11497 case OMPD_threadprivate:
11498 case OMPD_allocate:
11499 case OMPD_task:
11500 case OMPD_simd:
11501 case OMPD_tile:
11502 case OMPD_unroll:
11503 case OMPD_sections:
11504 case OMPD_section:
11505 case OMPD_single:
11506 case OMPD_master:
11507 case OMPD_critical:
11508 case OMPD_taskyield:
11509 case OMPD_barrier:
11510 case OMPD_taskwait:
11511 case OMPD_taskgroup:
11512 case OMPD_atomic:
11513 case OMPD_flush:
11514 case OMPD_depobj:
11515 case OMPD_scan:
11516 case OMPD_teams:
11517 case OMPD_target_data:
11518 case OMPD_target_exit_data:
11519 case OMPD_target_enter_data:
11520 case OMPD_distribute:
11521 case OMPD_distribute_simd:
11522 case OMPD_distribute_parallel_for:
11523 case OMPD_distribute_parallel_for_simd:
11524 case OMPD_teams_distribute:
11525 case OMPD_teams_distribute_simd:
11526 case OMPD_teams_distribute_parallel_for:
11527 case OMPD_teams_distribute_parallel_for_simd:
11528 case OMPD_target_update:
11529 case OMPD_declare_simd:
11530 case OMPD_declare_variant:
11531 case OMPD_begin_declare_variant:
11532 case OMPD_end_declare_variant:
11533 case OMPD_declare_target:
11534 case OMPD_end_declare_target:
11535 case OMPD_declare_reduction:
11536 case OMPD_declare_mapper:
11537 case OMPD_taskloop:
11538 case OMPD_taskloop_simd:
11539 case OMPD_master_taskloop:
11540 case OMPD_master_taskloop_simd:
11541 case OMPD_parallel_master_taskloop:
11542 case OMPD_parallel_master_taskloop_simd:
11543 case OMPD_requires:
11544 case OMPD_metadirective:
11545 case OMPD_unknown:
11546 default:
11547 llvm_unreachable("Unknown target directive for OpenMP device codegen.");
11548 }
11549 return;
11550 }
11551
11552 if (const auto *E = dyn_cast<OMPExecutableDirective>(Val: S)) {
11553 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt())
11554 return;
11555
11556 scanForTargetRegionsFunctions(S: E->getRawStmt(), ParentName);
11557 return;
11558 }
11559
11560 // If this is a lambda function, look into its body.
11561 if (const auto *L = dyn_cast<LambdaExpr>(Val: S))
11562 S = L->getBody();
11563
11564 // Keep looking for target regions recursively.
11565 for (const Stmt *II : S->children())
11566 scanForTargetRegionsFunctions(S: II, ParentName);
11567}
11568
11569static bool isAssumedToBeNotEmitted(const ValueDecl *VD, bool IsDevice) {
11570 std::optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
11571 OMPDeclareTargetDeclAttr::getDeviceType(VD);
11572 if (!DevTy)
11573 return false;
11574 // Do not emit device_type(nohost) functions for the host.
11575 if (!IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_NoHost)
11576 return true;
11577 // Do not emit device_type(host) functions for the device.
11578 if (IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_Host)
11579 return true;
11580 return false;
11581}
11582
11583bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) {
11584 // If emitting code for the host, we do not process FD here. Instead we do
11585 // the normal code generation.
11586 if (!CGM.getLangOpts().OpenMPIsTargetDevice) {
11587 if (const auto *FD = dyn_cast<FunctionDecl>(Val: GD.getDecl()))
11588 if (isAssumedToBeNotEmitted(VD: cast<ValueDecl>(Val: FD),
11589 IsDevice: CGM.getLangOpts().OpenMPIsTargetDevice))
11590 return true;
11591 return false;
11592 }
11593
11594 const ValueDecl *VD = cast<ValueDecl>(Val: GD.getDecl());
11595 // Try to detect target regions in the function.
11596 if (const auto *FD = dyn_cast<FunctionDecl>(Val: VD)) {
11597 StringRef Name = CGM.getMangledName(GD);
11598 scanForTargetRegionsFunctions(S: FD->getBody(), ParentName: Name);
11599 if (isAssumedToBeNotEmitted(VD: cast<ValueDecl>(Val: FD),
11600 IsDevice: CGM.getLangOpts().OpenMPIsTargetDevice))
11601 return true;
11602 }
11603
11604 // Do not emit function if it is not marked as declare target.
11605 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) &&
11606 AlreadyEmittedTargetDecls.count(V: VD) == 0;
11607}
11608
11609bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
11610 if (isAssumedToBeNotEmitted(VD: cast<ValueDecl>(Val: GD.getDecl()),
11611 IsDevice: CGM.getLangOpts().OpenMPIsTargetDevice))
11612 return true;
11613
11614 if (!CGM.getLangOpts().OpenMPIsTargetDevice)
11615 return false;
11616
11617 // Check if there are Ctors/Dtors in this declaration and look for target
11618 // regions in it. We use the complete variant to produce the kernel name
11619 // mangling.
11620 QualType RDTy = cast<VarDecl>(Val: GD.getDecl())->getType();
11621 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) {
11622 for (const CXXConstructorDecl *Ctor : RD->ctors()) {
11623 StringRef ParentName =
11624 CGM.getMangledName(GD: GlobalDecl(Ctor, Ctor_Complete));
11625 scanForTargetRegionsFunctions(S: Ctor->getBody(), ParentName);
11626 }
11627 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) {
11628 StringRef ParentName =
11629 CGM.getMangledName(GD: GlobalDecl(Dtor, Dtor_Complete));
11630 scanForTargetRegionsFunctions(S: Dtor->getBody(), ParentName);
11631 }
11632 }
11633
11634 // Do not emit variable if it is not marked as declare target.
11635 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
11636 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(
11637 VD: cast<VarDecl>(Val: GD.getDecl()));
11638 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
11639 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
11640 *Res == OMPDeclareTargetDeclAttr::MT_Enter) &&
11641 HasRequiresUnifiedSharedMemory)) {
11642 DeferredGlobalVariables.insert(V: cast<VarDecl>(Val: GD.getDecl()));
11643 return true;
11644 }
11645 return false;
11646}
11647
11648void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD,
11649 llvm::Constant *Addr) {
11650 if (CGM.getLangOpts().OMPTargetTriples.empty() &&
11651 !CGM.getLangOpts().OpenMPIsTargetDevice)
11652 return;
11653
11654 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
11655 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
11656
11657 // If this is an 'extern' declaration we defer to the canonical definition and
11658 // do not emit an offloading entry.
11659 if (Res && *Res != OMPDeclareTargetDeclAttr::MT_Link &&
11660 VD->hasExternalStorage())
11661 return;
11662
11663 // MT_Local variables use direct access with no host-device mapping.
11664 // No offload entry needed — the device global keeps its own initializer.
11665 if (Res && *Res == OMPDeclareTargetDeclAttr::MT_Local)
11666 return;
11667
11668 if (!Res) {
11669 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
11670 // Register non-target variables being emitted in device code (debug info
11671 // may cause this).
11672 StringRef VarName = CGM.getMangledName(GD: VD);
11673 EmittedNonTargetVariables.try_emplace(Key: VarName, Args&: Addr);
11674 }
11675 return;
11676 }
11677
11678 auto AddrOfGlobal = [&VD, this]() { return CGM.GetAddrOfGlobal(GD: VD); };
11679 auto LinkageForVariable = [&VD, this]() {
11680 return CGM.getLLVMLinkageVarDefinition(VD);
11681 };
11682
11683 std::vector<llvm::GlobalVariable *> GeneratedRefs;
11684 OMPBuilder.registerTargetGlobalVariable(
11685 CaptureClause: convertCaptureClause(VD), DeviceClause: convertDeviceClause(VD),
11686 IsDeclaration: VD->hasDefinition(CGM.getContext()) == VarDecl::DeclarationOnly,
11687 IsExternallyVisible: VD->isExternallyVisible(),
11688 EntryInfo: getEntryInfoFromPresumedLoc(CGM, OMPBuilder,
11689 BeginLoc: VD->getCanonicalDecl()->getBeginLoc()),
11690 MangledName: CGM.getMangledName(GD: VD), GeneratedRefs, OpenMPSIMD: CGM.getLangOpts().OpenMPSimd,
11691 TargetTriple: CGM.getLangOpts().OMPTargetTriples, GlobalInitializer: AddrOfGlobal, VariableLinkage: LinkageForVariable,
11692 LlvmPtrTy: CGM.getTypes().ConvertTypeForMem(
11693 T: CGM.getContext().getPointerType(T: VD->getType())),
11694 Addr);
11695
11696 for (auto *ref : GeneratedRefs)
11697 CGM.addCompilerUsedGlobal(GV: ref);
11698}
11699
11700bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) {
11701 if (isa<FunctionDecl>(Val: GD.getDecl()) ||
11702 isa<OMPDeclareReductionDecl>(Val: GD.getDecl()))
11703 return emitTargetFunctions(GD);
11704
11705 return emitTargetGlobalVariable(GD);
11706}
11707
11708void CGOpenMPRuntime::emitDeferredTargetDecls() const {
11709 for (const VarDecl *VD : DeferredGlobalVariables) {
11710 std::optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
11711 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
11712 if (!Res)
11713 continue;
11714 // MT_Local and MT_To/MT_Enter without USM are always emitted.
11715 if (*Res == OMPDeclareTargetDeclAttr::MT_Local ||
11716 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
11717 *Res == OMPDeclareTargetDeclAttr::MT_Enter) &&
11718 !HasRequiresUnifiedSharedMemory)) {
11719 CGM.EmitGlobal(D: VD);
11720 } else {
11721 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link ||
11722 ((*Res == OMPDeclareTargetDeclAttr::MT_To ||
11723 *Res == OMPDeclareTargetDeclAttr::MT_Enter ||
11724 *Res == OMPDeclareTargetDeclAttr::MT_Local) &&
11725 HasRequiresUnifiedSharedMemory)) &&
11726 "Expected link clause or to clause with unified memory.");
11727 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
11728 }
11729 }
11730}
11731
11732void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas(
11733 CodeGenFunction &CGF, const OMPExecutableDirective &D) const {
11734 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) &&
11735 " Expected target-based directive.");
11736}
11737
11738void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) {
11739 for (const OMPClause *Clause : D->clauselists()) {
11740 if (Clause->getClauseKind() == OMPC_unified_shared_memory) {
11741 HasRequiresUnifiedSharedMemory = true;
11742 OMPBuilder.Config.setHasRequiresUnifiedSharedMemory(true);
11743 } else if (const auto *AC =
11744 dyn_cast<OMPAtomicDefaultMemOrderClause>(Val: Clause)) {
11745 switch (AC->getAtomicDefaultMemOrderKind()) {
11746 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel:
11747 RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease;
11748 break;
11749 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst:
11750 RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent;
11751 break;
11752 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed:
11753 RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic;
11754 break;
11755 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown:
11756 break;
11757 }
11758 }
11759 }
11760}
11761
11762llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const {
11763 return RequiresAtomicOrdering;
11764}
11765
11766bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD,
11767 LangAS &AS) {
11768 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())
11769 return false;
11770 const auto *A = VD->getAttr<OMPAllocateDeclAttr>();
11771 switch(A->getAllocatorType()) {
11772 case OMPAllocateDeclAttr::OMPNullMemAlloc:
11773 case OMPAllocateDeclAttr::OMPDefaultMemAlloc:
11774 // Not supported, fallback to the default mem space.
11775 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc:
11776 case OMPAllocateDeclAttr::OMPCGroupMemAlloc:
11777 case OMPAllocateDeclAttr::OMPHighBWMemAlloc:
11778 case OMPAllocateDeclAttr::OMPLowLatMemAlloc:
11779 case OMPAllocateDeclAttr::OMPThreadMemAlloc:
11780 case OMPAllocateDeclAttr::OMPConstMemAlloc:
11781 case OMPAllocateDeclAttr::OMPPTeamMemAlloc:
11782 AS = LangAS::Default;
11783 return true;
11784 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc:
11785 llvm_unreachable("Expected predefined allocator for the variables with the "
11786 "static storage.");
11787 }
11788 return false;
11789}
11790
11791bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const {
11792 return HasRequiresUnifiedSharedMemory;
11793}
11794
11795CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII(
11796 CodeGenModule &CGM)
11797 : CGM(CGM) {
11798 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
11799 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal;
11800 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false;
11801 }
11802}
11803
11804CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() {
11805 if (CGM.getLangOpts().OpenMPIsTargetDevice)
11806 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal;
11807}
11808
11809bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) {
11810 if (!CGM.getLangOpts().OpenMPIsTargetDevice || !ShouldMarkAsGlobal)
11811 return true;
11812
11813 const auto *D = cast<FunctionDecl>(Val: GD.getDecl());
11814 // Do not emit function if it is marked as declare target as it was already
11815 // emitted.
11816 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD: D)) {
11817 if (D->hasBody() && AlreadyEmittedTargetDecls.count(V: D) == 0) {
11818 if (auto *F = dyn_cast_or_null<llvm::Function>(
11819 Val: CGM.GetGlobalValue(Ref: CGM.getMangledName(GD))))
11820 return !F->isDeclaration();
11821 return false;
11822 }
11823 return true;
11824 }
11825
11826 return !AlreadyEmittedTargetDecls.insert(V: D).second;
11827}
11828
11829void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF,
11830 const OMPExecutableDirective &D,
11831 SourceLocation Loc,
11832 llvm::Function *OutlinedFn,
11833 ArrayRef<llvm::Value *> CapturedVars) {
11834 if (!CGF.HaveInsertPoint())
11835 return;
11836
11837 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11838 CodeGenFunction::RunCleanupsScope Scope(CGF);
11839
11840 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn);
11841 llvm::Value *Args[] = {
11842 RTLoc,
11843 CGF.Builder.getInt32(C: CapturedVars.size()), // Number of captured vars
11844 OutlinedFn};
11845 llvm::SmallVector<llvm::Value *, 16> RealArgs;
11846 RealArgs.append(in_start: std::begin(arr&: Args), in_end: std::end(arr&: Args));
11847 RealArgs.append(in_start: CapturedVars.begin(), in_end: CapturedVars.end());
11848
11849 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction(
11850 M&: CGM.getModule(), FnID: OMPRTL___kmpc_fork_teams);
11851 CGF.EmitRuntimeCall(callee: RTLFn, args: RealArgs);
11852}
11853
11854void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
11855 const Expr *NumTeams,
11856 const Expr *ThreadLimit,
11857 SourceLocation Loc) {
11858 if (!CGF.HaveInsertPoint())
11859 return;
11860
11861 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11862
11863 llvm::Value *NumTeamsVal =
11864 NumTeams
11865 ? CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: NumTeams),
11866 DestTy: CGF.CGM.Int32Ty, /* isSigned = */ true)
11867 : CGF.Builder.getInt32(C: 0);
11868
11869 llvm::Value *ThreadLimitVal =
11870 ThreadLimit
11871 ? CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: ThreadLimit),
11872 DestTy: CGF.CGM.Int32Ty, /* isSigned = */ true)
11873 : CGF.Builder.getInt32(C: 0);
11874
11875 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit)
11876 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal,
11877 ThreadLimitVal};
11878 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
11879 M&: CGM.getModule(), FnID: OMPRTL___kmpc_push_num_teams),
11880 args: PushNumTeamsArgs);
11881}
11882
11883void CGOpenMPRuntime::emitThreadLimitClause(CodeGenFunction &CGF,
11884 const Expr *ThreadLimit,
11885 SourceLocation Loc) {
11886 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11887 llvm::Value *ThreadLimitVal =
11888 ThreadLimit
11889 ? CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: ThreadLimit),
11890 DestTy: CGF.CGM.Int32Ty, /* isSigned = */ true)
11891 : CGF.Builder.getInt32(C: 0);
11892
11893 // Build call __kmpc_set_thread_limit(&loc, global_tid, thread_limit)
11894 llvm::Value *ThreadLimitArgs[] = {RTLoc, getThreadID(CGF, Loc),
11895 ThreadLimitVal};
11896 CGF.EmitRuntimeCall(callee: OMPBuilder.getOrCreateRuntimeFunction(
11897 M&: CGM.getModule(), FnID: OMPRTL___kmpc_set_thread_limit),
11898 args: ThreadLimitArgs);
11899}
11900
11901void CGOpenMPRuntime::emitTargetDataCalls(
11902 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
11903 const Expr *Device, const RegionCodeGenTy &CodeGen,
11904 CGOpenMPRuntime::TargetDataInfo &Info) {
11905 if (!CGF.HaveInsertPoint())
11906 return;
11907
11908 // Action used to replace the default codegen action and turn privatization
11909 // off.
11910 PrePostActionTy NoPrivAction;
11911
11912 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
11913
11914 llvm::Value *IfCondVal = nullptr;
11915 if (IfCond)
11916 IfCondVal = CGF.EvaluateExprAsBool(E: IfCond);
11917
11918 // Emit device ID if any.
11919 llvm::Value *DeviceID = nullptr;
11920 if (Device) {
11921 DeviceID = CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: Device),
11922 DestTy: CGF.Int64Ty, /*isSigned=*/true);
11923 } else {
11924 DeviceID = CGF.Builder.getInt64(C: OMP_DEVICEID_UNDEF);
11925 }
11926
11927 // Fill up the arrays with all the mapped variables.
11928 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
11929 auto GenMapInfoCB =
11930 [&](InsertPointTy CodeGenIP) -> llvm::OpenMPIRBuilder::MapInfosTy & {
11931 CGF.Builder.restoreIP(IP: CodeGenIP);
11932 // Get map clause information.
11933 MappableExprsHandler MEHandler(D, CGF);
11934 MEHandler.generateAllInfo(CombinedInfo, OMPBuilder);
11935
11936 auto FillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) {
11937 return emitMappingInformation(CGF, OMPBuilder, MapExprs&: MapExpr);
11938 };
11939 if (CGM.getCodeGenOpts().getDebugInfo() !=
11940 llvm::codegenoptions::NoDebugInfo) {
11941 CombinedInfo.Names.resize(N: CombinedInfo.Exprs.size());
11942 llvm::transform(Range&: CombinedInfo.Exprs, d_first: CombinedInfo.Names.begin(),
11943 F: FillInfoMap);
11944 }
11945
11946 return CombinedInfo;
11947 };
11948 using BodyGenTy = llvm::OpenMPIRBuilder::BodyGenTy;
11949 auto BodyCB = [&](InsertPointTy CodeGenIP, BodyGenTy BodyGenType) {
11950 CGF.Builder.restoreIP(IP: CodeGenIP);
11951 switch (BodyGenType) {
11952 case BodyGenTy::Priv:
11953 if (!Info.CaptureDeviceAddrMap.empty())
11954 CodeGen(CGF);
11955 break;
11956 case BodyGenTy::DupNoPriv:
11957 if (!Info.CaptureDeviceAddrMap.empty()) {
11958 CodeGen.setAction(NoPrivAction);
11959 CodeGen(CGF);
11960 }
11961 break;
11962 case BodyGenTy::NoPriv:
11963 if (Info.CaptureDeviceAddrMap.empty()) {
11964 CodeGen.setAction(NoPrivAction);
11965 CodeGen(CGF);
11966 }
11967 break;
11968 }
11969 return InsertPointTy(CGF.Builder.GetInsertPoint());
11970 };
11971
11972 auto DeviceAddrCB = [&](unsigned int I, llvm::Value *NewDecl) {
11973 if (const ValueDecl *DevVD = CombinedInfo.DevicePtrDecls[I]) {
11974 Info.CaptureDeviceAddrMap.try_emplace(Key: DevVD, Args&: NewDecl);
11975 }
11976 };
11977
11978 auto CustomMapperCB = [&](unsigned int I) {
11979 llvm::Function *MFunc = nullptr;
11980 if (CombinedInfo.Mappers[I]) {
11981 Info.HasMapper = true;
11982 MFunc = CGF.CGM.getOpenMPRuntime().getOrCreateUserDefinedMapperFunc(
11983 D: cast<OMPDeclareMapperDecl>(Val: CombinedInfo.Mappers[I]));
11984 }
11985 return MFunc;
11986 };
11987
11988 // Source location for the ident struct
11989 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc: D.getBeginLoc());
11990
11991 InsertPointTy AllocaIP(CGF.AllocaInsertPt->getIterator());
11992 InsertPointTy CodeGenIP(CGF.Builder.GetInsertPoint());
11993 llvm::OpenMPIRBuilder::LocationDescription OmpLoc(CGF.Builder);
11994 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
11995 cantFail(ValOrErr: OMPBuilder.createTargetData(
11996 Loc: OmpLoc, AllocaIP, CodeGenIP, /*DeallocBlocks=*/{}, DeviceID,
11997 IfCond: IfCondVal, Info, GenMapInfoCB, CustomMapperCB,
11998 /*MapperFunc=*/nullptr, BodyGenCB: BodyCB, DeviceAddrCB, SrcLocInfo: RTLoc));
11999 CGF.Builder.restoreIP(IP: AfterIP);
12000}
12001
12002void CGOpenMPRuntime::emitTargetDataStandAloneCall(
12003 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12004 const Expr *Device) {
12005 if (!CGF.HaveInsertPoint())
12006 return;
12007
12008 assert((isa<OMPTargetEnterDataDirective>(D) ||
12009 isa<OMPTargetExitDataDirective>(D) ||
12010 isa<OMPTargetUpdateDirective>(D)) &&
12011 "Expecting either target enter, exit data, or update directives.");
12012
12013 CodeGenFunction::OMPTargetDataInfo InputInfo;
12014 llvm::Value *MapTypesArray = nullptr;
12015 llvm::Value *MapNamesArray = nullptr;
12016 // Generate the code for the opening of the data environment.
12017 auto &&ThenGen = [this, &D, Device, &InputInfo, &MapTypesArray,
12018 &MapNamesArray](CodeGenFunction &CGF, PrePostActionTy &) {
12019 // Emit device ID if any.
12020 llvm::Value *DeviceID = nullptr;
12021 if (Device) {
12022 DeviceID = CGF.Builder.CreateIntCast(V: CGF.EmitScalarExpr(E: Device),
12023 DestTy: CGF.Int64Ty, /*isSigned=*/true);
12024 } else {
12025 DeviceID = CGF.Builder.getInt64(C: OMP_DEVICEID_UNDEF);
12026 }
12027
12028 // Emit the number of elements in the offloading arrays.
12029 llvm::Constant *PointerNum =
12030 CGF.Builder.getInt32(C: InputInfo.NumberOfTargetItems);
12031
12032 // Source location for the ident struct
12033 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc: D.getBeginLoc());
12034
12035 SmallVector<llvm::Value *, 13> OffloadingArgs(
12036 {RTLoc, DeviceID, PointerNum,
12037 InputInfo.BasePointersArray.emitRawPointer(CGF),
12038 InputInfo.PointersArray.emitRawPointer(CGF),
12039 InputInfo.SizesArray.emitRawPointer(CGF), MapTypesArray, MapNamesArray,
12040 InputInfo.MappersArray.emitRawPointer(CGF)});
12041
12042 // Select the right runtime function call for each standalone
12043 // directive.
12044 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
12045 RuntimeFunction RTLFn;
12046 switch (D.getDirectiveKind()) {
12047 case OMPD_target_enter_data:
12048 RTLFn = HasNowait ? OMPRTL___tgt_target_data_begin_nowait_mapper
12049 : OMPRTL___tgt_target_data_begin_mapper;
12050 break;
12051 case OMPD_target_exit_data:
12052 RTLFn = HasNowait ? OMPRTL___tgt_target_data_end_nowait_mapper
12053 : OMPRTL___tgt_target_data_end_mapper;
12054 break;
12055 case OMPD_target_update:
12056 RTLFn = HasNowait ? OMPRTL___tgt_target_data_update_nowait_mapper
12057 : OMPRTL___tgt_target_data_update_mapper;
12058 break;
12059 case OMPD_parallel:
12060 case OMPD_for:
12061 case OMPD_parallel_for:
12062 case OMPD_parallel_master:
12063 case OMPD_parallel_sections:
12064 case OMPD_for_simd:
12065 case OMPD_parallel_for_simd:
12066 case OMPD_cancel:
12067 case OMPD_cancellation_point:
12068 case OMPD_ordered_standalone:
12069 case OMPD_ordered_blockassoc:
12070 case OMPD_threadprivate:
12071 case OMPD_allocate:
12072 case OMPD_task:
12073 case OMPD_simd:
12074 case OMPD_tile:
12075 case OMPD_unroll:
12076 case OMPD_sections:
12077 case OMPD_section:
12078 case OMPD_single:
12079 case OMPD_master:
12080 case OMPD_critical:
12081 case OMPD_taskyield:
12082 case OMPD_barrier:
12083 case OMPD_taskwait:
12084 case OMPD_taskgroup:
12085 case OMPD_atomic:
12086 case OMPD_flush:
12087 case OMPD_depobj:
12088 case OMPD_scan:
12089 case OMPD_teams:
12090 case OMPD_target_data:
12091 case OMPD_distribute:
12092 case OMPD_distribute_simd:
12093 case OMPD_distribute_parallel_for:
12094 case OMPD_distribute_parallel_for_simd:
12095 case OMPD_teams_distribute:
12096 case OMPD_teams_distribute_simd:
12097 case OMPD_teams_distribute_parallel_for:
12098 case OMPD_teams_distribute_parallel_for_simd:
12099 case OMPD_declare_simd:
12100 case OMPD_declare_variant:
12101 case OMPD_begin_declare_variant:
12102 case OMPD_end_declare_variant:
12103 case OMPD_declare_target:
12104 case OMPD_end_declare_target:
12105 case OMPD_declare_reduction:
12106 case OMPD_declare_mapper:
12107 case OMPD_taskloop:
12108 case OMPD_taskloop_simd:
12109 case OMPD_master_taskloop:
12110 case OMPD_master_taskloop_simd:
12111 case OMPD_parallel_master_taskloop:
12112 case OMPD_parallel_master_taskloop_simd:
12113 case OMPD_target:
12114 case OMPD_target_simd:
12115 case OMPD_target_teams_distribute:
12116 case OMPD_target_teams_distribute_simd:
12117 case OMPD_target_teams_distribute_parallel_for:
12118 case OMPD_target_teams_distribute_parallel_for_simd:
12119 case OMPD_target_teams:
12120 case OMPD_target_parallel:
12121 case OMPD_target_parallel_for:
12122 case OMPD_target_parallel_for_simd:
12123 case OMPD_requires:
12124 case OMPD_metadirective:
12125 case OMPD_unknown:
12126 default:
12127 llvm_unreachable("Unexpected standalone target data directive.");
12128 break;
12129 }
12130 if (HasNowait) {
12131 OffloadingArgs.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.Int32Ty));
12132 OffloadingArgs.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.VoidPtrTy));
12133 OffloadingArgs.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.Int32Ty));
12134 OffloadingArgs.push_back(Elt: llvm::Constant::getNullValue(Ty: CGF.VoidPtrTy));
12135 }
12136 CGF.EmitRuntimeCall(
12137 callee: OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(), FnID: RTLFn),
12138 args: OffloadingArgs);
12139 };
12140
12141 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
12142 &MapNamesArray](CodeGenFunction &CGF,
12143 PrePostActionTy &) {
12144 // Fill up the arrays with all the mapped variables.
12145 MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
12146 CGOpenMPRuntime::TargetDataInfo Info;
12147 MappableExprsHandler MEHandler(D, CGF);
12148 genMapInfo(MEHandler, CGF, CombinedInfo, OMPBuilder);
12149 emitOffloadingArraysAndArgs(CGF, CombinedInfo, Info, OMPBuilder,
12150 /*IsNonContiguous=*/true, /*ForEndCall=*/false);
12151
12152 bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() ||
12153 D.hasClausesOfKind<OMPNowaitClause>();
12154
12155 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
12156 InputInfo.BasePointersArray = Address(Info.RTArgs.BasePointersArray,
12157 CGF.VoidPtrTy, CGM.getPointerAlign());
12158 InputInfo.PointersArray = Address(Info.RTArgs.PointersArray, CGF.VoidPtrTy,
12159 CGM.getPointerAlign());
12160 InputInfo.SizesArray =
12161 Address(Info.RTArgs.SizesArray, CGF.Int64Ty, CGM.getPointerAlign());
12162 InputInfo.MappersArray =
12163 Address(Info.RTArgs.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
12164 MapTypesArray = Info.RTArgs.MapTypesArray;
12165 MapNamesArray = Info.RTArgs.MapNamesArray;
12166 if (RequiresOuterTask)
12167 CGF.EmitOMPTargetTaskBasedDirective(S: D, BodyGen: ThenGen, InputInfo);
12168 else
12169 emitInlinedDirective(CGF, InnerKind: D.getDirectiveKind(), CodeGen: ThenGen);
12170 };
12171
12172 if (IfCond) {
12173 emitIfClause(CGF, Cond: IfCond, ThenGen: TargetThenGen,
12174 ElseGen: [](CodeGenFunction &CGF, PrePostActionTy &) {});
12175 } else {
12176 RegionCodeGenTy ThenRCG(TargetThenGen);
12177 ThenRCG(CGF);
12178 }
12179}
12180
12181static unsigned
12182evaluateCDTSize(const FunctionDecl *FD,
12183 ArrayRef<llvm::OpenMPIRBuilder::DeclareSimdAttrTy> ParamAttrs) {
12184 // Every vector variant of a SIMD-enabled function has a vector length (VLEN).
12185 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument
12186 // of that clause. The VLEN value must be power of 2.
12187 // In other case the notion of the function`s "characteristic data type" (CDT)
12188 // is used to compute the vector length.
12189 // CDT is defined in the following order:
12190 // a) For non-void function, the CDT is the return type.
12191 // b) If the function has any non-uniform, non-linear parameters, then the
12192 // CDT is the type of the first such parameter.
12193 // c) If the CDT determined by a) or b) above is struct, union, or class
12194 // type which is pass-by-value (except for the type that maps to the
12195 // built-in complex data type), the characteristic data type is int.
12196 // d) If none of the above three cases is applicable, the CDT is int.
12197 // The VLEN is then determined based on the CDT and the size of vector
12198 // register of that ISA for which current vector version is generated. The
12199 // VLEN is computed using the formula below:
12200 // VLEN = sizeof(vector_register) / sizeof(CDT),
12201 // where vector register size specified in section 3.2.1 Registers and the
12202 // Stack Frame of original AMD64 ABI document.
12203 QualType RetType = FD->getReturnType();
12204 if (RetType.isNull())
12205 return 0;
12206 ASTContext &C = FD->getASTContext();
12207 QualType CDT;
12208 if (!RetType.isNull() && !RetType->isVoidType()) {
12209 CDT = RetType;
12210 } else {
12211 unsigned Offset = 0;
12212 if (const auto *MD = dyn_cast<CXXMethodDecl>(Val: FD)) {
12213 if (ParamAttrs[Offset].Kind ==
12214 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Vector)
12215 CDT = C.getPointerType(T: C.getCanonicalTagType(TD: MD->getParent()));
12216 ++Offset;
12217 }
12218 if (CDT.isNull()) {
12219 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
12220 if (ParamAttrs[I + Offset].Kind ==
12221 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Vector) {
12222 CDT = FD->getParamDecl(i: I)->getType();
12223 break;
12224 }
12225 }
12226 }
12227 }
12228 if (CDT.isNull())
12229 CDT = C.IntTy;
12230 CDT = CDT->getCanonicalTypeUnqualified();
12231 if (CDT->isRecordType() || CDT->isUnionType())
12232 CDT = C.IntTy;
12233 return C.getTypeSize(T: CDT);
12234}
12235
12236// This are the Functions that are needed to mangle the name of the
12237// vector functions generated by the compiler, according to the rules
12238// defined in the "Vector Function ABI specifications for AArch64",
12239// available at
12240// https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi.
12241
12242/// Maps To Vector (MTV), as defined in 4.1.1 of the AAVFABI (2021Q1).
12243static bool getAArch64MTV(QualType QT,
12244 llvm::OpenMPIRBuilder::DeclareSimdKindTy Kind) {
12245 QT = QT.getCanonicalType();
12246
12247 if (QT->isVoidType())
12248 return false;
12249
12250 if (Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::Uniform)
12251 return false;
12252
12253 if (Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearUVal ||
12254 Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearRef)
12255 return false;
12256
12257 if ((Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::Linear ||
12258 Kind == llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearVal) &&
12259 !QT->isReferenceType())
12260 return false;
12261
12262 return true;
12263}
12264
12265/// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
12266static bool getAArch64PBV(QualType QT, ASTContext &C) {
12267 QT = QT.getCanonicalType();
12268 unsigned Size = C.getTypeSize(T: QT);
12269
12270 // Only scalars and complex within 16 bytes wide set PVB to true.
12271 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128)
12272 return false;
12273
12274 if (QT->isFloatingType())
12275 return true;
12276
12277 if (QT->isIntegerType())
12278 return true;
12279
12280 if (QT->isPointerType())
12281 return true;
12282
12283 // TODO: Add support for complex types (section 3.1.2, item 2).
12284
12285 return false;
12286}
12287
12288/// Computes the lane size (LS) of a return type or of an input parameter,
12289/// as defined by `LS(P)` in 3.2.1 of the AAVFABI.
12290/// TODO: Add support for references, section 3.2.1, item 1.
12291static unsigned getAArch64LS(QualType QT,
12292 llvm::OpenMPIRBuilder::DeclareSimdKindTy Kind,
12293 ASTContext &C) {
12294 if (!getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) {
12295 QualType PTy = QT.getCanonicalType()->getPointeeType();
12296 if (getAArch64PBV(QT: PTy, C))
12297 return C.getTypeSize(T: PTy);
12298 }
12299 if (getAArch64PBV(QT, C))
12300 return C.getTypeSize(T: QT);
12301
12302 return C.getTypeSize(T: C.getUIntPtrType());
12303}
12304
12305// Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the
12306// signature of the scalar function, as defined in 3.2.2 of the
12307// AAVFABI.
12308static std::tuple<unsigned, unsigned, bool>
12309getNDSWDS(const FunctionDecl *FD,
12310 ArrayRef<llvm::OpenMPIRBuilder::DeclareSimdAttrTy> ParamAttrs) {
12311 QualType RetType = FD->getReturnType().getCanonicalType();
12312
12313 ASTContext &C = FD->getASTContext();
12314
12315 bool OutputBecomesInput = false;
12316
12317 llvm::SmallVector<unsigned, 8> Sizes;
12318 if (!RetType->isVoidType()) {
12319 Sizes.push_back(Elt: getAArch64LS(
12320 QT: RetType, Kind: llvm::OpenMPIRBuilder::DeclareSimdKindTy::Vector, C));
12321 if (!getAArch64PBV(QT: RetType, C) && getAArch64MTV(QT: RetType, Kind: {}))
12322 OutputBecomesInput = true;
12323 }
12324 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
12325 QualType QT = FD->getParamDecl(i: I)->getType().getCanonicalType();
12326 Sizes.push_back(Elt: getAArch64LS(QT, Kind: ParamAttrs[I].Kind, C));
12327 }
12328
12329 assert(!Sizes.empty() && "Unable to determine NDS and WDS.");
12330 // The LS of a function parameter / return value can only be a power
12331 // of 2, starting from 8 bits, up to 128.
12332 assert(llvm::all_of(Sizes,
12333 [](unsigned Size) {
12334 return Size == 8 || Size == 16 || Size == 32 ||
12335 Size == 64 || Size == 128;
12336 }) &&
12337 "Invalid size");
12338
12339 return std::make_tuple(args&: *llvm::min_element(Range&: Sizes), args&: *llvm::max_element(Range&: Sizes),
12340 args&: OutputBecomesInput);
12341}
12342
12343static llvm::OpenMPIRBuilder::DeclareSimdBranch
12344convertDeclareSimdBranch(OMPDeclareSimdDeclAttr::BranchStateTy State) {
12345 switch (State) {
12346 case OMPDeclareSimdDeclAttr::BS_Undefined:
12347 return llvm::OpenMPIRBuilder::DeclareSimdBranch::Undefined;
12348 case OMPDeclareSimdDeclAttr::BS_Inbranch:
12349 return llvm::OpenMPIRBuilder::DeclareSimdBranch::Inbranch;
12350 case OMPDeclareSimdDeclAttr::BS_Notinbranch:
12351 return llvm::OpenMPIRBuilder::DeclareSimdBranch::Notinbranch;
12352 }
12353 llvm_unreachable("unexpected declare simd branch state");
12354}
12355
12356// Check the values provided via `simdlen` by the user.
12357static bool validateAArch64Simdlen(CodeGenModule &CGM, SourceLocation SLoc,
12358 unsigned UserVLEN, unsigned WDS, char ISA) {
12359 // 1. A `simdlen(1)` doesn't produce vector signatures.
12360 if (UserVLEN == 1) {
12361 CGM.getDiags().Report(Loc: SLoc, DiagID: diag::warn_simdlen_1_no_effect);
12362 return false;
12363 }
12364
12365 // 2. Section 3.3.1, item 1: user input must be a power of 2 for Advanced
12366 // SIMD.
12367 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(Value: UserVLEN)) {
12368 CGM.getDiags().Report(Loc: SLoc, DiagID: diag::warn_simdlen_requires_power_of_2);
12369 return false;
12370 }
12371
12372 // 3. Section 3.4.1: SVE fixed length must obey the architectural limits.
12373 if (ISA == 's' && UserVLEN != 0 &&
12374 ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0))) {
12375 CGM.getDiags().Report(Loc: SLoc, DiagID: diag::warn_simdlen_must_fit_lanes) << WDS;
12376 return false;
12377 }
12378
12379 return true;
12380}
12381
12382void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD,
12383 llvm::Function *Fn) {
12384 ASTContext &C = CGM.getContext();
12385 FD = FD->getMostRecentDecl();
12386 while (FD) {
12387 // Map params to their positions in function decl.
12388 llvm::DenseMap<const Decl *, unsigned> ParamPositions;
12389 if (isa<CXXMethodDecl>(Val: FD))
12390 ParamPositions.try_emplace(Key: FD, Args: 0);
12391 unsigned ParamPos = ParamPositions.size();
12392 for (const ParmVarDecl *P : FD->parameters()) {
12393 ParamPositions.try_emplace(Key: P->getCanonicalDecl(), Args&: ParamPos);
12394 ++ParamPos;
12395 }
12396 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) {
12397 llvm::SmallVector<llvm::OpenMPIRBuilder::DeclareSimdAttrTy, 8> ParamAttrs(
12398 ParamPositions.size());
12399 // Mark uniform parameters.
12400 for (const Expr *E : Attr->uniforms()) {
12401 E = E->IgnoreParenImpCasts();
12402 unsigned Pos;
12403 if (isa<CXXThisExpr>(Val: E)) {
12404 Pos = ParamPositions[FD];
12405 } else {
12406 const auto *PVD = cast<ParmVarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl())
12407 ->getCanonicalDecl();
12408 auto It = ParamPositions.find(Val: PVD);
12409 assert(It != ParamPositions.end() && "Function parameter not found");
12410 Pos = It->second;
12411 }
12412 ParamAttrs[Pos].Kind =
12413 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Uniform;
12414 }
12415 // Get alignment info.
12416 auto *NI = Attr->alignments_begin();
12417 for (const Expr *E : Attr->aligneds()) {
12418 E = E->IgnoreParenImpCasts();
12419 unsigned Pos;
12420 QualType ParmTy;
12421 if (isa<CXXThisExpr>(Val: E)) {
12422 Pos = ParamPositions[FD];
12423 ParmTy = E->getType();
12424 } else {
12425 const auto *PVD = cast<ParmVarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl())
12426 ->getCanonicalDecl();
12427 auto It = ParamPositions.find(Val: PVD);
12428 assert(It != ParamPositions.end() && "Function parameter not found");
12429 Pos = It->second;
12430 ParmTy = PVD->getType();
12431 }
12432 ParamAttrs[Pos].Alignment =
12433 (*NI)
12434 ? (*NI)->EvaluateKnownConstInt(Ctx: C)
12435 : llvm::APSInt::getUnsigned(
12436 X: C.toCharUnitsFromBits(BitSize: C.getOpenMPDefaultSimdAlign(T: ParmTy))
12437 .getQuantity());
12438 ++NI;
12439 }
12440 // Mark linear parameters.
12441 auto *SI = Attr->steps_begin();
12442 auto *MI = Attr->modifiers_begin();
12443 for (const Expr *E : Attr->linears()) {
12444 E = E->IgnoreParenImpCasts();
12445 unsigned Pos;
12446 bool IsReferenceType = false;
12447 // Rescaling factor needed to compute the linear parameter
12448 // value in the mangled name.
12449 unsigned PtrRescalingFactor = 1;
12450 if (isa<CXXThisExpr>(Val: E)) {
12451 Pos = ParamPositions[FD];
12452 auto *P = cast<PointerType>(Val: E->getType());
12453 PtrRescalingFactor = CGM.getContext()
12454 .getTypeSizeInChars(T: P->getPointeeType())
12455 .getQuantity();
12456 } else {
12457 const auto *PVD = cast<ParmVarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl())
12458 ->getCanonicalDecl();
12459 auto It = ParamPositions.find(Val: PVD);
12460 assert(It != ParamPositions.end() && "Function parameter not found");
12461 Pos = It->second;
12462 if (auto *P = dyn_cast<PointerType>(Val: PVD->getType()))
12463 PtrRescalingFactor = CGM.getContext()
12464 .getTypeSizeInChars(T: P->getPointeeType())
12465 .getQuantity();
12466 else if (PVD->getType()->isReferenceType()) {
12467 IsReferenceType = true;
12468 PtrRescalingFactor =
12469 CGM.getContext()
12470 .getTypeSizeInChars(T: PVD->getType().getNonReferenceType())
12471 .getQuantity();
12472 }
12473 }
12474 llvm::OpenMPIRBuilder::DeclareSimdAttrTy &ParamAttr = ParamAttrs[Pos];
12475 if (*MI == OMPC_LINEAR_ref)
12476 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearRef;
12477 else if (*MI == OMPC_LINEAR_uval)
12478 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearUVal;
12479 else if (IsReferenceType)
12480 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearVal;
12481 else
12482 ParamAttr.Kind = llvm::OpenMPIRBuilder::DeclareSimdKindTy::Linear;
12483 // Assuming a stride of 1, for `linear` without modifiers.
12484 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(X: 1);
12485 if (*SI) {
12486 Expr::EvalResult Result;
12487 if (!(*SI)->EvaluateAsInt(Result, Ctx: C, AllowSideEffects: Expr::SE_AllowSideEffects)) {
12488 if (const auto *DRE =
12489 cast<DeclRefExpr>(Val: (*SI)->IgnoreParenImpCasts())) {
12490 if (const auto *StridePVD =
12491 dyn_cast<ParmVarDecl>(Val: DRE->getDecl())) {
12492 ParamAttr.HasVarStride = true;
12493 auto It = ParamPositions.find(Val: StridePVD->getCanonicalDecl());
12494 assert(It != ParamPositions.end() &&
12495 "Function parameter not found");
12496 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(X: It->second);
12497 }
12498 }
12499 } else {
12500 ParamAttr.StrideOrArg = Result.Val.getInt();
12501 }
12502 }
12503 // If we are using a linear clause on a pointer, we need to
12504 // rescale the value of linear_step with the byte size of the
12505 // pointee type.
12506 if (!ParamAttr.HasVarStride &&
12507 (ParamAttr.Kind ==
12508 llvm::OpenMPIRBuilder::DeclareSimdKindTy::Linear ||
12509 ParamAttr.Kind ==
12510 llvm::OpenMPIRBuilder::DeclareSimdKindTy::LinearRef))
12511 ParamAttr.StrideOrArg = ParamAttr.StrideOrArg * PtrRescalingFactor;
12512 ++SI;
12513 ++MI;
12514 }
12515 llvm::APSInt VLENVal;
12516 SourceLocation ExprLoc;
12517 const Expr *VLENExpr = Attr->getSimdlen();
12518 if (VLENExpr) {
12519 VLENVal = VLENExpr->EvaluateKnownConstInt(Ctx: C);
12520 ExprLoc = VLENExpr->getExprLoc();
12521 }
12522 llvm::OpenMPIRBuilder::DeclareSimdBranch State =
12523 convertDeclareSimdBranch(State: Attr->getBranchState());
12524 if (CGM.getTriple().isX86()) {
12525 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs);
12526 assert(NumElts && "Non-zero simdlen/cdtsize expected");
12527 OMPBuilder.emitX86DeclareSimdFunction(Fn, NumElements: NumElts, VLENVal, ParamAttrs,
12528 Branch: State);
12529 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) {
12530 unsigned VLEN = VLENVal.getExtValue();
12531 // Get basic data for building the vector signature.
12532 const auto Data = getNDSWDS(FD, ParamAttrs);
12533 const unsigned NDS = std::get<0>(t: Data);
12534 const unsigned WDS = std::get<1>(t: Data);
12535 const bool OutputBecomesInput = std::get<2>(t: Data);
12536 if (CGM.getTarget().hasFeature(Feature: "sve")) {
12537 if (validateAArch64Simdlen(CGM, SLoc: ExprLoc, UserVLEN: VLEN, WDS, ISA: 's'))
12538 OMPBuilder.emitAArch64DeclareSimdFunction(
12539 Fn, VLENVal: VLEN, ParamAttrs, Branch: State, ISA: 's', NarrowestDataSize: NDS, OutputBecomesInput);
12540 } else if (CGM.getTarget().hasFeature(Feature: "neon")) {
12541 if (validateAArch64Simdlen(CGM, SLoc: ExprLoc, UserVLEN: VLEN, WDS, ISA: 'n'))
12542 OMPBuilder.emitAArch64DeclareSimdFunction(
12543 Fn, VLENVal: VLEN, ParamAttrs, Branch: State, ISA: 'n', NarrowestDataSize: NDS, OutputBecomesInput);
12544 }
12545 }
12546 }
12547 FD = FD->getPreviousDecl();
12548 }
12549}
12550
12551namespace {
12552/// Cleanup action for doacross support.
12553class DoacrossCleanupTy final : public EHScopeStack::Cleanup {
12554public:
12555 static const int DoacrossFinArgs = 2;
12556
12557private:
12558 llvm::FunctionCallee RTLFn;
12559 llvm::Value *Args[DoacrossFinArgs];
12560
12561public:
12562 DoacrossCleanupTy(llvm::FunctionCallee RTLFn,
12563 ArrayRef<llvm::Value *> CallArgs)
12564 : RTLFn(RTLFn) {
12565 assert(CallArgs.size() == DoacrossFinArgs);
12566 std::copy(first: CallArgs.begin(), last: CallArgs.end(), result: std::begin(arr&: Args));
12567 }
12568 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
12569 if (!CGF.HaveInsertPoint())
12570 return;
12571 CGF.EmitRuntimeCall(callee: RTLFn, args: Args);
12572 }
12573};
12574} // namespace
12575
12576void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF,
12577 const OMPLoopDirective &D,
12578 ArrayRef<Expr *> NumIterations) {
12579 if (!CGF.HaveInsertPoint())
12580 return;
12581
12582 ASTContext &C = CGM.getContext();
12583 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
12584 RecordDecl *RD;
12585 if (KmpDimTy.isNull()) {
12586 // Build struct kmp_dim { // loop bounds info casted to kmp_int64
12587 // kmp_int64 lo; // lower
12588 // kmp_int64 up; // upper
12589 // kmp_int64 st; // stride
12590 // };
12591 RD = C.buildImplicitRecord(Name: "kmp_dim");
12592 RD->startDefinition();
12593 addFieldToRecordDecl(C, DC: RD, FieldTy: Int64Ty);
12594 addFieldToRecordDecl(C, DC: RD, FieldTy: Int64Ty);
12595 addFieldToRecordDecl(C, DC: RD, FieldTy: Int64Ty);
12596 RD->completeDefinition();
12597 KmpDimTy = C.getCanonicalTagType(TD: RD);
12598 } else {
12599 RD = KmpDimTy->castAsRecordDecl();
12600 }
12601 llvm::APInt Size(/*numBits=*/32, NumIterations.size());
12602 QualType ArrayTy = C.getConstantArrayType(EltTy: KmpDimTy, ArySize: Size, SizeExpr: nullptr,
12603 ASM: ArraySizeModifier::Normal, IndexTypeQuals: 0);
12604
12605 Address DimsAddr = CGF.CreateMemTemp(T: ArrayTy, Name: "dims");
12606 CGF.EmitNullInitialization(DestPtr: DimsAddr, Ty: ArrayTy);
12607 enum { LowerFD = 0, UpperFD, StrideFD };
12608 // Fill dims with data.
12609 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) {
12610 LValue DimsLVal = CGF.MakeAddrLValue(
12611 Addr: CGF.Builder.CreateConstArrayGEP(Addr: DimsAddr, Index: I), T: KmpDimTy);
12612 // dims.upper = num_iterations;
12613 LValue UpperLVal = CGF.EmitLValueForField(
12614 Base: DimsLVal, Field: *std::next(x: RD->field_begin(), n: UpperFD));
12615 llvm::Value *NumIterVal = CGF.EmitScalarConversion(
12616 Src: CGF.EmitScalarExpr(E: NumIterations[I]), SrcTy: NumIterations[I]->getType(),
12617 DstTy: Int64Ty, Loc: NumIterations[I]->getExprLoc());
12618 CGF.EmitStoreOfScalar(value: NumIterVal, lvalue: UpperLVal);
12619 // dims.stride = 1;
12620 LValue StrideLVal = CGF.EmitLValueForField(
12621 Base: DimsLVal, Field: *std::next(x: RD->field_begin(), n: StrideFD));
12622 CGF.EmitStoreOfScalar(value: llvm::ConstantInt::getSigned(Ty: CGM.Int64Ty, /*V=*/1),
12623 lvalue: StrideLVal);
12624 }
12625
12626 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid,
12627 // kmp_int32 num_dims, struct kmp_dim * dims);
12628 llvm::Value *Args[] = {
12629 emitUpdateLocation(CGF, Loc: D.getBeginLoc()),
12630 getThreadID(CGF, Loc: D.getBeginLoc()),
12631 llvm::ConstantInt::getSigned(Ty: CGM.Int32Ty, V: NumIterations.size()),
12632 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
12633 V: CGF.Builder.CreateConstArrayGEP(Addr: DimsAddr, Index: 0).emitRawPointer(CGF),
12634 DestTy: CGM.VoidPtrTy)};
12635
12636 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12637 M&: CGM.getModule(), FnID: OMPRTL___kmpc_doacross_init);
12638 CGF.EmitRuntimeCall(callee: RTLFn, args: Args);
12639 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = {
12640 emitUpdateLocation(CGF, Loc: D.getEndLoc()), getThreadID(CGF, Loc: D.getEndLoc())};
12641 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12642 M&: CGM.getModule(), FnID: OMPRTL___kmpc_doacross_fini);
12643 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(Kind: NormalAndEHCleanup, A: FiniRTLFn,
12644 A: llvm::ArrayRef(FiniArgs));
12645}
12646
12647template <typename T>
12648static void EmitDoacrossOrdered(CodeGenFunction &CGF, CodeGenModule &CGM,
12649 const T *C, llvm::Value *ULoc,
12650 llvm::Value *ThreadID) {
12651 QualType Int64Ty =
12652 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
12653 llvm::APInt Size(/*numBits=*/32, C->getNumLoops());
12654 QualType ArrayTy = CGM.getContext().getConstantArrayType(
12655 EltTy: Int64Ty, ArySize: Size, SizeExpr: nullptr, ASM: ArraySizeModifier::Normal, IndexTypeQuals: 0);
12656 Address CntAddr = CGF.CreateMemTemp(T: ArrayTy, Name: ".cnt.addr");
12657 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) {
12658 const Expr *CounterVal = C->getLoopData(I);
12659 assert(CounterVal);
12660 llvm::Value *CntVal = CGF.EmitScalarConversion(
12661 Src: CGF.EmitScalarExpr(E: CounterVal), SrcTy: CounterVal->getType(), DstTy: Int64Ty,
12662 Loc: CounterVal->getExprLoc());
12663 CGF.EmitStoreOfScalar(Value: CntVal, Addr: CGF.Builder.CreateConstArrayGEP(Addr: CntAddr, Index: I),
12664 /*Volatile=*/false, Ty: Int64Ty);
12665 }
12666 llvm::Value *Args[] = {
12667 ULoc, ThreadID,
12668 CGF.Builder.CreateConstArrayGEP(Addr: CntAddr, Index: 0).emitRawPointer(CGF)};
12669 llvm::FunctionCallee RTLFn;
12670 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
12671 OMPDoacrossKind<T> ODK;
12672 if (ODK.isSource(C)) {
12673 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(),
12674 FnID: OMPRTL___kmpc_doacross_post);
12675 } else {
12676 assert(ODK.isSink(C) && "Expect sink modifier.");
12677 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(),
12678 FnID: OMPRTL___kmpc_doacross_wait);
12679 }
12680 CGF.EmitRuntimeCall(callee: RTLFn, args: Args);
12681}
12682
12683void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
12684 const OMPDependClause *C) {
12685 return EmitDoacrossOrdered<OMPDependClause>(
12686 CGF, CGM, C, ULoc: emitUpdateLocation(CGF, Loc: C->getBeginLoc()),
12687 ThreadID: getThreadID(CGF, Loc: C->getBeginLoc()));
12688}
12689
12690void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
12691 const OMPDoacrossClause *C) {
12692 return EmitDoacrossOrdered<OMPDoacrossClause>(
12693 CGF, CGM, C, ULoc: emitUpdateLocation(CGF, Loc: C->getBeginLoc()),
12694 ThreadID: getThreadID(CGF, Loc: C->getBeginLoc()));
12695}
12696
12697void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc,
12698 llvm::FunctionCallee Callee,
12699 ArrayRef<llvm::Value *> Args) const {
12700 assert(Loc.isValid() && "Outlined function call location must be valid.");
12701 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, TemporaryLocation: Loc);
12702
12703 if (auto *Fn = dyn_cast<llvm::Function>(Val: Callee.getCallee())) {
12704 if (Fn->doesNotThrow()) {
12705 CGF.EmitNounwindRuntimeCall(callee: Fn, args: Args);
12706 return;
12707 }
12708 }
12709 CGF.EmitRuntimeCall(callee: Callee, args: Args);
12710}
12711
12712void CGOpenMPRuntime::emitOutlinedFunctionCall(
12713 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn,
12714 ArrayRef<llvm::Value *> Args) const {
12715 emitCall(CGF, Loc, Callee: OutlinedFn, Args);
12716}
12717
12718void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) {
12719 if (const auto *FD = dyn_cast<FunctionDecl>(Val: D))
12720 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD: FD))
12721 HasEmittedDeclareTargetRegion = true;
12722}
12723
12724Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF,
12725 const VarDecl *NativeParam,
12726 const VarDecl *TargetParam) const {
12727 return CGF.GetAddrOfLocalVar(VD: NativeParam);
12728}
12729
12730/// Return allocator value from expression, or return a null allocator (default
12731/// when no allocator specified).
12732static llvm::Value *getAllocatorVal(CodeGenFunction &CGF,
12733 const Expr *Allocator) {
12734 llvm::Value *AllocVal;
12735 if (Allocator) {
12736 AllocVal = CGF.EmitScalarExpr(E: Allocator);
12737 // According to the standard, the original allocator type is a enum
12738 // (integer). Convert to pointer type, if required.
12739 AllocVal = CGF.EmitScalarConversion(Src: AllocVal, SrcTy: Allocator->getType(),
12740 DstTy: CGF.getContext().VoidPtrTy,
12741 Loc: Allocator->getExprLoc());
12742 } else {
12743 // If no allocator specified, it defaults to the null allocator.
12744 AllocVal = llvm::Constant::getNullValue(
12745 Ty: CGF.CGM.getTypes().ConvertType(T: CGF.getContext().VoidPtrTy));
12746 }
12747 return AllocVal;
12748}
12749
12750/// Return the alignment from an allocate directive if present.
12751static llvm::Value *getAlignmentValue(CodeGenModule &CGM, const VarDecl *VD) {
12752 std::optional<CharUnits> AllocateAlignment = CGM.getOMPAllocateAlignment(VD);
12753
12754 if (!AllocateAlignment)
12755 return nullptr;
12756
12757 return llvm::ConstantInt::get(Ty: CGM.SizeTy, V: AllocateAlignment->getQuantity());
12758}
12759
12760Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF,
12761 const VarDecl *VD) {
12762 if (!VD)
12763 return Address::invalid();
12764 Address UntiedAddr = Address::invalid();
12765 Address UntiedRealAddr = Address::invalid();
12766 auto It = FunctionToUntiedTaskStackMap.find(Val: CGF.CurFn);
12767 if (It != FunctionToUntiedTaskStackMap.end()) {
12768 const UntiedLocalVarsAddressesMap &UntiedData =
12769 UntiedLocalVarsStack[It->second];
12770 auto I = UntiedData.find(Key: VD);
12771 if (I != UntiedData.end()) {
12772 UntiedAddr = I->second.first;
12773 UntiedRealAddr = I->second.second;
12774 }
12775 }
12776 const VarDecl *CVD = VD->getCanonicalDecl();
12777 if (CVD->hasAttr<OMPAllocateDeclAttr>()) {
12778 // Use the default allocation.
12779 if (!isAllocatableDecl(VD))
12780 return UntiedAddr;
12781 llvm::Value *Size;
12782 CharUnits Align = CGM.getContext().getDeclAlign(D: CVD);
12783 if (CVD->getType()->isVariablyModifiedType()) {
12784 Size = CGF.getTypeSize(Ty: CVD->getType());
12785 // Align the size: ((size + align - 1) / align) * align
12786 Size = CGF.Builder.CreateNUWAdd(
12787 LHS: Size, RHS: CGM.getSize(numChars: Align - CharUnits::fromQuantity(Quantity: 1)));
12788 Size = CGF.Builder.CreateUDiv(LHS: Size, RHS: CGM.getSize(numChars: Align));
12789 Size = CGF.Builder.CreateNUWMul(LHS: Size, RHS: CGM.getSize(numChars: Align));
12790 } else {
12791 CharUnits Sz = CGM.getContext().getTypeSizeInChars(T: CVD->getType());
12792 Size = CGM.getSize(numChars: Sz.alignTo(Align));
12793 }
12794 llvm::Value *ThreadID = getThreadID(CGF, Loc: CVD->getBeginLoc());
12795 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
12796 const Expr *Allocator = AA->getAllocator();
12797 llvm::Value *AllocVal = getAllocatorVal(CGF, Allocator);
12798 llvm::Value *Alignment = getAlignmentValue(CGM, VD: CVD);
12799 SmallVector<llvm::Value *, 4> Args;
12800 Args.push_back(Elt: ThreadID);
12801 if (Alignment)
12802 Args.push_back(Elt: Alignment);
12803 Args.push_back(Elt: Size);
12804 Args.push_back(Elt: AllocVal);
12805 llvm::omp::RuntimeFunction FnID =
12806 Alignment ? OMPRTL___kmpc_aligned_alloc : OMPRTL___kmpc_alloc;
12807 llvm::Value *Addr = CGF.EmitRuntimeCall(
12808 callee: OMPBuilder.getOrCreateRuntimeFunction(M&: CGM.getModule(), FnID), args: Args,
12809 name: getName(Parts: {CVD->getName(), ".void.addr"}));
12810 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12811 M&: CGM.getModule(), FnID: OMPRTL___kmpc_free);
12812 QualType Ty = CGM.getContext().getPointerType(T: CVD->getType());
12813 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
12814 V: Addr, DestTy: CGF.ConvertTypeForMem(T: Ty), Name: getName(Parts: {CVD->getName(), ".addr"}));
12815 if (UntiedAddr.isValid())
12816 CGF.EmitStoreOfScalar(Value: Addr, Addr: UntiedAddr, /*Volatile=*/false, Ty);
12817
12818 // Cleanup action for allocate support.
12819 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup {
12820 llvm::FunctionCallee RTLFn;
12821 SourceLocation::UIntTy LocEncoding;
12822 Address Addr;
12823 const Expr *AllocExpr;
12824
12825 public:
12826 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn,
12827 SourceLocation::UIntTy LocEncoding, Address Addr,
12828 const Expr *AllocExpr)
12829 : RTLFn(RTLFn), LocEncoding(LocEncoding), Addr(Addr),
12830 AllocExpr(AllocExpr) {}
12831 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
12832 if (!CGF.HaveInsertPoint())
12833 return;
12834 llvm::Value *Args[3];
12835 Args[0] = CGF.CGM.getOpenMPRuntime().getThreadID(
12836 CGF, Loc: SourceLocation::getFromRawEncoding(Encoding: LocEncoding));
12837 Args[1] = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
12838 V: Addr.emitRawPointer(CGF), DestTy: CGF.VoidPtrTy);
12839 llvm::Value *AllocVal = getAllocatorVal(CGF, Allocator: AllocExpr);
12840 Args[2] = AllocVal;
12841 CGF.EmitRuntimeCall(callee: RTLFn, args: Args);
12842 }
12843 };
12844 Address VDAddr =
12845 UntiedRealAddr.isValid()
12846 ? UntiedRealAddr
12847 : Address(Addr, CGF.ConvertTypeForMem(T: CVD->getType()), Align);
12848 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(
12849 Kind: NormalAndEHCleanup, A: FiniRTLFn, A: CVD->getLocation().getRawEncoding(),
12850 A: VDAddr, A: Allocator);
12851 if (UntiedRealAddr.isValid())
12852 if (auto *Region =
12853 dyn_cast_or_null<CGOpenMPRegionInfo>(Val: CGF.CapturedStmtInfo))
12854 Region->emitUntiedSwitch(CGF);
12855 return VDAddr;
12856 }
12857 return UntiedAddr;
12858}
12859
12860bool CGOpenMPRuntime::isLocalVarInUntiedTask(CodeGenFunction &CGF,
12861 const VarDecl *VD) const {
12862 auto It = FunctionToUntiedTaskStackMap.find(Val: CGF.CurFn);
12863 if (It == FunctionToUntiedTaskStackMap.end())
12864 return false;
12865 return UntiedLocalVarsStack[It->second].count(Key: VD) > 0;
12866}
12867
12868CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII(
12869 CodeGenModule &CGM, const OMPLoopDirective &S)
12870 : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) {
12871 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12872 if (!NeedToPush)
12873 return;
12874 NontemporalDeclsSet &DS =
12875 CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back();
12876 for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) {
12877 for (const Stmt *Ref : C->private_refs()) {
12878 const auto *SimpleRefExpr = cast<Expr>(Val: Ref)->IgnoreParenImpCasts();
12879 const ValueDecl *VD;
12880 if (const auto *DRE = dyn_cast<DeclRefExpr>(Val: SimpleRefExpr)) {
12881 VD = DRE->getDecl();
12882 } else {
12883 const auto *ME = cast<MemberExpr>(Val: SimpleRefExpr);
12884 assert((ME->isImplicitCXXThis() ||
12885 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) &&
12886 "Expected member of current class.");
12887 VD = ME->getMemberDecl();
12888 }
12889 DS.insert(V: VD);
12890 }
12891 }
12892}
12893
12894CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() {
12895 if (!NeedToPush)
12896 return;
12897 CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back();
12898}
12899
12900CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::UntiedTaskLocalDeclsRAII(
12901 CodeGenFunction &CGF,
12902 const llvm::MapVector<CanonicalDeclPtr<const VarDecl>,
12903 std::pair<Address, Address>> &LocalVars)
12904 : CGM(CGF.CGM), NeedToPush(!LocalVars.empty()) {
12905 if (!NeedToPush)
12906 return;
12907 CGM.getOpenMPRuntime().FunctionToUntiedTaskStackMap.try_emplace(
12908 Key: CGF.CurFn, Args: CGM.getOpenMPRuntime().UntiedLocalVarsStack.size());
12909 CGM.getOpenMPRuntime().UntiedLocalVarsStack.push_back(Elt: LocalVars);
12910}
12911
12912CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::~UntiedTaskLocalDeclsRAII() {
12913 if (!NeedToPush)
12914 return;
12915 CGM.getOpenMPRuntime().UntiedLocalVarsStack.pop_back();
12916}
12917
12918bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const {
12919 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12920
12921 return llvm::any_of(
12922 Range&: CGM.getOpenMPRuntime().NontemporalDeclsStack,
12923 P: [VD](const NontemporalDeclsSet &Set) { return Set.contains(V: VD); });
12924}
12925
12926void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis(
12927 const OMPExecutableDirective &S,
12928 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled)
12929 const {
12930 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs;
12931 // Vars in target/task regions must be excluded completely.
12932 if (isOpenMPTargetExecutionDirective(DKind: S.getDirectiveKind()) ||
12933 isOpenMPTaskingDirective(Kind: S.getDirectiveKind())) {
12934 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
12935 getOpenMPCaptureRegions(CaptureRegions, DKind: S.getDirectiveKind());
12936 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: CaptureRegions.front());
12937 for (const CapturedStmt::Capture &Cap : CS->captures()) {
12938 if (Cap.capturesVariable() || Cap.capturesVariableByCopy())
12939 NeedToCheckForLPCs.insert(V: Cap.getCapturedVar());
12940 }
12941 }
12942 // Exclude vars in private clauses.
12943 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) {
12944 for (const Expr *Ref : C->varlist()) {
12945 if (!Ref->getType()->isScalarType())
12946 continue;
12947 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
12948 if (!DRE)
12949 continue;
12950 NeedToCheckForLPCs.insert(V: DRE->getDecl());
12951 }
12952 }
12953 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
12954 for (const Expr *Ref : C->varlist()) {
12955 if (!Ref->getType()->isScalarType())
12956 continue;
12957 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
12958 if (!DRE)
12959 continue;
12960 NeedToCheckForLPCs.insert(V: DRE->getDecl());
12961 }
12962 }
12963 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
12964 for (const Expr *Ref : C->varlist()) {
12965 if (!Ref->getType()->isScalarType())
12966 continue;
12967 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
12968 if (!DRE)
12969 continue;
12970 NeedToCheckForLPCs.insert(V: DRE->getDecl());
12971 }
12972 }
12973 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
12974 for (const Expr *Ref : C->varlist()) {
12975 if (!Ref->getType()->isScalarType())
12976 continue;
12977 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
12978 if (!DRE)
12979 continue;
12980 NeedToCheckForLPCs.insert(V: DRE->getDecl());
12981 }
12982 }
12983 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) {
12984 for (const Expr *Ref : C->varlist()) {
12985 if (!Ref->getType()->isScalarType())
12986 continue;
12987 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
12988 if (!DRE)
12989 continue;
12990 NeedToCheckForLPCs.insert(V: DRE->getDecl());
12991 }
12992 }
12993 for (const Decl *VD : NeedToCheckForLPCs) {
12994 for (const LastprivateConditionalData &Data :
12995 llvm::reverse(C&: CGM.getOpenMPRuntime().LastprivateConditionalStack)) {
12996 if (Data.DeclToUniqueName.count(Key: VD) > 0) {
12997 if (!Data.Disabled)
12998 NeedToAddForLPCsAsDisabled.insert(V: VD);
12999 break;
13000 }
13001 }
13002 }
13003}
13004
13005CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
13006 CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal)
13007 : CGM(CGF.CGM),
13008 Action((CGM.getLangOpts().OpenMP >= 50 &&
13009 llvm::any_of(Range: S.getClausesOfKind<OMPLastprivateClause>(),
13010 P: [](const OMPLastprivateClause *C) {
13011 return C->getKind() ==
13012 OMPC_LASTPRIVATE_conditional;
13013 }))
13014 ? ActionToDo::PushAsLastprivateConditional
13015 : ActionToDo::DoNotPush) {
13016 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
13017 if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush)
13018 return;
13019 assert(Action == ActionToDo::PushAsLastprivateConditional &&
13020 "Expected a push action.");
13021 LastprivateConditionalData &Data =
13022 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
13023 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
13024 if (C->getKind() != OMPC_LASTPRIVATE_conditional)
13025 continue;
13026
13027 for (const Expr *Ref : C->varlist()) {
13028 Data.DeclToUniqueName.insert(KV: std::make_pair(
13029 x: cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts())->getDecl(),
13030 y: SmallString<16>(generateUniqueName(CGM, Prefix: "pl_cond", Ref))));
13031 }
13032 }
13033 Data.IVLVal = IVLVal;
13034 Data.Fn = CGF.CurFn;
13035}
13036
13037CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
13038 CodeGenFunction &CGF, const OMPExecutableDirective &S)
13039 : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) {
13040 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
13041 if (CGM.getLangOpts().OpenMP < 50)
13042 return;
13043 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled;
13044 tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled);
13045 if (!NeedToAddForLPCsAsDisabled.empty()) {
13046 Action = ActionToDo::DisableLastprivateConditional;
13047 LastprivateConditionalData &Data =
13048 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
13049 for (const Decl *VD : NeedToAddForLPCsAsDisabled)
13050 Data.DeclToUniqueName.try_emplace(Key: VD);
13051 Data.Fn = CGF.CurFn;
13052 Data.Disabled = true;
13053 }
13054}
13055
13056CGOpenMPRuntime::LastprivateConditionalRAII
13057CGOpenMPRuntime::LastprivateConditionalRAII::disable(
13058 CodeGenFunction &CGF, const OMPExecutableDirective &S) {
13059 return LastprivateConditionalRAII(CGF, S);
13060}
13061
13062CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() {
13063 if (CGM.getLangOpts().OpenMP < 50)
13064 return;
13065 if (Action == ActionToDo::DisableLastprivateConditional) {
13066 assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
13067 "Expected list of disabled private vars.");
13068 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
13069 }
13070 if (Action == ActionToDo::PushAsLastprivateConditional) {
13071 assert(
13072 !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
13073 "Expected list of lastprivate conditional vars.");
13074 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
13075 }
13076}
13077
13078Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF,
13079 const VarDecl *VD) {
13080 ASTContext &C = CGM.getContext();
13081 auto I = LastprivateConditionalToTypes.try_emplace(Key: CGF.CurFn).first;
13082 QualType NewType;
13083 const FieldDecl *VDField;
13084 const FieldDecl *FiredField;
13085 LValue BaseLVal;
13086 auto VI = I->getSecond().find(Val: VD);
13087 if (VI == I->getSecond().end()) {
13088 RecordDecl *RD = C.buildImplicitRecord(Name: "lasprivate.conditional");
13089 RD->startDefinition();
13090 VDField = addFieldToRecordDecl(C, DC: RD, FieldTy: VD->getType().getNonReferenceType());
13091 FiredField = addFieldToRecordDecl(C, DC: RD, FieldTy: C.CharTy);
13092 RD->completeDefinition();
13093 NewType = C.getCanonicalTagType(TD: RD);
13094 Address Addr = CGF.CreateMemTemp(T: NewType, Align: C.getDeclAlign(D: VD), Name: VD->getName());
13095 BaseLVal = CGF.MakeAddrLValue(Addr, T: NewType, Source: AlignmentSource::Decl);
13096 I->getSecond().try_emplace(Key: VD, Args&: NewType, Args&: VDField, Args&: FiredField, Args&: BaseLVal);
13097 } else {
13098 NewType = std::get<0>(t&: VI->getSecond());
13099 VDField = std::get<1>(t&: VI->getSecond());
13100 FiredField = std::get<2>(t&: VI->getSecond());
13101 BaseLVal = std::get<3>(t&: VI->getSecond());
13102 }
13103 LValue FiredLVal =
13104 CGF.EmitLValueForField(Base: BaseLVal, Field: FiredField);
13105 CGF.EmitStoreOfScalar(
13106 value: llvm::ConstantInt::getNullValue(Ty: CGF.ConvertTypeForMem(T: C.CharTy)),
13107 lvalue: FiredLVal);
13108 return CGF.EmitLValueForField(Base: BaseLVal, Field: VDField).getAddress();
13109}
13110
13111namespace {
13112/// Checks if the lastprivate conditional variable is referenced in LHS.
13113class LastprivateConditionalRefChecker final
13114 : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> {
13115 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM;
13116 const Expr *FoundE = nullptr;
13117 const Decl *FoundD = nullptr;
13118 StringRef UniqueDeclName;
13119 LValue IVLVal;
13120 llvm::Function *FoundFn = nullptr;
13121 SourceLocation Loc;
13122
13123public:
13124 bool VisitDeclRefExpr(const DeclRefExpr *E) {
13125 for (const CGOpenMPRuntime::LastprivateConditionalData &D :
13126 llvm::reverse(C&: LPM)) {
13127 auto It = D.DeclToUniqueName.find(Key: E->getDecl());
13128 if (It == D.DeclToUniqueName.end())
13129 continue;
13130 if (D.Disabled)
13131 return false;
13132 FoundE = E;
13133 FoundD = E->getDecl()->getCanonicalDecl();
13134 UniqueDeclName = It->second;
13135 IVLVal = D.IVLVal;
13136 FoundFn = D.Fn;
13137 break;
13138 }
13139 return FoundE == E;
13140 }
13141 bool VisitMemberExpr(const MemberExpr *E) {
13142 if (!CodeGenFunction::IsWrappedCXXThis(E: E->getBase()))
13143 return false;
13144 for (const CGOpenMPRuntime::LastprivateConditionalData &D :
13145 llvm::reverse(C&: LPM)) {
13146 auto It = D.DeclToUniqueName.find(Key: E->getMemberDecl());
13147 if (It == D.DeclToUniqueName.end())
13148 continue;
13149 if (D.Disabled)
13150 return false;
13151 FoundE = E;
13152 FoundD = E->getMemberDecl()->getCanonicalDecl();
13153 UniqueDeclName = It->second;
13154 IVLVal = D.IVLVal;
13155 FoundFn = D.Fn;
13156 break;
13157 }
13158 return FoundE == E;
13159 }
13160 bool VisitStmt(const Stmt *S) {
13161 for (const Stmt *Child : S->children()) {
13162 if (!Child)
13163 continue;
13164 if (const auto *E = dyn_cast<Expr>(Val: Child))
13165 if (!E->isGLValue())
13166 continue;
13167 if (Visit(S: Child))
13168 return true;
13169 }
13170 return false;
13171 }
13172 explicit LastprivateConditionalRefChecker(
13173 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM)
13174 : LPM(LPM) {}
13175 std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *>
13176 getFoundData() const {
13177 return std::make_tuple(args: FoundE, args: FoundD, args: UniqueDeclName, args: IVLVal, args: FoundFn);
13178 }
13179};
13180} // namespace
13181
13182void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF,
13183 LValue IVLVal,
13184 StringRef UniqueDeclName,
13185 LValue LVal,
13186 SourceLocation Loc) {
13187 // Last updated loop counter for the lastprivate conditional var.
13188 // int<xx> last_iv = 0;
13189 llvm::Type *LLIVTy = CGF.ConvertTypeForMem(T: IVLVal.getType());
13190 llvm::Constant *LastIV = OMPBuilder.getOrCreateInternalVariable(
13191 Ty: LLIVTy, Name: getName(Parts: {UniqueDeclName, "iv"}));
13192 cast<llvm::GlobalVariable>(Val: LastIV)->setAlignment(
13193 IVLVal.getAlignment().getAsAlign());
13194 LValue LastIVLVal =
13195 CGF.MakeNaturalAlignRawAddrLValue(V: LastIV, T: IVLVal.getType());
13196
13197 // Last value of the lastprivate conditional.
13198 // decltype(priv_a) last_a;
13199 llvm::GlobalVariable *Last = OMPBuilder.getOrCreateInternalVariable(
13200 Ty: CGF.ConvertTypeForMem(T: LVal.getType()), Name: UniqueDeclName);
13201 cast<llvm::GlobalVariable>(Val: Last)->setAlignment(
13202 LVal.getAlignment().getAsAlign());
13203 LValue LastLVal =
13204 CGF.MakeRawAddrLValue(V: Last, T: LVal.getType(), Alignment: LVal.getAlignment());
13205
13206 // Global loop counter. Required to handle inner parallel-for regions.
13207 // iv
13208 llvm::Value *IVVal = CGF.EmitLoadOfScalar(lvalue: IVLVal, Loc);
13209
13210 // #pragma omp critical(a)
13211 // if (last_iv <= iv) {
13212 // last_iv = iv;
13213 // last_a = priv_a;
13214 // }
13215 auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal,
13216 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
13217 Action.Enter(CGF);
13218 llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(lvalue: LastIVLVal, Loc);
13219 // (last_iv <= iv) ? Check if the variable is updated and store new
13220 // value in global var.
13221 llvm::Value *CmpRes;
13222 if (IVLVal.getType()->isSignedIntegerType()) {
13223 CmpRes = CGF.Builder.CreateICmpSLE(LHS: LastIVVal, RHS: IVVal);
13224 } else {
13225 assert(IVLVal.getType()->isUnsignedIntegerType() &&
13226 "Loop iteration variable must be integer.");
13227 CmpRes = CGF.Builder.CreateICmpULE(LHS: LastIVVal, RHS: IVVal);
13228 }
13229 llvm::BasicBlock *ThenBB = CGF.createBasicBlock(name: "lp_cond_then");
13230 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(name: "lp_cond_exit");
13231 CGF.Builder.CreateCondBr(Cond: CmpRes, True: ThenBB, False: ExitBB);
13232 // {
13233 CGF.EmitBlock(BB: ThenBB);
13234
13235 // last_iv = iv;
13236 CGF.EmitStoreOfScalar(value: IVVal, lvalue: LastIVLVal);
13237
13238 // last_a = priv_a;
13239 switch (CGF.getEvaluationKind(T: LVal.getType())) {
13240 case TEK_Scalar: {
13241 llvm::Value *PrivVal = CGF.EmitLoadOfScalar(lvalue: LVal, Loc);
13242 CGF.EmitStoreOfScalar(value: PrivVal, lvalue: LastLVal);
13243 break;
13244 }
13245 case TEK_Complex: {
13246 CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(src: LVal, loc: Loc);
13247 CGF.EmitStoreOfComplex(V: PrivVal, dest: LastLVal, /*isInit=*/false);
13248 break;
13249 }
13250 case TEK_Aggregate:
13251 llvm_unreachable(
13252 "Aggregates are not supported in lastprivate conditional.");
13253 }
13254 // }
13255 CGF.EmitBranch(Block: ExitBB);
13256 // There is no need to emit line number for unconditional branch.
13257 (void)ApplyDebugLocation::CreateEmpty(CGF);
13258 CGF.EmitBlock(BB: ExitBB, /*IsFinished=*/true);
13259 };
13260
13261 if (CGM.getLangOpts().OpenMPSimd) {
13262 // Do not emit as a critical region as no parallel region could be emitted.
13263 RegionCodeGenTy ThenRCG(CodeGen);
13264 ThenRCG(CGF);
13265 } else {
13266 emitCriticalRegion(CGF, CriticalName: UniqueDeclName, CriticalOpGen: CodeGen, Loc);
13267 }
13268}
13269
13270void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF,
13271 const Expr *LHS) {
13272 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
13273 return;
13274 LastprivateConditionalRefChecker Checker(LastprivateConditionalStack);
13275 if (!Checker.Visit(S: LHS))
13276 return;
13277 const Expr *FoundE;
13278 const Decl *FoundD;
13279 StringRef UniqueDeclName;
13280 LValue IVLVal;
13281 llvm::Function *FoundFn;
13282 std::tie(args&: FoundE, args&: FoundD, args&: UniqueDeclName, args&: IVLVal, args&: FoundFn) =
13283 Checker.getFoundData();
13284 if (FoundFn != CGF.CurFn) {
13285 // Special codegen for inner parallel regions.
13286 // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1;
13287 auto It = LastprivateConditionalToTypes[FoundFn].find(Val: FoundD);
13288 assert(It != LastprivateConditionalToTypes[FoundFn].end() &&
13289 "Lastprivate conditional is not found in outer region.");
13290 QualType StructTy = std::get<0>(t&: It->getSecond());
13291 const FieldDecl* FiredDecl = std::get<2>(t&: It->getSecond());
13292 LValue PrivLVal = CGF.EmitLValue(E: FoundE);
13293 Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
13294 Addr: PrivLVal.getAddress(),
13295 Ty: CGF.ConvertTypeForMem(T: CGF.getContext().getPointerType(T: StructTy)),
13296 ElementTy: CGF.ConvertTypeForMem(T: StructTy));
13297 LValue BaseLVal =
13298 CGF.MakeAddrLValue(Addr: StructAddr, T: StructTy, Source: AlignmentSource::Decl);
13299 LValue FiredLVal = CGF.EmitLValueForField(Base: BaseLVal, Field: FiredDecl);
13300 CGF.EmitAtomicStore(rvalue: RValue::get(V: llvm::ConstantInt::get(
13301 Ty: CGF.ConvertTypeForMem(T: FiredDecl->getType()), V: 1)),
13302 lvalue: FiredLVal, AO: llvm::AtomicOrdering::Unordered,
13303 /*IsVolatile=*/true, /*isInit=*/false);
13304 return;
13305 }
13306
13307 // Private address of the lastprivate conditional in the current context.
13308 // priv_a
13309 LValue LVal = CGF.EmitLValue(E: FoundE);
13310 emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal,
13311 Loc: FoundE->getExprLoc());
13312}
13313
13314void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional(
13315 CodeGenFunction &CGF, const OMPExecutableDirective &D,
13316 const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) {
13317 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
13318 return;
13319 auto Range = llvm::reverse(C&: LastprivateConditionalStack);
13320 auto It = llvm::find_if(
13321 Range, P: [](const LastprivateConditionalData &D) { return !D.Disabled; });
13322 if (It == Range.end() || It->Fn != CGF.CurFn)
13323 return;
13324 auto LPCI = LastprivateConditionalToTypes.find(Val: It->Fn);
13325 assert(LPCI != LastprivateConditionalToTypes.end() &&
13326 "Lastprivates must be registered already.");
13327 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
13328 getOpenMPCaptureRegions(CaptureRegions, DKind: D.getDirectiveKind());
13329 const CapturedStmt *CS = D.getCapturedStmt(RegionKind: CaptureRegions.back());
13330 for (const auto &Pair : It->DeclToUniqueName) {
13331 const auto *VD = cast<VarDecl>(Val: Pair.first->getCanonicalDecl());
13332 if (!CS->capturesVariable(Var: VD) || IgnoredDecls.contains(V: VD))
13333 continue;
13334 auto I = LPCI->getSecond().find(Val: Pair.first);
13335 assert(I != LPCI->getSecond().end() &&
13336 "Lastprivate must be rehistered already.");
13337 // bool Cmp = priv_a.Fired != 0;
13338 LValue BaseLVal = std::get<3>(t&: I->getSecond());
13339 LValue FiredLVal =
13340 CGF.EmitLValueForField(Base: BaseLVal, Field: std::get<2>(t&: I->getSecond()));
13341 llvm::Value *Res = CGF.EmitLoadOfScalar(lvalue: FiredLVal, Loc: D.getBeginLoc());
13342 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Arg: Res);
13343 llvm::BasicBlock *ThenBB = CGF.createBasicBlock(name: "lpc.then");
13344 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(name: "lpc.done");
13345 // if (Cmp) {
13346 CGF.Builder.CreateCondBr(Cond: Cmp, True: ThenBB, False: DoneBB);
13347 CGF.EmitBlock(BB: ThenBB);
13348 Address Addr = CGF.GetAddrOfLocalVar(VD);
13349 LValue LVal;
13350 if (VD->getType()->isReferenceType())
13351 LVal = CGF.EmitLoadOfReferenceLValue(RefAddr: Addr, RefTy: VD->getType(),
13352 Source: AlignmentSource::Decl);
13353 else
13354 LVal = CGF.MakeAddrLValue(Addr, T: VD->getType().getNonReferenceType(),
13355 Source: AlignmentSource::Decl);
13356 emitLastprivateConditionalUpdate(CGF, IVLVal: It->IVLVal, UniqueDeclName: Pair.second, LVal,
13357 Loc: D.getBeginLoc());
13358 auto AL = ApplyDebugLocation::CreateArtificial(CGF);
13359 CGF.EmitBlock(BB: DoneBB, /*IsFinal=*/IsFinished: true);
13360 // }
13361 }
13362}
13363
13364void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate(
13365 CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD,
13366 SourceLocation Loc) {
13367 if (CGF.getLangOpts().OpenMP < 50)
13368 return;
13369 auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(Key: VD);
13370 assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() &&
13371 "Unknown lastprivate conditional variable.");
13372 StringRef UniqueName = It->second;
13373 llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(Name: UniqueName);
13374 // The variable was not updated in the region - exit.
13375 if (!GV)
13376 return;
13377 LValue LPLVal = CGF.MakeRawAddrLValue(
13378 V: GV, T: PrivLVal.getType().getNonReferenceType(), Alignment: PrivLVal.getAlignment());
13379 llvm::Value *Res = CGF.EmitLoadOfScalar(lvalue: LPLVal, Loc);
13380 CGF.EmitStoreOfScalar(value: Res, lvalue: PrivLVal);
13381}
13382
13383llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction(
13384 CodeGenFunction &CGF, const OMPExecutableDirective &D,
13385 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
13386 const RegionCodeGenTy &CodeGen) {
13387 llvm_unreachable("Not supported in SIMD-only mode");
13388}
13389
13390llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction(
13391 CodeGenFunction &CGF, const OMPExecutableDirective &D,
13392 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
13393 const RegionCodeGenTy &CodeGen) {
13394 llvm_unreachable("Not supported in SIMD-only mode");
13395}
13396
13397llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction(
13398 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
13399 const VarDecl *PartIDVar, const VarDecl *TaskTVar,
13400 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
13401 bool Tied, unsigned &NumberOfParts) {
13402 llvm_unreachable("Not supported in SIMD-only mode");
13403}
13404
13405void CGOpenMPSIMDRuntime::emitParallelCall(
13406 CodeGenFunction &CGF, SourceLocation Loc, llvm::Function *OutlinedFn,
13407 ArrayRef<llvm::Value *> CapturedVars, const Expr *IfCond,
13408 llvm::Value *NumThreads, OpenMPNumThreadsClauseModifier NumThreadsModifier,
13409 OpenMPSeverityClauseKind Severity, const Expr *Message) {
13410 llvm_unreachable("Not supported in SIMD-only mode");
13411}
13412
13413void CGOpenMPSIMDRuntime::emitCriticalRegion(
13414 CodeGenFunction &CGF, StringRef CriticalName,
13415 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc,
13416 const Expr *Hint) {
13417 llvm_unreachable("Not supported in SIMD-only mode");
13418}
13419
13420void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF,
13421 const RegionCodeGenTy &MasterOpGen,
13422 SourceLocation Loc) {
13423 llvm_unreachable("Not supported in SIMD-only mode");
13424}
13425
13426void CGOpenMPSIMDRuntime::emitMaskedRegion(CodeGenFunction &CGF,
13427 const RegionCodeGenTy &MasterOpGen,
13428 SourceLocation Loc,
13429 const Expr *Filter) {
13430 llvm_unreachable("Not supported in SIMD-only mode");
13431}
13432
13433void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
13434 SourceLocation Loc) {
13435 llvm_unreachable("Not supported in SIMD-only mode");
13436}
13437
13438void CGOpenMPSIMDRuntime::emitTaskgroupRegion(
13439 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen,
13440 SourceLocation Loc) {
13441 llvm_unreachable("Not supported in SIMD-only mode");
13442}
13443
13444void CGOpenMPSIMDRuntime::emitSingleRegion(
13445 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen,
13446 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars,
13447 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs,
13448 ArrayRef<const Expr *> AssignmentOps) {
13449 llvm_unreachable("Not supported in SIMD-only mode");
13450}
13451
13452void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF,
13453 const RegionCodeGenTy &OrderedOpGen,
13454 SourceLocation Loc,
13455 bool IsThreads) {
13456 llvm_unreachable("Not supported in SIMD-only mode");
13457}
13458
13459void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF,
13460 SourceLocation Loc,
13461 OpenMPDirectiveKind Kind,
13462 bool EmitChecks,
13463 bool ForceSimpleCall) {
13464 llvm_unreachable("Not supported in SIMD-only mode");
13465}
13466
13467void CGOpenMPSIMDRuntime::emitForDispatchInit(
13468 CodeGenFunction &CGF, SourceLocation Loc,
13469 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
13470 bool Ordered, const DispatchRTInput &DispatchValues) {
13471 llvm_unreachable("Not supported in SIMD-only mode");
13472}
13473
13474void CGOpenMPSIMDRuntime::emitForDispatchDeinit(CodeGenFunction &CGF,
13475 SourceLocation Loc) {
13476 llvm_unreachable("Not supported in SIMD-only mode");
13477}
13478
13479void CGOpenMPSIMDRuntime::emitForStaticInit(
13480 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind,
13481 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) {
13482 llvm_unreachable("Not supported in SIMD-only mode");
13483}
13484
13485void CGOpenMPSIMDRuntime::emitDistributeStaticInit(
13486 CodeGenFunction &CGF, SourceLocation Loc,
13487 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) {
13488 llvm_unreachable("Not supported in SIMD-only mode");
13489}
13490
13491void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
13492 SourceLocation Loc,
13493 unsigned IVSize,
13494 bool IVSigned) {
13495 llvm_unreachable("Not supported in SIMD-only mode");
13496}
13497
13498void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF,
13499 SourceLocation Loc,
13500 OpenMPDirectiveKind DKind) {
13501 llvm_unreachable("Not supported in SIMD-only mode");
13502}
13503
13504llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF,
13505 SourceLocation Loc,
13506 unsigned IVSize, bool IVSigned,
13507 Address IL, Address LB,
13508 Address UB, Address ST) {
13509 llvm_unreachable("Not supported in SIMD-only mode");
13510}
13511
13512void CGOpenMPSIMDRuntime::emitNumThreadsClause(
13513 CodeGenFunction &CGF, llvm::Value *NumThreads, SourceLocation Loc,
13514 OpenMPNumThreadsClauseModifier Modifier, OpenMPSeverityClauseKind Severity,
13515 SourceLocation SeverityLoc, const Expr *Message,
13516 SourceLocation MessageLoc) {
13517 llvm_unreachable("Not supported in SIMD-only mode");
13518}
13519
13520void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF,
13521 ProcBindKind ProcBind,
13522 SourceLocation Loc) {
13523 llvm_unreachable("Not supported in SIMD-only mode");
13524}
13525
13526Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
13527 const VarDecl *VD,
13528 Address VDAddr,
13529 SourceLocation Loc) {
13530 llvm_unreachable("Not supported in SIMD-only mode");
13531}
13532
13533llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition(
13534 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit,
13535 CodeGenFunction *CGF) {
13536 llvm_unreachable("Not supported in SIMD-only mode");
13537}
13538
13539Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate(
13540 CodeGenFunction &CGF, QualType VarType, StringRef Name) {
13541 llvm_unreachable("Not supported in SIMD-only mode");
13542}
13543
13544void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF,
13545 ArrayRef<const Expr *> Vars,
13546 SourceLocation Loc,
13547 llvm::AtomicOrdering AO) {
13548 llvm_unreachable("Not supported in SIMD-only mode");
13549}
13550
13551void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
13552 const OMPExecutableDirective &D,
13553 llvm::Function *TaskFunction,
13554 QualType SharedsTy, Address Shareds,
13555 const Expr *IfCond,
13556 const OMPTaskDataTy &Data) {
13557 llvm_unreachable("Not supported in SIMD-only mode");
13558}
13559
13560void CGOpenMPSIMDRuntime::emitTaskLoopCall(
13561 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D,
13562 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds,
13563 const Expr *IfCond, const OMPTaskDataTy &Data) {
13564 llvm_unreachable("Not supported in SIMD-only mode");
13565}
13566
13567void CGOpenMPSIMDRuntime::emitReduction(
13568 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates,
13569 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
13570 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) {
13571 assert(Options.SimpleReduction && "Only simple reduction is expected.");
13572 CGOpenMPRuntime::emitReduction(CGF, Loc, OrgPrivates: Privates, OrgLHSExprs: LHSExprs, OrgRHSExprs: RHSExprs,
13573 OrgReductionOps: ReductionOps, Options);
13574}
13575
13576llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit(
13577 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
13578 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
13579 llvm_unreachable("Not supported in SIMD-only mode");
13580}
13581
13582void CGOpenMPSIMDRuntime::emitTaskReductionFini(CodeGenFunction &CGF,
13583 SourceLocation Loc,
13584 bool IsWorksharingReduction) {
13585 llvm_unreachable("Not supported in SIMD-only mode");
13586}
13587
13588void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
13589 SourceLocation Loc,
13590 ReductionCodeGen &RCG,
13591 unsigned N) {
13592 llvm_unreachable("Not supported in SIMD-only mode");
13593}
13594
13595Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF,
13596 SourceLocation Loc,
13597 llvm::Value *ReductionsPtr,
13598 LValue SharedLVal) {
13599 llvm_unreachable("Not supported in SIMD-only mode");
13600}
13601
13602void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
13603 SourceLocation Loc,
13604 const OMPTaskDataTy &Data) {
13605 llvm_unreachable("Not supported in SIMD-only mode");
13606}
13607
13608void CGOpenMPSIMDRuntime::emitCancellationPointCall(
13609 CodeGenFunction &CGF, SourceLocation Loc,
13610 OpenMPDirectiveKind CancelRegion) {
13611 llvm_unreachable("Not supported in SIMD-only mode");
13612}
13613
13614void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF,
13615 SourceLocation Loc, const Expr *IfCond,
13616 OpenMPDirectiveKind CancelRegion) {
13617 llvm_unreachable("Not supported in SIMD-only mode");
13618}
13619
13620void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction(
13621 const OMPExecutableDirective &D, StringRef ParentName,
13622 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
13623 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
13624 llvm_unreachable("Not supported in SIMD-only mode");
13625}
13626
13627void CGOpenMPSIMDRuntime::emitTargetCall(
13628 CodeGenFunction &CGF, const OMPExecutableDirective &D,
13629 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
13630 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
13631 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
13632 const OMPLoopDirective &D)>
13633 SizeEmitter) {
13634 llvm_unreachable("Not supported in SIMD-only mode");
13635}
13636
13637bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) {
13638 llvm_unreachable("Not supported in SIMD-only mode");
13639}
13640
13641bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
13642 llvm_unreachable("Not supported in SIMD-only mode");
13643}
13644
13645bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) {
13646 return false;
13647}
13648
13649void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF,
13650 const OMPExecutableDirective &D,
13651 SourceLocation Loc,
13652 llvm::Function *OutlinedFn,
13653 ArrayRef<llvm::Value *> CapturedVars) {
13654 llvm_unreachable("Not supported in SIMD-only mode");
13655}
13656
13657void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
13658 const Expr *NumTeams,
13659 const Expr *ThreadLimit,
13660 SourceLocation Loc) {
13661 llvm_unreachable("Not supported in SIMD-only mode");
13662}
13663
13664void CGOpenMPSIMDRuntime::emitTargetDataCalls(
13665 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
13666 const Expr *Device, const RegionCodeGenTy &CodeGen,
13667 CGOpenMPRuntime::TargetDataInfo &Info) {
13668 llvm_unreachable("Not supported in SIMD-only mode");
13669}
13670
13671void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall(
13672 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
13673 const Expr *Device) {
13674 llvm_unreachable("Not supported in SIMD-only mode");
13675}
13676
13677void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF,
13678 const OMPLoopDirective &D,
13679 ArrayRef<Expr *> NumIterations) {
13680 llvm_unreachable("Not supported in SIMD-only mode");
13681}
13682
13683void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
13684 const OMPDependClause *C) {
13685 llvm_unreachable("Not supported in SIMD-only mode");
13686}
13687
13688void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
13689 const OMPDoacrossClause *C) {
13690 llvm_unreachable("Not supported in SIMD-only mode");
13691}
13692
13693const VarDecl *
13694CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD,
13695 const VarDecl *NativeParam) const {
13696 llvm_unreachable("Not supported in SIMD-only mode");
13697}
13698
13699Address
13700CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF,
13701 const VarDecl *NativeParam,
13702 const VarDecl *TargetParam) const {
13703 llvm_unreachable("Not supported in SIMD-only mode");
13704}
13705