1//===--- CGStmtOpenMP.cpp - Emit LLVM Code from Statements ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This contains code to emit OpenMP nodes as LLVM code.
10//
11//===----------------------------------------------------------------------===//
12
13#include "CGCleanup.h"
14#include "CGDebugInfo.h"
15#include "CGOpenMPRuntime.h"
16#include "CodeGenFunction.h"
17#include "CodeGenModule.h"
18#include "CodeGenPGO.h"
19#include "TargetInfo.h"
20#include "clang/AST/ASTContext.h"
21#include "clang/AST/Attr.h"
22#include "clang/AST/DeclOpenMP.h"
23#include "clang/AST/OpenMPClause.h"
24#include "clang/AST/Stmt.h"
25#include "clang/AST/StmtOpenMP.h"
26#include "clang/AST/StmtVisitor.h"
27#include "clang/Basic/DiagnosticFrontend.h"
28#include "clang/Basic/OpenMPKinds.h"
29#include "clang/Basic/PrettyStackTrace.h"
30#include "clang/Basic/SourceManager.h"
31#include "llvm/ADT/SmallSet.h"
32#include "llvm/BinaryFormat/Dwarf.h"
33#include "llvm/Frontend/OpenMP/OMPConstants.h"
34#include "llvm/Frontend/OpenMP/OMPIRBuilder.h"
35#include "llvm/IR/Constants.h"
36#include "llvm/IR/DebugInfoMetadata.h"
37#include "llvm/IR/Instructions.h"
38#include "llvm/IR/IntrinsicInst.h"
39#include "llvm/IR/Metadata.h"
40#include "llvm/Support/AtomicOrdering.h"
41#include "llvm/Support/Debug.h"
42#include <optional>
43using namespace clang;
44using namespace CodeGen;
45using namespace llvm::omp;
46
47#define TTL_CODEGEN_TYPE "target-teams-loop-codegen"
48
49static const VarDecl *getBaseDecl(const Expr *Ref);
50static OpenMPDirectiveKind
51getEffectiveDirectiveKind(const OMPExecutableDirective &S);
52
53/// Whether a combined `distribute parallel for` may use the fused
54/// distr_static_chunk + static_chunkone schedule (enum 93): one
55/// for_static_init, no surrounding distribute_static_init.
56static bool canEmitGPUFusedDistSchedule(const CodeGenModule &CGM,
57 const OMPLoopDirective &S,
58 OpenMPDirectiveKind DKind) {
59 // 'teams loop' is always emitted as 'distribute', and 'target teams loop'
60 // only becomes 'distribute parallel for' if canBeParallelFor() holds.
61 // Without the inner worksharing loop, there is nothing that would schedule
62 // the iteration space if the outer distribute loop is omitted.
63 if (DKind == OMPD_teams_loop)
64 return false;
65 if (const auto *TTLD = dyn_cast<OMPTargetTeamsGenericLoopDirective>(Val: &S);
66 TTLD && !TTLD->canBeParallelFor())
67 return false;
68 // Reduction-only for now. Non-reduction cases might follow in the future, but
69 // need more analysis for maximum profit.
70 return CGM.getLangOpts().OpenMPIsTargetDevice && CGM.getTriple().isGPU() &&
71 isOpenMPLoopBoundSharingDirective(Kind: DKind) &&
72 S.hasClausesOfKind<OMPReductionClause>() &&
73 !S.getSingleClause<OMPDistScheduleClause>() &&
74 !S.getSingleClause<OMPScheduleClause>() &&
75 !S.getSingleClause<OMPOrderedClause>();
76}
77
78namespace {
79/// Lexical scope for OpenMP executable constructs, that handles correct codegen
80/// for captured expressions.
81class OMPLexicalScope : public CodeGenFunction::LexicalScope {
82 void emitPreInitStmt(CodeGenFunction &CGF, const OMPExecutableDirective &S) {
83 for (const auto *C : S.clauses()) {
84 if (const auto *CPI = OMPClauseWithPreInit::get(C)) {
85 if (const auto *PreInit =
86 cast_or_null<DeclStmt>(Val: CPI->getPreInitStmt())) {
87 for (const auto *I : PreInit->decls()) {
88 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
89 CGF.EmitVarDecl(D: cast<VarDecl>(Val: *I));
90 } else {
91 CodeGenFunction::AutoVarEmission Emission =
92 CGF.EmitAutoVarAlloca(var: cast<VarDecl>(Val: *I));
93 CGF.EmitAutoVarCleanups(emission: Emission);
94 }
95 }
96 }
97 }
98 }
99 }
100 CodeGenFunction::OMPPrivateScope InlinedShareds;
101
102 static bool isCapturedVar(CodeGenFunction &CGF, const VarDecl *VD) {
103 return CGF.LambdaCaptureFields.lookup(Val: VD) ||
104 (CGF.CapturedStmtInfo && CGF.CapturedStmtInfo->lookup(VD)) ||
105 (isa_and_nonnull<BlockDecl>(Val: CGF.CurCodeDecl) &&
106 cast<BlockDecl>(Val: CGF.CurCodeDecl)->capturesVariable(var: VD));
107 }
108
109public:
110 OMPLexicalScope(
111 CodeGenFunction &CGF, const OMPExecutableDirective &S,
112 const std::optional<OpenMPDirectiveKind> CapturedRegion = std::nullopt,
113 const bool EmitPreInitStmt = true)
114 : CodeGenFunction::LexicalScope(CGF, S.getSourceRange()),
115 InlinedShareds(CGF) {
116 if (EmitPreInitStmt)
117 emitPreInitStmt(CGF, S);
118 if (!CapturedRegion)
119 return;
120 assert(S.hasAssociatedStmt() &&
121 "Expected associated statement for inlined directive.");
122 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: *CapturedRegion);
123 for (const auto &C : CS->captures()) {
124 if (C.capturesVariable() || C.capturesVariableByCopy()) {
125 auto *VD = C.getCapturedVar();
126 assert(VD == VD->getCanonicalDecl() &&
127 "Canonical decl must be captured.");
128 DeclRefExpr DRE(
129 CGF.getContext(), const_cast<VarDecl *>(VD),
130 isCapturedVar(CGF, VD) || (CGF.CapturedStmtInfo &&
131 InlinedShareds.isGlobalVarCaptured(VD)),
132 VD->getType().getNonReferenceType(), VK_LValue, C.getLocation());
133 InlinedShareds.addPrivate(LocalVD: VD, Addr: CGF.EmitLValue(E: &DRE).getAddress());
134 }
135 }
136 (void)InlinedShareds.Privatize();
137 }
138};
139
140/// Lexical scope for OpenMP parallel construct, that handles correct codegen
141/// for captured expressions.
142class OMPParallelScope final : public OMPLexicalScope {
143 bool EmitPreInitStmt(const OMPExecutableDirective &S) {
144 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
145 return !(isOpenMPTargetExecutionDirective(DKind: EKind) ||
146 isOpenMPLoopBoundSharingDirective(Kind: EKind)) &&
147 isOpenMPParallelDirective(DKind: EKind);
148 }
149
150public:
151 OMPParallelScope(CodeGenFunction &CGF, const OMPExecutableDirective &S)
152 : OMPLexicalScope(CGF, S, /*CapturedRegion=*/std::nullopt,
153 EmitPreInitStmt(S)) {}
154};
155
156/// Lexical scope for OpenMP teams construct, that handles correct codegen
157/// for captured expressions.
158class OMPTeamsScope final : public OMPLexicalScope {
159 bool EmitPreInitStmt(const OMPExecutableDirective &S) {
160 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
161 return !isOpenMPTargetExecutionDirective(DKind: EKind) &&
162 isOpenMPTeamsDirective(DKind: EKind);
163 }
164
165public:
166 OMPTeamsScope(CodeGenFunction &CGF, const OMPExecutableDirective &S)
167 : OMPLexicalScope(CGF, S, /*CapturedRegion=*/std::nullopt,
168 EmitPreInitStmt(S)) {}
169};
170
171/// Private scope for OpenMP loop-based directives, that supports capturing
172/// of used expression from loop statement.
173class OMPLoopScope : public CodeGenFunction::RunCleanupsScope {
174 void emitPreInitStmt(CodeGenFunction &CGF, const OMPLoopBasedDirective &S) {
175 const Stmt *PreInits;
176 CodeGenFunction::OMPMapVars PreCondVars;
177 if (auto *LD = dyn_cast<OMPLoopDirective>(Val: &S)) {
178 // Emit init, __range, __begin and __end variables for C++ range loops.
179 (void)OMPLoopBasedDirective::doForAllLoops(
180 CurStmt: LD->getInnermostCapturedStmt()->getCapturedStmt(),
181 /*TryImperfectlyNestedLoops=*/true, NumLoops: LD->getLoopsNumber(),
182 Callback: [&CGF](unsigned Cnt, const Stmt *CurStmt) {
183 if (const auto *CXXFor = dyn_cast<CXXForRangeStmt>(Val: CurStmt)) {
184 if (const Stmt *Init = CXXFor->getInit())
185 CGF.EmitStmt(S: Init);
186 CGF.EmitStmt(S: CXXFor->getRangeStmt());
187 CGF.EmitStmt(S: CXXFor->getBeginStmt());
188 CGF.EmitStmt(S: CXXFor->getEndStmt());
189 }
190 return false;
191 });
192 llvm::DenseSet<const VarDecl *> EmittedAsPrivate;
193 for (const auto *E : LD->counters()) {
194 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl());
195 EmittedAsPrivate.insert(V: VD->getCanonicalDecl());
196 (void)PreCondVars.setVarAddr(
197 CGF, LocalVD: VD, TempAddr: CGF.CreateMemTemp(T: VD->getType().getNonReferenceType()));
198 }
199 // Mark private vars as undefs.
200 for (const auto *C : LD->getClausesOfKind<OMPPrivateClause>()) {
201 for (const Expr *IRef : C->varlist()) {
202 const auto *OrigDecl = cast<DeclRefExpr>(Val: IRef)->getDecl();
203 const auto *OrigVD = dyn_cast<VarDecl>(Val: OrigDecl);
204 if (!OrigVD)
205 continue;
206 if (EmittedAsPrivate.insert(V: OrigVD->getCanonicalDecl()).second) {
207 QualType OrigVDTy = OrigVD->getType().getNonReferenceType();
208 (void)PreCondVars.setVarAddr(
209 CGF, LocalVD: OrigVD,
210 TempAddr: Address(llvm::UndefValue::get(T: CGF.ConvertTypeForMem(
211 T: CGF.getContext().getPointerType(T: OrigVDTy))),
212 CGF.ConvertTypeForMem(T: OrigVDTy),
213 CGF.getContext().getDeclAlign(D: OrigVD)));
214 }
215 }
216 }
217 (void)PreCondVars.apply(CGF);
218 PreInits = LD->getPreInits();
219 } else if (const auto *Tile = dyn_cast<OMPTileDirective>(Val: &S)) {
220 PreInits = Tile->getPreInits();
221 } else if (const auto *Stripe = dyn_cast<OMPStripeDirective>(Val: &S)) {
222 PreInits = Stripe->getPreInits();
223 } else if (const auto *Unroll = dyn_cast<OMPUnrollDirective>(Val: &S)) {
224 PreInits = Unroll->getPreInits();
225 } else if (const auto *Reverse = dyn_cast<OMPReverseDirective>(Val: &S)) {
226 PreInits = Reverse->getPreInits();
227 } else if (const auto *Split = dyn_cast<OMPSplitDirective>(Val: &S)) {
228 PreInits = Split->getPreInits();
229 } else if (const auto *Interchange =
230 dyn_cast<OMPInterchangeDirective>(Val: &S)) {
231 PreInits = Interchange->getPreInits();
232 } else if (const auto *Flatten = dyn_cast<OMPFlattenDirective>(Val: &S)) {
233 PreInits = Flatten->getPreInits();
234 } else {
235 llvm_unreachable("Unknown loop-based directive kind.");
236 }
237 doEmitPreinits(PreInits);
238 PreCondVars.restore(CGF);
239 }
240
241 void
242 emitPreInitStmt(CodeGenFunction &CGF,
243 const OMPCanonicalLoopSequenceTransformationDirective &S) {
244 const Stmt *PreInits;
245 if (const auto *Fuse = dyn_cast<OMPFuseDirective>(Val: &S)) {
246 PreInits = Fuse->getPreInits();
247 } else {
248 llvm_unreachable(
249 "Unknown canonical loop sequence transform directive kind.");
250 }
251 doEmitPreinits(PreInits);
252 }
253
254 void doEmitPreinits(const Stmt *PreInits) {
255 if (PreInits) {
256 // CompoundStmts and DeclStmts are used as lists of PreInit statements and
257 // declarations. Since declarations must be visible in the the following
258 // that they initialize, unpack the CompoundStmt they are nested in.
259 SmallVector<const Stmt *> PreInitStmts;
260 if (auto *PreInitCompound = dyn_cast<CompoundStmt>(Val: PreInits))
261 llvm::append_range(C&: PreInitStmts, R: PreInitCompound->body());
262 else
263 PreInitStmts.push_back(Elt: PreInits);
264
265 for (const Stmt *S : PreInitStmts) {
266 // EmitStmt skips any OMPCapturedExprDecls, but needs to be emitted
267 // here.
268 if (auto *PreInitDecl = dyn_cast<DeclStmt>(Val: S)) {
269 for (Decl *I : PreInitDecl->decls())
270 CGF.EmitVarDecl(D: cast<VarDecl>(Val&: *I));
271 continue;
272 }
273 CGF.EmitStmt(S);
274 }
275 }
276 }
277
278public:
279 OMPLoopScope(CodeGenFunction &CGF, const OMPLoopBasedDirective &S)
280 : CodeGenFunction::RunCleanupsScope(CGF) {
281 emitPreInitStmt(CGF, S);
282 }
283 OMPLoopScope(CodeGenFunction &CGF,
284 const OMPCanonicalLoopSequenceTransformationDirective &S)
285 : CodeGenFunction::RunCleanupsScope(CGF) {
286 emitPreInitStmt(CGF, S);
287 }
288};
289
290class OMPSimdLexicalScope : public CodeGenFunction::LexicalScope {
291 CodeGenFunction::OMPPrivateScope InlinedShareds;
292
293 static bool isCapturedVar(CodeGenFunction &CGF, const VarDecl *VD) {
294 return CGF.LambdaCaptureFields.lookup(Val: VD) ||
295 (CGF.CapturedStmtInfo && CGF.CapturedStmtInfo->lookup(VD)) ||
296 (isa_and_nonnull<BlockDecl>(Val: CGF.CurCodeDecl) &&
297 cast<BlockDecl>(Val: CGF.CurCodeDecl)->capturesVariable(var: VD));
298 }
299
300public:
301 OMPSimdLexicalScope(CodeGenFunction &CGF, const OMPExecutableDirective &S)
302 : CodeGenFunction::LexicalScope(CGF, S.getSourceRange()),
303 InlinedShareds(CGF) {
304 for (const auto *C : S.clauses()) {
305 if (const auto *CPI = OMPClauseWithPreInit::get(C)) {
306 if (const auto *PreInit =
307 cast_or_null<DeclStmt>(Val: CPI->getPreInitStmt())) {
308 for (const auto *I : PreInit->decls()) {
309 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
310 CGF.EmitVarDecl(D: cast<VarDecl>(Val: *I));
311 } else {
312 CodeGenFunction::AutoVarEmission Emission =
313 CGF.EmitAutoVarAlloca(var: cast<VarDecl>(Val: *I));
314 CGF.EmitAutoVarCleanups(emission: Emission);
315 }
316 }
317 }
318 } else if (const auto *UDP = dyn_cast<OMPUseDevicePtrClause>(Val: C)) {
319 for (const Expr *E : UDP->varlist()) {
320 const Decl *D = cast<DeclRefExpr>(Val: E)->getDecl();
321 if (const auto *OED = dyn_cast<OMPCapturedExprDecl>(Val: D))
322 CGF.EmitVarDecl(D: *OED);
323 }
324 } else if (const auto *UDP = dyn_cast<OMPUseDeviceAddrClause>(Val: C)) {
325 for (const Expr *E : UDP->varlist()) {
326 const Decl *D = getBaseDecl(Ref: E);
327 if (const auto *OED = dyn_cast<OMPCapturedExprDecl>(Val: D))
328 CGF.EmitVarDecl(D: *OED);
329 }
330 }
331 }
332 if (!isOpenMPSimdDirective(DKind: getEffectiveDirectiveKind(S)))
333 CGF.EmitOMPPrivateClause(D: S, PrivateScope&: InlinedShareds);
334 if (const auto *TG = dyn_cast<OMPTaskgroupDirective>(Val: &S)) {
335 if (const Expr *E = TG->getReductionRef())
336 CGF.EmitVarDecl(D: *cast<VarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl()));
337 }
338 // Temp copy arrays for inscan reductions should not be emitted as they are
339 // not used in simd only mode.
340 llvm::DenseSet<CanonicalDeclPtr<const Decl>> CopyArrayTemps;
341 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
342 if (C->getModifier() != OMPC_REDUCTION_inscan)
343 continue;
344 for (const Expr *E : C->copy_array_temps())
345 CopyArrayTemps.insert(V: cast<DeclRefExpr>(Val: E)->getDecl());
346 }
347 const auto *CS = cast_or_null<CapturedStmt>(Val: S.getAssociatedStmt());
348 while (CS) {
349 for (auto &C : CS->captures()) {
350 if (C.capturesVariable() || C.capturesVariableByCopy()) {
351 auto *VD = C.getCapturedVar();
352 if (CopyArrayTemps.contains(V: VD))
353 continue;
354 assert(VD == VD->getCanonicalDecl() &&
355 "Canonical decl must be captured.");
356 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
357 isCapturedVar(CGF, VD) ||
358 (CGF.CapturedStmtInfo &&
359 InlinedShareds.isGlobalVarCaptured(VD)),
360 VD->getType().getNonReferenceType(), VK_LValue,
361 C.getLocation());
362 InlinedShareds.addPrivate(LocalVD: VD, Addr: CGF.EmitLValue(E: &DRE).getAddress());
363 }
364 }
365 CS = dyn_cast<CapturedStmt>(Val: CS->getCapturedStmt());
366 }
367 (void)InlinedShareds.Privatize();
368 }
369};
370
371} // namespace
372
373// The loop directive with a bind clause will be mapped to a different
374// directive with corresponding semantics.
375static OpenMPDirectiveKind
376getEffectiveDirectiveKind(const OMPExecutableDirective &S) {
377 OpenMPDirectiveKind Kind = S.getDirectiveKind();
378 if (Kind != OMPD_loop)
379 return Kind;
380
381 OpenMPBindClauseKind BindKind = OMPC_BIND_unknown;
382 if (const auto *C = S.getSingleClause<OMPBindClause>())
383 BindKind = C->getBindKind();
384
385 switch (BindKind) {
386 case OMPC_BIND_parallel:
387 return OMPD_for;
388 case OMPC_BIND_teams:
389 return OMPD_distribute;
390 case OMPC_BIND_thread:
391 return OMPD_simd;
392 default:
393 return OMPD_loop;
394 }
395}
396
397static void emitCommonOMPTargetDirective(CodeGenFunction &CGF,
398 const OMPExecutableDirective &S,
399 const RegionCodeGenTy &CodeGen);
400
401Address CodeGenFunction::EmitOMPBindingOriginalAddr(const BindingDecl *BD,
402 SourceLocation Loc) {
403 if (CapturedStmtInfo &&
404 CapturedStmtInfo->getKind() == CapturedRegionKind::CR_OpenMP) {
405 if (const auto *DD = dyn_cast<VarDecl>(Val: BD->getDecomposedDecl())) {
406 if (CapturedStmtInfo->lookup(VD: DD))
407 return EmitOMPCapturedBindingLValue(BD).getAddress();
408 }
409 }
410 DeclRefExpr DRE(getContext(), const_cast<BindingDecl *>(BD),
411 /*RefersToEnclosingVariableOrCapture=*/false, BD->getType(),
412 VK_LValue, Loc);
413 return EmitLValue(E: &DRE).getAddress();
414}
415
416LValue CodeGenFunction::EmitOMPSharedLValue(const Expr *E) {
417 if (const auto *OrigDRE = dyn_cast<DeclRefExpr>(Val: E)) {
418 if (const auto *OrigVD = dyn_cast<VarDecl>(Val: OrigDRE->getDecl())) {
419 OrigVD = OrigVD->getCanonicalDecl();
420 bool IsCaptured =
421 LambdaCaptureFields.lookup(Val: OrigVD) ||
422 (CapturedStmtInfo && CapturedStmtInfo->lookup(VD: OrigVD)) ||
423 (isa_and_nonnull<BlockDecl>(Val: CurCodeDecl));
424 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(OrigVD), IsCaptured,
425 OrigDRE->getType(), VK_LValue, OrigDRE->getExprLoc());
426 return EmitLValue(E: &DRE);
427 }
428 if (const auto *OrigBD = dyn_cast<BindingDecl>(Val: OrigDRE->getDecl())) {
429 OrigBD = cast<BindingDecl>(Val: OrigBD->getCanonicalDecl());
430 const auto *DD = cast<VarDecl>(Val: OrigBD->getDecomposedDecl());
431 bool IsCaptured = LambdaCaptureFields.lookup(Val: OrigBD) ||
432 (CapturedStmtInfo && CapturedStmtInfo->lookup(VD: DD)) ||
433 isa_and_nonnull<BlockDecl>(Val: CurCodeDecl);
434 DeclRefExpr DRE(getContext(), const_cast<BindingDecl *>(OrigBD),
435 IsCaptured, OrigDRE->getType(), VK_LValue,
436 OrigDRE->getExprLoc());
437 return EmitLValue(E: &DRE);
438 }
439 }
440 return EmitLValue(E);
441}
442
443llvm::Value *CodeGenFunction::getTypeSize(QualType Ty) {
444 ASTContext &C = getContext();
445 llvm::Value *Size = nullptr;
446 auto SizeInChars = C.getTypeSizeInChars(T: Ty);
447 if (SizeInChars.isZero()) {
448 // getTypeSizeInChars() returns 0 for a VLA.
449 while (const VariableArrayType *VAT = C.getAsVariableArrayType(T: Ty)) {
450 VlaSizePair VlaSize = getVLASize(vla: VAT);
451 Ty = VlaSize.Type;
452 Size =
453 Size ? Builder.CreateNUWMul(LHS: Size, RHS: VlaSize.NumElts) : VlaSize.NumElts;
454 }
455 SizeInChars = C.getTypeSizeInChars(T: Ty);
456 if (SizeInChars.isZero())
457 return llvm::ConstantInt::get(Ty: SizeTy, /*V=*/0);
458 return Builder.CreateNUWMul(LHS: Size, RHS: CGM.getSize(numChars: SizeInChars));
459 }
460 return CGM.getSize(numChars: SizeInChars);
461}
462
463void CodeGenFunction::GenerateOpenMPCapturedVars(
464 const CapturedStmt &S, SmallVectorImpl<llvm::Value *> &CapturedVars) {
465 const RecordDecl *RD = S.getCapturedRecordDecl();
466 auto CurField = RD->field_begin();
467 auto CurCap = S.captures().begin();
468 for (CapturedStmt::const_capture_init_iterator I = S.capture_init_begin(),
469 E = S.capture_init_end();
470 I != E; ++I, ++CurField, ++CurCap) {
471 if (CurField->hasCapturedVLAType()) {
472 const VariableArrayType *VAT = CurField->getCapturedVLAType();
473 llvm::Value *Val = VLASizeMap[VAT->getSizeExpr()];
474 CapturedVars.push_back(Elt: Val);
475 } else if (CurCap->capturesThis()) {
476 CapturedVars.push_back(Elt: CXXThisValue);
477 } else if (CurCap->capturesVariableByCopy()) {
478 llvm::Value *CV = EmitLoadOfScalar(lvalue: EmitLValue(E: *I), Loc: CurCap->getLocation());
479
480 // If the field is not a pointer, we need to save the actual value
481 // and load it as a void pointer.
482 if (!CurField->getType()->isAnyPointerType()) {
483 ASTContext &Ctx = getContext();
484 Address DstAddr = CreateMemTempWithoutCast(
485 T: Ctx.getUIntPtrType(),
486 Name: Twine(CurCap->getCapturedVar()->getName(), ".casted"));
487 LValue DstLV = MakeAddrLValue(Addr: DstAddr, T: Ctx.getUIntPtrType());
488
489 llvm::Value *SrcAddrVal = EmitScalarConversion(
490 Src: DstAddr.emitRawPointer(CGF&: *this),
491 SrcTy: Ctx.getPointerType(T: Ctx.getUIntPtrType()),
492 DstTy: Ctx.getPointerType(T: CurField->getType()), Loc: CurCap->getLocation());
493 LValue SrcLV =
494 MakeNaturalAlignAddrLValue(V: SrcAddrVal, T: CurField->getType());
495
496 // Store the value using the source type pointer.
497 EmitStoreThroughLValue(Src: RValue::get(V: CV), Dst: SrcLV);
498
499 // Load the value using the destination type pointer.
500 CV = EmitLoadOfScalar(lvalue: DstLV, Loc: CurCap->getLocation());
501 }
502 CapturedVars.push_back(Elt: CV);
503 } else {
504 assert(CurCap->capturesVariable() && "Expected capture by reference.");
505 llvm::Value *Addr = EmitLValue(E: *I).getAddress().emitRawPointer(CGF&: *this);
506 // Sema strips the address space from the type of the captured field.
507 llvm::Type *ArgTy = ConvertType(T: CurField->getType());
508 if (Addr->getType() != ArgTy)
509 Addr = performAddrSpaceCast(Src: Addr, DestTy: ArgTy);
510 CapturedVars.push_back(Elt: Addr);
511 }
512 }
513}
514
515static Address castValueFromUintptr(CodeGenFunction &CGF, SourceLocation Loc,
516 QualType DstType, StringRef Name,
517 LValue AddrLV) {
518 ASTContext &Ctx = CGF.getContext();
519
520 llvm::Value *CastedPtr = CGF.EmitScalarConversion(
521 Src: AddrLV.getAddress().emitRawPointer(CGF), SrcTy: Ctx.getUIntPtrType(),
522 DstTy: Ctx.getPointerType(T: DstType), Loc);
523 // FIXME: should the pointee type (DstType) be passed?
524 Address TmpAddr =
525 CGF.MakeNaturalAlignAddrLValue(V: CastedPtr, T: DstType).getAddress();
526 return TmpAddr;
527}
528
529static QualType getCanonicalParamType(ASTContext &C, QualType T) {
530 if (T->isLValueReferenceType())
531 return C.getLValueReferenceType(
532 T: getCanonicalParamType(C, T: T.getNonReferenceType()),
533 /*SpelledAsLValue=*/false);
534 if (T->isPointerType())
535 return C.getPointerType(T: getCanonicalParamType(C, T: T->getPointeeType()));
536 if (const ArrayType *A = T->getAsArrayTypeUnsafe()) {
537 if (const auto *VLA = dyn_cast<VariableArrayType>(Val: A))
538 return getCanonicalParamType(C, T: VLA->getElementType());
539 if (!A->isVariablyModifiedType())
540 return C.getCanonicalType(T);
541 }
542 return C.getCanonicalParamType(T);
543}
544
545namespace {
546/// Contains required data for proper outlined function codegen.
547struct FunctionOptions {
548 /// Captured statement for which the function is generated.
549 const CapturedStmt *S = nullptr;
550 /// true if cast to/from UIntPtr is required for variables captured by
551 /// value.
552 const bool UIntPtrCastRequired = true;
553 /// true if only casted arguments must be registered as local args or VLA
554 /// sizes.
555 const bool RegisterCastedArgsOnly = false;
556 /// Name of the generated function.
557 const StringRef FunctionName;
558 /// Location of the non-debug version of the outlined function.
559 SourceLocation Loc;
560 const bool IsDeviceKernel = false;
561 explicit FunctionOptions(const CapturedStmt *S, bool UIntPtrCastRequired,
562 bool RegisterCastedArgsOnly, StringRef FunctionName,
563 SourceLocation Loc, bool IsDeviceKernel)
564 : S(S), UIntPtrCastRequired(UIntPtrCastRequired),
565 RegisterCastedArgsOnly(UIntPtrCastRequired && RegisterCastedArgsOnly),
566 FunctionName(FunctionName), Loc(Loc), IsDeviceKernel(IsDeviceKernel) {}
567};
568} // namespace
569
570static llvm::Function *emitOutlinedFunctionPrologue(
571 CodeGenFunction &CGF, FunctionArgList &Args,
572 llvm::MapVector<const Decl *, std::pair<const VarDecl *, Address>>
573 &LocalAddrs,
574 llvm::DenseMap<const Decl *, std::pair<const Expr *, llvm::Value *>>
575 &VLASizes,
576 llvm::Value *&CXXThisValue, const FunctionOptions &FO) {
577 const CapturedDecl *CD = FO.S->getCapturedDecl();
578 const RecordDecl *RD = FO.S->getCapturedRecordDecl();
579 assert(CD->hasBody() && "missing CapturedDecl body");
580
581 CXXThisValue = nullptr;
582 // Build the argument list.
583 CodeGenModule &CGM = CGF.CGM;
584 ASTContext &Ctx = CGM.getContext();
585 FunctionArgList TargetArgs;
586 Args.append(in_start: CD->param_begin(),
587 in_end: std::next(x: CD->param_begin(), n: CD->getContextParamPosition()));
588 TargetArgs.append(
589 in_start: CD->param_begin(),
590 in_end: std::next(x: CD->param_begin(), n: CD->getContextParamPosition()));
591 auto I = FO.S->captures().begin();
592 FunctionDecl *DebugFunctionDecl = nullptr;
593 if (!FO.UIntPtrCastRequired) {
594 FunctionProtoType::ExtProtoInfo EPI;
595 QualType FunctionTy = Ctx.getFunctionType(ResultTy: Ctx.VoidTy, Args: {}, EPI);
596 DebugFunctionDecl = FunctionDecl::Create(
597 C&: Ctx, DC: Ctx.getTranslationUnitDecl(), StartLoc: FO.S->getBeginLoc(),
598 NLoc: SourceLocation(), N: DeclarationName(), T: FunctionTy,
599 TInfo: Ctx.getTrivialTypeSourceInfo(T: FunctionTy), SC: SC_Static,
600 /*UsesFPIntrin=*/false, /*isInlineSpecified=*/false,
601 /*hasWrittenPrototype=*/false);
602 }
603 for (const FieldDecl *FD : RD->fields()) {
604 QualType ArgType = FD->getType();
605 IdentifierInfo *II = nullptr;
606 VarDecl *CapVar = nullptr;
607
608 // If this is a capture by copy and the type is not a pointer, the outlined
609 // function argument type should be uintptr and the value properly casted to
610 // uintptr. This is necessary given that the runtime library is only able to
611 // deal with pointers. We can pass in the same way the VLA type sizes to the
612 // outlined function.
613 if (FO.UIntPtrCastRequired &&
614 ((I->capturesVariableByCopy() && !ArgType->isAnyPointerType()) ||
615 I->capturesVariableArrayType()))
616 ArgType = Ctx.getUIntPtrType();
617
618 if (I->capturesVariable() || I->capturesVariableByCopy()) {
619 CapVar = I->getCapturedVar();
620 II = CapVar->getIdentifier();
621 } else if (I->capturesThis()) {
622 II = &Ctx.Idents.get(Name: "this");
623 } else {
624 assert(I->capturesVariableArrayType());
625 II = &Ctx.Idents.get(Name: "vla");
626 }
627 if (ArgType->isVariablyModifiedType())
628 ArgType = getCanonicalParamType(C&: Ctx, T: ArgType);
629 VarDecl *Arg;
630 if (CapVar && (CapVar->getTLSKind() != clang::VarDecl::TLS_None)) {
631 Arg = ImplicitParamDecl::Create(C&: Ctx, /*DC=*/nullptr, IdLoc: FD->getLocation(),
632 Id: II, T: ArgType,
633 ParamKind: ImplicitParamKind::ThreadPrivateVar);
634 } else if (DebugFunctionDecl && (CapVar || I->capturesThis())) {
635 Arg = ParmVarDecl::Create(
636 C&: Ctx, DC: DebugFunctionDecl,
637 StartLoc: CapVar ? CapVar->getBeginLoc() : FD->getBeginLoc(),
638 IdLoc: CapVar ? CapVar->getLocation() : FD->getLocation(), Id: II, T: ArgType,
639 /*TInfo=*/nullptr, S: SC_None, /*DefArg=*/nullptr);
640 } else {
641 Arg = ImplicitParamDecl::Create(C&: Ctx, /*DC=*/nullptr, IdLoc: FD->getLocation(),
642 Id: II, T: ArgType, ParamKind: ImplicitParamKind::Other);
643 }
644 Args.emplace_back(Args&: Arg);
645 // Do not cast arguments if we emit function with non-original types.
646 TargetArgs.emplace_back(
647 Args: FO.UIntPtrCastRequired
648 ? Arg
649 : CGM.getOpenMPRuntime().translateParameter(FD, NativeParam: Arg));
650 ++I;
651 }
652 Args.append(in_start: std::next(x: CD->param_begin(), n: CD->getContextParamPosition() + 1),
653 in_end: CD->param_end());
654 TargetArgs.append(
655 in_start: std::next(x: CD->param_begin(), n: CD->getContextParamPosition() + 1),
656 in_end: CD->param_end());
657
658 // Create the function declaration.
659 const CGFunctionInfo &FuncInfo =
660 FO.IsDeviceKernel
661 ? CGM.getTypes().arrangeDeviceKernelCallerDeclaration(resultType: Ctx.VoidTy,
662 args: TargetArgs)
663 : CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: Ctx.VoidTy,
664 args: TargetArgs);
665 llvm::FunctionType *FuncLLVMTy = CGM.getTypes().GetFunctionType(Info: FuncInfo);
666
667 auto *F =
668 llvm::Function::Create(Ty: FuncLLVMTy, Linkage: llvm::GlobalValue::InternalLinkage,
669 N: FO.FunctionName, M: &CGM.getModule());
670 CGM.SetInternalFunctionAttributes(GD: CD, F, FI: FuncInfo);
671
672 // Adjust the calling convention for SPIR-V targets to avoid mismatches
673 // between callee and caller.
674 if (CGM.getTriple().isSPIRV() && !FO.IsDeviceKernel)
675 F->setCallingConv(llvm::CallingConv::SPIR_FUNC);
676
677 if (CD->isNothrow())
678 F->setDoesNotThrow();
679
680 // Always inline the outlined function if optimizations are enabled.
681 if (CGM.getCodeGenOpts().OptimizationLevel != 0) {
682 F->removeFnAttr(Kind: llvm::Attribute::NoInline);
683 F->addFnAttr(Kind: llvm::Attribute::AlwaysInline);
684 }
685 if (!CGM.getCodeGenOpts().SampleProfileFile.empty())
686 F->addFnAttr(Kind: "sample-profile-suffix-elision-policy", Val: "selected");
687
688 // Generate the function.
689 CGF.StartFunction(GD: CD, RetTy: Ctx.VoidTy, Fn: F, FnInfo: FuncInfo, Args: TargetArgs,
690 Loc: FO.UIntPtrCastRequired ? FO.Loc : FO.S->getBeginLoc(),
691 StartLoc: FO.UIntPtrCastRequired ? FO.Loc
692 : CD->getBody()->getBeginLoc());
693 unsigned Cnt = CD->getContextParamPosition();
694 I = FO.S->captures().begin();
695 for (const FieldDecl *FD : RD->fields()) {
696 // Do not map arguments if we emit function with non-original types.
697 Address LocalAddr(Address::invalid());
698 if (!FO.UIntPtrCastRequired && Args[Cnt] != TargetArgs[Cnt]) {
699 LocalAddr = CGM.getOpenMPRuntime().getParameterAddress(CGF, NativeParam: Args[Cnt],
700 TargetParam: TargetArgs[Cnt]);
701 } else {
702 LocalAddr = CGF.GetAddrOfLocalVar(VD: Args[Cnt]);
703 }
704 // If we are capturing a pointer by copy we don't need to do anything, just
705 // use the value that we get from the arguments.
706 if (I->capturesVariableByCopy() && FD->getType()->isAnyPointerType()) {
707 const VarDecl *CurVD = I->getCapturedVar();
708 if (!FO.RegisterCastedArgsOnly)
709 LocalAddrs.insert(KV: {Args[Cnt], {CurVD, LocalAddr}});
710 ++Cnt;
711 ++I;
712 continue;
713 }
714
715 LValue ArgLVal = CGF.MakeAddrLValue(Addr: LocalAddr, T: Args[Cnt]->getType(),
716 Source: AlignmentSource::Decl);
717 if (FD->hasCapturedVLAType()) {
718 if (FO.UIntPtrCastRequired) {
719 ArgLVal = CGF.MakeAddrLValue(
720 Addr: castValueFromUintptr(CGF, Loc: I->getLocation(), DstType: FD->getType(),
721 Name: Args[Cnt]->getName(), AddrLV: ArgLVal),
722 T: FD->getType(), Source: AlignmentSource::Decl);
723 }
724 llvm::Value *ExprArg = CGF.EmitLoadOfScalar(lvalue: ArgLVal, Loc: I->getLocation());
725 const VariableArrayType *VAT = FD->getCapturedVLAType();
726 VLASizes.try_emplace(Key: Args[Cnt], Args: VAT->getSizeExpr(), Args&: ExprArg);
727 } else if (I->capturesVariable()) {
728 const VarDecl *Var = I->getCapturedVar();
729 QualType VarTy = Var->getType();
730 Address ArgAddr = ArgLVal.getAddress();
731 if (ArgLVal.getType()->isLValueReferenceType()) {
732 ArgAddr = CGF.EmitLoadOfReference(RefLVal: ArgLVal);
733 } else if (!VarTy->isVariablyModifiedType() || !VarTy->isPointerType()) {
734 assert(ArgLVal.getType()->isPointerType());
735 ArgAddr = CGF.EmitLoadOfPointer(
736 Ptr: ArgAddr, PtrTy: ArgLVal.getType()->castAs<PointerType>());
737 }
738 if (!FO.RegisterCastedArgsOnly) {
739 LocalAddrs.insert(
740 KV: {Args[Cnt], {Var, ArgAddr.withAlignment(NewAlignment: Ctx.getDeclAlign(D: Var))}});
741 }
742 } else if (I->capturesVariableByCopy()) {
743 assert(!FD->getType()->isAnyPointerType() &&
744 "Not expecting a captured pointer.");
745 const VarDecl *Var = I->getCapturedVar();
746 LocalAddrs.insert(KV: {Args[Cnt],
747 {Var, FO.UIntPtrCastRequired
748 ? castValueFromUintptr(
749 CGF, Loc: I->getLocation(), DstType: FD->getType(),
750 Name: Args[Cnt]->getName(), AddrLV: ArgLVal)
751 : ArgLVal.getAddress()}});
752 } else {
753 // If 'this' is captured, load it into CXXThisValue.
754 assert(I->capturesThis());
755 CXXThisValue = CGF.EmitLoadOfScalar(lvalue: ArgLVal, Loc: I->getLocation());
756 LocalAddrs.insert(KV: {Args[Cnt], {nullptr, ArgLVal.getAddress()}});
757 }
758 ++Cnt;
759 ++I;
760 }
761
762 return F;
763}
764
765static llvm::Function *emitOutlinedFunctionPrologueAggregate(
766 CodeGenFunction &CGF, FunctionArgList &Args,
767 llvm::MapVector<const Decl *, std::pair<const VarDecl *, Address>>
768 &LocalAddrs,
769 llvm::DenseMap<const Decl *, std::pair<const Expr *, llvm::Value *>>
770 &VLASizes,
771 llvm::Value *&CXXThisValue, llvm::Value *&ContextV, const CapturedStmt &CS,
772 SourceLocation Loc, StringRef FunctionName) {
773 const CapturedDecl *CD = CS.getCapturedDecl();
774 const RecordDecl *RD = CS.getCapturedRecordDecl();
775
776 CXXThisValue = nullptr;
777 CodeGenModule &CGM = CGF.CGM;
778 ASTContext &Ctx = CGM.getContext();
779 Args.push_back(Elt: CD->getContextParam());
780
781 const CGFunctionInfo &FuncInfo =
782 CGM.getTypes().arrangeBuiltinFunctionDeclaration(resultType: Ctx.VoidTy, args: Args);
783 llvm::FunctionType *FuncLLVMTy = CGM.getTypes().GetFunctionType(Info: FuncInfo);
784
785 auto *F =
786 llvm::Function::Create(Ty: FuncLLVMTy, Linkage: llvm::GlobalValue::InternalLinkage,
787 N: FunctionName, M: &CGM.getModule());
788 CGM.SetInternalFunctionAttributes(GD: CD, F, FI: FuncInfo);
789 if (CD->isNothrow())
790 F->setDoesNotThrow();
791
792 CGF.StartFunction(GD: CD, RetTy: Ctx.VoidTy, Fn: F, FnInfo: FuncInfo, Args, Loc, StartLoc: Loc);
793 Address ContextAddr = CGF.GetAddrOfLocalVar(VD: CD->getContextParam());
794 ContextV = CGF.Builder.CreateLoad(Addr: ContextAddr);
795
796 // The runtime passes arguments as an array of pointers.
797 llvm::Type *PtrTy = CGF.Builder.getPtrTy();
798 llvm::Align PtrAlign = CGM.getDataLayout().getPointerABIAlignment(AS: 0);
799 CharUnits SlotAlign = CharUnits::fromQuantity(Quantity: PtrAlign.value());
800
801 for (auto [FD, C, FieldIdx] :
802 llvm::zip(t: RD->fields(), u: CS.captures(),
803 args: llvm::seq<unsigned>(Size: RD->getNumFields()))) {
804 llvm::Value *SlotPtr =
805 CGF.Builder.CreateConstInBoundsGEP1_32(Ty: PtrTy, Ptr: ContextV, Idx0: FieldIdx);
806 llvm::Value *Slot = CGF.Builder.CreateAlignedLoad(Ty: PtrTy, Ptr: SlotPtr, Align: PtrAlign);
807
808 // Generate the appropriate load from the per-argument storage. This
809 // includes all of the user arguments as well as the implicit kernel
810 // argument pointer.
811 if (C.capturesVariableByCopy() && FD->getType()->isAnyPointerType()) {
812 const VarDecl *CurVD = C.getCapturedVar();
813 Slot->setName(CurVD->getName());
814 Address SlotAddr(Slot, PtrTy, SlotAlign);
815 LocalAddrs.insert(KV: {FD, {CurVD, SlotAddr}});
816 } else if (FD->hasCapturedVLAType()) {
817 // VLA size is stored as intptr_t directly in the slot.
818 Address SlotAddr(Slot, CGF.ConvertTypeForMem(T: FD->getType()), SlotAlign);
819 LValue ArgLVal =
820 CGF.MakeAddrLValue(Addr: SlotAddr, T: FD->getType(), Source: AlignmentSource::Decl);
821 llvm::Value *ExprArg = CGF.EmitLoadOfScalar(lvalue: ArgLVal, Loc: C.getLocation());
822 const VariableArrayType *VAT = FD->getCapturedVLAType();
823 VLASizes.try_emplace(Key: FD, Args: VAT->getSizeExpr(), Args&: ExprArg);
824 } else if (C.capturesVariable()) {
825 const VarDecl *Var = C.getCapturedVar();
826 QualType VarTy = Var->getType();
827
828 if (VarTy->isVariablyModifiedType() && VarTy->isPointerType()) {
829 Slot->setName(Var->getName() + ".addr");
830 Address SlotAddr(Slot, PtrTy, SlotAlign);
831 LocalAddrs.insert(KV: {FD, {Var, SlotAddr}});
832 } else {
833 llvm::Value *VarAddr = CGF.Builder.CreateAlignedLoad(
834 Ty: PtrTy, Ptr: Slot, Align: PtrAlign, Name: Var->getName());
835 LocalAddrs.insert(KV: {FD,
836 {Var, Address(VarAddr, CGF.ConvertTypeForMem(T: VarTy),
837 Ctx.getDeclAlign(D: Var))}});
838 }
839 } else if (C.capturesVariableByCopy()) {
840 assert(!FD->getType()->isAnyPointerType() &&
841 "Not expecting a captured pointer.");
842 const VarDecl *Var = C.getCapturedVar();
843 QualType FieldTy = FD->getType();
844
845 // Scalar values are promoted and stored directly in the slot.
846 Address SlotAddr(Slot, CGF.ConvertTypeForMem(T: FieldTy), SlotAlign);
847 Address CopyAddr =
848 CGF.CreateMemTemp(T: FieldTy, Align: Ctx.getDeclAlign(D: FD), Name: Var->getName());
849 LValue SrcLVal =
850 CGF.MakeAddrLValue(Addr: SlotAddr, T: FieldTy, Source: AlignmentSource::Decl);
851 LValue CopyLVal =
852 CGF.MakeAddrLValue(Addr: CopyAddr, T: FieldTy, Source: AlignmentSource::Decl);
853
854 RValue ArgRVal = CGF.EmitLoadOfLValue(V: SrcLVal, Loc: C.getLocation());
855 CGF.EmitStoreThroughLValue(Src: ArgRVal, Dst: CopyLVal);
856
857 LocalAddrs.insert(KV: {FD, {Var, CopyAddr}});
858 } else {
859 assert(C.capturesThis() && "Default case expected to be CXX 'this'");
860 CXXThisValue =
861 CGF.Builder.CreateAlignedLoad(Ty: PtrTy, Ptr: Slot, Align: PtrAlign, Name: "this");
862 Address SlotAddr(Slot, PtrTy, SlotAlign);
863 LocalAddrs.insert(KV: {FD, {nullptr, SlotAddr}});
864 }
865 }
866
867 return F;
868}
869
870llvm::Function *CodeGenFunction::GenerateOpenMPCapturedStmtFunction(
871 const CapturedStmt &S, const OMPExecutableDirective &D) {
872 SourceLocation Loc = D.getBeginLoc();
873 assert(
874 CapturedStmtInfo &&
875 "CapturedStmtInfo should be set when generating the captured function");
876 const CapturedDecl *CD = S.getCapturedDecl();
877 // Build the argument list.
878 bool NeedWrapperFunction =
879 getDebugInfo() && CGM.getCodeGenOpts().hasReducedDebugInfo();
880 FunctionArgList Args, WrapperArgs;
881 llvm::MapVector<const Decl *, std::pair<const VarDecl *, Address>> LocalAddrs,
882 WrapperLocalAddrs;
883 llvm::DenseMap<const Decl *, std::pair<const Expr *, llvm::Value *>> VLASizes,
884 WrapperVLASizes;
885 SmallString<256> Buffer;
886 llvm::raw_svector_ostream Out(Buffer);
887 Out << CapturedStmtInfo->getHelperName();
888 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
889 bool IsDeviceKernel = CGM.getOpenMPRuntime().isGPU() &&
890 isOpenMPTargetExecutionDirective(DKind: EKind) &&
891 D.getCapturedStmt(RegionKind: OMPD_target) == &S;
892 CodeGenFunction WrapperCGF(CGM, /*suppressNewContext=*/true);
893 llvm::Function *WrapperF = nullptr;
894 if (NeedWrapperFunction) {
895 // Emit the final kernel early to allow attributes to be added by the
896 // OpenMPI-IR-Builder.
897 FunctionOptions WrapperFO(&S, /*UIntPtrCastRequired=*/true,
898 /*RegisterCastedArgsOnly=*/true,
899 CapturedStmtInfo->getHelperName(), Loc,
900 IsDeviceKernel);
901 WrapperCGF.CapturedStmtInfo = CapturedStmtInfo;
902 WrapperF =
903 emitOutlinedFunctionPrologue(CGF&: WrapperCGF, Args, LocalAddrs, VLASizes,
904 CXXThisValue&: WrapperCGF.CXXThisValue, FO: WrapperFO);
905 Out << "_debug__";
906 }
907 FunctionOptions FO(&S, !NeedWrapperFunction, /*RegisterCastedArgsOnly=*/false,
908 Out.str(), Loc, !NeedWrapperFunction && IsDeviceKernel);
909 llvm::Function *F = emitOutlinedFunctionPrologue(
910 CGF&: *this, Args&: WrapperArgs, LocalAddrs&: WrapperLocalAddrs, VLASizes&: WrapperVLASizes, CXXThisValue, FO);
911 CodeGenFunction::OMPPrivateScope LocalScope(*this);
912 for (const auto &LocalAddrPair : WrapperLocalAddrs) {
913 if (LocalAddrPair.second.first) {
914 LocalScope.addPrivate(LocalVD: LocalAddrPair.second.first,
915 Addr: LocalAddrPair.second.second);
916 }
917 }
918 (void)LocalScope.Privatize();
919 for (const auto &VLASizePair : WrapperVLASizes)
920 VLASizeMap[VLASizePair.second.first] = VLASizePair.second.second;
921 PGO->assignRegionCounters(GD: GlobalDecl(CD), Fn: F);
922 CapturedStmtInfo->EmitBody(CGF&: *this, S: CD->getBody());
923 LocalScope.ForceCleanup();
924 FinishFunction(EndLoc: CD->getBodyRBrace());
925 if (!NeedWrapperFunction)
926 return F;
927
928 // Reverse the order.
929 WrapperF->removeFromParent();
930 F->getParent()->getFunctionList().insertAfter(where: F->getIterator(), New: WrapperF);
931
932 llvm::SmallVector<llvm::Value *, 4> CallArgs;
933 auto *PI = F->arg_begin();
934 for (const auto *Arg : Args) {
935 llvm::Value *CallArg;
936 auto I = LocalAddrs.find(Key: Arg);
937 if (I != LocalAddrs.end()) {
938 LValue LV = WrapperCGF.MakeAddrLValue(
939 Addr: I->second.second,
940 T: I->second.first ? I->second.first->getType() : Arg->getType(),
941 Source: AlignmentSource::Decl);
942 if (LV.getType()->isAnyComplexType())
943 LV.setAddress(LV.getAddress().withElementType(ElemTy: PI->getType()));
944 CallArg = WrapperCGF.EmitLoadOfScalar(lvalue: LV, Loc: S.getBeginLoc());
945 } else {
946 auto EI = VLASizes.find(Val: Arg);
947 if (EI != VLASizes.end()) {
948 CallArg = EI->second.second;
949 } else {
950 LValue LV =
951 WrapperCGF.MakeAddrLValue(Addr: WrapperCGF.GetAddrOfLocalVar(VD: Arg),
952 T: Arg->getType(), Source: AlignmentSource::Decl);
953 CallArg = WrapperCGF.EmitLoadOfScalar(lvalue: LV, Loc: S.getBeginLoc());
954 }
955 }
956 CallArgs.emplace_back(Args: WrapperCGF.EmitFromMemory(Value: CallArg, Ty: Arg->getType()));
957 ++PI;
958 }
959 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF&: WrapperCGF, Loc, OutlinedFn: F, Args: CallArgs);
960 WrapperCGF.FinishFunction();
961 return WrapperF;
962}
963
964llvm::Function *CodeGenFunction::GenerateOpenMPCapturedStmtFunctionAggregate(
965 const CapturedStmt &S, const OMPExecutableDirective &D) {
966 SourceLocation Loc = D.getBeginLoc();
967 assert(
968 CapturedStmtInfo &&
969 "CapturedStmtInfo should be set when generating the captured function");
970 const CapturedDecl *CD = S.getCapturedDecl();
971 const RecordDecl *RD = S.getCapturedRecordDecl();
972 StringRef FunctionName = CapturedStmtInfo->getHelperName();
973 bool NeedWrapperFunction =
974 getDebugInfo() && CGM.getCodeGenOpts().hasReducedDebugInfo();
975
976 CodeGenFunction WrapperCGF(CGM, /*suppressNewContext=*/true);
977 llvm::Function *WrapperF = nullptr;
978 llvm::Value *WrapperContextV = nullptr;
979 if (NeedWrapperFunction) {
980 WrapperCGF.CapturedStmtInfo = CapturedStmtInfo;
981 FunctionArgList WrapperArgs;
982 llvm::MapVector<const Decl *, std::pair<const VarDecl *, Address>>
983 WrapperLocalAddrs;
984 llvm::DenseMap<const Decl *, std::pair<const Expr *, llvm::Value *>>
985 WrapperVLASizes;
986 WrapperF = emitOutlinedFunctionPrologueAggregate(
987 CGF&: WrapperCGF, Args&: WrapperArgs, LocalAddrs&: WrapperLocalAddrs, VLASizes&: WrapperVLASizes,
988 CXXThisValue&: WrapperCGF.CXXThisValue, ContextV&: WrapperContextV, CS: S, Loc, FunctionName);
989 }
990
991 FunctionArgList Args;
992 llvm::MapVector<const Decl *, std::pair<const VarDecl *, Address>> LocalAddrs;
993 llvm::DenseMap<const Decl *, std::pair<const Expr *, llvm::Value *>> VLASizes;
994 llvm::Function *F;
995
996 if (NeedWrapperFunction) {
997 SmallString<256> Buffer;
998 llvm::raw_svector_ostream Out(Buffer);
999 Out << FunctionName << "_debug__";
1000
1001 FunctionOptions FO(&S, /*UIntPtrCastRequired=*/false,
1002 /*RegisterCastedArgsOnly=*/false, Out.str(), Loc,
1003 /*IsDeviceKernel=*/false);
1004 F = emitOutlinedFunctionPrologue(CGF&: *this, Args, LocalAddrs, VLASizes,
1005 CXXThisValue, FO);
1006 } else {
1007 llvm::Value *ContextV = nullptr;
1008 F = emitOutlinedFunctionPrologueAggregate(CGF&: *this, Args, LocalAddrs, VLASizes,
1009 CXXThisValue, ContextV, CS: S, Loc,
1010 FunctionName);
1011
1012 const RecordDecl *RD = S.getCapturedRecordDecl();
1013 unsigned FieldIdx = RD->getNumFields();
1014 for (unsigned I = 0; I < CD->getNumParams(); ++I) {
1015 const ImplicitParamDecl *Param = CD->getParam(i: I);
1016 if (Param == CD->getContextParam())
1017 continue;
1018 llvm::Align PtrAlign = CGM.getDataLayout().getPointerABIAlignment(AS: 0);
1019 llvm::Value *SlotPtr = Builder.CreateConstInBoundsGEP1_32(
1020 Ty: Builder.getPtrTy(), Ptr: ContextV, Idx0: FieldIdx,
1021 Name: Twine(Param->getName()) + ".addr");
1022 llvm::Value *ParamAddr =
1023 Builder.CreateAlignedLoad(Ty: Builder.getPtrTy(), Ptr: SlotPtr, Align: PtrAlign);
1024 llvm::Value *ParamVal = Builder.CreateAlignedLoad(
1025 Ty: Builder.getPtrTy(), Ptr: ParamAddr, Align: PtrAlign, Name: Param->getName());
1026 Address ParamLocalAddr =
1027 CreateMemTemp(T: Param->getType(), Name: Param->getName());
1028 Builder.CreateStore(Val: ParamVal, Addr: ParamLocalAddr);
1029 LocalAddrs.insert(KV: {Param, {Param, ParamLocalAddr}});
1030 ++FieldIdx;
1031 }
1032 }
1033
1034 CodeGenFunction::OMPPrivateScope LocalScope(*this);
1035 for (const auto &LocalAddrPair : LocalAddrs) {
1036 if (LocalAddrPair.second.first)
1037 LocalScope.addPrivate(LocalVD: LocalAddrPair.second.first,
1038 Addr: LocalAddrPair.second.second);
1039 }
1040 (void)LocalScope.Privatize();
1041 for (const auto &VLASizePair : VLASizes)
1042 VLASizeMap[VLASizePair.second.first] = VLASizePair.second.second;
1043 PGO->assignRegionCounters(GD: GlobalDecl(CD), Fn: F);
1044 CapturedStmtInfo->EmitBody(CGF&: *this, S: CD->getBody());
1045 (void)LocalScope.ForceCleanup();
1046 FinishFunction(EndLoc: CD->getBodyRBrace());
1047
1048 if (!NeedWrapperFunction)
1049 return F;
1050
1051 // Reverse the order.
1052 WrapperF->removeFromParent();
1053 F->getParent()->getFunctionList().insertAfter(where: F->getIterator(), New: WrapperF);
1054
1055 llvm::Align PtrAlign = CGM.getDataLayout().getPointerABIAlignment(AS: 0);
1056 llvm::SmallVector<llvm::Value *, 16> CallArgs;
1057 assert(CD->getContextParamPosition() == 0 &&
1058 "Expected context param at position 0 for target regions");
1059 assert(RD->getNumFields() + 1 == F->getNumOperands() &&
1060 "Argument count mismatch");
1061
1062 for (auto [FD, InnerParam, SlotIdx] : llvm::zip(
1063 t: RD->fields(), u: F->args(), args: llvm::seq<unsigned>(Size: RD->getNumFields()))) {
1064 llvm::Value *SlotPtr = WrapperCGF.Builder.CreateConstInBoundsGEP1_32(
1065 Ty: WrapperCGF.Builder.getPtrTy(), Ptr: WrapperContextV, Idx0: SlotIdx);
1066 llvm::Value *Slot = WrapperCGF.Builder.CreateAlignedLoad(
1067 Ty: WrapperCGF.Builder.getPtrTy(), Ptr: SlotPtr, Align: PtrAlign);
1068 llvm::Value *Val = WrapperCGF.Builder.CreateAlignedLoad(
1069 Ty: InnerParam.getType(), Ptr: Slot, Align: PtrAlign, Name: InnerParam.getName());
1070 CallArgs.push_back(Elt: Val);
1071 }
1072
1073 // Handle the load from the implicit dyn_ptr at the end of the __context.
1074 unsigned SlotIdx = RD->getNumFields();
1075 auto InnerParam = F->arg_begin() + SlotIdx;
1076 llvm::Value *SlotPtr = WrapperCGF.Builder.CreateConstInBoundsGEP1_32(
1077 Ty: WrapperCGF.Builder.getPtrTy(), Ptr: WrapperContextV, Idx0: SlotIdx);
1078 llvm::Value *Slot = WrapperCGF.Builder.CreateAlignedLoad(
1079 Ty: WrapperCGF.Builder.getPtrTy(), Ptr: SlotPtr, Align: PtrAlign);
1080 llvm::Value *Val = WrapperCGF.Builder.CreateAlignedLoad(
1081 Ty: InnerParam->getType(), Ptr: Slot, Align: PtrAlign, Name: InnerParam->getName());
1082 CallArgs.push_back(Elt: Val);
1083
1084 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF&: WrapperCGF, Loc, OutlinedFn: F, Args: CallArgs);
1085 WrapperCGF.FinishFunction();
1086 return WrapperF;
1087}
1088
1089//===----------------------------------------------------------------------===//
1090// OpenMP Directive Emission
1091//===----------------------------------------------------------------------===//
1092void CodeGenFunction::EmitOMPAggregateAssign(
1093 Address DestAddr, Address SrcAddr, QualType OriginalType,
1094 const llvm::function_ref<void(Address, Address)> CopyGen) {
1095 // Perform element-by-element initialization.
1096 QualType ElementTy;
1097
1098 // Drill down to the base element type on both arrays.
1099 const ArrayType *ArrayTy = OriginalType->getAsArrayTypeUnsafe();
1100 llvm::Value *NumElements = emitArrayLength(arrayType: ArrayTy, baseType&: ElementTy, addr&: DestAddr);
1101 SrcAddr = SrcAddr.withElementType(ElemTy: DestAddr.getElementType());
1102
1103 llvm::Value *SrcBegin = SrcAddr.emitRawPointer(CGF&: *this);
1104 llvm::Value *DestBegin = DestAddr.emitRawPointer(CGF&: *this);
1105 // Cast from pointer to array type to pointer to single element.
1106 llvm::Value *DestEnd = Builder.CreateInBoundsGEP(Ty: DestAddr.getElementType(),
1107 Ptr: DestBegin, IdxList: NumElements);
1108
1109 // The basic structure here is a while-do loop.
1110 llvm::BasicBlock *BodyBB = createBasicBlock(name: "omp.arraycpy.body");
1111 llvm::BasicBlock *DoneBB = createBasicBlock(name: "omp.arraycpy.done");
1112 llvm::Value *IsEmpty =
1113 Builder.CreateICmpEQ(LHS: DestBegin, RHS: DestEnd, Name: "omp.arraycpy.isempty");
1114 Builder.CreateCondBr(Cond: IsEmpty, True: DoneBB, False: BodyBB);
1115
1116 // Enter the loop body, making that address the current address.
1117 llvm::BasicBlock *EntryBB = Builder.GetInsertBlock();
1118 EmitBlock(BB: BodyBB);
1119
1120 CharUnits ElementSize = getContext().getTypeSizeInChars(T: ElementTy);
1121
1122 llvm::PHINode *SrcElementPHI =
1123 Builder.CreatePHI(Ty: SrcBegin->getType(), NumReservedValues: 2, Name: "omp.arraycpy.srcElementPast");
1124 SrcElementPHI->addIncoming(V: SrcBegin, BB: EntryBB);
1125 Address SrcElementCurrent =
1126 Address(SrcElementPHI, SrcAddr.getElementType(),
1127 SrcAddr.getAlignment().alignmentOfArrayElement(elementSize: ElementSize));
1128
1129 llvm::PHINode *DestElementPHI = Builder.CreatePHI(
1130 Ty: DestBegin->getType(), NumReservedValues: 2, Name: "omp.arraycpy.destElementPast");
1131 DestElementPHI->addIncoming(V: DestBegin, BB: EntryBB);
1132 Address DestElementCurrent =
1133 Address(DestElementPHI, DestAddr.getElementType(),
1134 DestAddr.getAlignment().alignmentOfArrayElement(elementSize: ElementSize));
1135
1136 // Emit copy.
1137 CopyGen(DestElementCurrent, SrcElementCurrent);
1138
1139 // Shift the address forward by one element.
1140 llvm::Value *DestElementNext =
1141 Builder.CreateConstGEP1_32(Ty: DestAddr.getElementType(), Ptr: DestElementPHI,
1142 /*Idx0=*/1, Name: "omp.arraycpy.dest.element");
1143 llvm::Value *SrcElementNext =
1144 Builder.CreateConstGEP1_32(Ty: SrcAddr.getElementType(), Ptr: SrcElementPHI,
1145 /*Idx0=*/1, Name: "omp.arraycpy.src.element");
1146 // Check whether we've reached the end.
1147 llvm::Value *Done =
1148 Builder.CreateICmpEQ(LHS: DestElementNext, RHS: DestEnd, Name: "omp.arraycpy.done");
1149 Builder.CreateCondBr(Cond: Done, True: DoneBB, False: BodyBB);
1150 DestElementPHI->addIncoming(V: DestElementNext, BB: Builder.GetInsertBlock());
1151 SrcElementPHI->addIncoming(V: SrcElementNext, BB: Builder.GetInsertBlock());
1152
1153 // Done.
1154 EmitBlock(BB: DoneBB, /*IsFinished=*/true);
1155}
1156
1157void CodeGenFunction::EmitOMPCopy(QualType OriginalType, Address DestAddr,
1158 Address SrcAddr, const VarDecl *DestVD,
1159 const VarDecl *SrcVD, const Expr *Copy) {
1160 if (OriginalType->isArrayType()) {
1161 const auto *BO = dyn_cast<BinaryOperator>(Val: Copy);
1162 if (BO && BO->getOpcode() == BO_Assign) {
1163 // Perform simple memcpy for simple copying.
1164 LValue Dest = MakeAddrLValue(Addr: DestAddr, T: OriginalType);
1165 LValue Src = MakeAddrLValue(Addr: SrcAddr, T: OriginalType);
1166 EmitAggregateAssign(Dest, Src, EltTy: OriginalType);
1167 } else {
1168 // For arrays with complex element types perform element by element
1169 // copying.
1170 EmitOMPAggregateAssign(
1171 DestAddr, SrcAddr, OriginalType,
1172 CopyGen: [this, Copy, SrcVD, DestVD](Address DestElement, Address SrcElement) {
1173 // Working with the single array element, so have to remap
1174 // destination and source variables to corresponding array
1175 // elements.
1176 CodeGenFunction::OMPPrivateScope Remap(*this);
1177 Remap.addPrivate(LocalVD: DestVD, Addr: DestElement);
1178 Remap.addPrivate(LocalVD: SrcVD, Addr: SrcElement);
1179 (void)Remap.Privatize();
1180 EmitIgnoredExpr(E: Copy);
1181 });
1182 }
1183 } else {
1184 // Remap pseudo source variable to private copy.
1185 CodeGenFunction::OMPPrivateScope Remap(*this);
1186 Remap.addPrivate(LocalVD: SrcVD, Addr: SrcAddr);
1187 Remap.addPrivate(LocalVD: DestVD, Addr: DestAddr);
1188 (void)Remap.Privatize();
1189 // Emit copying of the whole variable.
1190 EmitIgnoredExpr(E: Copy);
1191 }
1192}
1193
1194bool CodeGenFunction::EmitOMPFirstprivateClause(const OMPExecutableDirective &D,
1195 OMPPrivateScope &PrivateScope) {
1196 if (!HaveInsertPoint())
1197 return false;
1198 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
1199 bool DeviceConstTarget = getLangOpts().OpenMPIsTargetDevice &&
1200 isOpenMPTargetExecutionDirective(DKind: EKind);
1201 bool FirstprivateIsLastprivate = false;
1202 llvm::SmallDenseMap<const Decl *, OpenMPLastprivateModifier> Lastprivates;
1203 for (const auto *C : D.getClausesOfKind<OMPLastprivateClause>()) {
1204 for (const auto *D : C->varlist()) {
1205 const auto *VD = cast<DeclRefExpr>(Val: D)->getDecl();
1206 Lastprivates.try_emplace(Key: VD->getCanonicalDecl(), Args: C->getKind());
1207 }
1208 }
1209 llvm::DenseSet<const VarDecl *> EmittedAsFirstprivate;
1210 llvm::SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
1211 getOpenMPCaptureRegions(CaptureRegions, DKind: EKind);
1212 // Force emission of the firstprivate copy if the directive does not emit
1213 // outlined function, like omp for, omp simd, omp distribute etc.
1214 bool MustEmitFirstprivateCopy =
1215 CaptureRegions.size() == 1 && CaptureRegions.back() == OMPD_unknown;
1216 for (const auto *C : D.getClausesOfKind<OMPFirstprivateClause>()) {
1217 const auto *IRef = C->varlist_begin();
1218 const auto *InitsRef = C->inits().begin();
1219 for (const Expr *IInit : C->private_copies()) {
1220 const auto *OrigDecl = cast<DeclRefExpr>(Val: *IRef)->getDecl();
1221 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: IInit)->getDecl());
1222
1223 if (const auto *BD = dyn_cast<BindingDecl>(Val: OrigDecl)) {
1224 // Check if this binding is also lastprivate.
1225 bool ThisFirstprivateIsLastprivate =
1226 Lastprivates.count(Val: BD->getCanonicalDecl()) > 0;
1227 const auto *DD = cast<VarDecl>(Val: BD->getDecomposedDecl());
1228 // If the decomposition is captured by copy, the captured field is
1229 // already a private copy; map the binding to its member directly.
1230 if (!MustEmitFirstprivateCopy && !ThisFirstprivateIsLastprivate) {
1231 if (const FieldDecl *FD = CapturedStmtInfo->lookup(VD: DD)) {
1232 if (!FD->getType()->isReferenceType()) {
1233 bool IsRegistered = PrivateScope.addPrivate(
1234 LocalVD: BD, Addr: EmitOMPCapturedBindingLValue(BD).getAddress());
1235 assert(IsRegistered &&
1236 "firstprivate var already registered as firstprivate");
1237 (void)IsRegistered;
1238 ++IRef;
1239 ++InitsRef;
1240 continue;
1241 }
1242 }
1243 }
1244 const auto *VDInit =
1245 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *InitsRef)->getDecl());
1246 Address OriginalAddr =
1247 EmitOMPBindingOriginalAddr(BD, Loc: (*IRef)->getExprLoc());
1248
1249 QualType Type = VD->getType();
1250 bool IsRegistered;
1251 if (Type->isArrayType()) {
1252 // For array bindings, use array copy logic
1253 AutoVarEmission Emission = EmitAutoVarAlloca(var: *VD);
1254 const Expr *Init = VD->getInit();
1255 LValue OriginalLVal = MakeAddrLValue(Addr: OriginalAddr, T: Type);
1256 if (!Init || !isa<CXXConstructExpr>(Val: Init) ||
1257 isTrivialInitializer(Init)) {
1258 // Perform simple memcpy.
1259 LValue Dest = MakeAddrLValue(Addr: Emission.getAllocatedAddress(), T: Type);
1260 EmitAggregateAssign(Dest, Src: OriginalLVal, EltTy: Type);
1261 } else {
1262 EmitOMPAggregateAssign(
1263 DestAddr: Emission.getAllocatedAddress(), SrcAddr: OriginalAddr, OriginalType: Type,
1264 CopyGen: [this, VDInit, Init](Address DestElement, Address SrcElement) {
1265 // Clean up any temporaries needed by the initialization.
1266 RunCleanupsScope InitScope(*this);
1267 // Emit initialization for single element.
1268 setAddrOfLocalVar(VD: VDInit, Addr: SrcElement);
1269 EmitAnyExprToMem(E: Init, Location: DestElement,
1270 Quals: Init->getType().getQualifiers(),
1271 /*IsInitializer*/ false);
1272 LocalDeclMap.erase(Val: VDInit);
1273 });
1274 }
1275 EmitAutoVarCleanups(emission: Emission);
1276 IsRegistered =
1277 PrivateScope.addPrivate(LocalVD: BD, Addr: Emission.getAllocatedAddress());
1278 } else {
1279 // VD now has the binding's type (e.g., int), not the struct type.
1280 // Emit VD initialized from the binding's field address.
1281 setAddrOfLocalVar(VD: VDInit, Addr: OriginalAddr);
1282 EmitDecl(D: *VD);
1283 LocalDeclMap.erase(Val: VDInit);
1284 Address VDAddr = GetAddrOfLocalVar(VD);
1285 // VD is the private copy of the binding, map BD to VDAddr directly
1286 IsRegistered = PrivateScope.addPrivate(LocalVD: BD, Addr: VDAddr);
1287 }
1288
1289 assert(IsRegistered &&
1290 "firstprivate var already registered as firstprivate");
1291 (void)IsRegistered;
1292 FirstprivateIsLastprivate =
1293 FirstprivateIsLastprivate || ThisFirstprivateIsLastprivate;
1294 ++IRef;
1295 ++InitsRef;
1296 continue;
1297 }
1298
1299 // Original VarDecl logic.
1300 const VarDecl *OrigVD = dyn_cast<VarDecl>(Val: OrigDecl);
1301 assert(OrigVD && "Expected VarDecl for non-BindingDecl firstprivate");
1302 bool ThisFirstprivateIsLastprivate =
1303 Lastprivates.count(Val: OrigVD->getCanonicalDecl()) > 0;
1304 const FieldDecl *FD = CapturedStmtInfo->lookup(VD: OrigVD);
1305 if (!MustEmitFirstprivateCopy && !ThisFirstprivateIsLastprivate && FD &&
1306 !FD->getType()->isReferenceType() &&
1307 (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())) {
1308 EmittedAsFirstprivate.insert(V: OrigVD->getCanonicalDecl());
1309 ++IRef;
1310 ++InitsRef;
1311 continue;
1312 }
1313 // Do not emit copy for firstprivate constant variables in target regions,
1314 // captured by reference.
1315 if (DeviceConstTarget && OrigVD->getType().isConstant(Ctx: getContext()) &&
1316 FD && FD->getType()->isReferenceType() &&
1317 (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())) {
1318 EmittedAsFirstprivate.insert(V: OrigVD->getCanonicalDecl());
1319 ++IRef;
1320 ++InitsRef;
1321 continue;
1322 }
1323 FirstprivateIsLastprivate =
1324 FirstprivateIsLastprivate || ThisFirstprivateIsLastprivate;
1325 if (EmittedAsFirstprivate.insert(V: OrigVD->getCanonicalDecl()).second) {
1326 const auto *VDInit =
1327 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *InitsRef)->getDecl());
1328 bool IsRegistered;
1329 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(OrigVD),
1330 /*RefersToEnclosingVariableOrCapture=*/FD != nullptr,
1331 (*IRef)->getType(), VK_LValue, (*IRef)->getExprLoc());
1332 LValue OriginalLVal;
1333 if (!FD) {
1334 // Check if the firstprivate variable is just a constant value.
1335 ConstantEmission CE = tryEmitAsConstant(RefExpr: &DRE);
1336 if (CE && !CE.isReference()) {
1337 // Constant value, no need to create a copy.
1338 ++IRef;
1339 ++InitsRef;
1340 continue;
1341 }
1342 if (CE && CE.isReference()) {
1343 OriginalLVal = CE.getReferenceLValue(CGF&: *this, RefExpr: &DRE);
1344 } else {
1345 assert(!CE && "Expected non-constant firstprivate.");
1346 OriginalLVal = EmitLValue(E: &DRE);
1347 }
1348 } else {
1349 OriginalLVal = EmitLValue(E: &DRE);
1350 }
1351 QualType Type = VD->getType();
1352 if (Type->isArrayType()) {
1353 // Emit VarDecl with copy init for arrays.
1354 // Get the address of the original variable captured in current
1355 // captured region.
1356 AutoVarEmission Emission = EmitAutoVarAlloca(var: *VD);
1357 const Expr *Init = VD->getInit();
1358 if (!isa<CXXConstructExpr>(Val: Init) || isTrivialInitializer(Init)) {
1359 // Perform simple memcpy.
1360 LValue Dest = MakeAddrLValue(Addr: Emission.getAllocatedAddress(), T: Type);
1361 EmitAggregateAssign(Dest, Src: OriginalLVal, EltTy: Type);
1362 } else {
1363 EmitOMPAggregateAssign(
1364 DestAddr: Emission.getAllocatedAddress(), SrcAddr: OriginalLVal.getAddress(), OriginalType: Type,
1365 CopyGen: [this, VDInit, Init](Address DestElement, Address SrcElement) {
1366 // Clean up any temporaries needed by the
1367 // initialization.
1368 RunCleanupsScope InitScope(*this);
1369 // Emit initialization for single element.
1370 setAddrOfLocalVar(VD: VDInit, Addr: SrcElement);
1371 EmitAnyExprToMem(E: Init, Location: DestElement,
1372 Quals: Init->getType().getQualifiers(),
1373 /*IsInitializer*/ false);
1374 LocalDeclMap.erase(Val: VDInit);
1375 });
1376 }
1377 EmitAutoVarCleanups(emission: Emission);
1378 IsRegistered =
1379 PrivateScope.addPrivate(LocalVD: OrigVD, Addr: Emission.getAllocatedAddress());
1380 } else {
1381 Address OriginalAddr = OriginalLVal.getAddress();
1382 // Emit private VarDecl with copy init.
1383 // Remap temp VDInit variable to the address of the original
1384 // variable (for proper handling of captured global variables).
1385 setAddrOfLocalVar(VD: VDInit, Addr: OriginalAddr);
1386 EmitDecl(D: *VD);
1387 LocalDeclMap.erase(Val: VDInit);
1388 Address VDAddr = GetAddrOfLocalVar(VD);
1389 if (ThisFirstprivateIsLastprivate &&
1390 Lastprivates[OrigVD->getCanonicalDecl()] ==
1391 OMPC_LASTPRIVATE_conditional) {
1392 // Create/init special variable for lastprivate conditionals.
1393 llvm::Value *V =
1394 EmitLoadOfScalar(lvalue: MakeAddrLValue(Addr: VDAddr, T: (*IRef)->getType(),
1395 Source: AlignmentSource::Decl),
1396 Loc: (*IRef)->getExprLoc());
1397 VDAddr = CGM.getOpenMPRuntime().emitLastprivateConditionalInit(
1398 CGF&: *this, VD: OrigVD);
1399 EmitStoreOfScalar(value: V, lvalue: MakeAddrLValue(Addr: VDAddr, T: (*IRef)->getType(),
1400 Source: AlignmentSource::Decl));
1401 LocalDeclMap.erase(Val: VD);
1402 setAddrOfLocalVar(VD, Addr: VDAddr);
1403 }
1404 IsRegistered = PrivateScope.addPrivate(LocalVD: OrigVD, Addr: VDAddr);
1405 }
1406 assert(IsRegistered &&
1407 "firstprivate var already registered as private");
1408 // Silence the warning about unused variable.
1409 (void)IsRegistered;
1410 }
1411 ++IRef;
1412 ++InitsRef;
1413 }
1414 }
1415 return FirstprivateIsLastprivate && !EmittedAsFirstprivate.empty();
1416}
1417
1418void CodeGenFunction::EmitOMPPrivateClause(
1419 const OMPExecutableDirective &D,
1420 CodeGenFunction::OMPPrivateScope &PrivateScope) {
1421 if (!HaveInsertPoint())
1422 return;
1423 llvm::SmallDenseSet<const ValueDecl *> EmittedAsPrivate;
1424 for (const auto *C : D.getClausesOfKind<OMPPrivateClause>()) {
1425 auto IRef = C->varlist_begin();
1426 for (const Expr *IInit : C->private_copies()) {
1427 const auto *OrigDecl = cast<DeclRefExpr>(Val: *IRef)->getDecl();
1428 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: IInit)->getDecl());
1429 if (EmittedAsPrivate.insert(V: cast<ValueDecl>(Val: OrigDecl->getCanonicalDecl()))
1430 .second) {
1431 EmitDecl(D: *VD);
1432 bool IsRegistered =
1433 PrivateScope.addPrivate(LocalVD: OrigDecl, Addr: GetAddrOfLocalVar(VD));
1434 assert(IsRegistered && "private var already registered as private");
1435 (void)IsRegistered;
1436 }
1437 ++IRef;
1438 }
1439 }
1440}
1441
1442bool CodeGenFunction::EmitOMPCopyinClause(const OMPExecutableDirective &D) {
1443 if (!HaveInsertPoint())
1444 return false;
1445 // threadprivate_var1 = master_threadprivate_var1;
1446 // operator=(threadprivate_var2, master_threadprivate_var2);
1447 // ...
1448 // __kmpc_barrier(&loc, global_tid);
1449 llvm::DenseSet<const VarDecl *> CopiedVars;
1450 llvm::BasicBlock *CopyBegin = nullptr, *CopyEnd = nullptr;
1451 for (const auto *C : D.getClausesOfKind<OMPCopyinClause>()) {
1452 auto IRef = C->varlist_begin();
1453 auto ISrcRef = C->source_exprs().begin();
1454 auto IDestRef = C->destination_exprs().begin();
1455 for (const Expr *AssignOp : C->assignment_ops()) {
1456 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IRef)->getDecl());
1457 QualType Type = VD->getType();
1458 if (CopiedVars.insert(V: VD->getCanonicalDecl()).second) {
1459 // Get the address of the master variable. If we are emitting code with
1460 // TLS support, the address is passed from the master as field in the
1461 // captured declaration.
1462 Address MasterAddr = Address::invalid();
1463 if (getLangOpts().OpenMPUseTLS &&
1464 getContext().getTargetInfo().isTLSSupported()) {
1465 assert(CapturedStmtInfo->lookup(VD) &&
1466 "Copyin threadprivates should have been captured!");
1467 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(VD), true,
1468 (*IRef)->getType(), VK_LValue, (*IRef)->getExprLoc());
1469 MasterAddr = EmitLValue(E: &DRE).getAddress();
1470 LocalDeclMap.erase(Val: VD);
1471 } else {
1472 MasterAddr =
1473 Address(VD->isStaticLocal() ? CGM.getStaticLocalDeclAddress(D: VD)
1474 : CGM.GetAddrOfGlobal(GD: VD),
1475 CGM.getTypes().ConvertTypeForMem(T: VD->getType()),
1476 getContext().getDeclAlign(D: VD));
1477 }
1478 // Get the address of the threadprivate variable.
1479 Address PrivateAddr = EmitLValue(E: *IRef).getAddress();
1480 if (CopiedVars.size() == 1) {
1481 // At first check if current thread is a master thread. If it is, no
1482 // need to copy data.
1483 CopyBegin = createBasicBlock(name: "copyin.not.master");
1484 CopyEnd = createBasicBlock(name: "copyin.not.master.end");
1485 // TODO: Avoid ptrtoint conversion.
1486 auto *MasterAddrInt = Builder.CreatePtrToInt(
1487 V: MasterAddr.emitRawPointer(CGF&: *this), DestTy: CGM.IntPtrTy);
1488 auto *PrivateAddrInt = Builder.CreatePtrToInt(
1489 V: PrivateAddr.emitRawPointer(CGF&: *this), DestTy: CGM.IntPtrTy);
1490 Builder.CreateCondBr(
1491 Cond: Builder.CreateICmpNE(LHS: MasterAddrInt, RHS: PrivateAddrInt), True: CopyBegin,
1492 False: CopyEnd);
1493 EmitBlock(BB: CopyBegin);
1494 }
1495 const auto *SrcVD =
1496 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ISrcRef)->getDecl());
1497 const auto *DestVD =
1498 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IDestRef)->getDecl());
1499 EmitOMPCopy(OriginalType: Type, DestAddr: PrivateAddr, SrcAddr: MasterAddr, DestVD, SrcVD, Copy: AssignOp);
1500 }
1501 ++IRef;
1502 ++ISrcRef;
1503 ++IDestRef;
1504 }
1505 }
1506 if (CopyEnd) {
1507 // Exit out of copying procedure for non-master thread.
1508 EmitBlock(BB: CopyEnd, /*IsFinished=*/true);
1509 return true;
1510 }
1511 return false;
1512}
1513
1514bool CodeGenFunction::EmitOMPLastprivateClauseInit(
1515 const OMPExecutableDirective &D, OMPPrivateScope &PrivateScope) {
1516 if (!HaveInsertPoint())
1517 return false;
1518 bool HasAtLeastOneLastprivate = false;
1519 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
1520 llvm::DenseSet<const VarDecl *> SIMDLCVs;
1521 if (isOpenMPSimdDirective(DKind: EKind)) {
1522 const auto *LoopDirective = cast<OMPLoopDirective>(Val: &D);
1523 for (const Expr *C : LoopDirective->counters()) {
1524 SIMDLCVs.insert(
1525 V: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: C)->getDecl())->getCanonicalDecl());
1526 }
1527 }
1528 llvm::SmallDenseSet<const ValueDecl *> AlreadyEmittedVars;
1529 for (const auto *C : D.getClausesOfKind<OMPLastprivateClause>()) {
1530 HasAtLeastOneLastprivate = true;
1531 if (isOpenMPTaskLoopDirective(DKind: EKind) && !getLangOpts().OpenMPSimd)
1532 break;
1533 const auto *IRef = C->varlist_begin();
1534 const auto *IDestRef = C->destination_exprs().begin();
1535 for (const Expr *IInit : C->private_copies()) {
1536 // Keep the address of the original variable for future update at the end
1537 // of the loop.
1538 const auto *OrigDecl = cast<DeclRefExpr>(Val: *IRef)->getDecl();
1539 // Handle BindingDecls with the same level of support as VarDecls.
1540 if (const auto *BD = dyn_cast<BindingDecl>(Val: OrigDecl)) {
1541 if (AlreadyEmittedVars.insert(V: cast<ValueDecl>(Val: BD->getCanonicalDecl()))
1542 .second) {
1543 const auto *DestVD =
1544 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IDestRef)->getDecl());
1545
1546 // Get the original binding address.
1547 Address OrigAddr =
1548 EmitOMPBindingOriginalAddr(BD, Loc: (*IRef)->getExprLoc());
1549 PrivateScope.addPrivate(LocalVD: DestVD, Addr: OrigAddr);
1550 if (IInit) {
1551 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: IInit)->getDecl());
1552 // Emit private VarDecl with copy init.
1553 EmitDecl(D: *VD);
1554 Address VDAddr = GetAddrOfLocalVar(VD);
1555 bool IsRegistered = PrivateScope.addPrivate(LocalVD: BD, Addr: VDAddr);
1556 assert(IsRegistered &&
1557 "lastprivate binding already registered as private");
1558 (void)IsRegistered;
1559 }
1560 }
1561 ++IRef;
1562 ++IDestRef;
1563 continue;
1564 }
1565 const auto *OrigVD = cast<VarDecl>(Val: OrigDecl);
1566 // Taskloops do not require additional initialization, it is done in
1567 // runtime support library.
1568 if (AlreadyEmittedVars.insert(V: OrigVD->getCanonicalDecl()).second) {
1569 const auto *DestVD =
1570 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IDestRef)->getDecl());
1571 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(OrigVD),
1572 /*RefersToEnclosingVariableOrCapture=*/
1573 CapturedStmtInfo->lookup(VD: OrigVD) != nullptr,
1574 (*IRef)->getType(), VK_LValue, (*IRef)->getExprLoc());
1575 PrivateScope.addPrivate(LocalVD: DestVD, Addr: EmitLValue(E: &DRE).getAddress());
1576 // Check if the variable is also a firstprivate: in this case IInit is
1577 // not generated. Initialization of this variable will happen in codegen
1578 // for 'firstprivate' clause.
1579 if (IInit && !SIMDLCVs.count(V: OrigVD->getCanonicalDecl())) {
1580 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: IInit)->getDecl());
1581 Address VDAddr = Address::invalid();
1582 if (C->getKind() == OMPC_LASTPRIVATE_conditional) {
1583 VDAddr = CGM.getOpenMPRuntime().emitLastprivateConditionalInit(
1584 CGF&: *this, VD: OrigVD);
1585 setAddrOfLocalVar(VD, Addr: VDAddr);
1586 } else {
1587 // Emit private VarDecl with copy init.
1588 EmitDecl(D: *VD);
1589 VDAddr = GetAddrOfLocalVar(VD);
1590 }
1591 bool IsRegistered = PrivateScope.addPrivate(LocalVD: OrigVD, Addr: VDAddr);
1592 assert(IsRegistered &&
1593 "lastprivate var already registered as private");
1594 (void)IsRegistered;
1595 }
1596 }
1597 ++IRef;
1598 ++IDestRef;
1599 }
1600 }
1601 return HasAtLeastOneLastprivate;
1602}
1603
1604void CodeGenFunction::EmitOMPLastprivateClauseFinal(
1605 const OMPExecutableDirective &D, bool NoFinals,
1606 llvm::Value *IsLastIterCond) {
1607 if (!HaveInsertPoint())
1608 return;
1609 // Emit following code:
1610 // if (<IsLastIterCond>) {
1611 // orig_var1 = private_orig_var1;
1612 // ...
1613 // orig_varn = private_orig_varn;
1614 // }
1615 llvm::BasicBlock *ThenBB = nullptr;
1616 llvm::BasicBlock *DoneBB = nullptr;
1617 if (IsLastIterCond) {
1618 // Emit implicit barrier if at least one lastprivate conditional is found
1619 // and this is not a simd mode.
1620 if (!getLangOpts().OpenMPSimd &&
1621 llvm::any_of(Range: D.getClausesOfKind<OMPLastprivateClause>(),
1622 P: [](const OMPLastprivateClause *C) {
1623 return C->getKind() == OMPC_LASTPRIVATE_conditional;
1624 })) {
1625 CGM.getOpenMPRuntime().emitBarrierCall(CGF&: *this, Loc: D.getBeginLoc(),
1626 Kind: OMPD_unknown,
1627 /*EmitChecks=*/false,
1628 /*ForceSimpleCall=*/true);
1629 }
1630 ThenBB = createBasicBlock(name: ".omp.lastprivate.then");
1631 DoneBB = createBasicBlock(name: ".omp.lastprivate.done");
1632 Builder.CreateCondBr(Cond: IsLastIterCond, True: ThenBB, False: DoneBB);
1633 EmitBlock(BB: ThenBB);
1634 }
1635 llvm::DenseSet<const ValueDecl *> AlreadyEmittedVars;
1636 llvm::SmallDenseMap<const VarDecl *, const Expr *> LoopCountersAndUpdates;
1637 if (const auto *LoopDirective = dyn_cast<OMPLoopDirective>(Val: &D)) {
1638 auto IC = LoopDirective->counters().begin();
1639 for (const Expr *F : LoopDirective->finals()) {
1640 const auto *D =
1641 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IC)->getDecl())->getCanonicalDecl();
1642 if (NoFinals)
1643 AlreadyEmittedVars.insert(V: D);
1644 else
1645 LoopCountersAndUpdates[D] = F;
1646 ++IC;
1647 }
1648 }
1649 for (const auto *C : D.getClausesOfKind<OMPLastprivateClause>()) {
1650 auto IRef = C->varlist_begin();
1651 auto ISrcRef = C->source_exprs().begin();
1652 auto IDestRef = C->destination_exprs().begin();
1653 for (const Expr *AssignOp : C->assignment_ops()) {
1654 const auto *PrivateDecl = cast<DeclRefExpr>(Val: *IRef)->getDecl();
1655
1656 // For BindingDecls, check if we should use .lastprivate.src or the BD
1657 // itself.
1658 const VarDecl *PrivateVD = nullptr;
1659 const BindingDecl *BD = nullptr;
1660 if ((BD = dyn_cast<BindingDecl>(Val: PrivateDecl))) {
1661 // Check if .lastprivate.src is available (taskloop case).
1662 const auto *SrcVD =
1663 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ISrcRef)->getDecl());
1664
1665 if (LocalDeclMap.count(Val: SrcVD)) {
1666 // Taskloop case: use .lastprivate.src.
1667 PrivateVD = SrcVD;
1668 } else if (OMPPrivatizedBindings.count(Val: BD)) {
1669 // Parallel for case: BindingDecl is directly privatized.
1670 // Leave PrivateVD as nullptr to handle specially below.
1671 }
1672 } else {
1673 PrivateVD = cast<VarDecl>(Val: PrivateDecl);
1674 }
1675
1676 QualType Type = PrivateVD ? PrivateVD->getType() : BD->getType();
1677 const auto *CanonicalVD =
1678 PrivateVD ? PrivateVD->getCanonicalDecl() : nullptr;
1679
1680 // Check if already emitted.
1681 bool ShouldEmit =
1682 AlreadyEmittedVars
1683 .insert(V: PrivateVD ? static_cast<const ValueDecl *>(CanonicalVD)
1684 : static_cast<const ValueDecl *>(BD))
1685 .second;
1686 if (ShouldEmit) {
1687 // If lastprivate variable is a loop control variable for loop-based
1688 // directive, update its value before copyin back to original
1689 // variable.
1690 if (CanonicalVD) {
1691 if (const Expr *FinalExpr =
1692 LoopCountersAndUpdates.lookup(Val: CanonicalVD))
1693 EmitIgnoredExpr(E: FinalExpr);
1694 }
1695 const auto *SrcVD =
1696 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ISrcRef)->getDecl());
1697 const auto *DestVD =
1698 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IDestRef)->getDecl());
1699
1700 // Get the address of the private variable.
1701 Address PrivateAddr = Address::invalid();
1702 if (PrivateVD) {
1703 PrivateAddr = GetAddrOfLocalVar(VD: PrivateVD);
1704 } else {
1705 auto It = OMPPrivatizedBindings.find(Val: BD);
1706 assert(It != OMPPrivatizedBindings.end() &&
1707 "BindingDecl should be privatized");
1708 PrivateAddr = It->second;
1709 }
1710 if (PrivateVD) {
1711 if (const auto *RefTy = PrivateVD->getType()->getAs<ReferenceType>())
1712 PrivateAddr = Address(
1713 Builder.CreateLoad(Addr: PrivateAddr),
1714 CGM.getTypes().ConvertTypeForMem(T: RefTy->getPointeeType()),
1715 CGM.getNaturalTypeAlignment(T: RefTy->getPointeeType()));
1716 }
1717 // Store the last value to the private copy in the last iteration.
1718 if (C->getKind() == OMPC_LASTPRIVATE_conditional)
1719 CGM.getOpenMPRuntime().emitLastprivateConditionalFinalUpdate(
1720 CGF&: *this, PrivLVal: MakeAddrLValue(Addr: PrivateAddr, T: (*IRef)->getType()), VD: PrivateVD,
1721 Loc: (*IRef)->getExprLoc());
1722 // Get the address of the original variable.
1723 Address OriginalAddr = GetAddrOfLocalVar(VD: DestVD);
1724 EmitOMPCopy(OriginalType: Type, DestAddr: OriginalAddr, SrcAddr: PrivateAddr, DestVD, SrcVD, Copy: AssignOp);
1725 }
1726 ++IRef;
1727 ++ISrcRef;
1728 ++IDestRef;
1729 }
1730 if (const Expr *PostUpdate = C->getPostUpdateExpr())
1731 EmitIgnoredExpr(E: PostUpdate);
1732 }
1733 if (IsLastIterCond)
1734 EmitBlock(BB: DoneBB, /*IsFinished=*/true);
1735}
1736
1737void CodeGenFunction::EmitOMPReductionClauseInit(
1738 const OMPExecutableDirective &D,
1739 CodeGenFunction::OMPPrivateScope &PrivateScope, bool ForInscan) {
1740 if (!HaveInsertPoint())
1741 return;
1742 SmallVector<const Expr *, 4> Shareds;
1743 SmallVector<const Expr *, 4> Privates;
1744 SmallVector<const Expr *, 4> ReductionOps;
1745 SmallVector<const Expr *, 4> LHSs;
1746 SmallVector<const Expr *, 4> RHSs;
1747 OMPTaskDataTy Data;
1748 SmallVector<const Expr *, 4> TaskLHSs;
1749 SmallVector<const Expr *, 4> TaskRHSs;
1750 for (const auto *C : D.getClausesOfKind<OMPReductionClause>()) {
1751 if (ForInscan != (C->getModifier() == OMPC_REDUCTION_inscan))
1752 continue;
1753 Shareds.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
1754 Privates.append(in_start: C->privates().begin(), in_end: C->privates().end());
1755 ReductionOps.append(in_start: C->reduction_ops().begin(), in_end: C->reduction_ops().end());
1756 LHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
1757 RHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
1758 if (C->getModifier() == OMPC_REDUCTION_task) {
1759 Data.ReductionVars.append(in_start: C->privates().begin(), in_end: C->privates().end());
1760 Data.ReductionOrigs.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
1761 Data.ReductionCopies.append(in_start: C->privates().begin(), in_end: C->privates().end());
1762 Data.ReductionOps.append(in_start: C->reduction_ops().begin(),
1763 in_end: C->reduction_ops().end());
1764 TaskLHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
1765 TaskRHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
1766 }
1767 }
1768 ReductionCodeGen RedCG(Shareds, Shareds, Privates, ReductionOps);
1769 unsigned Count = 0;
1770 auto *ILHS = LHSs.begin();
1771 auto *IRHS = RHSs.begin();
1772 auto *IPriv = Privates.begin();
1773 for (const Expr *IRef : Shareds) {
1774 const auto *PrivateVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IPriv)->getDecl());
1775 RedCG.emitSharedOrigLValue(CGF&: *this, N: Count);
1776 RedCG.emitAggregateType(CGF&: *this, N: Count);
1777 AutoVarEmission Emission = EmitAutoVarAlloca(var: *PrivateVD);
1778 RedCG.emitInitialization(CGF&: *this, N: Count, PrivateAddr: Emission.getAllocatedAddress(),
1779 SharedAddr: RedCG.getSharedLValue(N: Count).getAddress(),
1780 DefaultInit: [&Emission](CodeGenFunction &CGF) {
1781 CGF.EmitAutoVarInit(emission: Emission);
1782 return true;
1783 });
1784 EmitAutoVarCleanups(emission: Emission);
1785 Address BaseAddr = RedCG.adjustPrivateAddress(
1786 CGF&: *this, N: Count, PrivateAddr: Emission.getAllocatedAddress());
1787 bool IsRegistered =
1788 PrivateScope.addPrivate(LocalVD: RedCG.getBaseDecl(N: Count), Addr: BaseAddr);
1789 assert(IsRegistered && "private var already registered as private");
1790 // Silence the warning about unused variable.
1791 (void)IsRegistered;
1792
1793 const auto *LHSVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ILHS)->getDecl());
1794 const auto *RHSVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IRHS)->getDecl());
1795 QualType Type = PrivateVD->getType();
1796 bool isaOMPArraySectionExpr = isa<ArraySectionExpr>(Val: IRef);
1797 if (isaOMPArraySectionExpr && Type->isVariablyModifiedType()) {
1798 // Store the address of the original variable associated with the LHS
1799 // implicit variable.
1800 PrivateScope.addPrivate(LocalVD: LHSVD, Addr: RedCG.getSharedLValue(N: Count).getAddress());
1801 PrivateScope.addPrivate(LocalVD: RHSVD, Addr: GetAddrOfLocalVar(VD: PrivateVD));
1802 } else if ((isaOMPArraySectionExpr && Type->isScalarType()) ||
1803 isa<ArraySubscriptExpr>(Val: IRef)) {
1804 // Store the address of the original variable associated with the LHS
1805 // implicit variable.
1806 PrivateScope.addPrivate(LocalVD: LHSVD, Addr: RedCG.getSharedLValue(N: Count).getAddress());
1807 PrivateScope.addPrivate(LocalVD: RHSVD,
1808 Addr: GetAddrOfLocalVar(VD: PrivateVD).withElementType(
1809 ElemTy: ConvertTypeForMem(T: RHSVD->getType())));
1810 } else {
1811 QualType Type = PrivateVD->getType();
1812 bool IsArray = getContext().getAsArrayType(T: Type) != nullptr;
1813 Address OriginalAddr = RedCG.getSharedLValue(N: Count).getAddress();
1814 // Store the address of the original variable associated with the LHS
1815 // implicit variable.
1816 if (IsArray) {
1817 OriginalAddr =
1818 OriginalAddr.withElementType(ElemTy: ConvertTypeForMem(T: LHSVD->getType()));
1819 }
1820 PrivateScope.addPrivate(LocalVD: LHSVD, Addr: OriginalAddr);
1821 PrivateScope.addPrivate(
1822 LocalVD: RHSVD, Addr: IsArray ? GetAddrOfLocalVar(VD: PrivateVD).withElementType(
1823 ElemTy: ConvertTypeForMem(T: RHSVD->getType()))
1824 : GetAddrOfLocalVar(VD: PrivateVD));
1825 }
1826 ++ILHS;
1827 ++IRHS;
1828 ++IPriv;
1829 ++Count;
1830 }
1831 if (!Data.ReductionVars.empty()) {
1832 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
1833 Data.IsReductionWithTaskMod = true;
1834 Data.IsWorksharingReduction = isOpenMPWorksharingDirective(DKind: EKind);
1835 llvm::Value *ReductionDesc = CGM.getOpenMPRuntime().emitTaskReductionInit(
1836 CGF&: *this, Loc: D.getBeginLoc(), LHSExprs: TaskLHSs, RHSExprs: TaskRHSs, Data);
1837 const Expr *TaskRedRef = nullptr;
1838 switch (EKind) {
1839 case OMPD_parallel:
1840 TaskRedRef = cast<OMPParallelDirective>(Val: D).getTaskReductionRefExpr();
1841 break;
1842 case OMPD_for:
1843 TaskRedRef = cast<OMPForDirective>(Val: D).getTaskReductionRefExpr();
1844 break;
1845 case OMPD_sections:
1846 TaskRedRef = cast<OMPSectionsDirective>(Val: D).getTaskReductionRefExpr();
1847 break;
1848 case OMPD_parallel_for:
1849 TaskRedRef = cast<OMPParallelForDirective>(Val: D).getTaskReductionRefExpr();
1850 break;
1851 case OMPD_parallel_master:
1852 TaskRedRef =
1853 cast<OMPParallelMasterDirective>(Val: D).getTaskReductionRefExpr();
1854 break;
1855 case OMPD_parallel_sections:
1856 TaskRedRef =
1857 cast<OMPParallelSectionsDirective>(Val: D).getTaskReductionRefExpr();
1858 break;
1859 case OMPD_target_parallel:
1860 TaskRedRef =
1861 cast<OMPTargetParallelDirective>(Val: D).getTaskReductionRefExpr();
1862 break;
1863 case OMPD_target_parallel_for:
1864 TaskRedRef =
1865 cast<OMPTargetParallelForDirective>(Val: D).getTaskReductionRefExpr();
1866 break;
1867 case OMPD_distribute_parallel_for:
1868 TaskRedRef =
1869 cast<OMPDistributeParallelForDirective>(Val: D).getTaskReductionRefExpr();
1870 break;
1871 case OMPD_teams_distribute_parallel_for:
1872 TaskRedRef = cast<OMPTeamsDistributeParallelForDirective>(Val: D)
1873 .getTaskReductionRefExpr();
1874 break;
1875 case OMPD_target_teams_distribute_parallel_for:
1876 TaskRedRef = cast<OMPTargetTeamsDistributeParallelForDirective>(Val: D)
1877 .getTaskReductionRefExpr();
1878 break;
1879 case OMPD_simd:
1880 case OMPD_for_simd:
1881 case OMPD_section:
1882 case OMPD_single:
1883 case OMPD_master:
1884 case OMPD_critical:
1885 case OMPD_parallel_for_simd:
1886 case OMPD_task:
1887 case OMPD_taskyield:
1888 case OMPD_error:
1889 case OMPD_barrier:
1890 case OMPD_taskwait:
1891 case OMPD_taskgroup:
1892 case OMPD_flush:
1893 case OMPD_depobj:
1894 case OMPD_scan:
1895 case OMPD_ordered_standalone:
1896 case OMPD_ordered_blockassoc:
1897 case OMPD_atomic:
1898 case OMPD_teams:
1899 case OMPD_target:
1900 case OMPD_cancellation_point:
1901 case OMPD_cancel:
1902 case OMPD_target_data:
1903 case OMPD_target_enter_data:
1904 case OMPD_target_exit_data:
1905 case OMPD_taskloop:
1906 case OMPD_taskloop_simd:
1907 case OMPD_master_taskloop:
1908 case OMPD_master_taskloop_simd:
1909 case OMPD_parallel_master_taskloop:
1910 case OMPD_parallel_master_taskloop_simd:
1911 case OMPD_distribute:
1912 case OMPD_target_update:
1913 case OMPD_distribute_parallel_for_simd:
1914 case OMPD_distribute_simd:
1915 case OMPD_target_parallel_for_simd:
1916 case OMPD_target_simd:
1917 case OMPD_teams_distribute:
1918 case OMPD_teams_distribute_simd:
1919 case OMPD_teams_distribute_parallel_for_simd:
1920 case OMPD_target_teams:
1921 case OMPD_target_teams_distribute:
1922 case OMPD_target_teams_distribute_parallel_for_simd:
1923 case OMPD_target_teams_distribute_simd:
1924 case OMPD_declare_target:
1925 case OMPD_end_declare_target:
1926 case OMPD_threadprivate:
1927 case OMPD_allocate:
1928 case OMPD_declare_reduction:
1929 case OMPD_declare_mapper:
1930 case OMPD_declare_simd:
1931 case OMPD_requires:
1932 case OMPD_declare_variant:
1933 case OMPD_begin_declare_variant:
1934 case OMPD_end_declare_variant:
1935 case OMPD_unknown:
1936 default:
1937 llvm_unreachable("Unexpected directive with task reductions.");
1938 }
1939
1940 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: TaskRedRef)->getDecl());
1941 EmitVarDecl(D: *VD);
1942 EmitStoreOfScalar(Value: ReductionDesc, Addr: GetAddrOfLocalVar(VD),
1943 /*Volatile=*/false, Ty: TaskRedRef->getType());
1944 }
1945}
1946
1947void CodeGenFunction::EmitOMPReductionClauseFinal(
1948 const OMPExecutableDirective &D, const OpenMPDirectiveKind ReductionKind) {
1949 if (!HaveInsertPoint())
1950 return;
1951 llvm::SmallVector<const Expr *, 8> Privates;
1952 llvm::SmallVector<const Expr *, 8> LHSExprs;
1953 llvm::SmallVector<const Expr *, 8> RHSExprs;
1954 llvm::SmallVector<const Expr *, 8> ReductionOps;
1955 llvm::SmallVector<bool, 8> IsPrivateVarReduction;
1956 bool HasAtLeastOneReduction = false;
1957 bool IsReductionWithTaskMod = false;
1958 for (const auto *C : D.getClausesOfKind<OMPReductionClause>()) {
1959 // Do not emit for inscan reductions.
1960 if (C->getModifier() == OMPC_REDUCTION_inscan)
1961 continue;
1962 HasAtLeastOneReduction = true;
1963 Privates.append(in_start: C->privates().begin(), in_end: C->privates().end());
1964 LHSExprs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
1965 RHSExprs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
1966 IsPrivateVarReduction.append(in_start: C->private_var_reduction_flags().begin(),
1967 in_end: C->private_var_reduction_flags().end());
1968 ReductionOps.append(in_start: C->reduction_ops().begin(), in_end: C->reduction_ops().end());
1969 IsReductionWithTaskMod =
1970 IsReductionWithTaskMod || C->getModifier() == OMPC_REDUCTION_task;
1971 }
1972 if (HasAtLeastOneReduction) {
1973 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
1974 if (IsReductionWithTaskMod) {
1975 CGM.getOpenMPRuntime().emitTaskReductionFini(
1976 CGF&: *this, Loc: D.getBeginLoc(), IsWorksharingReduction: isOpenMPWorksharingDirective(DKind: EKind));
1977 }
1978 bool TeamsLoopCanBeParallel = false;
1979 if (auto *TTLD = dyn_cast<OMPTargetTeamsGenericLoopDirective>(Val: &D))
1980 TeamsLoopCanBeParallel = TTLD->canBeParallelFor();
1981 bool WithNowait = D.getSingleClause<OMPNowaitClause>() ||
1982 isOpenMPParallelDirective(DKind: EKind) ||
1983 TeamsLoopCanBeParallel || ReductionKind == OMPD_simd;
1984 bool SimpleReduction = ReductionKind == OMPD_simd;
1985 // Emit nowait reduction if nowait clause is present or directive is a
1986 // parallel directive (it always has implicit barrier).
1987 CGM.getOpenMPRuntime().emitReduction(
1988 CGF&: *this, Loc: D.getEndLoc(), Privates, LHSExprs, RHSExprs, ReductionOps,
1989 Options: {.WithNowait: WithNowait, .SimpleReduction: SimpleReduction, .IsPrivateVarReduction: IsPrivateVarReduction, .ReductionKind: ReductionKind});
1990 }
1991}
1992
1993static void emitPostUpdateForReductionClause(
1994 CodeGenFunction &CGF, const OMPExecutableDirective &D,
1995 const llvm::function_ref<llvm::Value *(CodeGenFunction &)> CondGen) {
1996 if (!CGF.HaveInsertPoint())
1997 return;
1998 llvm::BasicBlock *DoneBB = nullptr;
1999 for (const auto *C : D.getClausesOfKind<OMPReductionClause>()) {
2000 if (const Expr *PostUpdate = C->getPostUpdateExpr()) {
2001 if (!DoneBB) {
2002 if (llvm::Value *Cond = CondGen(CGF)) {
2003 // If the first post-update expression is found, emit conditional
2004 // block if it was requested.
2005 llvm::BasicBlock *ThenBB = CGF.createBasicBlock(name: ".omp.reduction.pu");
2006 DoneBB = CGF.createBasicBlock(name: ".omp.reduction.pu.done");
2007 CGF.Builder.CreateCondBr(Cond, True: ThenBB, False: DoneBB);
2008 CGF.EmitBlock(BB: ThenBB);
2009 }
2010 }
2011 CGF.EmitIgnoredExpr(E: PostUpdate);
2012 }
2013 }
2014 if (DoneBB)
2015 CGF.EmitBlock(BB: DoneBB, /*IsFinished=*/true);
2016}
2017
2018namespace {
2019/// Codegen lambda for appending distribute lower and upper bounds to outlined
2020/// parallel function. This is necessary for combined constructs such as
2021/// 'distribute parallel for'
2022typedef llvm::function_ref<void(CodeGenFunction &,
2023 const OMPExecutableDirective &,
2024 llvm::SmallVectorImpl<llvm::Value *> &)>
2025 CodeGenBoundParametersTy;
2026} // anonymous namespace
2027
2028static void
2029checkForLastprivateConditionalUpdate(CodeGenFunction &CGF,
2030 const OMPExecutableDirective &S) {
2031 if (CGF.getLangOpts().OpenMP < 50)
2032 return;
2033 llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> PrivateDecls;
2034 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
2035 for (const Expr *Ref : C->varlist()) {
2036 if (!Ref->getType()->isScalarType())
2037 continue;
2038 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
2039 if (!DRE)
2040 continue;
2041 // Skip BindingDecls - lastprivate conditional only applies to VarDecls.
2042 const auto *VD = dyn_cast<VarDecl>(Val: DRE->getDecl());
2043 if (!VD)
2044 continue;
2045 PrivateDecls.insert(V: VD);
2046 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: Ref);
2047 }
2048 }
2049 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
2050 for (const Expr *Ref : C->varlist()) {
2051 if (!Ref->getType()->isScalarType())
2052 continue;
2053 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
2054 if (!DRE)
2055 continue;
2056 // Skip BindingDecls - they don't use the same conditional lastprivate
2057 // mechanism.
2058 if (const auto *VD = dyn_cast<VarDecl>(Val: DRE->getDecl())) {
2059 PrivateDecls.insert(V: VD);
2060 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: Ref);
2061 }
2062 }
2063 }
2064 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) {
2065 for (const Expr *Ref : C->varlist()) {
2066 if (!Ref->getType()->isScalarType())
2067 continue;
2068 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
2069 if (!DRE)
2070 continue;
2071 if (const auto *VD = dyn_cast<VarDecl>(Val: DRE->getDecl())) {
2072 PrivateDecls.insert(V: VD);
2073 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: Ref);
2074 }
2075 }
2076 }
2077 // Privates should ne analyzed since they are not captured at all.
2078 // Task reductions may be skipped - tasks are ignored.
2079 // Firstprivates do not return value but may be passed by reference - no need
2080 // to check for updated lastprivate conditional.
2081 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
2082 for (const Expr *Ref : C->varlist()) {
2083 if (!Ref->getType()->isScalarType())
2084 continue;
2085 const auto *DRE = dyn_cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
2086 if (!DRE)
2087 continue;
2088 // Only track VarDecl, not BindingDecl.
2089 if (const auto *VD = dyn_cast<VarDecl>(Val: DRE->getDecl()))
2090 PrivateDecls.insert(V: VD);
2091 }
2092 }
2093 CGF.CGM.getOpenMPRuntime().checkAndEmitSharedLastprivateConditional(
2094 CGF, D: S, IgnoredDecls: PrivateDecls);
2095}
2096
2097static void emitCommonOMPParallelDirective(
2098 CodeGenFunction &CGF, const OMPExecutableDirective &S,
2099 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
2100 const CodeGenBoundParametersTy &CodeGenBoundParameters) {
2101 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: OMPD_parallel);
2102 llvm::Value *NumThreads = nullptr;
2103 OpenMPNumThreadsClauseModifier Modifier = OMPC_NUMTHREADS_unknown;
2104 // OpenMP 6.0, 10.4: "If no severity clause is specified then the effect is as
2105 // if sev-level is fatal."
2106 OpenMPSeverityClauseKind Severity = OMPC_SEVERITY_fatal;
2107 clang::Expr *Message = nullptr;
2108 SourceLocation SeverityLoc = SourceLocation();
2109 SourceLocation MessageLoc = SourceLocation();
2110
2111 llvm::Function *OutlinedFn =
2112 CGF.CGM.getOpenMPRuntime().emitParallelOutlinedFunction(
2113 CGF, D: S, ThreadIDVar: *CS->getCapturedDecl()->param_begin(), InnermostKind,
2114 CodeGen);
2115
2116 if (const auto *NumThreadsClause = S.getSingleClause<OMPNumThreadsClause>()) {
2117 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
2118 NumThreads = CGF.EmitScalarExpr(E: NumThreadsClause->getNumThreads().front(),
2119 /*IgnoreResultAssign=*/true);
2120 Modifier = NumThreadsClause->getPrescriptivenessModifier();
2121 if (const auto *MessageClause = S.getSingleClause<OMPMessageClause>()) {
2122 Message = MessageClause->getMessageString();
2123 MessageLoc = MessageClause->getBeginLoc();
2124 }
2125 if (const auto *SeverityClause = S.getSingleClause<OMPSeverityClause>()) {
2126 Severity = SeverityClause->getSeverityKind();
2127 SeverityLoc = SeverityClause->getBeginLoc();
2128 }
2129 CGF.CGM.getOpenMPRuntime().emitNumThreadsClause(
2130 CGF, NumThreads, Loc: NumThreadsClause->getBeginLoc(), Modifier, Severity,
2131 SeverityLoc, Message, MessageLoc);
2132 }
2133 if (const auto *ProcBindClause = S.getSingleClause<OMPProcBindClause>()) {
2134 CodeGenFunction::RunCleanupsScope ProcBindScope(CGF);
2135 CGF.CGM.getOpenMPRuntime().emitProcBindClause(
2136 CGF, ProcBind: ProcBindClause->getProcBindKind(), Loc: ProcBindClause->getBeginLoc());
2137 }
2138 const Expr *IfCond = nullptr;
2139 for (const auto *C : S.getClausesOfKind<OMPIfClause>()) {
2140 if (C->getNameModifier() == OMPD_unknown ||
2141 C->getNameModifier() == OMPD_parallel) {
2142 IfCond = C->getCondition();
2143 break;
2144 }
2145 }
2146
2147 OMPParallelScope Scope(CGF, S);
2148 llvm::SmallVector<llvm::Value *, 16> CapturedVars;
2149 // Combining 'distribute' with 'for' requires sharing each 'distribute' chunk
2150 // lower and upper bounds with the pragma 'for' chunking mechanism.
2151 // The following lambda takes care of appending the lower and upper bound
2152 // parameters when necessary
2153 CodeGenBoundParameters(CGF, S, CapturedVars);
2154 CGF.GenerateOpenMPCapturedVars(S: *CS, CapturedVars);
2155 CGF.CGM.getOpenMPRuntime().emitParallelCall(CGF, Loc: S.getBeginLoc(), OutlinedFn,
2156 CapturedVars, IfCond, NumThreads,
2157 NumThreadsModifier: Modifier, Severity, Message);
2158}
2159
2160static bool isAllocatableDecl(const VarDecl *VD) {
2161 const VarDecl *CVD = VD->getCanonicalDecl();
2162 if (!CVD->hasAttr<OMPAllocateDeclAttr>())
2163 return false;
2164 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
2165 // Use the default allocation.
2166 return !((AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc ||
2167 AA->getAllocatorType() == OMPAllocateDeclAttr::OMPNullMemAlloc) &&
2168 !AA->getAllocator());
2169}
2170
2171static void emitEmptyBoundParameters(CodeGenFunction &,
2172 const OMPExecutableDirective &,
2173 llvm::SmallVectorImpl<llvm::Value *> &) {}
2174
2175static void emitOMPCopyinClause(CodeGenFunction &CGF,
2176 const OMPExecutableDirective &S) {
2177 bool Copyins = CGF.EmitOMPCopyinClause(D: S);
2178 if (Copyins) {
2179 // Emit implicit barrier to synchronize threads and avoid data races on
2180 // propagation master's thread values of threadprivate variables to local
2181 // instances of that variables of all other implicit threads.
2182 CGF.CGM.getOpenMPRuntime().emitBarrierCall(
2183 CGF, Loc: S.getBeginLoc(), Kind: OMPD_unknown, /*EmitChecks=*/false,
2184 /*ForceSimpleCall=*/true);
2185 }
2186}
2187
2188Address CodeGenFunction::OMPBuilderCBHelpers::getAddressOfLocalVariable(
2189 CodeGenFunction &CGF, const VarDecl *VD) {
2190 CodeGenModule &CGM = CGF.CGM;
2191 auto &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
2192
2193 if (!VD)
2194 return Address::invalid();
2195 const VarDecl *CVD = VD->getCanonicalDecl();
2196 if (!isAllocatableDecl(VD: CVD))
2197 return Address::invalid();
2198 llvm::Value *Size;
2199 CharUnits Align = CGM.getContext().getDeclAlign(D: CVD);
2200 if (CVD->getType()->isVariablyModifiedType()) {
2201 Size = CGF.getTypeSize(Ty: CVD->getType());
2202 // Align the size: ((size + align - 1) / align) * align
2203 Size = CGF.Builder.CreateNUWAdd(
2204 LHS: Size, RHS: CGM.getSize(numChars: Align - CharUnits::fromQuantity(Quantity: 1)));
2205 Size = CGF.Builder.CreateUDiv(LHS: Size, RHS: CGM.getSize(numChars: Align));
2206 Size = CGF.Builder.CreateNUWMul(LHS: Size, RHS: CGM.getSize(numChars: Align));
2207 } else {
2208 CharUnits Sz = CGM.getContext().getTypeSizeInChars(T: CVD->getType());
2209 Size = CGM.getSize(numChars: Sz.alignTo(Align));
2210 }
2211
2212 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
2213 assert(AA->getAllocator() &&
2214 "Expected allocator expression for non-default allocator.");
2215 llvm::Value *Allocator = CGF.EmitScalarExpr(E: AA->getAllocator());
2216 // According to the standard, the original allocator type is a enum (integer).
2217 // Convert to pointer type, if required.
2218 if (Allocator->getType()->isIntegerTy())
2219 Allocator = CGF.Builder.CreateIntToPtr(V: Allocator, DestTy: CGM.VoidPtrTy);
2220 else if (Allocator->getType()->isPointerTy())
2221 Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(V: Allocator,
2222 DestTy: CGM.VoidPtrTy);
2223
2224 llvm::Value *Addr = OMPBuilder.createOMPAlloc(
2225 Loc: CGF.Builder, Size, Allocator,
2226 Name: getNameWithSeparators(Parts: {CVD->getName(), ".void.addr"}, FirstSeparator: ".", Separator: "."));
2227 llvm::CallInst *FreeCI =
2228 OMPBuilder.createOMPFree(Loc: CGF.Builder, Addr, Allocator);
2229
2230 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(Kind: NormalAndEHCleanup, A: FreeCI);
2231 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2232 V: Addr,
2233 DestTy: CGF.ConvertTypeForMem(T: CGM.getContext().getPointerType(T: CVD->getType())),
2234 Name: getNameWithSeparators(Parts: {CVD->getName(), ".addr"}, FirstSeparator: ".", Separator: "."));
2235 return Address(Addr, CGF.ConvertTypeForMem(T: CVD->getType()), Align);
2236}
2237
2238Address CodeGenFunction::OMPBuilderCBHelpers::getAddrOfThreadPrivate(
2239 CodeGenFunction &CGF, const VarDecl *VD, Address VDAddr,
2240 SourceLocation Loc) {
2241 CodeGenModule &CGM = CGF.CGM;
2242 if (CGM.getLangOpts().OpenMPUseTLS &&
2243 CGM.getContext().getTargetInfo().isTLSSupported())
2244 return VDAddr;
2245
2246 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
2247
2248 llvm::Type *VarTy = VDAddr.getElementType();
2249 llvm::Value *Data =
2250 CGF.Builder.CreatePointerCast(V: VDAddr.emitRawPointer(CGF), DestTy: CGM.Int8PtrTy);
2251 llvm::ConstantInt *Size = CGM.getSize(numChars: CGM.GetTargetTypeStoreSize(Ty: VarTy));
2252 std::string Suffix = getNameWithSeparators(Parts: {"cache", ""});
2253 llvm::Twine CacheName = Twine(CGM.getMangledName(GD: VD)).concat(Suffix);
2254
2255 llvm::CallInst *ThreadPrivateCacheCall =
2256 OMPBuilder.createCachedThreadPrivate(Loc: CGF.Builder, Pointer: Data, Size, Name: CacheName);
2257
2258 return Address(ThreadPrivateCacheCall, CGM.Int8Ty, VDAddr.getAlignment());
2259}
2260
2261std::string CodeGenFunction::OMPBuilderCBHelpers::getNameWithSeparators(
2262 ArrayRef<StringRef> Parts, StringRef FirstSeparator, StringRef Separator) {
2263 SmallString<128> Buffer;
2264 llvm::raw_svector_ostream OS(Buffer);
2265 StringRef Sep = FirstSeparator;
2266 for (StringRef Part : Parts) {
2267 OS << Sep << Part;
2268 Sep = Separator;
2269 }
2270 return OS.str().str();
2271}
2272
2273void CodeGenFunction::OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(
2274 CodeGenFunction &CGF, const Stmt *RegionBodyStmt, InsertPointTy AllocaIP,
2275 InsertPointTy CodeGenIP, Twine RegionName) {
2276 CGBuilderTy &Builder = CGF.Builder;
2277 Builder.restoreIP(IP: CodeGenIP);
2278 llvm::BasicBlock *FiniBB = splitBBWithSuffix(Builder, /*CreateBranch=*/false,
2279 Suffix: "." + RegionName + ".after");
2280
2281 {
2282 OMPBuilderCBHelpers::InlinedRegionBodyRAII IRB(CGF, AllocaIP, *FiniBB);
2283 CGF.EmitStmt(S: RegionBodyStmt);
2284 }
2285
2286 if (Builder.saveIP().isValid())
2287 Builder.CreateBr(Dest: FiniBB);
2288}
2289
2290void CodeGenFunction::OMPBuilderCBHelpers::EmitOMPOutlinedRegionBody(
2291 CodeGenFunction &CGF, const Stmt *RegionBodyStmt, InsertPointTy AllocaIP,
2292 InsertPointTy CodeGenIP, Twine RegionName) {
2293 CGBuilderTy &Builder = CGF.Builder;
2294 Builder.restoreIP(IP: CodeGenIP);
2295 llvm::BasicBlock *FiniBB = splitBBWithSuffix(Builder, /*CreateBranch=*/false,
2296 Suffix: "." + RegionName + ".after");
2297
2298 {
2299 OMPBuilderCBHelpers::OutlinedRegionBodyRAII IRB(CGF, AllocaIP, *FiniBB);
2300 CGF.EmitStmt(S: RegionBodyStmt);
2301 }
2302
2303 if (Builder.saveIP().isValid())
2304 Builder.CreateBr(Dest: FiniBB);
2305}
2306
2307void CodeGenFunction::EmitOMPParallelDirective(const OMPParallelDirective &S) {
2308 if (CGM.getLangOpts().OpenMPIRBuilder) {
2309 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
2310 // Check if we have any if clause associated with the directive.
2311 llvm::Value *IfCond = nullptr;
2312 if (const auto *C = S.getSingleClause<OMPIfClause>())
2313 IfCond = EmitScalarExpr(E: C->getCondition(),
2314 /*IgnoreResultAssign=*/true);
2315
2316 llvm::Value *NumThreads = nullptr;
2317 if (const auto *NumThreadsClause = S.getSingleClause<OMPNumThreadsClause>())
2318 NumThreads = EmitScalarExpr(E: NumThreadsClause->getNumThreads().front(),
2319 /*IgnoreResultAssign=*/true);
2320
2321 ProcBindKind ProcBind = OMP_PROC_BIND_default;
2322 if (const auto *ProcBindClause = S.getSingleClause<OMPProcBindClause>())
2323 ProcBind = ProcBindClause->getProcBindKind();
2324
2325 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
2326
2327 // The cleanup callback that finalizes all variables at the given location,
2328 // thus calls destructors etc.
2329 auto FiniCB = [this](InsertPointTy IP) {
2330 OMPBuilderCBHelpers::FinalizeOMPRegion(CGF&: *this, IP);
2331 return llvm::Error::success();
2332 };
2333
2334 // Privatization callback that performs appropriate action for
2335 // shared/private/firstprivate/lastprivate/copyin/... variables.
2336 //
2337 // TODO: This defaults to shared right now.
2338 auto PrivCB = [](InsertPointTy AllocaIP, InsertPointTy CodeGenIP,
2339 llvm::Value &, llvm::Value &Val, llvm::Value *&ReplVal) {
2340 // The next line is appropriate only for variables (Val) with the
2341 // data-sharing attribute "shared".
2342 ReplVal = &Val;
2343
2344 return CodeGenIP;
2345 };
2346
2347 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: OMPD_parallel);
2348 const Stmt *ParallelRegionBodyStmt = CS->getCapturedStmt();
2349
2350 auto BodyGenCB = [&, this](InsertPointTy AllocIP, InsertPointTy CodeGenIP,
2351 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
2352 OMPBuilderCBHelpers::EmitOMPOutlinedRegionBody(
2353 CGF&: *this, RegionBodyStmt: ParallelRegionBodyStmt, AllocaIP: AllocIP, CodeGenIP, RegionName: "parallel");
2354 return llvm::Error::success();
2355 };
2356
2357 CGCapturedStmtInfo CGSI(*CS, CR_OpenMP);
2358 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(*this, &CGSI);
2359 llvm::OpenMPIRBuilder::InsertPointTy AllocaIP(
2360 AllocaInsertPt->getIterator());
2361 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
2362 cantFail(ValOrErr: OMPBuilder.createParallel(
2363 Loc: Builder, AllocaIP, /*DeallocBlocks=*/{}, BodyGenCB, PrivCB, FiniCB,
2364 IfCondition: IfCond, NumThreads, ProcBind, IsCancellable: S.hasCancel()));
2365 Builder.restoreIP(IP: AfterIP);
2366 return;
2367 }
2368
2369 // Emit parallel region as a standalone region.
2370 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
2371 Action.Enter(CGF);
2372 OMPPrivateScope PrivateScope(CGF);
2373 emitOMPCopyinClause(CGF, S);
2374 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
2375 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
2376 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
2377 (void)PrivateScope.Privatize();
2378 CGF.EmitStmt(S: S.getCapturedStmt(RegionKind: OMPD_parallel)->getCapturedStmt());
2379 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_parallel);
2380 };
2381 {
2382 auto LPCRegion =
2383 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
2384 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_parallel, CodeGen,
2385 CodeGenBoundParameters: emitEmptyBoundParameters);
2386 emitPostUpdateForReductionClause(CGF&: *this, D: S,
2387 CondGen: [](CodeGenFunction &) { return nullptr; });
2388 }
2389 // Check for outer lastprivate conditional update.
2390 checkForLastprivateConditionalUpdate(CGF&: *this, S);
2391}
2392
2393void CodeGenFunction::EmitOMPMetaDirective(const OMPMetaDirective &S) {
2394 EmitStmt(S: S.getIfStmt());
2395}
2396
2397namespace {
2398/// RAII to handle scopes for loop transformation directives.
2399class OMPTransformDirectiveScopeRAII {
2400 OMPLoopScope *Scope = nullptr;
2401 CodeGenFunction::CGCapturedStmtInfo *CGSI = nullptr;
2402 CodeGenFunction::CGCapturedStmtRAII *CapInfoRAII = nullptr;
2403
2404 OMPTransformDirectiveScopeRAII(const OMPTransformDirectiveScopeRAII &) =
2405 delete;
2406 OMPTransformDirectiveScopeRAII &
2407 operator=(const OMPTransformDirectiveScopeRAII &) = delete;
2408
2409public:
2410 OMPTransformDirectiveScopeRAII(CodeGenFunction &CGF, const Stmt *S) {
2411 if (const auto *Dir = dyn_cast<OMPLoopBasedDirective>(Val: S)) {
2412 Scope = new OMPLoopScope(CGF, *Dir);
2413 CGSI = new CodeGenFunction::CGCapturedStmtInfo(CR_OpenMP);
2414 CapInfoRAII = new CodeGenFunction::CGCapturedStmtRAII(CGF, CGSI);
2415 } else if (const auto *Dir =
2416 dyn_cast<OMPCanonicalLoopSequenceTransformationDirective>(
2417 Val: S)) {
2418 // For simplicity we reuse the loop scope similarly to what we do with
2419 // OMPCanonicalLoopNestTransformationDirective do by being a subclass
2420 // of OMPLoopBasedDirective.
2421 Scope = new OMPLoopScope(CGF, *Dir);
2422 CGSI = new CodeGenFunction::CGCapturedStmtInfo(CR_OpenMP);
2423 CapInfoRAII = new CodeGenFunction::CGCapturedStmtRAII(CGF, CGSI);
2424 }
2425 }
2426 ~OMPTransformDirectiveScopeRAII() {
2427 if (!Scope)
2428 return;
2429 delete CapInfoRAII;
2430 delete CGSI;
2431 delete Scope;
2432 }
2433};
2434} // namespace
2435
2436static void emitBody(CodeGenFunction &CGF, const Stmt *S, const Stmt *NextLoop,
2437 int MaxLevel, int Level = 0) {
2438 assert(Level < MaxLevel && "Too deep lookup during loop body codegen.");
2439 const Stmt *SimplifiedS = S->IgnoreContainers();
2440 if (const auto *CS = dyn_cast<CompoundStmt>(Val: SimplifiedS)) {
2441 PrettyStackTraceLoc CrashInfo(
2442 CGF.getContext().getSourceManager(), CS->getLBracLoc(),
2443 "LLVM IR generation of compound statement ('{}')");
2444
2445 // Keep track of the current cleanup stack depth, including debug scopes.
2446 CodeGenFunction::LexicalScope Scope(CGF, S->getSourceRange());
2447 for (const Stmt *CurStmt : CS->body())
2448 emitBody(CGF, S: CurStmt, NextLoop, MaxLevel, Level);
2449 return;
2450 }
2451
2452 // `tryToFindNextInnerLoop` keeps the intra-tile hint wrapper around, so match
2453 // against the loop it annotates. The tile overshoot guard is emitted
2454 // separately via EmitOMPLoopBody's finals-conditions handling.
2455 if (SimplifiedS == OMPLoopBasedDirective::ignoreIntraTileHint(S: NextLoop)) {
2456 if (auto *Dir = dyn_cast<OMPLoopTransformationDirective>(Val: SimplifiedS))
2457 SimplifiedS = Dir->getTransformedStmt();
2458 if (const auto *CanonLoop = dyn_cast<OMPCanonicalLoop>(Val: SimplifiedS))
2459 SimplifiedS = CanonLoop->getLoopStmt();
2460 if (const auto *For = dyn_cast<ForStmt>(Val: SimplifiedS)) {
2461 S = For->getBody();
2462 } else {
2463 assert(isa<CXXForRangeStmt>(SimplifiedS) &&
2464 "Expected canonical for loop or range-based for loop.");
2465 const auto *CXXFor = cast<CXXForRangeStmt>(Val: SimplifiedS);
2466 CGF.EmitStmt(S: CXXFor->getLoopVarStmt());
2467 S = CXXFor->getBody();
2468 }
2469 if (Level + 1 < MaxLevel) {
2470 NextLoop = OMPLoopDirective::tryToFindNextInnerLoop(
2471 CurStmt: S, /*TryImperfectlyNestedLoops=*/true);
2472 emitBody(CGF, S, NextLoop, MaxLevel, Level: Level + 1);
2473 return;
2474 }
2475 }
2476 CGF.EmitStmt(S);
2477}
2478
2479void CodeGenFunction::EmitOMPLoopBody(const OMPLoopDirective &D,
2480 JumpDest LoopExit) {
2481 RunCleanupsScope BodyScope(*this);
2482 // Update counters values on current iteration.
2483 for (const Expr *UE : D.updates())
2484 EmitIgnoredExpr(E: UE);
2485 // Update the linear variables.
2486 // In distribute directives only loop counters may be marked as linear, no
2487 // need to generate the code for them.
2488 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
2489 if (!isOpenMPDistributeDirective(DKind: EKind)) {
2490 for (const auto *C : D.getClausesOfKind<OMPLinearClause>()) {
2491 for (const Expr *UE : C->updates())
2492 EmitIgnoredExpr(E: UE);
2493 }
2494 }
2495
2496 // On a continue in the body, jump to the end.
2497 JumpDest Continue = getJumpDestInCurrentScope(Name: "omp.body.continue");
2498 BreakContinueStack.push_back(Elt: BreakContinue(D, LoopExit, Continue));
2499 for (const Expr *E : D.finals_conditions()) {
2500 if (!E)
2501 continue;
2502 // Check that loop counter in non-rectangular nest fits into the iteration
2503 // space.
2504 llvm::BasicBlock *NextBB = createBasicBlock(name: "omp.body.next");
2505 EmitBranchOnBoolExpr(Cond: E, TrueBlock: NextBB, FalseBlock: Continue.getBlock(),
2506 TrueCount: getProfileCount(S: D.getBody()));
2507 EmitBlock(BB: NextBB);
2508 }
2509
2510 OMPPrivateScope InscanScope(*this);
2511 EmitOMPReductionClauseInit(D, PrivateScope&: InscanScope, /*ForInscan=*/true);
2512 bool IsInscanRegion = InscanScope.Privatize();
2513 if (IsInscanRegion) {
2514 // Need to remember the block before and after scan directive
2515 // to dispatch them correctly depending on the clause used in
2516 // this directive, inclusive or exclusive. For inclusive scan the natural
2517 // order of the blocks is used, for exclusive clause the blocks must be
2518 // executed in reverse order.
2519 OMPBeforeScanBlock = createBasicBlock(name: "omp.before.scan.bb");
2520 OMPAfterScanBlock = createBasicBlock(name: "omp.after.scan.bb");
2521 // No need to allocate inscan exit block, in simd mode it is selected in the
2522 // codegen for the scan directive.
2523 if (EKind != OMPD_simd && !getLangOpts().OpenMPSimd)
2524 OMPScanExitBlock = createBasicBlock(name: "omp.exit.inscan.bb");
2525 OMPScanDispatch = createBasicBlock(name: "omp.inscan.dispatch");
2526 EmitBranch(Block: OMPScanDispatch);
2527 EmitBlock(BB: OMPBeforeScanBlock);
2528 }
2529
2530 // Emit loop variables for C++ range loops.
2531 const Stmt *Body =
2532 D.getInnermostCapturedStmt()->getCapturedStmt()->IgnoreContainers();
2533 // Emit loop body.
2534 emitBody(CGF&: *this, S: Body,
2535 NextLoop: OMPLoopBasedDirective::tryToFindNextInnerLoop(
2536 CurStmt: Body, /*TryImperfectlyNestedLoops=*/true),
2537 MaxLevel: D.getLoopsNumber());
2538
2539 // Jump to the dispatcher at the end of the loop body.
2540 if (IsInscanRegion)
2541 EmitBranch(Block: OMPScanExitBlock);
2542
2543 // The end (updates/cleanups).
2544 EmitBlock(BB: Continue.getBlock());
2545 BreakContinueStack.pop_back();
2546}
2547
2548using EmittedClosureTy = std::pair<llvm::Function *, llvm::Value *>;
2549
2550/// Emit a captured statement and return the function as well as its captured
2551/// closure context.
2552static EmittedClosureTy emitCapturedStmtFunc(CodeGenFunction &ParentCGF,
2553 const CapturedStmt *S) {
2554 LValue CapStruct = ParentCGF.InitCapturedStruct(S: *S);
2555 CodeGenFunction CGF(ParentCGF.CGM, /*suppressNewContext=*/true);
2556 std::unique_ptr<CodeGenFunction::CGCapturedStmtInfo> CSI =
2557 std::make_unique<CodeGenFunction::CGCapturedStmtInfo>(args: *S);
2558 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, CSI.get());
2559 llvm::Function *F = CGF.GenerateCapturedStmtFunction(S: *S);
2560
2561 return {F, CapStruct.getPointer(CGF&: ParentCGF)};
2562}
2563
2564/// Emit a call to a previously captured closure.
2565static llvm::CallInst *
2566emitCapturedStmtCall(CodeGenFunction &ParentCGF, EmittedClosureTy Cap,
2567 llvm::ArrayRef<llvm::Value *> Args) {
2568 // Append the closure context to the argument.
2569 SmallVector<llvm::Value *> EffectiveArgs;
2570 EffectiveArgs.reserve(N: Args.size() + 1);
2571 llvm::append_range(C&: EffectiveArgs, R&: Args);
2572 EffectiveArgs.push_back(Elt: Cap.second);
2573
2574 return ParentCGF.Builder.CreateCall(Callee: Cap.first, Args: EffectiveArgs);
2575}
2576
2577llvm::CanonicalLoopInfo *
2578CodeGenFunction::EmitOMPCollapsedCanonicalLoopNest(const Stmt *S, int Depth) {
2579 assert(Depth == 1 && "Nested loops with OpenMPIRBuilder not yet implemented");
2580
2581 // The caller is processing the loop-associated directive processing the \p
2582 // Depth loops nested in \p S. Put the previous pending loop-associated
2583 // directive to the stack. If the current loop-associated directive is a loop
2584 // transformation directive, it will push its generated loops onto the stack
2585 // such that together with the loops left here they form the combined loop
2586 // nest for the parent loop-associated directive.
2587 int ParentExpectedOMPLoopDepth = ExpectedOMPLoopDepth;
2588 ExpectedOMPLoopDepth = Depth;
2589
2590 EmitStmt(S);
2591 assert(OMPLoopNestStack.size() >= (size_t)Depth && "Found too few loops");
2592
2593 // The last added loop is the outermost one.
2594 llvm::CanonicalLoopInfo *Result = OMPLoopNestStack.back();
2595
2596 // Pop the \p Depth loops requested by the call from that stack and restore
2597 // the previous context.
2598 OMPLoopNestStack.pop_back_n(NumItems: Depth);
2599 ExpectedOMPLoopDepth = ParentExpectedOMPLoopDepth;
2600
2601 return Result;
2602}
2603
2604void CodeGenFunction::EmitOMPCanonicalLoop(const OMPCanonicalLoop *S) {
2605 const Stmt *SyntacticalLoop = S->getLoopStmt();
2606 if (!getLangOpts().OpenMPIRBuilder) {
2607 // Ignore if OpenMPIRBuilder is not enabled.
2608 EmitStmt(S: SyntacticalLoop);
2609 return;
2610 }
2611
2612 LexicalScope ForScope(*this, S->getSourceRange());
2613
2614 // Emit init statements. The Distance/LoopVar funcs may reference variable
2615 // declarations they contain.
2616 const Stmt *BodyStmt;
2617 if (const auto *For = dyn_cast<ForStmt>(Val: SyntacticalLoop)) {
2618 if (const Stmt *InitStmt = For->getInit())
2619 EmitStmt(S: InitStmt);
2620 BodyStmt = For->getBody();
2621 } else if (const auto *RangeFor =
2622 dyn_cast<CXXForRangeStmt>(Val: SyntacticalLoop)) {
2623 if (const DeclStmt *RangeStmt = RangeFor->getRangeStmt())
2624 EmitStmt(S: RangeStmt);
2625 if (const DeclStmt *BeginStmt = RangeFor->getBeginStmt())
2626 EmitStmt(S: BeginStmt);
2627 if (const DeclStmt *EndStmt = RangeFor->getEndStmt())
2628 EmitStmt(S: EndStmt);
2629 if (const DeclStmt *LoopVarStmt = RangeFor->getLoopVarStmt())
2630 EmitStmt(S: LoopVarStmt);
2631 BodyStmt = RangeFor->getBody();
2632 } else
2633 llvm_unreachable("Expected for-stmt or range-based for-stmt");
2634
2635 // Emit closure for later use. By-value captures will be captured here.
2636 const CapturedStmt *DistanceFunc = S->getDistanceFunc();
2637 EmittedClosureTy DistanceClosure = emitCapturedStmtFunc(ParentCGF&: *this, S: DistanceFunc);
2638 const CapturedStmt *LoopVarFunc = S->getLoopVarFunc();
2639 EmittedClosureTy LoopVarClosure = emitCapturedStmtFunc(ParentCGF&: *this, S: LoopVarFunc);
2640
2641 // Call the distance function to get the number of iterations of the loop to
2642 // come.
2643 QualType LogicalTy = DistanceFunc->getCapturedDecl()
2644 ->getParam(i: 0)
2645 ->getType()
2646 .getNonReferenceType();
2647 RawAddress CountAddr = CreateMemTemp(T: LogicalTy, Name: ".count.addr");
2648 emitCapturedStmtCall(ParentCGF&: *this, Cap: DistanceClosure, Args: {CountAddr.getPointer()});
2649 llvm::Value *DistVal = Builder.CreateLoad(Addr: CountAddr, Name: ".count");
2650
2651 // Emit the loop structure.
2652 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
2653 auto BodyGen = [&, this](llvm::OpenMPIRBuilder::InsertPointTy CodeGenIP,
2654 llvm::Value *IndVar) {
2655 Builder.restoreIP(IP: CodeGenIP);
2656
2657 // Emit the loop body: Convert the logical iteration number to the loop
2658 // variable and emit the body.
2659 const DeclRefExpr *LoopVarRef = S->getLoopVarRef();
2660 LValue LCVal = EmitLValue(E: LoopVarRef);
2661 Address LoopVarAddress = LCVal.getAddress();
2662 emitCapturedStmtCall(ParentCGF&: *this, Cap: LoopVarClosure,
2663 Args: {LoopVarAddress.emitRawPointer(CGF&: *this), IndVar});
2664
2665 RunCleanupsScope BodyScope(*this);
2666 EmitStmt(S: BodyStmt);
2667 return llvm::Error::success();
2668 };
2669
2670 llvm::CanonicalLoopInfo *CL =
2671 cantFail(ValOrErr: OMPBuilder.createCanonicalLoop(Loc: Builder, BodyGenCB: BodyGen, TripCount: DistVal));
2672
2673 // Finish up the loop.
2674 Builder.restoreIP(IP: CL->getAfterIP());
2675 ForScope.ForceCleanup();
2676
2677 // Remember the CanonicalLoopInfo for parent AST nodes consuming it.
2678 OMPLoopNestStack.push_back(Elt: CL);
2679}
2680
2681void CodeGenFunction::EmitOMPInnerLoop(
2682 const OMPExecutableDirective &S, bool RequiresCleanup, const Expr *LoopCond,
2683 const Expr *IncExpr,
2684 const llvm::function_ref<void(CodeGenFunction &)> BodyGen,
2685 const llvm::function_ref<void(CodeGenFunction &)> PostIncGen) {
2686 auto LoopExit = getJumpDestInCurrentScope(Name: "omp.inner.for.end");
2687
2688 // Start the loop with a block that tests the condition.
2689 auto CondBlock = createBasicBlock(name: "omp.inner.for.cond");
2690 EmitBlock(BB: CondBlock);
2691 const SourceRange R = S.getSourceRange();
2692
2693 // If attributes are attached, push to the basic block with them.
2694 const auto &OMPED = cast<OMPExecutableDirective>(Val: S);
2695 const CapturedStmt *ICS = OMPED.getInnermostCapturedStmt();
2696 const Stmt *SS = ICS->getCapturedStmt();
2697 const AttributedStmt *AS = dyn_cast_or_null<AttributedStmt>(Val: SS);
2698 OMPLoopNestStack.clear();
2699 if (AS)
2700 LoopStack.push(Header: CondBlock, Ctx&: CGM.getContext(), CGOpts: CGM.getCodeGenOpts(),
2701 Attrs: AS->getAttrs(), StartLoc: SourceLocToDebugLoc(Location: R.getBegin()),
2702 EndLoc: SourceLocToDebugLoc(Location: R.getEnd()));
2703 else
2704 LoopStack.push(Header: CondBlock, StartLoc: SourceLocToDebugLoc(Location: R.getBegin()),
2705 EndLoc: SourceLocToDebugLoc(Location: R.getEnd()));
2706
2707 // If there are any cleanups between here and the loop-exit scope,
2708 // create a block to stage a loop exit along.
2709 llvm::BasicBlock *ExitBlock = LoopExit.getBlock();
2710 if (RequiresCleanup)
2711 ExitBlock = createBasicBlock(name: "omp.inner.for.cond.cleanup");
2712
2713 llvm::BasicBlock *LoopBody = createBasicBlock(name: "omp.inner.for.body");
2714
2715 // Emit condition.
2716 EmitBranchOnBoolExpr(Cond: LoopCond, TrueBlock: LoopBody, FalseBlock: ExitBlock, TrueCount: getProfileCount(S: &S));
2717 if (ExitBlock != LoopExit.getBlock()) {
2718 EmitBlock(BB: ExitBlock);
2719 EmitBranchThroughCleanup(Dest: LoopExit);
2720 }
2721
2722 EmitBlock(BB: LoopBody);
2723 incrementProfileCounter(S: &S);
2724
2725 // Create a block for the increment.
2726 JumpDest Continue = getJumpDestInCurrentScope(Name: "omp.inner.for.inc");
2727 BreakContinueStack.push_back(Elt: BreakContinue(S, LoopExit, Continue));
2728
2729 BodyGen(*this);
2730
2731 // Emit "IV = IV + 1" and a back-edge to the condition block.
2732 EmitBlock(BB: Continue.getBlock());
2733 EmitIgnoredExpr(E: IncExpr);
2734 PostIncGen(*this);
2735 BreakContinueStack.pop_back();
2736 EmitBranch(Block: CondBlock);
2737 LoopStack.pop();
2738 // Emit the fall-through block.
2739 EmitBlock(BB: LoopExit.getBlock());
2740}
2741
2742bool CodeGenFunction::EmitOMPLinearClauseInit(const OMPLoopDirective &D) {
2743 if (!HaveInsertPoint())
2744 return false;
2745 // Emit inits for the linear variables.
2746 bool HasLinears = false;
2747 for (const auto *C : D.getClausesOfKind<OMPLinearClause>()) {
2748 for (const Expr *Init : C->inits()) {
2749 HasLinears = true;
2750 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: Init)->getDecl());
2751 if (const auto *Ref =
2752 dyn_cast<DeclRefExpr>(Val: VD->getInit()->IgnoreImpCasts())) {
2753 if (isa<BindingDecl>(Val: Ref->getDecl())) {
2754 EmitVarDecl(D: *VD);
2755 } else {
2756 AutoVarEmission Emission = EmitAutoVarAlloca(var: *VD);
2757 const auto *OrigVD = cast<VarDecl>(Val: Ref->getDecl());
2758 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(OrigVD),
2759 CapturedStmtInfo->lookup(VD: OrigVD) != nullptr,
2760 VD->getInit()->getType(), VK_LValue,
2761 VD->getInit()->getExprLoc());
2762 EmitExprAsInit(
2763 init: &DRE, D: VD,
2764 lvalue: MakeAddrLValue(Addr: Emission.getAllocatedAddress(), T: VD->getType()),
2765 /*capturedByInit=*/false);
2766 EmitAutoVarCleanups(emission: Emission);
2767 }
2768 } else {
2769 EmitVarDecl(D: *VD);
2770 }
2771 }
2772 // Emit the linear steps for the linear clauses.
2773 // If a step is not constant, it is pre-calculated before the loop.
2774 if (const auto *CS = cast_or_null<BinaryOperator>(Val: C->getCalcStep()))
2775 if (const auto *SaveRef = cast<DeclRefExpr>(Val: CS->getLHS())) {
2776 EmitVarDecl(D: *cast<VarDecl>(Val: SaveRef->getDecl()));
2777 // Emit calculation of the linear step.
2778 EmitIgnoredExpr(E: CS);
2779 }
2780 }
2781 return HasLinears;
2782}
2783
2784void CodeGenFunction::EmitOMPLinearClauseFinal(
2785 const OMPLoopDirective &D,
2786 const llvm::function_ref<llvm::Value *(CodeGenFunction &)> CondGen) {
2787 if (!HaveInsertPoint())
2788 return;
2789 llvm::BasicBlock *DoneBB = nullptr;
2790 // Emit the final values of the linear variables.
2791 for (const auto *C : D.getClausesOfKind<OMPLinearClause>()) {
2792 auto IC = C->varlist_begin();
2793 for (const Expr *F : C->finals()) {
2794 if (!DoneBB) {
2795 if (llvm::Value *Cond = CondGen(*this)) {
2796 // If the first post-update expression is found, emit conditional
2797 // block if it was requested.
2798 llvm::BasicBlock *ThenBB = createBasicBlock(name: ".omp.linear.pu");
2799 DoneBB = createBasicBlock(name: ".omp.linear.pu.done");
2800 Builder.CreateCondBr(Cond, True: ThenBB, False: DoneBB);
2801 EmitBlock(BB: ThenBB);
2802 }
2803 }
2804 const auto *OrigDecl = cast<DeclRefExpr>(Val: *IC)->getDecl();
2805 Address OrigAddr = Address::invalid();
2806 if (const auto *BD = dyn_cast<BindingDecl>(Val: OrigDecl)) {
2807 OrigAddr = EmitOMPBindingOriginalAddr(BD, Loc: (*IC)->getExprLoc());
2808 } else {
2809 const auto *OrigVD = cast<VarDecl>(Val: OrigDecl);
2810 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(OrigVD),
2811 CapturedStmtInfo->lookup(VD: OrigVD) != nullptr,
2812 (*IC)->getType(), VK_LValue, (*IC)->getExprLoc());
2813 OrigAddr = EmitLValue(E: &DRE).getAddress();
2814 }
2815 CodeGenFunction::OMPPrivateScope VarScope(*this);
2816 VarScope.addPrivate(LocalVD: OrigDecl, Addr: OrigAddr);
2817 (void)VarScope.Privatize();
2818 EmitIgnoredExpr(E: F);
2819 ++IC;
2820 }
2821 if (const Expr *PostUpdate = C->getPostUpdateExpr())
2822 EmitIgnoredExpr(E: PostUpdate);
2823 }
2824 if (DoneBB)
2825 EmitBlock(BB: DoneBB, /*IsFinished=*/true);
2826}
2827
2828static void emitAlignedClause(CodeGenFunction &CGF,
2829 const OMPExecutableDirective &D) {
2830 if (!CGF.HaveInsertPoint())
2831 return;
2832 for (const auto *Clause : D.getClausesOfKind<OMPAlignedClause>()) {
2833 llvm::APInt ClauseAlignment(64, 0);
2834 if (const Expr *AlignmentExpr = Clause->getAlignment()) {
2835 auto *AlignmentCI =
2836 cast<llvm::ConstantInt>(Val: CGF.EmitScalarExpr(E: AlignmentExpr));
2837 ClauseAlignment = AlignmentCI->getValue();
2838 }
2839 for (const Expr *E : Clause->varlist()) {
2840 llvm::APInt Alignment(ClauseAlignment);
2841 if (Alignment == 0) {
2842 // OpenMP [2.8.1, Description]
2843 // If no optional parameter is specified, implementation-defined default
2844 // alignments for SIMD instructions on the target platforms are assumed.
2845 Alignment =
2846 CGF.getContext()
2847 .toCharUnitsFromBits(BitSize: CGF.getContext().getOpenMPDefaultSimdAlign(
2848 T: E->getType()->getPointeeType()))
2849 .getQuantity();
2850 }
2851 assert((Alignment == 0 || Alignment.isPowerOf2()) &&
2852 "alignment is not power of 2");
2853 if (Alignment != 0) {
2854 llvm::Value *PtrValue = CGF.EmitScalarExpr(E);
2855 CGF.emitAlignmentAssumption(
2856 PtrValue, E, /*No second loc needed*/ AssumptionLoc: SourceLocation(),
2857 Alignment: llvm::ConstantInt::get(Context&: CGF.getLLVMContext(), V: Alignment));
2858 }
2859 }
2860 }
2861}
2862
2863void CodeGenFunction::EmitOMPPrivateLoopCounters(
2864 const OMPLoopDirective &S, CodeGenFunction::OMPPrivateScope &LoopScope) {
2865 if (!HaveInsertPoint())
2866 return;
2867 auto I = S.private_counters().begin();
2868 for (const Expr *E : S.counters()) {
2869 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl());
2870 const auto *PrivateVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *I)->getDecl());
2871 // Emit var without initialization.
2872 AutoVarEmission VarEmission = EmitAutoVarAlloca(var: *PrivateVD);
2873 EmitAutoVarCleanups(emission: VarEmission);
2874 LocalDeclMap.erase(Val: PrivateVD);
2875 (void)LoopScope.addPrivate(LocalVD: VD, Addr: VarEmission.getAllocatedAddress());
2876 if (LocalDeclMap.count(Val: VD) || CapturedStmtInfo->lookup(VD) ||
2877 VD->hasGlobalStorage()) {
2878 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(VD),
2879 LocalDeclMap.count(Val: VD) || CapturedStmtInfo->lookup(VD),
2880 E->getType(), VK_LValue, E->getExprLoc());
2881 (void)LoopScope.addPrivate(LocalVD: PrivateVD, Addr: EmitLValue(E: &DRE).getAddress());
2882 } else {
2883 (void)LoopScope.addPrivate(LocalVD: PrivateVD, Addr: VarEmission.getAllocatedAddress());
2884 }
2885 ++I;
2886 }
2887 // Privatize extra loop counters used in loops for ordered(n) clauses.
2888 for (const auto *C : S.getClausesOfKind<OMPOrderedClause>()) {
2889 if (!C->getNumForLoops())
2890 continue;
2891 for (unsigned I = S.getLoopsNumber(), E = C->getLoopNumIterations().size();
2892 I < E; ++I) {
2893 const auto *DRE = cast<DeclRefExpr>(Val: C->getLoopCounter(NumLoop: I));
2894 const auto *VD = cast<VarDecl>(Val: DRE->getDecl());
2895 // Override only those variables that can be captured to avoid re-emission
2896 // of the variables declared within the loops.
2897 if (DRE->refersToEnclosingVariableOrCapture()) {
2898 (void)LoopScope.addPrivate(
2899 LocalVD: VD, Addr: CreateMemTemp(T: DRE->getType(), Name: VD->getName()));
2900 }
2901 }
2902 }
2903}
2904
2905static void emitPreCond(CodeGenFunction &CGF, const OMPLoopDirective &S,
2906 const Expr *Cond, llvm::BasicBlock *TrueBlock,
2907 llvm::BasicBlock *FalseBlock, uint64_t TrueCount) {
2908 if (!CGF.HaveInsertPoint())
2909 return;
2910 {
2911 CodeGenFunction::OMPPrivateScope PreCondScope(CGF);
2912 CGF.EmitOMPPrivateLoopCounters(S, LoopScope&: PreCondScope);
2913 (void)PreCondScope.Privatize();
2914 // Get initial values of real counters.
2915 for (const Expr *I : S.inits()) {
2916 CGF.EmitIgnoredExpr(E: I);
2917 }
2918 }
2919 // Create temp loop control variables with their init values to support
2920 // non-rectangular loops.
2921 CodeGenFunction::OMPMapVars PreCondVars;
2922 for (const Expr *E : S.dependent_counters()) {
2923 if (!E)
2924 continue;
2925 assert(!E->getType().getNonReferenceType()->isRecordType() &&
2926 "dependent counter must not be an iterator.");
2927 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl());
2928 Address CounterAddr =
2929 CGF.CreateMemTemp(T: VD->getType().getNonReferenceType());
2930 (void)PreCondVars.setVarAddr(CGF, LocalVD: VD, TempAddr: CounterAddr);
2931 }
2932 (void)PreCondVars.apply(CGF);
2933 for (const Expr *E : S.dependent_inits()) {
2934 if (!E)
2935 continue;
2936 CGF.EmitIgnoredExpr(E);
2937 }
2938 // Check that loop is executed at least one time.
2939 CGF.EmitBranchOnBoolExpr(Cond, TrueBlock, FalseBlock, TrueCount);
2940 PreCondVars.restore(CGF);
2941}
2942
2943void CodeGenFunction::EmitOMPLinearClause(
2944 const OMPLoopDirective &D, CodeGenFunction::OMPPrivateScope &PrivateScope) {
2945 if (!HaveInsertPoint())
2946 return;
2947 llvm::DenseSet<const VarDecl *> SIMDLCVs;
2948 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
2949 if (isOpenMPSimdDirective(DKind: EKind)) {
2950 const auto *LoopDirective = cast<OMPLoopDirective>(Val: &D);
2951 for (const Expr *C : LoopDirective->counters()) {
2952 SIMDLCVs.insert(
2953 V: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: C)->getDecl())->getCanonicalDecl());
2954 }
2955 }
2956 for (const auto *C : D.getClausesOfKind<OMPLinearClause>()) {
2957 auto CurPrivate = C->privates().begin();
2958 for (const Expr *E : C->varlist()) {
2959 const auto *VD = cast<DeclRefExpr>(Val: E)->getDecl();
2960 const auto *PrivateVD =
2961 cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *CurPrivate)->getDecl());
2962 bool IsSIMDLCV = false;
2963 if (const auto *VarD = dyn_cast<VarDecl>(Val: VD))
2964 IsSIMDLCV = SIMDLCVs.count(V: VarD->getCanonicalDecl());
2965 if (!IsSIMDLCV) {
2966 // Emit private VarDecl with copy init.
2967 EmitVarDecl(D: *PrivateVD);
2968 Address PrivateAddr = GetAddrOfLocalVar(VD: PrivateVD);
2969 bool IsRegistered = PrivateScope.addPrivate(LocalVD: VD, Addr: PrivateAddr);
2970 assert(IsRegistered && "linear var already registered as private");
2971 // Silence the warning about unused variable.
2972 (void)IsRegistered;
2973 } else {
2974 EmitVarDecl(D: *PrivateVD);
2975 }
2976 ++CurPrivate;
2977 }
2978 }
2979}
2980
2981static void emitSimdlenSafelenClause(CodeGenFunction &CGF,
2982 const OMPExecutableDirective &D) {
2983 if (!CGF.HaveInsertPoint())
2984 return;
2985 if (const auto *C = D.getSingleClause<OMPSimdlenClause>()) {
2986 RValue Len = CGF.EmitAnyExpr(E: C->getSimdlen(), aggSlot: AggValueSlot::ignored(),
2987 /*ignoreResult=*/true);
2988 auto *Val = cast<llvm::ConstantInt>(Val: Len.getScalarVal());
2989 CGF.LoopStack.setVectorizeWidth(Val->getZExtValue());
2990 // In presence of finite 'safelen', it may be unsafe to mark all
2991 // the memory instructions parallel, because loop-carried
2992 // dependences of 'safelen' iterations are possible.
2993 CGF.LoopStack.setParallel(!D.getSingleClause<OMPSafelenClause>());
2994 } else if (const auto *C = D.getSingleClause<OMPSafelenClause>()) {
2995 RValue Len = CGF.EmitAnyExpr(E: C->getSafelen(), aggSlot: AggValueSlot::ignored(),
2996 /*ignoreResult=*/true);
2997 auto *Val = cast<llvm::ConstantInt>(Val: Len.getScalarVal());
2998 CGF.LoopStack.setVectorizeWidth(Val->getZExtValue());
2999 // In presence of finite 'safelen', it may be unsafe to mark all
3000 // the memory instructions parallel, because loop-carried
3001 // dependences of 'safelen' iterations are possible.
3002 CGF.LoopStack.setParallel(/*Enable=*/false);
3003 }
3004}
3005
3006// Check for the presence of an `OMPOrderedBlockAssocDirective`,
3007// i.e., `ordered` in `#pragma omp ordered simd`.
3008//
3009// Consider the following source code:
3010// ```
3011// __attribute__((noinline)) void omp_simd_loop(float X[ARRAY_SIZE][ARRAY_SIZE])
3012// {
3013// for (int r = 1; r < ARRAY_SIZE; ++r) {
3014// for (int c = 1; c < ARRAY_SIZE; ++c) {
3015// #pragma omp simd
3016// for (int k = 2; k < ARRAY_SIZE; ++k) {
3017// #pragma omp ordered simd
3018// X[r][k] = X[r][k - 2] + sinf((float)(r / c));
3019// }
3020// }
3021// }
3022// }
3023// ```
3024//
3025// Suppose we are in `CodeGenFunction::EmitOMPSimdInit(const OMPLoopDirective
3026// &D)`. By examining `D.dump()` we have the following AST containing
3027// `OMPOrderedBlockAssocDirective`:
3028//
3029// ```
3030// OMPSimdDirective 0x1c32950
3031// `-CapturedStmt 0x1c32028
3032// |-CapturedDecl 0x1c310e8
3033// | |-ForStmt 0x1c31e30
3034// | | |-DeclStmt 0x1c31298
3035// | | | `-VarDecl 0x1c31208 used k 'int' cinit
3036// | | | `-IntegerLiteral 0x1c31278 'int' 2
3037// | | |-<<<NULL>>>
3038// | | |-BinaryOperator 0x1c31308 'int' '<'
3039// | | | |-ImplicitCastExpr 0x1c312f0 'int' <LValueToRValue>
3040// | | | | `-DeclRefExpr 0x1c312b0 'int' lvalue Var 0x1c31208 'k' 'int'
3041// | | | `-IntegerLiteral 0x1c312d0 'int' 256
3042// | | |-UnaryOperator 0x1c31348 'int' prefix '++'
3043// | | | `-DeclRefExpr 0x1c31328 'int' lvalue Var 0x1c31208 'k' 'int'
3044// | | `-CompoundStmt 0x1c31e18
3045// | | `-OMPOrderedBlockAssocDirective 0x1c31dd8
3046// | | |-OMPSimdClause 0x1c31380
3047// | | `-CapturedStmt 0x1c31cd0
3048// ```
3049//
3050// Note the presence of `OMPOrderedBlockAssocDirective` above:
3051// It's (transitively) nested in a `CapturedStmt` representing the pragma
3052// annotated compound statement. Thus, we need to consider this nesting and
3053// include checking the `getCapturedStmt` in this case.
3054static bool hasOrderedBlockAssocDirective(const Stmt *S) {
3055 if (isa<OMPOrderedBlockAssocDirective>(Val: S))
3056 return true;
3057
3058 if (const auto *CS = dyn_cast<CapturedStmt>(Val: S))
3059 return hasOrderedBlockAssocDirective(S: CS->getCapturedStmt());
3060
3061 for (const Stmt *Child : S->children()) {
3062 if (Child && hasOrderedBlockAssocDirective(S: Child))
3063 return true;
3064 }
3065
3066 return false;
3067}
3068
3069static void applyConservativeSimdOrderedDirective(const Stmt &AssociatedStmt,
3070 LoopInfoStack &LoopStack) {
3071 // Check for the presence of an `OMPOrderedBlockAssocDirective`
3072 // i.e., `ordered` in `#pragma omp ordered simd`
3073 bool HasOrderedDirective = hasOrderedBlockAssocDirective(S: &AssociatedStmt);
3074 // If present then conservatively disable loop vectorization
3075 // analogously to how `emitSimdlenSafelenClause` does.
3076 if (HasOrderedDirective)
3077 LoopStack.setParallel(/*Enable=*/false);
3078}
3079
3080void CodeGenFunction::EmitOMPSimdInit(const OMPLoopDirective &D) {
3081 // Walk clauses and process safelen/lastprivate.
3082 LoopStack.setParallel(/*Enable=*/true);
3083 LoopStack.setVectorizeEnable();
3084 const Stmt *AssociatedStmt = D.getAssociatedStmt();
3085 applyConservativeSimdOrderedDirective(AssociatedStmt: *AssociatedStmt, LoopStack);
3086 emitSimdlenSafelenClause(CGF&: *this, D);
3087 if (const auto *C = D.getSingleClause<OMPOrderClause>())
3088 if (C->getKind() == OMPC_ORDER_concurrent)
3089 LoopStack.setParallel(/*Enable=*/true);
3090 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S: D);
3091 if ((EKind == OMPD_simd ||
3092 (getLangOpts().OpenMPSimd && isOpenMPSimdDirective(DKind: EKind))) &&
3093 llvm::any_of(Range: D.getClausesOfKind<OMPReductionClause>(),
3094 P: [](const OMPReductionClause *C) {
3095 return C->getModifier() == OMPC_REDUCTION_inscan;
3096 }))
3097 // Disable parallel access in case of prefix sum.
3098 LoopStack.setParallel(/*Enable=*/false);
3099}
3100
3101void CodeGenFunction::EmitOMPSimdFinal(
3102 const OMPLoopDirective &D,
3103 const llvm::function_ref<llvm::Value *(CodeGenFunction &)> CondGen) {
3104 if (!HaveInsertPoint())
3105 return;
3106 llvm::BasicBlock *DoneBB = nullptr;
3107 auto IC = D.counters().begin();
3108 auto IPC = D.private_counters().begin();
3109 for (const Expr *F : D.finals()) {
3110 const auto *OrigVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: (*IC))->getDecl());
3111 const auto *PrivateVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: (*IPC))->getDecl());
3112 const auto *CED = dyn_cast<OMPCapturedExprDecl>(Val: OrigVD);
3113 if (LocalDeclMap.count(Val: OrigVD) || CapturedStmtInfo->lookup(VD: OrigVD) ||
3114 OrigVD->hasGlobalStorage() || CED) {
3115 if (!DoneBB) {
3116 if (llvm::Value *Cond = CondGen(*this)) {
3117 // If the first post-update expression is found, emit conditional
3118 // block if it was requested.
3119 llvm::BasicBlock *ThenBB = createBasicBlock(name: ".omp.final.then");
3120 DoneBB = createBasicBlock(name: ".omp.final.done");
3121 Builder.CreateCondBr(Cond, True: ThenBB, False: DoneBB);
3122 EmitBlock(BB: ThenBB);
3123 }
3124 }
3125 Address OrigAddr = Address::invalid();
3126 if (CED) {
3127 OrigAddr = EmitLValue(E: CED->getInit()->IgnoreImpCasts()).getAddress();
3128 } else {
3129 DeclRefExpr DRE(getContext(), const_cast<VarDecl *>(PrivateVD),
3130 /*RefersToEnclosingVariableOrCapture=*/false,
3131 (*IPC)->getType(), VK_LValue, (*IPC)->getExprLoc());
3132 OrigAddr = EmitLValue(E: &DRE).getAddress();
3133 }
3134 OMPPrivateScope VarScope(*this);
3135 VarScope.addPrivate(LocalVD: OrigVD, Addr: OrigAddr);
3136 (void)VarScope.Privatize();
3137 EmitIgnoredExpr(E: F);
3138 }
3139 ++IC;
3140 ++IPC;
3141 }
3142 if (DoneBB)
3143 EmitBlock(BB: DoneBB, /*IsFinished=*/true);
3144}
3145
3146static void emitOMPLoopBodyWithStopPoint(CodeGenFunction &CGF,
3147 const OMPLoopDirective &S,
3148 CodeGenFunction::JumpDest LoopExit) {
3149 CGF.EmitOMPLoopBody(D: S, LoopExit);
3150 CGF.EmitStopPoint(S: &S);
3151}
3152
3153/// Emit a helper variable and return corresponding lvalue.
3154static LValue EmitOMPHelperVar(CodeGenFunction &CGF,
3155 const DeclRefExpr *Helper) {
3156 auto VDecl = cast<VarDecl>(Val: Helper->getDecl());
3157 CGF.EmitVarDecl(D: *VDecl);
3158 return CGF.EmitLValue(E: Helper);
3159}
3160
3161static void emitCommonSimdLoop(CodeGenFunction &CGF, const OMPLoopDirective &S,
3162 const RegionCodeGenTy &SimdInitGen,
3163 const RegionCodeGenTy &BodyCodeGen) {
3164 auto &&ThenGen = [&S, &SimdInitGen, &BodyCodeGen](CodeGenFunction &CGF,
3165 PrePostActionTy &) {
3166 CGOpenMPRuntime::NontemporalDeclsRAII NontemporalsRegion(CGF.CGM, S);
3167 CodeGenFunction::OMPLocalDeclMapRAII Scope(CGF);
3168 SimdInitGen(CGF);
3169
3170 BodyCodeGen(CGF);
3171 };
3172 auto &&ElseGen = [&BodyCodeGen](CodeGenFunction &CGF, PrePostActionTy &) {
3173 CodeGenFunction::OMPLocalDeclMapRAII Scope(CGF);
3174 CGF.LoopStack.setVectorizeEnable(/*Enable=*/false);
3175
3176 BodyCodeGen(CGF);
3177 };
3178 const Expr *IfCond = nullptr;
3179 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
3180 if (isOpenMPSimdDirective(DKind: EKind)) {
3181 for (const auto *C : S.getClausesOfKind<OMPIfClause>()) {
3182 if (CGF.getLangOpts().OpenMP >= 50 &&
3183 (C->getNameModifier() == OMPD_unknown ||
3184 C->getNameModifier() == OMPD_simd)) {
3185 IfCond = C->getCondition();
3186 break;
3187 }
3188 }
3189 }
3190 if (IfCond) {
3191 CGF.CGM.getOpenMPRuntime().emitIfClause(CGF, Cond: IfCond, ThenGen, ElseGen);
3192 } else {
3193 RegionCodeGenTy ThenRCG(ThenGen);
3194 ThenRCG(CGF);
3195 }
3196}
3197
3198static void emitOMPSimdRegion(CodeGenFunction &CGF, const OMPLoopDirective &S,
3199 PrePostActionTy &Action) {
3200 Action.Enter(CGF);
3201 OMPLoopScope PreInitScope(CGF, S);
3202 // if (PreCond) {
3203 // for (IV in 0..LastIteration) BODY;
3204 // <Final counter/linear vars updates>;
3205 // }
3206
3207 // The presence of lower/upper bound variable depends on the actual directive
3208 // kind in the AST node. The variables must be emitted because some of the
3209 // expressions associated with the loop will use them.
3210 OpenMPDirectiveKind DKind = S.getDirectiveKind();
3211 if (isOpenMPDistributeDirective(DKind) ||
3212 isOpenMPWorksharingDirective(DKind) || isOpenMPTaskLoopDirective(DKind) ||
3213 isOpenMPGenericLoopDirective(DKind)) {
3214 (void)EmitOMPHelperVar(CGF, Helper: cast<DeclRefExpr>(Val: S.getLowerBoundVariable()));
3215 (void)EmitOMPHelperVar(CGF, Helper: cast<DeclRefExpr>(Val: S.getUpperBoundVariable()));
3216 }
3217
3218 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
3219 // Emit: if (PreCond) - begin.
3220 // If the condition constant folds and can be elided, avoid emitting the
3221 // whole loop.
3222 bool CondConstant;
3223 llvm::BasicBlock *ContBlock = nullptr;
3224 if (CGF.ConstantFoldsToSimpleInteger(Cond: S.getPreCond(), Result&: CondConstant)) {
3225 if (!CondConstant)
3226 return;
3227 } else {
3228 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock(name: "simd.if.then");
3229 ContBlock = CGF.createBasicBlock(name: "simd.if.end");
3230 emitPreCond(CGF, S, Cond: S.getPreCond(), TrueBlock: ThenBlock, FalseBlock: ContBlock,
3231 TrueCount: CGF.getProfileCount(S: &S));
3232 CGF.EmitBlock(BB: ThenBlock);
3233 CGF.incrementProfileCounter(S: &S);
3234 }
3235
3236 // Emit the loop iteration variable.
3237 const Expr *IVExpr = S.getIterationVariable();
3238 const auto *IVDecl = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: IVExpr)->getDecl());
3239 CGF.EmitVarDecl(D: *IVDecl);
3240 CGF.EmitIgnoredExpr(E: S.getInit());
3241
3242 // Emit the iterations count variable.
3243 // If it is not a variable, Sema decided to calculate iterations count on
3244 // each iteration (e.g., it is foldable into a constant).
3245 if (const auto *LIExpr = dyn_cast<DeclRefExpr>(Val: S.getLastIteration())) {
3246 CGF.EmitVarDecl(D: *cast<VarDecl>(Val: LIExpr->getDecl()));
3247 // Emit calculation of the iterations count.
3248 CGF.EmitIgnoredExpr(E: S.getCalcLastIteration());
3249 }
3250
3251 emitAlignedClause(CGF, D: S);
3252 (void)CGF.EmitOMPLinearClauseInit(D: S);
3253 {
3254 CodeGenFunction::OMPPrivateScope LoopScope(CGF);
3255 CGF.EmitOMPPrivateClause(D: S, PrivateScope&: LoopScope);
3256 CGF.EmitOMPPrivateLoopCounters(S, LoopScope);
3257 CGF.EmitOMPLinearClause(D: S, PrivateScope&: LoopScope);
3258 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope&: LoopScope);
3259 CGOpenMPRuntime::LastprivateConditionalRAII LPCRegion(
3260 CGF, S, CGF.EmitLValue(E: S.getIterationVariable()));
3261 bool HasLastprivateClause = CGF.EmitOMPLastprivateClauseInit(D: S, PrivateScope&: LoopScope);
3262 (void)LoopScope.Privatize();
3263 if (isOpenMPTargetExecutionDirective(DKind: EKind))
3264 CGF.CGM.getOpenMPRuntime().adjustTargetSpecificDataForLambdas(CGF, D: S);
3265
3266 emitCommonSimdLoop(
3267 CGF, S,
3268 SimdInitGen: [&S](CodeGenFunction &CGF, PrePostActionTy &) {
3269 CGF.EmitOMPSimdInit(D: S);
3270 },
3271 BodyCodeGen: [&S, &LoopScope](CodeGenFunction &CGF, PrePostActionTy &) {
3272 CGF.EmitOMPInnerLoop(
3273 S, RequiresCleanup: LoopScope.requiresCleanups(), LoopCond: S.getCond(), IncExpr: S.getInc(),
3274 BodyGen: [&S](CodeGenFunction &CGF) {
3275 emitOMPLoopBodyWithStopPoint(CGF, S,
3276 LoopExit: CodeGenFunction::JumpDest());
3277 },
3278 PostIncGen: [](CodeGenFunction &) {});
3279 });
3280 CGF.EmitOMPSimdFinal(D: S, CondGen: [](CodeGenFunction &) { return nullptr; });
3281 // Emit final copy of the lastprivate variables at the end of loops.
3282 if (HasLastprivateClause)
3283 CGF.EmitOMPLastprivateClauseFinal(D: S, /*NoFinals=*/true);
3284 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_simd);
3285 emitPostUpdateForReductionClause(CGF, D: S,
3286 CondGen: [](CodeGenFunction &) { return nullptr; });
3287 LoopScope.restoreMap();
3288 CGF.EmitOMPLinearClauseFinal(D: S, CondGen: [](CodeGenFunction &) { return nullptr; });
3289 }
3290 // Emit: if (PreCond) - end.
3291 if (ContBlock) {
3292 CGF.EmitBranch(Block: ContBlock);
3293 CGF.EmitBlock(BB: ContBlock, IsFinished: true);
3294 }
3295}
3296
3297// Pass OMPLoopDirective (instead of OMPSimdDirective) to make this function
3298// available for "loop bind(thread)", which maps to "simd".
3299static bool isSimdSupportedByOpenMPIRBuilder(const OMPLoopDirective &S) {
3300 // Check for unsupported clauses
3301 for (OMPClause *C : S.clauses()) {
3302 // Currently only order, simdlen and safelen clauses are supported
3303 if (!(isa<OMPSimdlenClause>(Val: C) || isa<OMPSafelenClause>(Val: C) ||
3304 isa<OMPOrderClause>(Val: C) || isa<OMPAlignedClause>(Val: C)))
3305 return false;
3306 }
3307
3308 // Check if we have a statement with the ordered-blockassoc directive.
3309 // Visit the statement hierarchy to find a compound statement
3310 // with a ordered-blockassoc directive in it.
3311 if (const auto *CanonLoop = dyn_cast<OMPCanonicalLoop>(Val: S.getRawStmt())) {
3312 if (const Stmt *SyntacticalLoop = CanonLoop->getLoopStmt()) {
3313 for (const Stmt *SubStmt : SyntacticalLoop->children()) {
3314 if (!SubStmt)
3315 continue;
3316 if (const CompoundStmt *CS = dyn_cast<CompoundStmt>(Val: SubStmt)) {
3317 for (const Stmt *CSSubStmt : CS->children()) {
3318 if (!CSSubStmt)
3319 continue;
3320 if (isa<OMPOrderedBlockAssocDirective>(Val: CSSubStmt)) {
3321 return false;
3322 }
3323 }
3324 }
3325 }
3326 }
3327 }
3328 return true;
3329}
3330
3331static llvm::MapVector<llvm::Value *, llvm::Value *>
3332GetAlignedMapping(const OMPLoopDirective &S, CodeGenFunction &CGF) {
3333 llvm::MapVector<llvm::Value *, llvm::Value *> AlignedVars;
3334 for (const auto *Clause : S.getClausesOfKind<OMPAlignedClause>()) {
3335 llvm::APInt ClauseAlignment(64, 0);
3336 if (const Expr *AlignmentExpr = Clause->getAlignment()) {
3337 auto *AlignmentCI =
3338 cast<llvm::ConstantInt>(Val: CGF.EmitScalarExpr(E: AlignmentExpr));
3339 ClauseAlignment = AlignmentCI->getValue();
3340 }
3341 for (const Expr *E : Clause->varlist()) {
3342 llvm::APInt Alignment(ClauseAlignment);
3343 if (Alignment == 0) {
3344 // OpenMP [2.8.1, Description]
3345 // If no optional parameter is specified, implementation-defined default
3346 // alignments for SIMD instructions on the target platforms are assumed.
3347 Alignment =
3348 CGF.getContext()
3349 .toCharUnitsFromBits(BitSize: CGF.getContext().getOpenMPDefaultSimdAlign(
3350 T: E->getType()->getPointeeType()))
3351 .getQuantity();
3352 }
3353 assert((Alignment == 0 || Alignment.isPowerOf2()) &&
3354 "alignment is not power of 2");
3355 llvm::Value *PtrValue = CGF.EmitScalarExpr(E);
3356 AlignedVars[PtrValue] = CGF.Builder.getInt64(C: Alignment.getSExtValue());
3357 }
3358 }
3359 return AlignedVars;
3360}
3361
3362// Pass OMPLoopDirective (instead of OMPSimdDirective) to make this function
3363// available for "loop bind(thread)", which maps to "simd".
3364static void emitOMPSimdDirective(const OMPLoopDirective &S,
3365 CodeGenFunction &CGF, CodeGenModule &CGM) {
3366 bool UseOMPIRBuilder =
3367 CGM.getLangOpts().OpenMPIRBuilder && isSimdSupportedByOpenMPIRBuilder(S);
3368 if (UseOMPIRBuilder) {
3369 auto &&CodeGenIRBuilder = [&S, &CGM, UseOMPIRBuilder](CodeGenFunction &CGF,
3370 PrePostActionTy &) {
3371 // Use the OpenMPIRBuilder if enabled.
3372 if (UseOMPIRBuilder) {
3373 llvm::MapVector<llvm::Value *, llvm::Value *> AlignedVars =
3374 GetAlignedMapping(S, CGF);
3375 // Emit the associated statement and get its loop representation.
3376 const Stmt *Inner = S.getRawStmt();
3377 llvm::CanonicalLoopInfo *CLI =
3378 CGF.EmitOMPCollapsedCanonicalLoopNest(S: Inner, Depth: 1);
3379
3380 llvm::OpenMPIRBuilder &OMPBuilder =
3381 CGM.getOpenMPRuntime().getOMPBuilder();
3382 // Add SIMD specific metadata
3383 llvm::ConstantInt *Simdlen = nullptr;
3384 if (const auto *C = S.getSingleClause<OMPSimdlenClause>()) {
3385 RValue Len = CGF.EmitAnyExpr(E: C->getSimdlen(), aggSlot: AggValueSlot::ignored(),
3386 /*ignoreResult=*/true);
3387 auto *Val = cast<llvm::ConstantInt>(Val: Len.getScalarVal());
3388 Simdlen = Val;
3389 }
3390 llvm::ConstantInt *Safelen = nullptr;
3391 if (const auto *C = S.getSingleClause<OMPSafelenClause>()) {
3392 RValue Len = CGF.EmitAnyExpr(E: C->getSafelen(), aggSlot: AggValueSlot::ignored(),
3393 /*ignoreResult=*/true);
3394 auto *Val = cast<llvm::ConstantInt>(Val: Len.getScalarVal());
3395 Safelen = Val;
3396 }
3397 llvm::omp::OrderKind Order = llvm::omp::OrderKind::OMP_ORDER_unknown;
3398 if (const auto *C = S.getSingleClause<OMPOrderClause>()) {
3399 if (C->getKind() == OpenMPOrderClauseKind::OMPC_ORDER_concurrent) {
3400 Order = llvm::omp::OrderKind::OMP_ORDER_concurrent;
3401 }
3402 }
3403 // Add simd metadata to the collapsed loop. Do not generate
3404 // another loop for if clause. Support for if clause is done earlier.
3405 OMPBuilder.applySimd(Loop: CLI, AlignedVars,
3406 /*IfCond*/ nullptr, Order, Simdlen, Safelen);
3407 return;
3408 }
3409 };
3410 {
3411 auto LPCRegion =
3412 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF, S);
3413 OMPLexicalScope Scope(CGF, S, OMPD_unknown);
3414 CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_simd,
3415 CodeGen: CodeGenIRBuilder);
3416 }
3417 return;
3418 }
3419
3420 CodeGenFunction::ParentLoopDirectiveForScanRegion ScanRegion(CGF, S);
3421 CGF.OMPFirstScanLoop = true;
3422 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
3423 emitOMPSimdRegion(CGF, S, Action);
3424 };
3425 {
3426 auto LPCRegion =
3427 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF, S);
3428 OMPLexicalScope Scope(CGF, S, OMPD_unknown);
3429 CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_simd, CodeGen);
3430 }
3431 // Check for outer lastprivate conditional update.
3432 checkForLastprivateConditionalUpdate(CGF, S);
3433}
3434
3435void CodeGenFunction::EmitOMPSimdDirective(const OMPSimdDirective &S) {
3436 emitOMPSimdDirective(S, CGF&: *this, CGM);
3437}
3438
3439void CodeGenFunction::EmitOMPTileDirective(const OMPTileDirective &S) {
3440 // Emit the de-sugared statement.
3441 OMPTransformDirectiveScopeRAII TileScope(*this, &S);
3442 EmitStmt(S: S.getTransformedStmt());
3443}
3444
3445void CodeGenFunction::EmitOMPStripeDirective(const OMPStripeDirective &S) {
3446 // Emit the de-sugared statement.
3447 OMPTransformDirectiveScopeRAII StripeScope(*this, &S);
3448 EmitStmt(S: S.getTransformedStmt());
3449}
3450
3451void CodeGenFunction::EmitOMPReverseDirective(const OMPReverseDirective &S) {
3452 // Emit the de-sugared statement.
3453 OMPTransformDirectiveScopeRAII ReverseScope(*this, &S);
3454 EmitStmt(S: S.getTransformedStmt());
3455}
3456
3457void CodeGenFunction::EmitOMPSplitDirective(const OMPSplitDirective &S) {
3458 // Emit the de-sugared statement (the split loops).
3459 OMPTransformDirectiveScopeRAII SplitScope(*this, &S);
3460 EmitStmt(S: S.getTransformedStmt());
3461}
3462
3463void CodeGenFunction::EmitOMPInterchangeDirective(
3464 const OMPInterchangeDirective &S) {
3465 // Emit the de-sugared statement.
3466 OMPTransformDirectiveScopeRAII InterchangeScope(*this, &S);
3467 EmitStmt(S: S.getTransformedStmt());
3468}
3469
3470void CodeGenFunction::EmitOMPFlattenDirective(const OMPFlattenDirective &S) {
3471 // Emit the de-sugared statement.
3472 OMPTransformDirectiveScopeRAII FlattenScope(*this, &S);
3473 EmitStmt(S: S.getTransformedStmt());
3474 EmitStmt(S: S.getFinals());
3475}
3476
3477void CodeGenFunction::EmitOMPFuseDirective(const OMPFuseDirective &S) {
3478 // Emit the de-sugared statement
3479 OMPTransformDirectiveScopeRAII FuseScope(*this, &S);
3480 EmitStmt(S: S.getTransformedStmt());
3481}
3482
3483void CodeGenFunction::EmitOMPUnrollDirective(const OMPUnrollDirective &S) {
3484 bool UseOMPIRBuilder = CGM.getLangOpts().OpenMPIRBuilder;
3485
3486 if (UseOMPIRBuilder) {
3487 auto DL = SourceLocToDebugLoc(Location: S.getBeginLoc());
3488 const Stmt *Inner = S.getRawStmt();
3489
3490 // Consume nested loop. Clear the entire remaining loop stack because a
3491 // fully unrolled loop is non-transformable. For partial unrolling the
3492 // generated outer loop is pushed back to the stack.
3493 llvm::CanonicalLoopInfo *CLI = EmitOMPCollapsedCanonicalLoopNest(S: Inner, Depth: 1);
3494 OMPLoopNestStack.clear();
3495
3496 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
3497
3498 bool NeedsUnrolledCLI = ExpectedOMPLoopDepth >= 1;
3499 llvm::CanonicalLoopInfo *UnrolledCLI = nullptr;
3500
3501 if (S.hasClausesOfKind<OMPFullClause>()) {
3502 assert(ExpectedOMPLoopDepth == 0);
3503 OMPBuilder.unrollLoopFull(DL, Loop: CLI);
3504 } else if (auto *PartialClause = S.getSingleClause<OMPPartialClause>()) {
3505 uint64_t Factor = 0;
3506 if (Expr *FactorExpr = PartialClause->getFactor()) {
3507 Factor =
3508 FactorExpr->EvaluateKnownConstInt(Ctx: getContext()).getLimitedValue();
3509 assert(Factor >= 1 && "Only positive factors are valid");
3510 }
3511 OMPBuilder.unrollLoopPartial(DL, Loop: CLI, Factor,
3512 UnrolledCLI: NeedsUnrolledCLI ? &UnrolledCLI : nullptr);
3513 } else {
3514 OMPBuilder.unrollLoopHeuristic(DL, Loop: CLI);
3515 }
3516
3517 assert((!NeedsUnrolledCLI || UnrolledCLI) &&
3518 "NeedsUnrolledCLI implies UnrolledCLI to be set");
3519 if (UnrolledCLI)
3520 OMPLoopNestStack.push_back(Elt: UnrolledCLI);
3521
3522 return;
3523 }
3524
3525 // This function is only called if the unrolled loop is not consumed by any
3526 // other loop-associated construct. Such a loop-associated construct will have
3527 // used the transformed AST.
3528
3529 // Set the unroll metadata for the next emitted loop.
3530 LoopStack.setUnrollState(LoopAttributes::Enable);
3531
3532 if (S.hasClausesOfKind<OMPFullClause>()) {
3533 LoopStack.setUnrollState(LoopAttributes::Full);
3534 } else if (auto *PartialClause = S.getSingleClause<OMPPartialClause>()) {
3535 if (Expr *FactorExpr = PartialClause->getFactor()) {
3536 uint64_t Factor =
3537 FactorExpr->EvaluateKnownConstInt(Ctx: getContext()).getLimitedValue();
3538 assert(Factor >= 1 && "Only positive factors are valid");
3539 LoopStack.setUnrollCount(Factor);
3540 }
3541 }
3542
3543 EmitStmt(S: S.getAssociatedStmt());
3544}
3545
3546void CodeGenFunction::EmitOMPOuterLoop(
3547 bool DynamicOrOrdered, bool IsMonotonic, const OMPLoopDirective &S,
3548 CodeGenFunction::OMPPrivateScope &LoopScope,
3549 const CodeGenFunction::OMPLoopArguments &LoopArgs,
3550 const CodeGenFunction::CodeGenLoopTy &CodeGenLoop,
3551 const CodeGenFunction::CodeGenOrderedTy &CodeGenOrdered) {
3552 CGOpenMPRuntime &RT = CGM.getOpenMPRuntime();
3553
3554 const Expr *IVExpr = S.getIterationVariable();
3555 const unsigned IVSize = getContext().getTypeSize(T: IVExpr->getType());
3556 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation();
3557
3558 JumpDest LoopExit = getJumpDestInCurrentScope(Name: "omp.dispatch.end");
3559
3560 // Start the loop with a block that tests the condition.
3561 llvm::BasicBlock *CondBlock = createBasicBlock(name: "omp.dispatch.cond");
3562 EmitBlock(BB: CondBlock);
3563 const SourceRange R = S.getSourceRange();
3564 OMPLoopNestStack.clear();
3565 LoopStack.push(Header: CondBlock, StartLoc: SourceLocToDebugLoc(Location: R.getBegin()),
3566 EndLoc: SourceLocToDebugLoc(Location: R.getEnd()));
3567
3568 llvm::Value *BoolCondVal = nullptr;
3569 if (!DynamicOrOrdered) {
3570 // UB = min(UB, GlobalUB) or
3571 // UB = min(UB, PrevUB) for combined loop sharing constructs (e.g.
3572 // 'distribute parallel for')
3573 EmitIgnoredExpr(E: LoopArgs.EUB);
3574 // IV = LB
3575 EmitIgnoredExpr(E: LoopArgs.Init);
3576 // IV < UB
3577 BoolCondVal = EvaluateExprAsBool(E: LoopArgs.Cond);
3578 } else {
3579 BoolCondVal =
3580 RT.emitForNext(CGF&: *this, Loc: S.getBeginLoc(), IVSize, IVSigned, IL: LoopArgs.IL,
3581 LB: LoopArgs.LB, UB: LoopArgs.UB, ST: LoopArgs.ST);
3582 }
3583
3584 // If there are any cleanups between here and the loop-exit scope,
3585 // create a block to stage a loop exit along.
3586 llvm::BasicBlock *ExitBlock = LoopExit.getBlock();
3587 if (LoopScope.requiresCleanups())
3588 ExitBlock = createBasicBlock(name: "omp.dispatch.cleanup");
3589
3590 llvm::BasicBlock *LoopBody = createBasicBlock(name: "omp.dispatch.body");
3591 Builder.CreateCondBr(Cond: BoolCondVal, True: LoopBody, False: ExitBlock);
3592 if (ExitBlock != LoopExit.getBlock()) {
3593 EmitBlock(BB: ExitBlock);
3594 EmitBranchThroughCleanup(Dest: LoopExit);
3595 }
3596 EmitBlock(BB: LoopBody);
3597
3598 // Emit "IV = LB" (in case of static schedule, we have already calculated new
3599 // LB for loop condition and emitted it above).
3600 if (DynamicOrOrdered)
3601 EmitIgnoredExpr(E: LoopArgs.Init);
3602
3603 // Create a block for the increment.
3604 JumpDest Continue = getJumpDestInCurrentScope(Name: "omp.dispatch.inc");
3605 BreakContinueStack.push_back(Elt: BreakContinue(S, LoopExit, Continue));
3606
3607 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
3608 emitCommonSimdLoop(
3609 CGF&: *this, S,
3610 SimdInitGen: [&S, IsMonotonic, EKind](CodeGenFunction &CGF, PrePostActionTy &) {
3611 // Generate !llvm.loop.parallel metadata for loads and stores for loops
3612 // with dynamic/guided scheduling and without ordered clause.
3613 if (!isOpenMPSimdDirective(DKind: EKind)) {
3614 CGF.LoopStack.setParallel(!IsMonotonic);
3615 if (const auto *C = S.getSingleClause<OMPOrderClause>())
3616 if (C->getKind() == OMPC_ORDER_concurrent)
3617 CGF.LoopStack.setParallel(/*Enable=*/true);
3618 } else {
3619 CGF.EmitOMPSimdInit(D: S);
3620 }
3621 },
3622 BodyCodeGen: [&S, &LoopArgs, LoopExit, &CodeGenLoop, IVSize, IVSigned, &CodeGenOrdered,
3623 &LoopScope](CodeGenFunction &CGF, PrePostActionTy &) {
3624 SourceLocation Loc = S.getBeginLoc();
3625 // when 'distribute' is not combined with a 'for':
3626 // while (idx <= UB) { BODY; ++idx; }
3627 // when 'distribute' is combined with a 'for'
3628 // (e.g. 'distribute parallel for')
3629 // while (idx <= UB) { <CodeGen rest of pragma>; idx += ST; }
3630 CGF.EmitOMPInnerLoop(
3631 S, RequiresCleanup: LoopScope.requiresCleanups(), LoopCond: LoopArgs.Cond, IncExpr: LoopArgs.IncExpr,
3632 BodyGen: [&S, LoopExit, &CodeGenLoop](CodeGenFunction &CGF) {
3633 CodeGenLoop(CGF, S, LoopExit);
3634 },
3635 PostIncGen: [IVSize, IVSigned, Loc, &CodeGenOrdered](CodeGenFunction &CGF) {
3636 CodeGenOrdered(CGF, Loc, IVSize, IVSigned);
3637 });
3638 });
3639
3640 EmitBlock(BB: Continue.getBlock());
3641 BreakContinueStack.pop_back();
3642 if (!DynamicOrOrdered) {
3643 // Emit "LB = LB + Stride", "UB = UB + Stride".
3644 EmitIgnoredExpr(E: LoopArgs.NextLB);
3645 EmitIgnoredExpr(E: LoopArgs.NextUB);
3646 }
3647
3648 EmitBranch(Block: CondBlock);
3649 OMPLoopNestStack.clear();
3650 LoopStack.pop();
3651 // Emit the fall-through block.
3652 EmitBlock(BB: LoopExit.getBlock());
3653
3654 // Tell the runtime we are done.
3655 auto &&CodeGen = [DynamicOrOrdered, &S, &LoopArgs](CodeGenFunction &CGF) {
3656 if (!DynamicOrOrdered)
3657 CGF.CGM.getOpenMPRuntime().emitForStaticFinish(CGF, Loc: S.getEndLoc(),
3658 DKind: LoopArgs.DKind);
3659 };
3660 OMPCancelStack.emitExit(CGF&: *this, Kind: EKind, CodeGen);
3661}
3662
3663void CodeGenFunction::EmitOMPForOuterLoop(
3664 const OpenMPScheduleTy &ScheduleKind, bool IsMonotonic,
3665 const OMPLoopDirective &S, OMPPrivateScope &LoopScope, bool Ordered,
3666 const OMPLoopArguments &LoopArgs,
3667 const CodeGenDispatchBoundsTy &CGDispatchBounds) {
3668 CGOpenMPRuntime &RT = CGM.getOpenMPRuntime();
3669
3670 // Dynamic scheduling of the outer loop (dynamic, guided, auto, runtime).
3671 const bool DynamicOrOrdered = Ordered || RT.isDynamic(ScheduleKind: ScheduleKind.Schedule);
3672
3673 assert((Ordered || !RT.isStaticNonchunked(ScheduleKind.Schedule,
3674 LoopArgs.Chunk != nullptr)) &&
3675 "static non-chunked schedule does not need outer loop");
3676
3677 // Emit outer loop.
3678 //
3679 // OpenMP [2.7.1, Loop Construct, Description, table 2-1]
3680 // When schedule(dynamic,chunk_size) is specified, the iterations are
3681 // distributed to threads in the team in chunks as the threads request them.
3682 // Each thread executes a chunk of iterations, then requests another chunk,
3683 // until no chunks remain to be distributed. Each chunk contains chunk_size
3684 // iterations, except for the last chunk to be distributed, which may have
3685 // fewer iterations. When no chunk_size is specified, it defaults to 1.
3686 //
3687 // When schedule(guided,chunk_size) is specified, the iterations are assigned
3688 // to threads in the team in chunks as the executing threads request them.
3689 // Each thread executes a chunk of iterations, then requests another chunk,
3690 // until no chunks remain to be assigned. For a chunk_size of 1, the size of
3691 // each chunk is proportional to the number of unassigned iterations divided
3692 // by the number of threads in the team, decreasing to 1. For a chunk_size
3693 // with value k (greater than 1), the size of each chunk is determined in the
3694 // same way, with the restriction that the chunks do not contain fewer than k
3695 // iterations (except for the last chunk to be assigned, which may have fewer
3696 // than k iterations).
3697 //
3698 // When schedule(auto) is specified, the decision regarding scheduling is
3699 // delegated to the compiler and/or runtime system. The programmer gives the
3700 // implementation the freedom to choose any possible mapping of iterations to
3701 // threads in the team.
3702 //
3703 // When schedule(runtime) is specified, the decision regarding scheduling is
3704 // deferred until run time, and the schedule and chunk size are taken from the
3705 // run-sched-var ICV. If the ICV is set to auto, the schedule is
3706 // implementation defined
3707 //
3708 // __kmpc_dispatch_init();
3709 // while(__kmpc_dispatch_next(&LB, &UB)) {
3710 // idx = LB;
3711 // while (idx <= UB) { BODY; ++idx;
3712 // __kmpc_dispatch_fini_(4|8)[u](); // For ordered loops only.
3713 // } // inner loop
3714 // }
3715 // __kmpc_dispatch_deinit();
3716 //
3717 // OpenMP [2.7.1, Loop Construct, Description, table 2-1]
3718 // When schedule(static, chunk_size) is specified, iterations are divided into
3719 // chunks of size chunk_size, and the chunks are assigned to the threads in
3720 // the team in a round-robin fashion in the order of the thread number.
3721 //
3722 // while(UB = min(UB, GlobalUB), idx = LB, idx < UB) {
3723 // while (idx <= UB) { BODY; ++idx; } // inner loop
3724 // LB = LB + ST;
3725 // UB = UB + ST;
3726 // }
3727 //
3728
3729 const Expr *IVExpr = S.getIterationVariable();
3730 const unsigned IVSize = getContext().getTypeSize(T: IVExpr->getType());
3731 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation();
3732
3733 if (DynamicOrOrdered) {
3734 const std::pair<llvm::Value *, llvm::Value *> DispatchBounds =
3735 CGDispatchBounds(*this, S, LoopArgs.LB, LoopArgs.UB);
3736 llvm::Value *LBVal = DispatchBounds.first;
3737 llvm::Value *UBVal = DispatchBounds.second;
3738 CGOpenMPRuntime::DispatchRTInput DipatchRTInputValues = {LBVal, UBVal,
3739 LoopArgs.Chunk};
3740 RT.emitForDispatchInit(CGF&: *this, Loc: S.getBeginLoc(), ScheduleKind, IVSize,
3741 IVSigned, Ordered, DispatchValues: DipatchRTInputValues);
3742 } else {
3743 CGOpenMPRuntime::StaticRTInput StaticInit(
3744 IVSize, IVSigned, Ordered, LoopArgs.IL, LoopArgs.LB, LoopArgs.UB,
3745 LoopArgs.ST, LoopArgs.Chunk);
3746 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
3747 RT.emitForStaticInit(CGF&: *this, Loc: S.getBeginLoc(), DKind: EKind, ScheduleKind,
3748 Values: StaticInit);
3749 }
3750
3751 auto &&CodeGenOrdered = [Ordered](CodeGenFunction &CGF, SourceLocation Loc,
3752 const unsigned IVSize,
3753 const bool IVSigned) {
3754 if (Ordered) {
3755 CGF.CGM.getOpenMPRuntime().emitForOrderedIterationEnd(CGF, Loc, IVSize,
3756 IVSigned);
3757 }
3758 };
3759
3760 OMPLoopArguments OuterLoopArgs(LoopArgs.LB, LoopArgs.UB, LoopArgs.ST,
3761 LoopArgs.IL, LoopArgs.Chunk, LoopArgs.EUB);
3762 OuterLoopArgs.IncExpr = S.getInc();
3763 OuterLoopArgs.Init = S.getInit();
3764 OuterLoopArgs.Cond = S.getCond();
3765 OuterLoopArgs.NextLB = S.getNextLowerBound();
3766 OuterLoopArgs.NextUB = S.getNextUpperBound();
3767 OuterLoopArgs.DKind = LoopArgs.DKind;
3768 EmitOMPOuterLoop(DynamicOrOrdered, IsMonotonic, S, LoopScope, LoopArgs: OuterLoopArgs,
3769 CodeGenLoop: emitOMPLoopBodyWithStopPoint, CodeGenOrdered);
3770 if (DynamicOrOrdered) {
3771 RT.emitForDispatchDeinit(CGF&: *this, Loc: S.getBeginLoc());
3772 }
3773}
3774
3775static void emitEmptyOrdered(CodeGenFunction &, SourceLocation Loc,
3776 const unsigned IVSize, const bool IVSigned) {}
3777
3778void CodeGenFunction::EmitOMPDistributeOuterLoop(
3779 OpenMPDistScheduleClauseKind ScheduleKind, const OMPLoopDirective &S,
3780 OMPPrivateScope &LoopScope, const OMPLoopArguments &LoopArgs,
3781 const CodeGenLoopTy &CodeGenLoopContent) {
3782
3783 CGOpenMPRuntime &RT = CGM.getOpenMPRuntime();
3784
3785 // Emit outer loop.
3786 // Same behavior as a OMPForOuterLoop, except that schedule cannot be
3787 // dynamic
3788 //
3789
3790 const Expr *IVExpr = S.getIterationVariable();
3791 const unsigned IVSize = getContext().getTypeSize(T: IVExpr->getType());
3792 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation();
3793 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
3794
3795 CGOpenMPRuntime::StaticRTInput StaticInit(
3796 IVSize, IVSigned, /* Ordered = */ false, LoopArgs.IL, LoopArgs.LB,
3797 LoopArgs.UB, LoopArgs.ST, LoopArgs.Chunk);
3798 RT.emitDistributeStaticInit(CGF&: *this, Loc: S.getBeginLoc(), SchedKind: ScheduleKind, Values: StaticInit);
3799
3800 // for combined 'distribute' and 'for' the increment expression of distribute
3801 // is stored in DistInc. For 'distribute' alone, it is in Inc.
3802 Expr *IncExpr;
3803 if (isOpenMPLoopBoundSharingDirective(Kind: EKind))
3804 IncExpr = S.getDistInc();
3805 else
3806 IncExpr = S.getInc();
3807
3808 // this routine is shared by 'omp distribute parallel for' and
3809 // 'omp distribute': select the right EUB expression depending on the
3810 // directive
3811 OMPLoopArguments OuterLoopArgs;
3812 OuterLoopArgs.LB = LoopArgs.LB;
3813 OuterLoopArgs.UB = LoopArgs.UB;
3814 OuterLoopArgs.ST = LoopArgs.ST;
3815 OuterLoopArgs.IL = LoopArgs.IL;
3816 OuterLoopArgs.Chunk = LoopArgs.Chunk;
3817 OuterLoopArgs.EUB = isOpenMPLoopBoundSharingDirective(Kind: EKind)
3818 ? S.getCombinedEnsureUpperBound()
3819 : S.getEnsureUpperBound();
3820 OuterLoopArgs.IncExpr = IncExpr;
3821 OuterLoopArgs.Init = isOpenMPLoopBoundSharingDirective(Kind: EKind)
3822 ? S.getCombinedInit()
3823 : S.getInit();
3824 OuterLoopArgs.Cond = isOpenMPLoopBoundSharingDirective(Kind: EKind)
3825 ? S.getCombinedCond()
3826 : S.getCond();
3827 OuterLoopArgs.NextLB = isOpenMPLoopBoundSharingDirective(Kind: EKind)
3828 ? S.getCombinedNextLowerBound()
3829 : S.getNextLowerBound();
3830 OuterLoopArgs.NextUB = isOpenMPLoopBoundSharingDirective(Kind: EKind)
3831 ? S.getCombinedNextUpperBound()
3832 : S.getNextUpperBound();
3833 OuterLoopArgs.DKind = OMPD_distribute;
3834
3835 EmitOMPOuterLoop(/* DynamicOrOrdered = */ false, /* IsMonotonic = */ false, S,
3836 LoopScope, LoopArgs: OuterLoopArgs, CodeGenLoop: CodeGenLoopContent,
3837 CodeGenOrdered: emitEmptyOrdered);
3838}
3839
3840static std::pair<LValue, LValue>
3841emitDistributeParallelForInnerBounds(CodeGenFunction &CGF,
3842 const OMPExecutableDirective &S) {
3843 const OMPLoopDirective &LS = cast<OMPLoopDirective>(Val: S);
3844 LValue LB =
3845 EmitOMPHelperVar(CGF, Helper: cast<DeclRefExpr>(Val: LS.getLowerBoundVariable()));
3846 LValue UB =
3847 EmitOMPHelperVar(CGF, Helper: cast<DeclRefExpr>(Val: LS.getUpperBoundVariable()));
3848
3849 // When composing 'distribute' with 'for' (e.g. as in 'distribute
3850 // parallel for') we need to use the 'distribute'
3851 // chunk lower and upper bounds rather than the whole loop iteration
3852 // space. These are parameters to the outlined function for 'parallel'
3853 // and we copy the bounds of the previous schedule into the
3854 // the current ones.
3855 LValue PrevLB = CGF.EmitLValue(E: LS.getPrevLowerBoundVariable());
3856 LValue PrevUB = CGF.EmitLValue(E: LS.getPrevUpperBoundVariable());
3857 llvm::Value *PrevLBVal = CGF.EmitLoadOfScalar(
3858 lvalue: PrevLB, Loc: LS.getPrevLowerBoundVariable()->getExprLoc());
3859 PrevLBVal = CGF.EmitScalarConversion(
3860 Src: PrevLBVal, SrcTy: LS.getPrevLowerBoundVariable()->getType(),
3861 DstTy: LS.getIterationVariable()->getType(),
3862 Loc: LS.getPrevLowerBoundVariable()->getExprLoc());
3863 llvm::Value *PrevUBVal = CGF.EmitLoadOfScalar(
3864 lvalue: PrevUB, Loc: LS.getPrevUpperBoundVariable()->getExprLoc());
3865 PrevUBVal = CGF.EmitScalarConversion(
3866 Src: PrevUBVal, SrcTy: LS.getPrevUpperBoundVariable()->getType(),
3867 DstTy: LS.getIterationVariable()->getType(),
3868 Loc: LS.getPrevUpperBoundVariable()->getExprLoc());
3869
3870 CGF.EmitStoreOfScalar(value: PrevLBVal, lvalue: LB);
3871 CGF.EmitStoreOfScalar(value: PrevUBVal, lvalue: UB);
3872
3873 return {LB, UB};
3874}
3875
3876/// if the 'for' loop has a dispatch schedule (e.g. dynamic, guided) then
3877/// we need to use the LB and UB expressions generated by the worksharing
3878/// code generation support, whereas in non combined situations we would
3879/// just emit 0 and the LastIteration expression
3880/// This function is necessary due to the difference of the LB and UB
3881/// types for the RT emission routines for 'for_static_init' and
3882/// 'for_dispatch_init'
3883static std::pair<llvm::Value *, llvm::Value *>
3884emitDistributeParallelForDispatchBounds(CodeGenFunction &CGF,
3885 const OMPExecutableDirective &S,
3886 Address LB, Address UB) {
3887 const OMPLoopDirective &LS = cast<OMPLoopDirective>(Val: S);
3888 const Expr *IVExpr = LS.getIterationVariable();
3889 // when implementing a dynamic schedule for a 'for' combined with a
3890 // 'distribute' (e.g. 'distribute parallel for'), the 'for' loop
3891 // is not normalized as each team only executes its own assigned
3892 // distribute chunk
3893 QualType IteratorTy = IVExpr->getType();
3894 llvm::Value *LBVal =
3895 CGF.EmitLoadOfScalar(Addr: LB, /*Volatile=*/false, Ty: IteratorTy, Loc: S.getBeginLoc());
3896 llvm::Value *UBVal =
3897 CGF.EmitLoadOfScalar(Addr: UB, /*Volatile=*/false, Ty: IteratorTy, Loc: S.getBeginLoc());
3898 return {LBVal, UBVal};
3899}
3900
3901static void emitDistributeParallelForDistributeInnerBoundParams(
3902 CodeGenFunction &CGF, const OMPExecutableDirective &S,
3903 llvm::SmallVectorImpl<llvm::Value *> &CapturedVars) {
3904 const auto &Dir = cast<OMPLoopDirective>(Val: S);
3905 LValue LB =
3906 CGF.EmitLValue(E: cast<DeclRefExpr>(Val: Dir.getCombinedLowerBoundVariable()));
3907 llvm::Value *LBCast = CGF.Builder.CreateIntCast(
3908 V: CGF.Builder.CreateLoad(Addr: LB.getAddress()), DestTy: CGF.SizeTy, /*isSigned=*/false);
3909 CapturedVars.push_back(Elt: LBCast);
3910 LValue UB =
3911 CGF.EmitLValue(E: cast<DeclRefExpr>(Val: Dir.getCombinedUpperBoundVariable()));
3912
3913 llvm::Value *UBCast = CGF.Builder.CreateIntCast(
3914 V: CGF.Builder.CreateLoad(Addr: UB.getAddress()), DestTy: CGF.SizeTy, /*isSigned=*/false);
3915 CapturedVars.push_back(Elt: UBCast);
3916}
3917
3918static void
3919emitInnerParallelForWhenCombined(CodeGenFunction &CGF,
3920 const OMPLoopDirective &S,
3921 CodeGenFunction::JumpDest LoopExit) {
3922 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
3923 auto &&CGInlinedWorksharingLoop = [&S, EKind](CodeGenFunction &CGF,
3924 PrePostActionTy &Action) {
3925 Action.Enter(CGF);
3926 bool HasCancel = false;
3927 if (!isOpenMPSimdDirective(DKind: EKind)) {
3928 if (const auto *D = dyn_cast<OMPTeamsDistributeParallelForDirective>(Val: &S))
3929 HasCancel = D->hasCancel();
3930 else if (const auto *D = dyn_cast<OMPDistributeParallelForDirective>(Val: &S))
3931 HasCancel = D->hasCancel();
3932 else if (const auto *D =
3933 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(Val: &S))
3934 HasCancel = D->hasCancel();
3935 }
3936 CodeGenFunction::OMPCancelStackRAII CancelRegion(CGF, EKind, HasCancel);
3937 CGF.EmitOMPWorksharingLoop(S, EUB: S.getPrevEnsureUpperBound(),
3938 CodeGenLoopBounds: emitDistributeParallelForInnerBounds,
3939 CGDispatchBounds: emitDistributeParallelForDispatchBounds);
3940 };
3941
3942 emitCommonOMPParallelDirective(
3943 CGF, S, InnermostKind: isOpenMPSimdDirective(DKind: EKind) ? OMPD_for_simd : OMPD_for,
3944 CodeGen: CGInlinedWorksharingLoop,
3945 CodeGenBoundParameters: emitDistributeParallelForDistributeInnerBoundParams);
3946}
3947
3948void CodeGenFunction::EmitOMPDistributeParallelForDirective(
3949 const OMPDistributeParallelForDirective &S) {
3950 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
3951 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitInnerParallelForWhenCombined,
3952 IncExpr: S.getDistInc());
3953 };
3954 OMPLexicalScope Scope(*this, S, OMPD_parallel);
3955 CGM.getOpenMPRuntime().emitInlinedDirective(CGF&: *this, InnermostKind: OMPD_distribute, CodeGen);
3956}
3957
3958void CodeGenFunction::EmitOMPDistributeParallelForSimdDirective(
3959 const OMPDistributeParallelForSimdDirective &S) {
3960 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
3961 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitInnerParallelForWhenCombined,
3962 IncExpr: S.getDistInc());
3963 };
3964 OMPLexicalScope Scope(*this, S, OMPD_parallel);
3965 CGM.getOpenMPRuntime().emitInlinedDirective(CGF&: *this, InnermostKind: OMPD_distribute, CodeGen);
3966}
3967
3968void CodeGenFunction::EmitOMPDistributeSimdDirective(
3969 const OMPDistributeSimdDirective &S) {
3970 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
3971 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
3972 };
3973 OMPLexicalScope Scope(*this, S, OMPD_unknown);
3974 CGM.getOpenMPRuntime().emitInlinedDirective(CGF&: *this, InnermostKind: OMPD_simd, CodeGen);
3975}
3976
3977void CodeGenFunction::EmitOMPTargetSimdDeviceFunction(
3978 CodeGenModule &CGM, StringRef ParentName, const OMPTargetSimdDirective &S) {
3979 // Emit SPMD target parallel for region as a standalone region.
3980 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
3981 emitOMPSimdRegion(CGF, S, Action);
3982 };
3983 llvm::Function *Fn;
3984 llvm::Constant *Addr;
3985 // Emit target region as a standalone region.
3986 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
3987 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
3988 assert(Fn && Addr && "Target device function emission failed.");
3989}
3990
3991void CodeGenFunction::EmitOMPTargetSimdDirective(
3992 const OMPTargetSimdDirective &S) {
3993 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
3994 emitOMPSimdRegion(CGF, S, Action);
3995 };
3996 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
3997}
3998
3999namespace {
4000struct ScheduleKindModifiersTy {
4001 OpenMPScheduleClauseKind Kind;
4002 OpenMPScheduleClauseModifier M1;
4003 OpenMPScheduleClauseModifier M2;
4004 ScheduleKindModifiersTy(OpenMPScheduleClauseKind Kind,
4005 OpenMPScheduleClauseModifier M1,
4006 OpenMPScheduleClauseModifier M2)
4007 : Kind(Kind), M1(M1), M2(M2) {}
4008};
4009} // namespace
4010
4011bool CodeGenFunction::EmitOMPWorksharingLoop(
4012 const OMPLoopDirective &S, Expr *EUB,
4013 const CodeGenLoopBoundsTy &CodeGenLoopBounds,
4014 const CodeGenDispatchBoundsTy &CGDispatchBounds) {
4015 // Emit the loop iteration variable.
4016 const auto *IVExpr = cast<DeclRefExpr>(Val: S.getIterationVariable());
4017 const auto *IVDecl = cast<VarDecl>(Val: IVExpr->getDecl());
4018 EmitVarDecl(D: *IVDecl);
4019
4020 // Emit the iterations count variable.
4021 // If it is not a variable, Sema decided to calculate iterations count on each
4022 // iteration (e.g., it is foldable into a constant).
4023 if (const auto *LIExpr = dyn_cast<DeclRefExpr>(Val: S.getLastIteration())) {
4024 EmitVarDecl(D: *cast<VarDecl>(Val: LIExpr->getDecl()));
4025 // Emit calculation of the iterations count.
4026 EmitIgnoredExpr(E: S.getCalcLastIteration());
4027 }
4028
4029 CGOpenMPRuntime &RT = CGM.getOpenMPRuntime();
4030
4031 bool HasLastprivateClause;
4032 // Check pre-condition.
4033 {
4034 OMPLoopScope PreInitScope(*this, S);
4035 // Skip the entire loop if we don't meet the precondition.
4036 // If the condition constant folds and can be elided, avoid emitting the
4037 // whole loop.
4038 bool CondConstant;
4039 llvm::BasicBlock *ContBlock = nullptr;
4040 if (ConstantFoldsToSimpleInteger(Cond: S.getPreCond(), Result&: CondConstant)) {
4041 if (!CondConstant)
4042 return false;
4043 } else {
4044 llvm::BasicBlock *ThenBlock = createBasicBlock(name: "omp.precond.then");
4045 ContBlock = createBasicBlock(name: "omp.precond.end");
4046 emitPreCond(CGF&: *this, S, Cond: S.getPreCond(), TrueBlock: ThenBlock, FalseBlock: ContBlock,
4047 TrueCount: getProfileCount(S: &S));
4048 EmitBlock(BB: ThenBlock);
4049 incrementProfileCounter(S: &S);
4050 }
4051
4052 RunCleanupsScope DoacrossCleanupScope(*this);
4053 bool Ordered = false;
4054 if (const auto *OrderedClause = S.getSingleClause<OMPOrderedClause>()) {
4055 if (OrderedClause->getNumForLoops())
4056 RT.emitDoacrossInit(CGF&: *this, D: S, NumIterations: OrderedClause->getLoopNumIterations());
4057 else
4058 Ordered = true;
4059 }
4060
4061 emitAlignedClause(CGF&: *this, D: S);
4062 bool HasLinears = EmitOMPLinearClauseInit(D: S);
4063 // Emit helper vars inits.
4064
4065 std::pair<LValue, LValue> Bounds = CodeGenLoopBounds(*this, S);
4066 LValue LB = Bounds.first;
4067 LValue UB = Bounds.second;
4068 LValue ST =
4069 EmitOMPHelperVar(CGF&: *this, Helper: cast<DeclRefExpr>(Val: S.getStrideVariable()));
4070 LValue IL =
4071 EmitOMPHelperVar(CGF&: *this, Helper: cast<DeclRefExpr>(Val: S.getIsLastIterVariable()));
4072
4073 // Emit 'then' code.
4074 {
4075 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
4076 OMPPrivateScope LoopScope(*this);
4077 if (EmitOMPFirstprivateClause(D: S, PrivateScope&: LoopScope) || HasLinears) {
4078 // Emit implicit barrier to synchronize threads and avoid data races on
4079 // initialization of firstprivate variables and post-update of
4080 // lastprivate variables.
4081 CGM.getOpenMPRuntime().emitBarrierCall(
4082 CGF&: *this, Loc: S.getBeginLoc(), Kind: OMPD_unknown, /*EmitChecks=*/false,
4083 /*ForceSimpleCall=*/true);
4084 }
4085 EmitOMPPrivateClause(D: S, PrivateScope&: LoopScope);
4086 CGOpenMPRuntime::LastprivateConditionalRAII LPCRegion(
4087 *this, S, EmitLValue(E: S.getIterationVariable()));
4088 HasLastprivateClause = EmitOMPLastprivateClauseInit(D: S, PrivateScope&: LoopScope);
4089 EmitOMPReductionClauseInit(D: S, PrivateScope&: LoopScope);
4090 EmitOMPPrivateLoopCounters(S, LoopScope);
4091 EmitOMPLinearClause(D: S, PrivateScope&: LoopScope);
4092 (void)LoopScope.Privatize();
4093 if (isOpenMPTargetExecutionDirective(DKind: EKind))
4094 CGM.getOpenMPRuntime().adjustTargetSpecificDataForLambdas(CGF&: *this, D: S);
4095
4096 // Detect the loop schedule kind and chunk.
4097 const Expr *ChunkExpr = nullptr;
4098 OpenMPScheduleTy ScheduleKind;
4099 if (const auto *C = S.getSingleClause<OMPScheduleClause>()) {
4100 ScheduleKind.Schedule = C->getScheduleKind();
4101 ScheduleKind.M1 = C->getFirstScheduleModifier();
4102 ScheduleKind.M2 = C->getSecondScheduleModifier();
4103 ChunkExpr = C->getChunkSize();
4104 } else {
4105 // Default behaviour for schedule clause.
4106 CGM.getOpenMPRuntime().getDefaultScheduleAndChunk(
4107 CGF&: *this, S, ScheduleKind&: ScheduleKind.Schedule, ChunkExpr);
4108 }
4109 bool HasChunkSizeOne = false;
4110 llvm::Value *Chunk = nullptr;
4111 if (ChunkExpr) {
4112 Chunk = EmitScalarExpr(E: ChunkExpr);
4113 Chunk = EmitScalarConversion(Src: Chunk, SrcTy: ChunkExpr->getType(),
4114 DstTy: S.getIterationVariable()->getType(),
4115 Loc: S.getBeginLoc());
4116 Expr::EvalResult Result;
4117 if (ChunkExpr->EvaluateAsInt(Result, Ctx: getContext())) {
4118 llvm::APSInt EvaluatedChunk = Result.Val.getInt();
4119 HasChunkSizeOne = (EvaluatedChunk.getLimitedValue() == 1);
4120 }
4121 }
4122 const unsigned IVSize = getContext().getTypeSize(T: IVExpr->getType());
4123 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation();
4124 // OpenMP 4.5, 2.7.1 Loop Construct, Description.
4125 // If the static schedule kind is specified or if the ordered clause is
4126 // specified, and if no monotonic modifier is specified, the effect will
4127 // be as if the monotonic modifier was specified.
4128 bool StaticChunkedOne =
4129 RT.isStaticChunked(ScheduleKind: ScheduleKind.Schedule,
4130 /* Chunked */ Chunk != nullptr) &&
4131 HasChunkSizeOne && isOpenMPLoopBoundSharingDirective(Kind: EKind);
4132 // GPU combined `distribute parallel for`: emit a single
4133 // for_static_init with the fused distr_static_chunk + static_chunkone
4134 // schedule (enum 93). The surrounding EmitOMPDistributeLoop must skip
4135 // its distribute_static_init under the same conditions. Both sites are
4136 // guarded by canEmitGPUFusedDistSchedule() alone so they cannot
4137 // disagree; the assert guards the invariant that makes this safe today,
4138 // aka that the implicit GPU default schedule is always static chunk-one.
4139 ScheduleKind.UseFusedDistChunkSchedule =
4140 canEmitGPUFusedDistSchedule(CGM, S, DKind: EKind);
4141 assert((!ScheduleKind.UseFusedDistChunkSchedule || StaticChunkedOne) &&
4142 "fused distribute schedule requires a static chunk-one schedule");
4143 bool IsMonotonic =
4144 Ordered ||
4145 (ScheduleKind.Schedule == OMPC_SCHEDULE_static &&
4146 !(ScheduleKind.M1 == OMPC_SCHEDULE_MODIFIER_nonmonotonic ||
4147 ScheduleKind.M2 == OMPC_SCHEDULE_MODIFIER_nonmonotonic)) ||
4148 ScheduleKind.M1 == OMPC_SCHEDULE_MODIFIER_monotonic ||
4149 ScheduleKind.M2 == OMPC_SCHEDULE_MODIFIER_monotonic;
4150 if ((RT.isStaticNonchunked(ScheduleKind: ScheduleKind.Schedule,
4151 /* Chunked */ Chunk != nullptr) ||
4152 StaticChunkedOne) &&
4153 !Ordered) {
4154 JumpDest LoopExit =
4155 getJumpDestInCurrentScope(Target: createBasicBlock(name: "omp.loop.exit"));
4156 emitCommonSimdLoop(
4157 CGF&: *this, S,
4158 SimdInitGen: [&S, EKind](CodeGenFunction &CGF, PrePostActionTy &) {
4159 if (isOpenMPSimdDirective(DKind: EKind)) {
4160 CGF.EmitOMPSimdInit(D: S);
4161 } else if (const auto *C = S.getSingleClause<OMPOrderClause>()) {
4162 if (C->getKind() == OMPC_ORDER_concurrent)
4163 CGF.LoopStack.setParallel(/*Enable=*/true);
4164 }
4165 },
4166 BodyCodeGen: [IVSize, IVSigned, Ordered, IL, LB, UB, ST, StaticChunkedOne, Chunk,
4167 &S, ScheduleKind, LoopExit, EKind,
4168 &LoopScope](CodeGenFunction &CGF, PrePostActionTy &) {
4169 // OpenMP [2.7.1, Loop Construct, Description, table 2-1]
4170 // When no chunk_size is specified, the iteration space is divided
4171 // into chunks that are approximately equal in size, and at most
4172 // one chunk is distributed to each thread. Note that the size of
4173 // the chunks is unspecified in this case.
4174 CGOpenMPRuntime::StaticRTInput StaticInit(
4175 IVSize, IVSigned, Ordered, IL.getAddress(), LB.getAddress(),
4176 UB.getAddress(), ST.getAddress(),
4177 StaticChunkedOne ? Chunk : nullptr);
4178 CGF.CGM.getOpenMPRuntime().emitForStaticInit(
4179 CGF, Loc: S.getBeginLoc(), DKind: EKind, ScheduleKind, Values: StaticInit);
4180 // UB = min(UB, GlobalUB);
4181 if (!StaticChunkedOne)
4182 CGF.EmitIgnoredExpr(E: S.getEnsureUpperBound());
4183 // IV = LB;
4184 CGF.EmitIgnoredExpr(E: S.getInit());
4185 // For unchunked static schedule generate:
4186 //
4187 // while (idx <= UB) {
4188 // BODY;
4189 // ++idx;
4190 // }
4191 //
4192 // For static schedule with chunk one:
4193 //
4194 // while (IV <= PrevUB) {
4195 // BODY;
4196 // IV += ST;
4197 // }
4198 CGF.EmitOMPInnerLoop(
4199 S, RequiresCleanup: LoopScope.requiresCleanups(),
4200 LoopCond: StaticChunkedOne ? S.getCombinedParForInDistCond()
4201 : S.getCond(),
4202 IncExpr: StaticChunkedOne ? S.getDistInc() : S.getInc(),
4203 BodyGen: [&S, LoopExit](CodeGenFunction &CGF) {
4204 emitOMPLoopBodyWithStopPoint(CGF, S, LoopExit);
4205 },
4206 PostIncGen: [](CodeGenFunction &) {});
4207 });
4208 EmitBlock(BB: LoopExit.getBlock());
4209 // Tell the runtime we are done.
4210 auto &&CodeGen = [&S](CodeGenFunction &CGF) {
4211 CGF.CGM.getOpenMPRuntime().emitForStaticFinish(CGF, Loc: S.getEndLoc(),
4212 DKind: OMPD_for);
4213 };
4214 OMPCancelStack.emitExit(CGF&: *this, Kind: EKind, CodeGen);
4215 } else {
4216 // Emit the outer loop, which requests its work chunk [LB..UB] from
4217 // runtime and runs the inner loop to process it.
4218 OMPLoopArguments LoopArguments(LB.getAddress(), UB.getAddress(),
4219 ST.getAddress(), IL.getAddress(), Chunk,
4220 EUB);
4221 LoopArguments.DKind = OMPD_for;
4222 EmitOMPForOuterLoop(ScheduleKind, IsMonotonic, S, LoopScope, Ordered,
4223 LoopArgs: LoopArguments, CGDispatchBounds);
4224 }
4225 if (isOpenMPSimdDirective(DKind: EKind)) {
4226 EmitOMPSimdFinal(D: S, CondGen: [IL, &S](CodeGenFunction &CGF) {
4227 return CGF.Builder.CreateIsNotNull(
4228 Arg: CGF.EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc()));
4229 });
4230 }
4231 EmitOMPReductionClauseFinal(
4232 D: S, /*ReductionKind=*/isOpenMPSimdDirective(DKind: EKind)
4233 ? /*Parallel and Simd*/ OMPD_parallel_for_simd
4234 : /*Parallel only*/ OMPD_parallel);
4235 // Emit post-update of the reduction variables if IsLastIter != 0.
4236 emitPostUpdateForReductionClause(
4237 CGF&: *this, D: S, CondGen: [IL, &S](CodeGenFunction &CGF) {
4238 return CGF.Builder.CreateIsNotNull(
4239 Arg: CGF.EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc()));
4240 });
4241 // Emit final copy of the lastprivate variables if IsLastIter != 0.
4242 if (HasLastprivateClause)
4243 EmitOMPLastprivateClauseFinal(
4244 D: S, NoFinals: isOpenMPSimdDirective(DKind: EKind),
4245 IsLastIterCond: Builder.CreateIsNotNull(Arg: EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc())));
4246 LoopScope.restoreMap();
4247 EmitOMPLinearClauseFinal(D: S, CondGen: [IL, &S](CodeGenFunction &CGF) {
4248 return CGF.Builder.CreateIsNotNull(
4249 Arg: CGF.EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc()));
4250 });
4251 }
4252 DoacrossCleanupScope.ForceCleanup();
4253 // We're now done with the loop, so jump to the continuation block.
4254 if (ContBlock) {
4255 EmitBranch(Block: ContBlock);
4256 EmitBlock(BB: ContBlock, /*IsFinished=*/true);
4257 }
4258 }
4259 return HasLastprivateClause;
4260}
4261
4262/// The following two functions generate expressions for the loop lower
4263/// and upper bounds in case of static and dynamic (dispatch) schedule
4264/// of the associated 'for' or 'distribute' loop.
4265static std::pair<LValue, LValue>
4266emitForLoopBounds(CodeGenFunction &CGF, const OMPExecutableDirective &S) {
4267 const auto &LS = cast<OMPLoopDirective>(Val: S);
4268 LValue LB =
4269 EmitOMPHelperVar(CGF, Helper: cast<DeclRefExpr>(Val: LS.getLowerBoundVariable()));
4270 LValue UB =
4271 EmitOMPHelperVar(CGF, Helper: cast<DeclRefExpr>(Val: LS.getUpperBoundVariable()));
4272 return {LB, UB};
4273}
4274
4275/// When dealing with dispatch schedules (e.g. dynamic, guided) we do not
4276/// consider the lower and upper bound expressions generated by the
4277/// worksharing loop support, but we use 0 and the iteration space size as
4278/// constants
4279static std::pair<llvm::Value *, llvm::Value *>
4280emitDispatchForLoopBounds(CodeGenFunction &CGF, const OMPExecutableDirective &S,
4281 Address LB, Address UB) {
4282 const auto &LS = cast<OMPLoopDirective>(Val: S);
4283 const Expr *IVExpr = LS.getIterationVariable();
4284 const unsigned IVSize = CGF.getContext().getTypeSize(T: IVExpr->getType());
4285 llvm::Value *LBVal = CGF.Builder.getIntN(N: IVSize, C: 0);
4286 llvm::Value *UBVal = CGF.EmitScalarExpr(E: LS.getLastIteration());
4287 return {LBVal, UBVal};
4288}
4289
4290/// Emits internal temp array declarations for the directive with inscan
4291/// reductions.
4292/// The code is the following:
4293/// \code
4294/// size num_iters = <num_iters>;
4295/// <type> buffer[num_iters];
4296/// \endcode
4297static void emitScanBasedDirectiveDecls(
4298 CodeGenFunction &CGF, const OMPLoopDirective &S,
4299 llvm::function_ref<llvm::Value *(CodeGenFunction &)> NumIteratorsGen) {
4300 llvm::Value *OMPScanNumIterations = CGF.Builder.CreateIntCast(
4301 V: NumIteratorsGen(CGF), DestTy: CGF.SizeTy, /*isSigned=*/false);
4302 SmallVector<const Expr *, 4> Shareds;
4303 SmallVector<const Expr *, 4> Privates;
4304 SmallVector<const Expr *, 4> ReductionOps;
4305 SmallVector<const Expr *, 4> CopyArrayTemps;
4306 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
4307 assert(C->getModifier() == OMPC_REDUCTION_inscan &&
4308 "Only inscan reductions are expected.");
4309 Shareds.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
4310 Privates.append(in_start: C->privates().begin(), in_end: C->privates().end());
4311 ReductionOps.append(in_start: C->reduction_ops().begin(), in_end: C->reduction_ops().end());
4312 CopyArrayTemps.append(in_start: C->copy_array_temps().begin(),
4313 in_end: C->copy_array_temps().end());
4314 }
4315 {
4316 // Emit buffers for each reduction variables.
4317 // ReductionCodeGen is required to emit correctly the code for array
4318 // reductions.
4319 ReductionCodeGen RedCG(Shareds, Shareds, Privates, ReductionOps);
4320 unsigned Count = 0;
4321 auto *ITA = CopyArrayTemps.begin();
4322 for (const Expr *IRef : Privates) {
4323 const auto *PrivateVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: IRef)->getDecl());
4324 // Emit variably modified arrays, used for arrays/array sections
4325 // reductions.
4326 if (PrivateVD->getType()->isVariablyModifiedType()) {
4327 RedCG.emitSharedOrigLValue(CGF, N: Count);
4328 RedCG.emitAggregateType(CGF, N: Count);
4329 }
4330 CodeGenFunction::OpaqueValueMapping DimMapping(
4331 CGF,
4332 cast<OpaqueValueExpr>(
4333 Val: cast<VariableArrayType>(Val: (*ITA)->getType()->getAsArrayTypeUnsafe())
4334 ->getSizeExpr()),
4335 RValue::get(V: OMPScanNumIterations));
4336 // Emit temp buffer.
4337 CGF.EmitVarDecl(D: *cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ITA)->getDecl()));
4338 ++ITA;
4339 ++Count;
4340 }
4341 }
4342}
4343
4344/// Copies final inscan reductions values to the original variables.
4345/// The code is the following:
4346/// \code
4347/// <orig_var> = buffer[num_iters-1];
4348/// \endcode
4349static void emitScanBasedDirectiveFinals(
4350 CodeGenFunction &CGF, const OMPLoopDirective &S,
4351 llvm::function_ref<llvm::Value *(CodeGenFunction &)> NumIteratorsGen) {
4352 llvm::Value *OMPScanNumIterations = CGF.Builder.CreateIntCast(
4353 V: NumIteratorsGen(CGF), DestTy: CGF.SizeTy, /*isSigned=*/false);
4354 SmallVector<const Expr *, 4> Shareds;
4355 SmallVector<const Expr *, 4> LHSs;
4356 SmallVector<const Expr *, 4> RHSs;
4357 SmallVector<const Expr *, 4> Privates;
4358 SmallVector<const Expr *, 4> CopyOps;
4359 SmallVector<const Expr *, 4> CopyArrayElems;
4360 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
4361 assert(C->getModifier() == OMPC_REDUCTION_inscan &&
4362 "Only inscan reductions are expected.");
4363 Shareds.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
4364 LHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
4365 RHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
4366 Privates.append(in_start: C->privates().begin(), in_end: C->privates().end());
4367 CopyOps.append(in_start: C->copy_ops().begin(), in_end: C->copy_ops().end());
4368 CopyArrayElems.append(in_start: C->copy_array_elems().begin(),
4369 in_end: C->copy_array_elems().end());
4370 }
4371 // Create temp var and copy LHS value to this temp value.
4372 // LHS = TMP[LastIter];
4373 llvm::Value *OMPLast = CGF.Builder.CreateNSWSub(
4374 LHS: OMPScanNumIterations,
4375 RHS: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 1, /*isSigned=*/IsSigned: false));
4376 for (unsigned I = 0, E = CopyArrayElems.size(); I < E; ++I) {
4377 const Expr *PrivateExpr = Privates[I];
4378 const Expr *OrigExpr = Shareds[I];
4379 const Expr *CopyArrayElem = CopyArrayElems[I];
4380 CodeGenFunction::OpaqueValueMapping IdxMapping(
4381 CGF,
4382 cast<OpaqueValueExpr>(
4383 Val: cast<ArraySubscriptExpr>(Val: CopyArrayElem)->getIdx()),
4384 RValue::get(V: OMPLast));
4385 LValue DestLVal = CGF.EmitLValue(E: OrigExpr);
4386 LValue SrcLVal = CGF.EmitLValue(E: CopyArrayElem);
4387 CGF.EmitOMPCopy(
4388 OriginalType: PrivateExpr->getType(), DestAddr: DestLVal.getAddress(), SrcAddr: SrcLVal.getAddress(),
4389 DestVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: LHSs[I])->getDecl()),
4390 SrcVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: RHSs[I])->getDecl()), Copy: CopyOps[I]);
4391 }
4392}
4393
4394/// Emits the code for the directive with inscan reductions.
4395/// The code is the following:
4396/// \code
4397/// #pragma omp ...
4398/// for (i: 0..<num_iters>) {
4399/// <input phase>;
4400/// buffer[i] = red;
4401/// }
4402/// #pragma omp master // in parallel region
4403/// for (int k = 0; k != ceil(log2(num_iters)); ++k)
4404/// for (size cnt = last_iter; cnt >= pow(2, k); --k)
4405/// buffer[i] op= buffer[i-pow(2,k)];
4406/// #pragma omp barrier // in parallel region
4407/// #pragma omp ...
4408/// for (0..<num_iters>) {
4409/// red = InclusiveScan ? buffer[i] : buffer[i-1];
4410/// <scan phase>;
4411/// }
4412/// \endcode
4413static void emitScanBasedDirective(
4414 CodeGenFunction &CGF, const OMPLoopDirective &S,
4415 llvm::function_ref<llvm::Value *(CodeGenFunction &)> NumIteratorsGen,
4416 llvm::function_ref<void(CodeGenFunction &)> FirstGen,
4417 llvm::function_ref<void(CodeGenFunction &)> SecondGen) {
4418 llvm::Value *OMPScanNumIterations = CGF.Builder.CreateIntCast(
4419 V: NumIteratorsGen(CGF), DestTy: CGF.SizeTy, /*isSigned=*/false);
4420 SmallVector<const Expr *, 4> Privates;
4421 SmallVector<const Expr *, 4> ReductionOps;
4422 SmallVector<const Expr *, 4> LHSs;
4423 SmallVector<const Expr *, 4> RHSs;
4424 SmallVector<const Expr *, 4> CopyArrayElems;
4425 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
4426 assert(C->getModifier() == OMPC_REDUCTION_inscan &&
4427 "Only inscan reductions are expected.");
4428 Privates.append(in_start: C->privates().begin(), in_end: C->privates().end());
4429 ReductionOps.append(in_start: C->reduction_ops().begin(), in_end: C->reduction_ops().end());
4430 LHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
4431 RHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
4432 CopyArrayElems.append(in_start: C->copy_array_elems().begin(),
4433 in_end: C->copy_array_elems().end());
4434 }
4435 CodeGenFunction::ParentLoopDirectiveForScanRegion ScanRegion(CGF, S);
4436 {
4437 // Emit loop with input phase:
4438 // #pragma omp ...
4439 // for (i: 0..<num_iters>) {
4440 // <input phase>;
4441 // buffer[i] = red;
4442 // }
4443 CGF.OMPFirstScanLoop = true;
4444 CodeGenFunction::OMPLocalDeclMapRAII Scope(CGF);
4445 FirstGen(CGF);
4446 }
4447 // #pragma omp barrier // in parallel region
4448 auto &&CodeGen = [&S, OMPScanNumIterations, &LHSs, &RHSs, &CopyArrayElems,
4449 &ReductionOps,
4450 &Privates](CodeGenFunction &CGF, PrePostActionTy &Action) {
4451 Action.Enter(CGF);
4452 // Emit prefix reduction:
4453 // #pragma omp master // in parallel region
4454 // for (int k = 0; k <= ceil(log2(n)); ++k)
4455 llvm::BasicBlock *InputBB = CGF.Builder.GetInsertBlock();
4456 llvm::BasicBlock *LoopBB = CGF.createBasicBlock(name: "omp.outer.log.scan.body");
4457 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(name: "omp.outer.log.scan.exit");
4458 llvm::Function *F =
4459 CGF.CGM.getIntrinsic(IID: llvm::Intrinsic::log2, Tys: CGF.DoubleTy);
4460 llvm::Value *Arg =
4461 CGF.Builder.CreateUIToFP(V: OMPScanNumIterations, DestTy: CGF.DoubleTy);
4462 llvm::Value *LogVal = CGF.EmitNounwindRuntimeCall(callee: F, args: Arg);
4463 F = CGF.CGM.getIntrinsic(IID: llvm::Intrinsic::ceil, Tys: CGF.DoubleTy);
4464 LogVal = CGF.EmitNounwindRuntimeCall(callee: F, args: LogVal);
4465 LogVal = CGF.Builder.CreateFPToUI(V: LogVal, DestTy: CGF.IntTy);
4466 llvm::Value *NMin1 = CGF.Builder.CreateNUWSub(
4467 LHS: OMPScanNumIterations, RHS: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 1));
4468 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, TemporaryLocation: S.getBeginLoc());
4469 CGF.EmitBlock(BB: LoopBB);
4470 auto *Counter = CGF.Builder.CreatePHI(Ty: CGF.IntTy, NumReservedValues: 2);
4471 // size pow2k = 1;
4472 auto *Pow2K = CGF.Builder.CreatePHI(Ty: CGF.SizeTy, NumReservedValues: 2);
4473 Counter->addIncoming(V: llvm::ConstantInt::get(Ty: CGF.IntTy, V: 0), BB: InputBB);
4474 Pow2K->addIncoming(V: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 1), BB: InputBB);
4475 // for (size i = n - 1; i >= 2 ^ k; --i)
4476 // tmp[i] op= tmp[i-pow2k];
4477 llvm::BasicBlock *InnerLoopBB =
4478 CGF.createBasicBlock(name: "omp.inner.log.scan.body");
4479 llvm::BasicBlock *InnerExitBB =
4480 CGF.createBasicBlock(name: "omp.inner.log.scan.exit");
4481 llvm::Value *CmpI = CGF.Builder.CreateICmpUGE(LHS: NMin1, RHS: Pow2K);
4482 CGF.Builder.CreateCondBr(Cond: CmpI, True: InnerLoopBB, False: InnerExitBB);
4483 CGF.EmitBlock(BB: InnerLoopBB);
4484 auto *IVal = CGF.Builder.CreatePHI(Ty: CGF.SizeTy, NumReservedValues: 2);
4485 IVal->addIncoming(V: NMin1, BB: LoopBB);
4486 {
4487 CodeGenFunction::OMPPrivateScope PrivScope(CGF);
4488 auto *ILHS = LHSs.begin();
4489 auto *IRHS = RHSs.begin();
4490 for (const Expr *CopyArrayElem : CopyArrayElems) {
4491 const auto *LHSVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *ILHS)->getDecl());
4492 const auto *RHSVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: *IRHS)->getDecl());
4493 Address LHSAddr = Address::invalid();
4494 {
4495 CodeGenFunction::OpaqueValueMapping IdxMapping(
4496 CGF,
4497 cast<OpaqueValueExpr>(
4498 Val: cast<ArraySubscriptExpr>(Val: CopyArrayElem)->getIdx()),
4499 RValue::get(V: IVal));
4500 LHSAddr = CGF.EmitLValue(E: CopyArrayElem).getAddress();
4501 }
4502 PrivScope.addPrivate(LocalVD: LHSVD, Addr: LHSAddr);
4503 Address RHSAddr = Address::invalid();
4504 {
4505 llvm::Value *OffsetIVal = CGF.Builder.CreateNUWSub(LHS: IVal, RHS: Pow2K);
4506 CodeGenFunction::OpaqueValueMapping IdxMapping(
4507 CGF,
4508 cast<OpaqueValueExpr>(
4509 Val: cast<ArraySubscriptExpr>(Val: CopyArrayElem)->getIdx()),
4510 RValue::get(V: OffsetIVal));
4511 RHSAddr = CGF.EmitLValue(E: CopyArrayElem).getAddress();
4512 }
4513 PrivScope.addPrivate(LocalVD: RHSVD, Addr: RHSAddr);
4514 ++ILHS;
4515 ++IRHS;
4516 }
4517 PrivScope.Privatize();
4518 CGF.CGM.getOpenMPRuntime().emitReduction(
4519 CGF, Loc: S.getEndLoc(), Privates, LHSExprs: LHSs, RHSExprs: RHSs, ReductionOps,
4520 Options: {/*WithNowait=*/true, /*SimpleReduction=*/true,
4521 /*IsPrivateVarReduction*/ {}, .ReductionKind: OMPD_unknown});
4522 }
4523 llvm::Value *NextIVal =
4524 CGF.Builder.CreateNUWSub(LHS: IVal, RHS: llvm::ConstantInt::get(Ty: CGF.SizeTy, V: 1));
4525 IVal->addIncoming(V: NextIVal, BB: CGF.Builder.GetInsertBlock());
4526 CmpI = CGF.Builder.CreateICmpUGE(LHS: NextIVal, RHS: Pow2K);
4527 CGF.Builder.CreateCondBr(Cond: CmpI, True: InnerLoopBB, False: InnerExitBB);
4528 CGF.EmitBlock(BB: InnerExitBB);
4529 llvm::Value *Next =
4530 CGF.Builder.CreateNUWAdd(LHS: Counter, RHS: llvm::ConstantInt::get(Ty: CGF.IntTy, V: 1));
4531 Counter->addIncoming(V: Next, BB: CGF.Builder.GetInsertBlock());
4532 // pow2k <<= 1;
4533 llvm::Value *NextPow2K =
4534 CGF.Builder.CreateShl(LHS: Pow2K, RHS: 1, Name: "", /*HasNUW=*/true);
4535 Pow2K->addIncoming(V: NextPow2K, BB: CGF.Builder.GetInsertBlock());
4536 llvm::Value *Cmp = CGF.Builder.CreateICmpNE(LHS: Next, RHS: LogVal);
4537 CGF.Builder.CreateCondBr(Cond: Cmp, True: LoopBB, False: ExitBB);
4538 auto DL1 = ApplyDebugLocation::CreateDefaultArtificial(CGF, TemporaryLocation: S.getEndLoc());
4539 CGF.EmitBlock(BB: ExitBB);
4540 };
4541 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
4542 if (isOpenMPParallelDirective(DKind: EKind)) {
4543 CGF.CGM.getOpenMPRuntime().emitMasterRegion(CGF, MasterOpGen: CodeGen, Loc: S.getBeginLoc());
4544 CGF.CGM.getOpenMPRuntime().emitBarrierCall(
4545 CGF, Loc: S.getBeginLoc(), Kind: OMPD_unknown, /*EmitChecks=*/false,
4546 /*ForceSimpleCall=*/true);
4547 } else {
4548 RegionCodeGenTy RCG(CodeGen);
4549 RCG(CGF);
4550 }
4551
4552 CGF.OMPFirstScanLoop = false;
4553 SecondGen(CGF);
4554}
4555
4556static bool emitWorksharingDirective(CodeGenFunction &CGF,
4557 const OMPLoopDirective &S,
4558 bool HasCancel) {
4559 bool HasLastprivates;
4560 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
4561 if (llvm::any_of(Range: S.getClausesOfKind<OMPReductionClause>(),
4562 P: [](const OMPReductionClause *C) {
4563 return C->getModifier() == OMPC_REDUCTION_inscan;
4564 })) {
4565 const auto &&NumIteratorsGen = [&S](CodeGenFunction &CGF) {
4566 CodeGenFunction::OMPLocalDeclMapRAII Scope(CGF);
4567 OMPLoopScope LoopScope(CGF, S);
4568 return CGF.EmitScalarExpr(E: S.getNumIterations());
4569 };
4570 const auto &&FirstGen = [&S, HasCancel, EKind](CodeGenFunction &CGF) {
4571 CodeGenFunction::OMPCancelStackRAII CancelRegion(CGF, EKind, HasCancel);
4572 (void)CGF.EmitOMPWorksharingLoop(S, EUB: S.getEnsureUpperBound(),
4573 CodeGenLoopBounds: emitForLoopBounds,
4574 CGDispatchBounds: emitDispatchForLoopBounds);
4575 // Emit an implicit barrier at the end.
4576 CGF.CGM.getOpenMPRuntime().emitBarrierCall(CGF, Loc: S.getBeginLoc(),
4577 Kind: OMPD_for);
4578 };
4579 const auto &&SecondGen = [&S, HasCancel, EKind,
4580 &HasLastprivates](CodeGenFunction &CGF) {
4581 CodeGenFunction::OMPCancelStackRAII CancelRegion(CGF, EKind, HasCancel);
4582 HasLastprivates = CGF.EmitOMPWorksharingLoop(S, EUB: S.getEnsureUpperBound(),
4583 CodeGenLoopBounds: emitForLoopBounds,
4584 CGDispatchBounds: emitDispatchForLoopBounds);
4585 };
4586 if (!isOpenMPParallelDirective(DKind: EKind))
4587 emitScanBasedDirectiveDecls(CGF, S, NumIteratorsGen);
4588 emitScanBasedDirective(CGF, S, NumIteratorsGen, FirstGen, SecondGen);
4589 if (!isOpenMPParallelDirective(DKind: EKind))
4590 emitScanBasedDirectiveFinals(CGF, S, NumIteratorsGen);
4591 } else {
4592 CodeGenFunction::OMPCancelStackRAII CancelRegion(CGF, EKind, HasCancel);
4593 HasLastprivates = CGF.EmitOMPWorksharingLoop(S, EUB: S.getEnsureUpperBound(),
4594 CodeGenLoopBounds: emitForLoopBounds,
4595 CGDispatchBounds: emitDispatchForLoopBounds);
4596 }
4597 return HasLastprivates;
4598}
4599
4600// Pass OMPLoopDirective (instead of OMPForDirective) to make this check
4601// available for "loop bind(parallel)", which maps to "for".
4602static bool isForSupportedByOpenMPIRBuilder(const OMPLoopDirective &S,
4603 bool HasCancel) {
4604 if (HasCancel)
4605 return false;
4606 for (OMPClause *C : S.clauses()) {
4607 if (isa<OMPNowaitClause, OMPBindClause>(Val: C))
4608 continue;
4609
4610 if (auto *SC = dyn_cast<OMPScheduleClause>(Val: C)) {
4611 if (SC->getFirstScheduleModifier() != OMPC_SCHEDULE_MODIFIER_unknown)
4612 return false;
4613 if (SC->getSecondScheduleModifier() != OMPC_SCHEDULE_MODIFIER_unknown)
4614 return false;
4615 switch (SC->getScheduleKind()) {
4616 case OMPC_SCHEDULE_auto:
4617 case OMPC_SCHEDULE_dynamic:
4618 case OMPC_SCHEDULE_runtime:
4619 case OMPC_SCHEDULE_guided:
4620 case OMPC_SCHEDULE_static:
4621 continue;
4622 case OMPC_SCHEDULE_unknown:
4623 return false;
4624 }
4625 }
4626
4627 return false;
4628 }
4629
4630 return true;
4631}
4632
4633static llvm::omp::ScheduleKind
4634convertClauseKindToSchedKind(OpenMPScheduleClauseKind ScheduleClauseKind) {
4635 switch (ScheduleClauseKind) {
4636 case OMPC_SCHEDULE_unknown:
4637 return llvm::omp::OMP_SCHEDULE_Default;
4638 case OMPC_SCHEDULE_auto:
4639 return llvm::omp::OMP_SCHEDULE_Auto;
4640 case OMPC_SCHEDULE_dynamic:
4641 return llvm::omp::OMP_SCHEDULE_Dynamic;
4642 case OMPC_SCHEDULE_guided:
4643 return llvm::omp::OMP_SCHEDULE_Guided;
4644 case OMPC_SCHEDULE_runtime:
4645 return llvm::omp::OMP_SCHEDULE_Runtime;
4646 case OMPC_SCHEDULE_static:
4647 return llvm::omp::OMP_SCHEDULE_Static;
4648 }
4649 llvm_unreachable("Unhandled schedule kind");
4650}
4651
4652// Pass OMPLoopDirective (instead of OMPForDirective) to make this function
4653// available for "loop bind(parallel)", which maps to "for".
4654static void emitOMPForDirective(const OMPLoopDirective &S, CodeGenFunction &CGF,
4655 CodeGenModule &CGM, bool HasCancel) {
4656 bool HasLastprivates = false;
4657 bool UseOMPIRBuilder = CGM.getLangOpts().OpenMPIRBuilder &&
4658 isForSupportedByOpenMPIRBuilder(S, HasCancel);
4659 auto &&CodeGen = [&S, &CGM, HasCancel, &HasLastprivates,
4660 UseOMPIRBuilder](CodeGenFunction &CGF, PrePostActionTy &) {
4661 // Use the OpenMPIRBuilder if enabled.
4662 if (UseOMPIRBuilder) {
4663 bool NeedsBarrier = !S.getSingleClause<OMPNowaitClause>();
4664
4665 llvm::omp::ScheduleKind SchedKind = llvm::omp::OMP_SCHEDULE_Default;
4666 llvm::Value *ChunkSize = nullptr;
4667 if (auto *SchedClause = S.getSingleClause<OMPScheduleClause>()) {
4668 SchedKind =
4669 convertClauseKindToSchedKind(ScheduleClauseKind: SchedClause->getScheduleKind());
4670 if (const Expr *ChunkSizeExpr = SchedClause->getChunkSize())
4671 ChunkSize = CGF.EmitScalarExpr(E: ChunkSizeExpr);
4672 }
4673
4674 // Emit the associated statement and get its loop representation.
4675 const Stmt *Inner = S.getRawStmt();
4676 llvm::CanonicalLoopInfo *CLI =
4677 CGF.EmitOMPCollapsedCanonicalLoopNest(S: Inner, Depth: 1);
4678
4679 llvm::OpenMPIRBuilder &OMPBuilder =
4680 CGM.getOpenMPRuntime().getOMPBuilder();
4681 llvm::OpenMPIRBuilder::InsertPointTy AllocaIP(
4682 CGF.AllocaInsertPt->getIterator());
4683 cantFail(ValOrErr: OMPBuilder.applyWorkshareLoop(
4684 DL: CGF.Builder.getCurrentDebugLocation(), CLI, AllocaIP, NeedsBarrier,
4685 SchedKind, ChunkSize, /*HasSimdModifier=*/false,
4686 /*HasMonotonicModifier=*/false, /*HasNonmonotonicModifier=*/false,
4687 /*HasOrderedClause=*/false));
4688 return;
4689 }
4690
4691 HasLastprivates = emitWorksharingDirective(CGF, S, HasCancel);
4692 };
4693 {
4694 auto LPCRegion =
4695 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF, S);
4696 OMPLexicalScope Scope(CGF, S, OMPD_unknown);
4697 CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_for, CodeGen,
4698 HasCancel);
4699 }
4700
4701 if (!UseOMPIRBuilder) {
4702 // Emit an implicit barrier at the end.
4703 if (!S.getSingleClause<OMPNowaitClause>() || HasLastprivates)
4704 CGM.getOpenMPRuntime().emitBarrierCall(CGF, Loc: S.getBeginLoc(), Kind: OMPD_for);
4705 }
4706 // Check for outer lastprivate conditional update.
4707 checkForLastprivateConditionalUpdate(CGF, S);
4708}
4709
4710void CodeGenFunction::EmitOMPForDirective(const OMPForDirective &S) {
4711 return emitOMPForDirective(S, CGF&: *this, CGM, HasCancel: S.hasCancel());
4712}
4713
4714void CodeGenFunction::EmitOMPForSimdDirective(const OMPForSimdDirective &S) {
4715 bool HasLastprivates = false;
4716 auto &&CodeGen = [&S, &HasLastprivates](CodeGenFunction &CGF,
4717 PrePostActionTy &) {
4718 HasLastprivates = emitWorksharingDirective(CGF, S, /*HasCancel=*/false);
4719 };
4720 {
4721 auto LPCRegion =
4722 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
4723 OMPLexicalScope Scope(*this, S, OMPD_unknown);
4724 CGM.getOpenMPRuntime().emitInlinedDirective(CGF&: *this, InnermostKind: OMPD_simd, CodeGen);
4725 }
4726
4727 // Emit an implicit barrier at the end.
4728 if (!S.getSingleClause<OMPNowaitClause>() || HasLastprivates)
4729 CGM.getOpenMPRuntime().emitBarrierCall(CGF&: *this, Loc: S.getBeginLoc(), Kind: OMPD_for);
4730 // Check for outer lastprivate conditional update.
4731 checkForLastprivateConditionalUpdate(CGF&: *this, S);
4732}
4733
4734static LValue createSectionLVal(CodeGenFunction &CGF, QualType Ty,
4735 const Twine &Name,
4736 llvm::Value *Init = nullptr) {
4737 LValue LVal = CGF.MakeAddrLValue(Addr: CGF.CreateMemTemp(T: Ty, Name), T: Ty);
4738 if (Init)
4739 CGF.EmitStoreThroughLValue(Src: RValue::get(V: Init), Dst: LVal, /*isInit*/ true);
4740 return LVal;
4741}
4742
4743void CodeGenFunction::EmitSections(const OMPExecutableDirective &S) {
4744 const Stmt *CapturedStmt = S.getInnermostCapturedStmt()->getCapturedStmt();
4745 const auto *CS = dyn_cast<CompoundStmt>(Val: CapturedStmt);
4746 bool HasLastprivates = false;
4747 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
4748 auto &&CodeGen = [&S, CapturedStmt, CS, EKind,
4749 &HasLastprivates](CodeGenFunction &CGF, PrePostActionTy &) {
4750 const ASTContext &C = CGF.getContext();
4751 QualType KmpInt32Ty =
4752 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
4753 // Emit helper vars inits.
4754 LValue LB = createSectionLVal(CGF, Ty: KmpInt32Ty, Name: ".omp.sections.lb.",
4755 Init: CGF.Builder.getInt32(C: 0));
4756 llvm::ConstantInt *GlobalUBVal = CS != nullptr
4757 ? CGF.Builder.getInt32(C: CS->size() - 1)
4758 : CGF.Builder.getInt32(C: 0);
4759 LValue UB =
4760 createSectionLVal(CGF, Ty: KmpInt32Ty, Name: ".omp.sections.ub.", Init: GlobalUBVal);
4761 LValue ST = createSectionLVal(CGF, Ty: KmpInt32Ty, Name: ".omp.sections.st.",
4762 Init: CGF.Builder.getInt32(C: 1));
4763 LValue IL = createSectionLVal(CGF, Ty: KmpInt32Ty, Name: ".omp.sections.il.",
4764 Init: CGF.Builder.getInt32(C: 0));
4765 // Loop counter.
4766 LValue IV = createSectionLVal(CGF, Ty: KmpInt32Ty, Name: ".omp.sections.iv.");
4767 OpaqueValueExpr IVRefExpr(S.getBeginLoc(), KmpInt32Ty, VK_LValue);
4768 CodeGenFunction::OpaqueValueMapping OpaqueIV(CGF, &IVRefExpr, IV);
4769 OpaqueValueExpr UBRefExpr(S.getBeginLoc(), KmpInt32Ty, VK_LValue);
4770 CodeGenFunction::OpaqueValueMapping OpaqueUB(CGF, &UBRefExpr, UB);
4771 // Generate condition for loop.
4772 BinaryOperator *Cond = BinaryOperator::Create(
4773 C, lhs: &IVRefExpr, rhs: &UBRefExpr, opc: BO_LE, ResTy: C.BoolTy, VK: VK_PRValue, OK: OK_Ordinary,
4774 opLoc: S.getBeginLoc(), FPFeatures: FPOptionsOverride());
4775 // Increment for loop counter.
4776 UnaryOperator *Inc = UnaryOperator::Create(
4777 C, input: &IVRefExpr, opc: UO_PreInc, type: KmpInt32Ty, VK: VK_PRValue, OK: OK_Ordinary,
4778 l: S.getBeginLoc(), CanOverflow: true, FPFeatures: FPOptionsOverride());
4779 auto &&BodyGen = [CapturedStmt, CS, &S, &IV](CodeGenFunction &CGF) {
4780 // Iterate through all sections and emit a switch construct:
4781 // switch (IV) {
4782 // case 0:
4783 // <SectionStmt[0]>;
4784 // break;
4785 // ...
4786 // case <NumSection> - 1:
4787 // <SectionStmt[<NumSection> - 1]>;
4788 // break;
4789 // }
4790 // .omp.sections.exit:
4791 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(name: ".omp.sections.exit");
4792 llvm::SwitchInst *SwitchStmt =
4793 CGF.Builder.CreateSwitch(V: CGF.EmitLoadOfScalar(lvalue: IV, Loc: S.getBeginLoc()),
4794 Dest: ExitBB, NumCases: CS == nullptr ? 1 : CS->size());
4795 if (CS) {
4796 unsigned CaseNumber = 0;
4797 for (const Stmt *SubStmt : CS->children()) {
4798 auto CaseBB = CGF.createBasicBlock(name: ".omp.sections.case");
4799 CGF.EmitBlock(BB: CaseBB);
4800 SwitchStmt->addCase(OnVal: CGF.Builder.getInt32(C: CaseNumber), Dest: CaseBB);
4801 CGF.EmitStmt(S: SubStmt);
4802 CGF.EmitBranch(Block: ExitBB);
4803 ++CaseNumber;
4804 }
4805 } else {
4806 llvm::BasicBlock *CaseBB = CGF.createBasicBlock(name: ".omp.sections.case");
4807 CGF.EmitBlock(BB: CaseBB);
4808 SwitchStmt->addCase(OnVal: CGF.Builder.getInt32(C: 0), Dest: CaseBB);
4809 CGF.EmitStmt(S: CapturedStmt);
4810 CGF.EmitBranch(Block: ExitBB);
4811 }
4812 CGF.EmitBlock(BB: ExitBB, /*IsFinished=*/true);
4813 };
4814
4815 CodeGenFunction::OMPPrivateScope LoopScope(CGF);
4816 if (CGF.EmitOMPFirstprivateClause(D: S, PrivateScope&: LoopScope)) {
4817 // Emit implicit barrier to synchronize threads and avoid data races on
4818 // initialization of firstprivate variables and post-update of lastprivate
4819 // variables.
4820 CGF.CGM.getOpenMPRuntime().emitBarrierCall(
4821 CGF, Loc: S.getBeginLoc(), Kind: OMPD_unknown, /*EmitChecks=*/false,
4822 /*ForceSimpleCall=*/true);
4823 }
4824 CGF.EmitOMPPrivateClause(D: S, PrivateScope&: LoopScope);
4825 CGOpenMPRuntime::LastprivateConditionalRAII LPCRegion(CGF, S, IV);
4826 HasLastprivates = CGF.EmitOMPLastprivateClauseInit(D: S, PrivateScope&: LoopScope);
4827 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope&: LoopScope);
4828 (void)LoopScope.Privatize();
4829 if (isOpenMPTargetExecutionDirective(DKind: EKind))
4830 CGF.CGM.getOpenMPRuntime().adjustTargetSpecificDataForLambdas(CGF, D: S);
4831
4832 // Emit static non-chunked loop.
4833 OpenMPScheduleTy ScheduleKind;
4834 ScheduleKind.Schedule = OMPC_SCHEDULE_static;
4835 CGOpenMPRuntime::StaticRTInput StaticInit(
4836 /*IVSize=*/32, /*IVSigned=*/true, /*Ordered=*/false, IL.getAddress(),
4837 LB.getAddress(), UB.getAddress(), ST.getAddress());
4838 CGF.CGM.getOpenMPRuntime().emitForStaticInit(CGF, Loc: S.getBeginLoc(), DKind: EKind,
4839 ScheduleKind, Values: StaticInit);
4840 // UB = min(UB, GlobalUB);
4841 llvm::Value *UBVal = CGF.EmitLoadOfScalar(lvalue: UB, Loc: S.getBeginLoc());
4842 llvm::Value *MinUBGlobalUB = CGF.Builder.CreateSelect(
4843 C: CGF.Builder.CreateICmpSLT(LHS: UBVal, RHS: GlobalUBVal), True: UBVal, False: GlobalUBVal);
4844 CGF.EmitStoreOfScalar(value: MinUBGlobalUB, lvalue: UB);
4845 // IV = LB;
4846 CGF.EmitStoreOfScalar(value: CGF.EmitLoadOfScalar(lvalue: LB, Loc: S.getBeginLoc()), lvalue: IV);
4847 // while (idx <= UB) { BODY; ++idx; }
4848 CGF.EmitOMPInnerLoop(S, /*RequiresCleanup=*/false, LoopCond: Cond, IncExpr: Inc, BodyGen,
4849 PostIncGen: [](CodeGenFunction &) {});
4850 // Tell the runtime we are done.
4851 auto &&CodeGen = [&S](CodeGenFunction &CGF) {
4852 CGF.CGM.getOpenMPRuntime().emitForStaticFinish(CGF, Loc: S.getEndLoc(),
4853 DKind: OMPD_sections);
4854 };
4855 CGF.OMPCancelStack.emitExit(CGF, Kind: EKind, CodeGen);
4856 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_parallel);
4857 // Emit post-update of the reduction variables if IsLastIter != 0.
4858 emitPostUpdateForReductionClause(CGF, D: S, CondGen: [IL, &S](CodeGenFunction &CGF) {
4859 return CGF.Builder.CreateIsNotNull(
4860 Arg: CGF.EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc()));
4861 });
4862
4863 // Emit final copy of the lastprivate variables if IsLastIter != 0.
4864 if (HasLastprivates)
4865 CGF.EmitOMPLastprivateClauseFinal(
4866 D: S, /*NoFinals=*/false,
4867 IsLastIterCond: CGF.Builder.CreateIsNotNull(
4868 Arg: CGF.EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc())));
4869 };
4870
4871 bool HasCancel = false;
4872 if (auto *OSD = dyn_cast<OMPSectionsDirective>(Val: &S))
4873 HasCancel = OSD->hasCancel();
4874 else if (auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(Val: &S))
4875 HasCancel = OPSD->hasCancel();
4876 OMPCancelStackRAII CancelRegion(*this, EKind, HasCancel);
4877 CGM.getOpenMPRuntime().emitInlinedDirective(CGF&: *this, InnermostKind: OMPD_sections, CodeGen,
4878 HasCancel);
4879 // Emit barrier for lastprivates only if 'sections' directive has 'nowait'
4880 // clause. Otherwise the barrier will be generated by the codegen for the
4881 // directive.
4882 if (HasLastprivates && S.getSingleClause<OMPNowaitClause>()) {
4883 // Emit implicit barrier to synchronize threads and avoid data races on
4884 // initialization of firstprivate variables.
4885 CGM.getOpenMPRuntime().emitBarrierCall(CGF&: *this, Loc: S.getBeginLoc(),
4886 Kind: OMPD_unknown);
4887 }
4888}
4889
4890void CodeGenFunction::EmitOMPScopeDirective(const OMPScopeDirective &S) {
4891 {
4892 // Emit code for 'scope' region
4893 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
4894 Action.Enter(CGF);
4895 OMPPrivateScope PrivateScope(CGF);
4896 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
4897 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
4898 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
4899 (void)PrivateScope.Privatize();
4900 CGF.EmitStmt(S: S.getInnermostCapturedStmt()->getCapturedStmt());
4901 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_parallel);
4902 };
4903 auto LPCRegion =
4904 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
4905 OMPLexicalScope Scope(*this, S, OMPD_unknown);
4906 CGM.getOpenMPRuntime().emitInlinedDirective(CGF&: *this, InnermostKind: OMPD_scope, CodeGen);
4907 }
4908 // Emit an implicit barrier at the end.
4909 if (!S.getSingleClause<OMPNowaitClause>()) {
4910 CGM.getOpenMPRuntime().emitBarrierCall(CGF&: *this, Loc: S.getBeginLoc(), Kind: OMPD_scope);
4911 }
4912 // Check for outer lastprivate conditional update.
4913 checkForLastprivateConditionalUpdate(CGF&: *this, S);
4914}
4915
4916void CodeGenFunction::EmitOMPSectionsDirective(const OMPSectionsDirective &S) {
4917 if (CGM.getLangOpts().OpenMPIRBuilder) {
4918 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
4919 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
4920 using BodyGenCallbackTy = llvm::OpenMPIRBuilder::StorableBodyGenCallbackTy;
4921
4922 auto FiniCB = [](InsertPointTy IP) {
4923 // Don't FinalizeOMPRegion because this is done inside of OMPIRBuilder for
4924 // sections.
4925 return llvm::Error::success();
4926 };
4927
4928 const CapturedStmt *ICS = S.getInnermostCapturedStmt();
4929 const Stmt *CapturedStmt = S.getInnermostCapturedStmt()->getCapturedStmt();
4930 const auto *CS = dyn_cast<CompoundStmt>(Val: CapturedStmt);
4931 llvm::SmallVector<BodyGenCallbackTy, 4> SectionCBVector;
4932 if (CS) {
4933 for (const Stmt *SubStmt : CS->children()) {
4934 auto SectionCB = [this, SubStmt](
4935 InsertPointTy AllocIP, InsertPointTy CodeGenIP,
4936 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
4937 OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(CGF&: *this, RegionBodyStmt: SubStmt, AllocaIP: AllocIP,
4938 CodeGenIP, RegionName: "section");
4939 return llvm::Error::success();
4940 };
4941 SectionCBVector.push_back(Elt: SectionCB);
4942 }
4943 } else {
4944 auto SectionCB =
4945 [this, CapturedStmt](InsertPointTy AllocIP, InsertPointTy CodeGenIP,
4946 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
4947 OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(
4948 CGF&: *this, RegionBodyStmt: CapturedStmt, AllocaIP: AllocIP, CodeGenIP, RegionName: "section");
4949 return llvm::Error::success();
4950 };
4951 SectionCBVector.push_back(Elt: SectionCB);
4952 }
4953
4954 // Privatization callback that performs appropriate action for
4955 // shared/private/firstprivate/lastprivate/copyin/... variables.
4956 //
4957 // TODO: This defaults to shared right now.
4958 auto PrivCB = [](InsertPointTy AllocaIP, InsertPointTy CodeGenIP,
4959 llvm::Value &, llvm::Value &Val, llvm::Value *&ReplVal) {
4960 // The next line is appropriate only for variables (Val) with the
4961 // data-sharing attribute "shared".
4962 ReplVal = &Val;
4963
4964 return CodeGenIP;
4965 };
4966
4967 CGCapturedStmtInfo CGSI(*ICS, CR_OpenMP);
4968 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(*this, &CGSI);
4969 llvm::OpenMPIRBuilder::InsertPointTy AllocaIP(
4970 AllocaInsertPt->getIterator());
4971 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
4972 cantFail(ValOrErr: OMPBuilder.createSections(
4973 Loc: Builder, AllocaIP, SectionCBs: SectionCBVector, PrivCB, FiniCB, IsCancellable: S.hasCancel(),
4974 IsNowait: S.getSingleClause<OMPNowaitClause>()));
4975 Builder.restoreIP(IP: AfterIP);
4976 return;
4977 }
4978 {
4979 auto LPCRegion =
4980 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
4981 OMPLexicalScope Scope(*this, S, OMPD_unknown);
4982 EmitSections(S);
4983 }
4984 // Emit an implicit barrier at the end.
4985 if (!S.getSingleClause<OMPNowaitClause>()) {
4986 CGM.getOpenMPRuntime().emitBarrierCall(CGF&: *this, Loc: S.getBeginLoc(),
4987 Kind: OMPD_sections);
4988 }
4989 // Check for outer lastprivate conditional update.
4990 checkForLastprivateConditionalUpdate(CGF&: *this, S);
4991}
4992
4993void CodeGenFunction::EmitOMPSectionDirective(const OMPSectionDirective &S) {
4994 if (CGM.getLangOpts().OpenMPIRBuilder) {
4995 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
4996 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
4997
4998 const Stmt *SectionRegionBodyStmt = S.getAssociatedStmt();
4999 auto FiniCB = [this](InsertPointTy IP) {
5000 OMPBuilderCBHelpers::FinalizeOMPRegion(CGF&: *this, IP);
5001 return llvm::Error::success();
5002 };
5003
5004 auto BodyGenCB = [SectionRegionBodyStmt,
5005 this](InsertPointTy AllocIP, InsertPointTy CodeGenIP,
5006 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
5007 OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(
5008 CGF&: *this, RegionBodyStmt: SectionRegionBodyStmt, AllocaIP: AllocIP, CodeGenIP, RegionName: "section");
5009 return llvm::Error::success();
5010 };
5011
5012 LexicalScope Scope(*this, S.getSourceRange());
5013 EmitStopPoint(S: &S);
5014 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
5015 cantFail(ValOrErr: OMPBuilder.createSection(Loc: Builder, BodyGenCB, FiniCB));
5016 Builder.restoreIP(IP: AfterIP);
5017
5018 return;
5019 }
5020 LexicalScope Scope(*this, S.getSourceRange());
5021 EmitStopPoint(S: &S);
5022 EmitStmt(S: S.getAssociatedStmt());
5023}
5024
5025void CodeGenFunction::EmitOMPSingleDirective(const OMPSingleDirective &S) {
5026 llvm::SmallVector<const Expr *, 8> CopyprivateVars;
5027 llvm::SmallVector<const Expr *, 8> DestExprs;
5028 llvm::SmallVector<const Expr *, 8> SrcExprs;
5029 llvm::SmallVector<const Expr *, 8> AssignmentOps;
5030 // Check if there are any 'copyprivate' clauses associated with this
5031 // 'single' construct.
5032 // Build a list of copyprivate variables along with helper expressions
5033 // (<source>, <destination>, <destination>=<source> expressions)
5034 for (const auto *C : S.getClausesOfKind<OMPCopyprivateClause>()) {
5035 CopyprivateVars.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
5036 DestExprs.append(in_start: C->destination_exprs().begin(),
5037 in_end: C->destination_exprs().end());
5038 SrcExprs.append(in_start: C->source_exprs().begin(), in_end: C->source_exprs().end());
5039 AssignmentOps.append(in_start: C->assignment_ops().begin(),
5040 in_end: C->assignment_ops().end());
5041 }
5042 // Emit code for 'single' region along with 'copyprivate' clauses
5043 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5044 Action.Enter(CGF);
5045 OMPPrivateScope SingleScope(CGF);
5046 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope&: SingleScope);
5047 CGF.EmitOMPPrivateClause(D: S, PrivateScope&: SingleScope);
5048 (void)SingleScope.Privatize();
5049 CGF.EmitStmt(S: S.getInnermostCapturedStmt()->getCapturedStmt());
5050 };
5051 {
5052 auto LPCRegion =
5053 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
5054 OMPLexicalScope Scope(*this, S, OMPD_unknown);
5055 CGM.getOpenMPRuntime().emitSingleRegion(CGF&: *this, SingleOpGen: CodeGen, Loc: S.getBeginLoc(),
5056 CopyprivateVars, DestExprs,
5057 SrcExprs, AssignmentOps);
5058 }
5059 // Emit an implicit barrier at the end (to avoid data race on firstprivate
5060 // init or if no 'nowait' clause was specified and no 'copyprivate' clause).
5061 if (!S.getSingleClause<OMPNowaitClause>() && CopyprivateVars.empty()) {
5062 CGM.getOpenMPRuntime().emitBarrierCall(
5063 CGF&: *this, Loc: S.getBeginLoc(),
5064 Kind: S.getSingleClause<OMPNowaitClause>() ? OMPD_unknown : OMPD_single);
5065 }
5066 // Check for outer lastprivate conditional update.
5067 checkForLastprivateConditionalUpdate(CGF&: *this, S);
5068}
5069
5070static void emitMaster(CodeGenFunction &CGF, const OMPExecutableDirective &S) {
5071 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5072 Action.Enter(CGF);
5073 CGF.EmitStmt(S: S.getRawStmt());
5074 };
5075 CGF.CGM.getOpenMPRuntime().emitMasterRegion(CGF, MasterOpGen: CodeGen, Loc: S.getBeginLoc());
5076}
5077
5078void CodeGenFunction::EmitOMPMasterDirective(const OMPMasterDirective &S) {
5079 if (CGM.getLangOpts().OpenMPIRBuilder) {
5080 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
5081 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
5082
5083 const Stmt *MasterRegionBodyStmt = S.getAssociatedStmt();
5084
5085 auto FiniCB = [this](InsertPointTy IP) {
5086 OMPBuilderCBHelpers::FinalizeOMPRegion(CGF&: *this, IP);
5087 return llvm::Error::success();
5088 };
5089
5090 auto BodyGenCB = [MasterRegionBodyStmt,
5091 this](InsertPointTy AllocIP, InsertPointTy CodeGenIP,
5092 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
5093 OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(
5094 CGF&: *this, RegionBodyStmt: MasterRegionBodyStmt, AllocaIP: AllocIP, CodeGenIP, RegionName: "master");
5095 return llvm::Error::success();
5096 };
5097
5098 LexicalScope Scope(*this, S.getSourceRange());
5099 EmitStopPoint(S: &S);
5100 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
5101 cantFail(ValOrErr: OMPBuilder.createMaster(Loc: Builder, BodyGenCB, FiniCB));
5102 Builder.restoreIP(IP: AfterIP);
5103
5104 return;
5105 }
5106 LexicalScope Scope(*this, S.getSourceRange());
5107 EmitStopPoint(S: &S);
5108 emitMaster(CGF&: *this, S);
5109}
5110
5111static void emitMasked(CodeGenFunction &CGF, const OMPExecutableDirective &S) {
5112 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5113 Action.Enter(CGF);
5114 CGF.EmitStmt(S: S.getRawStmt());
5115 };
5116 Expr *Filter = nullptr;
5117 if (const auto *FilterClause = S.getSingleClause<OMPFilterClause>())
5118 Filter = FilterClause->getThreadID();
5119 CGF.CGM.getOpenMPRuntime().emitMaskedRegion(CGF, MaskedOpGen: CodeGen, Loc: S.getBeginLoc(),
5120 Filter);
5121}
5122
5123void CodeGenFunction::EmitOMPMaskedDirective(const OMPMaskedDirective &S) {
5124 if (CGM.getLangOpts().OpenMPIRBuilder) {
5125 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
5126 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
5127
5128 const Stmt *MaskedRegionBodyStmt = S.getAssociatedStmt();
5129 const Expr *Filter = nullptr;
5130 if (const auto *FilterClause = S.getSingleClause<OMPFilterClause>())
5131 Filter = FilterClause->getThreadID();
5132 llvm::Value *FilterVal = Filter
5133 ? EmitScalarExpr(E: Filter, IgnoreResultAssign: CGM.Int32Ty)
5134 : llvm::ConstantInt::get(Ty: CGM.Int32Ty, /*V=*/0);
5135
5136 auto FiniCB = [this](InsertPointTy IP) {
5137 OMPBuilderCBHelpers::FinalizeOMPRegion(CGF&: *this, IP);
5138 return llvm::Error::success();
5139 };
5140
5141 auto BodyGenCB = [MaskedRegionBodyStmt,
5142 this](InsertPointTy AllocIP, InsertPointTy CodeGenIP,
5143 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
5144 OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(
5145 CGF&: *this, RegionBodyStmt: MaskedRegionBodyStmt, AllocaIP: AllocIP, CodeGenIP, RegionName: "masked");
5146 return llvm::Error::success();
5147 };
5148
5149 LexicalScope Scope(*this, S.getSourceRange());
5150 EmitStopPoint(S: &S);
5151 llvm::OpenMPIRBuilder::InsertPointTy AfterIP = cantFail(
5152 ValOrErr: OMPBuilder.createMasked(Loc: Builder, BodyGenCB, FiniCB, Filter: FilterVal));
5153 Builder.restoreIP(IP: AfterIP);
5154
5155 return;
5156 }
5157 LexicalScope Scope(*this, S.getSourceRange());
5158 EmitStopPoint(S: &S);
5159 emitMasked(CGF&: *this, S);
5160}
5161
5162void CodeGenFunction::EmitOMPCriticalDirective(const OMPCriticalDirective &S) {
5163 if (CGM.getLangOpts().OpenMPIRBuilder) {
5164 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
5165 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
5166
5167 const Stmt *CriticalRegionBodyStmt = S.getAssociatedStmt();
5168 const Expr *Hint = nullptr;
5169 if (const auto *HintClause = S.getSingleClause<OMPHintClause>())
5170 Hint = HintClause->getHint();
5171
5172 // TODO: This is slightly different from what's currently being done in
5173 // clang. Fix the Int32Ty to IntPtrTy (pointer width size) when everything
5174 // about typing is final.
5175 llvm::Value *HintInst = nullptr;
5176 if (Hint)
5177 HintInst =
5178 Builder.CreateIntCast(V: EmitScalarExpr(E: Hint), DestTy: CGM.Int32Ty, isSigned: false);
5179
5180 auto FiniCB = [this](InsertPointTy IP) {
5181 OMPBuilderCBHelpers::FinalizeOMPRegion(CGF&: *this, IP);
5182 return llvm::Error::success();
5183 };
5184
5185 auto BodyGenCB = [CriticalRegionBodyStmt,
5186 this](InsertPointTy AllocIP, InsertPointTy CodeGenIP,
5187 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
5188 OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(
5189 CGF&: *this, RegionBodyStmt: CriticalRegionBodyStmt, AllocaIP: AllocIP, CodeGenIP, RegionName: "critical");
5190 return llvm::Error::success();
5191 };
5192
5193 LexicalScope Scope(*this, S.getSourceRange());
5194 EmitStopPoint(S: &S);
5195 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
5196 cantFail(ValOrErr: OMPBuilder.createCritical(Loc: Builder, BodyGenCB, FiniCB,
5197 CriticalName: S.getDirectiveName().getAsString(),
5198 HintInst));
5199 Builder.restoreIP(IP: AfterIP);
5200
5201 return;
5202 }
5203
5204 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5205 Action.Enter(CGF);
5206 CGF.EmitStmt(S: S.getAssociatedStmt());
5207 };
5208 const Expr *Hint = nullptr;
5209 if (const auto *HintClause = S.getSingleClause<OMPHintClause>())
5210 Hint = HintClause->getHint();
5211 LexicalScope Scope(*this, S.getSourceRange());
5212 EmitStopPoint(S: &S);
5213 CGM.getOpenMPRuntime().emitCriticalRegion(CGF&: *this,
5214 CriticalName: S.getDirectiveName().getAsString(),
5215 CriticalOpGen: CodeGen, Loc: S.getBeginLoc(), Hint);
5216}
5217
5218void CodeGenFunction::EmitOMPParallelForDirective(
5219 const OMPParallelForDirective &S) {
5220 // Emit directive as a combined directive that consists of two implicit
5221 // directives: 'parallel' with 'for' directive.
5222 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5223 Action.Enter(CGF);
5224 emitOMPCopyinClause(CGF, S);
5225 (void)emitWorksharingDirective(CGF, S, HasCancel: S.hasCancel());
5226 };
5227 {
5228 const auto &&NumIteratorsGen = [&S](CodeGenFunction &CGF) {
5229 CodeGenFunction::OMPLocalDeclMapRAII Scope(CGF);
5230 CGCapturedStmtInfo CGSI(CR_OpenMP);
5231 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGSI);
5232 OMPLoopScope LoopScope(CGF, S);
5233 return CGF.EmitScalarExpr(E: S.getNumIterations());
5234 };
5235 bool IsInscan = llvm::any_of(Range: S.getClausesOfKind<OMPReductionClause>(),
5236 P: [](const OMPReductionClause *C) {
5237 return C->getModifier() == OMPC_REDUCTION_inscan;
5238 });
5239 if (IsInscan)
5240 emitScanBasedDirectiveDecls(CGF&: *this, S, NumIteratorsGen);
5241 auto LPCRegion =
5242 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
5243 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_for, CodeGen,
5244 CodeGenBoundParameters: emitEmptyBoundParameters);
5245 if (IsInscan)
5246 emitScanBasedDirectiveFinals(CGF&: *this, S, NumIteratorsGen);
5247 }
5248 // Check for outer lastprivate conditional update.
5249 checkForLastprivateConditionalUpdate(CGF&: *this, S);
5250}
5251
5252void CodeGenFunction::EmitOMPParallelForSimdDirective(
5253 const OMPParallelForSimdDirective &S) {
5254 // Emit directive as a combined directive that consists of two implicit
5255 // directives: 'parallel' with 'for' directive.
5256 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5257 Action.Enter(CGF);
5258 emitOMPCopyinClause(CGF, S);
5259 (void)emitWorksharingDirective(CGF, S, /*HasCancel=*/false);
5260 };
5261 {
5262 const auto &&NumIteratorsGen = [&S](CodeGenFunction &CGF) {
5263 CodeGenFunction::OMPLocalDeclMapRAII Scope(CGF);
5264 CGCapturedStmtInfo CGSI(CR_OpenMP);
5265 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGSI);
5266 OMPLoopScope LoopScope(CGF, S);
5267 return CGF.EmitScalarExpr(E: S.getNumIterations());
5268 };
5269 bool IsInscan = llvm::any_of(Range: S.getClausesOfKind<OMPReductionClause>(),
5270 P: [](const OMPReductionClause *C) {
5271 return C->getModifier() == OMPC_REDUCTION_inscan;
5272 });
5273 if (IsInscan)
5274 emitScanBasedDirectiveDecls(CGF&: *this, S, NumIteratorsGen);
5275 auto LPCRegion =
5276 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
5277 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_for_simd, CodeGen,
5278 CodeGenBoundParameters: emitEmptyBoundParameters);
5279 if (IsInscan)
5280 emitScanBasedDirectiveFinals(CGF&: *this, S, NumIteratorsGen);
5281 }
5282 // Check for outer lastprivate conditional update.
5283 checkForLastprivateConditionalUpdate(CGF&: *this, S);
5284}
5285
5286void CodeGenFunction::EmitOMPParallelMasterDirective(
5287 const OMPParallelMasterDirective &S) {
5288 // Emit directive as a combined directive that consists of two implicit
5289 // directives: 'parallel' with 'master' directive.
5290 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5291 Action.Enter(CGF);
5292 OMPPrivateScope PrivateScope(CGF);
5293 emitOMPCopyinClause(CGF, S);
5294 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
5295 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
5296 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
5297 (void)PrivateScope.Privatize();
5298 emitMaster(CGF, S);
5299 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_parallel);
5300 };
5301 {
5302 auto LPCRegion =
5303 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
5304 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_master, CodeGen,
5305 CodeGenBoundParameters: emitEmptyBoundParameters);
5306 emitPostUpdateForReductionClause(CGF&: *this, D: S,
5307 CondGen: [](CodeGenFunction &) { return nullptr; });
5308 }
5309 // Check for outer lastprivate conditional update.
5310 checkForLastprivateConditionalUpdate(CGF&: *this, S);
5311}
5312
5313void CodeGenFunction::EmitOMPParallelMaskedDirective(
5314 const OMPParallelMaskedDirective &S) {
5315 // Emit directive as a combined directive that consists of two implicit
5316 // directives: 'parallel' with 'masked' directive.
5317 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5318 Action.Enter(CGF);
5319 OMPPrivateScope PrivateScope(CGF);
5320 emitOMPCopyinClause(CGF, S);
5321 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
5322 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
5323 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
5324 (void)PrivateScope.Privatize();
5325 emitMasked(CGF, S);
5326 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_parallel);
5327 };
5328 {
5329 auto LPCRegion =
5330 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
5331 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_masked, CodeGen,
5332 CodeGenBoundParameters: emitEmptyBoundParameters);
5333 emitPostUpdateForReductionClause(CGF&: *this, D: S,
5334 CondGen: [](CodeGenFunction &) { return nullptr; });
5335 }
5336 // Check for outer lastprivate conditional update.
5337 checkForLastprivateConditionalUpdate(CGF&: *this, S);
5338}
5339
5340void CodeGenFunction::EmitOMPParallelSectionsDirective(
5341 const OMPParallelSectionsDirective &S) {
5342 // Emit directive as a combined directive that consists of two implicit
5343 // directives: 'parallel' with 'sections' directive.
5344 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
5345 Action.Enter(CGF);
5346 emitOMPCopyinClause(CGF, S);
5347 CGF.EmitSections(S);
5348 };
5349 {
5350 auto LPCRegion =
5351 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
5352 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_sections, CodeGen,
5353 CodeGenBoundParameters: emitEmptyBoundParameters);
5354 }
5355 // Check for outer lastprivate conditional update.
5356 checkForLastprivateConditionalUpdate(CGF&: *this, S);
5357}
5358
5359namespace {
5360/// Get the list of variables declared in the context of the untied tasks.
5361class CheckVarsEscapingUntiedTaskDeclContext final
5362 : public ConstStmtVisitor<CheckVarsEscapingUntiedTaskDeclContext> {
5363 llvm::SmallVector<const VarDecl *, 4> PrivateDecls;
5364
5365public:
5366 explicit CheckVarsEscapingUntiedTaskDeclContext() = default;
5367 ~CheckVarsEscapingUntiedTaskDeclContext() = default;
5368 void VisitDeclStmt(const DeclStmt *S) {
5369 if (!S)
5370 return;
5371 // Need to privatize only local vars, static locals can be processed as is.
5372 for (const Decl *D : S->decls()) {
5373 if (const auto *VD = dyn_cast_or_null<VarDecl>(Val: D))
5374 if (VD->hasLocalStorage())
5375 PrivateDecls.push_back(Elt: VD);
5376 }
5377 }
5378 void VisitOMPExecutableDirective(const OMPExecutableDirective *) {}
5379 void VisitCapturedStmt(const CapturedStmt *) {}
5380 void VisitLambdaExpr(const LambdaExpr *) {}
5381 void VisitBlockExpr(const BlockExpr *) {}
5382 void VisitStmt(const Stmt *S) {
5383 if (!S)
5384 return;
5385 for (const Stmt *Child : S->children())
5386 if (Child)
5387 Visit(S: Child);
5388 }
5389
5390 /// Swaps list of vars with the provided one.
5391 ArrayRef<const VarDecl *> getPrivateDecls() const { return PrivateDecls; }
5392};
5393} // anonymous namespace
5394
5395static void buildDependences(const OMPExecutableDirective &S,
5396 OMPTaskDataTy &Data) {
5397
5398 // First look for 'omp_all_memory' and add this first.
5399 bool OmpAllMemory = false;
5400 if (llvm::any_of(
5401 Range: S.getClausesOfKind<OMPDependClause>(), P: [](const OMPDependClause *C) {
5402 return C->getDependencyKind() == OMPC_DEPEND_outallmemory ||
5403 C->getDependencyKind() == OMPC_DEPEND_inoutallmemory;
5404 })) {
5405 OmpAllMemory = true;
5406 // Since both OMPC_DEPEND_outallmemory and OMPC_DEPEND_inoutallmemory are
5407 // equivalent to the runtime, always use OMPC_DEPEND_outallmemory to
5408 // simplify.
5409 OMPTaskDataTy::DependData &DD =
5410 Data.Dependences.emplace_back(Args: OMPC_DEPEND_outallmemory,
5411 /*IteratorExpr=*/Args: nullptr);
5412 // Add a nullptr Expr to simplify the codegen in emitDependData.
5413 DD.DepExprs.push_back(Elt: nullptr);
5414 }
5415 // Add remaining dependences skipping any 'out' or 'inout' if they are
5416 // overridden by 'omp_all_memory'.
5417 for (const auto *C : S.getClausesOfKind<OMPDependClause>()) {
5418 OpenMPDependClauseKind Kind = C->getDependencyKind();
5419 if (Kind == OMPC_DEPEND_outallmemory || Kind == OMPC_DEPEND_inoutallmemory)
5420 continue;
5421 if (OmpAllMemory && (Kind == OMPC_DEPEND_out || Kind == OMPC_DEPEND_inout))
5422 continue;
5423 OMPTaskDataTy::DependData &DD =
5424 Data.Dependences.emplace_back(Args: C->getDependencyKind(), Args: C->getModifier());
5425 DD.DepExprs.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
5426 }
5427}
5428
5429void CodeGenFunction::EmitOMPTaskBasedDirective(
5430 const OMPExecutableDirective &S, const OpenMPDirectiveKind CapturedRegion,
5431 const RegionCodeGenTy &BodyGen, const TaskGenTy &TaskGen,
5432 OMPTaskDataTy &Data) {
5433 // Emit outlined function for task construct.
5434 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: CapturedRegion);
5435 auto I = CS->getCapturedDecl()->param_begin();
5436 auto PartId = std::next(x: I);
5437 auto TaskT = std::next(x: I, n: 4);
5438 // Check if the task is final
5439 if (const auto *Clause = S.getSingleClause<OMPFinalClause>()) {
5440 // If the condition constant folds and can be elided, try to avoid emitting
5441 // the condition and the dead arm of the if/else.
5442 const Expr *Cond = Clause->getCondition();
5443 bool CondConstant;
5444 if (ConstantFoldsToSimpleInteger(Cond, Result&: CondConstant))
5445 Data.Final.setInt(CondConstant);
5446 else
5447 Data.Final.setPointer(EvaluateExprAsBool(E: Cond));
5448 } else {
5449 // By default the task is not final.
5450 Data.Final.setInt(/*IntVal=*/false);
5451 }
5452 // Check if the task has 'priority' clause.
5453 if (const auto *Clause = S.getSingleClause<OMPPriorityClause>()) {
5454 const Expr *Prio = Clause->getPriority();
5455 Data.Priority.setInt(/*IntVal=*/true);
5456 Data.Priority.setPointer(EmitScalarConversion(
5457 Src: EmitScalarExpr(E: Prio), SrcTy: Prio->getType(),
5458 DstTy: getContext().getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1),
5459 Loc: Prio->getExprLoc()));
5460 }
5461 // The first function argument for tasks is a thread id, the second one is a
5462 // part id (0 for tied tasks, >=0 for untied task).
5463 llvm::DenseSet<const ValueDecl *> EmittedAsPrivate;
5464 // Get list of private variables.
5465 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) {
5466 auto IRef = C->varlist_begin();
5467 for (const Expr *IInit : C->private_copies()) {
5468 const auto *OrigDecl = cast<DeclRefExpr>(Val: *IRef)->getDecl();
5469 if (EmittedAsPrivate.insert(V: cast<ValueDecl>(Val: OrigDecl->getCanonicalDecl()))
5470 .second) {
5471 Data.PrivateVars.push_back(Elt: *IRef);
5472 Data.PrivateCopies.push_back(Elt: IInit);
5473 }
5474 ++IRef;
5475 }
5476 }
5477 EmittedAsPrivate.clear();
5478 // Get list of firstprivate variables.
5479 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
5480 auto IRef = C->varlist_begin();
5481 auto IElemInitRef = C->inits().begin();
5482 for (const Expr *IInit : C->private_copies()) {
5483 const auto *OrigDecl = cast<DeclRefExpr>(Val: *IRef)->getDecl();
5484 if (EmittedAsPrivate.insert(V: cast<ValueDecl>(Val: OrigDecl->getCanonicalDecl()))
5485 .second) {
5486 Data.FirstprivateVars.push_back(Elt: *IRef);
5487 Data.FirstprivateCopies.push_back(Elt: IInit);
5488 Data.FirstprivateInits.push_back(Elt: *IElemInitRef);
5489 }
5490 ++IRef;
5491 ++IElemInitRef;
5492 }
5493 }
5494 // Get list of lastprivate variables (for taskloops).
5495 llvm::MapVector<const ValueDecl *, const DeclRefExpr *> LastprivateDstsOrigs;
5496 llvm::MapVector<const ValueDecl *, const DeclRefExpr *> LastprivateSrcsOrigs;
5497 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
5498 auto IRef = C->varlist_begin();
5499 auto ID = C->destination_exprs().begin();
5500 auto IS = C->source_exprs().begin();
5501 for (const Expr *IInit : C->private_copies()) {
5502 const auto *OrigDecl = cast<DeclRefExpr>(Val: *IRef)->getDecl();
5503 if (EmittedAsPrivate.insert(V: cast<ValueDecl>(Val: OrigDecl->getCanonicalDecl()))
5504 .second) {
5505 Data.LastprivateVars.push_back(Elt: *IRef);
5506 Data.LastprivateCopies.push_back(Elt: IInit);
5507 }
5508 LastprivateDstsOrigs.insert(KV: std::make_pair(
5509 x: cast<DeclRefExpr>(Val: *ID)->getDecl(), y: cast<DeclRefExpr>(Val: *IRef)));
5510 LastprivateSrcsOrigs.insert(KV: std::make_pair(
5511 x: cast<DeclRefExpr>(Val: *IS)->getDecl(), y: cast<DeclRefExpr>(Val: *IRef)));
5512 ++IRef;
5513 ++ID;
5514 ++IS;
5515 }
5516 }
5517 SmallVector<const Expr *, 4> LHSs;
5518 SmallVector<const Expr *, 4> RHSs;
5519 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
5520 Data.ReductionVars.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
5521 Data.ReductionOrigs.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
5522 Data.ReductionCopies.append(in_start: C->privates().begin(), in_end: C->privates().end());
5523 Data.ReductionOps.append(in_start: C->reduction_ops().begin(),
5524 in_end: C->reduction_ops().end());
5525 LHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
5526 RHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
5527 }
5528 Data.Reductions = CGM.getOpenMPRuntime().emitTaskReductionInit(
5529 CGF&: *this, Loc: S.getBeginLoc(), LHSExprs: LHSs, RHSExprs: RHSs, Data);
5530 // Build list of dependences.
5531 buildDependences(S, Data);
5532 // Get list of local vars for untied tasks.
5533 if (!Data.Tied) {
5534 CheckVarsEscapingUntiedTaskDeclContext Checker;
5535 Checker.Visit(S: S.getInnermostCapturedStmt()->getCapturedStmt());
5536 Data.PrivateLocals.append(in_start: Checker.getPrivateDecls().begin(),
5537 in_end: Checker.getPrivateDecls().end());
5538 }
5539 auto &&CodeGen = [&Data, &S, CS, &BodyGen, &LastprivateDstsOrigs,
5540 &LastprivateSrcsOrigs, CapturedRegion](
5541 CodeGenFunction &CGF, PrePostActionTy &Action) {
5542 llvm::MapVector<CanonicalDeclPtr<const VarDecl>,
5543 std::pair<Address, Address>>
5544 UntiedLocalVars;
5545 // Set proper addresses for generated private copies.
5546 OMPPrivateScope Scope(CGF);
5547 // Generate debug info for variables present in shared clause.
5548 if (auto *DI = CGF.getDebugInfo()) {
5549 llvm::SmallDenseMap<const VarDecl *, FieldDecl *> CaptureFields =
5550 CGF.CapturedStmtInfo->getCaptureFields();
5551 llvm::Value *ContextValue = CGF.CapturedStmtInfo->getContextValue();
5552 if (CaptureFields.size() && ContextValue) {
5553 unsigned CharWidth = CGF.getContext().getCharWidth();
5554 // The shared variables are packed together as members of structure.
5555 // So the address of each shared variable can be computed by adding
5556 // offset of it (within record) to the base address of record. For each
5557 // shared variable, debug intrinsic llvm.dbg.declare is generated with
5558 // appropriate expressions (DIExpression).
5559 // Ex:
5560 // %12 = load %struct.anon*, %struct.anon** %__context.addr.i
5561 // call void @llvm.dbg.declare(metadata %struct.anon* %12,
5562 // metadata !svar1,
5563 // metadata !DIExpression(DW_OP_deref))
5564 // call void @llvm.dbg.declare(metadata %struct.anon* %12,
5565 // metadata !svar2,
5566 // metadata !DIExpression(DW_OP_plus_uconst, 8, DW_OP_deref))
5567 for (auto It = CaptureFields.begin(); It != CaptureFields.end(); ++It) {
5568 const VarDecl *SharedVar = It->first;
5569 RecordDecl *CaptureRecord = It->second->getParent();
5570 const ASTRecordLayout &Layout =
5571 CGF.getContext().getASTRecordLayout(D: CaptureRecord);
5572 unsigned Offset =
5573 Layout.getFieldOffset(FieldNo: It->second->getFieldIndex()) / CharWidth;
5574 if (CGF.CGM.getCodeGenOpts().hasReducedDebugInfo())
5575 (void)DI->EmitDeclareOfAutoVariable(Decl: SharedVar, AI: ContextValue,
5576 Builder&: CGF.Builder, UsePointerValue: false);
5577 // Get the call dbg.declare instruction we just created and update
5578 // its DIExpression to add offset to base address.
5579 auto UpdateExpr = [](llvm::LLVMContext &Ctx, auto *Declare,
5580 unsigned Offset) {
5581 SmallVector<uint64_t, 8> Ops;
5582 // Add offset to the base address if non zero.
5583 if (Offset) {
5584 Ops.push_back(Elt: llvm::dwarf::DW_OP_plus_uconst);
5585 Ops.push_back(Elt: Offset);
5586 }
5587 Ops.push_back(Elt: llvm::dwarf::DW_OP_deref);
5588 Declare->setExpression(llvm::DIExpression::get(Context&: Ctx, Elements: Ops));
5589 };
5590 llvm::Instruction &Last = CGF.Builder.GetInsertBlock()->back();
5591 if (auto DDI = dyn_cast<llvm::DbgVariableIntrinsic>(Val: &Last))
5592 UpdateExpr(DDI->getContext(), DDI, Offset);
5593 // If we're emitting using the new debug info format into a block
5594 // without a terminator, the record will be "trailing".
5595 assert(!Last.isTerminator() && "unexpected terminator");
5596 if (auto *Marker =
5597 CGF.Builder.GetInsertBlock()->getTrailingDbgRecords()) {
5598 for (llvm::DbgVariableRecord &DVR : llvm::reverse(
5599 C: llvm::filterDbgVars(R: Marker->getDbgRecordRange()))) {
5600 UpdateExpr(Last.getContext(), &DVR, Offset);
5601 break;
5602 }
5603 }
5604 }
5605 }
5606 }
5607 llvm::SmallVector<std::pair<const ValueDecl *, Address>, 16>
5608 FirstprivatePtrs;
5609 if (!Data.PrivateVars.empty() || !Data.FirstprivateVars.empty() ||
5610 !Data.LastprivateVars.empty() || !Data.PrivateLocals.empty()) {
5611 enum { PrivatesParam = 2, CopyFnParam = 3 };
5612 llvm::Value *CopyFn = CGF.Builder.CreateLoad(
5613 Addr: CGF.GetAddrOfLocalVar(VD: CS->getCapturedDecl()->getParam(i: CopyFnParam)));
5614 llvm::Value *PrivatesPtr = CGF.Builder.CreateLoad(Addr: CGF.GetAddrOfLocalVar(
5615 VD: CS->getCapturedDecl()->getParam(i: PrivatesParam)));
5616 // Map privates.
5617 llvm::SmallVector<std::pair<const ValueDecl *, Address>, 16> PrivatePtrs;
5618 llvm::SmallVector<llvm::Value *, 16> CallArgs;
5619 llvm::SmallVector<llvm::Type *, 4> ParamTypes;
5620 CallArgs.push_back(Elt: PrivatesPtr);
5621 ParamTypes.push_back(Elt: PrivatesPtr->getType());
5622 for (const Expr *E : Data.PrivateVars) {
5623 const auto *VD = cast<DeclRefExpr>(Val: E)->getDecl();
5624 RawAddress PrivatePtr = CGF.CreateMemTempWithoutCast(
5625 T: CGF.getContext().getPointerType(T: E->getType()), Name: ".priv.ptr.addr");
5626 PrivatePtrs.emplace_back(Args&: VD, Args&: PrivatePtr);
5627 CallArgs.push_back(Elt: PrivatePtr.getPointer());
5628 ParamTypes.push_back(Elt: PrivatePtr.getType());
5629 }
5630 for (const Expr *E : Data.FirstprivateVars) {
5631 const auto *VD = cast<DeclRefExpr>(Val: E)->getDecl();
5632 RawAddress PrivatePtr = CGF.CreateMemTempWithoutCast(
5633 T: CGF.getContext().getPointerType(T: E->getType()),
5634 Name: ".firstpriv.ptr.addr");
5635 PrivatePtrs.emplace_back(Args&: VD, Args&: PrivatePtr);
5636 FirstprivatePtrs.emplace_back(Args&: VD, Args&: PrivatePtr);
5637 CallArgs.push_back(Elt: PrivatePtr.getPointer());
5638 ParamTypes.push_back(Elt: PrivatePtr.getType());
5639 }
5640 for (const Expr *E : Data.LastprivateVars) {
5641 const auto *VD = cast<DeclRefExpr>(Val: E)->getDecl();
5642 RawAddress PrivatePtr = CGF.CreateMemTempWithoutCast(
5643 T: CGF.getContext().getPointerType(T: E->getType()),
5644 Name: ".lastpriv.ptr.addr");
5645 PrivatePtrs.emplace_back(Args&: VD, Args&: PrivatePtr);
5646 CallArgs.push_back(Elt: PrivatePtr.getPointer());
5647 ParamTypes.push_back(Elt: PrivatePtr.getType());
5648 }
5649 for (const VarDecl *VD : Data.PrivateLocals) {
5650 QualType Ty = VD->getType().getNonReferenceType();
5651 if (VD->getType()->isLValueReferenceType())
5652 Ty = CGF.getContext().getPointerType(T: Ty);
5653 if (isAllocatableDecl(VD))
5654 Ty = CGF.getContext().getPointerType(T: Ty);
5655 RawAddress PrivatePtr = CGF.CreateMemTempWithoutCast(
5656 T: CGF.getContext().getPointerType(T: Ty), Name: ".local.ptr.addr");
5657 auto Result = UntiedLocalVars.insert(
5658 KV: std::make_pair(x&: VD, y: std::make_pair(x&: PrivatePtr, y: Address::invalid())));
5659 // If key exists update in place.
5660 if (Result.second == false)
5661 *Result.first = std::make_pair(
5662 x&: VD, y: std::make_pair(x&: PrivatePtr, y: Address::invalid()));
5663 CallArgs.push_back(Elt: PrivatePtr.getPointer());
5664 ParamTypes.push_back(Elt: PrivatePtr.getType());
5665 }
5666 auto *CopyFnTy = llvm::FunctionType::get(Result: CGF.Builder.getVoidTy(),
5667 Params: ParamTypes, /*isVarArg=*/false);
5668 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(
5669 CGF, Loc: S.getBeginLoc(), OutlinedFn: {CopyFnTy, CopyFn}, Args: CallArgs);
5670 for (const auto &Pair : LastprivateDstsOrigs) {
5671 const auto *OrigDecl = Pair.second->getDecl();
5672 if (const auto *BD = dyn_cast<BindingDecl>(Val: OrigDecl)) {
5673 // For BindingDecls, emit the binding's LValue directly.
5674 Address OrigAddr =
5675 CGF.EmitOMPBindingOriginalAddr(BD, Loc: Pair.second->getExprLoc());
5676 Scope.addPrivate(LocalVD: Pair.first, Addr: OrigAddr);
5677 } else {
5678 const auto *OrigVD = cast<VarDecl>(Val: OrigDecl);
5679 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(OrigVD),
5680 /*RefersToEnclosingVariableOrCapture=*/
5681 CGF.CapturedStmtInfo->lookup(VD: OrigVD) != nullptr,
5682 Pair.second->getType(), VK_LValue,
5683 Pair.second->getExprLoc());
5684 Scope.addPrivate(LocalVD: Pair.first, Addr: CGF.EmitLValue(E: &DRE).getAddress());
5685 }
5686 }
5687 for (const auto &Pair : PrivatePtrs) {
5688 Address Replacement = Address(
5689 CGF.Builder.CreateLoad(Addr: Pair.second),
5690 CGF.ConvertTypeForMem(T: Pair.first->getType().getNonReferenceType()),
5691 CGF.getContext().getDeclAlign(D: Pair.first));
5692 Scope.addPrivate(LocalVD: Pair.first, Addr: Replacement);
5693
5694 // For BindingDecls with lastprivate, also map the .lastprivate.src
5695 // pseudo-variable to the same private address.
5696 if (isa<BindingDecl>(Val: Pair.first)) {
5697 for (const auto &SrcPair : LastprivateSrcsOrigs) {
5698 if (SrcPair.second->getDecl() == Pair.first) {
5699 Scope.addPrivate(LocalVD: SrcPair.first, Addr: Replacement);
5700 break;
5701 }
5702 }
5703 }
5704
5705 if (auto *DI = CGF.getDebugInfo())
5706 if (CGF.CGM.getCodeGenOpts().hasReducedDebugInfo())
5707 // Only emit debug info for VarDecls, not BindingDecls.
5708 if (const auto *VD = dyn_cast<VarDecl>(Val: Pair.first))
5709 (void)DI->EmitDeclareOfAutoVariable(
5710 Decl: VD, AI: Pair.second.getBasePointer(), Builder&: CGF.Builder,
5711 /*UsePointerValue*/ true);
5712 }
5713 // Adjust mapping for internal locals by mapping actual memory instead of
5714 // a pointer to this memory.
5715 for (auto &Pair : UntiedLocalVars) {
5716 QualType VDType = Pair.first->getType().getNonReferenceType();
5717 if (Pair.first->getType()->isLValueReferenceType())
5718 VDType = CGF.getContext().getPointerType(T: VDType);
5719 if (isAllocatableDecl(VD: Pair.first)) {
5720 llvm::Value *Ptr = CGF.Builder.CreateLoad(Addr: Pair.second.first);
5721 Address Replacement(
5722 Ptr,
5723 CGF.ConvertTypeForMem(T: CGF.getContext().getPointerType(T: VDType)),
5724 CGF.getPointerAlign());
5725 Pair.second.first = Replacement;
5726 Ptr = CGF.Builder.CreateLoad(Addr: Replacement);
5727 Replacement = Address(Ptr, CGF.ConvertTypeForMem(T: VDType),
5728 CGF.getContext().getDeclAlign(D: Pair.first));
5729 Pair.second.second = Replacement;
5730 } else {
5731 llvm::Value *Ptr = CGF.Builder.CreateLoad(Addr: Pair.second.first);
5732 Address Replacement(Ptr, CGF.ConvertTypeForMem(T: VDType),
5733 CGF.getContext().getDeclAlign(D: Pair.first));
5734 Pair.second.first = Replacement;
5735 }
5736 }
5737 }
5738 if (Data.Reductions) {
5739 OMPPrivateScope FirstprivateScope(CGF);
5740 for (const auto &Pair : FirstprivatePtrs) {
5741 Address Replacement(
5742 CGF.Builder.CreateLoad(Addr: Pair.second),
5743 CGF.ConvertTypeForMem(T: Pair.first->getType().getNonReferenceType()),
5744 CGF.getContext().getDeclAlign(D: Pair.first));
5745 FirstprivateScope.addPrivate(LocalVD: Pair.first, Addr: Replacement);
5746 }
5747 (void)FirstprivateScope.Privatize();
5748 OMPLexicalScope LexScope(CGF, S, CapturedRegion);
5749 ReductionCodeGen RedCG(Data.ReductionVars, Data.ReductionVars,
5750 Data.ReductionCopies, Data.ReductionOps);
5751 llvm::Value *ReductionsPtr = CGF.Builder.CreateLoad(
5752 Addr: CGF.GetAddrOfLocalVar(VD: CS->getCapturedDecl()->getParam(i: 9)));
5753 for (unsigned Cnt = 0, E = Data.ReductionVars.size(); Cnt < E; ++Cnt) {
5754 RedCG.emitSharedOrigLValue(CGF, N: Cnt);
5755 RedCG.emitAggregateType(CGF, N: Cnt);
5756 // FIXME: This must removed once the runtime library is fixed.
5757 // Emit required threadprivate variables for
5758 // initializer/combiner/finalizer.
5759 CGF.CGM.getOpenMPRuntime().emitTaskReductionFixups(CGF, Loc: S.getBeginLoc(),
5760 RCG&: RedCG, N: Cnt);
5761 Address Replacement = CGF.CGM.getOpenMPRuntime().getTaskReductionItem(
5762 CGF, Loc: S.getBeginLoc(), ReductionsPtr, SharedLVal: RedCG.getSharedLValue(N: Cnt));
5763 Replacement = Address(
5764 CGF.EmitScalarConversion(Src: Replacement.emitRawPointer(CGF),
5765 SrcTy: CGF.getContext().VoidPtrTy,
5766 DstTy: CGF.getContext().getPointerType(
5767 T: Data.ReductionCopies[Cnt]->getType()),
5768 Loc: Data.ReductionCopies[Cnt]->getExprLoc()),
5769 CGF.ConvertTypeForMem(T: Data.ReductionCopies[Cnt]->getType()),
5770 Replacement.getAlignment());
5771 Replacement = RedCG.adjustPrivateAddress(CGF, N: Cnt, PrivateAddr: Replacement);
5772 Scope.addPrivate(LocalVD: RedCG.getBaseDecl(N: Cnt), Addr: Replacement);
5773 }
5774 }
5775 // Privatize all private variables except for in_reduction items.
5776 (void)Scope.Privatize();
5777 SmallVector<const Expr *, 4> InRedVars;
5778 SmallVector<const Expr *, 4> InRedPrivs;
5779 SmallVector<const Expr *, 4> InRedOps;
5780 SmallVector<const Expr *, 4> TaskgroupDescriptors;
5781 for (const auto *C : S.getClausesOfKind<OMPInReductionClause>()) {
5782 auto IPriv = C->privates().begin();
5783 auto IRed = C->reduction_ops().begin();
5784 auto ITD = C->taskgroup_descriptors().begin();
5785 for (const Expr *Ref : C->varlist()) {
5786 InRedVars.emplace_back(Args&: Ref);
5787 InRedPrivs.emplace_back(Args: *IPriv);
5788 InRedOps.emplace_back(Args: *IRed);
5789 TaskgroupDescriptors.emplace_back(Args: *ITD);
5790 std::advance(i&: IPriv, n: 1);
5791 std::advance(i&: IRed, n: 1);
5792 std::advance(i&: ITD, n: 1);
5793 }
5794 }
5795 // Privatize in_reduction items here, because taskgroup descriptors must be
5796 // privatized earlier.
5797 OMPPrivateScope InRedScope(CGF);
5798 if (!InRedVars.empty()) {
5799 ReductionCodeGen RedCG(InRedVars, InRedVars, InRedPrivs, InRedOps);
5800 for (unsigned Cnt = 0, E = InRedVars.size(); Cnt < E; ++Cnt) {
5801 RedCG.emitSharedOrigLValue(CGF, N: Cnt);
5802 RedCG.emitAggregateType(CGF, N: Cnt);
5803 // The taskgroup descriptor variable is always implicit firstprivate and
5804 // privatized already during processing of the firstprivates.
5805 // FIXME: This must removed once the runtime library is fixed.
5806 // Emit required threadprivate variables for
5807 // initializer/combiner/finalizer.
5808 CGF.CGM.getOpenMPRuntime().emitTaskReductionFixups(CGF, Loc: S.getBeginLoc(),
5809 RCG&: RedCG, N: Cnt);
5810 llvm::Value *ReductionsPtr;
5811 if (const Expr *TRExpr = TaskgroupDescriptors[Cnt]) {
5812 ReductionsPtr = CGF.EmitLoadOfScalar(lvalue: CGF.EmitLValue(E: TRExpr),
5813 Loc: TRExpr->getExprLoc());
5814 } else {
5815 ReductionsPtr = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
5816 }
5817 Address Replacement = CGF.CGM.getOpenMPRuntime().getTaskReductionItem(
5818 CGF, Loc: S.getBeginLoc(), ReductionsPtr, SharedLVal: RedCG.getSharedLValue(N: Cnt));
5819 Replacement = Address(
5820 CGF.EmitScalarConversion(
5821 Src: Replacement.emitRawPointer(CGF), SrcTy: CGF.getContext().VoidPtrTy,
5822 DstTy: CGF.getContext().getPointerType(T: InRedPrivs[Cnt]->getType()),
5823 Loc: InRedPrivs[Cnt]->getExprLoc()),
5824 CGF.ConvertTypeForMem(T: InRedPrivs[Cnt]->getType()),
5825 Replacement.getAlignment());
5826 Replacement = RedCG.adjustPrivateAddress(CGF, N: Cnt, PrivateAddr: Replacement);
5827 InRedScope.addPrivate(LocalVD: RedCG.getBaseDecl(N: Cnt), Addr: Replacement);
5828 }
5829 }
5830 (void)InRedScope.Privatize();
5831
5832 CGOpenMPRuntime::UntiedTaskLocalDeclsRAII LocalVarsScope(CGF,
5833 UntiedLocalVars);
5834 Action.Enter(CGF);
5835 BodyGen(CGF);
5836 };
5837 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
5838 llvm::Function *OutlinedFn = CGM.getOpenMPRuntime().emitTaskOutlinedFunction(
5839 D: S, ThreadIDVar: *I, PartIDVar: *PartId, TaskTVar: *TaskT, InnermostKind: EKind, CodeGen, Tied: Data.Tied, NumberOfParts&: Data.NumberOfParts);
5840 OMPLexicalScope Scope(*this, S, std::nullopt,
5841 !isOpenMPParallelDirective(DKind: EKind) &&
5842 !isOpenMPSimdDirective(DKind: EKind));
5843 TaskGen(*this, OutlinedFn, Data);
5844}
5845
5846static ImplicitParamDecl *
5847createImplicitFirstprivateForType(ASTContext &C, OMPTaskDataTy &Data,
5848 QualType Ty, CapturedDecl *CD,
5849 SourceLocation Loc) {
5850 auto *OrigVD = ImplicitParamDecl::Create(C, DC: CD, IdLoc: Loc, /*Id=*/nullptr, T: Ty,
5851 ParamKind: ImplicitParamKind::Other);
5852 auto *OrigRef = DeclRefExpr::Create(
5853 Context: C, QualifierLoc: NestedNameSpecifierLoc(), TemplateKWLoc: SourceLocation(), D: OrigVD,
5854 /*RefersToEnclosingVariableOrCapture=*/false, NameLoc: Loc, T: Ty, VK: VK_LValue);
5855 auto *PrivateVD = ImplicitParamDecl::Create(C, DC: CD, IdLoc: Loc, /*Id=*/nullptr, T: Ty,
5856 ParamKind: ImplicitParamKind::Other);
5857 auto *PrivateRef = DeclRefExpr::Create(
5858 Context: C, QualifierLoc: NestedNameSpecifierLoc(), TemplateKWLoc: SourceLocation(), D: PrivateVD,
5859 /*RefersToEnclosingVariableOrCapture=*/false, NameLoc: Loc, T: Ty, VK: VK_LValue);
5860 QualType ElemType = C.getBaseElementType(QT: Ty);
5861 auto *InitVD = ImplicitParamDecl::Create(C, DC: CD, IdLoc: Loc, /*Id=*/nullptr, T: ElemType,
5862 ParamKind: ImplicitParamKind::Other);
5863 auto *InitRef = DeclRefExpr::Create(
5864 Context: C, QualifierLoc: NestedNameSpecifierLoc(), TemplateKWLoc: SourceLocation(), D: InitVD,
5865 /*RefersToEnclosingVariableOrCapture=*/false, NameLoc: Loc, T: ElemType, VK: VK_LValue);
5866 PrivateVD->setInitStyle(VarDecl::CInit);
5867 PrivateVD->setInit(ImplicitCastExpr::Create(Context: C, T: ElemType, Kind: CK_LValueToRValue,
5868 Operand: InitRef, /*BasePath=*/nullptr,
5869 Cat: VK_PRValue, FPO: FPOptionsOverride()));
5870 Data.FirstprivateVars.emplace_back(Args&: OrigRef);
5871 Data.FirstprivateCopies.emplace_back(Args&: PrivateRef);
5872 Data.FirstprivateInits.emplace_back(Args&: InitRef);
5873 return OrigVD;
5874}
5875
5876void CodeGenFunction::EmitOMPTargetTaskBasedDirective(
5877 const OMPExecutableDirective &S, const RegionCodeGenTy &BodyGen,
5878 OMPTargetDataInfo &InputInfo) {
5879 // Emit outlined function for task construct.
5880 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: OMPD_task);
5881 Address CapturedStruct = GenerateCapturedStmtArgument(S: *CS);
5882 CanQualType SharedsTy =
5883 getContext().getCanonicalTagType(TD: CS->getCapturedRecordDecl());
5884 auto I = CS->getCapturedDecl()->param_begin();
5885 auto PartId = std::next(x: I);
5886 auto TaskT = std::next(x: I, n: 4);
5887 OMPTaskDataTy Data;
5888 // The task is not final.
5889 Data.Final.setInt(/*IntVal=*/false);
5890 // Get list of firstprivate variables.
5891 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
5892 auto IRef = C->varlist_begin();
5893 auto IElemInitRef = C->inits().begin();
5894 for (auto *IInit : C->private_copies()) {
5895 Data.FirstprivateVars.push_back(Elt: *IRef);
5896 Data.FirstprivateCopies.push_back(Elt: IInit);
5897 Data.FirstprivateInits.push_back(Elt: *IElemInitRef);
5898 ++IRef;
5899 ++IElemInitRef;
5900 }
5901 }
5902 SmallVector<const Expr *, 4> LHSs;
5903 SmallVector<const Expr *, 4> RHSs;
5904 for (const auto *C : S.getClausesOfKind<OMPInReductionClause>()) {
5905 Data.ReductionVars.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
5906 Data.ReductionOrigs.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
5907 Data.ReductionCopies.append(in_start: C->privates().begin(), in_end: C->privates().end());
5908 Data.ReductionOps.append(in_start: C->reduction_ops().begin(),
5909 in_end: C->reduction_ops().end());
5910 LHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
5911 RHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
5912 }
5913 OMPPrivateScope TargetScope(*this);
5914 VarDecl *BPVD = nullptr;
5915 VarDecl *PVD = nullptr;
5916 VarDecl *SVD = nullptr;
5917 VarDecl *MVD = nullptr;
5918 if (InputInfo.NumberOfTargetItems > 0) {
5919 auto *CD = CapturedDecl::Create(
5920 C&: getContext(), DC: getContext().getTranslationUnitDecl(), /*NumParams=*/0);
5921 llvm::APInt ArrSize(/*numBits=*/32, InputInfo.NumberOfTargetItems);
5922 QualType BaseAndPointerAndMapperType = getContext().getConstantArrayType(
5923 EltTy: getContext().VoidPtrTy, ArySize: ArrSize, SizeExpr: nullptr, ASM: ArraySizeModifier::Normal,
5924 /*IndexTypeQuals=*/0);
5925 BPVD = createImplicitFirstprivateForType(
5926 C&: getContext(), Data, Ty: BaseAndPointerAndMapperType, CD, Loc: S.getBeginLoc());
5927 PVD = createImplicitFirstprivateForType(
5928 C&: getContext(), Data, Ty: BaseAndPointerAndMapperType, CD, Loc: S.getBeginLoc());
5929 QualType SizesType = getContext().getConstantArrayType(
5930 EltTy: getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1),
5931 ArySize: ArrSize, SizeExpr: nullptr, ASM: ArraySizeModifier::Normal,
5932 /*IndexTypeQuals=*/0);
5933 SVD = createImplicitFirstprivateForType(C&: getContext(), Data, Ty: SizesType, CD,
5934 Loc: S.getBeginLoc());
5935 TargetScope.addPrivate(LocalVD: BPVD, Addr: InputInfo.BasePointersArray);
5936 TargetScope.addPrivate(LocalVD: PVD, Addr: InputInfo.PointersArray);
5937 TargetScope.addPrivate(LocalVD: SVD, Addr: InputInfo.SizesArray);
5938 // If there is no user-defined mapper, the mapper array will be nullptr. In
5939 // this case, we don't need to privatize it.
5940 if (!isa_and_nonnull<llvm::ConstantPointerNull>(
5941 Val: InputInfo.MappersArray.emitRawPointer(CGF&: *this))) {
5942 MVD = createImplicitFirstprivateForType(
5943 C&: getContext(), Data, Ty: BaseAndPointerAndMapperType, CD, Loc: S.getBeginLoc());
5944 TargetScope.addPrivate(LocalVD: MVD, Addr: InputInfo.MappersArray);
5945 }
5946 }
5947 (void)TargetScope.Privatize();
5948 buildDependences(S, Data);
5949 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
5950 auto &&CodeGen = [&Data, &S, CS, &BodyGen, BPVD, PVD, SVD, MVD, EKind,
5951 &InputInfo](CodeGenFunction &CGF, PrePostActionTy &Action) {
5952 // Set proper addresses for generated private copies.
5953 OMPPrivateScope Scope(CGF);
5954 if (!Data.FirstprivateVars.empty()) {
5955 enum { PrivatesParam = 2, CopyFnParam = 3 };
5956 llvm::Value *CopyFn = CGF.Builder.CreateLoad(
5957 Addr: CGF.GetAddrOfLocalVar(VD: CS->getCapturedDecl()->getParam(i: CopyFnParam)));
5958 llvm::Value *PrivatesPtr = CGF.Builder.CreateLoad(Addr: CGF.GetAddrOfLocalVar(
5959 VD: CS->getCapturedDecl()->getParam(i: PrivatesParam)));
5960 // Map privates.
5961 llvm::SmallVector<std::pair<const ValueDecl *, Address>, 16> PrivatePtrs;
5962 llvm::SmallVector<llvm::Value *, 16> CallArgs;
5963 llvm::SmallVector<llvm::Type *, 4> ParamTypes;
5964 CallArgs.push_back(Elt: PrivatesPtr);
5965 ParamTypes.push_back(Elt: PrivatesPtr->getType());
5966 for (const Expr *E : Data.FirstprivateVars) {
5967 const auto *VD = cast<DeclRefExpr>(Val: E)->getDecl();
5968 RawAddress PrivatePtr = CGF.CreateMemTempWithoutCast(
5969 T: CGF.getContext().getPointerType(T: E->getType()),
5970 Name: ".firstpriv.ptr.addr");
5971 PrivatePtrs.emplace_back(Args&: VD, Args&: PrivatePtr);
5972 CallArgs.push_back(Elt: PrivatePtr.getPointer());
5973 ParamTypes.push_back(Elt: PrivatePtr.getType());
5974 }
5975 auto *CopyFnTy = llvm::FunctionType::get(Result: CGF.Builder.getVoidTy(),
5976 Params: ParamTypes, /*isVarArg=*/false);
5977 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(
5978 CGF, Loc: S.getBeginLoc(), OutlinedFn: {CopyFnTy, CopyFn}, Args: CallArgs);
5979 for (const auto &Pair : PrivatePtrs) {
5980 Address Replacement(
5981 CGF.Builder.CreateLoad(Addr: Pair.second),
5982 CGF.ConvertTypeForMem(T: Pair.first->getType().getNonReferenceType()),
5983 CGF.getContext().getDeclAlign(D: Pair.first));
5984 Scope.addPrivate(LocalVD: Pair.first, Addr: Replacement);
5985 }
5986 }
5987 CGF.processInReduction(S, Data, CGF, CS, Scope);
5988 if (InputInfo.NumberOfTargetItems > 0) {
5989 InputInfo.BasePointersArray = CGF.Builder.CreateConstArrayGEP(
5990 Addr: CGF.GetAddrOfLocalVar(VD: BPVD), /*Index=*/0);
5991 InputInfo.PointersArray = CGF.Builder.CreateConstArrayGEP(
5992 Addr: CGF.GetAddrOfLocalVar(VD: PVD), /*Index=*/0);
5993 InputInfo.SizesArray = CGF.Builder.CreateConstArrayGEP(
5994 Addr: CGF.GetAddrOfLocalVar(VD: SVD), /*Index=*/0);
5995 // If MVD is nullptr, the mapper array is not privatized
5996 if (MVD)
5997 InputInfo.MappersArray = CGF.Builder.CreateConstArrayGEP(
5998 Addr: CGF.GetAddrOfLocalVar(VD: MVD), /*Index=*/0);
5999 }
6000
6001 Action.Enter(CGF);
6002 OMPLexicalScope LexScope(CGF, S, OMPD_task, /*EmitPreInitStmt=*/false);
6003 auto *TL = S.getSingleClause<OMPThreadLimitClause>();
6004 if (CGF.CGM.getLangOpts().OpenMP >= 51 &&
6005 needsTaskBasedThreadLimit(DKind: EKind) && TL) {
6006 // Emit __kmpc_set_thread_limit() to set the thread_limit for the task
6007 // enclosing this target region. This will indirectly set the thread_limit
6008 // for every applicable construct within target region.
6009 CGF.CGM.getOpenMPRuntime().emitThreadLimitClause(
6010 CGF, ThreadLimit: TL->getThreadLimit().front(), Loc: S.getBeginLoc());
6011 }
6012 BodyGen(CGF);
6013 };
6014 llvm::Function *OutlinedFn = CGM.getOpenMPRuntime().emitTaskOutlinedFunction(
6015 D: S, ThreadIDVar: *I, PartIDVar: *PartId, TaskTVar: *TaskT, InnermostKind: EKind, CodeGen, /*Tied=*/true,
6016 NumberOfParts&: Data.NumberOfParts);
6017 llvm::APInt TrueOrFalse(32, S.hasClausesOfKind<OMPNowaitClause>() ? 1 : 0);
6018 IntegerLiteral IfCond(getContext(), TrueOrFalse,
6019 getContext().getIntTypeForBitwidth(DestWidth: 32, /*Signed=*/0),
6020 SourceLocation());
6021 CGM.getOpenMPRuntime().emitTaskCall(CGF&: *this, Loc: S.getBeginLoc(), D: S, TaskFunction: OutlinedFn,
6022 SharedsTy, Shareds: CapturedStruct, IfCond: &IfCond, Data);
6023}
6024
6025void CodeGenFunction::processInReduction(const OMPExecutableDirective &S,
6026 OMPTaskDataTy &Data,
6027 CodeGenFunction &CGF,
6028 const CapturedStmt *CS,
6029 OMPPrivateScope &Scope) {
6030 OpenMPDirectiveKind EKind = getEffectiveDirectiveKind(S);
6031 if (Data.Reductions) {
6032 OpenMPDirectiveKind CapturedRegion = EKind;
6033 OMPLexicalScope LexScope(CGF, S, CapturedRegion);
6034 ReductionCodeGen RedCG(Data.ReductionVars, Data.ReductionVars,
6035 Data.ReductionCopies, Data.ReductionOps);
6036 llvm::Value *ReductionsPtr = CGF.Builder.CreateLoad(
6037 Addr: CGF.GetAddrOfLocalVar(VD: CS->getCapturedDecl()->getParam(i: 4)));
6038 for (unsigned Cnt = 0, E = Data.ReductionVars.size(); Cnt < E; ++Cnt) {
6039 RedCG.emitSharedOrigLValue(CGF, N: Cnt);
6040 RedCG.emitAggregateType(CGF, N: Cnt);
6041 // FIXME: This must removed once the runtime library is fixed.
6042 // Emit required threadprivate variables for
6043 // initializer/combiner/finalizer.
6044 CGF.CGM.getOpenMPRuntime().emitTaskReductionFixups(CGF, Loc: S.getBeginLoc(),
6045 RCG&: RedCG, N: Cnt);
6046 Address Replacement = CGF.CGM.getOpenMPRuntime().getTaskReductionItem(
6047 CGF, Loc: S.getBeginLoc(), ReductionsPtr, SharedLVal: RedCG.getSharedLValue(N: Cnt));
6048 Replacement = Address(
6049 CGF.EmitScalarConversion(Src: Replacement.emitRawPointer(CGF),
6050 SrcTy: CGF.getContext().VoidPtrTy,
6051 DstTy: CGF.getContext().getPointerType(
6052 T: Data.ReductionCopies[Cnt]->getType()),
6053 Loc: Data.ReductionCopies[Cnt]->getExprLoc()),
6054 CGF.ConvertTypeForMem(T: Data.ReductionCopies[Cnt]->getType()),
6055 Replacement.getAlignment());
6056 Replacement = RedCG.adjustPrivateAddress(CGF, N: Cnt, PrivateAddr: Replacement);
6057 Scope.addPrivate(LocalVD: RedCG.getBaseDecl(N: Cnt), Addr: Replacement);
6058 }
6059 }
6060 (void)Scope.Privatize();
6061 SmallVector<const Expr *, 4> InRedVars;
6062 SmallVector<const Expr *, 4> InRedPrivs;
6063 SmallVector<const Expr *, 4> InRedOps;
6064 SmallVector<const Expr *, 4> TaskgroupDescriptors;
6065 for (const auto *C : S.getClausesOfKind<OMPInReductionClause>()) {
6066 auto IPriv = C->privates().begin();
6067 auto IRed = C->reduction_ops().begin();
6068 auto ITD = C->taskgroup_descriptors().begin();
6069 for (const Expr *Ref : C->varlist()) {
6070 InRedVars.emplace_back(Args&: Ref);
6071 InRedPrivs.emplace_back(Args: *IPriv);
6072 InRedOps.emplace_back(Args: *IRed);
6073 TaskgroupDescriptors.emplace_back(Args: *ITD);
6074 std::advance(i&: IPriv, n: 1);
6075 std::advance(i&: IRed, n: 1);
6076 std::advance(i&: ITD, n: 1);
6077 }
6078 }
6079 OMPPrivateScope InRedScope(CGF);
6080 if (!InRedVars.empty()) {
6081 ReductionCodeGen RedCG(InRedVars, InRedVars, InRedPrivs, InRedOps);
6082 for (unsigned Cnt = 0, E = InRedVars.size(); Cnt < E; ++Cnt) {
6083 RedCG.emitSharedOrigLValue(CGF, N: Cnt);
6084 RedCG.emitAggregateType(CGF, N: Cnt);
6085 // FIXME: This must removed once the runtime library is fixed.
6086 // Emit required threadprivate variables for
6087 // initializer/combiner/finalizer.
6088 CGF.CGM.getOpenMPRuntime().emitTaskReductionFixups(CGF, Loc: S.getBeginLoc(),
6089 RCG&: RedCG, N: Cnt);
6090 llvm::Value *ReductionsPtr;
6091 if (const Expr *TRExpr = TaskgroupDescriptors[Cnt]) {
6092 ReductionsPtr =
6093 CGF.EmitLoadOfScalar(lvalue: CGF.EmitLValue(E: TRExpr), Loc: TRExpr->getExprLoc());
6094 } else {
6095 ReductionsPtr = llvm::ConstantPointerNull::get(T: CGF.VoidPtrTy);
6096 }
6097 Address Replacement = CGF.CGM.getOpenMPRuntime().getTaskReductionItem(
6098 CGF, Loc: S.getBeginLoc(), ReductionsPtr, SharedLVal: RedCG.getSharedLValue(N: Cnt));
6099 Replacement = Address(
6100 CGF.EmitScalarConversion(
6101 Src: Replacement.emitRawPointer(CGF), SrcTy: CGF.getContext().VoidPtrTy,
6102 DstTy: CGF.getContext().getPointerType(T: InRedPrivs[Cnt]->getType()),
6103 Loc: InRedPrivs[Cnt]->getExprLoc()),
6104 CGF.ConvertTypeForMem(T: InRedPrivs[Cnt]->getType()),
6105 Replacement.getAlignment());
6106 Replacement = RedCG.adjustPrivateAddress(CGF, N: Cnt, PrivateAddr: Replacement);
6107 InRedScope.addPrivate(LocalVD: RedCG.getBaseDecl(N: Cnt), Addr: Replacement);
6108 }
6109 }
6110 (void)InRedScope.Privatize();
6111}
6112
6113void CodeGenFunction::EmitOMPTaskDirective(const OMPTaskDirective &S) {
6114 // Emit outlined function for task construct.
6115 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: OMPD_task);
6116 Address CapturedStruct = GenerateCapturedStmtArgument(S: *CS);
6117 CanQualType SharedsTy =
6118 getContext().getCanonicalTagType(TD: CS->getCapturedRecordDecl());
6119 const Expr *IfCond = nullptr;
6120 for (const auto *C : S.getClausesOfKind<OMPIfClause>()) {
6121 if (C->getNameModifier() == OMPD_unknown ||
6122 C->getNameModifier() == OMPD_task) {
6123 IfCond = C->getCondition();
6124 break;
6125 }
6126 }
6127
6128 OMPTaskDataTy Data;
6129 // Check if we should emit tied or untied task.
6130 Data.Tied = !S.getSingleClause<OMPUntiedClause>();
6131 auto &&BodyGen = [CS](CodeGenFunction &CGF, PrePostActionTy &) {
6132 CGF.EmitStmt(S: CS->getCapturedStmt());
6133 };
6134 auto &&TaskGen = [&S, SharedsTy, CapturedStruct,
6135 IfCond](CodeGenFunction &CGF, llvm::Function *OutlinedFn,
6136 const OMPTaskDataTy &Data) {
6137 CGF.CGM.getOpenMPRuntime().emitTaskCall(CGF, Loc: S.getBeginLoc(), D: S, TaskFunction: OutlinedFn,
6138 SharedsTy, Shareds: CapturedStruct, IfCond,
6139 Data);
6140 };
6141 auto LPCRegion =
6142 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
6143 EmitOMPTaskBasedDirective(S, CapturedRegion: OMPD_task, BodyGen, TaskGen, Data);
6144}
6145
6146void CodeGenFunction::EmitOMPTaskyieldDirective(
6147 const OMPTaskyieldDirective &S) {
6148 CGM.getOpenMPRuntime().emitTaskyieldCall(CGF&: *this, Loc: S.getBeginLoc());
6149}
6150
6151void CodeGenFunction::EmitOMPErrorDirective(const OMPErrorDirective &S) {
6152 const OMPMessageClause *MC = S.getSingleClause<OMPMessageClause>();
6153 Expr *ME = MC ? MC->getMessageString() : nullptr;
6154 const OMPSeverityClause *SC = S.getSingleClause<OMPSeverityClause>();
6155 bool IsFatal = false;
6156 if (!SC || SC->getSeverityKind() == OMPC_SEVERITY_fatal)
6157 IsFatal = true;
6158 CGM.getOpenMPRuntime().emitErrorCall(CGF&: *this, Loc: S.getBeginLoc(), ME, IsFatal);
6159}
6160
6161void CodeGenFunction::EmitOMPBarrierDirective(const OMPBarrierDirective &S) {
6162 CGM.getOpenMPRuntime().emitBarrierCall(CGF&: *this, Loc: S.getBeginLoc(), Kind: OMPD_barrier);
6163}
6164
6165void CodeGenFunction::EmitOMPTaskwaitDirective(const OMPTaskwaitDirective &S) {
6166 OMPTaskDataTy Data;
6167 // Build list of dependences
6168 buildDependences(S, Data);
6169 Data.HasNowaitClause = S.hasClausesOfKind<OMPNowaitClause>();
6170 CGM.getOpenMPRuntime().emitTaskwaitCall(CGF&: *this, Loc: S.getBeginLoc(), Data);
6171}
6172
6173static bool isSupportedByOpenMPIRBuilder(const OMPTaskgroupDirective &T) {
6174 return T.clauses().empty();
6175}
6176
6177void CodeGenFunction::EmitOMPTaskgroupDirective(
6178 const OMPTaskgroupDirective &S) {
6179 OMPLexicalScope Scope(*this, S, OMPD_unknown);
6180 if (CGM.getLangOpts().OpenMPIRBuilder && isSupportedByOpenMPIRBuilder(T: S)) {
6181 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
6182 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
6183 InsertPointTy AllocaIP(AllocaInsertPt->getIterator());
6184
6185 auto BodyGenCB = [&, this](InsertPointTy AllocIP, InsertPointTy CodeGenIP,
6186 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
6187 Builder.restoreIP(IP: CodeGenIP);
6188 EmitStmt(S: S.getInnermostCapturedStmt()->getCapturedStmt());
6189 return llvm::Error::success();
6190 };
6191 CodeGenFunction::CGCapturedStmtInfo CapStmtInfo;
6192 if (!CapturedStmtInfo)
6193 CapturedStmtInfo = &CapStmtInfo;
6194 llvm::OpenMPIRBuilder::InsertPointTy AfterIP =
6195 cantFail(ValOrErr: OMPBuilder.createTaskgroup(Loc: Builder, AllocaIP,
6196 /*DeallocBlocks=*/{}, BodyGenCB));
6197 Builder.restoreIP(IP: AfterIP);
6198 return;
6199 }
6200 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
6201 Action.Enter(CGF);
6202 if (const Expr *E = S.getReductionRef()) {
6203 SmallVector<const Expr *, 4> LHSs;
6204 SmallVector<const Expr *, 4> RHSs;
6205 OMPTaskDataTy Data;
6206 for (const auto *C : S.getClausesOfKind<OMPTaskReductionClause>()) {
6207 Data.ReductionVars.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
6208 Data.ReductionOrigs.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
6209 Data.ReductionCopies.append(in_start: C->privates().begin(), in_end: C->privates().end());
6210 Data.ReductionOps.append(in_start: C->reduction_ops().begin(),
6211 in_end: C->reduction_ops().end());
6212 LHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
6213 RHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
6214 }
6215 llvm::Value *ReductionDesc =
6216 CGF.CGM.getOpenMPRuntime().emitTaskReductionInit(CGF, Loc: S.getBeginLoc(),
6217 LHSExprs: LHSs, RHSExprs: RHSs, Data);
6218 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl());
6219 CGF.EmitVarDecl(D: *VD);
6220 CGF.EmitStoreOfScalar(Value: ReductionDesc, Addr: CGF.GetAddrOfLocalVar(VD),
6221 /*Volatile=*/false, Ty: E->getType());
6222 }
6223 CGF.EmitStmt(S: S.getInnermostCapturedStmt()->getCapturedStmt());
6224 };
6225 CGM.getOpenMPRuntime().emitTaskgroupRegion(CGF&: *this, TaskgroupOpGen: CodeGen, Loc: S.getBeginLoc());
6226}
6227
6228void CodeGenFunction::EmitOMPFlushDirective(const OMPFlushDirective &S) {
6229 llvm::AtomicOrdering AO = S.getSingleClause<OMPFlushClause>()
6230 ? llvm::AtomicOrdering::NotAtomic
6231 : llvm::AtomicOrdering::AcquireRelease;
6232 CGM.getOpenMPRuntime().emitFlush(
6233 CGF&: *this,
6234 Vars: [&S]() -> ArrayRef<const Expr *> {
6235 if (const auto *FlushClause = S.getSingleClause<OMPFlushClause>())
6236 return llvm::ArrayRef(FlushClause->varlist_begin(),
6237 FlushClause->varlist_end());
6238 return {};
6239 }(),
6240 Loc: S.getBeginLoc(), AO);
6241}
6242
6243void CodeGenFunction::EmitOMPDepobjDirective(const OMPDepobjDirective &S) {
6244 const auto *DO = S.getSingleClause<OMPDepobjClause>();
6245 LValue DOLVal = EmitLValue(E: DO->getDepobj());
6246 if (const auto *DC = S.getSingleClause<OMPDependClause>()) {
6247 // Build list and emit dependences
6248 OMPTaskDataTy Data;
6249 buildDependences(S, Data);
6250 for (auto &Dep : Data.Dependences) {
6251 Address DepAddr = CGM.getOpenMPRuntime().emitDepobjDependClause(
6252 CGF&: *this, Dependencies: Dep, Loc: DC->getBeginLoc());
6253 EmitStoreOfScalar(value: DepAddr.emitRawPointer(CGF&: *this), lvalue: DOLVal);
6254 }
6255 return;
6256 }
6257 if (const auto *DC = S.getSingleClause<OMPDestroyClause>()) {
6258 CGM.getOpenMPRuntime().emitDestroyClause(CGF&: *this, DepobjLVal: DOLVal, Loc: DC->getBeginLoc());
6259 return;
6260 }
6261 if (const auto *UC = S.getSingleClause<OMPUpdateDependObjectsClause>()) {
6262 CGM.getOpenMPRuntime().emitUpdateDependObjectsClause(
6263 CGF&: *this, DepobjLVal: DOLVal, NewDepKind: UC->getDependencyKind(), Loc: UC->getBeginLoc());
6264 return;
6265 }
6266}
6267
6268void CodeGenFunction::EmitOMPScanDirective(const OMPScanDirective &S) {
6269 if (!OMPParentLoopDirectiveForScan)
6270 return;
6271 const OMPExecutableDirective &ParentDir = *OMPParentLoopDirectiveForScan;
6272 bool IsInclusive = S.hasClausesOfKind<OMPInclusiveClause>();
6273 SmallVector<const Expr *, 4> Shareds;
6274 SmallVector<const Expr *, 4> Privates;
6275 SmallVector<const Expr *, 4> LHSs;
6276 SmallVector<const Expr *, 4> RHSs;
6277 SmallVector<const Expr *, 4> ReductionOps;
6278 SmallVector<const Expr *, 4> CopyOps;
6279 SmallVector<const Expr *, 4> CopyArrayTemps;
6280 SmallVector<const Expr *, 4> CopyArrayElems;
6281 for (const auto *C : ParentDir.getClausesOfKind<OMPReductionClause>()) {
6282 if (C->getModifier() != OMPC_REDUCTION_inscan)
6283 continue;
6284 Shareds.append(in_start: C->varlist_begin(), in_end: C->varlist_end());
6285 Privates.append(in_start: C->privates().begin(), in_end: C->privates().end());
6286 LHSs.append(in_start: C->lhs_exprs().begin(), in_end: C->lhs_exprs().end());
6287 RHSs.append(in_start: C->rhs_exprs().begin(), in_end: C->rhs_exprs().end());
6288 ReductionOps.append(in_start: C->reduction_ops().begin(), in_end: C->reduction_ops().end());
6289 CopyOps.append(in_start: C->copy_ops().begin(), in_end: C->copy_ops().end());
6290 CopyArrayTemps.append(in_start: C->copy_array_temps().begin(),
6291 in_end: C->copy_array_temps().end());
6292 CopyArrayElems.append(in_start: C->copy_array_elems().begin(),
6293 in_end: C->copy_array_elems().end());
6294 }
6295 if (ParentDir.getDirectiveKind() == OMPD_simd ||
6296 (getLangOpts().OpenMPSimd &&
6297 isOpenMPSimdDirective(DKind: ParentDir.getDirectiveKind()))) {
6298 // For simd directive and simd-based directives in simd only mode, use the
6299 // following codegen:
6300 // int x = 0;
6301 // #pragma omp simd reduction(inscan, +: x)
6302 // for (..) {
6303 // <first part>
6304 // #pragma omp scan inclusive(x)
6305 // <second part>
6306 // }
6307 // is transformed to:
6308 // int x = 0;
6309 // for (..) {
6310 // int x_priv = 0;
6311 // <first part>
6312 // x = x_priv + x;
6313 // x_priv = x;
6314 // <second part>
6315 // }
6316 // and
6317 // int x = 0;
6318 // #pragma omp simd reduction(inscan, +: x)
6319 // for (..) {
6320 // <first part>
6321 // #pragma omp scan exclusive(x)
6322 // <second part>
6323 // }
6324 // to
6325 // int x = 0;
6326 // for (..) {
6327 // int x_priv = 0;
6328 // <second part>
6329 // int temp = x;
6330 // x = x_priv + x;
6331 // x_priv = temp;
6332 // <first part>
6333 // }
6334 llvm::BasicBlock *OMPScanReduce = createBasicBlock(name: "omp.inscan.reduce");
6335 EmitBranch(Block: IsInclusive
6336 ? OMPScanReduce
6337 : BreakContinueStack.back().ContinueBlock.getBlock());
6338 EmitBlock(BB: OMPScanDispatch);
6339 {
6340 // New scope for correct construction/destruction of temp variables for
6341 // exclusive scan.
6342 LexicalScope Scope(*this, S.getSourceRange());
6343 EmitBranch(Block: IsInclusive ? OMPBeforeScanBlock : OMPAfterScanBlock);
6344 EmitBlock(BB: OMPScanReduce);
6345 if (!IsInclusive) {
6346 // Create temp var and copy LHS value to this temp value.
6347 // TMP = LHS;
6348 for (unsigned I = 0, E = CopyArrayElems.size(); I < E; ++I) {
6349 const Expr *PrivateExpr = Privates[I];
6350 const Expr *TempExpr = CopyArrayTemps[I];
6351 EmitAutoVarDecl(
6352 D: *cast<VarDecl>(Val: cast<DeclRefExpr>(Val: TempExpr)->getDecl()));
6353 LValue DestLVal = EmitLValue(E: TempExpr);
6354 LValue SrcLVal = EmitLValue(E: LHSs[I]);
6355 EmitOMPCopy(OriginalType: PrivateExpr->getType(), DestAddr: DestLVal.getAddress(),
6356 SrcAddr: SrcLVal.getAddress(),
6357 DestVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: LHSs[I])->getDecl()),
6358 SrcVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: RHSs[I])->getDecl()),
6359 Copy: CopyOps[I]);
6360 }
6361 }
6362 CGM.getOpenMPRuntime().emitReduction(
6363 CGF&: *this, Loc: ParentDir.getEndLoc(), Privates, LHSExprs: LHSs, RHSExprs: RHSs, ReductionOps,
6364 Options: {/*WithNowait=*/true, /*SimpleReduction=*/true,
6365 /*IsPrivateVarReduction*/ {}, .ReductionKind: OMPD_simd});
6366 for (unsigned I = 0, E = CopyArrayElems.size(); I < E; ++I) {
6367 const Expr *PrivateExpr = Privates[I];
6368 LValue DestLVal;
6369 LValue SrcLVal;
6370 if (IsInclusive) {
6371 DestLVal = EmitLValue(E: RHSs[I]);
6372 SrcLVal = EmitLValue(E: LHSs[I]);
6373 } else {
6374 const Expr *TempExpr = CopyArrayTemps[I];
6375 DestLVal = EmitLValue(E: RHSs[I]);
6376 SrcLVal = EmitLValue(E: TempExpr);
6377 }
6378 EmitOMPCopy(
6379 OriginalType: PrivateExpr->getType(), DestAddr: DestLVal.getAddress(), SrcAddr: SrcLVal.getAddress(),
6380 DestVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: LHSs[I])->getDecl()),
6381 SrcVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: RHSs[I])->getDecl()), Copy: CopyOps[I]);
6382 }
6383 }
6384 EmitBranch(Block: IsInclusive ? OMPAfterScanBlock : OMPBeforeScanBlock);
6385 OMPScanExitBlock = IsInclusive
6386 ? BreakContinueStack.back().ContinueBlock.getBlock()
6387 : OMPScanReduce;
6388 EmitBlock(BB: OMPAfterScanBlock);
6389 return;
6390 }
6391 if (!IsInclusive) {
6392 EmitBranch(Block: BreakContinueStack.back().ContinueBlock.getBlock());
6393 EmitBlock(BB: OMPScanExitBlock);
6394 }
6395 if (OMPFirstScanLoop) {
6396 // Emit buffer[i] = red; at the end of the input phase.
6397 const auto *IVExpr = cast<OMPLoopDirective>(Val: ParentDir)
6398 .getIterationVariable()
6399 ->IgnoreParenImpCasts();
6400 LValue IdxLVal = EmitLValue(E: IVExpr);
6401 llvm::Value *IdxVal = EmitLoadOfScalar(lvalue: IdxLVal, Loc: IVExpr->getExprLoc());
6402 IdxVal = Builder.CreateIntCast(V: IdxVal, DestTy: SizeTy, /*isSigned=*/false);
6403 for (unsigned I = 0, E = CopyArrayElems.size(); I < E; ++I) {
6404 const Expr *PrivateExpr = Privates[I];
6405 const Expr *OrigExpr = Shareds[I];
6406 const Expr *CopyArrayElem = CopyArrayElems[I];
6407 OpaqueValueMapping IdxMapping(
6408 *this,
6409 cast<OpaqueValueExpr>(
6410 Val: cast<ArraySubscriptExpr>(Val: CopyArrayElem)->getIdx()),
6411 RValue::get(V: IdxVal));
6412 LValue DestLVal = EmitLValue(E: CopyArrayElem);
6413 LValue SrcLVal = EmitLValue(E: OrigExpr);
6414 EmitOMPCopy(
6415 OriginalType: PrivateExpr->getType(), DestAddr: DestLVal.getAddress(), SrcAddr: SrcLVal.getAddress(),
6416 DestVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: LHSs[I])->getDecl()),
6417 SrcVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: RHSs[I])->getDecl()), Copy: CopyOps[I]);
6418 }
6419 }
6420 EmitBranch(Block: BreakContinueStack.back().ContinueBlock.getBlock());
6421 if (IsInclusive) {
6422 EmitBlock(BB: OMPScanExitBlock);
6423 EmitBranch(Block: BreakContinueStack.back().ContinueBlock.getBlock());
6424 }
6425 EmitBlock(BB: OMPScanDispatch);
6426 if (!OMPFirstScanLoop) {
6427 // Emit red = buffer[i]; at the entrance to the scan phase.
6428 const auto *IVExpr = cast<OMPLoopDirective>(Val: ParentDir)
6429 .getIterationVariable()
6430 ->IgnoreParenImpCasts();
6431 LValue IdxLVal = EmitLValue(E: IVExpr);
6432 llvm::Value *IdxVal = EmitLoadOfScalar(lvalue: IdxLVal, Loc: IVExpr->getExprLoc());
6433 IdxVal = Builder.CreateIntCast(V: IdxVal, DestTy: SizeTy, /*isSigned=*/false);
6434 llvm::BasicBlock *ExclusiveExitBB = nullptr;
6435 if (!IsInclusive) {
6436 llvm::BasicBlock *ContBB = createBasicBlock(name: "omp.exclusive.dec");
6437 ExclusiveExitBB = createBasicBlock(name: "omp.exclusive.copy.exit");
6438 llvm::Value *Cmp = Builder.CreateIsNull(Arg: IdxVal);
6439 Builder.CreateCondBr(Cond: Cmp, True: ExclusiveExitBB, False: ContBB);
6440 EmitBlock(BB: ContBB);
6441 // Use idx - 1 iteration for exclusive scan.
6442 IdxVal = Builder.CreateNUWSub(LHS: IdxVal, RHS: llvm::ConstantInt::get(Ty: SizeTy, V: 1));
6443 }
6444 for (unsigned I = 0, E = CopyArrayElems.size(); I < E; ++I) {
6445 const Expr *PrivateExpr = Privates[I];
6446 const Expr *OrigExpr = Shareds[I];
6447 const Expr *CopyArrayElem = CopyArrayElems[I];
6448 OpaqueValueMapping IdxMapping(
6449 *this,
6450 cast<OpaqueValueExpr>(
6451 Val: cast<ArraySubscriptExpr>(Val: CopyArrayElem)->getIdx()),
6452 RValue::get(V: IdxVal));
6453 LValue SrcLVal = EmitLValue(E: CopyArrayElem);
6454 LValue DestLVal = EmitLValue(E: OrigExpr);
6455 EmitOMPCopy(
6456 OriginalType: PrivateExpr->getType(), DestAddr: DestLVal.getAddress(), SrcAddr: SrcLVal.getAddress(),
6457 DestVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: LHSs[I])->getDecl()),
6458 SrcVD: cast<VarDecl>(Val: cast<DeclRefExpr>(Val: RHSs[I])->getDecl()), Copy: CopyOps[I]);
6459 }
6460 if (!IsInclusive) {
6461 EmitBlock(BB: ExclusiveExitBB);
6462 }
6463 }
6464 EmitBranch(Block: (OMPFirstScanLoop == IsInclusive) ? OMPBeforeScanBlock
6465 : OMPAfterScanBlock);
6466 EmitBlock(BB: OMPAfterScanBlock);
6467}
6468
6469void CodeGenFunction::EmitOMPDistributeLoop(const OMPLoopDirective &S,
6470 const CodeGenLoopTy &CodeGenLoop,
6471 Expr *IncExpr) {
6472 // Emit the loop iteration variable.
6473 const auto *IVExpr = cast<DeclRefExpr>(Val: S.getIterationVariable());
6474 const auto *IVDecl = cast<VarDecl>(Val: IVExpr->getDecl());
6475 EmitVarDecl(D: *IVDecl);
6476
6477 // Emit the iterations count variable.
6478 // If it is not a variable, Sema decided to calculate iterations count on each
6479 // iteration (e.g., it is foldable into a constant).
6480 if (const auto *LIExpr = dyn_cast<DeclRefExpr>(Val: S.getLastIteration())) {
6481 EmitVarDecl(D: *cast<VarDecl>(Val: LIExpr->getDecl()));
6482 // Emit calculation of the iterations count.
6483 EmitIgnoredExpr(E: S.getCalcLastIteration());
6484 }
6485
6486 CGOpenMPRuntime &RT = CGM.getOpenMPRuntime();
6487
6488 bool HasLastprivateClause = false;
6489 // Check pre-condition.
6490 {
6491 OMPLoopScope PreInitScope(*this, S);
6492 // Skip the entire loop if we don't meet the precondition.
6493 // If the condition constant folds and can be elided, avoid emitting the
6494 // whole loop.
6495 bool CondConstant;
6496 llvm::BasicBlock *ContBlock = nullptr;
6497 if (ConstantFoldsToSimpleInteger(Cond: S.getPreCond(), Result&: CondConstant)) {
6498 if (!CondConstant)
6499 return;
6500 } else {
6501 llvm::BasicBlock *ThenBlock = createBasicBlock(name: "omp.precond.then");
6502 ContBlock = createBasicBlock(name: "omp.precond.end");
6503 emitPreCond(CGF&: *this, S, Cond: S.getPreCond(), TrueBlock: ThenBlock, FalseBlock: ContBlock,
6504 TrueCount: getProfileCount(S: &S));
6505 EmitBlock(BB: ThenBlock);
6506 incrementProfileCounter(S: &S);
6507 }
6508
6509 emitAlignedClause(CGF&: *this, D: S);
6510 // Emit 'then' code.
6511 {
6512 // Emit helper vars inits.
6513
6514 LValue LB = EmitOMPHelperVar(
6515 CGF&: *this, Helper: cast<DeclRefExpr>(
6516 Val: (isOpenMPLoopBoundSharingDirective(Kind: S.getDirectiveKind())
6517 ? S.getCombinedLowerBoundVariable()
6518 : S.getLowerBoundVariable())));
6519 LValue UB = EmitOMPHelperVar(
6520 CGF&: *this, Helper: cast<DeclRefExpr>(
6521 Val: (isOpenMPLoopBoundSharingDirective(Kind: S.getDirectiveKind())
6522 ? S.getCombinedUpperBoundVariable()
6523 : S.getUpperBoundVariable())));
6524 LValue ST =
6525 EmitOMPHelperVar(CGF&: *this, Helper: cast<DeclRefExpr>(Val: S.getStrideVariable()));
6526 LValue IL =
6527 EmitOMPHelperVar(CGF&: *this, Helper: cast<DeclRefExpr>(Val: S.getIsLastIterVariable()));
6528
6529 OMPPrivateScope LoopScope(*this);
6530 if (EmitOMPFirstprivateClause(D: S, PrivateScope&: LoopScope)) {
6531 // Emit implicit barrier to synchronize threads and avoid data races
6532 // on initialization of firstprivate variables and post-update of
6533 // lastprivate variables.
6534 CGM.getOpenMPRuntime().emitBarrierCall(
6535 CGF&: *this, Loc: S.getBeginLoc(), Kind: OMPD_unknown, /*EmitChecks=*/false,
6536 /*ForceSimpleCall=*/true);
6537 }
6538 EmitOMPPrivateClause(D: S, PrivateScope&: LoopScope);
6539 if (isOpenMPSimdDirective(DKind: S.getDirectiveKind()) &&
6540 !isOpenMPParallelDirective(DKind: S.getDirectiveKind()) &&
6541 !isOpenMPTeamsDirective(DKind: S.getDirectiveKind()))
6542 EmitOMPReductionClauseInit(D: S, PrivateScope&: LoopScope);
6543 HasLastprivateClause = EmitOMPLastprivateClauseInit(D: S, PrivateScope&: LoopScope);
6544 EmitOMPPrivateLoopCounters(S, LoopScope);
6545 (void)LoopScope.Privatize();
6546 if (isOpenMPTargetExecutionDirective(DKind: S.getDirectiveKind()))
6547 CGM.getOpenMPRuntime().adjustTargetSpecificDataForLambdas(CGF&: *this, D: S);
6548
6549 // Detect the distribute schedule kind and chunk.
6550 llvm::Value *Chunk = nullptr;
6551 OpenMPDistScheduleClauseKind ScheduleKind = OMPC_DIST_SCHEDULE_unknown;
6552 if (const auto *C = S.getSingleClause<OMPDistScheduleClause>()) {
6553 ScheduleKind = C->getDistScheduleKind();
6554 if (const Expr *Ch = C->getChunkSize()) {
6555 Chunk = EmitScalarExpr(E: Ch);
6556 Chunk = EmitScalarConversion(Src: Chunk, SrcTy: Ch->getType(),
6557 DstTy: S.getIterationVariable()->getType(),
6558 Loc: S.getBeginLoc());
6559 }
6560 } else {
6561 // Default behaviour for dist_schedule clause.
6562 CGM.getOpenMPRuntime().getDefaultDistScheduleAndChunk(
6563 CGF&: *this, S, ScheduleKind, Chunk);
6564 }
6565 const unsigned IVSize = getContext().getTypeSize(T: IVExpr->getType());
6566 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation();
6567
6568 // GPU fused schedule: omit the outer distribute loop and let the inner
6569 // worksharing loop schedule the flattened team/thread iteration space.
6570 if (canEmitGPUFusedDistSchedule(CGM, S, DKind: S.getDirectiveKind())) {
6571 JumpDest LoopExit =
6572 getJumpDestInCurrentScope(Target: createBasicBlock(name: "omp.loop.exit"));
6573 CodeGenLoop(*this, S, LoopExit);
6574 EmitBlock(BB: LoopExit.getBlock());
6575 } else {
6576 // OpenMP [2.10.8, distribute Construct, Description]
6577 // If dist_schedule is specified, kind must be static. If specified,
6578 // iterations are divided into chunks of size chunk_size, chunks are
6579 // assigned to the teams of the league in a round-robin fashion in the
6580 // order of the team number. When no chunk_size is specified, the
6581 // iteration space is divided into chunks that are approximately equal
6582 // in size, and at most one chunk is distributed to each team of the
6583 // league. The size of the chunks is unspecified in this case.
6584 bool StaticChunked =
6585 RT.isStaticChunked(ScheduleKind, /* Chunked */ Chunk != nullptr) &&
6586 isOpenMPLoopBoundSharingDirective(Kind: S.getDirectiveKind());
6587 if (RT.isStaticNonchunked(ScheduleKind,
6588 /* Chunked */ Chunk != nullptr) ||
6589 StaticChunked) {
6590 CGOpenMPRuntime::StaticRTInput StaticInit(
6591 IVSize, IVSigned, /* Ordered = */ false, IL.getAddress(),
6592 LB.getAddress(), UB.getAddress(), ST.getAddress(),
6593 StaticChunked ? Chunk : nullptr);
6594 RT.emitDistributeStaticInit(CGF&: *this, Loc: S.getBeginLoc(), SchedKind: ScheduleKind,
6595 Values: StaticInit);
6596 JumpDest LoopExit =
6597 getJumpDestInCurrentScope(Target: createBasicBlock(name: "omp.loop.exit"));
6598 // UB = min(UB, GlobalUB);
6599 EmitIgnoredExpr(
6600 E: isOpenMPLoopBoundSharingDirective(Kind: S.getDirectiveKind())
6601 ? S.getCombinedEnsureUpperBound()
6602 : S.getEnsureUpperBound());
6603 // IV = LB;
6604 EmitIgnoredExpr(
6605 E: isOpenMPLoopBoundSharingDirective(Kind: S.getDirectiveKind())
6606 ? S.getCombinedInit()
6607 : S.getInit());
6608
6609 const Expr *Cond =
6610 isOpenMPLoopBoundSharingDirective(Kind: S.getDirectiveKind())
6611 ? S.getCombinedCond()
6612 : S.getCond();
6613
6614 if (StaticChunked)
6615 Cond = S.getCombinedDistCond();
6616
6617 // For static unchunked schedules generate:
6618 //
6619 // 1. For distribute alone, codegen
6620 // while (idx <= UB) {
6621 // BODY;
6622 // ++idx;
6623 // }
6624 //
6625 // 2. When combined with 'for' (e.g. as in 'distribute parallel for')
6626 // while (idx <= UB) {
6627 // <CodeGen rest of pragma>(LB, UB);
6628 // idx += ST;
6629 // }
6630 //
6631 // For static chunk one schedule generate:
6632 //
6633 // while (IV <= GlobalUB) {
6634 // <CodeGen rest of pragma>(LB, UB);
6635 // LB += ST;
6636 // UB += ST;
6637 // UB = min(UB, GlobalUB);
6638 // IV = LB;
6639 // }
6640 //
6641 emitCommonSimdLoop(
6642 CGF&: *this, S,
6643 SimdInitGen: [&S](CodeGenFunction &CGF, PrePostActionTy &) {
6644 if (isOpenMPSimdDirective(DKind: S.getDirectiveKind()))
6645 CGF.EmitOMPSimdInit(D: S);
6646 },
6647 BodyCodeGen: [&S, &LoopScope, Cond, IncExpr, LoopExit, &CodeGenLoop,
6648 StaticChunked](CodeGenFunction &CGF, PrePostActionTy &) {
6649 CGF.EmitOMPInnerLoop(
6650 S, RequiresCleanup: LoopScope.requiresCleanups(), LoopCond: Cond, IncExpr,
6651 BodyGen: [&S, LoopExit, &CodeGenLoop](CodeGenFunction &CGF) {
6652 CodeGenLoop(CGF, S, LoopExit);
6653 },
6654 PostIncGen: [&S, StaticChunked](CodeGenFunction &CGF) {
6655 if (StaticChunked) {
6656 CGF.EmitIgnoredExpr(E: S.getCombinedNextLowerBound());
6657 CGF.EmitIgnoredExpr(E: S.getCombinedNextUpperBound());
6658 CGF.EmitIgnoredExpr(E: S.getCombinedEnsureUpperBound());
6659 CGF.EmitIgnoredExpr(E: S.getCombinedInit());
6660 }
6661 });
6662 });
6663 EmitBlock(BB: LoopExit.getBlock());
6664 // Tell the runtime we are done.
6665 RT.emitForStaticFinish(CGF&: *this, Loc: S.getEndLoc(), DKind: OMPD_distribute);
6666 } else {
6667 // Emit the outer loop, which requests its work chunk [LB..UB] from
6668 // runtime and runs the inner loop to process it.
6669 const OMPLoopArguments LoopArguments = {
6670 LB.getAddress(), UB.getAddress(), ST.getAddress(),
6671 IL.getAddress(), Chunk};
6672 EmitOMPDistributeOuterLoop(ScheduleKind, S, LoopScope, LoopArgs: LoopArguments,
6673 CodeGenLoopContent: CodeGenLoop);
6674 }
6675 }
6676 if (isOpenMPSimdDirective(DKind: S.getDirectiveKind())) {
6677 EmitOMPSimdFinal(D: S, CondGen: [IL, &S](CodeGenFunction &CGF) {
6678 return CGF.Builder.CreateIsNotNull(
6679 Arg: CGF.EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc()));
6680 });
6681 }
6682 if (isOpenMPSimdDirective(DKind: S.getDirectiveKind()) &&
6683 !isOpenMPParallelDirective(DKind: S.getDirectiveKind()) &&
6684 !isOpenMPTeamsDirective(DKind: S.getDirectiveKind())) {
6685 EmitOMPReductionClauseFinal(D: S, ReductionKind: OMPD_simd);
6686 // Emit post-update of the reduction variables if IsLastIter != 0.
6687 emitPostUpdateForReductionClause(
6688 CGF&: *this, D: S, CondGen: [IL, &S](CodeGenFunction &CGF) {
6689 return CGF.Builder.CreateIsNotNull(
6690 Arg: CGF.EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc()));
6691 });
6692 }
6693 // Emit final copy of the lastprivate variables if IsLastIter != 0.
6694 if (HasLastprivateClause) {
6695 EmitOMPLastprivateClauseFinal(
6696 D: S, /*NoFinals=*/false,
6697 IsLastIterCond: Builder.CreateIsNotNull(Arg: EmitLoadOfScalar(lvalue: IL, Loc: S.getBeginLoc())));
6698 }
6699 }
6700
6701 // We're now done with the loop, so jump to the continuation block.
6702 if (ContBlock) {
6703 EmitBranch(Block: ContBlock);
6704 EmitBlock(BB: ContBlock, IsFinished: true);
6705 }
6706 }
6707}
6708
6709// Pass OMPLoopDirective (instead of OMPDistributeDirective) to make this
6710// function available for "loop bind(teams)", which maps to "distribute".
6711static void emitOMPDistributeDirective(const OMPLoopDirective &S,
6712 CodeGenFunction &CGF,
6713 CodeGenModule &CGM) {
6714 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
6715 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
6716 };
6717 OMPLexicalScope Scope(CGF, S, OMPD_unknown);
6718 CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_distribute, CodeGen);
6719}
6720
6721void CodeGenFunction::EmitOMPDistributeDirective(
6722 const OMPDistributeDirective &S) {
6723 emitOMPDistributeDirective(S, CGF&: *this, CGM);
6724}
6725
6726static llvm::Function *
6727emitOutlinedOrderedFunction(CodeGenModule &CGM, const CapturedStmt *S,
6728 const OMPExecutableDirective &D) {
6729 CodeGenFunction CGF(CGM, /*suppressNewContext=*/true);
6730 CodeGenFunction::CGCapturedStmtInfo CapStmtInfo;
6731 CGF.CapturedStmtInfo = &CapStmtInfo;
6732 llvm::Function *Fn = CGF.GenerateOpenMPCapturedStmtFunction(S: *S, D);
6733 Fn->setDoesNotRecurse();
6734 return Fn;
6735}
6736
6737template <typename T>
6738static void emitRestoreIP(CodeGenFunction &CGF, const T *C,
6739 llvm::OpenMPIRBuilder::InsertPointTy AllocaIP,
6740 llvm::OpenMPIRBuilder &OMPBuilder) {
6741
6742 unsigned NumLoops = C->getNumLoops();
6743 QualType Int64Ty = CGF.CGM.getContext().getIntTypeForBitwidth(
6744 /*DestWidth=*/64, /*Signed=*/1);
6745 llvm::SmallVector<llvm::Value *> StoreValues;
6746 for (unsigned I = 0; I < NumLoops; I++) {
6747 const Expr *CounterVal = C->getLoopData(I);
6748 assert(CounterVal);
6749 llvm::Value *StoreValue = CGF.EmitScalarConversion(
6750 Src: CGF.EmitScalarExpr(E: CounterVal), SrcTy: CounterVal->getType(), DstTy: Int64Ty,
6751 Loc: CounterVal->getExprLoc());
6752 StoreValues.emplace_back(Args&: StoreValue);
6753 }
6754 OMPDoacrossKind<T> ODK;
6755 bool IsDependSource = ODK.isSource(C);
6756 CGF.Builder.restoreIP(
6757 IP: OMPBuilder.createOrderedDepend(Loc: CGF.Builder, AllocaIP, NumLoops,
6758 StoreValues, Name: ".cnt.addr", IsDependSource));
6759}
6760
6761void CodeGenFunction::EmitOMPOrderedStandaloneDirective(
6762 const OMPOrderedStandaloneDirective &S) {
6763 assert((S.hasClausesOfKind<OMPDependClause>() ||
6764 S.hasClausesOfKind<OMPDoacrossClause>()) &&
6765 "Standalone ordered directive should have either depend or doacross "
6766 "clause");
6767 // The ordered-standalone directive.
6768 assert(!S.hasAssociatedStmt() && "No associated statement must be in "
6769 "ordered depend|doacross construct.");
6770
6771 if (CGM.getLangOpts().OpenMPIRBuilder) {
6772 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
6773 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
6774
6775 InsertPointTy AllocaIP(AllocaInsertPt->getIterator());
6776 for (const auto *DC : S.getClausesOfKind<OMPDependClause>())
6777 emitRestoreIP(CGF&: *this, C: DC, AllocaIP, OMPBuilder);
6778 for (const auto *DC : S.getClausesOfKind<OMPDoacrossClause>())
6779 emitRestoreIP(CGF&: *this, C: DC, AllocaIP, OMPBuilder);
6780 return;
6781 }
6782
6783 if (S.hasClausesOfKind<OMPDependClause>()) {
6784 for (const auto *DC : S.getClausesOfKind<OMPDependClause>())
6785 CGM.getOpenMPRuntime().emitDoacrossOrdered(CGF&: *this, C: DC);
6786 } else if (S.hasClausesOfKind<OMPDoacrossClause>()) {
6787 for (const auto *DC : S.getClausesOfKind<OMPDoacrossClause>())
6788 CGM.getOpenMPRuntime().emitDoacrossOrdered(CGF&: *this, C: DC);
6789 }
6790}
6791
6792void CodeGenFunction::EmitOMPOrderedBlockAssocDirective(
6793 const OMPOrderedBlockAssocDirective &S) {
6794 if (CGM.getLangOpts().OpenMPIRBuilder) {
6795 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
6796 using InsertPointTy = llvm::OpenMPIRBuilder::InsertPointTy;
6797
6798 // The ordered directive with threads or simd clause, or without clause.
6799 // Without clause, it behaves as if the threads clause is specified.
6800 const auto *C = S.getSingleClause<OMPSIMDClause>();
6801
6802 auto FiniCB = [this](InsertPointTy IP) {
6803 OMPBuilderCBHelpers::FinalizeOMPRegion(CGF&: *this, IP);
6804 return llvm::Error::success();
6805 };
6806
6807 auto BodyGenCB = [&S, C, this](InsertPointTy AllocIP,
6808 InsertPointTy CodeGenIP,
6809 ArrayRef<llvm::BasicBlock *> DeallocBlocks) {
6810 Builder.restoreIP(IP: CodeGenIP);
6811
6812 const CapturedStmt *CS = S.getInnermostCapturedStmt();
6813 if (C) {
6814 llvm::BasicBlock *CodeGenBB = CodeGenIP.getNodeParent();
6815 llvm::BasicBlock *FiniBB = splitBBWithSuffix(
6816 Builder, /*CreateBranch=*/false, Suffix: ".ordered.after");
6817 llvm::SmallVector<llvm::Value *, 16> CapturedVars;
6818 GenerateOpenMPCapturedVars(S: *CS, CapturedVars);
6819 llvm::Function *OutlinedFn = emitOutlinedOrderedFunction(CGM, S: CS, D: S);
6820 assert(S.getBeginLoc().isValid() &&
6821 "Outlined function call location must be valid.");
6822 ApplyDebugLocation::CreateDefaultArtificial(CGF&: *this, TemporaryLocation: S.getBeginLoc());
6823 OMPBuilderCBHelpers::EmitCaptureStmt(CGF&: *this, CodeGenIPBB: CodeGenBB, FiniBB&: *FiniBB,
6824 Fn: OutlinedFn, Args: CapturedVars);
6825 } else {
6826 OMPBuilderCBHelpers::EmitOMPInlinedRegionBody(
6827 CGF&: *this, RegionBodyStmt: CS->getCapturedStmt(), AllocaIP: AllocIP, CodeGenIP, RegionName: "ordered");
6828 }
6829 return llvm::Error::success();
6830 };
6831
6832 OMPLexicalScope Scope(*this, S, OMPD_unknown);
6833 llvm::OpenMPIRBuilder::InsertPointTy AfterIP = cantFail(
6834 ValOrErr: OMPBuilder.createOrderedThreadsSimd(Loc: Builder, BodyGenCB, FiniCB, IsThreads: !C));
6835 Builder.restoreIP(IP: AfterIP);
6836 return;
6837 }
6838
6839 const auto *C = S.getSingleClause<OMPSIMDClause>();
6840 auto &&CodeGen = [&S, C, this](CodeGenFunction &CGF,
6841 PrePostActionTy &Action) {
6842 const CapturedStmt *CS = S.getInnermostCapturedStmt();
6843 if (C) {
6844 llvm::SmallVector<llvm::Value *, 16> CapturedVars;
6845 CGF.GenerateOpenMPCapturedVars(S: *CS, CapturedVars);
6846 llvm::Function *OutlinedFn = emitOutlinedOrderedFunction(CGM, S: CS, D: S);
6847 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc: S.getBeginLoc(),
6848 OutlinedFn, Args: CapturedVars);
6849 } else {
6850 Action.Enter(CGF);
6851 CGF.EmitStmt(S: CS->getCapturedStmt());
6852 }
6853 };
6854 OMPLexicalScope Scope(*this, S, OMPD_unknown);
6855 CGM.getOpenMPRuntime().emitOrderedRegion(CGF&: *this, OrderedOpGen: CodeGen, Loc: S.getBeginLoc(), IsThreads: !C);
6856}
6857
6858static llvm::Value *convertToScalarValue(CodeGenFunction &CGF, RValue Val,
6859 QualType SrcType, QualType DestType,
6860 SourceLocation Loc) {
6861 assert(CGF.hasScalarEvaluationKind(DestType) &&
6862 "DestType must have scalar evaluation kind.");
6863 assert(!Val.isAggregate() && "Must be a scalar or complex.");
6864 return Val.isScalar() ? CGF.EmitScalarConversion(Src: Val.getScalarVal(), SrcTy: SrcType,
6865 DstTy: DestType, Loc)
6866 : CGF.EmitComplexToScalarConversion(
6867 Src: Val.getComplexVal(), SrcTy: SrcType, DstTy: DestType, Loc);
6868}
6869
6870static CodeGenFunction::ComplexPairTy
6871convertToComplexValue(CodeGenFunction &CGF, RValue Val, QualType SrcType,
6872 QualType DestType, SourceLocation Loc) {
6873 assert(CGF.getEvaluationKind(DestType) == TEK_Complex &&
6874 "DestType must have complex evaluation kind.");
6875 CodeGenFunction::ComplexPairTy ComplexVal;
6876 if (Val.isScalar()) {
6877 // Convert the input element to the element type of the complex.
6878 QualType DestElementType =
6879 DestType->castAs<ComplexType>()->getElementType();
6880 llvm::Value *ScalarVal = CGF.EmitScalarConversion(
6881 Src: Val.getScalarVal(), SrcTy: SrcType, DstTy: DestElementType, Loc);
6882 ComplexVal = CodeGenFunction::ComplexPairTy(
6883 ScalarVal, llvm::Constant::getNullValue(Ty: ScalarVal->getType()));
6884 } else {
6885 assert(Val.isComplex() && "Must be a scalar or complex.");
6886 QualType SrcElementType = SrcType->castAs<ComplexType>()->getElementType();
6887 QualType DestElementType =
6888 DestType->castAs<ComplexType>()->getElementType();
6889 ComplexVal.first = CGF.EmitScalarConversion(
6890 Src: Val.getComplexVal().first, SrcTy: SrcElementType, DstTy: DestElementType, Loc);
6891 ComplexVal.second = CGF.EmitScalarConversion(
6892 Src: Val.getComplexVal().second, SrcTy: SrcElementType, DstTy: DestElementType, Loc);
6893 }
6894 return ComplexVal;
6895}
6896
6897static void emitSimpleAtomicStore(CodeGenFunction &CGF, llvm::AtomicOrdering AO,
6898 LValue LVal, RValue RVal) {
6899 if (LVal.isGlobalReg())
6900 CGF.EmitStoreThroughGlobalRegLValue(Src: RVal, Dst: LVal);
6901 else
6902 CGF.EmitAtomicStore(rvalue: RVal, lvalue: LVal, AO, IsVolatile: LVal.isVolatile(), /*isInit=*/false);
6903}
6904
6905static RValue emitSimpleAtomicLoad(CodeGenFunction &CGF,
6906 llvm::AtomicOrdering AO, LValue LVal,
6907 SourceLocation Loc) {
6908 if (LVal.isGlobalReg())
6909 return CGF.EmitLoadOfLValue(V: LVal, Loc);
6910 return CGF.EmitAtomicLoad(
6911 lvalue: LVal, loc: Loc, AO: llvm::AtomicCmpXchgInst::getStrongestFailureOrdering(SuccessOrdering: AO),
6912 IsVolatile: LVal.isVolatile());
6913}
6914
6915void CodeGenFunction::emitOMPSimpleStore(LValue LVal, RValue RVal,
6916 QualType RValTy, SourceLocation Loc) {
6917 switch (getEvaluationKind(T: LVal.getType())) {
6918 case TEK_Scalar:
6919 EmitStoreThroughLValue(Src: RValue::get(V: convertToScalarValue(
6920 CGF&: *this, Val: RVal, SrcType: RValTy, DestType: LVal.getType(), Loc)),
6921 Dst: LVal);
6922 break;
6923 case TEK_Complex:
6924 EmitStoreOfComplex(
6925 V: convertToComplexValue(CGF&: *this, Val: RVal, SrcType: RValTy, DestType: LVal.getType(), Loc), dest: LVal,
6926 /*isInit=*/false);
6927 break;
6928 case TEK_Aggregate:
6929 llvm_unreachable("Must be a scalar or complex.");
6930 }
6931}
6932
6933static void emitOMPAtomicReadExpr(CodeGenFunction &CGF, llvm::AtomicOrdering AO,
6934 const Expr *X, const Expr *V,
6935 SourceLocation Loc) {
6936 // v = x;
6937 assert(V->isLValue() && "V of 'omp atomic read' is not lvalue");
6938 assert(X->isLValue() && "X of 'omp atomic read' is not lvalue");
6939 LValue XLValue = CGF.EmitLValue(E: X);
6940 LValue VLValue = CGF.EmitLValue(E: V);
6941 RValue Res = emitSimpleAtomicLoad(CGF, AO, LVal: XLValue, Loc);
6942 // OpenMP, 2.17.7, atomic Construct
6943 // If the read or capture clause is specified and the acquire, acq_rel, or
6944 // seq_cst clause is specified then the strong flush on exit from the atomic
6945 // operation is also an acquire flush.
6946 switch (AO) {
6947 case llvm::AtomicOrdering::Acquire:
6948 case llvm::AtomicOrdering::AcquireRelease:
6949 case llvm::AtomicOrdering::SequentiallyConsistent:
6950 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, Vars: {}, Loc,
6951 AO: llvm::AtomicOrdering::Acquire);
6952 break;
6953 case llvm::AtomicOrdering::Monotonic:
6954 case llvm::AtomicOrdering::Release:
6955 break;
6956 case llvm::AtomicOrdering::NotAtomic:
6957 case llvm::AtomicOrdering::Unordered:
6958 llvm_unreachable("Unexpected ordering.");
6959 }
6960 CGF.emitOMPSimpleStore(LVal: VLValue, RVal: Res, RValTy: X->getType().getNonReferenceType(), Loc);
6961 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: V);
6962}
6963
6964static void emitOMPAtomicWriteExpr(CodeGenFunction &CGF,
6965 llvm::AtomicOrdering AO, const Expr *X,
6966 const Expr *E, SourceLocation Loc) {
6967 // x = expr;
6968 assert(X->isLValue() && "X of 'omp atomic write' is not lvalue");
6969 emitSimpleAtomicStore(CGF, AO, LVal: CGF.EmitLValue(E: X), RVal: CGF.EmitAnyExpr(E));
6970 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: X);
6971 // OpenMP, 2.17.7, atomic Construct
6972 // If the write, update, or capture clause is specified and the release,
6973 // acq_rel, or seq_cst clause is specified then the strong flush on entry to
6974 // the atomic operation is also a release flush.
6975 switch (AO) {
6976 case llvm::AtomicOrdering::Release:
6977 case llvm::AtomicOrdering::AcquireRelease:
6978 case llvm::AtomicOrdering::SequentiallyConsistent:
6979 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, Vars: {}, Loc,
6980 AO: llvm::AtomicOrdering::Release);
6981 break;
6982 case llvm::AtomicOrdering::Acquire:
6983 case llvm::AtomicOrdering::Monotonic:
6984 break;
6985 case llvm::AtomicOrdering::NotAtomic:
6986 case llvm::AtomicOrdering::Unordered:
6987 llvm_unreachable("Unexpected ordering.");
6988 }
6989}
6990
6991static std::pair<bool, RValue> emitOMPAtomicRMW(CodeGenFunction &CGF, LValue X,
6992 RValue Update,
6993 BinaryOperatorKind BO,
6994 llvm::AtomicOrdering AO,
6995 bool IsXLHSInRHSPart) {
6996 ASTContext &Context = CGF.getContext();
6997 // Allow atomicrmw only if 'x' and 'update' are integer values, lvalue for 'x'
6998 // expression is simple and atomic is allowed for the given type for the
6999 // target platform.
7000 if (BO == BO_Comma || !Update.isScalar() || !X.isSimple() ||
7001 (!isa<llvm::ConstantInt>(Val: Update.getScalarVal()) &&
7002 (Update.getScalarVal()->getType() != X.getAddress().getElementType())) ||
7003 !Context.getTargetInfo().hasBuiltinAtomic(
7004 AtomicSizeInBits: Context.getTypeSize(T: X.getType()), AlignmentInBits: Context.toBits(CharSize: X.getAlignment())))
7005 return std::make_pair(x: false, y: RValue::get(V: nullptr));
7006
7007 auto &&CheckAtomicSupport = [&CGF](llvm::Type *T, BinaryOperatorKind BO) {
7008 if (T->isIntegerTy())
7009 return true;
7010
7011 if (T->isFloatingPointTy() && (BO == BO_Add || BO == BO_Sub))
7012 return llvm::isPowerOf2_64(Value: CGF.CGM.getDataLayout().getTypeStoreSize(Ty: T));
7013
7014 return false;
7015 };
7016
7017 if (!CheckAtomicSupport(Update.getScalarVal()->getType(), BO) ||
7018 !CheckAtomicSupport(X.getAddress().getElementType(), BO))
7019 return std::make_pair(x: false, y: RValue::get(V: nullptr));
7020
7021 bool IsInteger = X.getAddress().getElementType()->isIntegerTy();
7022 llvm::AtomicRMWInst::BinOp RMWOp;
7023 switch (BO) {
7024 case BO_Add:
7025 RMWOp = IsInteger ? llvm::AtomicRMWInst::Add : llvm::AtomicRMWInst::FAdd;
7026 break;
7027 case BO_Sub:
7028 if (!IsXLHSInRHSPart)
7029 return std::make_pair(x: false, y: RValue::get(V: nullptr));
7030 RMWOp = IsInteger ? llvm::AtomicRMWInst::Sub : llvm::AtomicRMWInst::FSub;
7031 break;
7032 case BO_And:
7033 RMWOp = llvm::AtomicRMWInst::And;
7034 break;
7035 case BO_Or:
7036 RMWOp = llvm::AtomicRMWInst::Or;
7037 break;
7038 case BO_Xor:
7039 RMWOp = llvm::AtomicRMWInst::Xor;
7040 break;
7041 case BO_LT:
7042 if (IsInteger)
7043 RMWOp = X.getType()->hasSignedIntegerRepresentation()
7044 ? (IsXLHSInRHSPart ? llvm::AtomicRMWInst::Min
7045 : llvm::AtomicRMWInst::Max)
7046 : (IsXLHSInRHSPart ? llvm::AtomicRMWInst::UMin
7047 : llvm::AtomicRMWInst::UMax);
7048 else
7049 RMWOp = IsXLHSInRHSPart ? llvm::AtomicRMWInst::FMin
7050 : llvm::AtomicRMWInst::FMax;
7051 break;
7052 case BO_GT:
7053 if (IsInteger)
7054 RMWOp = X.getType()->hasSignedIntegerRepresentation()
7055 ? (IsXLHSInRHSPart ? llvm::AtomicRMWInst::Max
7056 : llvm::AtomicRMWInst::Min)
7057 : (IsXLHSInRHSPart ? llvm::AtomicRMWInst::UMax
7058 : llvm::AtomicRMWInst::UMin);
7059 else
7060 RMWOp = IsXLHSInRHSPart ? llvm::AtomicRMWInst::FMax
7061 : llvm::AtomicRMWInst::FMin;
7062 break;
7063 case BO_Assign:
7064 RMWOp = llvm::AtomicRMWInst::Xchg;
7065 break;
7066 case BO_Mul:
7067 case BO_Div:
7068 case BO_Rem:
7069 case BO_Shl:
7070 case BO_Shr:
7071 case BO_LAnd:
7072 case BO_LOr:
7073 return std::make_pair(x: false, y: RValue::get(V: nullptr));
7074 case BO_PtrMemD:
7075 case BO_PtrMemI:
7076 case BO_LE:
7077 case BO_GE:
7078 case BO_EQ:
7079 case BO_NE:
7080 case BO_Cmp:
7081 case BO_AddAssign:
7082 case BO_SubAssign:
7083 case BO_AndAssign:
7084 case BO_OrAssign:
7085 case BO_XorAssign:
7086 case BO_MulAssign:
7087 case BO_DivAssign:
7088 case BO_RemAssign:
7089 case BO_ShlAssign:
7090 case BO_ShrAssign:
7091 case BO_Comma:
7092 llvm_unreachable("Unsupported atomic update operation");
7093 }
7094 llvm::Value *UpdateVal = Update.getScalarVal();
7095 if (auto *IC = dyn_cast<llvm::ConstantInt>(Val: UpdateVal)) {
7096 if (IsInteger)
7097 UpdateVal = CGF.Builder.CreateIntCast(
7098 V: IC, DestTy: X.getAddress().getElementType(),
7099 isSigned: X.getType()->hasSignedIntegerRepresentation());
7100 else
7101 UpdateVal = CGF.Builder.CreateCast(Op: llvm::Instruction::CastOps::UIToFP, V: IC,
7102 DestTy: X.getAddress().getElementType());
7103 }
7104 llvm::AtomicRMWInst *Res =
7105 CGF.emitAtomicRMWInst(Op: RMWOp, Addr: X.getAddress(), Val: UpdateVal, Order: AO);
7106 return std::make_pair(x: true, y: RValue::get(V: Res));
7107}
7108
7109std::pair<bool, RValue> CodeGenFunction::EmitOMPAtomicSimpleUpdateExpr(
7110 LValue X, RValue E, BinaryOperatorKind BO, bool IsXLHSInRHSPart,
7111 llvm::AtomicOrdering AO, SourceLocation Loc,
7112 const llvm::function_ref<RValue(RValue)> CommonGen) {
7113 // Update expressions are allowed to have the following forms:
7114 // x binop= expr; -> xrval + expr;
7115 // x++, ++x -> xrval + 1;
7116 // x--, --x -> xrval - 1;
7117 // x = x binop expr; -> xrval binop expr
7118 // x = expr Op x; - > expr binop xrval;
7119 auto Res = emitOMPAtomicRMW(CGF&: *this, X, Update: E, BO, AO, IsXLHSInRHSPart);
7120 if (!Res.first) {
7121 if (X.isGlobalReg()) {
7122 // Emit an update expression: 'xrval' binop 'expr' or 'expr' binop
7123 // 'xrval'.
7124 EmitStoreThroughLValue(Src: CommonGen(EmitLoadOfLValue(V: X, Loc)), Dst: X);
7125 } else {
7126 // Perform compare-and-swap procedure.
7127 EmitAtomicUpdate(LVal: X, AO, UpdateOp: CommonGen, IsVolatile: X.getType().isVolatileQualified());
7128 }
7129 }
7130 return Res;
7131}
7132
7133static void emitOMPAtomicUpdateExpr(CodeGenFunction &CGF,
7134 llvm::AtomicOrdering AO, const Expr *X,
7135 const Expr *E, const Expr *UE,
7136 bool IsXLHSInRHSPart, SourceLocation Loc) {
7137 assert(isa<BinaryOperator>(UE->IgnoreImpCasts()) &&
7138 "Update expr in 'atomic update' must be a binary operator.");
7139 const auto *BOUE = cast<BinaryOperator>(Val: UE->IgnoreImpCasts());
7140 // Update expressions are allowed to have the following forms:
7141 // x binop= expr; -> xrval + expr;
7142 // x++, ++x -> xrval + 1;
7143 // x--, --x -> xrval - 1;
7144 // x = x binop expr; -> xrval binop expr
7145 // x = expr Op x; - > expr binop xrval;
7146 assert(X->isLValue() && "X of 'omp atomic update' is not lvalue");
7147 LValue XLValue = CGF.EmitLValue(E: X);
7148 RValue ExprRValue = CGF.EmitAnyExpr(E);
7149 const auto *LHS = cast<OpaqueValueExpr>(Val: BOUE->getLHS()->IgnoreImpCasts());
7150 const auto *RHS = cast<OpaqueValueExpr>(Val: BOUE->getRHS()->IgnoreImpCasts());
7151 const OpaqueValueExpr *XRValExpr = IsXLHSInRHSPart ? LHS : RHS;
7152 const OpaqueValueExpr *ERValExpr = IsXLHSInRHSPart ? RHS : LHS;
7153 auto &&Gen = [&CGF, UE, ExprRValue, XRValExpr, ERValExpr](RValue XRValue) {
7154 CodeGenFunction::OpaqueValueMapping MapExpr(CGF, ERValExpr, ExprRValue);
7155 CodeGenFunction::OpaqueValueMapping MapX(CGF, XRValExpr, XRValue);
7156 return CGF.EmitAnyExpr(E: UE);
7157 };
7158 (void)CGF.EmitOMPAtomicSimpleUpdateExpr(
7159 X: XLValue, E: ExprRValue, BO: BOUE->getOpcode(), IsXLHSInRHSPart, AO, Loc, CommonGen: Gen);
7160 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: X);
7161 // OpenMP, 2.17.7, atomic Construct
7162 // If the write, update, or capture clause is specified and the release,
7163 // acq_rel, or seq_cst clause is specified then the strong flush on entry to
7164 // the atomic operation is also a release flush.
7165 switch (AO) {
7166 case llvm::AtomicOrdering::Release:
7167 case llvm::AtomicOrdering::AcquireRelease:
7168 case llvm::AtomicOrdering::SequentiallyConsistent:
7169 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, Vars: {}, Loc,
7170 AO: llvm::AtomicOrdering::Release);
7171 break;
7172 case llvm::AtomicOrdering::Acquire:
7173 case llvm::AtomicOrdering::Monotonic:
7174 break;
7175 case llvm::AtomicOrdering::NotAtomic:
7176 case llvm::AtomicOrdering::Unordered:
7177 llvm_unreachable("Unexpected ordering.");
7178 }
7179}
7180
7181static RValue convertToType(CodeGenFunction &CGF, RValue Value,
7182 QualType SourceType, QualType ResType,
7183 SourceLocation Loc) {
7184 switch (CGF.getEvaluationKind(T: ResType)) {
7185 case TEK_Scalar:
7186 return RValue::get(
7187 V: convertToScalarValue(CGF, Val: Value, SrcType: SourceType, DestType: ResType, Loc));
7188 case TEK_Complex: {
7189 auto Res = convertToComplexValue(CGF, Val: Value, SrcType: SourceType, DestType: ResType, Loc);
7190 return RValue::getComplex(V1: Res.first, V2: Res.second);
7191 }
7192 case TEK_Aggregate:
7193 break;
7194 }
7195 llvm_unreachable("Must be a scalar or complex.");
7196}
7197
7198static void emitOMPAtomicCaptureExpr(CodeGenFunction &CGF,
7199 llvm::AtomicOrdering AO,
7200 bool IsPostfixUpdate, const Expr *V,
7201 const Expr *X, const Expr *E,
7202 const Expr *UE, bool IsXLHSInRHSPart,
7203 SourceLocation Loc) {
7204 assert(X->isLValue() && "X of 'omp atomic capture' is not lvalue");
7205 assert(V->isLValue() && "V of 'omp atomic capture' is not lvalue");
7206 RValue NewVVal;
7207 LValue VLValue = CGF.EmitLValue(E: V);
7208 LValue XLValue = CGF.EmitLValue(E: X);
7209 RValue ExprRValue = CGF.EmitAnyExpr(E);
7210 QualType NewVValType;
7211 if (UE) {
7212 // 'x' is updated with some additional value.
7213 assert(isa<BinaryOperator>(UE->IgnoreImpCasts()) &&
7214 "Update expr in 'atomic capture' must be a binary operator.");
7215 const auto *BOUE = cast<BinaryOperator>(Val: UE->IgnoreImpCasts());
7216 // Update expressions are allowed to have the following forms:
7217 // x binop= expr; -> xrval + expr;
7218 // x++, ++x -> xrval + 1;
7219 // x--, --x -> xrval - 1;
7220 // x = x binop expr; -> xrval binop expr
7221 // x = expr Op x; - > expr binop xrval;
7222 const auto *LHS = cast<OpaqueValueExpr>(Val: BOUE->getLHS()->IgnoreImpCasts());
7223 const auto *RHS = cast<OpaqueValueExpr>(Val: BOUE->getRHS()->IgnoreImpCasts());
7224 const OpaqueValueExpr *XRValExpr = IsXLHSInRHSPart ? LHS : RHS;
7225 NewVValType = XRValExpr->getType();
7226 const OpaqueValueExpr *ERValExpr = IsXLHSInRHSPart ? RHS : LHS;
7227 auto &&Gen = [&CGF, &NewVVal, UE, ExprRValue, XRValExpr, ERValExpr,
7228 IsPostfixUpdate](RValue XRValue) {
7229 CodeGenFunction::OpaqueValueMapping MapExpr(CGF, ERValExpr, ExprRValue);
7230 CodeGenFunction::OpaqueValueMapping MapX(CGF, XRValExpr, XRValue);
7231 RValue Res = CGF.EmitAnyExpr(E: UE);
7232 NewVVal = IsPostfixUpdate ? XRValue : Res;
7233 return Res;
7234 };
7235 auto Res = CGF.EmitOMPAtomicSimpleUpdateExpr(
7236 X: XLValue, E: ExprRValue, BO: BOUE->getOpcode(), IsXLHSInRHSPart, AO, Loc, CommonGen: Gen);
7237 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: X);
7238 if (Res.first) {
7239 // 'atomicrmw' instruction was generated.
7240 if (IsPostfixUpdate) {
7241 // Use old value from 'atomicrmw'.
7242 NewVVal = Res.second;
7243 } else {
7244 // 'atomicrmw' does not provide new value, so evaluate it using old
7245 // value of 'x'.
7246 CodeGenFunction::OpaqueValueMapping MapExpr(CGF, ERValExpr, ExprRValue);
7247 CodeGenFunction::OpaqueValueMapping MapX(CGF, XRValExpr, Res.second);
7248 NewVVal = CGF.EmitAnyExpr(E: UE);
7249 }
7250 }
7251 } else {
7252 // 'x' is simply rewritten with some 'expr'.
7253 NewVValType = X->getType().getNonReferenceType();
7254 ExprRValue = convertToType(CGF, Value: ExprRValue, SourceType: E->getType(),
7255 ResType: X->getType().getNonReferenceType(), Loc);
7256 auto &&Gen = [&NewVVal, ExprRValue](RValue XRValue) {
7257 NewVVal = XRValue;
7258 return ExprRValue;
7259 };
7260 // Try to perform atomicrmw xchg, otherwise simple exchange.
7261 auto Res = CGF.EmitOMPAtomicSimpleUpdateExpr(
7262 X: XLValue, E: ExprRValue, /*BO=*/BO_Assign, /*IsXLHSInRHSPart=*/false, AO,
7263 Loc, CommonGen: Gen);
7264 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: X);
7265 if (Res.first) {
7266 // 'atomicrmw' instruction was generated.
7267 NewVVal = IsPostfixUpdate ? Res.second : ExprRValue;
7268 }
7269 }
7270 // Emit post-update store to 'v' of old/new 'x' value.
7271 CGF.emitOMPSimpleStore(LVal: VLValue, RVal: NewVVal, RValTy: NewVValType, Loc);
7272 CGF.CGM.getOpenMPRuntime().checkAndEmitLastprivateConditional(CGF, LHS: V);
7273 // OpenMP 5.1 removes the required flush for capture clause.
7274 if (CGF.CGM.getLangOpts().OpenMP < 51) {
7275 // OpenMP, 2.17.7, atomic Construct
7276 // If the write, update, or capture clause is specified and the release,
7277 // acq_rel, or seq_cst clause is specified then the strong flush on entry to
7278 // the atomic operation is also a release flush.
7279 // If the read or capture clause is specified and the acquire, acq_rel, or
7280 // seq_cst clause is specified then the strong flush on exit from the atomic
7281 // operation is also an acquire flush.
7282 switch (AO) {
7283 case llvm::AtomicOrdering::Release:
7284 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, Vars: {}, Loc,
7285 AO: llvm::AtomicOrdering::Release);
7286 break;
7287 case llvm::AtomicOrdering::Acquire:
7288 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, Vars: {}, Loc,
7289 AO: llvm::AtomicOrdering::Acquire);
7290 break;
7291 case llvm::AtomicOrdering::AcquireRelease:
7292 case llvm::AtomicOrdering::SequentiallyConsistent:
7293 CGF.CGM.getOpenMPRuntime().emitFlush(
7294 CGF, Vars: {}, Loc, AO: llvm::AtomicOrdering::AcquireRelease);
7295 break;
7296 case llvm::AtomicOrdering::Monotonic:
7297 break;
7298 case llvm::AtomicOrdering::NotAtomic:
7299 case llvm::AtomicOrdering::Unordered:
7300 llvm_unreachable("Unexpected ordering.");
7301 }
7302 }
7303}
7304
7305static void emitOMPAtomicCompareExpr(
7306 CodeGenFunction &CGF, llvm::AtomicOrdering AO, llvm::AtomicOrdering FailAO,
7307 const Expr *X, const Expr *V, const Expr *R, const Expr *E, const Expr *D,
7308 const Expr *CE, bool IsXBinopExpr, bool IsPostfixUpdate, bool IsFailOnly,
7309 SourceLocation Loc) {
7310 llvm::OpenMPIRBuilder &OMPBuilder =
7311 CGF.CGM.getOpenMPRuntime().getOMPBuilder();
7312
7313 OMPAtomicCompareOp Op;
7314 assert(isa<BinaryOperator>(CE) && "CE is not a BinaryOperator");
7315 switch (cast<BinaryOperator>(Val: CE)->getOpcode()) {
7316 case BO_EQ:
7317 Op = OMPAtomicCompareOp::EQ;
7318 break;
7319 case BO_LT:
7320 Op = OMPAtomicCompareOp::MIN;
7321 break;
7322 case BO_GT:
7323 Op = OMPAtomicCompareOp::MAX;
7324 break;
7325 default:
7326 llvm_unreachable("unsupported atomic compare binary operator");
7327 }
7328
7329 LValue XLVal = CGF.EmitLValue(E: X);
7330 Address XAddr = XLVal.getAddress();
7331
7332 auto EmitRValueWithCastIfNeeded = [&CGF, Loc](const Expr *X, const Expr *E) {
7333 if (X->getType() == E->getType())
7334 return CGF.EmitScalarExpr(E);
7335 const Expr *NewE = E->IgnoreImplicitAsWritten();
7336 llvm::Value *V = CGF.EmitScalarExpr(E: NewE);
7337 if (NewE->getType() == X->getType())
7338 return V;
7339 return CGF.EmitScalarConversion(Src: V, SrcTy: NewE->getType(), DstTy: X->getType(), Loc);
7340 };
7341
7342 llvm::Value *EVal = EmitRValueWithCastIfNeeded(X, E);
7343 llvm::Value *DVal = D ? EmitRValueWithCastIfNeeded(X, D) : nullptr;
7344 if (auto *CI = dyn_cast<llvm::ConstantInt>(Val: EVal))
7345 EVal = CGF.Builder.CreateIntCast(
7346 V: CI, DestTy: XLVal.getAddress().getElementType(),
7347 isSigned: E->getType()->hasSignedIntegerRepresentation());
7348 if (DVal)
7349 if (auto *CI = dyn_cast<llvm::ConstantInt>(Val: DVal))
7350 DVal = CGF.Builder.CreateIntCast(
7351 V: CI, DestTy: XLVal.getAddress().getElementType(),
7352 isSigned: D->getType()->hasSignedIntegerRepresentation());
7353
7354 llvm::OpenMPIRBuilder::AtomicOpValue XOpVal{
7355 .Var: XAddr.emitRawPointer(CGF), .ElemTy: XAddr.getElementType(),
7356 .IsSigned: X->getType()->hasSignedIntegerRepresentation(),
7357 .IsVolatile: X->getType().isVolatileQualified()};
7358 llvm::OpenMPIRBuilder::AtomicOpValue VOpVal, ROpVal;
7359 if (V) {
7360 LValue LV = CGF.EmitLValue(E: V);
7361 Address Addr = LV.getAddress();
7362 VOpVal = {.Var: Addr.emitRawPointer(CGF), .ElemTy: Addr.getElementType(),
7363 .IsSigned: V->getType()->hasSignedIntegerRepresentation(),
7364 .IsVolatile: V->getType().isVolatileQualified()};
7365 }
7366 if (R) {
7367 LValue LV = CGF.EmitLValue(E: R);
7368 Address Addr = LV.getAddress();
7369 ROpVal = {.Var: Addr.emitRawPointer(CGF), .ElemTy: Addr.getElementType(),
7370 .IsSigned: R->getType()->hasSignedIntegerRepresentation(),
7371 .IsVolatile: R->getType().isVolatileQualified()};
7372 }
7373
7374 if (FailAO == llvm::AtomicOrdering::NotAtomic) {
7375 // fail clause was not mentioned on the
7376 // "#pragma omp atomic compare" construct.
7377 CGF.Builder.restoreIP(IP: OMPBuilder.createAtomicCompare(
7378 Loc: CGF.Builder, X&: XOpVal, V&: VOpVal, R&: ROpVal, E: EVal, D: DVal, AO, Op, IsXBinopExpr,
7379 IsPostfixUpdate, IsFailOnly));
7380 } else
7381 CGF.Builder.restoreIP(IP: OMPBuilder.createAtomicCompare(
7382 Loc: CGF.Builder, X&: XOpVal, V&: VOpVal, R&: ROpVal, E: EVal, D: DVal, AO, Op, IsXBinopExpr,
7383 IsPostfixUpdate, IsFailOnly, Failure: FailAO));
7384}
7385
7386static void emitOMPAtomicExpr(CodeGenFunction &CGF, OpenMPClauseKind Kind,
7387 llvm::AtomicOrdering AO,
7388 llvm::AtomicOrdering FailAO, bool IsPostfixUpdate,
7389 const Expr *X, const Expr *V, const Expr *R,
7390 const Expr *E, const Expr *UE, const Expr *D,
7391 const Expr *CE, bool IsXLHSInRHSPart,
7392 bool IsFailOnly, SourceLocation Loc) {
7393 switch (Kind) {
7394 case OMPC_read:
7395 emitOMPAtomicReadExpr(CGF, AO, X, V, Loc);
7396 break;
7397 case OMPC_write:
7398 emitOMPAtomicWriteExpr(CGF, AO, X, E, Loc);
7399 break;
7400 case OMPC_unknown:
7401 case OMPC_update:
7402 emitOMPAtomicUpdateExpr(CGF, AO, X, E, UE, IsXLHSInRHSPart, Loc);
7403 break;
7404 case OMPC_capture:
7405 emitOMPAtomicCaptureExpr(CGF, AO, IsPostfixUpdate, V, X, E, UE,
7406 IsXLHSInRHSPart, Loc);
7407 break;
7408 case OMPC_compare: {
7409 emitOMPAtomicCompareExpr(CGF, AO, FailAO, X, V, R, E, D, CE,
7410 IsXBinopExpr: IsXLHSInRHSPart, IsPostfixUpdate, IsFailOnly, Loc);
7411 break;
7412 }
7413 default:
7414 llvm_unreachable("Clause is not allowed in 'omp atomic'.");
7415 }
7416}
7417
7418void CodeGenFunction::EmitOMPAtomicDirective(const OMPAtomicDirective &S) {
7419 llvm::AtomicOrdering AO = CGM.getOpenMPRuntime().getDefaultMemoryOrdering();
7420 // Fail Memory Clause Ordering.
7421 llvm::AtomicOrdering FailAO = llvm::AtomicOrdering::NotAtomic;
7422 bool MemOrderingSpecified = false;
7423 if (S.getSingleClause<OMPSeqCstClause>()) {
7424 AO = llvm::AtomicOrdering::SequentiallyConsistent;
7425 MemOrderingSpecified = true;
7426 } else if (S.getSingleClause<OMPAcqRelClause>()) {
7427 AO = llvm::AtomicOrdering::AcquireRelease;
7428 MemOrderingSpecified = true;
7429 } else if (S.getSingleClause<OMPAcquireClause>()) {
7430 AO = llvm::AtomicOrdering::Acquire;
7431 MemOrderingSpecified = true;
7432 } else if (S.getSingleClause<OMPReleaseClause>()) {
7433 AO = llvm::AtomicOrdering::Release;
7434 MemOrderingSpecified = true;
7435 } else if (S.getSingleClause<OMPRelaxedClause>()) {
7436 AO = llvm::AtomicOrdering::Monotonic;
7437 MemOrderingSpecified = true;
7438 }
7439 llvm::SmallSet<OpenMPClauseKind, 2> KindsEncountered;
7440 OpenMPClauseKind Kind = OMPC_unknown;
7441 for (const OMPClause *C : S.clauses()) {
7442 // Find first clause (skip seq_cst|acq_rel|aqcuire|release|relaxed clause,
7443 // if it is first).
7444 OpenMPClauseKind K = C->getClauseKind();
7445 // TBD
7446 if (K == OMPC_weak)
7447 return;
7448 if (K == OMPC_seq_cst || K == OMPC_acq_rel || K == OMPC_acquire ||
7449 K == OMPC_release || K == OMPC_relaxed || K == OMPC_hint)
7450 continue;
7451 Kind = K;
7452 KindsEncountered.insert(V: K);
7453 }
7454 // We just need to correct Kind here. No need to set a bool saying it is
7455 // actually compare capture because we can tell from whether V and R are
7456 // nullptr.
7457 if (KindsEncountered.contains(V: OMPC_compare) &&
7458 KindsEncountered.contains(V: OMPC_capture))
7459 Kind = OMPC_compare;
7460 if (!MemOrderingSpecified) {
7461 llvm::AtomicOrdering DefaultOrder =
7462 CGM.getOpenMPRuntime().getDefaultMemoryOrdering();
7463 if (DefaultOrder == llvm::AtomicOrdering::Monotonic ||
7464 DefaultOrder == llvm::AtomicOrdering::SequentiallyConsistent ||
7465 (DefaultOrder == llvm::AtomicOrdering::AcquireRelease &&
7466 Kind == OMPC_capture)) {
7467 AO = DefaultOrder;
7468 } else if (DefaultOrder == llvm::AtomicOrdering::AcquireRelease) {
7469 if (Kind == OMPC_unknown || Kind == OMPC_update || Kind == OMPC_write) {
7470 AO = llvm::AtomicOrdering::Release;
7471 } else if (Kind == OMPC_read) {
7472 assert(Kind == OMPC_read && "Unexpected atomic kind.");
7473 AO = llvm::AtomicOrdering::Acquire;
7474 }
7475 }
7476 }
7477
7478 if (KindsEncountered.contains(V: OMPC_compare) &&
7479 KindsEncountered.contains(V: OMPC_fail)) {
7480 Kind = OMPC_compare;
7481 const auto *FailClause = S.getSingleClause<OMPFailClause>();
7482 if (FailClause) {
7483 OpenMPClauseKind FailParameter = FailClause->getFailParameter();
7484 if (FailParameter == llvm::omp::OMPC_relaxed)
7485 FailAO = llvm::AtomicOrdering::Monotonic;
7486 else if (FailParameter == llvm::omp::OMPC_acquire)
7487 FailAO = llvm::AtomicOrdering::Acquire;
7488 else if (FailParameter == llvm::omp::OMPC_seq_cst)
7489 FailAO = llvm::AtomicOrdering::SequentiallyConsistent;
7490 }
7491 }
7492
7493 LexicalScope Scope(*this, S.getSourceRange());
7494 EmitStopPoint(S: S.getAssociatedStmt());
7495 emitOMPAtomicExpr(CGF&: *this, Kind, AO, FailAO, IsPostfixUpdate: S.isPostfixUpdate(), X: S.getX(),
7496 V: S.getV(), R: S.getR(), E: S.getExpr(), UE: S.getUpdateExpr(),
7497 D: S.getD(), CE: S.getCondExpr(), IsXLHSInRHSPart: S.isXLHSInRHSPart(),
7498 IsFailOnly: S.isFailOnly(), Loc: S.getBeginLoc());
7499}
7500
7501static void emitCommonOMPTargetDirective(CodeGenFunction &CGF,
7502 const OMPExecutableDirective &S,
7503 const RegionCodeGenTy &CodeGen) {
7504 assert(isOpenMPTargetExecutionDirective(S.getDirectiveKind()));
7505 CodeGenModule &CGM = CGF.CGM;
7506
7507 // On device emit this construct as inlined code.
7508 if (CGM.getLangOpts().OpenMPIsTargetDevice) {
7509 OMPLexicalScope Scope(CGF, S, OMPD_target);
7510 CGM.getOpenMPRuntime().emitInlinedDirective(
7511 CGF, InnermostKind: OMPD_target, CodeGen: [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7512 CGF.EmitStmt(S: S.getInnermostCapturedStmt()->getCapturedStmt());
7513 });
7514 return;
7515 }
7516
7517 auto LPCRegion = CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF, S);
7518 llvm::Function *Fn = nullptr;
7519 llvm::Constant *FnID = nullptr;
7520
7521 const Expr *IfCond = nullptr;
7522 // Check for the at most one if clause associated with the target region.
7523 for (const auto *C : S.getClausesOfKind<OMPIfClause>()) {
7524 if (C->getNameModifier() == OMPD_unknown ||
7525 C->getNameModifier() == OMPD_target) {
7526 IfCond = C->getCondition();
7527 break;
7528 }
7529 }
7530
7531 // Check if we have any device clause associated with the directive.
7532 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device(
7533 nullptr, OMPC_DEVICE_unknown);
7534 if (auto *C = S.getSingleClause<OMPDeviceClause>())
7535 Device.setPointerAndInt(PtrVal: C->getDevice(), IntVal: C->getModifier());
7536
7537 // Check if we have an if clause whose conditional always evaluates to false
7538 // or if we do not have any targets specified. If so the target region is not
7539 // an offload entry point.
7540 bool IsOffloadEntry = true;
7541 if (IfCond) {
7542 bool Val;
7543 if (CGF.ConstantFoldsToSimpleInteger(Cond: IfCond, Result&: Val) && !Val)
7544 IsOffloadEntry = false;
7545 }
7546 if (CGM.getLangOpts().OMPTargetTriples.empty())
7547 IsOffloadEntry = false;
7548
7549 if (CGM.getLangOpts().OpenMPOffloadMandatory && !IsOffloadEntry) {
7550 CGM.getDiags().Report(DiagID: diag::err_missing_mandatory_offloading);
7551 }
7552
7553 StringRef ParentName;
7554 // In case we have Ctors/Dtors we use the complete type variant to produce
7555 // the mangling of the device outlined kernel. Lambdas and blocks at
7556 // namespace scope have no parent function.
7557 if (!CGF.CurFuncDecl)
7558 ParentName = CGF.CurFn->getName();
7559 else if (const auto *D = dyn_cast<CXXConstructorDecl>(Val: CGF.CurFuncDecl))
7560 ParentName = CGM.getMangledName(GD: GlobalDecl(D, Ctor_Complete));
7561 else if (const auto *D = dyn_cast<CXXDestructorDecl>(Val: CGF.CurFuncDecl))
7562 ParentName = CGM.getMangledName(GD: GlobalDecl(D, Dtor_Complete));
7563 else
7564 ParentName =
7565 CGM.getMangledName(GD: GlobalDecl(cast<FunctionDecl>(Val: CGF.CurFuncDecl)));
7566
7567 // Emit target region as a standalone region.
7568 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: FnID,
7569 IsOffloadEntry, CodeGen);
7570 OMPLexicalScope Scope(CGF, S, OMPD_task);
7571 auto &&SizeEmitter =
7572 [IsOffloadEntry](CodeGenFunction &CGF,
7573 const OMPLoopDirective &D) -> llvm::Value * {
7574 if (IsOffloadEntry) {
7575 OMPLoopScope PreInitScope(CGF, D);
7576 // Emit calculation of the iterations count.
7577 llvm::Value *NumIterations = CGF.EmitScalarExpr(E: D.getNumIterations());
7578 NumIterations = CGF.Builder.CreateIntCast(V: NumIterations, DestTy: CGF.Int64Ty,
7579 /*isSigned=*/false);
7580 return NumIterations;
7581 }
7582 return nullptr;
7583 };
7584 CGM.getOpenMPRuntime().emitTargetCall(CGF, D: S, OutlinedFn: Fn, OutlinedFnID: FnID, IfCond, Device,
7585 SizeEmitter);
7586}
7587
7588static void emitTargetRegion(CodeGenFunction &CGF, const OMPTargetDirective &S,
7589 PrePostActionTy &Action) {
7590 Action.Enter(CGF);
7591 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
7592 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
7593 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
7594 (void)PrivateScope.Privatize();
7595 if (isOpenMPTargetExecutionDirective(DKind: S.getDirectiveKind()))
7596 CGF.CGM.getOpenMPRuntime().adjustTargetSpecificDataForLambdas(CGF, D: S);
7597
7598 CGF.EmitStmt(S: S.getCapturedStmt(RegionKind: OMPD_target)->getCapturedStmt());
7599 CGF.EnsureInsertPoint();
7600}
7601
7602void CodeGenFunction::EmitOMPTargetDeviceFunction(CodeGenModule &CGM,
7603 StringRef ParentName,
7604 const OMPTargetDirective &S) {
7605 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7606 emitTargetRegion(CGF, S, Action);
7607 };
7608 llvm::Function *Fn;
7609 llvm::Constant *Addr;
7610 // Emit target region as a standalone region.
7611 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
7612 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
7613 assert(Fn && Addr && "Target device function emission failed.");
7614}
7615
7616void CodeGenFunction::EmitOMPTargetDirective(const OMPTargetDirective &S) {
7617 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7618 emitTargetRegion(CGF, S, Action);
7619 };
7620 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
7621}
7622
7623static void emitCommonOMPTeamsDirective(CodeGenFunction &CGF,
7624 const OMPExecutableDirective &S,
7625 OpenMPDirectiveKind InnermostKind,
7626 const RegionCodeGenTy &CodeGen) {
7627 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: OMPD_teams);
7628 llvm::Function *OutlinedFn =
7629 CGF.CGM.getOpenMPRuntime().emitTeamsOutlinedFunction(
7630 CGF, D: S, ThreadIDVar: *CS->getCapturedDecl()->param_begin(), InnermostKind,
7631 CodeGen);
7632
7633 OMPTeamsScope Scope(CGF, S);
7634 auto ParallelLeague = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7635 const auto *NT = S.getSingleClause<OMPNumTeamsClause>();
7636 const auto *TL = S.getSingleClause<OMPThreadLimitClause>();
7637 if (NT || TL) {
7638 const Expr *NumTeams = NT ? NT->getNumTeams().front() : nullptr;
7639 const Expr *ThreadLimit = TL ? TL->getThreadLimit().front() : nullptr;
7640
7641 CGF.CGM.getOpenMPRuntime().emitNumTeamsClause(CGF, NumTeams, ThreadLimit,
7642 Loc: S.getBeginLoc());
7643 }
7644 };
7645
7646 const Expr *IfCond = nullptr;
7647 for (const auto *C : S.getClausesOfKind<OMPIfClause>()) {
7648 if (C->getNameModifier() == OMPD_unknown ||
7649 C->getNameModifier() == OMPD_teams) {
7650 IfCond = C->getCondition();
7651 break;
7652 }
7653 }
7654 if (IfCond && CGF.CGM.getLangOpts().OpenMP >= 52) {
7655 auto SerialLeague = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7656 // OpenMP 5.2, 10.2, teams Construct
7657 // When an if clause is present on a teams construct and the if clause
7658 // expression evaluates to false, the number of created teams is one.
7659 const llvm::APInt One(32, 1);
7660 IntegerLiteral NumTeams(
7661 CGF.getContext(), One,
7662 CGF.getContext().getIntTypeForBitwidth(DestWidth: 32, /*Signed=*/0),
7663 SourceLocation());
7664 // The thread_limit clause is unaffected by the if clause.
7665 const auto *TL = S.getSingleClause<OMPThreadLimitClause>();
7666 const Expr *ThreadLimit = TL ? TL->getThreadLimit().front() : nullptr;
7667 CGF.CGM.getOpenMPRuntime().emitNumTeamsClause(CGF, NumTeams: &NumTeams, ThreadLimit,
7668 Loc: S.getBeginLoc());
7669 };
7670 CGF.CGM.getOpenMPRuntime().emitIfClause(CGF, Cond: IfCond, ThenGen: ParallelLeague,
7671 ElseGen: SerialLeague);
7672 } else {
7673 const RegionCodeGenTy ThenRCG(ParallelLeague);
7674 ThenRCG(CGF);
7675 }
7676
7677 llvm::SmallVector<llvm::Value *, 16> CapturedVars;
7678 CGF.GenerateOpenMPCapturedVars(S: *CS, CapturedVars);
7679 CGF.CGM.getOpenMPRuntime().emitTeamsCall(CGF, D: S, Loc: S.getBeginLoc(), OutlinedFn,
7680 CapturedVars);
7681}
7682
7683void CodeGenFunction::EmitOMPTeamsDirective(const OMPTeamsDirective &S) {
7684 // Emit teams region as a standalone region.
7685 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7686 Action.Enter(CGF);
7687 OMPPrivateScope PrivateScope(CGF);
7688 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
7689 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
7690 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7691 (void)PrivateScope.Privatize();
7692 CGF.EmitStmt(S: S.getCapturedStmt(RegionKind: OMPD_teams)->getCapturedStmt());
7693 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7694 };
7695 emitCommonOMPTeamsDirective(CGF&: *this, S, InnermostKind: OMPD_distribute, CodeGen);
7696 emitPostUpdateForReductionClause(CGF&: *this, D: S,
7697 CondGen: [](CodeGenFunction &) { return nullptr; });
7698}
7699
7700static void emitTargetTeamsRegion(CodeGenFunction &CGF, PrePostActionTy &Action,
7701 const OMPTargetTeamsDirective &S) {
7702 auto *CS = S.getCapturedStmt(RegionKind: OMPD_teams);
7703 Action.Enter(CGF);
7704 // Emit teams region as a standalone region.
7705 auto &&CodeGen = [&S, CS](CodeGenFunction &CGF, PrePostActionTy &Action) {
7706 Action.Enter(CGF);
7707 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
7708 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
7709 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
7710 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7711 (void)PrivateScope.Privatize();
7712 if (isOpenMPTargetExecutionDirective(DKind: S.getDirectiveKind()))
7713 CGF.CGM.getOpenMPRuntime().adjustTargetSpecificDataForLambdas(CGF, D: S);
7714 CGF.EmitStmt(S: CS->getCapturedStmt());
7715 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7716 };
7717 emitCommonOMPTeamsDirective(CGF, S, InnermostKind: OMPD_teams, CodeGen);
7718 emitPostUpdateForReductionClause(CGF, D: S,
7719 CondGen: [](CodeGenFunction &) { return nullptr; });
7720}
7721
7722void CodeGenFunction::EmitOMPTargetTeamsDeviceFunction(
7723 CodeGenModule &CGM, StringRef ParentName,
7724 const OMPTargetTeamsDirective &S) {
7725 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7726 emitTargetTeamsRegion(CGF, Action, S);
7727 };
7728 llvm::Function *Fn;
7729 llvm::Constant *Addr;
7730 // Emit target region as a standalone region.
7731 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
7732 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
7733 assert(Fn && Addr && "Target device function emission failed.");
7734}
7735
7736void CodeGenFunction::EmitOMPTargetTeamsDirective(
7737 const OMPTargetTeamsDirective &S) {
7738 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7739 emitTargetTeamsRegion(CGF, Action, S);
7740 };
7741 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
7742}
7743
7744static void
7745emitTargetTeamsDistributeRegion(CodeGenFunction &CGF, PrePostActionTy &Action,
7746 const OMPTargetTeamsDistributeDirective &S) {
7747 Action.Enter(CGF);
7748 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7749 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
7750 };
7751
7752 // Emit teams region as a standalone region.
7753 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
7754 PrePostActionTy &Action) {
7755 Action.Enter(CGF);
7756 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
7757 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7758 (void)PrivateScope.Privatize();
7759 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_distribute,
7760 CodeGen: CodeGenDistribute);
7761 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7762 };
7763 emitCommonOMPTeamsDirective(CGF, S, InnermostKind: OMPD_distribute, CodeGen);
7764 emitPostUpdateForReductionClause(CGF, D: S,
7765 CondGen: [](CodeGenFunction &) { return nullptr; });
7766}
7767
7768void CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction(
7769 CodeGenModule &CGM, StringRef ParentName,
7770 const OMPTargetTeamsDistributeDirective &S) {
7771 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7772 emitTargetTeamsDistributeRegion(CGF, Action, S);
7773 };
7774 llvm::Function *Fn;
7775 llvm::Constant *Addr;
7776 // Emit target region as a standalone region.
7777 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
7778 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
7779 assert(Fn && Addr && "Target device function emission failed.");
7780}
7781
7782void CodeGenFunction::EmitOMPTargetTeamsDistributeDirective(
7783 const OMPTargetTeamsDistributeDirective &S) {
7784 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7785 emitTargetTeamsDistributeRegion(CGF, Action, S);
7786 };
7787 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
7788}
7789
7790static void emitTargetTeamsDistributeSimdRegion(
7791 CodeGenFunction &CGF, PrePostActionTy &Action,
7792 const OMPTargetTeamsDistributeSimdDirective &S) {
7793 Action.Enter(CGF);
7794 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7795 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
7796 };
7797
7798 // Emit teams region as a standalone region.
7799 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
7800 PrePostActionTy &Action) {
7801 Action.Enter(CGF);
7802 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
7803 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7804 (void)PrivateScope.Privatize();
7805 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_distribute,
7806 CodeGen: CodeGenDistribute);
7807 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7808 };
7809 emitCommonOMPTeamsDirective(CGF, S, InnermostKind: OMPD_distribute_simd, CodeGen);
7810 emitPostUpdateForReductionClause(CGF, D: S,
7811 CondGen: [](CodeGenFunction &) { return nullptr; });
7812}
7813
7814void CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction(
7815 CodeGenModule &CGM, StringRef ParentName,
7816 const OMPTargetTeamsDistributeSimdDirective &S) {
7817 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7818 emitTargetTeamsDistributeSimdRegion(CGF, Action, S);
7819 };
7820 llvm::Function *Fn;
7821 llvm::Constant *Addr;
7822 // Emit target region as a standalone region.
7823 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
7824 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
7825 assert(Fn && Addr && "Target device function emission failed.");
7826}
7827
7828void CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDirective(
7829 const OMPTargetTeamsDistributeSimdDirective &S) {
7830 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
7831 emitTargetTeamsDistributeSimdRegion(CGF, Action, S);
7832 };
7833 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
7834}
7835
7836void CodeGenFunction::EmitOMPTeamsDistributeDirective(
7837 const OMPTeamsDistributeDirective &S) {
7838
7839 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7840 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
7841 };
7842
7843 // Emit teams region as a standalone region.
7844 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
7845 PrePostActionTy &Action) {
7846 Action.Enter(CGF);
7847 OMPPrivateScope PrivateScope(CGF);
7848 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7849 (void)PrivateScope.Privatize();
7850 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_distribute,
7851 CodeGen: CodeGenDistribute);
7852 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7853 };
7854 emitCommonOMPTeamsDirective(CGF&: *this, S, InnermostKind: OMPD_distribute, CodeGen);
7855 emitPostUpdateForReductionClause(CGF&: *this, D: S,
7856 CondGen: [](CodeGenFunction &) { return nullptr; });
7857}
7858
7859void CodeGenFunction::EmitOMPTeamsDistributeSimdDirective(
7860 const OMPTeamsDistributeSimdDirective &S) {
7861 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7862 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
7863 };
7864
7865 // Emit teams region as a standalone region.
7866 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
7867 PrePostActionTy &Action) {
7868 Action.Enter(CGF);
7869 OMPPrivateScope PrivateScope(CGF);
7870 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7871 (void)PrivateScope.Privatize();
7872 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_simd,
7873 CodeGen: CodeGenDistribute);
7874 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7875 };
7876 emitCommonOMPTeamsDirective(CGF&: *this, S, InnermostKind: OMPD_distribute_simd, CodeGen);
7877 emitPostUpdateForReductionClause(CGF&: *this, D: S,
7878 CondGen: [](CodeGenFunction &) { return nullptr; });
7879}
7880
7881void CodeGenFunction::EmitOMPTeamsDistributeParallelForDirective(
7882 const OMPTeamsDistributeParallelForDirective &S) {
7883 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7884 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitInnerParallelForWhenCombined,
7885 IncExpr: S.getDistInc());
7886 };
7887
7888 // Emit teams region as a standalone region.
7889 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
7890 PrePostActionTy &Action) {
7891 Action.Enter(CGF);
7892 OMPPrivateScope PrivateScope(CGF);
7893 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7894 (void)PrivateScope.Privatize();
7895 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_distribute,
7896 CodeGen: CodeGenDistribute);
7897 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7898 };
7899 emitCommonOMPTeamsDirective(CGF&: *this, S, InnermostKind: OMPD_distribute_parallel_for, CodeGen);
7900 emitPostUpdateForReductionClause(CGF&: *this, D: S,
7901 CondGen: [](CodeGenFunction &) { return nullptr; });
7902}
7903
7904void CodeGenFunction::EmitOMPTeamsDistributeParallelForSimdDirective(
7905 const OMPTeamsDistributeParallelForSimdDirective &S) {
7906 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
7907 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitInnerParallelForWhenCombined,
7908 IncExpr: S.getDistInc());
7909 };
7910
7911 // Emit teams region as a standalone region.
7912 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
7913 PrePostActionTy &Action) {
7914 Action.Enter(CGF);
7915 OMPPrivateScope PrivateScope(CGF);
7916 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
7917 (void)PrivateScope.Privatize();
7918 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(
7919 CGF, InnermostKind: OMPD_distribute, CodeGen: CodeGenDistribute, /*HasCancel=*/false);
7920 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
7921 };
7922 emitCommonOMPTeamsDirective(CGF&: *this, S, InnermostKind: OMPD_distribute_parallel_for_simd,
7923 CodeGen);
7924 emitPostUpdateForReductionClause(CGF&: *this, D: S,
7925 CondGen: [](CodeGenFunction &) { return nullptr; });
7926}
7927
7928void CodeGenFunction::EmitOMPInteropDirective(const OMPInteropDirective &S) {
7929 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
7930 llvm::Value *Device = nullptr;
7931 llvm::Value *NumDependences = nullptr;
7932 llvm::Value *DependenceList = nullptr;
7933
7934 if (const auto *C = S.getSingleClause<OMPDeviceClause>())
7935 Device = EmitScalarExpr(E: C->getDevice());
7936
7937 // Build list and emit dependences
7938 OMPTaskDataTy Data;
7939 buildDependences(S, Data);
7940 if (!Data.Dependences.empty()) {
7941 Address DependenciesArray = Address::invalid();
7942 std::tie(args&: NumDependences, args&: DependenciesArray) =
7943 CGM.getOpenMPRuntime().emitDependClause(CGF&: *this, Dependencies: Data.Dependences,
7944 Loc: S.getBeginLoc());
7945 DependenceList = DependenciesArray.emitRawPointer(CGF&: *this);
7946 }
7947 Data.HasNowaitClause = S.hasClausesOfKind<OMPNowaitClause>();
7948
7949 assert(!(Data.HasNowaitClause && !(S.getSingleClause<OMPInitClause>() ||
7950 S.getSingleClause<OMPDestroyClause>() ||
7951 S.getSingleClause<OMPUseClause>())) &&
7952 "OMPNowaitClause clause is used separately in OMPInteropDirective.");
7953
7954 auto ItOMPInitClause = S.getClausesOfKind<OMPInitClause>();
7955 if (!ItOMPInitClause.empty()) {
7956 // Look at the multiple init clauses
7957 for (const OMPInitClause *C : ItOMPInitClause) {
7958 llvm::Value *InteropvarPtr =
7959 EmitLValue(E: C->getInteropVar()).getPointer(CGF&: *this);
7960 llvm::omp::OMPInteropType InteropType =
7961 llvm::omp::OMPInteropType::Unknown;
7962 if (C->getIsTarget()) {
7963 InteropType = llvm::omp::OMPInteropType::Target;
7964 } else {
7965 assert(C->getIsTargetSync() &&
7966 "Expected interop-type target/targetsync");
7967 InteropType = llvm::omp::OMPInteropType::TargetSync;
7968 }
7969 OMPBuilder.createOMPInteropInit(Loc: Builder, InteropVar: InteropvarPtr, InteropType,
7970 Device, NumDependences, DependenceAddress: DependenceList,
7971 HaveNowaitClause: Data.HasNowaitClause);
7972 }
7973 }
7974 auto ItOMPDestroyClause = S.getClausesOfKind<OMPDestroyClause>();
7975 if (!ItOMPDestroyClause.empty()) {
7976 // Look at the multiple destroy clauses
7977 for (const OMPDestroyClause *C : ItOMPDestroyClause) {
7978 llvm::Value *InteropvarPtr =
7979 EmitLValue(E: C->getInteropVar()).getPointer(CGF&: *this);
7980 OMPBuilder.createOMPInteropDestroy(Loc: Builder, InteropVar: InteropvarPtr, Device,
7981 NumDependences, DependenceAddress: DependenceList,
7982 HaveNowaitClause: Data.HasNowaitClause);
7983 }
7984 }
7985 auto ItOMPUseClause = S.getClausesOfKind<OMPUseClause>();
7986 if (!ItOMPUseClause.empty()) {
7987 // Look at the multiple use clauses
7988 for (const OMPUseClause *C : ItOMPUseClause) {
7989 llvm::Value *InteropvarPtr =
7990 EmitLValue(E: C->getInteropVar()).getPointer(CGF&: *this);
7991 OMPBuilder.createOMPInteropUse(Loc: Builder, InteropVar: InteropvarPtr, Device,
7992 NumDependences, DependenceAddress: DependenceList,
7993 HaveNowaitClause: Data.HasNowaitClause);
7994 }
7995 }
7996}
7997
7998static void emitTargetTeamsDistributeParallelForRegion(
7999 CodeGenFunction &CGF, const OMPTargetTeamsDistributeParallelForDirective &S,
8000 PrePostActionTy &Action) {
8001 Action.Enter(CGF);
8002 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
8003 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitInnerParallelForWhenCombined,
8004 IncExpr: S.getDistInc());
8005 };
8006
8007 // Emit teams region as a standalone region.
8008 auto &&CodeGenTeams = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
8009 PrePostActionTy &Action) {
8010 Action.Enter(CGF);
8011 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
8012 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
8013 (void)PrivateScope.Privatize();
8014 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(
8015 CGF, InnermostKind: OMPD_distribute, CodeGen: CodeGenDistribute, /*HasCancel=*/false);
8016 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
8017 };
8018
8019 emitCommonOMPTeamsDirective(CGF, S, InnermostKind: OMPD_distribute_parallel_for,
8020 CodeGen: CodeGenTeams);
8021 emitPostUpdateForReductionClause(CGF, D: S,
8022 CondGen: [](CodeGenFunction &) { return nullptr; });
8023}
8024
8025void CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction(
8026 CodeGenModule &CGM, StringRef ParentName,
8027 const OMPTargetTeamsDistributeParallelForDirective &S) {
8028 // Emit SPMD target teams distribute parallel for region as a standalone
8029 // region.
8030 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8031 emitTargetTeamsDistributeParallelForRegion(CGF, S, Action);
8032 };
8033 llvm::Function *Fn;
8034 llvm::Constant *Addr;
8035 // Emit target region as a standalone region.
8036 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
8037 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
8038 assert(Fn && Addr && "Target device function emission failed.");
8039}
8040
8041void CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDirective(
8042 const OMPTargetTeamsDistributeParallelForDirective &S) {
8043 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8044 emitTargetTeamsDistributeParallelForRegion(CGF, S, Action);
8045 };
8046 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
8047}
8048
8049static void emitTargetTeamsDistributeParallelForSimdRegion(
8050 CodeGenFunction &CGF,
8051 const OMPTargetTeamsDistributeParallelForSimdDirective &S,
8052 PrePostActionTy &Action) {
8053 Action.Enter(CGF);
8054 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
8055 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitInnerParallelForWhenCombined,
8056 IncExpr: S.getDistInc());
8057 };
8058
8059 // Emit teams region as a standalone region.
8060 auto &&CodeGenTeams = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
8061 PrePostActionTy &Action) {
8062 Action.Enter(CGF);
8063 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
8064 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
8065 (void)PrivateScope.Privatize();
8066 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(
8067 CGF, InnermostKind: OMPD_distribute, CodeGen: CodeGenDistribute, /*HasCancel=*/false);
8068 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
8069 };
8070
8071 emitCommonOMPTeamsDirective(CGF, S, InnermostKind: OMPD_distribute_parallel_for_simd,
8072 CodeGen: CodeGenTeams);
8073 emitPostUpdateForReductionClause(CGF, D: S,
8074 CondGen: [](CodeGenFunction &) { return nullptr; });
8075}
8076
8077void CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(
8078 CodeGenModule &CGM, StringRef ParentName,
8079 const OMPTargetTeamsDistributeParallelForSimdDirective &S) {
8080 // Emit SPMD target teams distribute parallel for simd region as a standalone
8081 // region.
8082 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8083 emitTargetTeamsDistributeParallelForSimdRegion(CGF, S, Action);
8084 };
8085 llvm::Function *Fn;
8086 llvm::Constant *Addr;
8087 // Emit target region as a standalone region.
8088 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
8089 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
8090 assert(Fn && Addr && "Target device function emission failed.");
8091}
8092
8093void CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForSimdDirective(
8094 const OMPTargetTeamsDistributeParallelForSimdDirective &S) {
8095 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8096 emitTargetTeamsDistributeParallelForSimdRegion(CGF, S, Action);
8097 };
8098 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
8099}
8100
8101void CodeGenFunction::EmitOMPCancellationPointDirective(
8102 const OMPCancellationPointDirective &S) {
8103 CGM.getOpenMPRuntime().emitCancellationPointCall(CGF&: *this, Loc: S.getBeginLoc(),
8104 CancelRegion: S.getCancelRegion());
8105}
8106
8107void CodeGenFunction::EmitOMPCancelDirective(const OMPCancelDirective &S) {
8108 const Expr *IfCond = nullptr;
8109 for (const auto *C : S.getClausesOfKind<OMPIfClause>()) {
8110 if (C->getNameModifier() == OMPD_unknown ||
8111 C->getNameModifier() == OMPD_cancel) {
8112 IfCond = C->getCondition();
8113 break;
8114 }
8115 }
8116 if (CGM.getLangOpts().OpenMPIRBuilder) {
8117 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
8118 // TODO: This check is necessary as we only generate `omp parallel` through
8119 // the OpenMPIRBuilder for now.
8120 if (S.getCancelRegion() == OMPD_parallel ||
8121 S.getCancelRegion() == OMPD_sections ||
8122 S.getCancelRegion() == OMPD_section) {
8123 llvm::Value *IfCondition = nullptr;
8124 if (IfCond)
8125 IfCondition = EmitScalarExpr(E: IfCond,
8126 /*IgnoreResultAssign=*/true);
8127 llvm::OpenMPIRBuilder::InsertPointTy AfterIP = cantFail(
8128 ValOrErr: OMPBuilder.createCancel(Loc: Builder, IfCondition, CanceledDirective: S.getCancelRegion()));
8129 return Builder.restoreIP(IP: AfterIP);
8130 }
8131 }
8132
8133 CGM.getOpenMPRuntime().emitCancelCall(CGF&: *this, Loc: S.getBeginLoc(), IfCond,
8134 CancelRegion: S.getCancelRegion());
8135}
8136
8137CodeGenFunction::JumpDest
8138CodeGenFunction::getOMPCancelDestination(OpenMPDirectiveKind Kind) {
8139 if (Kind == OMPD_parallel || Kind == OMPD_task ||
8140 Kind == OMPD_target_parallel || Kind == OMPD_taskloop ||
8141 Kind == OMPD_master_taskloop || Kind == OMPD_parallel_master_taskloop)
8142 return ReturnBlock;
8143 assert(Kind == OMPD_for || Kind == OMPD_section || Kind == OMPD_sections ||
8144 Kind == OMPD_parallel_sections || Kind == OMPD_parallel_for ||
8145 Kind == OMPD_distribute_parallel_for ||
8146 Kind == OMPD_target_parallel_for ||
8147 Kind == OMPD_teams_distribute_parallel_for ||
8148 Kind == OMPD_target_teams_distribute_parallel_for);
8149 return OMPCancelStack.getExitBlock();
8150}
8151
8152void CodeGenFunction::EmitOMPUseDevicePtrClause(
8153 const OMPUseDevicePtrClause &C, OMPPrivateScope &PrivateScope,
8154 const llvm::DenseMap<const ValueDecl *, llvm::Value *>
8155 CaptureDeviceAddrMap) {
8156 llvm::SmallDenseSet<CanonicalDeclPtr<const Decl>, 4> Processed;
8157 for (const Expr *OrigVarIt : C.varlist()) {
8158 const auto *OrigVD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: OrigVarIt)->getDecl());
8159 if (!Processed.insert(V: OrigVD).second)
8160 continue;
8161
8162 // In order to identify the right initializer we need to match the
8163 // declaration used by the mapping logic. In some cases we may get
8164 // OMPCapturedExprDecl that refers to the original declaration.
8165 const ValueDecl *MatchingVD = OrigVD;
8166 if (const auto *OED = dyn_cast<OMPCapturedExprDecl>(Val: MatchingVD)) {
8167 // OMPCapturedExprDecl are used to privative fields of the current
8168 // structure.
8169 const auto *ME = cast<MemberExpr>(Val: OED->getInit());
8170 assert(isa<CXXThisExpr>(ME->getBase()->IgnoreImpCasts()) &&
8171 "Base should be the current struct!");
8172 MatchingVD = ME->getMemberDecl();
8173 }
8174
8175 // If we don't have information about the current list item, move on to
8176 // the next one.
8177 auto InitAddrIt = CaptureDeviceAddrMap.find(Val: MatchingVD);
8178 if (InitAddrIt == CaptureDeviceAddrMap.end())
8179 continue;
8180
8181 llvm::Type *Ty = ConvertTypeForMem(T: OrigVD->getType().getNonReferenceType());
8182
8183 // Return the address of the private variable.
8184 bool IsRegistered = PrivateScope.addPrivate(
8185 LocalVD: OrigVD,
8186 Addr: Address(InitAddrIt->second, Ty,
8187 getContext().getTypeAlignInChars(T: getContext().VoidPtrTy)));
8188 assert(IsRegistered && "firstprivate var already registered as private");
8189 // Silence the warning about unused variable.
8190 (void)IsRegistered;
8191 }
8192}
8193
8194static const VarDecl *getBaseDecl(const Expr *Ref) {
8195 const Expr *Base = Ref->IgnoreParenImpCasts();
8196 while (const auto *OASE = dyn_cast<ArraySectionExpr>(Val: Base))
8197 Base = OASE->getBase()->IgnoreParenImpCasts();
8198 while (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Val: Base))
8199 Base = ASE->getBase()->IgnoreParenImpCasts();
8200 return cast<VarDecl>(Val: cast<DeclRefExpr>(Val: Base)->getDecl());
8201}
8202
8203void CodeGenFunction::EmitOMPUseDeviceAddrClause(
8204 const OMPUseDeviceAddrClause &C, OMPPrivateScope &PrivateScope,
8205 const llvm::DenseMap<const ValueDecl *, llvm::Value *>
8206 CaptureDeviceAddrMap) {
8207 llvm::SmallDenseSet<CanonicalDeclPtr<const Decl>, 4> Processed;
8208 for (const Expr *Ref : C.varlist()) {
8209 const VarDecl *OrigVD = getBaseDecl(Ref);
8210 if (!Processed.insert(V: OrigVD).second)
8211 continue;
8212 // In order to identify the right initializer we need to match the
8213 // declaration used by the mapping logic. In some cases we may get
8214 // OMPCapturedExprDecl that refers to the original declaration.
8215 const ValueDecl *MatchingVD = OrigVD;
8216 if (const auto *OED = dyn_cast<OMPCapturedExprDecl>(Val: MatchingVD)) {
8217 // OMPCapturedExprDecl are used to privative fields of the current
8218 // structure.
8219 const auto *ME = cast<MemberExpr>(Val: OED->getInit());
8220 assert(isa<CXXThisExpr>(ME->getBase()) &&
8221 "Base should be the current struct!");
8222 MatchingVD = ME->getMemberDecl();
8223 }
8224
8225 // If we don't have information about the current list item, move on to
8226 // the next one.
8227 auto InitAddrIt = CaptureDeviceAddrMap.find(Val: MatchingVD);
8228 if (InitAddrIt == CaptureDeviceAddrMap.end())
8229 continue;
8230
8231 llvm::Type *Ty = ConvertTypeForMem(T: OrigVD->getType().getNonReferenceType());
8232
8233 Address PrivAddr =
8234 Address(InitAddrIt->second, Ty,
8235 getContext().getTypeAlignInChars(T: getContext().VoidPtrTy));
8236 // For declrefs and variable length array need to load the pointer for
8237 // correct mapping, since the pointer to the data was passed to the runtime.
8238 if (isa<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts()) ||
8239 MatchingVD->getType()->isArrayType()) {
8240 QualType PtrTy = getContext().getPointerType(
8241 T: OrigVD->getType().getNonReferenceType());
8242 PrivAddr =
8243 EmitLoadOfPointer(Ptr: PrivAddr.withElementType(ElemTy: ConvertTypeForMem(T: PtrTy)),
8244 PtrTy: PtrTy->castAs<PointerType>());
8245 }
8246
8247 (void)PrivateScope.addPrivate(LocalVD: OrigVD, Addr: PrivAddr);
8248 }
8249}
8250
8251// Generate the instructions for '#pragma omp target data' directive.
8252void CodeGenFunction::EmitOMPTargetDataDirective(
8253 const OMPTargetDataDirective &S) {
8254 // Emit vtable only from host for target data directive.
8255 if (!CGM.getLangOpts().OpenMPIsTargetDevice)
8256 CGM.getOpenMPRuntime().registerVTable(D: S);
8257
8258 CGOpenMPRuntime::TargetDataInfo Info(/*RequiresDevicePointerInfo=*/true,
8259 /*SeparateBeginEndCalls=*/true);
8260
8261 // Create a pre/post action to signal the privatization of the device pointer.
8262 // This action can be replaced by the OpenMP runtime code generation to
8263 // deactivate privatization.
8264 bool PrivatizeDevicePointers = false;
8265 class DevicePointerPrivActionTy : public PrePostActionTy {
8266 bool &PrivatizeDevicePointers;
8267
8268 public:
8269 explicit DevicePointerPrivActionTy(bool &PrivatizeDevicePointers)
8270 : PrivatizeDevicePointers(PrivatizeDevicePointers) {}
8271 void Enter(CodeGenFunction &CGF) override {
8272 PrivatizeDevicePointers = true;
8273 }
8274 };
8275 DevicePointerPrivActionTy PrivAction(PrivatizeDevicePointers);
8276
8277 auto &&CodeGen = [&](CodeGenFunction &CGF, PrePostActionTy &Action) {
8278 auto &&InnermostCodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
8279 CGF.EmitStmt(S: S.getInnermostCapturedStmt()->getCapturedStmt());
8280 };
8281
8282 // Codegen that selects whether to generate the privatization code or not.
8283 auto &&PrivCodeGen = [&](CodeGenFunction &CGF, PrePostActionTy &Action) {
8284 RegionCodeGenTy RCG(InnermostCodeGen);
8285 PrivatizeDevicePointers = false;
8286
8287 // Call the pre-action to change the status of PrivatizeDevicePointers if
8288 // needed.
8289 Action.Enter(CGF);
8290
8291 if (PrivatizeDevicePointers) {
8292 OMPPrivateScope PrivateScope(CGF);
8293 // Emit all instances of the use_device_ptr clause.
8294 for (const auto *C : S.getClausesOfKind<OMPUseDevicePtrClause>())
8295 CGF.EmitOMPUseDevicePtrClause(C: *C, PrivateScope,
8296 CaptureDeviceAddrMap: Info.CaptureDeviceAddrMap);
8297 for (const auto *C : S.getClausesOfKind<OMPUseDeviceAddrClause>())
8298 CGF.EmitOMPUseDeviceAddrClause(C: *C, PrivateScope,
8299 CaptureDeviceAddrMap: Info.CaptureDeviceAddrMap);
8300 (void)PrivateScope.Privatize();
8301 RCG(CGF);
8302 } else {
8303 // If we don't have target devices, don't bother emitting the data
8304 // mapping code.
8305 std::optional<OpenMPDirectiveKind> CaptureRegion;
8306 if (CGM.getLangOpts().OMPTargetTriples.empty()) {
8307 // Emit helper decls of the use_device_ptr/use_device_addr clauses.
8308 for (const auto *C : S.getClausesOfKind<OMPUseDevicePtrClause>())
8309 for (const Expr *E : C->varlist()) {
8310 const Decl *D = cast<DeclRefExpr>(Val: E)->getDecl();
8311 if (const auto *OED = dyn_cast<OMPCapturedExprDecl>(Val: D))
8312 CGF.EmitVarDecl(D: *OED);
8313 }
8314 for (const auto *C : S.getClausesOfKind<OMPUseDeviceAddrClause>())
8315 for (const Expr *E : C->varlist()) {
8316 const Decl *D = getBaseDecl(Ref: E);
8317 if (const auto *OED = dyn_cast<OMPCapturedExprDecl>(Val: D))
8318 CGF.EmitVarDecl(D: *OED);
8319 }
8320 } else {
8321 CaptureRegion = OMPD_unknown;
8322 }
8323
8324 OMPLexicalScope Scope(CGF, S, CaptureRegion);
8325 RCG(CGF);
8326 }
8327 };
8328
8329 // Forward the provided action to the privatization codegen.
8330 RegionCodeGenTy PrivRCG(PrivCodeGen);
8331 PrivRCG.setAction(Action);
8332
8333 // Notwithstanding the body of the region is emitted as inlined directive,
8334 // we don't use an inline scope as changes in the references inside the
8335 // region are expected to be visible outside, so we do not privative them.
8336 OMPLexicalScope Scope(CGF, S);
8337 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_target_data,
8338 CodeGen: PrivRCG);
8339 };
8340
8341 RegionCodeGenTy RCG(CodeGen);
8342
8343 // If we don't have target devices, don't bother emitting the data mapping
8344 // code.
8345 if (CGM.getLangOpts().OMPTargetTriples.empty()) {
8346 RCG(*this);
8347 return;
8348 }
8349
8350 // Check if we have any if clause associated with the directive.
8351 const Expr *IfCond = nullptr;
8352 if (const auto *C = S.getSingleClause<OMPIfClause>())
8353 IfCond = C->getCondition();
8354
8355 // Check if we have any device clause associated with the directive.
8356 const Expr *Device = nullptr;
8357 if (const auto *C = S.getSingleClause<OMPDeviceClause>())
8358 Device = C->getDevice();
8359
8360 // Set the action to signal privatization of device pointers.
8361 RCG.setAction(PrivAction);
8362
8363 // Emit region code.
8364 CGM.getOpenMPRuntime().emitTargetDataCalls(CGF&: *this, D: S, IfCond, Device, CodeGen: RCG,
8365 Info);
8366}
8367
8368void CodeGenFunction::EmitOMPTargetEnterDataDirective(
8369 const OMPTargetEnterDataDirective &S) {
8370 // If we don't have target devices, don't bother emitting the data mapping
8371 // code.
8372 if (CGM.getLangOpts().OMPTargetTriples.empty())
8373 return;
8374
8375 // Check if we have any if clause associated with the directive.
8376 const Expr *IfCond = nullptr;
8377 if (const auto *C = S.getSingleClause<OMPIfClause>())
8378 IfCond = C->getCondition();
8379
8380 // Check if we have any device clause associated with the directive.
8381 const Expr *Device = nullptr;
8382 if (const auto *C = S.getSingleClause<OMPDeviceClause>())
8383 Device = C->getDevice();
8384
8385 OMPLexicalScope Scope(*this, S, OMPD_task);
8386 CGM.getOpenMPRuntime().emitTargetDataStandAloneCall(CGF&: *this, D: S, IfCond, Device);
8387}
8388
8389void CodeGenFunction::EmitOMPTargetExitDataDirective(
8390 const OMPTargetExitDataDirective &S) {
8391 // If we don't have target devices, don't bother emitting the data mapping
8392 // code.
8393 if (CGM.getLangOpts().OMPTargetTriples.empty())
8394 return;
8395
8396 // Check if we have any if clause associated with the directive.
8397 const Expr *IfCond = nullptr;
8398 if (const auto *C = S.getSingleClause<OMPIfClause>())
8399 IfCond = C->getCondition();
8400
8401 // Check if we have any device clause associated with the directive.
8402 const Expr *Device = nullptr;
8403 if (const auto *C = S.getSingleClause<OMPDeviceClause>())
8404 Device = C->getDevice();
8405
8406 OMPLexicalScope Scope(*this, S, OMPD_task);
8407 CGM.getOpenMPRuntime().emitTargetDataStandAloneCall(CGF&: *this, D: S, IfCond, Device);
8408}
8409
8410static void emitTargetParallelRegion(CodeGenFunction &CGF,
8411 const OMPTargetParallelDirective &S,
8412 PrePostActionTy &Action) {
8413 // Get the captured statement associated with the 'parallel' region.
8414 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: OMPD_parallel);
8415 Action.Enter(CGF);
8416 auto &&CodeGen = [&S, CS](CodeGenFunction &CGF, PrePostActionTy &Action) {
8417 Action.Enter(CGF);
8418 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
8419 (void)CGF.EmitOMPFirstprivateClause(D: S, PrivateScope);
8420 CGF.EmitOMPPrivateClause(D: S, PrivateScope);
8421 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
8422 (void)PrivateScope.Privatize();
8423 if (isOpenMPTargetExecutionDirective(DKind: S.getDirectiveKind()))
8424 CGF.CGM.getOpenMPRuntime().adjustTargetSpecificDataForLambdas(CGF, D: S);
8425 // TODO: Add support for clauses.
8426 CGF.EmitStmt(S: CS->getCapturedStmt());
8427 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_parallel);
8428 };
8429 emitCommonOMPParallelDirective(CGF, S, InnermostKind: OMPD_parallel, CodeGen,
8430 CodeGenBoundParameters: emitEmptyBoundParameters);
8431 emitPostUpdateForReductionClause(CGF, D: S,
8432 CondGen: [](CodeGenFunction &) { return nullptr; });
8433}
8434
8435void CodeGenFunction::EmitOMPTargetParallelDeviceFunction(
8436 CodeGenModule &CGM, StringRef ParentName,
8437 const OMPTargetParallelDirective &S) {
8438 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8439 emitTargetParallelRegion(CGF, S, Action);
8440 };
8441 llvm::Function *Fn;
8442 llvm::Constant *Addr;
8443 // Emit target region as a standalone region.
8444 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
8445 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
8446 assert(Fn && Addr && "Target device function emission failed.");
8447}
8448
8449void CodeGenFunction::EmitOMPTargetParallelDirective(
8450 const OMPTargetParallelDirective &S) {
8451 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8452 emitTargetParallelRegion(CGF, S, Action);
8453 };
8454 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
8455}
8456
8457static void emitTargetParallelForRegion(CodeGenFunction &CGF,
8458 const OMPTargetParallelForDirective &S,
8459 PrePostActionTy &Action) {
8460 Action.Enter(CGF);
8461 // Emit directive as a combined directive that consists of two implicit
8462 // directives: 'parallel' with 'for' directive.
8463 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8464 Action.Enter(CGF);
8465 CodeGenFunction::OMPCancelStackRAII CancelRegion(
8466 CGF, OMPD_target_parallel_for, S.hasCancel());
8467 CGF.EmitOMPWorksharingLoop(S, EUB: S.getEnsureUpperBound(), CodeGenLoopBounds: emitForLoopBounds,
8468 CGDispatchBounds: emitDispatchForLoopBounds);
8469 };
8470 emitCommonOMPParallelDirective(CGF, S, InnermostKind: OMPD_for, CodeGen,
8471 CodeGenBoundParameters: emitEmptyBoundParameters);
8472}
8473
8474void CodeGenFunction::EmitOMPTargetParallelForDeviceFunction(
8475 CodeGenModule &CGM, StringRef ParentName,
8476 const OMPTargetParallelForDirective &S) {
8477 // Emit SPMD target parallel for region as a standalone region.
8478 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8479 emitTargetParallelForRegion(CGF, S, Action);
8480 };
8481 llvm::Function *Fn;
8482 llvm::Constant *Addr;
8483 // Emit target region as a standalone region.
8484 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
8485 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
8486 assert(Fn && Addr && "Target device function emission failed.");
8487}
8488
8489void CodeGenFunction::EmitOMPTargetParallelForDirective(
8490 const OMPTargetParallelForDirective &S) {
8491 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8492 emitTargetParallelForRegion(CGF, S, Action);
8493 };
8494 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
8495}
8496
8497static void
8498emitTargetParallelForSimdRegion(CodeGenFunction &CGF,
8499 const OMPTargetParallelForSimdDirective &S,
8500 PrePostActionTy &Action) {
8501 Action.Enter(CGF);
8502 // Emit directive as a combined directive that consists of two implicit
8503 // directives: 'parallel' with 'for' directive.
8504 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8505 Action.Enter(CGF);
8506 CGF.EmitOMPWorksharingLoop(S, EUB: S.getEnsureUpperBound(), CodeGenLoopBounds: emitForLoopBounds,
8507 CGDispatchBounds: emitDispatchForLoopBounds);
8508 };
8509 emitCommonOMPParallelDirective(CGF, S, InnermostKind: OMPD_simd, CodeGen,
8510 CodeGenBoundParameters: emitEmptyBoundParameters);
8511}
8512
8513void CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction(
8514 CodeGenModule &CGM, StringRef ParentName,
8515 const OMPTargetParallelForSimdDirective &S) {
8516 // Emit SPMD target parallel for region as a standalone region.
8517 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8518 emitTargetParallelForSimdRegion(CGF, S, Action);
8519 };
8520 llvm::Function *Fn;
8521 llvm::Constant *Addr;
8522 // Emit target region as a standalone region.
8523 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
8524 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
8525 assert(Fn && Addr && "Target device function emission failed.");
8526}
8527
8528void CodeGenFunction::EmitOMPTargetParallelForSimdDirective(
8529 const OMPTargetParallelForSimdDirective &S) {
8530 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8531 emitTargetParallelForSimdRegion(CGF, S, Action);
8532 };
8533 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
8534}
8535
8536/// Emit a helper variable and return corresponding lvalue.
8537static void mapParam(CodeGenFunction &CGF, const DeclRefExpr *Helper,
8538 const ImplicitParamDecl *PVD,
8539 CodeGenFunction::OMPPrivateScope &Privates) {
8540 const auto *VDecl = cast<VarDecl>(Val: Helper->getDecl());
8541 Privates.addPrivate(LocalVD: VDecl, Addr: CGF.GetAddrOfLocalVar(VD: PVD));
8542}
8543
8544void CodeGenFunction::EmitOMPTaskLoopBasedDirective(const OMPLoopDirective &S) {
8545 assert(isOpenMPTaskLoopDirective(S.getDirectiveKind()));
8546 // Emit outlined function for task construct.
8547 const CapturedStmt *CS = S.getCapturedStmt(RegionKind: OMPD_taskloop);
8548 Address CapturedStruct = Address::invalid();
8549 {
8550 OMPLexicalScope Scope(*this, S, OMPD_taskloop, /*EmitPreInitStmt=*/false);
8551 CapturedStruct = GenerateCapturedStmtArgument(S: *CS);
8552 }
8553 CanQualType SharedsTy =
8554 getContext().getCanonicalTagType(TD: CS->getCapturedRecordDecl());
8555 const Expr *IfCond = nullptr;
8556 for (const auto *C : S.getClausesOfKind<OMPIfClause>()) {
8557 if (C->getNameModifier() == OMPD_unknown ||
8558 C->getNameModifier() == OMPD_taskloop) {
8559 IfCond = C->getCondition();
8560 break;
8561 }
8562 }
8563
8564 OMPTaskDataTy Data;
8565 // Check if taskloop must be emitted without taskgroup.
8566 Data.Nogroup = S.getSingleClause<OMPNogroupClause>();
8567 // TODO: Check if we should emit tied or untied task.
8568 Data.Tied = true;
8569 // Set scheduling for taskloop
8570 if (const auto *Clause = S.getSingleClause<OMPGrainsizeClause>()) {
8571 // grainsize clause
8572 Data.Schedule.setInt(/*IntVal=*/false);
8573 Data.Schedule.setPointer(EmitScalarExpr(E: Clause->getGrainsize()));
8574 Data.HasModifier =
8575 (Clause->getModifier() == OMPC_GRAINSIZE_strict) ? true : false;
8576 } else if (const auto *Clause = S.getSingleClause<OMPNumTasksClause>()) {
8577 // num_tasks clause
8578 Data.Schedule.setInt(/*IntVal=*/true);
8579 Data.Schedule.setPointer(EmitScalarExpr(E: Clause->getNumTasks()));
8580 Data.HasModifier =
8581 (Clause->getModifier() == OMPC_NUMTASKS_strict) ? true : false;
8582 }
8583
8584 auto &&BodyGen = [CS, &S](CodeGenFunction &CGF, PrePostActionTy &) {
8585 // if (PreCond) {
8586 // for (IV in 0..LastIteration) BODY;
8587 // <Final counter/linear vars updates>;
8588 // }
8589 //
8590
8591 // Emit: if (PreCond) - begin.
8592 // If the condition constant folds and can be elided, avoid emitting the
8593 // whole loop.
8594 bool CondConstant;
8595 llvm::BasicBlock *ContBlock = nullptr;
8596 OMPLoopScope PreInitScope(CGF, S);
8597 if (CGF.ConstantFoldsToSimpleInteger(Cond: S.getPreCond(), Result&: CondConstant)) {
8598 if (!CondConstant)
8599 return;
8600 } else {
8601 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock(name: "taskloop.if.then");
8602 ContBlock = CGF.createBasicBlock(name: "taskloop.if.end");
8603 emitPreCond(CGF, S, Cond: S.getPreCond(), TrueBlock: ThenBlock, FalseBlock: ContBlock,
8604 TrueCount: CGF.getProfileCount(S: &S));
8605 CGF.EmitBlock(BB: ThenBlock);
8606 CGF.incrementProfileCounter(S: &S);
8607 }
8608
8609 (void)CGF.EmitOMPLinearClauseInit(D: S);
8610
8611 OMPPrivateScope LoopScope(CGF);
8612 // Emit helper vars inits.
8613 enum { LowerBound = 5, UpperBound, Stride, LastIter };
8614 auto *I = CS->getCapturedDecl()->param_begin();
8615 auto *LBP = std::next(x: I, n: LowerBound);
8616 auto *UBP = std::next(x: I, n: UpperBound);
8617 auto *STP = std::next(x: I, n: Stride);
8618 auto *LIP = std::next(x: I, n: LastIter);
8619 mapParam(CGF, Helper: cast<DeclRefExpr>(Val: S.getLowerBoundVariable()), PVD: *LBP,
8620 Privates&: LoopScope);
8621 mapParam(CGF, Helper: cast<DeclRefExpr>(Val: S.getUpperBoundVariable()), PVD: *UBP,
8622 Privates&: LoopScope);
8623 mapParam(CGF, Helper: cast<DeclRefExpr>(Val: S.getStrideVariable()), PVD: *STP, Privates&: LoopScope);
8624 mapParam(CGF, Helper: cast<DeclRefExpr>(Val: S.getIsLastIterVariable()), PVD: *LIP,
8625 Privates&: LoopScope);
8626 CGF.EmitOMPPrivateLoopCounters(S, LoopScope);
8627 CGF.EmitOMPLinearClause(D: S, PrivateScope&: LoopScope);
8628 bool HasLastprivateClause = CGF.EmitOMPLastprivateClauseInit(D: S, PrivateScope&: LoopScope);
8629 (void)LoopScope.Privatize();
8630 // Emit the loop iteration variable.
8631 const Expr *IVExpr = S.getIterationVariable();
8632 const auto *IVDecl = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: IVExpr)->getDecl());
8633 CGF.EmitVarDecl(D: *IVDecl);
8634 CGF.EmitIgnoredExpr(E: S.getInit());
8635
8636 // Emit the iterations count variable.
8637 // If it is not a variable, Sema decided to calculate iterations count on
8638 // each iteration (e.g., it is foldable into a constant).
8639 if (const auto *LIExpr = dyn_cast<DeclRefExpr>(Val: S.getLastIteration())) {
8640 CGF.EmitVarDecl(D: *cast<VarDecl>(Val: LIExpr->getDecl()));
8641 // Emit calculation of the iterations count.
8642 CGF.EmitIgnoredExpr(E: S.getCalcLastIteration());
8643 }
8644
8645 {
8646 OMPLexicalScope Scope(CGF, S, OMPD_taskloop, /*EmitPreInitStmt=*/false);
8647 emitCommonSimdLoop(
8648 CGF, S,
8649 SimdInitGen: [&S](CodeGenFunction &CGF, PrePostActionTy &) {
8650 if (isOpenMPSimdDirective(DKind: S.getDirectiveKind()))
8651 CGF.EmitOMPSimdInit(D: S);
8652 },
8653 BodyCodeGen: [&S, &LoopScope](CodeGenFunction &CGF, PrePostActionTy &) {
8654 CGF.EmitOMPInnerLoop(
8655 S, RequiresCleanup: LoopScope.requiresCleanups(), LoopCond: S.getCond(), IncExpr: S.getInc(),
8656 BodyGen: [&S](CodeGenFunction &CGF) {
8657 emitOMPLoopBodyWithStopPoint(CGF, S,
8658 LoopExit: CodeGenFunction::JumpDest());
8659 },
8660 PostIncGen: [](CodeGenFunction &) {});
8661 });
8662 }
8663 // Emit: if (PreCond) - end.
8664 if (ContBlock) {
8665 CGF.EmitBranch(Block: ContBlock);
8666 CGF.EmitBlock(BB: ContBlock, IsFinished: true);
8667 }
8668 // Emit final copy of the lastprivate variables if IsLastIter != 0.
8669 if (HasLastprivateClause) {
8670 CGF.EmitOMPLastprivateClauseFinal(
8671 D: S, NoFinals: isOpenMPSimdDirective(DKind: S.getDirectiveKind()),
8672 IsLastIterCond: CGF.Builder.CreateIsNotNull(Arg: CGF.EmitLoadOfScalar(
8673 Addr: CGF.GetAddrOfLocalVar(VD: *LIP), /*Volatile=*/false,
8674 Ty: (*LIP)->getType(), Loc: S.getBeginLoc())));
8675 }
8676 LoopScope.restoreMap();
8677 CGF.EmitOMPLinearClauseFinal(D: S, CondGen: [LIP, &S](CodeGenFunction &CGF) {
8678 return CGF.Builder.CreateIsNotNull(
8679 Arg: CGF.EmitLoadOfScalar(Addr: CGF.GetAddrOfLocalVar(VD: *LIP), /*Volatile=*/false,
8680 Ty: (*LIP)->getType(), Loc: S.getBeginLoc()));
8681 });
8682 };
8683 auto &&TaskGen = [&S, SharedsTy, CapturedStruct,
8684 IfCond](CodeGenFunction &CGF, llvm::Function *OutlinedFn,
8685 const OMPTaskDataTy &Data) {
8686 auto &&CodeGen = [&S, OutlinedFn, SharedsTy, CapturedStruct, IfCond,
8687 &Data](CodeGenFunction &CGF, PrePostActionTy &) {
8688 OMPLoopScope PreInitScope(CGF, S);
8689 CGF.CGM.getOpenMPRuntime().emitTaskLoopCall(CGF, Loc: S.getBeginLoc(), D: S,
8690 TaskFunction: OutlinedFn, SharedsTy,
8691 Shareds: CapturedStruct, IfCond, Data);
8692 };
8693 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_taskloop,
8694 CodeGen);
8695 };
8696 if (Data.Nogroup) {
8697 EmitOMPTaskBasedDirective(S, CapturedRegion: OMPD_taskloop, BodyGen, TaskGen, Data);
8698 } else {
8699 CGM.getOpenMPRuntime().emitTaskgroupRegion(
8700 CGF&: *this,
8701 TaskgroupOpGen: [&S, &BodyGen, &TaskGen, &Data](CodeGenFunction &CGF,
8702 PrePostActionTy &Action) {
8703 Action.Enter(CGF);
8704 CGF.EmitOMPTaskBasedDirective(S, CapturedRegion: OMPD_taskloop, BodyGen, TaskGen,
8705 Data);
8706 },
8707 Loc: S.getBeginLoc());
8708 }
8709}
8710
8711void CodeGenFunction::EmitOMPTaskLoopDirective(const OMPTaskLoopDirective &S) {
8712 auto LPCRegion =
8713 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8714 EmitOMPTaskLoopBasedDirective(S);
8715}
8716
8717void CodeGenFunction::EmitOMPTaskLoopSimdDirective(
8718 const OMPTaskLoopSimdDirective &S) {
8719 auto LPCRegion =
8720 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8721 OMPLexicalScope Scope(*this, S);
8722 EmitOMPTaskLoopBasedDirective(S);
8723}
8724
8725void CodeGenFunction::EmitOMPMasterTaskLoopDirective(
8726 const OMPMasterTaskLoopDirective &S) {
8727 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8728 Action.Enter(CGF);
8729 EmitOMPTaskLoopBasedDirective(S);
8730 };
8731 auto LPCRegion =
8732 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8733 OMPLexicalScope Scope(*this, S, std::nullopt, /*EmitPreInitStmt=*/false);
8734 CGM.getOpenMPRuntime().emitMasterRegion(CGF&: *this, MasterOpGen: CodeGen, Loc: S.getBeginLoc());
8735}
8736
8737void CodeGenFunction::EmitOMPMaskedTaskLoopDirective(
8738 const OMPMaskedTaskLoopDirective &S) {
8739 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8740 Action.Enter(CGF);
8741 EmitOMPTaskLoopBasedDirective(S);
8742 };
8743 auto LPCRegion =
8744 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8745 OMPLexicalScope Scope(*this, S, std::nullopt, /*EmitPreInitStmt=*/false);
8746 CGM.getOpenMPRuntime().emitMaskedRegion(CGF&: *this, MaskedOpGen: CodeGen, Loc: S.getBeginLoc());
8747}
8748
8749void CodeGenFunction::EmitOMPMasterTaskLoopSimdDirective(
8750 const OMPMasterTaskLoopSimdDirective &S) {
8751 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8752 Action.Enter(CGF);
8753 EmitOMPTaskLoopBasedDirective(S);
8754 };
8755 auto LPCRegion =
8756 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8757 OMPLexicalScope Scope(*this, S);
8758 CGM.getOpenMPRuntime().emitMasterRegion(CGF&: *this, MasterOpGen: CodeGen, Loc: S.getBeginLoc());
8759}
8760
8761void CodeGenFunction::EmitOMPMaskedTaskLoopSimdDirective(
8762 const OMPMaskedTaskLoopSimdDirective &S) {
8763 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8764 Action.Enter(CGF);
8765 EmitOMPTaskLoopBasedDirective(S);
8766 };
8767 auto LPCRegion =
8768 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8769 OMPLexicalScope Scope(*this, S);
8770 CGM.getOpenMPRuntime().emitMaskedRegion(CGF&: *this, MaskedOpGen: CodeGen, Loc: S.getBeginLoc());
8771}
8772
8773void CodeGenFunction::EmitOMPParallelMasterTaskLoopDirective(
8774 const OMPParallelMasterTaskLoopDirective &S) {
8775 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8776 auto &&TaskLoopCodeGen = [&S](CodeGenFunction &CGF,
8777 PrePostActionTy &Action) {
8778 Action.Enter(CGF);
8779 CGF.EmitOMPTaskLoopBasedDirective(S);
8780 };
8781 OMPLexicalScope Scope(CGF, S, OMPD_parallel, /*EmitPreInitStmt=*/false);
8782 CGM.getOpenMPRuntime().emitMasterRegion(CGF, MasterOpGen: TaskLoopCodeGen,
8783 Loc: S.getBeginLoc());
8784 };
8785 auto LPCRegion =
8786 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8787 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_master_taskloop, CodeGen,
8788 CodeGenBoundParameters: emitEmptyBoundParameters);
8789}
8790
8791void CodeGenFunction::EmitOMPParallelMaskedTaskLoopDirective(
8792 const OMPParallelMaskedTaskLoopDirective &S) {
8793 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8794 auto &&TaskLoopCodeGen = [&S](CodeGenFunction &CGF,
8795 PrePostActionTy &Action) {
8796 Action.Enter(CGF);
8797 CGF.EmitOMPTaskLoopBasedDirective(S);
8798 };
8799 OMPLexicalScope Scope(CGF, S, OMPD_parallel, /*EmitPreInitStmt=*/false);
8800 CGM.getOpenMPRuntime().emitMaskedRegion(CGF, MaskedOpGen: TaskLoopCodeGen,
8801 Loc: S.getBeginLoc());
8802 };
8803 auto LPCRegion =
8804 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8805 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_masked_taskloop, CodeGen,
8806 CodeGenBoundParameters: emitEmptyBoundParameters);
8807}
8808
8809void CodeGenFunction::EmitOMPParallelMasterTaskLoopSimdDirective(
8810 const OMPParallelMasterTaskLoopSimdDirective &S) {
8811 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8812 auto &&TaskLoopCodeGen = [&S](CodeGenFunction &CGF,
8813 PrePostActionTy &Action) {
8814 Action.Enter(CGF);
8815 CGF.EmitOMPTaskLoopBasedDirective(S);
8816 };
8817 OMPLexicalScope Scope(CGF, S, OMPD_parallel, /*EmitPreInitStmt=*/false);
8818 CGM.getOpenMPRuntime().emitMasterRegion(CGF, MasterOpGen: TaskLoopCodeGen,
8819 Loc: S.getBeginLoc());
8820 };
8821 auto LPCRegion =
8822 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8823 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_master_taskloop_simd, CodeGen,
8824 CodeGenBoundParameters: emitEmptyBoundParameters);
8825}
8826
8827void CodeGenFunction::EmitOMPParallelMaskedTaskLoopSimdDirective(
8828 const OMPParallelMaskedTaskLoopSimdDirective &S) {
8829 auto &&CodeGen = [this, &S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8830 auto &&TaskLoopCodeGen = [&S](CodeGenFunction &CGF,
8831 PrePostActionTy &Action) {
8832 Action.Enter(CGF);
8833 CGF.EmitOMPTaskLoopBasedDirective(S);
8834 };
8835 OMPLexicalScope Scope(CGF, S, OMPD_parallel, /*EmitPreInitStmt=*/false);
8836 CGM.getOpenMPRuntime().emitMaskedRegion(CGF, MaskedOpGen: TaskLoopCodeGen,
8837 Loc: S.getBeginLoc());
8838 };
8839 auto LPCRegion =
8840 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8841 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_masked_taskloop_simd, CodeGen,
8842 CodeGenBoundParameters: emitEmptyBoundParameters);
8843}
8844
8845// Generate the instructions for '#pragma omp target update' directive.
8846void CodeGenFunction::EmitOMPTargetUpdateDirective(
8847 const OMPTargetUpdateDirective &S) {
8848 // If we don't have target devices, don't bother emitting the data mapping
8849 // code.
8850 if (CGM.getLangOpts().OMPTargetTriples.empty())
8851 return;
8852
8853 // Check if we have any if clause associated with the directive.
8854 const Expr *IfCond = nullptr;
8855 if (const auto *C = S.getSingleClause<OMPIfClause>())
8856 IfCond = C->getCondition();
8857
8858 // Check if we have any device clause associated with the directive.
8859 const Expr *Device = nullptr;
8860 if (const auto *C = S.getSingleClause<OMPDeviceClause>())
8861 Device = C->getDevice();
8862
8863 OMPLexicalScope Scope(*this, S, OMPD_task);
8864 CGM.getOpenMPRuntime().emitTargetDataStandAloneCall(CGF&: *this, D: S, IfCond, Device);
8865}
8866
8867void CodeGenFunction::EmitOMPGenericLoopDirective(
8868 const OMPGenericLoopDirective &S) {
8869 // Always expect a bind clause on the loop directive. It it wasn't
8870 // in the source, it should have been added in sema.
8871
8872 OpenMPBindClauseKind BindKind = OMPC_BIND_unknown;
8873 if (const auto *C = S.getSingleClause<OMPBindClause>())
8874 BindKind = C->getBindKind();
8875
8876 switch (BindKind) {
8877 case OMPC_BIND_parallel: // for
8878 return emitOMPForDirective(S, CGF&: *this, CGM, /*HasCancel=*/false);
8879 case OMPC_BIND_teams: // distribute
8880 return emitOMPDistributeDirective(S, CGF&: *this, CGM);
8881 case OMPC_BIND_thread: // simd
8882 return emitOMPSimdDirective(S, CGF&: *this, CGM);
8883 case OMPC_BIND_unknown:
8884 break;
8885 }
8886
8887 // Unimplemented, just inline the underlying statement for now.
8888 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8889 // Emit the loop iteration variable.
8890 const Stmt *CS =
8891 cast<CapturedStmt>(Val: S.getAssociatedStmt())->getCapturedStmt();
8892 const auto *ForS = dyn_cast<ForStmt>(Val: CS);
8893 if (ForS && !isa<DeclStmt>(Val: ForS->getInit())) {
8894 OMPPrivateScope LoopScope(CGF);
8895 CGF.EmitOMPPrivateLoopCounters(S, LoopScope);
8896 (void)LoopScope.Privatize();
8897 CGF.EmitStmt(S: CS);
8898 LoopScope.restoreMap();
8899 } else {
8900 CGF.EmitStmt(S: CS);
8901 }
8902 };
8903 OMPLexicalScope Scope(*this, S, OMPD_unknown);
8904 CGM.getOpenMPRuntime().emitInlinedDirective(CGF&: *this, InnermostKind: OMPD_loop, CodeGen);
8905}
8906
8907void CodeGenFunction::EmitOMPParallelGenericLoopDirective(
8908 const OMPLoopDirective &S) {
8909 // Emit combined directive as if its constituent constructs are 'parallel'
8910 // and 'for'.
8911 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
8912 Action.Enter(CGF);
8913 emitOMPCopyinClause(CGF, S);
8914 (void)emitWorksharingDirective(CGF, S, /*HasCancel=*/false);
8915 };
8916 {
8917 auto LPCRegion =
8918 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S);
8919 emitCommonOMPParallelDirective(CGF&: *this, S, InnermostKind: OMPD_for, CodeGen,
8920 CodeGenBoundParameters: emitEmptyBoundParameters);
8921 }
8922 // Check for outer lastprivate conditional update.
8923 checkForLastprivateConditionalUpdate(CGF&: *this, S);
8924}
8925
8926void CodeGenFunction::EmitOMPTeamsGenericLoopDirective(
8927 const OMPTeamsGenericLoopDirective &S) {
8928 // To be consistent with current behavior of 'target teams loop', emit
8929 // 'teams loop' as if its constituent constructs are 'teams' and 'distribute'.
8930 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
8931 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
8932 };
8933
8934 // Emit teams region as a standalone region.
8935 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
8936 PrePostActionTy &Action) {
8937 Action.Enter(CGF);
8938 OMPPrivateScope PrivateScope(CGF);
8939 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
8940 (void)PrivateScope.Privatize();
8941 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, InnermostKind: OMPD_distribute,
8942 CodeGen: CodeGenDistribute);
8943 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
8944 };
8945 emitCommonOMPTeamsDirective(CGF&: *this, S, InnermostKind: OMPD_distribute, CodeGen);
8946 emitPostUpdateForReductionClause(CGF&: *this, D: S,
8947 CondGen: [](CodeGenFunction &) { return nullptr; });
8948}
8949
8950#ifndef NDEBUG
8951static void emitTargetTeamsLoopCodegenStatus(CodeGenFunction &CGF,
8952 std::string StatusMsg,
8953 const OMPExecutableDirective &D) {
8954 bool IsDevice = CGF.CGM.getLangOpts().OpenMPIsTargetDevice;
8955 if (IsDevice)
8956 StatusMsg += ": DEVICE";
8957 else
8958 StatusMsg += ": HOST";
8959 SourceLocation L = D.getBeginLoc();
8960 auto &SM = CGF.getContext().getSourceManager();
8961 PresumedLoc PLoc = SM.getPresumedLoc(L);
8962 const char *FileName = PLoc.isValid() ? PLoc.getFilename() : nullptr;
8963 unsigned LineNo =
8964 PLoc.isValid() ? PLoc.getLine() : SM.getExpansionLineNumber(L);
8965 llvm::dbgs() << StatusMsg << ": " << FileName << ": " << LineNo << "\n";
8966}
8967#endif
8968
8969static void emitTargetTeamsGenericLoopRegionAsParallel(
8970 CodeGenFunction &CGF, PrePostActionTy &Action,
8971 const OMPTargetTeamsGenericLoopDirective &S) {
8972 Action.Enter(CGF);
8973 // Emit 'teams loop' as if its constituent constructs are 'distribute,
8974 // 'parallel, and 'for'.
8975 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
8976 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitInnerParallelForWhenCombined,
8977 IncExpr: S.getDistInc());
8978 };
8979
8980 // Emit teams region as a standalone region.
8981 auto &&CodeGenTeams = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
8982 PrePostActionTy &Action) {
8983 Action.Enter(CGF);
8984 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
8985 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
8986 (void)PrivateScope.Privatize();
8987 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(
8988 CGF, InnermostKind: OMPD_distribute, CodeGen: CodeGenDistribute, /*HasCancel=*/false);
8989 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
8990 };
8991 DEBUG_WITH_TYPE(TTL_CODEGEN_TYPE,
8992 emitTargetTeamsLoopCodegenStatus(
8993 CGF, TTL_CODEGEN_TYPE " as parallel for", S));
8994 emitCommonOMPTeamsDirective(CGF, S, InnermostKind: OMPD_distribute_parallel_for,
8995 CodeGen: CodeGenTeams);
8996 emitPostUpdateForReductionClause(CGF, D: S,
8997 CondGen: [](CodeGenFunction &) { return nullptr; });
8998}
8999
9000static void emitTargetTeamsGenericLoopRegionAsDistribute(
9001 CodeGenFunction &CGF, PrePostActionTy &Action,
9002 const OMPTargetTeamsGenericLoopDirective &S) {
9003 Action.Enter(CGF);
9004 // Emit 'teams loop' as if its constituent construct is 'distribute'.
9005 auto &&CodeGenDistribute = [&S](CodeGenFunction &CGF, PrePostActionTy &) {
9006 CGF.EmitOMPDistributeLoop(S, CodeGenLoop: emitOMPLoopBodyWithStopPoint, IncExpr: S.getInc());
9007 };
9008
9009 // Emit teams region as a standalone region.
9010 auto &&CodeGen = [&S, &CodeGenDistribute](CodeGenFunction &CGF,
9011 PrePostActionTy &Action) {
9012 Action.Enter(CGF);
9013 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
9014 CGF.EmitOMPReductionClauseInit(D: S, PrivateScope);
9015 (void)PrivateScope.Privatize();
9016 CGF.CGM.getOpenMPRuntime().emitInlinedDirective(
9017 CGF, InnermostKind: OMPD_distribute, CodeGen: CodeGenDistribute, /*HasCancel=*/false);
9018 CGF.EmitOMPReductionClauseFinal(D: S, /*ReductionKind=*/OMPD_teams);
9019 };
9020 DEBUG_WITH_TYPE(TTL_CODEGEN_TYPE,
9021 emitTargetTeamsLoopCodegenStatus(
9022 CGF, TTL_CODEGEN_TYPE " as distribute", S));
9023 emitCommonOMPTeamsDirective(CGF, S, InnermostKind: OMPD_distribute, CodeGen);
9024 emitPostUpdateForReductionClause(CGF, D: S,
9025 CondGen: [](CodeGenFunction &) { return nullptr; });
9026}
9027
9028void CodeGenFunction::EmitOMPTargetTeamsGenericLoopDirective(
9029 const OMPTargetTeamsGenericLoopDirective &S) {
9030 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
9031 if (S.canBeParallelFor())
9032 emitTargetTeamsGenericLoopRegionAsParallel(CGF, Action, S);
9033 else
9034 emitTargetTeamsGenericLoopRegionAsDistribute(CGF, Action, S);
9035 };
9036 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
9037}
9038
9039void CodeGenFunction::EmitOMPTargetTeamsGenericLoopDeviceFunction(
9040 CodeGenModule &CGM, StringRef ParentName,
9041 const OMPTargetTeamsGenericLoopDirective &S) {
9042 // Emit SPMD target parallel loop region as a standalone region.
9043 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
9044 if (S.canBeParallelFor())
9045 emitTargetTeamsGenericLoopRegionAsParallel(CGF, Action, S);
9046 else
9047 emitTargetTeamsGenericLoopRegionAsDistribute(CGF, Action, S);
9048 };
9049 llvm::Function *Fn;
9050 llvm::Constant *Addr;
9051 // Emit target region as a standalone region.
9052 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
9053 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
9054 assert(Fn && Addr &&
9055 "Target device function emission failed for 'target teams loop'.");
9056}
9057
9058static void emitTargetParallelGenericLoopRegion(
9059 CodeGenFunction &CGF, const OMPTargetParallelGenericLoopDirective &S,
9060 PrePostActionTy &Action) {
9061 Action.Enter(CGF);
9062 // Emit as 'parallel for'.
9063 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
9064 Action.Enter(CGF);
9065 CodeGenFunction::OMPCancelStackRAII CancelRegion(
9066 CGF, OMPD_target_parallel_loop, /*hasCancel=*/false);
9067 CGF.EmitOMPWorksharingLoop(S, EUB: S.getEnsureUpperBound(), CodeGenLoopBounds: emitForLoopBounds,
9068 CGDispatchBounds: emitDispatchForLoopBounds);
9069 };
9070 emitCommonOMPParallelDirective(CGF, S, InnermostKind: OMPD_for, CodeGen,
9071 CodeGenBoundParameters: emitEmptyBoundParameters);
9072}
9073
9074void CodeGenFunction::EmitOMPTargetParallelGenericLoopDeviceFunction(
9075 CodeGenModule &CGM, StringRef ParentName,
9076 const OMPTargetParallelGenericLoopDirective &S) {
9077 // Emit target parallel loop region as a standalone region.
9078 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
9079 emitTargetParallelGenericLoopRegion(CGF, S, Action);
9080 };
9081 llvm::Function *Fn;
9082 llvm::Constant *Addr;
9083 // Emit target region as a standalone region.
9084 CGM.getOpenMPRuntime().emitTargetOutlinedFunction(
9085 D: S, ParentName, OutlinedFn&: Fn, OutlinedFnID&: Addr, /*IsOffloadEntry=*/true, CodeGen);
9086 assert(Fn && Addr && "Target device function emission failed.");
9087}
9088
9089/// Emit combined directive 'target parallel loop' as if its constituent
9090/// constructs are 'target', 'parallel', and 'for'.
9091void CodeGenFunction::EmitOMPTargetParallelGenericLoopDirective(
9092 const OMPTargetParallelGenericLoopDirective &S) {
9093 auto &&CodeGen = [&S](CodeGenFunction &CGF, PrePostActionTy &Action) {
9094 emitTargetParallelGenericLoopRegion(CGF, S, Action);
9095 };
9096 emitCommonOMPTargetDirective(CGF&: *this, S, CodeGen);
9097}
9098
9099void CodeGenFunction::EmitSimpleOMPExecutableDirective(
9100 const OMPExecutableDirective &D) {
9101 if (const auto *SD = dyn_cast<OMPScanDirective>(Val: &D)) {
9102 EmitOMPScanDirective(S: *SD);
9103 return;
9104 }
9105 if (!D.hasAssociatedStmt() || !D.getAssociatedStmt())
9106 return;
9107 auto &&CodeGen = [&D](CodeGenFunction &CGF, PrePostActionTy &Action) {
9108 OMPPrivateScope GlobalsScope(CGF);
9109 if (isOpenMPTaskingDirective(Kind: D.getDirectiveKind())) {
9110 // Capture global firstprivates to avoid crash.
9111 for (const auto *C : D.getClausesOfKind<OMPFirstprivateClause>()) {
9112 for (const Expr *Ref : C->varlist()) {
9113 const auto *DRE = cast<DeclRefExpr>(Val: Ref->IgnoreParenImpCasts());
9114 if (!DRE)
9115 continue;
9116 const auto *VD = dyn_cast<VarDecl>(Val: DRE->getDecl());
9117 if (!VD || VD->hasLocalStorage())
9118 continue;
9119 if (!CGF.LocalDeclMap.count(Val: VD)) {
9120 LValue GlobLVal = CGF.EmitLValue(E: Ref);
9121 GlobalsScope.addPrivate(LocalVD: VD, Addr: GlobLVal.getAddress());
9122 }
9123 }
9124 }
9125 }
9126 if (isOpenMPSimdDirective(DKind: D.getDirectiveKind())) {
9127 (void)GlobalsScope.Privatize();
9128 ParentLoopDirectiveForScanRegion ScanRegion(CGF, D);
9129 emitOMPSimdRegion(CGF, S: cast<OMPLoopDirective>(Val: D), Action);
9130 } else {
9131 if (const auto *LD = dyn_cast<OMPLoopDirective>(Val: &D)) {
9132 for (const Expr *E : LD->counters()) {
9133 const auto *VD = cast<VarDecl>(Val: cast<DeclRefExpr>(Val: E)->getDecl());
9134 if (!VD->hasLocalStorage() && !CGF.LocalDeclMap.count(Val: VD)) {
9135 LValue GlobLVal = CGF.EmitLValue(E);
9136 GlobalsScope.addPrivate(LocalVD: VD, Addr: GlobLVal.getAddress());
9137 }
9138 if (isa<OMPCapturedExprDecl>(Val: VD)) {
9139 // Emit only those that were not explicitly referenced in clauses.
9140 if (!CGF.LocalDeclMap.count(Val: VD))
9141 CGF.EmitVarDecl(D: *VD);
9142 }
9143 }
9144 for (const auto *C : D.getClausesOfKind<OMPOrderedClause>()) {
9145 if (!C->getNumForLoops())
9146 continue;
9147 for (unsigned I = LD->getLoopsNumber(),
9148 E = C->getLoopNumIterations().size();
9149 I < E; ++I) {
9150 if (const auto *VD = dyn_cast<OMPCapturedExprDecl>(
9151 Val: cast<DeclRefExpr>(Val: C->getLoopCounter(NumLoop: I))->getDecl())) {
9152 // Emit only those that were not explicitly referenced in clauses.
9153 if (!CGF.LocalDeclMap.count(Val: VD))
9154 CGF.EmitVarDecl(D: *VD);
9155 }
9156 }
9157 }
9158 }
9159 (void)GlobalsScope.Privatize();
9160 CGF.EmitStmt(S: D.getInnermostCapturedStmt()->getCapturedStmt());
9161 }
9162 };
9163 if (D.getDirectiveKind() == OMPD_atomic ||
9164 D.getDirectiveKind() == OMPD_critical ||
9165 D.getDirectiveKind() == OMPD_section ||
9166 D.getDirectiveKind() == OMPD_master ||
9167 D.getDirectiveKind() == OMPD_masked ||
9168 D.getDirectiveKind() == OMPD_unroll ||
9169 D.getDirectiveKind() == OMPD_assume) {
9170 EmitStmt(S: D.getAssociatedStmt());
9171 } else {
9172 auto LPCRegion =
9173 CGOpenMPRuntime::LastprivateConditionalRAII::disable(CGF&: *this, S: D);
9174 OMPSimdLexicalScope Scope(*this, D);
9175 CGM.getOpenMPRuntime().emitInlinedDirective(
9176 CGF&: *this,
9177 InnermostKind: isOpenMPSimdDirective(DKind: D.getDirectiveKind()) ? OMPD_simd
9178 : D.getDirectiveKind(),
9179 CodeGen);
9180 }
9181 // Check for outer lastprivate conditional update.
9182 checkForLastprivateConditionalUpdate(CGF&: *this, S: D);
9183}
9184
9185void CodeGenFunction::EmitOMPAssumeDirective(const OMPAssumeDirective &S) {
9186 for (const auto *C : S.getClausesOfKind<OMPHoldsClause>()) {
9187 const Expr *E = C->getExpr();
9188 assert(E && "holds clause requires an expression");
9189 if (!E->HasSideEffects(Ctx: getContext()))
9190 Builder.CreateAssumption(Cond: EvaluateExprAsBool(E));
9191 }
9192 EmitStmt(S: S.getAssociatedStmt());
9193}
9194