1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This provides a class for OpenMP runtime code generation.
10 //
11 //===----------------------------------------------------------------------===//
12 
13 #include "CGOpenMPRuntime.h"
14 #include "CGCXXABI.h"
15 #include "CGCleanup.h"
16 #include "CGRecordLayout.h"
17 #include "CodeGenFunction.h"
18 #include "clang/AST/Attr.h"
19 #include "clang/AST/Decl.h"
20 #include "clang/AST/OpenMPClause.h"
21 #include "clang/AST/StmtOpenMP.h"
22 #include "clang/AST/StmtVisitor.h"
23 #include "clang/Basic/BitmaskEnum.h"
24 #include "clang/Basic/OpenMPKinds.h"
25 #include "clang/CodeGen/ConstantInitBuilder.h"
26 #include "llvm/ADT/ArrayRef.h"
27 #include "llvm/ADT/SetOperations.h"
28 #include "llvm/ADT/StringExtras.h"
29 #include "llvm/Bitcode/BitcodeReader.h"
30 #include "llvm/Frontend/OpenMP/OMPIRBuilder.h"
31 #include "llvm/IR/DerivedTypes.h"
32 #include "llvm/IR/GlobalValue.h"
33 #include "llvm/IR/Value.h"
34 #include "llvm/Support/AtomicOrdering.h"
35 #include "llvm/Support/Format.h"
36 #include "llvm/Support/raw_ostream.h"
37 #include <cassert>
38 
39 using namespace clang;
40 using namespace CodeGen;
41 using namespace llvm::omp;
42 
43 namespace {
44 /// Base class for handling code generation inside OpenMP regions.
45 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
46 public:
47   /// Kinds of OpenMP regions used in codegen.
48   enum CGOpenMPRegionKind {
49     /// Region with outlined function for standalone 'parallel'
50     /// directive.
51     ParallelOutlinedRegion,
52     /// Region with outlined function for standalone 'task' directive.
53     TaskOutlinedRegion,
54     /// Region for constructs that do not require function outlining,
55     /// like 'for', 'sections', 'atomic' etc. directives.
56     InlinedRegion,
57     /// Region with outlined function for standalone 'target' directive.
58     TargetRegion,
59   };
60 
61   CGOpenMPRegionInfo(const CapturedStmt &CS,
62                      const CGOpenMPRegionKind RegionKind,
63                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
64                      bool HasCancel)
65       : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
66         CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {}
67 
68   CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
69                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
70                      bool HasCancel)
71       : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen),
72         Kind(Kind), HasCancel(HasCancel) {}
73 
74   /// Get a variable or parameter for storing global thread id
75   /// inside OpenMP construct.
76   virtual const VarDecl *getThreadIDVariable() const = 0;
77 
78   /// Emit the captured statement body.
79   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
80 
81   /// Get an LValue for the current ThreadID variable.
82   /// \return LValue for thread id variable. This LValue always has type int32*.
83   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
84 
85   virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {}
86 
87   CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
88 
89   OpenMPDirectiveKind getDirectiveKind() const { return Kind; }
90 
91   bool hasCancel() const { return HasCancel; }
92 
93   static bool classof(const CGCapturedStmtInfo *Info) {
94     return Info->getKind() == CR_OpenMP;
95   }
96 
97   ~CGOpenMPRegionInfo() override = default;
98 
99 protected:
100   CGOpenMPRegionKind RegionKind;
101   RegionCodeGenTy CodeGen;
102   OpenMPDirectiveKind Kind;
103   bool HasCancel;
104 };
105 
106 /// API for captured statement code generation in OpenMP constructs.
107 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo {
108 public:
109   CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
110                              const RegionCodeGenTy &CodeGen,
111                              OpenMPDirectiveKind Kind, bool HasCancel,
112                              StringRef HelperName)
113       : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind,
114                            HasCancel),
115         ThreadIDVar(ThreadIDVar), HelperName(HelperName) {
116     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
117   }
118 
119   /// Get a variable or parameter for storing global thread id
120   /// inside OpenMP construct.
121   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
122 
123   /// Get the name of the capture helper.
124   StringRef getHelperName() const override { return HelperName; }
125 
126   static bool classof(const CGCapturedStmtInfo *Info) {
127     return CGOpenMPRegionInfo::classof(Info) &&
128            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
129                ParallelOutlinedRegion;
130   }
131 
132 private:
133   /// A variable or parameter storing global thread id for OpenMP
134   /// constructs.
135   const VarDecl *ThreadIDVar;
136   StringRef HelperName;
137 };
138 
139 /// API for captured statement code generation in OpenMP constructs.
140 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo {
141 public:
142   class UntiedTaskActionTy final : public PrePostActionTy {
143     bool Untied;
144     const VarDecl *PartIDVar;
145     const RegionCodeGenTy UntiedCodeGen;
146     llvm::SwitchInst *UntiedSwitch = nullptr;
147 
148   public:
149     UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar,
150                        const RegionCodeGenTy &UntiedCodeGen)
151         : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {}
152     void Enter(CodeGenFunction &CGF) override {
153       if (Untied) {
154         // Emit task switching point.
155         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
156             CGF.GetAddrOfLocalVar(PartIDVar),
157             PartIDVar->getType()->castAs<PointerType>());
158         llvm::Value *Res =
159             CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation());
160         llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done.");
161         UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB);
162         CGF.EmitBlock(DoneBB);
163         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
164         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
165         UntiedSwitch->addCase(CGF.Builder.getInt32(0),
166                               CGF.Builder.GetInsertBlock());
167         emitUntiedSwitch(CGF);
168       }
169     }
170     void emitUntiedSwitch(CodeGenFunction &CGF) const {
171       if (Untied) {
172         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
173             CGF.GetAddrOfLocalVar(PartIDVar),
174             PartIDVar->getType()->castAs<PointerType>());
175         CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
176                               PartIdLVal);
177         UntiedCodeGen(CGF);
178         CodeGenFunction::JumpDest CurPoint =
179             CGF.getJumpDestInCurrentScope(".untied.next.");
180         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
181         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
182         UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
183                               CGF.Builder.GetInsertBlock());
184         CGF.EmitBranchThroughCleanup(CurPoint);
185         CGF.EmitBlock(CurPoint.getBlock());
186       }
187     }
188     unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); }
189   };
190   CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
191                                  const VarDecl *ThreadIDVar,
192                                  const RegionCodeGenTy &CodeGen,
193                                  OpenMPDirectiveKind Kind, bool HasCancel,
194                                  const UntiedTaskActionTy &Action)
195       : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel),
196         ThreadIDVar(ThreadIDVar), Action(Action) {
197     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
198   }
199 
200   /// Get a variable or parameter for storing global thread id
201   /// inside OpenMP construct.
202   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
203 
204   /// Get an LValue for the current ThreadID variable.
205   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
206 
207   /// Get the name of the capture helper.
208   StringRef getHelperName() const override { return ".omp_outlined."; }
209 
210   void emitUntiedSwitch(CodeGenFunction &CGF) override {
211     Action.emitUntiedSwitch(CGF);
212   }
213 
214   static bool classof(const CGCapturedStmtInfo *Info) {
215     return CGOpenMPRegionInfo::classof(Info) &&
216            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
217                TaskOutlinedRegion;
218   }
219 
220 private:
221   /// A variable or parameter storing global thread id for OpenMP
222   /// constructs.
223   const VarDecl *ThreadIDVar;
224   /// Action for emitting code for untied tasks.
225   const UntiedTaskActionTy &Action;
226 };
227 
228 /// API for inlined captured statement code generation in OpenMP
229 /// constructs.
230 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
231 public:
232   CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
233                             const RegionCodeGenTy &CodeGen,
234                             OpenMPDirectiveKind Kind, bool HasCancel)
235       : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel),
236         OldCSI(OldCSI),
237         OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
238 
239   // Retrieve the value of the context parameter.
240   llvm::Value *getContextValue() const override {
241     if (OuterRegionInfo)
242       return OuterRegionInfo->getContextValue();
243     llvm_unreachable("No context value for inlined OpenMP region");
244   }
245 
246   void setContextValue(llvm::Value *V) override {
247     if (OuterRegionInfo) {
248       OuterRegionInfo->setContextValue(V);
249       return;
250     }
251     llvm_unreachable("No context value for inlined OpenMP region");
252   }
253 
254   /// Lookup the captured field decl for a variable.
255   const FieldDecl *lookup(const VarDecl *VD) const override {
256     if (OuterRegionInfo)
257       return OuterRegionInfo->lookup(VD);
258     // If there is no outer outlined region,no need to lookup in a list of
259     // captured variables, we can use the original one.
260     return nullptr;
261   }
262 
263   FieldDecl *getThisFieldDecl() const override {
264     if (OuterRegionInfo)
265       return OuterRegionInfo->getThisFieldDecl();
266     return nullptr;
267   }
268 
269   /// Get a variable or parameter for storing global thread id
270   /// inside OpenMP construct.
271   const VarDecl *getThreadIDVariable() const override {
272     if (OuterRegionInfo)
273       return OuterRegionInfo->getThreadIDVariable();
274     return nullptr;
275   }
276 
277   /// Get an LValue for the current ThreadID variable.
278   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override {
279     if (OuterRegionInfo)
280       return OuterRegionInfo->getThreadIDVariableLValue(CGF);
281     llvm_unreachable("No LValue for inlined OpenMP construct");
282   }
283 
284   /// Get the name of the capture helper.
285   StringRef getHelperName() const override {
286     if (auto *OuterRegionInfo = getOldCSI())
287       return OuterRegionInfo->getHelperName();
288     llvm_unreachable("No helper name for inlined OpenMP construct");
289   }
290 
291   void emitUntiedSwitch(CodeGenFunction &CGF) override {
292     if (OuterRegionInfo)
293       OuterRegionInfo->emitUntiedSwitch(CGF);
294   }
295 
296   CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
297 
298   static bool classof(const CGCapturedStmtInfo *Info) {
299     return CGOpenMPRegionInfo::classof(Info) &&
300            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion;
301   }
302 
303   ~CGOpenMPInlinedRegionInfo() override = default;
304 
305 private:
306   /// CodeGen info about outer OpenMP region.
307   CodeGenFunction::CGCapturedStmtInfo *OldCSI;
308   CGOpenMPRegionInfo *OuterRegionInfo;
309 };
310 
311 /// API for captured statement code generation in OpenMP target
312 /// constructs. For this captures, implicit parameters are used instead of the
313 /// captured fields. The name of the target region has to be unique in a given
314 /// application so it is provided by the client, because only the client has
315 /// the information to generate that.
316 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo {
317 public:
318   CGOpenMPTargetRegionInfo(const CapturedStmt &CS,
319                            const RegionCodeGenTy &CodeGen, StringRef HelperName)
320       : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target,
321                            /*HasCancel=*/false),
322         HelperName(HelperName) {}
323 
324   /// This is unused for target regions because each starts executing
325   /// with a single thread.
326   const VarDecl *getThreadIDVariable() const override { return nullptr; }
327 
328   /// Get the name of the capture helper.
329   StringRef getHelperName() const override { return HelperName; }
330 
331   static bool classof(const CGCapturedStmtInfo *Info) {
332     return CGOpenMPRegionInfo::classof(Info) &&
333            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion;
334   }
335 
336 private:
337   StringRef HelperName;
338 };
339 
340 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) {
341   llvm_unreachable("No codegen for expressions");
342 }
343 /// API for generation of expressions captured in a innermost OpenMP
344 /// region.
345 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo {
346 public:
347   CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS)
348       : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen,
349                                   OMPD_unknown,
350                                   /*HasCancel=*/false),
351         PrivScope(CGF) {
352     // Make sure the globals captured in the provided statement are local by
353     // using the privatization logic. We assume the same variable is not
354     // captured more than once.
355     for (const auto &C : CS.captures()) {
356       if (!C.capturesVariable() && !C.capturesVariableByCopy())
357         continue;
358 
359       const VarDecl *VD = C.getCapturedVar();
360       if (VD->isLocalVarDeclOrParm())
361         continue;
362 
363       DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
364                       /*RefersToEnclosingVariableOrCapture=*/false,
365                       VD->getType().getNonReferenceType(), VK_LValue,
366                       C.getLocation());
367       PrivScope.addPrivate(
368           VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(CGF); });
369     }
370     (void)PrivScope.Privatize();
371   }
372 
373   /// Lookup the captured field decl for a variable.
374   const FieldDecl *lookup(const VarDecl *VD) const override {
375     if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD))
376       return FD;
377     return nullptr;
378   }
379 
380   /// Emit the captured statement body.
381   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override {
382     llvm_unreachable("No body for expressions");
383   }
384 
385   /// Get a variable or parameter for storing global thread id
386   /// inside OpenMP construct.
387   const VarDecl *getThreadIDVariable() const override {
388     llvm_unreachable("No thread id for expressions");
389   }
390 
391   /// Get the name of the capture helper.
392   StringRef getHelperName() const override {
393     llvm_unreachable("No helper name for expressions");
394   }
395 
396   static bool classof(const CGCapturedStmtInfo *Info) { return false; }
397 
398 private:
399   /// Private scope to capture global variables.
400   CodeGenFunction::OMPPrivateScope PrivScope;
401 };
402 
403 /// RAII for emitting code of OpenMP constructs.
404 class InlinedOpenMPRegionRAII {
405   CodeGenFunction &CGF;
406   llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields;
407   FieldDecl *LambdaThisCaptureField = nullptr;
408   const CodeGen::CGBlockInfo *BlockInfo = nullptr;
409 
410 public:
411   /// Constructs region for combined constructs.
412   /// \param CodeGen Code generation sequence for combined directives. Includes
413   /// a list of functions used for code generation of implicitly inlined
414   /// regions.
415   InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen,
416                           OpenMPDirectiveKind Kind, bool HasCancel)
417       : CGF(CGF) {
418     // Start emission for the construct.
419     CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(
420         CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel);
421     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
422     LambdaThisCaptureField = CGF.LambdaThisCaptureField;
423     CGF.LambdaThisCaptureField = nullptr;
424     BlockInfo = CGF.BlockInfo;
425     CGF.BlockInfo = nullptr;
426   }
427 
428   ~InlinedOpenMPRegionRAII() {
429     // Restore original CapturedStmtInfo only if we're done with code emission.
430     auto *OldCSI =
431         cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
432     delete CGF.CapturedStmtInfo;
433     CGF.CapturedStmtInfo = OldCSI;
434     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
435     CGF.LambdaThisCaptureField = LambdaThisCaptureField;
436     CGF.BlockInfo = BlockInfo;
437   }
438 };
439 
440 /// Values for bit flags used in the ident_t to describe the fields.
441 /// All enumeric elements are named and described in accordance with the code
442 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
443 enum OpenMPLocationFlags : unsigned {
444   /// Use trampoline for internal microtask.
445   OMP_IDENT_IMD = 0x01,
446   /// Use c-style ident structure.
447   OMP_IDENT_KMPC = 0x02,
448   /// Atomic reduction option for kmpc_reduce.
449   OMP_ATOMIC_REDUCE = 0x10,
450   /// Explicit 'barrier' directive.
451   OMP_IDENT_BARRIER_EXPL = 0x20,
452   /// Implicit barrier in code.
453   OMP_IDENT_BARRIER_IMPL = 0x40,
454   /// Implicit barrier in 'for' directive.
455   OMP_IDENT_BARRIER_IMPL_FOR = 0x40,
456   /// Implicit barrier in 'sections' directive.
457   OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0,
458   /// Implicit barrier in 'single' directive.
459   OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140,
460   /// Call of __kmp_for_static_init for static loop.
461   OMP_IDENT_WORK_LOOP = 0x200,
462   /// Call of __kmp_for_static_init for sections.
463   OMP_IDENT_WORK_SECTIONS = 0x400,
464   /// Call of __kmp_for_static_init for distribute.
465   OMP_IDENT_WORK_DISTRIBUTE = 0x800,
466   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE)
467 };
468 
469 namespace {
470 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
471 /// Values for bit flags for marking which requires clauses have been used.
472 enum OpenMPOffloadingRequiresDirFlags : int64_t {
473   /// flag undefined.
474   OMP_REQ_UNDEFINED               = 0x000,
475   /// no requires clause present.
476   OMP_REQ_NONE                    = 0x001,
477   /// reverse_offload clause.
478   OMP_REQ_REVERSE_OFFLOAD         = 0x002,
479   /// unified_address clause.
480   OMP_REQ_UNIFIED_ADDRESS         = 0x004,
481   /// unified_shared_memory clause.
482   OMP_REQ_UNIFIED_SHARED_MEMORY   = 0x008,
483   /// dynamic_allocators clause.
484   OMP_REQ_DYNAMIC_ALLOCATORS      = 0x010,
485   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS)
486 };
487 
488 enum OpenMPOffloadingReservedDeviceIDs {
489   /// Device ID if the device was not defined, runtime should get it
490   /// from environment variables in the spec.
491   OMP_DEVICEID_UNDEF = -1,
492 };
493 } // anonymous namespace
494 
495 /// Describes ident structure that describes a source location.
496 /// All descriptions are taken from
497 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
498 /// Original structure:
499 /// typedef struct ident {
500 ///    kmp_int32 reserved_1;   /**<  might be used in Fortran;
501 ///                                  see above  */
502 ///    kmp_int32 flags;        /**<  also f.flags; KMP_IDENT_xxx flags;
503 ///                                  KMP_IDENT_KMPC identifies this union
504 ///                                  member  */
505 ///    kmp_int32 reserved_2;   /**<  not really used in Fortran any more;
506 ///                                  see above */
507 ///#if USE_ITT_BUILD
508 ///                            /*  but currently used for storing
509 ///                                region-specific ITT */
510 ///                            /*  contextual information. */
511 ///#endif /* USE_ITT_BUILD */
512 ///    kmp_int32 reserved_3;   /**< source[4] in Fortran, do not use for
513 ///                                 C++  */
514 ///    char const *psource;    /**< String describing the source location.
515 ///                            The string is composed of semi-colon separated
516 //                             fields which describe the source file,
517 ///                            the function and a pair of line numbers that
518 ///                            delimit the construct.
519 ///                             */
520 /// } ident_t;
521 enum IdentFieldIndex {
522   /// might be used in Fortran
523   IdentField_Reserved_1,
524   /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member.
525   IdentField_Flags,
526   /// Not really used in Fortran any more
527   IdentField_Reserved_2,
528   /// Source[4] in Fortran, do not use for C++
529   IdentField_Reserved_3,
530   /// String describing the source location. The string is composed of
531   /// semi-colon separated fields which describe the source file, the function
532   /// and a pair of line numbers that delimit the construct.
533   IdentField_PSource
534 };
535 
536 /// Schedule types for 'omp for' loops (these enumerators are taken from
537 /// the enum sched_type in kmp.h).
538 enum OpenMPSchedType {
539   /// Lower bound for default (unordered) versions.
540   OMP_sch_lower = 32,
541   OMP_sch_static_chunked = 33,
542   OMP_sch_static = 34,
543   OMP_sch_dynamic_chunked = 35,
544   OMP_sch_guided_chunked = 36,
545   OMP_sch_runtime = 37,
546   OMP_sch_auto = 38,
547   /// static with chunk adjustment (e.g., simd)
548   OMP_sch_static_balanced_chunked = 45,
549   /// Lower bound for 'ordered' versions.
550   OMP_ord_lower = 64,
551   OMP_ord_static_chunked = 65,
552   OMP_ord_static = 66,
553   OMP_ord_dynamic_chunked = 67,
554   OMP_ord_guided_chunked = 68,
555   OMP_ord_runtime = 69,
556   OMP_ord_auto = 70,
557   OMP_sch_default = OMP_sch_static,
558   /// dist_schedule types
559   OMP_dist_sch_static_chunked = 91,
560   OMP_dist_sch_static = 92,
561   /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers.
562   /// Set if the monotonic schedule modifier was present.
563   OMP_sch_modifier_monotonic = (1 << 29),
564   /// Set if the nonmonotonic schedule modifier was present.
565   OMP_sch_modifier_nonmonotonic = (1 << 30),
566 };
567 
568 enum OpenMPRTLFunction {
569   /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc,
570   /// kmpc_micro microtask, ...);
571   OMPRTL__kmpc_fork_call,
572   /// Call to void *__kmpc_threadprivate_cached(ident_t *loc,
573   /// kmp_int32 global_tid, void *data, size_t size, void ***cache);
574   OMPRTL__kmpc_threadprivate_cached,
575   /// Call to void __kmpc_threadprivate_register( ident_t *,
576   /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
577   OMPRTL__kmpc_threadprivate_register,
578   // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc);
579   OMPRTL__kmpc_global_thread_num,
580   // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
581   // kmp_critical_name *crit);
582   OMPRTL__kmpc_critical,
583   // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32
584   // global_tid, kmp_critical_name *crit, uintptr_t hint);
585   OMPRTL__kmpc_critical_with_hint,
586   // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
587   // kmp_critical_name *crit);
588   OMPRTL__kmpc_end_critical,
589   // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
590   // global_tid);
591   OMPRTL__kmpc_cancel_barrier,
592   // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
593   OMPRTL__kmpc_barrier,
594   // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
595   OMPRTL__kmpc_for_static_fini,
596   // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
597   // global_tid);
598   OMPRTL__kmpc_serialized_parallel,
599   // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
600   // global_tid);
601   OMPRTL__kmpc_end_serialized_parallel,
602   // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
603   // kmp_int32 num_threads);
604   OMPRTL__kmpc_push_num_threads,
605   // Call to void __kmpc_flush(ident_t *loc);
606   OMPRTL__kmpc_flush,
607   // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid);
608   OMPRTL__kmpc_master,
609   // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid);
610   OMPRTL__kmpc_end_master,
611   // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
612   // int end_part);
613   OMPRTL__kmpc_omp_taskyield,
614   // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid);
615   OMPRTL__kmpc_single,
616   // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid);
617   OMPRTL__kmpc_end_single,
618   // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
619   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
620   // kmp_routine_entry_t *task_entry);
621   OMPRTL__kmpc_omp_task_alloc,
622   // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *,
623   // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t,
624   // size_t sizeof_shareds, kmp_routine_entry_t *task_entry,
625   // kmp_int64 device_id);
626   OMPRTL__kmpc_omp_target_task_alloc,
627   // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t *
628   // new_task);
629   OMPRTL__kmpc_omp_task,
630   // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
631   // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
632   // kmp_int32 didit);
633   OMPRTL__kmpc_copyprivate,
634   // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
635   // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
636   // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
637   OMPRTL__kmpc_reduce,
638   // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
639   // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
640   // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
641   // *lck);
642   OMPRTL__kmpc_reduce_nowait,
643   // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
644   // kmp_critical_name *lck);
645   OMPRTL__kmpc_end_reduce,
646   // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
647   // kmp_critical_name *lck);
648   OMPRTL__kmpc_end_reduce_nowait,
649   // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
650   // kmp_task_t * new_task);
651   OMPRTL__kmpc_omp_task_begin_if0,
652   // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
653   // kmp_task_t * new_task);
654   OMPRTL__kmpc_omp_task_complete_if0,
655   // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
656   OMPRTL__kmpc_ordered,
657   // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
658   OMPRTL__kmpc_end_ordered,
659   // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
660   // global_tid);
661   OMPRTL__kmpc_omp_taskwait,
662   // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
663   OMPRTL__kmpc_taskgroup,
664   // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
665   OMPRTL__kmpc_end_taskgroup,
666   // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
667   // int proc_bind);
668   OMPRTL__kmpc_push_proc_bind,
669   // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32
670   // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t
671   // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
672   OMPRTL__kmpc_omp_task_with_deps,
673   // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32
674   // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
675   // ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
676   OMPRTL__kmpc_omp_wait_deps,
677   // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
678   // global_tid, kmp_int32 cncl_kind);
679   OMPRTL__kmpc_cancellationpoint,
680   // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
681   // kmp_int32 cncl_kind);
682   OMPRTL__kmpc_cancel,
683   // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid,
684   // kmp_int32 num_teams, kmp_int32 thread_limit);
685   OMPRTL__kmpc_push_num_teams,
686   // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
687   // microtask, ...);
688   OMPRTL__kmpc_fork_teams,
689   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
690   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
691   // sched, kmp_uint64 grainsize, void *task_dup);
692   OMPRTL__kmpc_taskloop,
693   // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
694   // num_dims, struct kmp_dim *dims);
695   OMPRTL__kmpc_doacross_init,
696   // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
697   OMPRTL__kmpc_doacross_fini,
698   // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
699   // *vec);
700   OMPRTL__kmpc_doacross_post,
701   // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
702   // *vec);
703   OMPRTL__kmpc_doacross_wait,
704   // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void
705   // *data);
706   OMPRTL__kmpc_task_reduction_init,
707   // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
708   // *d);
709   OMPRTL__kmpc_task_reduction_get_th_data,
710   // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al);
711   OMPRTL__kmpc_alloc,
712   // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al);
713   OMPRTL__kmpc_free,
714 
715   //
716   // Offloading related calls
717   //
718   // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
719   // size);
720   OMPRTL__kmpc_push_target_tripcount,
721   // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
722   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
723   // *arg_types);
724   OMPRTL__tgt_target,
725   // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
726   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
727   // *arg_types);
728   OMPRTL__tgt_target_nowait,
729   // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
730   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
731   // *arg_types, int32_t num_teams, int32_t thread_limit);
732   OMPRTL__tgt_target_teams,
733   // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void
734   // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
735   // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
736   OMPRTL__tgt_target_teams_nowait,
737   // Call to void __tgt_register_requires(int64_t flags);
738   OMPRTL__tgt_register_requires,
739   // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
740   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
741   OMPRTL__tgt_target_data_begin,
742   // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
743   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
744   // *arg_types);
745   OMPRTL__tgt_target_data_begin_nowait,
746   // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
747   // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types);
748   OMPRTL__tgt_target_data_end,
749   // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t
750   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
751   // *arg_types);
752   OMPRTL__tgt_target_data_end_nowait,
753   // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
754   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
755   OMPRTL__tgt_target_data_update,
756   // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t
757   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
758   // *arg_types);
759   OMPRTL__tgt_target_data_update_nowait,
760   // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
761   OMPRTL__tgt_mapper_num_components,
762   // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void
763   // *base, void *begin, int64_t size, int64_t type);
764   OMPRTL__tgt_push_mapper_component,
765 };
766 
767 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP
768 /// region.
769 class CleanupTy final : public EHScopeStack::Cleanup {
770   PrePostActionTy *Action;
771 
772 public:
773   explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {}
774   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
775     if (!CGF.HaveInsertPoint())
776       return;
777     Action->Exit(CGF);
778   }
779 };
780 
781 } // anonymous namespace
782 
783 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const {
784   CodeGenFunction::RunCleanupsScope Scope(CGF);
785   if (PrePostAction) {
786     CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction);
787     Callback(CodeGen, CGF, *PrePostAction);
788   } else {
789     PrePostActionTy Action;
790     Callback(CodeGen, CGF, Action);
791   }
792 }
793 
794 /// Check if the combiner is a call to UDR combiner and if it is so return the
795 /// UDR decl used for reduction.
796 static const OMPDeclareReductionDecl *
797 getReductionInit(const Expr *ReductionOp) {
798   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
799     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
800       if (const auto *DRE =
801               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
802         if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl()))
803           return DRD;
804   return nullptr;
805 }
806 
807 static void emitInitWithReductionInitializer(CodeGenFunction &CGF,
808                                              const OMPDeclareReductionDecl *DRD,
809                                              const Expr *InitOp,
810                                              Address Private, Address Original,
811                                              QualType Ty) {
812   if (DRD->getInitializer()) {
813     std::pair<llvm::Function *, llvm::Function *> Reduction =
814         CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
815     const auto *CE = cast<CallExpr>(InitOp);
816     const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee());
817     const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts();
818     const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts();
819     const auto *LHSDRE =
820         cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr());
821     const auto *RHSDRE =
822         cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr());
823     CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
824     PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()),
825                             [=]() { return Private; });
826     PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()),
827                             [=]() { return Original; });
828     (void)PrivateScope.Privatize();
829     RValue Func = RValue::get(Reduction.second);
830     CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
831     CGF.EmitIgnoredExpr(InitOp);
832   } else {
833     llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty);
834     std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"});
835     auto *GV = new llvm::GlobalVariable(
836         CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true,
837         llvm::GlobalValue::PrivateLinkage, Init, Name);
838     LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty);
839     RValue InitRVal;
840     switch (CGF.getEvaluationKind(Ty)) {
841     case TEK_Scalar:
842       InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation());
843       break;
844     case TEK_Complex:
845       InitRVal =
846           RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation()));
847       break;
848     case TEK_Aggregate:
849       InitRVal = RValue::getAggregate(LV.getAddress(CGF));
850       break;
851     }
852     OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue);
853     CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal);
854     CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
855                          /*IsInitializer=*/false);
856   }
857 }
858 
859 /// Emit initialization of arrays of complex types.
860 /// \param DestAddr Address of the array.
861 /// \param Type Type of array.
862 /// \param Init Initial expression of array.
863 /// \param SrcAddr Address of the original array.
864 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr,
865                                  QualType Type, bool EmitDeclareReductionInit,
866                                  const Expr *Init,
867                                  const OMPDeclareReductionDecl *DRD,
868                                  Address SrcAddr = Address::invalid()) {
869   // Perform element-by-element initialization.
870   QualType ElementTy;
871 
872   // Drill down to the base element type on both arrays.
873   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
874   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr);
875   DestAddr =
876       CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType());
877   if (DRD)
878     SrcAddr =
879         CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType());
880 
881   llvm::Value *SrcBegin = nullptr;
882   if (DRD)
883     SrcBegin = SrcAddr.getPointer();
884   llvm::Value *DestBegin = DestAddr.getPointer();
885   // Cast from pointer to array type to pointer to single element.
886   llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements);
887   // The basic structure here is a while-do loop.
888   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body");
889   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done");
890   llvm::Value *IsEmpty =
891       CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty");
892   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
893 
894   // Enter the loop body, making that address the current address.
895   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
896   CGF.EmitBlock(BodyBB);
897 
898   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
899 
900   llvm::PHINode *SrcElementPHI = nullptr;
901   Address SrcElementCurrent = Address::invalid();
902   if (DRD) {
903     SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2,
904                                           "omp.arraycpy.srcElementPast");
905     SrcElementPHI->addIncoming(SrcBegin, EntryBB);
906     SrcElementCurrent =
907         Address(SrcElementPHI,
908                 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize));
909   }
910   llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI(
911       DestBegin->getType(), 2, "omp.arraycpy.destElementPast");
912   DestElementPHI->addIncoming(DestBegin, EntryBB);
913   Address DestElementCurrent =
914       Address(DestElementPHI,
915               DestAddr.getAlignment().alignmentOfArrayElement(ElementSize));
916 
917   // Emit copy.
918   {
919     CodeGenFunction::RunCleanupsScope InitScope(CGF);
920     if (EmitDeclareReductionInit) {
921       emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent,
922                                        SrcElementCurrent, ElementTy);
923     } else
924       CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(),
925                            /*IsInitializer=*/false);
926   }
927 
928   if (DRD) {
929     // Shift the address forward by one element.
930     llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32(
931         SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
932     SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock());
933   }
934 
935   // Shift the address forward by one element.
936   llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32(
937       DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
938   // Check whether we've reached the end.
939   llvm::Value *Done =
940       CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done");
941   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
942   DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock());
943 
944   // Done.
945   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
946 }
947 
948 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) {
949   return CGF.EmitOMPSharedLValue(E);
950 }
951 
952 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF,
953                                             const Expr *E) {
954   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E))
955     return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false);
956   return LValue();
957 }
958 
959 void ReductionCodeGen::emitAggregateInitialization(
960     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
961     const OMPDeclareReductionDecl *DRD) {
962   // Emit VarDecl with copy init for arrays.
963   // Get the address of the original variable captured in current
964   // captured region.
965   const auto *PrivateVD =
966       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
967   bool EmitDeclareReductionInit =
968       DRD && (DRD->getInitializer() || !PrivateVD->hasInit());
969   EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(),
970                        EmitDeclareReductionInit,
971                        EmitDeclareReductionInit ? ClausesData[N].ReductionOp
972                                                 : PrivateVD->getInit(),
973                        DRD, SharedLVal.getAddress(CGF));
974 }
975 
976 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds,
977                                    ArrayRef<const Expr *> Privates,
978                                    ArrayRef<const Expr *> ReductionOps) {
979   ClausesData.reserve(Shareds.size());
980   SharedAddresses.reserve(Shareds.size());
981   Sizes.reserve(Shareds.size());
982   BaseDecls.reserve(Shareds.size());
983   auto IPriv = Privates.begin();
984   auto IRed = ReductionOps.begin();
985   for (const Expr *Ref : Shareds) {
986     ClausesData.emplace_back(Ref, *IPriv, *IRed);
987     std::advance(IPriv, 1);
988     std::advance(IRed, 1);
989   }
990 }
991 
992 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) {
993   assert(SharedAddresses.size() == N &&
994          "Number of generated lvalues must be exactly N.");
995   LValue First = emitSharedLValue(CGF, ClausesData[N].Ref);
996   LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref);
997   SharedAddresses.emplace_back(First, Second);
998 }
999 
1000 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) {
1001   const auto *PrivateVD =
1002       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1003   QualType PrivateType = PrivateVD->getType();
1004   bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref);
1005   if (!PrivateType->isVariablyModifiedType()) {
1006     Sizes.emplace_back(
1007         CGF.getTypeSize(
1008             SharedAddresses[N].first.getType().getNonReferenceType()),
1009         nullptr);
1010     return;
1011   }
1012   llvm::Value *Size;
1013   llvm::Value *SizeInChars;
1014   auto *ElemType = cast<llvm::PointerType>(
1015                        SharedAddresses[N].first.getPointer(CGF)->getType())
1016                        ->getElementType();
1017   auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType);
1018   if (AsArraySection) {
1019     Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(CGF),
1020                                      SharedAddresses[N].first.getPointer(CGF));
1021     Size = CGF.Builder.CreateNUWAdd(
1022         Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1));
1023     SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf);
1024   } else {
1025     SizeInChars = CGF.getTypeSize(
1026         SharedAddresses[N].first.getType().getNonReferenceType());
1027     Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf);
1028   }
1029   Sizes.emplace_back(SizeInChars, Size);
1030   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1031       CGF,
1032       cast<OpaqueValueExpr>(
1033           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1034       RValue::get(Size));
1035   CGF.EmitVariablyModifiedType(PrivateType);
1036 }
1037 
1038 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N,
1039                                          llvm::Value *Size) {
1040   const auto *PrivateVD =
1041       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1042   QualType PrivateType = PrivateVD->getType();
1043   if (!PrivateType->isVariablyModifiedType()) {
1044     assert(!Size && !Sizes[N].second &&
1045            "Size should be nullptr for non-variably modified reduction "
1046            "items.");
1047     return;
1048   }
1049   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1050       CGF,
1051       cast<OpaqueValueExpr>(
1052           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1053       RValue::get(Size));
1054   CGF.EmitVariablyModifiedType(PrivateType);
1055 }
1056 
1057 void ReductionCodeGen::emitInitialization(
1058     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
1059     llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) {
1060   assert(SharedAddresses.size() > N && "No variable was generated");
1061   const auto *PrivateVD =
1062       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1063   const OMPDeclareReductionDecl *DRD =
1064       getReductionInit(ClausesData[N].ReductionOp);
1065   QualType PrivateType = PrivateVD->getType();
1066   PrivateAddr = CGF.Builder.CreateElementBitCast(
1067       PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1068   QualType SharedType = SharedAddresses[N].first.getType();
1069   SharedLVal = CGF.MakeAddrLValue(
1070       CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(CGF),
1071                                        CGF.ConvertTypeForMem(SharedType)),
1072       SharedType, SharedAddresses[N].first.getBaseInfo(),
1073       CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType));
1074   if (CGF.getContext().getAsArrayType(PrivateVD->getType())) {
1075     emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD);
1076   } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) {
1077     emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp,
1078                                      PrivateAddr, SharedLVal.getAddress(CGF),
1079                                      SharedLVal.getType());
1080   } else if (!DefaultInit(CGF) && PrivateVD->hasInit() &&
1081              !CGF.isTrivialInitializer(PrivateVD->getInit())) {
1082     CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr,
1083                          PrivateVD->getType().getQualifiers(),
1084                          /*IsInitializer=*/false);
1085   }
1086 }
1087 
1088 bool ReductionCodeGen::needCleanups(unsigned N) {
1089   const auto *PrivateVD =
1090       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1091   QualType PrivateType = PrivateVD->getType();
1092   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1093   return DTorKind != QualType::DK_none;
1094 }
1095 
1096 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N,
1097                                     Address PrivateAddr) {
1098   const auto *PrivateVD =
1099       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1100   QualType PrivateType = PrivateVD->getType();
1101   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1102   if (needCleanups(N)) {
1103     PrivateAddr = CGF.Builder.CreateElementBitCast(
1104         PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1105     CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType);
1106   }
1107 }
1108 
1109 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1110                           LValue BaseLV) {
1111   BaseTy = BaseTy.getNonReferenceType();
1112   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1113          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1114     if (const auto *PtrTy = BaseTy->getAs<PointerType>()) {
1115       BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy);
1116     } else {
1117       LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy);
1118       BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal);
1119     }
1120     BaseTy = BaseTy->getPointeeType();
1121   }
1122   return CGF.MakeAddrLValue(
1123       CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF),
1124                                        CGF.ConvertTypeForMem(ElTy)),
1125       BaseLV.getType(), BaseLV.getBaseInfo(),
1126       CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType()));
1127 }
1128 
1129 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1130                           llvm::Type *BaseLVType, CharUnits BaseLVAlignment,
1131                           llvm::Value *Addr) {
1132   Address Tmp = Address::invalid();
1133   Address TopTmp = Address::invalid();
1134   Address MostTopTmp = Address::invalid();
1135   BaseTy = BaseTy.getNonReferenceType();
1136   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1137          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1138     Tmp = CGF.CreateMemTemp(BaseTy);
1139     if (TopTmp.isValid())
1140       CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp);
1141     else
1142       MostTopTmp = Tmp;
1143     TopTmp = Tmp;
1144     BaseTy = BaseTy->getPointeeType();
1145   }
1146   llvm::Type *Ty = BaseLVType;
1147   if (Tmp.isValid())
1148     Ty = Tmp.getElementType();
1149   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty);
1150   if (Tmp.isValid()) {
1151     CGF.Builder.CreateStore(Addr, Tmp);
1152     return MostTopTmp;
1153   }
1154   return Address(Addr, BaseLVAlignment);
1155 }
1156 
1157 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) {
1158   const VarDecl *OrigVD = nullptr;
1159   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) {
1160     const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
1161     while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base))
1162       Base = TempOASE->getBase()->IgnoreParenImpCasts();
1163     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1164       Base = TempASE->getBase()->IgnoreParenImpCasts();
1165     DE = cast<DeclRefExpr>(Base);
1166     OrigVD = cast<VarDecl>(DE->getDecl());
1167   } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) {
1168     const Expr *Base = ASE->getBase()->IgnoreParenImpCasts();
1169     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1170       Base = TempASE->getBase()->IgnoreParenImpCasts();
1171     DE = cast<DeclRefExpr>(Base);
1172     OrigVD = cast<VarDecl>(DE->getDecl());
1173   }
1174   return OrigVD;
1175 }
1176 
1177 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N,
1178                                                Address PrivateAddr) {
1179   const DeclRefExpr *DE;
1180   if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) {
1181     BaseDecls.emplace_back(OrigVD);
1182     LValue OriginalBaseLValue = CGF.EmitLValue(DE);
1183     LValue BaseLValue =
1184         loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(),
1185                     OriginalBaseLValue);
1186     llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff(
1187         BaseLValue.getPointer(CGF), SharedAddresses[N].first.getPointer(CGF));
1188     llvm::Value *PrivatePointer =
1189         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1190             PrivateAddr.getPointer(),
1191             SharedAddresses[N].first.getAddress(CGF).getType());
1192     llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment);
1193     return castToBase(CGF, OrigVD->getType(),
1194                       SharedAddresses[N].first.getType(),
1195                       OriginalBaseLValue.getAddress(CGF).getType(),
1196                       OriginalBaseLValue.getAlignment(), Ptr);
1197   }
1198   BaseDecls.emplace_back(
1199       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl()));
1200   return PrivateAddr;
1201 }
1202 
1203 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const {
1204   const OMPDeclareReductionDecl *DRD =
1205       getReductionInit(ClausesData[N].ReductionOp);
1206   return DRD && DRD->getInitializer();
1207 }
1208 
1209 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
1210   return CGF.EmitLoadOfPointerLValue(
1211       CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1212       getThreadIDVariable()->getType()->castAs<PointerType>());
1213 }
1214 
1215 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) {
1216   if (!CGF.HaveInsertPoint())
1217     return;
1218   // 1.2.2 OpenMP Language Terminology
1219   // Structured block - An executable statement with a single entry at the
1220   // top and a single exit at the bottom.
1221   // The point of exit cannot be a branch out of the structured block.
1222   // longjmp() and throw() must not violate the entry/exit criteria.
1223   CGF.EHStack.pushTerminate();
1224   CodeGen(CGF);
1225   CGF.EHStack.popTerminate();
1226 }
1227 
1228 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
1229     CodeGenFunction &CGF) {
1230   return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1231                             getThreadIDVariable()->getType(),
1232                             AlignmentSource::Decl);
1233 }
1234 
1235 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1236                                        QualType FieldTy) {
1237   auto *Field = FieldDecl::Create(
1238       C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1239       C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1240       /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1241   Field->setAccess(AS_public);
1242   DC->addDecl(Field);
1243   return Field;
1244 }
1245 
1246 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator,
1247                                  StringRef Separator)
1248     : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator),
1249       OffloadEntriesInfoManager(CGM) {
1250   ASTContext &C = CGM.getContext();
1251   RecordDecl *RD = C.buildImplicitRecord("ident_t");
1252   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
1253   RD->startDefinition();
1254   // reserved_1
1255   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1256   // flags
1257   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1258   // reserved_2
1259   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1260   // reserved_3
1261   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1262   // psource
1263   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
1264   RD->completeDefinition();
1265   IdentQTy = C.getRecordType(RD);
1266   IdentTy = CGM.getTypes().ConvertRecordDeclType(RD);
1267   KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
1268 
1269   loadOffloadInfoMetadata();
1270 }
1271 
1272 bool CGOpenMPRuntime::tryEmitDeclareVariant(const GlobalDecl &NewGD,
1273                                             const GlobalDecl &OldGD,
1274                                             llvm::GlobalValue *OrigAddr,
1275                                             bool IsForDefinition) {
1276   // Emit at least a definition for the aliasee if the the address of the
1277   // original function is requested.
1278   if (IsForDefinition || OrigAddr)
1279     (void)CGM.GetAddrOfGlobal(NewGD);
1280   StringRef NewMangledName = CGM.getMangledName(NewGD);
1281   llvm::GlobalValue *Addr = CGM.GetGlobalValue(NewMangledName);
1282   if (Addr && !Addr->isDeclaration()) {
1283     const auto *D = cast<FunctionDecl>(OldGD.getDecl());
1284     const CGFunctionInfo &FI = CGM.getTypes().arrangeGlobalDeclaration(NewGD);
1285     llvm::Type *DeclTy = CGM.getTypes().GetFunctionType(FI);
1286 
1287     // Create a reference to the named value.  This ensures that it is emitted
1288     // if a deferred decl.
1289     llvm::GlobalValue::LinkageTypes LT = CGM.getFunctionLinkage(OldGD);
1290 
1291     // Create the new alias itself, but don't set a name yet.
1292     auto *GA =
1293         llvm::GlobalAlias::create(DeclTy, 0, LT, "", Addr, &CGM.getModule());
1294 
1295     if (OrigAddr) {
1296       assert(OrigAddr->isDeclaration() && "Expected declaration");
1297 
1298       GA->takeName(OrigAddr);
1299       OrigAddr->replaceAllUsesWith(
1300           llvm::ConstantExpr::getBitCast(GA, OrigAddr->getType()));
1301       OrigAddr->eraseFromParent();
1302     } else {
1303       GA->setName(CGM.getMangledName(OldGD));
1304     }
1305 
1306     // Set attributes which are particular to an alias; this is a
1307     // specialization of the attributes which may be set on a global function.
1308     if (D->hasAttr<WeakAttr>() || D->hasAttr<WeakRefAttr>() ||
1309         D->isWeakImported())
1310       GA->setLinkage(llvm::Function::WeakAnyLinkage);
1311 
1312     CGM.SetCommonAttributes(OldGD, GA);
1313     return true;
1314   }
1315   return false;
1316 }
1317 
1318 void CGOpenMPRuntime::clear() {
1319   InternalVars.clear();
1320   // Clean non-target variable declarations possibly used only in debug info.
1321   for (const auto &Data : EmittedNonTargetVariables) {
1322     if (!Data.getValue().pointsToAliveValue())
1323       continue;
1324     auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue());
1325     if (!GV)
1326       continue;
1327     if (!GV->isDeclaration() || GV->getNumUses() > 0)
1328       continue;
1329     GV->eraseFromParent();
1330   }
1331   // Emit aliases for the deferred aliasees.
1332   for (const auto &Pair : DeferredVariantFunction) {
1333     StringRef MangledName = CGM.getMangledName(Pair.second.second);
1334     llvm::GlobalValue *Addr = CGM.GetGlobalValue(MangledName);
1335     // If not able to emit alias, just emit original declaration.
1336     (void)tryEmitDeclareVariant(Pair.second.first, Pair.second.second, Addr,
1337                                 /*IsForDefinition=*/false);
1338   }
1339 }
1340 
1341 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const {
1342   SmallString<128> Buffer;
1343   llvm::raw_svector_ostream OS(Buffer);
1344   StringRef Sep = FirstSeparator;
1345   for (StringRef Part : Parts) {
1346     OS << Sep << Part;
1347     Sep = Separator;
1348   }
1349   return std::string(OS.str());
1350 }
1351 
1352 static llvm::Function *
1353 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty,
1354                           const Expr *CombinerInitializer, const VarDecl *In,
1355                           const VarDecl *Out, bool IsCombiner) {
1356   // void .omp_combiner.(Ty *in, Ty *out);
1357   ASTContext &C = CGM.getContext();
1358   QualType PtrTy = C.getPointerType(Ty).withRestrict();
1359   FunctionArgList Args;
1360   ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(),
1361                                /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1362   ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(),
1363                               /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1364   Args.push_back(&OmpOutParm);
1365   Args.push_back(&OmpInParm);
1366   const CGFunctionInfo &FnInfo =
1367       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
1368   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
1369   std::string Name = CGM.getOpenMPRuntime().getName(
1370       {IsCombiner ? "omp_combiner" : "omp_initializer", ""});
1371   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
1372                                     Name, &CGM.getModule());
1373   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
1374   if (CGM.getLangOpts().Optimize) {
1375     Fn->removeFnAttr(llvm::Attribute::NoInline);
1376     Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
1377     Fn->addFnAttr(llvm::Attribute::AlwaysInline);
1378   }
1379   CodeGenFunction CGF(CGM);
1380   // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions.
1381   // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions.
1382   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(),
1383                     Out->getLocation());
1384   CodeGenFunction::OMPPrivateScope Scope(CGF);
1385   Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm);
1386   Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() {
1387     return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>())
1388         .getAddress(CGF);
1389   });
1390   Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm);
1391   Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() {
1392     return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>())
1393         .getAddress(CGF);
1394   });
1395   (void)Scope.Privatize();
1396   if (!IsCombiner && Out->hasInit() &&
1397       !CGF.isTrivialInitializer(Out->getInit())) {
1398     CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out),
1399                          Out->getType().getQualifiers(),
1400                          /*IsInitializer=*/true);
1401   }
1402   if (CombinerInitializer)
1403     CGF.EmitIgnoredExpr(CombinerInitializer);
1404   Scope.ForceCleanup();
1405   CGF.FinishFunction();
1406   return Fn;
1407 }
1408 
1409 void CGOpenMPRuntime::emitUserDefinedReduction(
1410     CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) {
1411   if (UDRMap.count(D) > 0)
1412     return;
1413   llvm::Function *Combiner = emitCombinerOrInitializer(
1414       CGM, D->getType(), D->getCombiner(),
1415       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()),
1416       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()),
1417       /*IsCombiner=*/true);
1418   llvm::Function *Initializer = nullptr;
1419   if (const Expr *Init = D->getInitializer()) {
1420     Initializer = emitCombinerOrInitializer(
1421         CGM, D->getType(),
1422         D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init
1423                                                                      : nullptr,
1424         cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()),
1425         cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()),
1426         /*IsCombiner=*/false);
1427   }
1428   UDRMap.try_emplace(D, Combiner, Initializer);
1429   if (CGF) {
1430     auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn);
1431     Decls.second.push_back(D);
1432   }
1433 }
1434 
1435 std::pair<llvm::Function *, llvm::Function *>
1436 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) {
1437   auto I = UDRMap.find(D);
1438   if (I != UDRMap.end())
1439     return I->second;
1440   emitUserDefinedReduction(/*CGF=*/nullptr, D);
1441   return UDRMap.lookup(D);
1442 }
1443 
1444 namespace {
1445 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR
1446 // Builder if one is present.
1447 struct PushAndPopStackRAII {
1448   PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF,
1449                       bool HasCancel)
1450       : OMPBuilder(OMPBuilder) {
1451     if (!OMPBuilder)
1452       return;
1453 
1454     // The following callback is the crucial part of clangs cleanup process.
1455     //
1456     // NOTE:
1457     // Once the OpenMPIRBuilder is used to create parallel regions (and
1458     // similar), the cancellation destination (Dest below) is determined via
1459     // IP. That means if we have variables to finalize we split the block at IP,
1460     // use the new block (=BB) as destination to build a JumpDest (via
1461     // getJumpDestInCurrentScope(BB)) which then is fed to
1462     // EmitBranchThroughCleanup. Furthermore, there will not be the need
1463     // to push & pop an FinalizationInfo object.
1464     // The FiniCB will still be needed but at the point where the
1465     // OpenMPIRBuilder is asked to construct a parallel (or similar) construct.
1466     auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) {
1467       assert(IP.getBlock()->end() == IP.getPoint() &&
1468              "Clang CG should cause non-terminated block!");
1469       CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1470       CGF.Builder.restoreIP(IP);
1471       CodeGenFunction::JumpDest Dest =
1472           CGF.getOMPCancelDestination(OMPD_parallel);
1473       CGF.EmitBranchThroughCleanup(Dest);
1474     };
1475 
1476     // TODO: Remove this once we emit parallel regions through the
1477     //       OpenMPIRBuilder as it can do this setup internally.
1478     llvm::OpenMPIRBuilder::FinalizationInfo FI(
1479         {FiniCB, OMPD_parallel, HasCancel});
1480     OMPBuilder->pushFinalizationCB(std::move(FI));
1481   }
1482   ~PushAndPopStackRAII() {
1483     if (OMPBuilder)
1484       OMPBuilder->popFinalizationCB();
1485   }
1486   llvm::OpenMPIRBuilder *OMPBuilder;
1487 };
1488 } // namespace
1489 
1490 static llvm::Function *emitParallelOrTeamsOutlinedFunction(
1491     CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS,
1492     const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1493     const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) {
1494   assert(ThreadIDVar->getType()->isPointerType() &&
1495          "thread id variable must be of type kmp_int32 *");
1496   CodeGenFunction CGF(CGM, true);
1497   bool HasCancel = false;
1498   if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D))
1499     HasCancel = OPD->hasCancel();
1500   else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D))
1501     HasCancel = OPSD->hasCancel();
1502   else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D))
1503     HasCancel = OPFD->hasCancel();
1504   else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D))
1505     HasCancel = OPFD->hasCancel();
1506   else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D))
1507     HasCancel = OPFD->hasCancel();
1508   else if (const auto *OPFD =
1509                dyn_cast<OMPTeamsDistributeParallelForDirective>(&D))
1510     HasCancel = OPFD->hasCancel();
1511   else if (const auto *OPFD =
1512                dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D))
1513     HasCancel = OPFD->hasCancel();
1514 
1515   // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new
1516   //       parallel region to make cancellation barriers work properly.
1517   llvm::OpenMPIRBuilder *OMPBuilder = CGM.getOpenMPIRBuilder();
1518   PushAndPopStackRAII PSR(OMPBuilder, CGF, HasCancel);
1519   CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind,
1520                                     HasCancel, OutlinedHelperName);
1521   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1522   return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc());
1523 }
1524 
1525 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction(
1526     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1527     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1528   const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel);
1529   return emitParallelOrTeamsOutlinedFunction(
1530       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1531 }
1532 
1533 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction(
1534     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1535     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1536   const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams);
1537   return emitParallelOrTeamsOutlinedFunction(
1538       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1539 }
1540 
1541 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction(
1542     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1543     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
1544     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
1545     bool Tied, unsigned &NumberOfParts) {
1546   auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF,
1547                                               PrePostActionTy &) {
1548     llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc());
1549     llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc());
1550     llvm::Value *TaskArgs[] = {
1551         UpLoc, ThreadID,
1552         CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar),
1553                                     TaskTVar->getType()->castAs<PointerType>())
1554             .getPointer(CGF)};
1555     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs);
1556   };
1557   CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar,
1558                                                             UntiedCodeGen);
1559   CodeGen.setAction(Action);
1560   assert(!ThreadIDVar->getType()->isPointerType() &&
1561          "thread id variable must be of type kmp_int32 for tasks");
1562   const OpenMPDirectiveKind Region =
1563       isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop
1564                                                       : OMPD_task;
1565   const CapturedStmt *CS = D.getCapturedStmt(Region);
1566   const auto *TD = dyn_cast<OMPTaskDirective>(&D);
1567   CodeGenFunction CGF(CGM, true);
1568   CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen,
1569                                         InnermostKind,
1570                                         TD ? TD->hasCancel() : false, Action);
1571   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1572   llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS);
1573   if (!Tied)
1574     NumberOfParts = Action.getNumberOfParts();
1575   return Res;
1576 }
1577 
1578 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM,
1579                              const RecordDecl *RD, const CGRecordLayout &RL,
1580                              ArrayRef<llvm::Constant *> Data) {
1581   llvm::StructType *StructTy = RL.getLLVMType();
1582   unsigned PrevIdx = 0;
1583   ConstantInitBuilder CIBuilder(CGM);
1584   auto DI = Data.begin();
1585   for (const FieldDecl *FD : RD->fields()) {
1586     unsigned Idx = RL.getLLVMFieldNo(FD);
1587     // Fill the alignment.
1588     for (unsigned I = PrevIdx; I < Idx; ++I)
1589       Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I)));
1590     PrevIdx = Idx + 1;
1591     Fields.add(*DI);
1592     ++DI;
1593   }
1594 }
1595 
1596 template <class... As>
1597 static llvm::GlobalVariable *
1598 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant,
1599                    ArrayRef<llvm::Constant *> Data, const Twine &Name,
1600                    As &&... Args) {
1601   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1602   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1603   ConstantInitBuilder CIBuilder(CGM);
1604   ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType());
1605   buildStructValue(Fields, CGM, RD, RL, Data);
1606   return Fields.finishAndCreateGlobal(
1607       Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant,
1608       std::forward<As>(Args)...);
1609 }
1610 
1611 template <typename T>
1612 static void
1613 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty,
1614                                          ArrayRef<llvm::Constant *> Data,
1615                                          T &Parent) {
1616   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1617   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1618   ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType());
1619   buildStructValue(Fields, CGM, RD, RL, Data);
1620   Fields.finishAndAddTo(Parent);
1621 }
1622 
1623 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) {
1624   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1625   unsigned Reserved2Flags = getDefaultLocationReserved2Flags();
1626   FlagsTy FlagsKey(Flags, Reserved2Flags);
1627   llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey);
1628   if (!Entry) {
1629     if (!DefaultOpenMPPSource) {
1630       // Initialize default location for psource field of ident_t structure of
1631       // all ident_t objects. Format is ";file;function;line;column;;".
1632       // Taken from
1633       // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp
1634       DefaultOpenMPPSource =
1635           CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer();
1636       DefaultOpenMPPSource =
1637           llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy);
1638     }
1639 
1640     llvm::Constant *Data[] = {
1641         llvm::ConstantInt::getNullValue(CGM.Int32Ty),
1642         llvm::ConstantInt::get(CGM.Int32Ty, Flags),
1643         llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags),
1644         llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource};
1645     llvm::GlobalValue *DefaultOpenMPLocation =
1646         createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "",
1647                            llvm::GlobalValue::PrivateLinkage);
1648     DefaultOpenMPLocation->setUnnamedAddr(
1649         llvm::GlobalValue::UnnamedAddr::Global);
1650 
1651     OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation;
1652   }
1653   return Address(Entry, Align);
1654 }
1655 
1656 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF,
1657                                              bool AtCurrentPoint) {
1658   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1659   assert(!Elem.second.ServiceInsertPt && "Insert point is set already.");
1660 
1661   llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty);
1662   if (AtCurrentPoint) {
1663     Elem.second.ServiceInsertPt = new llvm::BitCastInst(
1664         Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock());
1665   } else {
1666     Elem.second.ServiceInsertPt =
1667         new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt");
1668     Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt);
1669   }
1670 }
1671 
1672 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) {
1673   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1674   if (Elem.second.ServiceInsertPt) {
1675     llvm::Instruction *Ptr = Elem.second.ServiceInsertPt;
1676     Elem.second.ServiceInsertPt = nullptr;
1677     Ptr->eraseFromParent();
1678   }
1679 }
1680 
1681 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
1682                                                  SourceLocation Loc,
1683                                                  unsigned Flags) {
1684   Flags |= OMP_IDENT_KMPC;
1685   // If no debug info is generated - return global default location.
1686   if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo ||
1687       Loc.isInvalid())
1688     return getOrCreateDefaultLocation(Flags).getPointer();
1689 
1690   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1691 
1692   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1693   Address LocValue = Address::invalid();
1694   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1695   if (I != OpenMPLocThreadIDMap.end())
1696     LocValue = Address(I->second.DebugLoc, Align);
1697 
1698   // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if
1699   // GetOpenMPThreadID was called before this routine.
1700   if (!LocValue.isValid()) {
1701     // Generate "ident_t .kmpc_loc.addr;"
1702     Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr");
1703     auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1704     Elem.second.DebugLoc = AI.getPointer();
1705     LocValue = AI;
1706 
1707     if (!Elem.second.ServiceInsertPt)
1708       setLocThreadIdInsertPt(CGF);
1709     CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1710     CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1711     CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags),
1712                              CGF.getTypeSize(IdentQTy));
1713   }
1714 
1715   // char **psource = &.kmpc_loc_<flags>.addr.psource;
1716   LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy);
1717   auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin();
1718   LValue PSource =
1719       CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource));
1720 
1721   llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding());
1722   if (OMPDebugLoc == nullptr) {
1723     SmallString<128> Buffer2;
1724     llvm::raw_svector_ostream OS2(Buffer2);
1725     // Build debug location
1726     PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1727     OS2 << ";" << PLoc.getFilename() << ";";
1728     if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1729       OS2 << FD->getQualifiedNameAsString();
1730     OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
1731     OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str());
1732     OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc;
1733   }
1734   // *psource = ";<File>;<Function>;<Line>;<Column>;;";
1735   CGF.EmitStoreOfScalar(OMPDebugLoc, PSource);
1736 
1737   // Our callers always pass this to a runtime function, so for
1738   // convenience, go ahead and return a naked pointer.
1739   return LocValue.getPointer();
1740 }
1741 
1742 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
1743                                           SourceLocation Loc) {
1744   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1745 
1746   llvm::Value *ThreadID = nullptr;
1747   // Check whether we've already cached a load of the thread id in this
1748   // function.
1749   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1750   if (I != OpenMPLocThreadIDMap.end()) {
1751     ThreadID = I->second.ThreadID;
1752     if (ThreadID != nullptr)
1753       return ThreadID;
1754   }
1755   // If exceptions are enabled, do not use parameter to avoid possible crash.
1756   if (auto *OMPRegionInfo =
1757           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
1758     if (OMPRegionInfo->getThreadIDVariable()) {
1759       // Check if this an outlined function with thread id passed as argument.
1760       LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
1761       llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent();
1762       if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions ||
1763           !CGF.getLangOpts().CXXExceptions ||
1764           CGF.Builder.GetInsertBlock() == TopBlock ||
1765           !isa<llvm::Instruction>(LVal.getPointer(CGF)) ||
1766           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1767               TopBlock ||
1768           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1769               CGF.Builder.GetInsertBlock()) {
1770         ThreadID = CGF.EmitLoadOfScalar(LVal, Loc);
1771         // If value loaded in entry block, cache it and use it everywhere in
1772         // function.
1773         if (CGF.Builder.GetInsertBlock() == TopBlock) {
1774           auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1775           Elem.second.ThreadID = ThreadID;
1776         }
1777         return ThreadID;
1778       }
1779     }
1780   }
1781 
1782   // This is not an outlined function region - need to call __kmpc_int32
1783   // kmpc_global_thread_num(ident_t *loc).
1784   // Generate thread id value and cache this value for use across the
1785   // function.
1786   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1787   if (!Elem.second.ServiceInsertPt)
1788     setLocThreadIdInsertPt(CGF);
1789   CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1790   CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1791   llvm::CallInst *Call = CGF.Builder.CreateCall(
1792       createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
1793       emitUpdateLocation(CGF, Loc));
1794   Call->setCallingConv(CGF.getRuntimeCC());
1795   Elem.second.ThreadID = Call;
1796   return Call;
1797 }
1798 
1799 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
1800   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1801   if (OpenMPLocThreadIDMap.count(CGF.CurFn)) {
1802     clearLocThreadIdInsertPt(CGF);
1803     OpenMPLocThreadIDMap.erase(CGF.CurFn);
1804   }
1805   if (FunctionUDRMap.count(CGF.CurFn) > 0) {
1806     for(const auto *D : FunctionUDRMap[CGF.CurFn])
1807       UDRMap.erase(D);
1808     FunctionUDRMap.erase(CGF.CurFn);
1809   }
1810   auto I = FunctionUDMMap.find(CGF.CurFn);
1811   if (I != FunctionUDMMap.end()) {
1812     for(const auto *D : I->second)
1813       UDMMap.erase(D);
1814     FunctionUDMMap.erase(I);
1815   }
1816   LastprivateConditionalToTypes.erase(CGF.CurFn);
1817 }
1818 
1819 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
1820   return IdentTy->getPointerTo();
1821 }
1822 
1823 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() {
1824   if (!Kmpc_MicroTy) {
1825     // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...)
1826     llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty),
1827                                  llvm::PointerType::getUnqual(CGM.Int32Ty)};
1828     Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true);
1829   }
1830   return llvm::PointerType::getUnqual(Kmpc_MicroTy);
1831 }
1832 
1833 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) {
1834   llvm::FunctionCallee RTLFn = nullptr;
1835   switch (static_cast<OpenMPRTLFunction>(Function)) {
1836   case OMPRTL__kmpc_fork_call: {
1837     // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro
1838     // microtask, ...);
1839     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1840                                 getKmpc_MicroPointerTy()};
1841     auto *FnTy =
1842         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
1843     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call");
1844     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
1845       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
1846         llvm::LLVMContext &Ctx = F->getContext();
1847         llvm::MDBuilder MDB(Ctx);
1848         // Annotate the callback behavior of the __kmpc_fork_call:
1849         //  - The callback callee is argument number 2 (microtask).
1850         //  - The first two arguments of the callback callee are unknown (-1).
1851         //  - All variadic arguments to the __kmpc_fork_call are passed to the
1852         //    callback callee.
1853         F->addMetadata(
1854             llvm::LLVMContext::MD_callback,
1855             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
1856                                         2, {-1, -1},
1857                                         /* VarArgsArePassed */ true)}));
1858       }
1859     }
1860     break;
1861   }
1862   case OMPRTL__kmpc_global_thread_num: {
1863     // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc);
1864     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1865     auto *FnTy =
1866         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1867     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num");
1868     break;
1869   }
1870   case OMPRTL__kmpc_threadprivate_cached: {
1871     // Build void *__kmpc_threadprivate_cached(ident_t *loc,
1872     // kmp_int32 global_tid, void *data, size_t size, void ***cache);
1873     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1874                                 CGM.VoidPtrTy, CGM.SizeTy,
1875                                 CGM.VoidPtrTy->getPointerTo()->getPointerTo()};
1876     auto *FnTy =
1877         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false);
1878     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached");
1879     break;
1880   }
1881   case OMPRTL__kmpc_critical: {
1882     // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
1883     // kmp_critical_name *crit);
1884     llvm::Type *TypeParams[] = {
1885         getIdentTyPointerTy(), CGM.Int32Ty,
1886         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1887     auto *FnTy =
1888         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1889     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical");
1890     break;
1891   }
1892   case OMPRTL__kmpc_critical_with_hint: {
1893     // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid,
1894     // kmp_critical_name *crit, uintptr_t hint);
1895     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1896                                 llvm::PointerType::getUnqual(KmpCriticalNameTy),
1897                                 CGM.IntPtrTy};
1898     auto *FnTy =
1899         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1900     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint");
1901     break;
1902   }
1903   case OMPRTL__kmpc_threadprivate_register: {
1904     // Build void __kmpc_threadprivate_register(ident_t *, void *data,
1905     // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
1906     // typedef void *(*kmpc_ctor)(void *);
1907     auto *KmpcCtorTy =
1908         llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
1909                                 /*isVarArg*/ false)->getPointerTo();
1910     // typedef void *(*kmpc_cctor)(void *, void *);
1911     llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
1912     auto *KmpcCopyCtorTy =
1913         llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs,
1914                                 /*isVarArg*/ false)
1915             ->getPointerTo();
1916     // typedef void (*kmpc_dtor)(void *);
1917     auto *KmpcDtorTy =
1918         llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false)
1919             ->getPointerTo();
1920     llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy,
1921                               KmpcCopyCtorTy, KmpcDtorTy};
1922     auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs,
1923                                         /*isVarArg*/ false);
1924     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register");
1925     break;
1926   }
1927   case OMPRTL__kmpc_end_critical: {
1928     // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
1929     // kmp_critical_name *crit);
1930     llvm::Type *TypeParams[] = {
1931         getIdentTyPointerTy(), CGM.Int32Ty,
1932         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1933     auto *FnTy =
1934         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1935     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical");
1936     break;
1937   }
1938   case OMPRTL__kmpc_cancel_barrier: {
1939     // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
1940     // global_tid);
1941     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1942     auto *FnTy =
1943         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1944     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier");
1945     break;
1946   }
1947   case OMPRTL__kmpc_barrier: {
1948     // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
1949     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1950     auto *FnTy =
1951         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1952     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier");
1953     break;
1954   }
1955   case OMPRTL__kmpc_for_static_fini: {
1956     // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
1957     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1958     auto *FnTy =
1959         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1960     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini");
1961     break;
1962   }
1963   case OMPRTL__kmpc_push_num_threads: {
1964     // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
1965     // kmp_int32 num_threads)
1966     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1967                                 CGM.Int32Ty};
1968     auto *FnTy =
1969         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1970     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads");
1971     break;
1972   }
1973   case OMPRTL__kmpc_serialized_parallel: {
1974     // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
1975     // global_tid);
1976     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1977     auto *FnTy =
1978         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1979     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel");
1980     break;
1981   }
1982   case OMPRTL__kmpc_end_serialized_parallel: {
1983     // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
1984     // global_tid);
1985     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1986     auto *FnTy =
1987         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1988     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel");
1989     break;
1990   }
1991   case OMPRTL__kmpc_flush: {
1992     // Build void __kmpc_flush(ident_t *loc);
1993     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1994     auto *FnTy =
1995         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1996     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush");
1997     break;
1998   }
1999   case OMPRTL__kmpc_master: {
2000     // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid);
2001     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2002     auto *FnTy =
2003         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2004     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master");
2005     break;
2006   }
2007   case OMPRTL__kmpc_end_master: {
2008     // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid);
2009     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2010     auto *FnTy =
2011         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2012     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master");
2013     break;
2014   }
2015   case OMPRTL__kmpc_omp_taskyield: {
2016     // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
2017     // int end_part);
2018     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2019     auto *FnTy =
2020         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2021     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield");
2022     break;
2023   }
2024   case OMPRTL__kmpc_single: {
2025     // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid);
2026     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2027     auto *FnTy =
2028         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2029     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single");
2030     break;
2031   }
2032   case OMPRTL__kmpc_end_single: {
2033     // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid);
2034     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2035     auto *FnTy =
2036         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2037     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single");
2038     break;
2039   }
2040   case OMPRTL__kmpc_omp_task_alloc: {
2041     // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
2042     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
2043     // kmp_routine_entry_t *task_entry);
2044     assert(KmpRoutineEntryPtrTy != nullptr &&
2045            "Type kmp_routine_entry_t must be created.");
2046     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2047                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy};
2048     // Return void * and then cast to particular kmp_task_t type.
2049     auto *FnTy =
2050         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2051     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc");
2052     break;
2053   }
2054   case OMPRTL__kmpc_omp_target_task_alloc: {
2055     // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid,
2056     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
2057     // kmp_routine_entry_t *task_entry, kmp_int64 device_id);
2058     assert(KmpRoutineEntryPtrTy != nullptr &&
2059            "Type kmp_routine_entry_t must be created.");
2060     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2061                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy,
2062                                 CGM.Int64Ty};
2063     // Return void * and then cast to particular kmp_task_t type.
2064     auto *FnTy =
2065         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2066     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc");
2067     break;
2068   }
2069   case OMPRTL__kmpc_omp_task: {
2070     // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2071     // *new_task);
2072     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2073                                 CGM.VoidPtrTy};
2074     auto *FnTy =
2075         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2076     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task");
2077     break;
2078   }
2079   case OMPRTL__kmpc_copyprivate: {
2080     // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
2081     // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
2082     // kmp_int32 didit);
2083     llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2084     auto *CpyFnTy =
2085         llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false);
2086     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy,
2087                                 CGM.VoidPtrTy, CpyFnTy->getPointerTo(),
2088                                 CGM.Int32Ty};
2089     auto *FnTy =
2090         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2091     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate");
2092     break;
2093   }
2094   case OMPRTL__kmpc_reduce: {
2095     // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
2096     // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
2097     // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
2098     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2099     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2100                                                /*isVarArg=*/false);
2101     llvm::Type *TypeParams[] = {
2102         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2103         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2104         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2105     auto *FnTy =
2106         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2107     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce");
2108     break;
2109   }
2110   case OMPRTL__kmpc_reduce_nowait: {
2111     // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
2112     // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
2113     // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
2114     // *lck);
2115     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2116     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2117                                                /*isVarArg=*/false);
2118     llvm::Type *TypeParams[] = {
2119         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2120         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2121         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2122     auto *FnTy =
2123         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2124     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait");
2125     break;
2126   }
2127   case OMPRTL__kmpc_end_reduce: {
2128     // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
2129     // kmp_critical_name *lck);
2130     llvm::Type *TypeParams[] = {
2131         getIdentTyPointerTy(), CGM.Int32Ty,
2132         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2133     auto *FnTy =
2134         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2135     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce");
2136     break;
2137   }
2138   case OMPRTL__kmpc_end_reduce_nowait: {
2139     // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
2140     // kmp_critical_name *lck);
2141     llvm::Type *TypeParams[] = {
2142         getIdentTyPointerTy(), CGM.Int32Ty,
2143         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2144     auto *FnTy =
2145         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2146     RTLFn =
2147         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait");
2148     break;
2149   }
2150   case OMPRTL__kmpc_omp_task_begin_if0: {
2151     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2152     // *new_task);
2153     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2154                                 CGM.VoidPtrTy};
2155     auto *FnTy =
2156         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2157     RTLFn =
2158         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0");
2159     break;
2160   }
2161   case OMPRTL__kmpc_omp_task_complete_if0: {
2162     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2163     // *new_task);
2164     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2165                                 CGM.VoidPtrTy};
2166     auto *FnTy =
2167         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2168     RTLFn = CGM.CreateRuntimeFunction(FnTy,
2169                                       /*Name=*/"__kmpc_omp_task_complete_if0");
2170     break;
2171   }
2172   case OMPRTL__kmpc_ordered: {
2173     // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
2174     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2175     auto *FnTy =
2176         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2177     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered");
2178     break;
2179   }
2180   case OMPRTL__kmpc_end_ordered: {
2181     // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
2182     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2183     auto *FnTy =
2184         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2185     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered");
2186     break;
2187   }
2188   case OMPRTL__kmpc_omp_taskwait: {
2189     // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid);
2190     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2191     auto *FnTy =
2192         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2193     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait");
2194     break;
2195   }
2196   case OMPRTL__kmpc_taskgroup: {
2197     // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
2198     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2199     auto *FnTy =
2200         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2201     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup");
2202     break;
2203   }
2204   case OMPRTL__kmpc_end_taskgroup: {
2205     // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
2206     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2207     auto *FnTy =
2208         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2209     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup");
2210     break;
2211   }
2212   case OMPRTL__kmpc_push_proc_bind: {
2213     // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
2214     // int proc_bind)
2215     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2216     auto *FnTy =
2217         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2218     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind");
2219     break;
2220   }
2221   case OMPRTL__kmpc_omp_task_with_deps: {
2222     // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
2223     // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
2224     // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
2225     llvm::Type *TypeParams[] = {
2226         getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty,
2227         CGM.VoidPtrTy,         CGM.Int32Ty, CGM.VoidPtrTy};
2228     auto *FnTy =
2229         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2230     RTLFn =
2231         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps");
2232     break;
2233   }
2234   case OMPRTL__kmpc_omp_wait_deps: {
2235     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
2236     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias,
2237     // kmp_depend_info_t *noalias_dep_list);
2238     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2239                                 CGM.Int32Ty,           CGM.VoidPtrTy,
2240                                 CGM.Int32Ty,           CGM.VoidPtrTy};
2241     auto *FnTy =
2242         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2243     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps");
2244     break;
2245   }
2246   case OMPRTL__kmpc_cancellationpoint: {
2247     // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
2248     // global_tid, kmp_int32 cncl_kind)
2249     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2250     auto *FnTy =
2251         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2252     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint");
2253     break;
2254   }
2255   case OMPRTL__kmpc_cancel: {
2256     // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
2257     // kmp_int32 cncl_kind)
2258     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2259     auto *FnTy =
2260         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2261     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel");
2262     break;
2263   }
2264   case OMPRTL__kmpc_push_num_teams: {
2265     // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid,
2266     // kmp_int32 num_teams, kmp_int32 num_threads)
2267     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2268         CGM.Int32Ty};
2269     auto *FnTy =
2270         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2271     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams");
2272     break;
2273   }
2274   case OMPRTL__kmpc_fork_teams: {
2275     // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
2276     // microtask, ...);
2277     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2278                                 getKmpc_MicroPointerTy()};
2279     auto *FnTy =
2280         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
2281     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams");
2282     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
2283       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
2284         llvm::LLVMContext &Ctx = F->getContext();
2285         llvm::MDBuilder MDB(Ctx);
2286         // Annotate the callback behavior of the __kmpc_fork_teams:
2287         //  - The callback callee is argument number 2 (microtask).
2288         //  - The first two arguments of the callback callee are unknown (-1).
2289         //  - All variadic arguments to the __kmpc_fork_teams are passed to the
2290         //    callback callee.
2291         F->addMetadata(
2292             llvm::LLVMContext::MD_callback,
2293             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
2294                                         2, {-1, -1},
2295                                         /* VarArgsArePassed */ true)}));
2296       }
2297     }
2298     break;
2299   }
2300   case OMPRTL__kmpc_taskloop: {
2301     // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
2302     // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
2303     // sched, kmp_uint64 grainsize, void *task_dup);
2304     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2305                                 CGM.IntTy,
2306                                 CGM.VoidPtrTy,
2307                                 CGM.IntTy,
2308                                 CGM.Int64Ty->getPointerTo(),
2309                                 CGM.Int64Ty->getPointerTo(),
2310                                 CGM.Int64Ty,
2311                                 CGM.IntTy,
2312                                 CGM.IntTy,
2313                                 CGM.Int64Ty,
2314                                 CGM.VoidPtrTy};
2315     auto *FnTy =
2316         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2317     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop");
2318     break;
2319   }
2320   case OMPRTL__kmpc_doacross_init: {
2321     // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
2322     // num_dims, struct kmp_dim *dims);
2323     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2324                                 CGM.Int32Ty,
2325                                 CGM.Int32Ty,
2326                                 CGM.VoidPtrTy};
2327     auto *FnTy =
2328         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2329     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init");
2330     break;
2331   }
2332   case OMPRTL__kmpc_doacross_fini: {
2333     // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
2334     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2335     auto *FnTy =
2336         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2337     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini");
2338     break;
2339   }
2340   case OMPRTL__kmpc_doacross_post: {
2341     // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
2342     // *vec);
2343     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2344                                 CGM.Int64Ty->getPointerTo()};
2345     auto *FnTy =
2346         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2347     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post");
2348     break;
2349   }
2350   case OMPRTL__kmpc_doacross_wait: {
2351     // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
2352     // *vec);
2353     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2354                                 CGM.Int64Ty->getPointerTo()};
2355     auto *FnTy =
2356         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2357     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait");
2358     break;
2359   }
2360   case OMPRTL__kmpc_task_reduction_init: {
2361     // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void
2362     // *data);
2363     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy};
2364     auto *FnTy =
2365         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2366     RTLFn =
2367         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init");
2368     break;
2369   }
2370   case OMPRTL__kmpc_task_reduction_get_th_data: {
2371     // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
2372     // *d);
2373     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2374     auto *FnTy =
2375         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2376     RTLFn = CGM.CreateRuntimeFunction(
2377         FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data");
2378     break;
2379   }
2380   case OMPRTL__kmpc_alloc: {
2381     // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t
2382     // al); omp_allocator_handle_t type is void *.
2383     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy};
2384     auto *FnTy =
2385         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2386     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc");
2387     break;
2388   }
2389   case OMPRTL__kmpc_free: {
2390     // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t
2391     // al); omp_allocator_handle_t type is void *.
2392     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2393     auto *FnTy =
2394         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2395     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free");
2396     break;
2397   }
2398   case OMPRTL__kmpc_push_target_tripcount: {
2399     // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
2400     // size);
2401     llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty};
2402     llvm::FunctionType *FnTy =
2403         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2404     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount");
2405     break;
2406   }
2407   case OMPRTL__tgt_target: {
2408     // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
2409     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2410     // *arg_types);
2411     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2412                                 CGM.VoidPtrTy,
2413                                 CGM.Int32Ty,
2414                                 CGM.VoidPtrPtrTy,
2415                                 CGM.VoidPtrPtrTy,
2416                                 CGM.Int64Ty->getPointerTo(),
2417                                 CGM.Int64Ty->getPointerTo()};
2418     auto *FnTy =
2419         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2420     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target");
2421     break;
2422   }
2423   case OMPRTL__tgt_target_nowait: {
2424     // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
2425     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2426     // int64_t *arg_types);
2427     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2428                                 CGM.VoidPtrTy,
2429                                 CGM.Int32Ty,
2430                                 CGM.VoidPtrPtrTy,
2431                                 CGM.VoidPtrPtrTy,
2432                                 CGM.Int64Ty->getPointerTo(),
2433                                 CGM.Int64Ty->getPointerTo()};
2434     auto *FnTy =
2435         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2436     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait");
2437     break;
2438   }
2439   case OMPRTL__tgt_target_teams: {
2440     // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
2441     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2442     // int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2443     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2444                                 CGM.VoidPtrTy,
2445                                 CGM.Int32Ty,
2446                                 CGM.VoidPtrPtrTy,
2447                                 CGM.VoidPtrPtrTy,
2448                                 CGM.Int64Ty->getPointerTo(),
2449                                 CGM.Int64Ty->getPointerTo(),
2450                                 CGM.Int32Ty,
2451                                 CGM.Int32Ty};
2452     auto *FnTy =
2453         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2454     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams");
2455     break;
2456   }
2457   case OMPRTL__tgt_target_teams_nowait: {
2458     // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void
2459     // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
2460     // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2461     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2462                                 CGM.VoidPtrTy,
2463                                 CGM.Int32Ty,
2464                                 CGM.VoidPtrPtrTy,
2465                                 CGM.VoidPtrPtrTy,
2466                                 CGM.Int64Ty->getPointerTo(),
2467                                 CGM.Int64Ty->getPointerTo(),
2468                                 CGM.Int32Ty,
2469                                 CGM.Int32Ty};
2470     auto *FnTy =
2471         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2472     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait");
2473     break;
2474   }
2475   case OMPRTL__tgt_register_requires: {
2476     // Build void __tgt_register_requires(int64_t flags);
2477     llvm::Type *TypeParams[] = {CGM.Int64Ty};
2478     auto *FnTy =
2479         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2480     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires");
2481     break;
2482   }
2483   case OMPRTL__tgt_target_data_begin: {
2484     // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
2485     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2486     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2487                                 CGM.Int32Ty,
2488                                 CGM.VoidPtrPtrTy,
2489                                 CGM.VoidPtrPtrTy,
2490                                 CGM.Int64Ty->getPointerTo(),
2491                                 CGM.Int64Ty->getPointerTo()};
2492     auto *FnTy =
2493         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2494     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin");
2495     break;
2496   }
2497   case OMPRTL__tgt_target_data_begin_nowait: {
2498     // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
2499     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2500     // *arg_types);
2501     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2502                                 CGM.Int32Ty,
2503                                 CGM.VoidPtrPtrTy,
2504                                 CGM.VoidPtrPtrTy,
2505                                 CGM.Int64Ty->getPointerTo(),
2506                                 CGM.Int64Ty->getPointerTo()};
2507     auto *FnTy =
2508         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2509     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait");
2510     break;
2511   }
2512   case OMPRTL__tgt_target_data_end: {
2513     // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
2514     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2515     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2516                                 CGM.Int32Ty,
2517                                 CGM.VoidPtrPtrTy,
2518                                 CGM.VoidPtrPtrTy,
2519                                 CGM.Int64Ty->getPointerTo(),
2520                                 CGM.Int64Ty->getPointerTo()};
2521     auto *FnTy =
2522         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2523     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end");
2524     break;
2525   }
2526   case OMPRTL__tgt_target_data_end_nowait: {
2527     // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t
2528     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2529     // *arg_types);
2530     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2531                                 CGM.Int32Ty,
2532                                 CGM.VoidPtrPtrTy,
2533                                 CGM.VoidPtrPtrTy,
2534                                 CGM.Int64Ty->getPointerTo(),
2535                                 CGM.Int64Ty->getPointerTo()};
2536     auto *FnTy =
2537         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2538     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait");
2539     break;
2540   }
2541   case OMPRTL__tgt_target_data_update: {
2542     // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
2543     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2544     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2545                                 CGM.Int32Ty,
2546                                 CGM.VoidPtrPtrTy,
2547                                 CGM.VoidPtrPtrTy,
2548                                 CGM.Int64Ty->getPointerTo(),
2549                                 CGM.Int64Ty->getPointerTo()};
2550     auto *FnTy =
2551         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2552     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update");
2553     break;
2554   }
2555   case OMPRTL__tgt_target_data_update_nowait: {
2556     // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t
2557     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2558     // *arg_types);
2559     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2560                                 CGM.Int32Ty,
2561                                 CGM.VoidPtrPtrTy,
2562                                 CGM.VoidPtrPtrTy,
2563                                 CGM.Int64Ty->getPointerTo(),
2564                                 CGM.Int64Ty->getPointerTo()};
2565     auto *FnTy =
2566         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2567     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait");
2568     break;
2569   }
2570   case OMPRTL__tgt_mapper_num_components: {
2571     // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
2572     llvm::Type *TypeParams[] = {CGM.VoidPtrTy};
2573     auto *FnTy =
2574         llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false);
2575     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components");
2576     break;
2577   }
2578   case OMPRTL__tgt_push_mapper_component: {
2579     // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void
2580     // *base, void *begin, int64_t size, int64_t type);
2581     llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy,
2582                                 CGM.Int64Ty, CGM.Int64Ty};
2583     auto *FnTy =
2584         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2585     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component");
2586     break;
2587   }
2588   }
2589   assert(RTLFn && "Unable to find OpenMP runtime function");
2590   return RTLFn;
2591 }
2592 
2593 llvm::FunctionCallee
2594 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) {
2595   assert((IVSize == 32 || IVSize == 64) &&
2596          "IV size is not compatible with the omp runtime");
2597   StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4"
2598                                             : "__kmpc_for_static_init_4u")
2599                                 : (IVSigned ? "__kmpc_for_static_init_8"
2600                                             : "__kmpc_for_static_init_8u");
2601   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2602   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2603   llvm::Type *TypeParams[] = {
2604     getIdentTyPointerTy(),                     // loc
2605     CGM.Int32Ty,                               // tid
2606     CGM.Int32Ty,                               // schedtype
2607     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2608     PtrTy,                                     // p_lower
2609     PtrTy,                                     // p_upper
2610     PtrTy,                                     // p_stride
2611     ITy,                                       // incr
2612     ITy                                        // chunk
2613   };
2614   auto *FnTy =
2615       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2616   return CGM.CreateRuntimeFunction(FnTy, Name);
2617 }
2618 
2619 llvm::FunctionCallee
2620 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) {
2621   assert((IVSize == 32 || IVSize == 64) &&
2622          "IV size is not compatible with the omp runtime");
2623   StringRef Name =
2624       IVSize == 32
2625           ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u")
2626           : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u");
2627   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2628   llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc
2629                                CGM.Int32Ty,           // tid
2630                                CGM.Int32Ty,           // schedtype
2631                                ITy,                   // lower
2632                                ITy,                   // upper
2633                                ITy,                   // stride
2634                                ITy                    // chunk
2635   };
2636   auto *FnTy =
2637       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2638   return CGM.CreateRuntimeFunction(FnTy, Name);
2639 }
2640 
2641 llvm::FunctionCallee
2642 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) {
2643   assert((IVSize == 32 || IVSize == 64) &&
2644          "IV size is not compatible with the omp runtime");
2645   StringRef Name =
2646       IVSize == 32
2647           ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u")
2648           : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u");
2649   llvm::Type *TypeParams[] = {
2650       getIdentTyPointerTy(), // loc
2651       CGM.Int32Ty,           // tid
2652   };
2653   auto *FnTy =
2654       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2655   return CGM.CreateRuntimeFunction(FnTy, Name);
2656 }
2657 
2658 llvm::FunctionCallee
2659 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) {
2660   assert((IVSize == 32 || IVSize == 64) &&
2661          "IV size is not compatible with the omp runtime");
2662   StringRef Name =
2663       IVSize == 32
2664           ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u")
2665           : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u");
2666   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2667   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2668   llvm::Type *TypeParams[] = {
2669     getIdentTyPointerTy(),                     // loc
2670     CGM.Int32Ty,                               // tid
2671     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2672     PtrTy,                                     // p_lower
2673     PtrTy,                                     // p_upper
2674     PtrTy                                      // p_stride
2675   };
2676   auto *FnTy =
2677       llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2678   return CGM.CreateRuntimeFunction(FnTy, Name);
2679 }
2680 
2681 /// Obtain information that uniquely identifies a target entry. This
2682 /// consists of the file and device IDs as well as line number associated with
2683 /// the relevant entry source location.
2684 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc,
2685                                      unsigned &DeviceID, unsigned &FileID,
2686                                      unsigned &LineNum) {
2687   SourceManager &SM = C.getSourceManager();
2688 
2689   // The loc should be always valid and have a file ID (the user cannot use
2690   // #pragma directives in macros)
2691 
2692   assert(Loc.isValid() && "Source location is expected to be always valid.");
2693 
2694   PresumedLoc PLoc = SM.getPresumedLoc(Loc);
2695   assert(PLoc.isValid() && "Source location is expected to be always valid.");
2696 
2697   llvm::sys::fs::UniqueID ID;
2698   if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID))
2699     SM.getDiagnostics().Report(diag::err_cannot_open_file)
2700         << PLoc.getFilename() << EC.message();
2701 
2702   DeviceID = ID.getDevice();
2703   FileID = ID.getFile();
2704   LineNum = PLoc.getLine();
2705 }
2706 
2707 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) {
2708   if (CGM.getLangOpts().OpenMPSimd)
2709     return Address::invalid();
2710   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2711       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2712   if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link ||
2713               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2714                HasRequiresUnifiedSharedMemory))) {
2715     SmallString<64> PtrName;
2716     {
2717       llvm::raw_svector_ostream OS(PtrName);
2718       OS << CGM.getMangledName(GlobalDecl(VD));
2719       if (!VD->isExternallyVisible()) {
2720         unsigned DeviceID, FileID, Line;
2721         getTargetEntryUniqueInfo(CGM.getContext(),
2722                                  VD->getCanonicalDecl()->getBeginLoc(),
2723                                  DeviceID, FileID, Line);
2724         OS << llvm::format("_%x", FileID);
2725       }
2726       OS << "_decl_tgt_ref_ptr";
2727     }
2728     llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName);
2729     if (!Ptr) {
2730       QualType PtrTy = CGM.getContext().getPointerType(VD->getType());
2731       Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy),
2732                                         PtrName);
2733 
2734       auto *GV = cast<llvm::GlobalVariable>(Ptr);
2735       GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
2736 
2737       if (!CGM.getLangOpts().OpenMPIsDevice)
2738         GV->setInitializer(CGM.GetAddrOfGlobal(VD));
2739       registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr));
2740     }
2741     return Address(Ptr, CGM.getContext().getDeclAlign(VD));
2742   }
2743   return Address::invalid();
2744 }
2745 
2746 llvm::Constant *
2747 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
2748   assert(!CGM.getLangOpts().OpenMPUseTLS ||
2749          !CGM.getContext().getTargetInfo().isTLSSupported());
2750   // Lookup the entry, lazily creating it if necessary.
2751   std::string Suffix = getName({"cache", ""});
2752   return getOrCreateInternalVariable(
2753       CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix));
2754 }
2755 
2756 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
2757                                                 const VarDecl *VD,
2758                                                 Address VDAddr,
2759                                                 SourceLocation Loc) {
2760   if (CGM.getLangOpts().OpenMPUseTLS &&
2761       CGM.getContext().getTargetInfo().isTLSSupported())
2762     return VDAddr;
2763 
2764   llvm::Type *VarTy = VDAddr.getElementType();
2765   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2766                          CGF.Builder.CreatePointerCast(VDAddr.getPointer(),
2767                                                        CGM.Int8PtrTy),
2768                          CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
2769                          getOrCreateThreadPrivateCache(VD)};
2770   return Address(CGF.EmitRuntimeCall(
2771       createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
2772                  VDAddr.getAlignment());
2773 }
2774 
2775 void CGOpenMPRuntime::emitThreadPrivateVarInit(
2776     CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor,
2777     llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
2778   // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
2779   // library.
2780   llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc);
2781   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
2782                       OMPLoc);
2783   // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
2784   // to register constructor/destructor for variable.
2785   llvm::Value *Args[] = {
2786       OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy),
2787       Ctor, CopyCtor, Dtor};
2788   CGF.EmitRuntimeCall(
2789       createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args);
2790 }
2791 
2792 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
2793     const VarDecl *VD, Address VDAddr, SourceLocation Loc,
2794     bool PerformInit, CodeGenFunction *CGF) {
2795   if (CGM.getLangOpts().OpenMPUseTLS &&
2796       CGM.getContext().getTargetInfo().isTLSSupported())
2797     return nullptr;
2798 
2799   VD = VD->getDefinition(CGM.getContext());
2800   if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) {
2801     QualType ASTTy = VD->getType();
2802 
2803     llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
2804     const Expr *Init = VD->getAnyInitializer();
2805     if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2806       // Generate function that re-emits the declaration's initializer into the
2807       // threadprivate copy of the variable VD
2808       CodeGenFunction CtorCGF(CGM);
2809       FunctionArgList Args;
2810       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2811                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2812                             ImplicitParamDecl::Other);
2813       Args.push_back(&Dst);
2814 
2815       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2816           CGM.getContext().VoidPtrTy, Args);
2817       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2818       std::string Name = getName({"__kmpc_global_ctor_", ""});
2819       llvm::Function *Fn =
2820           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2821       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
2822                             Args, Loc, Loc);
2823       llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar(
2824           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2825           CGM.getContext().VoidPtrTy, Dst.getLocation());
2826       Address Arg = Address(ArgVal, VDAddr.getAlignment());
2827       Arg = CtorCGF.Builder.CreateElementBitCast(
2828           Arg, CtorCGF.ConvertTypeForMem(ASTTy));
2829       CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
2830                                /*IsInitializer=*/true);
2831       ArgVal = CtorCGF.EmitLoadOfScalar(
2832           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2833           CGM.getContext().VoidPtrTy, Dst.getLocation());
2834       CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
2835       CtorCGF.FinishFunction();
2836       Ctor = Fn;
2837     }
2838     if (VD->getType().isDestructedType() != QualType::DK_none) {
2839       // Generate function that emits destructor call for the threadprivate copy
2840       // of the variable VD
2841       CodeGenFunction DtorCGF(CGM);
2842       FunctionArgList Args;
2843       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2844                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2845                             ImplicitParamDecl::Other);
2846       Args.push_back(&Dst);
2847 
2848       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2849           CGM.getContext().VoidTy, Args);
2850       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2851       std::string Name = getName({"__kmpc_global_dtor_", ""});
2852       llvm::Function *Fn =
2853           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2854       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2855       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
2856                             Loc, Loc);
2857       // Create a scope with an artificial location for the body of this function.
2858       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
2859       llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar(
2860           DtorCGF.GetAddrOfLocalVar(&Dst),
2861           /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation());
2862       DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy,
2863                           DtorCGF.getDestroyer(ASTTy.isDestructedType()),
2864                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
2865       DtorCGF.FinishFunction();
2866       Dtor = Fn;
2867     }
2868     // Do not emit init function if it is not required.
2869     if (!Ctor && !Dtor)
2870       return nullptr;
2871 
2872     llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2873     auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs,
2874                                                /*isVarArg=*/false)
2875                            ->getPointerTo();
2876     // Copying constructor for the threadprivate variable.
2877     // Must be NULL - reserved by runtime, but currently it requires that this
2878     // parameter is always NULL. Otherwise it fires assertion.
2879     CopyCtor = llvm::Constant::getNullValue(CopyCtorTy);
2880     if (Ctor == nullptr) {
2881       auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
2882                                              /*isVarArg=*/false)
2883                          ->getPointerTo();
2884       Ctor = llvm::Constant::getNullValue(CtorTy);
2885     }
2886     if (Dtor == nullptr) {
2887       auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy,
2888                                              /*isVarArg=*/false)
2889                          ->getPointerTo();
2890       Dtor = llvm::Constant::getNullValue(DtorTy);
2891     }
2892     if (!CGF) {
2893       auto *InitFunctionTy =
2894           llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
2895       std::string Name = getName({"__omp_threadprivate_init_", ""});
2896       llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction(
2897           InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction());
2898       CodeGenFunction InitCGF(CGM);
2899       FunctionArgList ArgList;
2900       InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
2901                             CGM.getTypes().arrangeNullaryFunction(), ArgList,
2902                             Loc, Loc);
2903       emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2904       InitCGF.FinishFunction();
2905       return InitFunction;
2906     }
2907     emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2908   }
2909   return nullptr;
2910 }
2911 
2912 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD,
2913                                                      llvm::GlobalVariable *Addr,
2914                                                      bool PerformInit) {
2915   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
2916       !CGM.getLangOpts().OpenMPIsDevice)
2917     return false;
2918   Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2919       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2920   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
2921       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2922        HasRequiresUnifiedSharedMemory))
2923     return CGM.getLangOpts().OpenMPIsDevice;
2924   VD = VD->getDefinition(CGM.getContext());
2925   if (VD && !DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second)
2926     return CGM.getLangOpts().OpenMPIsDevice;
2927 
2928   QualType ASTTy = VD->getType();
2929 
2930   SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc();
2931   // Produce the unique prefix to identify the new target regions. We use
2932   // the source location of the variable declaration which we know to not
2933   // conflict with any target region.
2934   unsigned DeviceID;
2935   unsigned FileID;
2936   unsigned Line;
2937   getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line);
2938   SmallString<128> Buffer, Out;
2939   {
2940     llvm::raw_svector_ostream OS(Buffer);
2941     OS << "__omp_offloading_" << llvm::format("_%x", DeviceID)
2942        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
2943   }
2944 
2945   const Expr *Init = VD->getAnyInitializer();
2946   if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2947     llvm::Constant *Ctor;
2948     llvm::Constant *ID;
2949     if (CGM.getLangOpts().OpenMPIsDevice) {
2950       // Generate function that re-emits the declaration's initializer into
2951       // the threadprivate copy of the variable VD
2952       CodeGenFunction CtorCGF(CGM);
2953 
2954       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2955       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2956       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2957           FTy, Twine(Buffer, "_ctor"), FI, Loc);
2958       auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF);
2959       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2960                             FunctionArgList(), Loc, Loc);
2961       auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF);
2962       CtorCGF.EmitAnyExprToMem(Init,
2963                                Address(Addr, CGM.getContext().getDeclAlign(VD)),
2964                                Init->getType().getQualifiers(),
2965                                /*IsInitializer=*/true);
2966       CtorCGF.FinishFunction();
2967       Ctor = Fn;
2968       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
2969       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor));
2970     } else {
2971       Ctor = new llvm::GlobalVariable(
2972           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
2973           llvm::GlobalValue::PrivateLinkage,
2974           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor"));
2975       ID = Ctor;
2976     }
2977 
2978     // Register the information for the entry associated with the constructor.
2979     Out.clear();
2980     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2981         DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor,
2982         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor);
2983   }
2984   if (VD->getType().isDestructedType() != QualType::DK_none) {
2985     llvm::Constant *Dtor;
2986     llvm::Constant *ID;
2987     if (CGM.getLangOpts().OpenMPIsDevice) {
2988       // Generate function that emits destructor call for the threadprivate
2989       // copy of the variable VD
2990       CodeGenFunction DtorCGF(CGM);
2991 
2992       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2993       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2994       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2995           FTy, Twine(Buffer, "_dtor"), FI, Loc);
2996       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2997       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2998                             FunctionArgList(), Loc, Loc);
2999       // Create a scope with an artificial location for the body of this
3000       // function.
3001       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
3002       DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)),
3003                           ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()),
3004                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
3005       DtorCGF.FinishFunction();
3006       Dtor = Fn;
3007       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
3008       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor));
3009     } else {
3010       Dtor = new llvm::GlobalVariable(
3011           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
3012           llvm::GlobalValue::PrivateLinkage,
3013           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor"));
3014       ID = Dtor;
3015     }
3016     // Register the information for the entry associated with the destructor.
3017     Out.clear();
3018     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
3019         DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor,
3020         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor);
3021   }
3022   return CGM.getLangOpts().OpenMPIsDevice;
3023 }
3024 
3025 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF,
3026                                                           QualType VarType,
3027                                                           StringRef Name) {
3028   std::string Suffix = getName({"artificial", ""});
3029   llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType);
3030   llvm::Value *GAddr =
3031       getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix));
3032   if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS &&
3033       CGM.getTarget().isTLSSupported()) {
3034     cast<llvm::GlobalVariable>(GAddr)->setThreadLocal(/*Val=*/true);
3035     return Address(GAddr, CGM.getContext().getTypeAlignInChars(VarType));
3036   }
3037   std::string CacheSuffix = getName({"cache", ""});
3038   llvm::Value *Args[] = {
3039       emitUpdateLocation(CGF, SourceLocation()),
3040       getThreadID(CGF, SourceLocation()),
3041       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy),
3042       CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy,
3043                                 /*isSigned=*/false),
3044       getOrCreateInternalVariable(
3045           CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))};
3046   return Address(
3047       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3048           CGF.EmitRuntimeCall(
3049               createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
3050           VarLVType->getPointerTo(/*AddrSpace=*/0)),
3051       CGM.getContext().getTypeAlignInChars(VarType));
3052 }
3053 
3054 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond,
3055                                    const RegionCodeGenTy &ThenGen,
3056                                    const RegionCodeGenTy &ElseGen) {
3057   CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
3058 
3059   // If the condition constant folds and can be elided, try to avoid emitting
3060   // the condition and the dead arm of the if/else.
3061   bool CondConstant;
3062   if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
3063     if (CondConstant)
3064       ThenGen(CGF);
3065     else
3066       ElseGen(CGF);
3067     return;
3068   }
3069 
3070   // Otherwise, the condition did not fold, or we couldn't elide it.  Just
3071   // emit the conditional branch.
3072   llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then");
3073   llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else");
3074   llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end");
3075   CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0);
3076 
3077   // Emit the 'then' code.
3078   CGF.EmitBlock(ThenBlock);
3079   ThenGen(CGF);
3080   CGF.EmitBranch(ContBlock);
3081   // Emit the 'else' code if present.
3082   // There is no need to emit line number for unconditional branch.
3083   (void)ApplyDebugLocation::CreateEmpty(CGF);
3084   CGF.EmitBlock(ElseBlock);
3085   ElseGen(CGF);
3086   // There is no need to emit line number for unconditional branch.
3087   (void)ApplyDebugLocation::CreateEmpty(CGF);
3088   CGF.EmitBranch(ContBlock);
3089   // Emit the continuation block for code after the if.
3090   CGF.EmitBlock(ContBlock, /*IsFinished=*/true);
3091 }
3092 
3093 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc,
3094                                        llvm::Function *OutlinedFn,
3095                                        ArrayRef<llvm::Value *> CapturedVars,
3096                                        const Expr *IfCond) {
3097   if (!CGF.HaveInsertPoint())
3098     return;
3099   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
3100   auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF,
3101                                                      PrePostActionTy &) {
3102     // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn);
3103     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3104     llvm::Value *Args[] = {
3105         RTLoc,
3106         CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
3107         CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())};
3108     llvm::SmallVector<llvm::Value *, 16> RealArgs;
3109     RealArgs.append(std::begin(Args), std::end(Args));
3110     RealArgs.append(CapturedVars.begin(), CapturedVars.end());
3111 
3112     llvm::FunctionCallee RTLFn =
3113         RT.createRuntimeFunction(OMPRTL__kmpc_fork_call);
3114     CGF.EmitRuntimeCall(RTLFn, RealArgs);
3115   };
3116   auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF,
3117                                                           PrePostActionTy &) {
3118     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3119     llvm::Value *ThreadID = RT.getThreadID(CGF, Loc);
3120     // Build calls:
3121     // __kmpc_serialized_parallel(&Loc, GTid);
3122     llvm::Value *Args[] = {RTLoc, ThreadID};
3123     CGF.EmitRuntimeCall(
3124         RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args);
3125 
3126     // OutlinedFn(&GTid, &zero_bound, CapturedStruct);
3127     Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc);
3128     Address ZeroAddrBound =
3129         CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty,
3130                                          /*Name=*/".bound.zero.addr");
3131     CGF.InitTempAlloca(ZeroAddrBound, CGF.Builder.getInt32(/*C*/ 0));
3132     llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs;
3133     // ThreadId for serialized parallels is 0.
3134     OutlinedFnArgs.push_back(ThreadIDAddr.getPointer());
3135     OutlinedFnArgs.push_back(ZeroAddrBound.getPointer());
3136     OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end());
3137     RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs);
3138 
3139     // __kmpc_end_serialized_parallel(&Loc, GTid);
3140     llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID};
3141     CGF.EmitRuntimeCall(
3142         RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel),
3143         EndArgs);
3144   };
3145   if (IfCond) {
3146     emitIfClause(CGF, IfCond, ThenGen, ElseGen);
3147   } else {
3148     RegionCodeGenTy ThenRCG(ThenGen);
3149     ThenRCG(CGF);
3150   }
3151 }
3152 
3153 // If we're inside an (outlined) parallel region, use the region info's
3154 // thread-ID variable (it is passed in a first argument of the outlined function
3155 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
3156 // regular serial code region, get thread ID by calling kmp_int32
3157 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
3158 // return the address of that temp.
3159 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
3160                                              SourceLocation Loc) {
3161   if (auto *OMPRegionInfo =
3162           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3163     if (OMPRegionInfo->getThreadIDVariable())
3164       return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF);
3165 
3166   llvm::Value *ThreadID = getThreadID(CGF, Loc);
3167   QualType Int32Ty =
3168       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
3169   Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp.");
3170   CGF.EmitStoreOfScalar(ThreadID,
3171                         CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty));
3172 
3173   return ThreadIDTemp;
3174 }
3175 
3176 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable(
3177     llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) {
3178   SmallString<256> Buffer;
3179   llvm::raw_svector_ostream Out(Buffer);
3180   Out << Name;
3181   StringRef RuntimeName = Out.str();
3182   auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first;
3183   if (Elem.second) {
3184     assert(Elem.second->getType()->getPointerElementType() == Ty &&
3185            "OMP internal variable has different type than requested");
3186     return &*Elem.second;
3187   }
3188 
3189   return Elem.second = new llvm::GlobalVariable(
3190              CGM.getModule(), Ty, /*IsConstant*/ false,
3191              llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty),
3192              Elem.first(), /*InsertBefore=*/nullptr,
3193              llvm::GlobalValue::NotThreadLocal, AddressSpace);
3194 }
3195 
3196 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
3197   std::string Prefix = Twine("gomp_critical_user_", CriticalName).str();
3198   std::string Name = getName({Prefix, "var"});
3199   return getOrCreateInternalVariable(KmpCriticalNameTy, Name);
3200 }
3201 
3202 namespace {
3203 /// Common pre(post)-action for different OpenMP constructs.
3204 class CommonActionTy final : public PrePostActionTy {
3205   llvm::FunctionCallee EnterCallee;
3206   ArrayRef<llvm::Value *> EnterArgs;
3207   llvm::FunctionCallee ExitCallee;
3208   ArrayRef<llvm::Value *> ExitArgs;
3209   bool Conditional;
3210   llvm::BasicBlock *ContBlock = nullptr;
3211 
3212 public:
3213   CommonActionTy(llvm::FunctionCallee EnterCallee,
3214                  ArrayRef<llvm::Value *> EnterArgs,
3215                  llvm::FunctionCallee ExitCallee,
3216                  ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false)
3217       : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee),
3218         ExitArgs(ExitArgs), Conditional(Conditional) {}
3219   void Enter(CodeGenFunction &CGF) override {
3220     llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs);
3221     if (Conditional) {
3222       llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes);
3223       auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
3224       ContBlock = CGF.createBasicBlock("omp_if.end");
3225       // Generate the branch (If-stmt)
3226       CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
3227       CGF.EmitBlock(ThenBlock);
3228     }
3229   }
3230   void Done(CodeGenFunction &CGF) {
3231     // Emit the rest of blocks/branches
3232     CGF.EmitBranch(ContBlock);
3233     CGF.EmitBlock(ContBlock, true);
3234   }
3235   void Exit(CodeGenFunction &CGF) override {
3236     CGF.EmitRuntimeCall(ExitCallee, ExitArgs);
3237   }
3238 };
3239 } // anonymous namespace
3240 
3241 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF,
3242                                          StringRef CriticalName,
3243                                          const RegionCodeGenTy &CriticalOpGen,
3244                                          SourceLocation Loc, const Expr *Hint) {
3245   // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]);
3246   // CriticalOpGen();
3247   // __kmpc_end_critical(ident_t *, gtid, Lock);
3248   // Prepare arguments and build a call to __kmpc_critical
3249   if (!CGF.HaveInsertPoint())
3250     return;
3251   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3252                          getCriticalRegionLock(CriticalName)};
3253   llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args),
3254                                                 std::end(Args));
3255   if (Hint) {
3256     EnterArgs.push_back(CGF.Builder.CreateIntCast(
3257         CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false));
3258   }
3259   CommonActionTy Action(
3260       createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint
3261                                  : OMPRTL__kmpc_critical),
3262       EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args);
3263   CriticalOpGen.setAction(Action);
3264   emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen);
3265 }
3266 
3267 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
3268                                        const RegionCodeGenTy &MasterOpGen,
3269                                        SourceLocation Loc) {
3270   if (!CGF.HaveInsertPoint())
3271     return;
3272   // if(__kmpc_master(ident_t *, gtid)) {
3273   //   MasterOpGen();
3274   //   __kmpc_end_master(ident_t *, gtid);
3275   // }
3276   // Prepare arguments and build a call to __kmpc_master
3277   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3278   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args,
3279                         createRuntimeFunction(OMPRTL__kmpc_end_master), Args,
3280                         /*Conditional=*/true);
3281   MasterOpGen.setAction(Action);
3282   emitInlinedDirective(CGF, OMPD_master, MasterOpGen);
3283   Action.Done(CGF);
3284 }
3285 
3286 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
3287                                         SourceLocation Loc) {
3288   if (!CGF.HaveInsertPoint())
3289     return;
3290   // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
3291   llvm::Value *Args[] = {
3292       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3293       llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
3294   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), Args);
3295   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3296     Region->emitUntiedSwitch(CGF);
3297 }
3298 
3299 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF,
3300                                           const RegionCodeGenTy &TaskgroupOpGen,
3301                                           SourceLocation Loc) {
3302   if (!CGF.HaveInsertPoint())
3303     return;
3304   // __kmpc_taskgroup(ident_t *, gtid);
3305   // TaskgroupOpGen();
3306   // __kmpc_end_taskgroup(ident_t *, gtid);
3307   // Prepare arguments and build a call to __kmpc_taskgroup
3308   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3309   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args,
3310                         createRuntimeFunction(OMPRTL__kmpc_end_taskgroup),
3311                         Args);
3312   TaskgroupOpGen.setAction(Action);
3313   emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen);
3314 }
3315 
3316 /// Given an array of pointers to variables, project the address of a
3317 /// given variable.
3318 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array,
3319                                       unsigned Index, const VarDecl *Var) {
3320   // Pull out the pointer to the variable.
3321   Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index);
3322   llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr);
3323 
3324   Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var));
3325   Addr = CGF.Builder.CreateElementBitCast(
3326       Addr, CGF.ConvertTypeForMem(Var->getType()));
3327   return Addr;
3328 }
3329 
3330 static llvm::Value *emitCopyprivateCopyFunction(
3331     CodeGenModule &CGM, llvm::Type *ArgsType,
3332     ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
3333     ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps,
3334     SourceLocation Loc) {
3335   ASTContext &C = CGM.getContext();
3336   // void copy_func(void *LHSArg, void *RHSArg);
3337   FunctionArgList Args;
3338   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3339                            ImplicitParamDecl::Other);
3340   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3341                            ImplicitParamDecl::Other);
3342   Args.push_back(&LHSArg);
3343   Args.push_back(&RHSArg);
3344   const auto &CGFI =
3345       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
3346   std::string Name =
3347       CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"});
3348   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
3349                                     llvm::GlobalValue::InternalLinkage, Name,
3350                                     &CGM.getModule());
3351   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
3352   Fn->setDoesNotRecurse();
3353   CodeGenFunction CGF(CGM);
3354   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
3355   // Dest = (void*[n])(LHSArg);
3356   // Src = (void*[n])(RHSArg);
3357   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3358       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
3359       ArgsType), CGF.getPointerAlign());
3360   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3361       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
3362       ArgsType), CGF.getPointerAlign());
3363   // *(Type0*)Dst[0] = *(Type0*)Src[0];
3364   // *(Type1*)Dst[1] = *(Type1*)Src[1];
3365   // ...
3366   // *(Typen*)Dst[n] = *(Typen*)Src[n];
3367   for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
3368     const auto *DestVar =
3369         cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl());
3370     Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar);
3371 
3372     const auto *SrcVar =
3373         cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl());
3374     Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar);
3375 
3376     const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl();
3377     QualType Type = VD->getType();
3378     CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]);
3379   }
3380   CGF.FinishFunction();
3381   return Fn;
3382 }
3383 
3384 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
3385                                        const RegionCodeGenTy &SingleOpGen,
3386                                        SourceLocation Loc,
3387                                        ArrayRef<const Expr *> CopyprivateVars,
3388                                        ArrayRef<const Expr *> SrcExprs,
3389                                        ArrayRef<const Expr *> DstExprs,
3390                                        ArrayRef<const Expr *> AssignmentOps) {
3391   if (!CGF.HaveInsertPoint())
3392     return;
3393   assert(CopyprivateVars.size() == SrcExprs.size() &&
3394          CopyprivateVars.size() == DstExprs.size() &&
3395          CopyprivateVars.size() == AssignmentOps.size());
3396   ASTContext &C = CGM.getContext();
3397   // int32 did_it = 0;
3398   // if(__kmpc_single(ident_t *, gtid)) {
3399   //   SingleOpGen();
3400   //   __kmpc_end_single(ident_t *, gtid);
3401   //   did_it = 1;
3402   // }
3403   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3404   // <copy_func>, did_it);
3405 
3406   Address DidIt = Address::invalid();
3407   if (!CopyprivateVars.empty()) {
3408     // int32 did_it = 0;
3409     QualType KmpInt32Ty =
3410         C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
3411     DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it");
3412     CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt);
3413   }
3414   // Prepare arguments and build a call to __kmpc_single
3415   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3416   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args,
3417                         createRuntimeFunction(OMPRTL__kmpc_end_single), Args,
3418                         /*Conditional=*/true);
3419   SingleOpGen.setAction(Action);
3420   emitInlinedDirective(CGF, OMPD_single, SingleOpGen);
3421   if (DidIt.isValid()) {
3422     // did_it = 1;
3423     CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt);
3424   }
3425   Action.Done(CGF);
3426   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3427   // <copy_func>, did_it);
3428   if (DidIt.isValid()) {
3429     llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
3430     QualType CopyprivateArrayTy = C.getConstantArrayType(
3431         C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
3432         /*IndexTypeQuals=*/0);
3433     // Create a list of all private variables for copyprivate.
3434     Address CopyprivateList =
3435         CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list");
3436     for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
3437       Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I);
3438       CGF.Builder.CreateStore(
3439           CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3440               CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF),
3441               CGF.VoidPtrTy),
3442           Elem);
3443     }
3444     // Build function that copies private values from single region to all other
3445     // threads in the corresponding parallel region.
3446     llvm::Value *CpyFn = emitCopyprivateCopyFunction(
3447         CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(),
3448         CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc);
3449     llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy);
3450     Address CL =
3451       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList,
3452                                                       CGF.VoidPtrTy);
3453     llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt);
3454     llvm::Value *Args[] = {
3455         emitUpdateLocation(CGF, Loc), // ident_t *<loc>
3456         getThreadID(CGF, Loc),        // i32 <gtid>
3457         BufSize,                      // size_t <buf_size>
3458         CL.getPointer(),              // void *<copyprivate list>
3459         CpyFn,                        // void (*) (void *, void *) <copy_func>
3460         DidItVal                      // i32 did_it
3461     };
3462     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args);
3463   }
3464 }
3465 
3466 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF,
3467                                         const RegionCodeGenTy &OrderedOpGen,
3468                                         SourceLocation Loc, bool IsThreads) {
3469   if (!CGF.HaveInsertPoint())
3470     return;
3471   // __kmpc_ordered(ident_t *, gtid);
3472   // OrderedOpGen();
3473   // __kmpc_end_ordered(ident_t *, gtid);
3474   // Prepare arguments and build a call to __kmpc_ordered
3475   if (IsThreads) {
3476     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3477     CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args,
3478                           createRuntimeFunction(OMPRTL__kmpc_end_ordered),
3479                           Args);
3480     OrderedOpGen.setAction(Action);
3481     emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3482     return;
3483   }
3484   emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3485 }
3486 
3487 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) {
3488   unsigned Flags;
3489   if (Kind == OMPD_for)
3490     Flags = OMP_IDENT_BARRIER_IMPL_FOR;
3491   else if (Kind == OMPD_sections)
3492     Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS;
3493   else if (Kind == OMPD_single)
3494     Flags = OMP_IDENT_BARRIER_IMPL_SINGLE;
3495   else if (Kind == OMPD_barrier)
3496     Flags = OMP_IDENT_BARRIER_EXPL;
3497   else
3498     Flags = OMP_IDENT_BARRIER_IMPL;
3499   return Flags;
3500 }
3501 
3502 void CGOpenMPRuntime::getDefaultScheduleAndChunk(
3503     CodeGenFunction &CGF, const OMPLoopDirective &S,
3504     OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const {
3505   // Check if the loop directive is actually a doacross loop directive. In this
3506   // case choose static, 1 schedule.
3507   if (llvm::any_of(
3508           S.getClausesOfKind<OMPOrderedClause>(),
3509           [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) {
3510     ScheduleKind = OMPC_SCHEDULE_static;
3511     // Chunk size is 1 in this case.
3512     llvm::APInt ChunkSize(32, 1);
3513     ChunkExpr = IntegerLiteral::Create(
3514         CGF.getContext(), ChunkSize,
3515         CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0),
3516         SourceLocation());
3517   }
3518 }
3519 
3520 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
3521                                       OpenMPDirectiveKind Kind, bool EmitChecks,
3522                                       bool ForceSimpleCall) {
3523   // Check if we should use the OMPBuilder
3524   auto *OMPRegionInfo =
3525       dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo);
3526   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3527   if (OMPBuilder) {
3528     CGF.Builder.restoreIP(OMPBuilder->CreateBarrier(
3529         CGF.Builder, Kind, ForceSimpleCall, EmitChecks));
3530     return;
3531   }
3532 
3533   if (!CGF.HaveInsertPoint())
3534     return;
3535   // Build call __kmpc_cancel_barrier(loc, thread_id);
3536   // Build call __kmpc_barrier(loc, thread_id);
3537   unsigned Flags = getDefaultFlagsForBarriers(Kind);
3538   // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc,
3539   // thread_id);
3540   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
3541                          getThreadID(CGF, Loc)};
3542   if (OMPRegionInfo) {
3543     if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) {
3544       llvm::Value *Result = CGF.EmitRuntimeCall(
3545           createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args);
3546       if (EmitChecks) {
3547         // if (__kmpc_cancel_barrier()) {
3548         //   exit from construct;
3549         // }
3550         llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
3551         llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
3552         llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
3553         CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
3554         CGF.EmitBlock(ExitBB);
3555         //   exit from construct;
3556         CodeGenFunction::JumpDest CancelDestination =
3557             CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
3558         CGF.EmitBranchThroughCleanup(CancelDestination);
3559         CGF.EmitBlock(ContBB, /*IsFinished=*/true);
3560       }
3561       return;
3562     }
3563   }
3564   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args);
3565 }
3566 
3567 /// Map the OpenMP loop schedule to the runtime enumeration.
3568 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
3569                                           bool Chunked, bool Ordered) {
3570   switch (ScheduleKind) {
3571   case OMPC_SCHEDULE_static:
3572     return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked)
3573                    : (Ordered ? OMP_ord_static : OMP_sch_static);
3574   case OMPC_SCHEDULE_dynamic:
3575     return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked;
3576   case OMPC_SCHEDULE_guided:
3577     return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked;
3578   case OMPC_SCHEDULE_runtime:
3579     return Ordered ? OMP_ord_runtime : OMP_sch_runtime;
3580   case OMPC_SCHEDULE_auto:
3581     return Ordered ? OMP_ord_auto : OMP_sch_auto;
3582   case OMPC_SCHEDULE_unknown:
3583     assert(!Chunked && "chunk was specified but schedule kind not known");
3584     return Ordered ? OMP_ord_static : OMP_sch_static;
3585   }
3586   llvm_unreachable("Unexpected runtime schedule");
3587 }
3588 
3589 /// Map the OpenMP distribute schedule to the runtime enumeration.
3590 static OpenMPSchedType
3591 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) {
3592   // only static is allowed for dist_schedule
3593   return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static;
3594 }
3595 
3596 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
3597                                          bool Chunked) const {
3598   OpenMPSchedType Schedule =
3599       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3600   return Schedule == OMP_sch_static;
3601 }
3602 
3603 bool CGOpenMPRuntime::isStaticNonchunked(
3604     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3605   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3606   return Schedule == OMP_dist_sch_static;
3607 }
3608 
3609 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind,
3610                                       bool Chunked) const {
3611   OpenMPSchedType Schedule =
3612       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3613   return Schedule == OMP_sch_static_chunked;
3614 }
3615 
3616 bool CGOpenMPRuntime::isStaticChunked(
3617     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3618   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3619   return Schedule == OMP_dist_sch_static_chunked;
3620 }
3621 
3622 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
3623   OpenMPSchedType Schedule =
3624       getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false);
3625   assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
3626   return Schedule != OMP_sch_static;
3627 }
3628 
3629 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule,
3630                                   OpenMPScheduleClauseModifier M1,
3631                                   OpenMPScheduleClauseModifier M2) {
3632   int Modifier = 0;
3633   switch (M1) {
3634   case OMPC_SCHEDULE_MODIFIER_monotonic:
3635     Modifier = OMP_sch_modifier_monotonic;
3636     break;
3637   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3638     Modifier = OMP_sch_modifier_nonmonotonic;
3639     break;
3640   case OMPC_SCHEDULE_MODIFIER_simd:
3641     if (Schedule == OMP_sch_static_chunked)
3642       Schedule = OMP_sch_static_balanced_chunked;
3643     break;
3644   case OMPC_SCHEDULE_MODIFIER_last:
3645   case OMPC_SCHEDULE_MODIFIER_unknown:
3646     break;
3647   }
3648   switch (M2) {
3649   case OMPC_SCHEDULE_MODIFIER_monotonic:
3650     Modifier = OMP_sch_modifier_monotonic;
3651     break;
3652   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3653     Modifier = OMP_sch_modifier_nonmonotonic;
3654     break;
3655   case OMPC_SCHEDULE_MODIFIER_simd:
3656     if (Schedule == OMP_sch_static_chunked)
3657       Schedule = OMP_sch_static_balanced_chunked;
3658     break;
3659   case OMPC_SCHEDULE_MODIFIER_last:
3660   case OMPC_SCHEDULE_MODIFIER_unknown:
3661     break;
3662   }
3663   // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription.
3664   // If the static schedule kind is specified or if the ordered clause is
3665   // specified, and if the nonmonotonic modifier is not specified, the effect is
3666   // as if the monotonic modifier is specified. Otherwise, unless the monotonic
3667   // modifier is specified, the effect is as if the nonmonotonic modifier is
3668   // specified.
3669   if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) {
3670     if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static ||
3671           Schedule == OMP_sch_static_balanced_chunked ||
3672           Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static ||
3673           Schedule == OMP_dist_sch_static_chunked ||
3674           Schedule == OMP_dist_sch_static))
3675       Modifier = OMP_sch_modifier_nonmonotonic;
3676   }
3677   return Schedule | Modifier;
3678 }
3679 
3680 void CGOpenMPRuntime::emitForDispatchInit(
3681     CodeGenFunction &CGF, SourceLocation Loc,
3682     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
3683     bool Ordered, const DispatchRTInput &DispatchValues) {
3684   if (!CGF.HaveInsertPoint())
3685     return;
3686   OpenMPSchedType Schedule = getRuntimeSchedule(
3687       ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered);
3688   assert(Ordered ||
3689          (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked &&
3690           Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked &&
3691           Schedule != OMP_sch_static_balanced_chunked));
3692   // Call __kmpc_dispatch_init(
3693   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
3694   //          kmp_int[32|64] lower, kmp_int[32|64] upper,
3695   //          kmp_int[32|64] stride, kmp_int[32|64] chunk);
3696 
3697   // If the Chunk was not specified in the clause - use default value 1.
3698   llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk
3699                                             : CGF.Builder.getIntN(IVSize, 1);
3700   llvm::Value *Args[] = {
3701       emitUpdateLocation(CGF, Loc),
3702       getThreadID(CGF, Loc),
3703       CGF.Builder.getInt32(addMonoNonMonoModifier(
3704           CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type
3705       DispatchValues.LB,                                     // Lower
3706       DispatchValues.UB,                                     // Upper
3707       CGF.Builder.getIntN(IVSize, 1),                        // Stride
3708       Chunk                                                  // Chunk
3709   };
3710   CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args);
3711 }
3712 
3713 static void emitForStaticInitCall(
3714     CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId,
3715     llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule,
3716     OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2,
3717     const CGOpenMPRuntime::StaticRTInput &Values) {
3718   if (!CGF.HaveInsertPoint())
3719     return;
3720 
3721   assert(!Values.Ordered);
3722   assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked ||
3723          Schedule == OMP_sch_static_balanced_chunked ||
3724          Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked ||
3725          Schedule == OMP_dist_sch_static ||
3726          Schedule == OMP_dist_sch_static_chunked);
3727 
3728   // Call __kmpc_for_static_init(
3729   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
3730   //          kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
3731   //          kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
3732   //          kmp_int[32|64] incr, kmp_int[32|64] chunk);
3733   llvm::Value *Chunk = Values.Chunk;
3734   if (Chunk == nullptr) {
3735     assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static ||
3736             Schedule == OMP_dist_sch_static) &&
3737            "expected static non-chunked schedule");
3738     // If the Chunk was not specified in the clause - use default value 1.
3739     Chunk = CGF.Builder.getIntN(Values.IVSize, 1);
3740   } else {
3741     assert((Schedule == OMP_sch_static_chunked ||
3742             Schedule == OMP_sch_static_balanced_chunked ||
3743             Schedule == OMP_ord_static_chunked ||
3744             Schedule == OMP_dist_sch_static_chunked) &&
3745            "expected static chunked schedule");
3746   }
3747   llvm::Value *Args[] = {
3748       UpdateLocation,
3749       ThreadId,
3750       CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1,
3751                                                   M2)), // Schedule type
3752       Values.IL.getPointer(),                           // &isLastIter
3753       Values.LB.getPointer(),                           // &LB
3754       Values.UB.getPointer(),                           // &UB
3755       Values.ST.getPointer(),                           // &Stride
3756       CGF.Builder.getIntN(Values.IVSize, 1),            // Incr
3757       Chunk                                             // Chunk
3758   };
3759   CGF.EmitRuntimeCall(ForStaticInitFunction, Args);
3760 }
3761 
3762 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF,
3763                                         SourceLocation Loc,
3764                                         OpenMPDirectiveKind DKind,
3765                                         const OpenMPScheduleTy &ScheduleKind,
3766                                         const StaticRTInput &Values) {
3767   OpenMPSchedType ScheduleNum = getRuntimeSchedule(
3768       ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered);
3769   assert(isOpenMPWorksharingDirective(DKind) &&
3770          "Expected loop-based or sections-based directive.");
3771   llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc,
3772                                              isOpenMPLoopDirective(DKind)
3773                                                  ? OMP_IDENT_WORK_LOOP
3774                                                  : OMP_IDENT_WORK_SECTIONS);
3775   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3776   llvm::FunctionCallee StaticInitFunction =
3777       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3778   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
3779   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3780                         ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values);
3781 }
3782 
3783 void CGOpenMPRuntime::emitDistributeStaticInit(
3784     CodeGenFunction &CGF, SourceLocation Loc,
3785     OpenMPDistScheduleClauseKind SchedKind,
3786     const CGOpenMPRuntime::StaticRTInput &Values) {
3787   OpenMPSchedType ScheduleNum =
3788       getRuntimeSchedule(SchedKind, Values.Chunk != nullptr);
3789   llvm::Value *UpdatedLocation =
3790       emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE);
3791   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3792   llvm::FunctionCallee StaticInitFunction =
3793       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3794   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3795                         ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown,
3796                         OMPC_SCHEDULE_MODIFIER_unknown, Values);
3797 }
3798 
3799 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF,
3800                                           SourceLocation Loc,
3801                                           OpenMPDirectiveKind DKind) {
3802   if (!CGF.HaveInsertPoint())
3803     return;
3804   // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
3805   llvm::Value *Args[] = {
3806       emitUpdateLocation(CGF, Loc,
3807                          isOpenMPDistributeDirective(DKind)
3808                              ? OMP_IDENT_WORK_DISTRIBUTE
3809                              : isOpenMPLoopDirective(DKind)
3810                                    ? OMP_IDENT_WORK_LOOP
3811                                    : OMP_IDENT_WORK_SECTIONS),
3812       getThreadID(CGF, Loc)};
3813   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
3814   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini),
3815                       Args);
3816 }
3817 
3818 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
3819                                                  SourceLocation Loc,
3820                                                  unsigned IVSize,
3821                                                  bool IVSigned) {
3822   if (!CGF.HaveInsertPoint())
3823     return;
3824   // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
3825   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3826   CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args);
3827 }
3828 
3829 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
3830                                           SourceLocation Loc, unsigned IVSize,
3831                                           bool IVSigned, Address IL,
3832                                           Address LB, Address UB,
3833                                           Address ST) {
3834   // Call __kmpc_dispatch_next(
3835   //          ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
3836   //          kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
3837   //          kmp_int[32|64] *p_stride);
3838   llvm::Value *Args[] = {
3839       emitUpdateLocation(CGF, Loc),
3840       getThreadID(CGF, Loc),
3841       IL.getPointer(), // &isLastIter
3842       LB.getPointer(), // &Lower
3843       UB.getPointer(), // &Upper
3844       ST.getPointer()  // &Stride
3845   };
3846   llvm::Value *Call =
3847       CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args);
3848   return CGF.EmitScalarConversion(
3849       Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1),
3850       CGF.getContext().BoolTy, Loc);
3851 }
3852 
3853 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
3854                                            llvm::Value *NumThreads,
3855                                            SourceLocation Loc) {
3856   if (!CGF.HaveInsertPoint())
3857     return;
3858   // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
3859   llvm::Value *Args[] = {
3860       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3861       CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)};
3862   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads),
3863                       Args);
3864 }
3865 
3866 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF,
3867                                          ProcBindKind ProcBind,
3868                                          SourceLocation Loc) {
3869   if (!CGF.HaveInsertPoint())
3870     return;
3871   assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value.");
3872   // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind)
3873   llvm::Value *Args[] = {
3874       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3875       llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)};
3876   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args);
3877 }
3878 
3879 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
3880                                 SourceLocation Loc, llvm::AtomicOrdering AO) {
3881   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3882   if (OMPBuilder) {
3883     OMPBuilder->CreateFlush(CGF.Builder);
3884   } else {
3885     if (!CGF.HaveInsertPoint())
3886       return;
3887     // Build call void __kmpc_flush(ident_t *loc)
3888     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush),
3889                         emitUpdateLocation(CGF, Loc));
3890   }
3891 }
3892 
3893 namespace {
3894 /// Indexes of fields for type kmp_task_t.
3895 enum KmpTaskTFields {
3896   /// List of shared variables.
3897   KmpTaskTShareds,
3898   /// Task routine.
3899   KmpTaskTRoutine,
3900   /// Partition id for the untied tasks.
3901   KmpTaskTPartId,
3902   /// Function with call of destructors for private variables.
3903   Data1,
3904   /// Task priority.
3905   Data2,
3906   /// (Taskloops only) Lower bound.
3907   KmpTaskTLowerBound,
3908   /// (Taskloops only) Upper bound.
3909   KmpTaskTUpperBound,
3910   /// (Taskloops only) Stride.
3911   KmpTaskTStride,
3912   /// (Taskloops only) Is last iteration flag.
3913   KmpTaskTLastIter,
3914   /// (Taskloops only) Reduction data.
3915   KmpTaskTReductions,
3916 };
3917 } // anonymous namespace
3918 
3919 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const {
3920   return OffloadEntriesTargetRegion.empty() &&
3921          OffloadEntriesDeviceGlobalVar.empty();
3922 }
3923 
3924 /// Initialize target region entry.
3925 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3926     initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3927                                     StringRef ParentName, unsigned LineNum,
3928                                     unsigned Order) {
3929   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3930                                              "only required for the device "
3931                                              "code generation.");
3932   OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] =
3933       OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr,
3934                                    OMPTargetRegionEntryTargetRegion);
3935   ++OffloadingEntriesNum;
3936 }
3937 
3938 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3939     registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3940                                   StringRef ParentName, unsigned LineNum,
3941                                   llvm::Constant *Addr, llvm::Constant *ID,
3942                                   OMPTargetRegionEntryKind Flags) {
3943   // If we are emitting code for a target, the entry is already initialized,
3944   // only has to be registered.
3945   if (CGM.getLangOpts().OpenMPIsDevice) {
3946     if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) {
3947       unsigned DiagID = CGM.getDiags().getCustomDiagID(
3948           DiagnosticsEngine::Error,
3949           "Unable to find target region on line '%0' in the device code.");
3950       CGM.getDiags().Report(DiagID) << LineNum;
3951       return;
3952     }
3953     auto &Entry =
3954         OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum];
3955     assert(Entry.isValid() && "Entry not initialized!");
3956     Entry.setAddress(Addr);
3957     Entry.setID(ID);
3958     Entry.setFlags(Flags);
3959   } else {
3960     OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags);
3961     OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry;
3962     ++OffloadingEntriesNum;
3963   }
3964 }
3965 
3966 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo(
3967     unsigned DeviceID, unsigned FileID, StringRef ParentName,
3968     unsigned LineNum) const {
3969   auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID);
3970   if (PerDevice == OffloadEntriesTargetRegion.end())
3971     return false;
3972   auto PerFile = PerDevice->second.find(FileID);
3973   if (PerFile == PerDevice->second.end())
3974     return false;
3975   auto PerParentName = PerFile->second.find(ParentName);
3976   if (PerParentName == PerFile->second.end())
3977     return false;
3978   auto PerLine = PerParentName->second.find(LineNum);
3979   if (PerLine == PerParentName->second.end())
3980     return false;
3981   // Fail if this entry is already registered.
3982   if (PerLine->second.getAddress() || PerLine->second.getID())
3983     return false;
3984   return true;
3985 }
3986 
3987 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo(
3988     const OffloadTargetRegionEntryInfoActTy &Action) {
3989   // Scan all target region entries and perform the provided action.
3990   for (const auto &D : OffloadEntriesTargetRegion)
3991     for (const auto &F : D.second)
3992       for (const auto &P : F.second)
3993         for (const auto &L : P.second)
3994           Action(D.first, F.first, P.first(), L.first, L.second);
3995 }
3996 
3997 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3998     initializeDeviceGlobalVarEntryInfo(StringRef Name,
3999                                        OMPTargetGlobalVarEntryKind Flags,
4000                                        unsigned Order) {
4001   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
4002                                              "only required for the device "
4003                                              "code generation.");
4004   OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
4005   ++OffloadingEntriesNum;
4006 }
4007 
4008 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
4009     registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr,
4010                                      CharUnits VarSize,
4011                                      OMPTargetGlobalVarEntryKind Flags,
4012                                      llvm::GlobalValue::LinkageTypes Linkage) {
4013   if (CGM.getLangOpts().OpenMPIsDevice) {
4014     auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
4015     assert(Entry.isValid() && Entry.getFlags() == Flags &&
4016            "Entry not initialized!");
4017     assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
4018            "Resetting with the new address.");
4019     if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) {
4020       if (Entry.getVarSize().isZero()) {
4021         Entry.setVarSize(VarSize);
4022         Entry.setLinkage(Linkage);
4023       }
4024       return;
4025     }
4026     Entry.setVarSize(VarSize);
4027     Entry.setLinkage(Linkage);
4028     Entry.setAddress(Addr);
4029   } else {
4030     if (hasDeviceGlobalVarEntryInfo(VarName)) {
4031       auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
4032       assert(Entry.isValid() && Entry.getFlags() == Flags &&
4033              "Entry not initialized!");
4034       assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
4035              "Resetting with the new address.");
4036       if (Entry.getVarSize().isZero()) {
4037         Entry.setVarSize(VarSize);
4038         Entry.setLinkage(Linkage);
4039       }
4040       return;
4041     }
4042     OffloadEntriesDeviceGlobalVar.try_emplace(
4043         VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage);
4044     ++OffloadingEntriesNum;
4045   }
4046 }
4047 
4048 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
4049     actOnDeviceGlobalVarEntriesInfo(
4050         const OffloadDeviceGlobalVarEntryInfoActTy &Action) {
4051   // Scan all target region entries and perform the provided action.
4052   for (const auto &E : OffloadEntriesDeviceGlobalVar)
4053     Action(E.getKey(), E.getValue());
4054 }
4055 
4056 void CGOpenMPRuntime::createOffloadEntry(
4057     llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags,
4058     llvm::GlobalValue::LinkageTypes Linkage) {
4059   StringRef Name = Addr->getName();
4060   llvm::Module &M = CGM.getModule();
4061   llvm::LLVMContext &C = M.getContext();
4062 
4063   // Create constant string with the name.
4064   llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name);
4065 
4066   std::string StringName = getName({"omp_offloading", "entry_name"});
4067   auto *Str = new llvm::GlobalVariable(
4068       M, StrPtrInit->getType(), /*isConstant=*/true,
4069       llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName);
4070   Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
4071 
4072   llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy),
4073                             llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy),
4074                             llvm::ConstantInt::get(CGM.SizeTy, Size),
4075                             llvm::ConstantInt::get(CGM.Int32Ty, Flags),
4076                             llvm::ConstantInt::get(CGM.Int32Ty, 0)};
4077   std::string EntryName = getName({"omp_offloading", "entry", ""});
4078   llvm::GlobalVariable *Entry = createGlobalStruct(
4079       CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data,
4080       Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage);
4081 
4082   // The entry has to be created in the section the linker expects it to be.
4083   Entry->setSection("omp_offloading_entries");
4084 }
4085 
4086 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() {
4087   // Emit the offloading entries and metadata so that the device codegen side
4088   // can easily figure out what to emit. The produced metadata looks like
4089   // this:
4090   //
4091   // !omp_offload.info = !{!1, ...}
4092   //
4093   // Right now we only generate metadata for function that contain target
4094   // regions.
4095 
4096   // If we are in simd mode or there are no entries, we don't need to do
4097   // anything.
4098   if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty())
4099     return;
4100 
4101   llvm::Module &M = CGM.getModule();
4102   llvm::LLVMContext &C = M.getContext();
4103   SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *,
4104                          SourceLocation, StringRef>,
4105               16>
4106       OrderedEntries(OffloadEntriesInfoManager.size());
4107   llvm::SmallVector<StringRef, 16> ParentFunctions(
4108       OffloadEntriesInfoManager.size());
4109 
4110   // Auxiliary methods to create metadata values and strings.
4111   auto &&GetMDInt = [this](unsigned V) {
4112     return llvm::ConstantAsMetadata::get(
4113         llvm::ConstantInt::get(CGM.Int32Ty, V));
4114   };
4115 
4116   auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); };
4117 
4118   // Create the offloading info metadata node.
4119   llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info");
4120 
4121   // Create function that emits metadata for each target region entry;
4122   auto &&TargetRegionMetadataEmitter =
4123       [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt,
4124        &GetMDString](
4125           unsigned DeviceID, unsigned FileID, StringRef ParentName,
4126           unsigned Line,
4127           const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) {
4128         // Generate metadata for target regions. Each entry of this metadata
4129         // contains:
4130         // - Entry 0 -> Kind of this type of metadata (0).
4131         // - Entry 1 -> Device ID of the file where the entry was identified.
4132         // - Entry 2 -> File ID of the file where the entry was identified.
4133         // - Entry 3 -> Mangled name of the function where the entry was
4134         // identified.
4135         // - Entry 4 -> Line in the file where the entry was identified.
4136         // - Entry 5 -> Order the entry was created.
4137         // The first element of the metadata node is the kind.
4138         llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID),
4139                                  GetMDInt(FileID),      GetMDString(ParentName),
4140                                  GetMDInt(Line),        GetMDInt(E.getOrder())};
4141 
4142         SourceLocation Loc;
4143         for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(),
4144                   E = CGM.getContext().getSourceManager().fileinfo_end();
4145              I != E; ++I) {
4146           if (I->getFirst()->getUniqueID().getDevice() == DeviceID &&
4147               I->getFirst()->getUniqueID().getFile() == FileID) {
4148             Loc = CGM.getContext().getSourceManager().translateFileLineCol(
4149                 I->getFirst(), Line, 1);
4150             break;
4151           }
4152         }
4153         // Save this entry in the right position of the ordered entries array.
4154         OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName);
4155         ParentFunctions[E.getOrder()] = ParentName;
4156 
4157         // Add metadata to the named metadata node.
4158         MD->addOperand(llvm::MDNode::get(C, Ops));
4159       };
4160 
4161   OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo(
4162       TargetRegionMetadataEmitter);
4163 
4164   // Create function that emits metadata for each device global variable entry;
4165   auto &&DeviceGlobalVarMetadataEmitter =
4166       [&C, &OrderedEntries, &GetMDInt, &GetMDString,
4167        MD](StringRef MangledName,
4168            const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar
4169                &E) {
4170         // Generate metadata for global variables. Each entry of this metadata
4171         // contains:
4172         // - Entry 0 -> Kind of this type of metadata (1).
4173         // - Entry 1 -> Mangled name of the variable.
4174         // - Entry 2 -> Declare target kind.
4175         // - Entry 3 -> Order the entry was created.
4176         // The first element of the metadata node is the kind.
4177         llvm::Metadata *Ops[] = {
4178             GetMDInt(E.getKind()), GetMDString(MangledName),
4179             GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
4180 
4181         // Save this entry in the right position of the ordered entries array.
4182         OrderedEntries[E.getOrder()] =
4183             std::make_tuple(&E, SourceLocation(), MangledName);
4184 
4185         // Add metadata to the named metadata node.
4186         MD->addOperand(llvm::MDNode::get(C, Ops));
4187       };
4188 
4189   OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo(
4190       DeviceGlobalVarMetadataEmitter);
4191 
4192   for (const auto &E : OrderedEntries) {
4193     assert(std::get<0>(E) && "All ordered entries must exist!");
4194     if (const auto *CE =
4195             dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>(
4196                 std::get<0>(E))) {
4197       if (!CE->getID() || !CE->getAddress()) {
4198         // Do not blame the entry if the parent funtion is not emitted.
4199         StringRef FnName = ParentFunctions[CE->getOrder()];
4200         if (!CGM.GetGlobalValue(FnName))
4201           continue;
4202         unsigned DiagID = CGM.getDiags().getCustomDiagID(
4203             DiagnosticsEngine::Error,
4204             "Offloading entry for target region in %0 is incorrect: either the "
4205             "address or the ID is invalid.");
4206         CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName;
4207         continue;
4208       }
4209       createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0,
4210                          CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage);
4211     } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy::
4212                                              OffloadEntryInfoDeviceGlobalVar>(
4213                    std::get<0>(E))) {
4214       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags =
4215           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4216               CE->getFlags());
4217       switch (Flags) {
4218       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: {
4219         if (CGM.getLangOpts().OpenMPIsDevice &&
4220             CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())
4221           continue;
4222         if (!CE->getAddress()) {
4223           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4224               DiagnosticsEngine::Error, "Offloading entry for declare target "
4225                                         "variable %0 is incorrect: the "
4226                                         "address is invalid.");
4227           CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E);
4228           continue;
4229         }
4230         // The vaiable has no definition - no need to add the entry.
4231         if (CE->getVarSize().isZero())
4232           continue;
4233         break;
4234       }
4235       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink:
4236         assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) ||
4237                 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) &&
4238                "Declaret target link address is set.");
4239         if (CGM.getLangOpts().OpenMPIsDevice)
4240           continue;
4241         if (!CE->getAddress()) {
4242           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4243               DiagnosticsEngine::Error,
4244               "Offloading entry for declare target variable is incorrect: the "
4245               "address is invalid.");
4246           CGM.getDiags().Report(DiagID);
4247           continue;
4248         }
4249         break;
4250       }
4251       createOffloadEntry(CE->getAddress(), CE->getAddress(),
4252                          CE->getVarSize().getQuantity(), Flags,
4253                          CE->getLinkage());
4254     } else {
4255       llvm_unreachable("Unsupported entry kind.");
4256     }
4257   }
4258 }
4259 
4260 /// Loads all the offload entries information from the host IR
4261 /// metadata.
4262 void CGOpenMPRuntime::loadOffloadInfoMetadata() {
4263   // If we are in target mode, load the metadata from the host IR. This code has
4264   // to match the metadaata creation in createOffloadEntriesAndInfoMetadata().
4265 
4266   if (!CGM.getLangOpts().OpenMPIsDevice)
4267     return;
4268 
4269   if (CGM.getLangOpts().OMPHostIRFile.empty())
4270     return;
4271 
4272   auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile);
4273   if (auto EC = Buf.getError()) {
4274     CGM.getDiags().Report(diag::err_cannot_open_file)
4275         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4276     return;
4277   }
4278 
4279   llvm::LLVMContext C;
4280   auto ME = expectedToErrorOrAndEmitErrors(
4281       C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C));
4282 
4283   if (auto EC = ME.getError()) {
4284     unsigned DiagID = CGM.getDiags().getCustomDiagID(
4285         DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'");
4286     CGM.getDiags().Report(DiagID)
4287         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4288     return;
4289   }
4290 
4291   llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info");
4292   if (!MD)
4293     return;
4294 
4295   for (llvm::MDNode *MN : MD->operands()) {
4296     auto &&GetMDInt = [MN](unsigned Idx) {
4297       auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx));
4298       return cast<llvm::ConstantInt>(V->getValue())->getZExtValue();
4299     };
4300 
4301     auto &&GetMDString = [MN](unsigned Idx) {
4302       auto *V = cast<llvm::MDString>(MN->getOperand(Idx));
4303       return V->getString();
4304     };
4305 
4306     switch (GetMDInt(0)) {
4307     default:
4308       llvm_unreachable("Unexpected metadata!");
4309       break;
4310     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4311         OffloadingEntryInfoTargetRegion:
4312       OffloadEntriesInfoManager.initializeTargetRegionEntryInfo(
4313           /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2),
4314           /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4),
4315           /*Order=*/GetMDInt(5));
4316       break;
4317     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4318         OffloadingEntryInfoDeviceGlobalVar:
4319       OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo(
4320           /*MangledName=*/GetMDString(1),
4321           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4322               /*Flags=*/GetMDInt(2)),
4323           /*Order=*/GetMDInt(3));
4324       break;
4325     }
4326   }
4327 }
4328 
4329 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
4330   if (!KmpRoutineEntryPtrTy) {
4331     // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
4332     ASTContext &C = CGM.getContext();
4333     QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
4334     FunctionProtoType::ExtProtoInfo EPI;
4335     KmpRoutineEntryPtrQTy = C.getPointerType(
4336         C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
4337     KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
4338   }
4339 }
4340 
4341 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() {
4342   // Make sure the type of the entry is already created. This is the type we
4343   // have to create:
4344   // struct __tgt_offload_entry{
4345   //   void      *addr;       // Pointer to the offload entry info.
4346   //                          // (function or global)
4347   //   char      *name;       // Name of the function or global.
4348   //   size_t     size;       // Size of the entry info (0 if it a function).
4349   //   int32_t    flags;      // Flags associated with the entry, e.g. 'link'.
4350   //   int32_t    reserved;   // Reserved, to use by the runtime library.
4351   // };
4352   if (TgtOffloadEntryQTy.isNull()) {
4353     ASTContext &C = CGM.getContext();
4354     RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry");
4355     RD->startDefinition();
4356     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4357     addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy));
4358     addFieldToRecordDecl(C, RD, C.getSizeType());
4359     addFieldToRecordDecl(
4360         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4361     addFieldToRecordDecl(
4362         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4363     RD->completeDefinition();
4364     RD->addAttr(PackedAttr::CreateImplicit(C));
4365     TgtOffloadEntryQTy = C.getRecordType(RD);
4366   }
4367   return TgtOffloadEntryQTy;
4368 }
4369 
4370 namespace {
4371 struct PrivateHelpersTy {
4372   PrivateHelpersTy(const VarDecl *Original, const VarDecl *PrivateCopy,
4373                    const VarDecl *PrivateElemInit)
4374       : Original(Original), PrivateCopy(PrivateCopy),
4375         PrivateElemInit(PrivateElemInit) {}
4376   const VarDecl *Original;
4377   const VarDecl *PrivateCopy;
4378   const VarDecl *PrivateElemInit;
4379 };
4380 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
4381 } // anonymous namespace
4382 
4383 static RecordDecl *
4384 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) {
4385   if (!Privates.empty()) {
4386     ASTContext &C = CGM.getContext();
4387     // Build struct .kmp_privates_t. {
4388     //         /*  private vars  */
4389     //       };
4390     RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t");
4391     RD->startDefinition();
4392     for (const auto &Pair : Privates) {
4393       const VarDecl *VD = Pair.second.Original;
4394       QualType Type = VD->getType().getNonReferenceType();
4395       FieldDecl *FD = addFieldToRecordDecl(C, RD, Type);
4396       if (VD->hasAttrs()) {
4397         for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()),
4398              E(VD->getAttrs().end());
4399              I != E; ++I)
4400           FD->addAttr(*I);
4401       }
4402     }
4403     RD->completeDefinition();
4404     return RD;
4405   }
4406   return nullptr;
4407 }
4408 
4409 static RecordDecl *
4410 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind,
4411                          QualType KmpInt32Ty,
4412                          QualType KmpRoutineEntryPointerQTy) {
4413   ASTContext &C = CGM.getContext();
4414   // Build struct kmp_task_t {
4415   //         void *              shareds;
4416   //         kmp_routine_entry_t routine;
4417   //         kmp_int32           part_id;
4418   //         kmp_cmplrdata_t data1;
4419   //         kmp_cmplrdata_t data2;
4420   // For taskloops additional fields:
4421   //         kmp_uint64          lb;
4422   //         kmp_uint64          ub;
4423   //         kmp_int64           st;
4424   //         kmp_int32           liter;
4425   //         void *              reductions;
4426   //       };
4427   RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union);
4428   UD->startDefinition();
4429   addFieldToRecordDecl(C, UD, KmpInt32Ty);
4430   addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy);
4431   UD->completeDefinition();
4432   QualType KmpCmplrdataTy = C.getRecordType(UD);
4433   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t");
4434   RD->startDefinition();
4435   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4436   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
4437   addFieldToRecordDecl(C, RD, KmpInt32Ty);
4438   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4439   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4440   if (isOpenMPTaskLoopDirective(Kind)) {
4441     QualType KmpUInt64Ty =
4442         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
4443     QualType KmpInt64Ty =
4444         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
4445     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4446     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4447     addFieldToRecordDecl(C, RD, KmpInt64Ty);
4448     addFieldToRecordDecl(C, RD, KmpInt32Ty);
4449     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4450   }
4451   RD->completeDefinition();
4452   return RD;
4453 }
4454 
4455 static RecordDecl *
4456 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy,
4457                                      ArrayRef<PrivateDataTy> Privates) {
4458   ASTContext &C = CGM.getContext();
4459   // Build struct kmp_task_t_with_privates {
4460   //         kmp_task_t task_data;
4461   //         .kmp_privates_t. privates;
4462   //       };
4463   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates");
4464   RD->startDefinition();
4465   addFieldToRecordDecl(C, RD, KmpTaskTQTy);
4466   if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates))
4467     addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD));
4468   RD->completeDefinition();
4469   return RD;
4470 }
4471 
4472 /// Emit a proxy function which accepts kmp_task_t as the second
4473 /// argument.
4474 /// \code
4475 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
4476 ///   TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt,
4477 ///   For taskloops:
4478 ///   tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4479 ///   tt->reductions, tt->shareds);
4480 ///   return 0;
4481 /// }
4482 /// \endcode
4483 static llvm::Function *
4484 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
4485                       OpenMPDirectiveKind Kind, QualType KmpInt32Ty,
4486                       QualType KmpTaskTWithPrivatesPtrQTy,
4487                       QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
4488                       QualType SharedsPtrTy, llvm::Function *TaskFunction,
4489                       llvm::Value *TaskPrivatesMap) {
4490   ASTContext &C = CGM.getContext();
4491   FunctionArgList Args;
4492   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4493                             ImplicitParamDecl::Other);
4494   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4495                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4496                                 ImplicitParamDecl::Other);
4497   Args.push_back(&GtidArg);
4498   Args.push_back(&TaskTypeArg);
4499   const auto &TaskEntryFnInfo =
4500       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4501   llvm::FunctionType *TaskEntryTy =
4502       CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
4503   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""});
4504   auto *TaskEntry = llvm::Function::Create(
4505       TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4506   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo);
4507   TaskEntry->setDoesNotRecurse();
4508   CodeGenFunction CGF(CGM);
4509   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args,
4510                     Loc, Loc);
4511 
4512   // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map,
4513   // tt,
4514   // For taskloops:
4515   // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4516   // tt->task_data.shareds);
4517   llvm::Value *GtidParam = CGF.EmitLoadOfScalar(
4518       CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc);
4519   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4520       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4521       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4522   const auto *KmpTaskTWithPrivatesQTyRD =
4523       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4524   LValue Base =
4525       CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4526   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
4527   auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
4528   LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI);
4529   llvm::Value *PartidParam = PartIdLVal.getPointer(CGF);
4530 
4531   auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds);
4532   LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI);
4533   llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4534       CGF.EmitLoadOfScalar(SharedsLVal, Loc),
4535       CGF.ConvertTypeForMem(SharedsPtrTy));
4536 
4537   auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
4538   llvm::Value *PrivatesParam;
4539   if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) {
4540     LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI);
4541     PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4542         PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy);
4543   } else {
4544     PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4545   }
4546 
4547   llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam,
4548                                TaskPrivatesMap,
4549                                CGF.Builder
4550                                    .CreatePointerBitCastOrAddrSpaceCast(
4551                                        TDBase.getAddress(CGF), CGF.VoidPtrTy)
4552                                    .getPointer()};
4553   SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs),
4554                                           std::end(CommonArgs));
4555   if (isOpenMPTaskLoopDirective(Kind)) {
4556     auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound);
4557     LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI);
4558     llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc);
4559     auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound);
4560     LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI);
4561     llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc);
4562     auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride);
4563     LValue StLVal = CGF.EmitLValueForField(Base, *StFI);
4564     llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc);
4565     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4566     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4567     llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc);
4568     auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions);
4569     LValue RLVal = CGF.EmitLValueForField(Base, *RFI);
4570     llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc);
4571     CallArgs.push_back(LBParam);
4572     CallArgs.push_back(UBParam);
4573     CallArgs.push_back(StParam);
4574     CallArgs.push_back(LIParam);
4575     CallArgs.push_back(RParam);
4576   }
4577   CallArgs.push_back(SharedsParam);
4578 
4579   CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction,
4580                                                   CallArgs);
4581   CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)),
4582                              CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty));
4583   CGF.FinishFunction();
4584   return TaskEntry;
4585 }
4586 
4587 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM,
4588                                             SourceLocation Loc,
4589                                             QualType KmpInt32Ty,
4590                                             QualType KmpTaskTWithPrivatesPtrQTy,
4591                                             QualType KmpTaskTWithPrivatesQTy) {
4592   ASTContext &C = CGM.getContext();
4593   FunctionArgList Args;
4594   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4595                             ImplicitParamDecl::Other);
4596   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4597                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4598                                 ImplicitParamDecl::Other);
4599   Args.push_back(&GtidArg);
4600   Args.push_back(&TaskTypeArg);
4601   const auto &DestructorFnInfo =
4602       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4603   llvm::FunctionType *DestructorFnTy =
4604       CGM.getTypes().GetFunctionType(DestructorFnInfo);
4605   std::string Name =
4606       CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""});
4607   auto *DestructorFn =
4608       llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage,
4609                              Name, &CGM.getModule());
4610   CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn,
4611                                     DestructorFnInfo);
4612   DestructorFn->setDoesNotRecurse();
4613   CodeGenFunction CGF(CGM);
4614   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo,
4615                     Args, Loc, Loc);
4616 
4617   LValue Base = CGF.EmitLoadOfPointerLValue(
4618       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4619       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4620   const auto *KmpTaskTWithPrivatesQTyRD =
4621       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4622   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4623   Base = CGF.EmitLValueForField(Base, *FI);
4624   for (const auto *Field :
4625        cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) {
4626     if (QualType::DestructionKind DtorKind =
4627             Field->getType().isDestructedType()) {
4628       LValue FieldLValue = CGF.EmitLValueForField(Base, Field);
4629       CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType());
4630     }
4631   }
4632   CGF.FinishFunction();
4633   return DestructorFn;
4634 }
4635 
4636 /// Emit a privates mapping function for correct handling of private and
4637 /// firstprivate variables.
4638 /// \code
4639 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1>
4640 /// **noalias priv1,...,  <tyn> **noalias privn) {
4641 ///   *priv1 = &.privates.priv1;
4642 ///   ...;
4643 ///   *privn = &.privates.privn;
4644 /// }
4645 /// \endcode
4646 static llvm::Value *
4647 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc,
4648                                ArrayRef<const Expr *> PrivateVars,
4649                                ArrayRef<const Expr *> FirstprivateVars,
4650                                ArrayRef<const Expr *> LastprivateVars,
4651                                QualType PrivatesQTy,
4652                                ArrayRef<PrivateDataTy> Privates) {
4653   ASTContext &C = CGM.getContext();
4654   FunctionArgList Args;
4655   ImplicitParamDecl TaskPrivatesArg(
4656       C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4657       C.getPointerType(PrivatesQTy).withConst().withRestrict(),
4658       ImplicitParamDecl::Other);
4659   Args.push_back(&TaskPrivatesArg);
4660   llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos;
4661   unsigned Counter = 1;
4662   for (const Expr *E : PrivateVars) {
4663     Args.push_back(ImplicitParamDecl::Create(
4664         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4665         C.getPointerType(C.getPointerType(E->getType()))
4666             .withConst()
4667             .withRestrict(),
4668         ImplicitParamDecl::Other));
4669     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4670     PrivateVarsPos[VD] = Counter;
4671     ++Counter;
4672   }
4673   for (const Expr *E : FirstprivateVars) {
4674     Args.push_back(ImplicitParamDecl::Create(
4675         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4676         C.getPointerType(C.getPointerType(E->getType()))
4677             .withConst()
4678             .withRestrict(),
4679         ImplicitParamDecl::Other));
4680     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4681     PrivateVarsPos[VD] = Counter;
4682     ++Counter;
4683   }
4684   for (const Expr *E : LastprivateVars) {
4685     Args.push_back(ImplicitParamDecl::Create(
4686         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4687         C.getPointerType(C.getPointerType(E->getType()))
4688             .withConst()
4689             .withRestrict(),
4690         ImplicitParamDecl::Other));
4691     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4692     PrivateVarsPos[VD] = Counter;
4693     ++Counter;
4694   }
4695   const auto &TaskPrivatesMapFnInfo =
4696       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4697   llvm::FunctionType *TaskPrivatesMapTy =
4698       CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo);
4699   std::string Name =
4700       CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""});
4701   auto *TaskPrivatesMap = llvm::Function::Create(
4702       TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name,
4703       &CGM.getModule());
4704   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap,
4705                                     TaskPrivatesMapFnInfo);
4706   if (CGM.getLangOpts().Optimize) {
4707     TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline);
4708     TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone);
4709     TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline);
4710   }
4711   CodeGenFunction CGF(CGM);
4712   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap,
4713                     TaskPrivatesMapFnInfo, Args, Loc, Loc);
4714 
4715   // *privi = &.privates.privi;
4716   LValue Base = CGF.EmitLoadOfPointerLValue(
4717       CGF.GetAddrOfLocalVar(&TaskPrivatesArg),
4718       TaskPrivatesArg.getType()->castAs<PointerType>());
4719   const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl());
4720   Counter = 0;
4721   for (const FieldDecl *Field : PrivatesQTyRD->fields()) {
4722     LValue FieldLVal = CGF.EmitLValueForField(Base, Field);
4723     const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]];
4724     LValue RefLVal =
4725         CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType());
4726     LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue(
4727         RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>());
4728     CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal);
4729     ++Counter;
4730   }
4731   CGF.FinishFunction();
4732   return TaskPrivatesMap;
4733 }
4734 
4735 /// Emit initialization for private variables in task-based directives.
4736 static void emitPrivatesInit(CodeGenFunction &CGF,
4737                              const OMPExecutableDirective &D,
4738                              Address KmpTaskSharedsPtr, LValue TDBase,
4739                              const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4740                              QualType SharedsTy, QualType SharedsPtrTy,
4741                              const OMPTaskDataTy &Data,
4742                              ArrayRef<PrivateDataTy> Privates, bool ForDup) {
4743   ASTContext &C = CGF.getContext();
4744   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4745   LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI);
4746   OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind())
4747                                  ? OMPD_taskloop
4748                                  : OMPD_task;
4749   const CapturedStmt &CS = *D.getCapturedStmt(Kind);
4750   CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS);
4751   LValue SrcBase;
4752   bool IsTargetTask =
4753       isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) ||
4754       isOpenMPTargetExecutionDirective(D.getDirectiveKind());
4755   // For target-based directives skip 3 firstprivate arrays BasePointersArray,
4756   // PointersArray and SizesArray. The original variables for these arrays are
4757   // not captured and we get their addresses explicitly.
4758   if ((!IsTargetTask && !Data.FirstprivateVars.empty()) ||
4759       (IsTargetTask && KmpTaskSharedsPtr.isValid())) {
4760     SrcBase = CGF.MakeAddrLValue(
4761         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4762             KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)),
4763         SharedsTy);
4764   }
4765   FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin();
4766   for (const PrivateDataTy &Pair : Privates) {
4767     const VarDecl *VD = Pair.second.PrivateCopy;
4768     const Expr *Init = VD->getAnyInitializer();
4769     if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) &&
4770                              !CGF.isTrivialInitializer(Init)))) {
4771       LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
4772       if (const VarDecl *Elem = Pair.second.PrivateElemInit) {
4773         const VarDecl *OriginalVD = Pair.second.Original;
4774         // Check if the variable is the target-based BasePointersArray,
4775         // PointersArray or SizesArray.
4776         LValue SharedRefLValue;
4777         QualType Type = PrivateLValue.getType();
4778         const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD);
4779         if (IsTargetTask && !SharedField) {
4780           assert(isa<ImplicitParamDecl>(OriginalVD) &&
4781                  isa<CapturedDecl>(OriginalVD->getDeclContext()) &&
4782                  cast<CapturedDecl>(OriginalVD->getDeclContext())
4783                          ->getNumParams() == 0 &&
4784                  isa<TranslationUnitDecl>(
4785                      cast<CapturedDecl>(OriginalVD->getDeclContext())
4786                          ->getDeclContext()) &&
4787                  "Expected artificial target data variable.");
4788           SharedRefLValue =
4789               CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type);
4790         } else {
4791           SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField);
4792           SharedRefLValue = CGF.MakeAddrLValue(
4793               Address(SharedRefLValue.getPointer(CGF),
4794                       C.getDeclAlign(OriginalVD)),
4795               SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl),
4796               SharedRefLValue.getTBAAInfo());
4797         }
4798         if (Type->isArrayType()) {
4799           // Initialize firstprivate array.
4800           if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) {
4801             // Perform simple memcpy.
4802             CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type);
4803           } else {
4804             // Initialize firstprivate array using element-by-element
4805             // initialization.
4806             CGF.EmitOMPAggregateAssign(
4807                 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF),
4808                 Type,
4809                 [&CGF, Elem, Init, &CapturesInfo](Address DestElement,
4810                                                   Address SrcElement) {
4811                   // Clean up any temporaries needed by the initialization.
4812                   CodeGenFunction::OMPPrivateScope InitScope(CGF);
4813                   InitScope.addPrivate(
4814                       Elem, [SrcElement]() -> Address { return SrcElement; });
4815                   (void)InitScope.Privatize();
4816                   // Emit initialization for single element.
4817                   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(
4818                       CGF, &CapturesInfo);
4819                   CGF.EmitAnyExprToMem(Init, DestElement,
4820                                        Init->getType().getQualifiers(),
4821                                        /*IsInitializer=*/false);
4822                 });
4823           }
4824         } else {
4825           CodeGenFunction::OMPPrivateScope InitScope(CGF);
4826           InitScope.addPrivate(Elem, [SharedRefLValue, &CGF]() -> Address {
4827             return SharedRefLValue.getAddress(CGF);
4828           });
4829           (void)InitScope.Privatize();
4830           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo);
4831           CGF.EmitExprAsInit(Init, VD, PrivateLValue,
4832                              /*capturedByInit=*/false);
4833         }
4834       } else {
4835         CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false);
4836       }
4837     }
4838     ++FI;
4839   }
4840 }
4841 
4842 /// Check if duplication function is required for taskloops.
4843 static bool checkInitIsRequired(CodeGenFunction &CGF,
4844                                 ArrayRef<PrivateDataTy> Privates) {
4845   bool InitRequired = false;
4846   for (const PrivateDataTy &Pair : Privates) {
4847     const VarDecl *VD = Pair.second.PrivateCopy;
4848     const Expr *Init = VD->getAnyInitializer();
4849     InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) &&
4850                                     !CGF.isTrivialInitializer(Init));
4851     if (InitRequired)
4852       break;
4853   }
4854   return InitRequired;
4855 }
4856 
4857 
4858 /// Emit task_dup function (for initialization of
4859 /// private/firstprivate/lastprivate vars and last_iter flag)
4860 /// \code
4861 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int
4862 /// lastpriv) {
4863 /// // setup lastprivate flag
4864 ///    task_dst->last = lastpriv;
4865 /// // could be constructor calls here...
4866 /// }
4867 /// \endcode
4868 static llvm::Value *
4869 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc,
4870                     const OMPExecutableDirective &D,
4871                     QualType KmpTaskTWithPrivatesPtrQTy,
4872                     const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4873                     const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy,
4874                     QualType SharedsPtrTy, const OMPTaskDataTy &Data,
4875                     ArrayRef<PrivateDataTy> Privates, bool WithLastIter) {
4876   ASTContext &C = CGM.getContext();
4877   FunctionArgList Args;
4878   ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4879                            KmpTaskTWithPrivatesPtrQTy,
4880                            ImplicitParamDecl::Other);
4881   ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4882                            KmpTaskTWithPrivatesPtrQTy,
4883                            ImplicitParamDecl::Other);
4884   ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy,
4885                                 ImplicitParamDecl::Other);
4886   Args.push_back(&DstArg);
4887   Args.push_back(&SrcArg);
4888   Args.push_back(&LastprivArg);
4889   const auto &TaskDupFnInfo =
4890       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4891   llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo);
4892   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""});
4893   auto *TaskDup = llvm::Function::Create(
4894       TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4895   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo);
4896   TaskDup->setDoesNotRecurse();
4897   CodeGenFunction CGF(CGM);
4898   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc,
4899                     Loc);
4900 
4901   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4902       CGF.GetAddrOfLocalVar(&DstArg),
4903       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4904   // task_dst->liter = lastpriv;
4905   if (WithLastIter) {
4906     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4907     LValue Base = CGF.EmitLValueForField(
4908         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4909     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4910     llvm::Value *Lastpriv = CGF.EmitLoadOfScalar(
4911         CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc);
4912     CGF.EmitStoreOfScalar(Lastpriv, LILVal);
4913   }
4914 
4915   // Emit initial values for private copies (if any).
4916   assert(!Privates.empty());
4917   Address KmpTaskSharedsPtr = Address::invalid();
4918   if (!Data.FirstprivateVars.empty()) {
4919     LValue TDBase = CGF.EmitLoadOfPointerLValue(
4920         CGF.GetAddrOfLocalVar(&SrcArg),
4921         KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4922     LValue Base = CGF.EmitLValueForField(
4923         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4924     KmpTaskSharedsPtr = Address(
4925         CGF.EmitLoadOfScalar(CGF.EmitLValueForField(
4926                                  Base, *std::next(KmpTaskTQTyRD->field_begin(),
4927                                                   KmpTaskTShareds)),
4928                              Loc),
4929         CGF.getNaturalTypeAlignment(SharedsTy));
4930   }
4931   emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD,
4932                    SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true);
4933   CGF.FinishFunction();
4934   return TaskDup;
4935 }
4936 
4937 /// Checks if destructor function is required to be generated.
4938 /// \return true if cleanups are required, false otherwise.
4939 static bool
4940 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) {
4941   bool NeedsCleanup = false;
4942   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
4943   const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl());
4944   for (const FieldDecl *FD : PrivateRD->fields()) {
4945     NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType();
4946     if (NeedsCleanup)
4947       break;
4948   }
4949   return NeedsCleanup;
4950 }
4951 
4952 CGOpenMPRuntime::TaskResultTy
4953 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc,
4954                               const OMPExecutableDirective &D,
4955                               llvm::Function *TaskFunction, QualType SharedsTy,
4956                               Address Shareds, const OMPTaskDataTy &Data) {
4957   ASTContext &C = CGM.getContext();
4958   llvm::SmallVector<PrivateDataTy, 4> Privates;
4959   // Aggregate privates and sort them by the alignment.
4960   auto I = Data.PrivateCopies.begin();
4961   for (const Expr *E : Data.PrivateVars) {
4962     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4963     Privates.emplace_back(
4964         C.getDeclAlign(VD),
4965         PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4966                          /*PrivateElemInit=*/nullptr));
4967     ++I;
4968   }
4969   I = Data.FirstprivateCopies.begin();
4970   auto IElemInitRef = Data.FirstprivateInits.begin();
4971   for (const Expr *E : Data.FirstprivateVars) {
4972     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4973     Privates.emplace_back(
4974         C.getDeclAlign(VD),
4975         PrivateHelpersTy(
4976             VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4977             cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl())));
4978     ++I;
4979     ++IElemInitRef;
4980   }
4981   I = Data.LastprivateCopies.begin();
4982   for (const Expr *E : Data.LastprivateVars) {
4983     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4984     Privates.emplace_back(
4985         C.getDeclAlign(VD),
4986         PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4987                          /*PrivateElemInit=*/nullptr));
4988     ++I;
4989   }
4990   llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) {
4991     return L.first > R.first;
4992   });
4993   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
4994   // Build type kmp_routine_entry_t (if not built yet).
4995   emitKmpRoutineEntryT(KmpInt32Ty);
4996   // Build type kmp_task_t (if not built yet).
4997   if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) {
4998     if (SavedKmpTaskloopTQTy.isNull()) {
4999       SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl(
5000           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
5001     }
5002     KmpTaskTQTy = SavedKmpTaskloopTQTy;
5003   } else {
5004     assert((D.getDirectiveKind() == OMPD_task ||
5005             isOpenMPTargetExecutionDirective(D.getDirectiveKind()) ||
5006             isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) &&
5007            "Expected taskloop, task or target directive");
5008     if (SavedKmpTaskTQTy.isNull()) {
5009       SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl(
5010           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
5011     }
5012     KmpTaskTQTy = SavedKmpTaskTQTy;
5013   }
5014   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
5015   // Build particular struct kmp_task_t for the given task.
5016   const RecordDecl *KmpTaskTWithPrivatesQTyRD =
5017       createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates);
5018   QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD);
5019   QualType KmpTaskTWithPrivatesPtrQTy =
5020       C.getPointerType(KmpTaskTWithPrivatesQTy);
5021   llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy);
5022   llvm::Type *KmpTaskTWithPrivatesPtrTy =
5023       KmpTaskTWithPrivatesTy->getPointerTo();
5024   llvm::Value *KmpTaskTWithPrivatesTySize =
5025       CGF.getTypeSize(KmpTaskTWithPrivatesQTy);
5026   QualType SharedsPtrTy = C.getPointerType(SharedsTy);
5027 
5028   // Emit initial values for private copies (if any).
5029   llvm::Value *TaskPrivatesMap = nullptr;
5030   llvm::Type *TaskPrivatesMapTy =
5031       std::next(TaskFunction->arg_begin(), 3)->getType();
5032   if (!Privates.empty()) {
5033     auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
5034     TaskPrivatesMap = emitTaskPrivateMappingFunction(
5035         CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars,
5036         FI->getType(), Privates);
5037     TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5038         TaskPrivatesMap, TaskPrivatesMapTy);
5039   } else {
5040     TaskPrivatesMap = llvm::ConstantPointerNull::get(
5041         cast<llvm::PointerType>(TaskPrivatesMapTy));
5042   }
5043   // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
5044   // kmp_task_t *tt);
5045   llvm::Function *TaskEntry = emitProxyTaskFunction(
5046       CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5047       KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction,
5048       TaskPrivatesMap);
5049 
5050   // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
5051   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
5052   // kmp_routine_entry_t *task_entry);
5053   // Task flags. Format is taken from
5054   // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h,
5055   // description of kmp_tasking_flags struct.
5056   enum {
5057     TiedFlag = 0x1,
5058     FinalFlag = 0x2,
5059     DestructorsFlag = 0x8,
5060     PriorityFlag = 0x20
5061   };
5062   unsigned Flags = Data.Tied ? TiedFlag : 0;
5063   bool NeedsCleanup = false;
5064   if (!Privates.empty()) {
5065     NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD);
5066     if (NeedsCleanup)
5067       Flags = Flags | DestructorsFlag;
5068   }
5069   if (Data.Priority.getInt())
5070     Flags = Flags | PriorityFlag;
5071   llvm::Value *TaskFlags =
5072       Data.Final.getPointer()
5073           ? CGF.Builder.CreateSelect(Data.Final.getPointer(),
5074                                      CGF.Builder.getInt32(FinalFlag),
5075                                      CGF.Builder.getInt32(/*C=*/0))
5076           : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0);
5077   TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
5078   llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy));
5079   SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc),
5080       getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize,
5081       SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5082           TaskEntry, KmpRoutineEntryPtrTy)};
5083   llvm::Value *NewTask;
5084   if (D.hasClausesOfKind<OMPNowaitClause>()) {
5085     // Check if we have any device clause associated with the directive.
5086     const Expr *Device = nullptr;
5087     if (auto *C = D.getSingleClause<OMPDeviceClause>())
5088       Device = C->getDevice();
5089     // Emit device ID if any otherwise use default value.
5090     llvm::Value *DeviceID;
5091     if (Device)
5092       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
5093                                            CGF.Int64Ty, /*isSigned=*/true);
5094     else
5095       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
5096     AllocArgs.push_back(DeviceID);
5097     NewTask = CGF.EmitRuntimeCall(
5098       createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs);
5099   } else {
5100     NewTask = CGF.EmitRuntimeCall(
5101       createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs);
5102   }
5103   llvm::Value *NewTaskNewTaskTTy =
5104       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5105           NewTask, KmpTaskTWithPrivatesPtrTy);
5106   LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy,
5107                                                KmpTaskTWithPrivatesQTy);
5108   LValue TDBase =
5109       CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
5110   // Fill the data in the resulting kmp_task_t record.
5111   // Copy shareds if there are any.
5112   Address KmpTaskSharedsPtr = Address::invalid();
5113   if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) {
5114     KmpTaskSharedsPtr =
5115         Address(CGF.EmitLoadOfScalar(
5116                     CGF.EmitLValueForField(
5117                         TDBase, *std::next(KmpTaskTQTyRD->field_begin(),
5118                                            KmpTaskTShareds)),
5119                     Loc),
5120                 CGF.getNaturalTypeAlignment(SharedsTy));
5121     LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy);
5122     LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy);
5123     CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap);
5124   }
5125   // Emit initial values for private copies (if any).
5126   TaskResultTy Result;
5127   if (!Privates.empty()) {
5128     emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD,
5129                      SharedsTy, SharedsPtrTy, Data, Privates,
5130                      /*ForDup=*/false);
5131     if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) &&
5132         (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) {
5133       Result.TaskDupFn = emitTaskDupFunction(
5134           CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD,
5135           KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates,
5136           /*WithLastIter=*/!Data.LastprivateVars.empty());
5137     }
5138   }
5139   // Fields of union "kmp_cmplrdata_t" for destructors and priority.
5140   enum { Priority = 0, Destructors = 1 };
5141   // Provide pointer to function with destructors for privates.
5142   auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1);
5143   const RecordDecl *KmpCmplrdataUD =
5144       (*FI)->getType()->getAsUnionType()->getDecl();
5145   if (NeedsCleanup) {
5146     llvm::Value *DestructorFn = emitDestructorsFunction(
5147         CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5148         KmpTaskTWithPrivatesQTy);
5149     LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI);
5150     LValue DestructorsLV = CGF.EmitLValueForField(
5151         Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors));
5152     CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5153                               DestructorFn, KmpRoutineEntryPtrTy),
5154                           DestructorsLV);
5155   }
5156   // Set priority.
5157   if (Data.Priority.getInt()) {
5158     LValue Data2LV = CGF.EmitLValueForField(
5159         TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2));
5160     LValue PriorityLV = CGF.EmitLValueForField(
5161         Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority));
5162     CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV);
5163   }
5164   Result.NewTask = NewTask;
5165   Result.TaskEntry = TaskEntry;
5166   Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy;
5167   Result.TDBase = TDBase;
5168   Result.KmpTaskTQTyRD = KmpTaskTQTyRD;
5169   return Result;
5170 }
5171 
5172 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
5173                                    const OMPExecutableDirective &D,
5174                                    llvm::Function *TaskFunction,
5175                                    QualType SharedsTy, Address Shareds,
5176                                    const Expr *IfCond,
5177                                    const OMPTaskDataTy &Data) {
5178   if (!CGF.HaveInsertPoint())
5179     return;
5180 
5181   TaskResultTy Result =
5182       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5183   llvm::Value *NewTask = Result.NewTask;
5184   llvm::Function *TaskEntry = Result.TaskEntry;
5185   llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy;
5186   LValue TDBase = Result.TDBase;
5187   const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD;
5188   ASTContext &C = CGM.getContext();
5189   // Process list of dependences.
5190   Address DependenciesArray = Address::invalid();
5191   unsigned NumDependencies = Data.Dependences.size();
5192   if (NumDependencies) {
5193     // Dependence kind for RTL.
5194     enum RTLDependenceKindTy { DepIn = 0x01, DepInOut = 0x3, DepMutexInOutSet = 0x4 };
5195     enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags };
5196     RecordDecl *KmpDependInfoRD;
5197     QualType FlagsTy =
5198         C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false);
5199     llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
5200     if (KmpDependInfoTy.isNull()) {
5201       KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info");
5202       KmpDependInfoRD->startDefinition();
5203       addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType());
5204       addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType());
5205       addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy);
5206       KmpDependInfoRD->completeDefinition();
5207       KmpDependInfoTy = C.getRecordType(KmpDependInfoRD);
5208     } else {
5209       KmpDependInfoRD = cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5210     }
5211     // Define type kmp_depend_info[<Dependences.size()>];
5212     QualType KmpDependInfoArrayTy = C.getConstantArrayType(
5213         KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies),
5214         nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
5215     // kmp_depend_info[<Dependences.size()>] deps;
5216     DependenciesArray =
5217         CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr");
5218     for (unsigned I = 0; I < NumDependencies; ++I) {
5219       const Expr *E = Data.Dependences[I].second;
5220       LValue Addr = CGF.EmitLValue(E);
5221       llvm::Value *Size;
5222       QualType Ty = E->getType();
5223       if (const auto *ASE =
5224               dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) {
5225         LValue UpAddrLVal =
5226             CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false);
5227         llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32(
5228             UpAddrLVal.getPointer(CGF), /*Idx0=*/1);
5229         llvm::Value *LowIntPtr =
5230             CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGM.SizeTy);
5231         llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy);
5232         Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr);
5233       } else {
5234         Size = CGF.getTypeSize(Ty);
5235       }
5236       LValue Base = CGF.MakeAddrLValue(
5237           CGF.Builder.CreateConstArrayGEP(DependenciesArray, I),
5238           KmpDependInfoTy);
5239       // deps[i].base_addr = &<Dependences[i].second>;
5240       LValue BaseAddrLVal = CGF.EmitLValueForField(
5241           Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5242       CGF.EmitStoreOfScalar(
5243           CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGF.IntPtrTy),
5244           BaseAddrLVal);
5245       // deps[i].len = sizeof(<Dependences[i].second>);
5246       LValue LenLVal = CGF.EmitLValueForField(
5247           Base, *std::next(KmpDependInfoRD->field_begin(), Len));
5248       CGF.EmitStoreOfScalar(Size, LenLVal);
5249       // deps[i].flags = <Dependences[i].first>;
5250       RTLDependenceKindTy DepKind;
5251       switch (Data.Dependences[I].first) {
5252       case OMPC_DEPEND_in:
5253         DepKind = DepIn;
5254         break;
5255       // Out and InOut dependencies must use the same code.
5256       case OMPC_DEPEND_out:
5257       case OMPC_DEPEND_inout:
5258         DepKind = DepInOut;
5259         break;
5260       case OMPC_DEPEND_mutexinoutset:
5261         DepKind = DepMutexInOutSet;
5262         break;
5263       case OMPC_DEPEND_source:
5264       case OMPC_DEPEND_sink:
5265       case OMPC_DEPEND_unknown:
5266         llvm_unreachable("Unknown task dependence type");
5267       }
5268       LValue FlagsLVal = CGF.EmitLValueForField(
5269           Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
5270       CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
5271                             FlagsLVal);
5272     }
5273     DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5274         CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0), CGF.VoidPtrTy);
5275   }
5276 
5277   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5278   // libcall.
5279   // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
5280   // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
5281   // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence
5282   // list is not empty
5283   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5284   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5285   llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask };
5286   llvm::Value *DepTaskArgs[7];
5287   if (NumDependencies) {
5288     DepTaskArgs[0] = UpLoc;
5289     DepTaskArgs[1] = ThreadID;
5290     DepTaskArgs[2] = NewTask;
5291     DepTaskArgs[3] = CGF.Builder.getInt32(NumDependencies);
5292     DepTaskArgs[4] = DependenciesArray.getPointer();
5293     DepTaskArgs[5] = CGF.Builder.getInt32(0);
5294     DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5295   }
5296   auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, NumDependencies,
5297                         &TaskArgs,
5298                         &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) {
5299     if (!Data.Tied) {
5300       auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
5301       LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI);
5302       CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal);
5303     }
5304     if (NumDependencies) {
5305       CGF.EmitRuntimeCall(
5306           createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs);
5307     } else {
5308       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task),
5309                           TaskArgs);
5310     }
5311     // Check if parent region is untied and build return for untied task;
5312     if (auto *Region =
5313             dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
5314       Region->emitUntiedSwitch(CGF);
5315   };
5316 
5317   llvm::Value *DepWaitTaskArgs[6];
5318   if (NumDependencies) {
5319     DepWaitTaskArgs[0] = UpLoc;
5320     DepWaitTaskArgs[1] = ThreadID;
5321     DepWaitTaskArgs[2] = CGF.Builder.getInt32(NumDependencies);
5322     DepWaitTaskArgs[3] = DependenciesArray.getPointer();
5323     DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
5324     DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5325   }
5326   auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry,
5327                         NumDependencies, &DepWaitTaskArgs,
5328                         Loc](CodeGenFunction &CGF, PrePostActionTy &) {
5329     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5330     CodeGenFunction::RunCleanupsScope LocalScope(CGF);
5331     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
5332     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
5333     // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
5334     // is specified.
5335     if (NumDependencies)
5336       CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps),
5337                           DepWaitTaskArgs);
5338     // Call proxy_task_entry(gtid, new_task);
5339     auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy,
5340                       Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
5341       Action.Enter(CGF);
5342       llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
5343       CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry,
5344                                                           OutlinedFnArgs);
5345     };
5346 
5347     // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
5348     // kmp_task_t *new_task);
5349     // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
5350     // kmp_task_t *new_task);
5351     RegionCodeGenTy RCG(CodeGen);
5352     CommonActionTy Action(
5353         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs,
5354         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs);
5355     RCG.setAction(Action);
5356     RCG(CGF);
5357   };
5358 
5359   if (IfCond) {
5360     emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen);
5361   } else {
5362     RegionCodeGenTy ThenRCG(ThenCodeGen);
5363     ThenRCG(CGF);
5364   }
5365 }
5366 
5367 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc,
5368                                        const OMPLoopDirective &D,
5369                                        llvm::Function *TaskFunction,
5370                                        QualType SharedsTy, Address Shareds,
5371                                        const Expr *IfCond,
5372                                        const OMPTaskDataTy &Data) {
5373   if (!CGF.HaveInsertPoint())
5374     return;
5375   TaskResultTy Result =
5376       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5377   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5378   // libcall.
5379   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
5380   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
5381   // sched, kmp_uint64 grainsize, void *task_dup);
5382   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5383   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5384   llvm::Value *IfVal;
5385   if (IfCond) {
5386     IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy,
5387                                       /*isSigned=*/true);
5388   } else {
5389     IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1);
5390   }
5391 
5392   LValue LBLVal = CGF.EmitLValueForField(
5393       Result.TDBase,
5394       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound));
5395   const auto *LBVar =
5396       cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl());
5397   CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF),
5398                        LBLVal.getQuals(),
5399                        /*IsInitializer=*/true);
5400   LValue UBLVal = CGF.EmitLValueForField(
5401       Result.TDBase,
5402       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound));
5403   const auto *UBVar =
5404       cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl());
5405   CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF),
5406                        UBLVal.getQuals(),
5407                        /*IsInitializer=*/true);
5408   LValue StLVal = CGF.EmitLValueForField(
5409       Result.TDBase,
5410       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride));
5411   const auto *StVar =
5412       cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl());
5413   CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF),
5414                        StLVal.getQuals(),
5415                        /*IsInitializer=*/true);
5416   // Store reductions address.
5417   LValue RedLVal = CGF.EmitLValueForField(
5418       Result.TDBase,
5419       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions));
5420   if (Data.Reductions) {
5421     CGF.EmitStoreOfScalar(Data.Reductions, RedLVal);
5422   } else {
5423     CGF.EmitNullInitialization(RedLVal.getAddress(CGF),
5424                                CGF.getContext().VoidPtrTy);
5425   }
5426   enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 };
5427   llvm::Value *TaskArgs[] = {
5428       UpLoc,
5429       ThreadID,
5430       Result.NewTask,
5431       IfVal,
5432       LBLVal.getPointer(CGF),
5433       UBLVal.getPointer(CGF),
5434       CGF.EmitLoadOfScalar(StLVal, Loc),
5435       llvm::ConstantInt::getSigned(
5436           CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler
5437       llvm::ConstantInt::getSigned(
5438           CGF.IntTy, Data.Schedule.getPointer()
5439                          ? Data.Schedule.getInt() ? NumTasks : Grainsize
5440                          : NoSchedule),
5441       Data.Schedule.getPointer()
5442           ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty,
5443                                       /*isSigned=*/false)
5444           : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0),
5445       Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5446                              Result.TaskDupFn, CGF.VoidPtrTy)
5447                        : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)};
5448   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs);
5449 }
5450 
5451 /// Emit reduction operation for each element of array (required for
5452 /// array sections) LHS op = RHS.
5453 /// \param Type Type of array.
5454 /// \param LHSVar Variable on the left side of the reduction operation
5455 /// (references element of array in original variable).
5456 /// \param RHSVar Variable on the right side of the reduction operation
5457 /// (references element of array in original variable).
5458 /// \param RedOpGen Generator of reduction operation with use of LHSVar and
5459 /// RHSVar.
5460 static void EmitOMPAggregateReduction(
5461     CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar,
5462     const VarDecl *RHSVar,
5463     const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *,
5464                                   const Expr *, const Expr *)> &RedOpGen,
5465     const Expr *XExpr = nullptr, const Expr *EExpr = nullptr,
5466     const Expr *UpExpr = nullptr) {
5467   // Perform element-by-element initialization.
5468   QualType ElementTy;
5469   Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar);
5470   Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar);
5471 
5472   // Drill down to the base element type on both arrays.
5473   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
5474   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr);
5475 
5476   llvm::Value *RHSBegin = RHSAddr.getPointer();
5477   llvm::Value *LHSBegin = LHSAddr.getPointer();
5478   // Cast from pointer to array type to pointer to single element.
5479   llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements);
5480   // The basic structure here is a while-do loop.
5481   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body");
5482   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done");
5483   llvm::Value *IsEmpty =
5484       CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty");
5485   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5486 
5487   // Enter the loop body, making that address the current address.
5488   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5489   CGF.EmitBlock(BodyBB);
5490 
5491   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
5492 
5493   llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI(
5494       RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast");
5495   RHSElementPHI->addIncoming(RHSBegin, EntryBB);
5496   Address RHSElementCurrent =
5497       Address(RHSElementPHI,
5498               RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5499 
5500   llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI(
5501       LHSBegin->getType(), 2, "omp.arraycpy.destElementPast");
5502   LHSElementPHI->addIncoming(LHSBegin, EntryBB);
5503   Address LHSElementCurrent =
5504       Address(LHSElementPHI,
5505               LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5506 
5507   // Emit copy.
5508   CodeGenFunction::OMPPrivateScope Scope(CGF);
5509   Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; });
5510   Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; });
5511   Scope.Privatize();
5512   RedOpGen(CGF, XExpr, EExpr, UpExpr);
5513   Scope.ForceCleanup();
5514 
5515   // Shift the address forward by one element.
5516   llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32(
5517       LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
5518   llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32(
5519       RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element");
5520   // Check whether we've reached the end.
5521   llvm::Value *Done =
5522       CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done");
5523   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
5524   LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock());
5525   RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock());
5526 
5527   // Done.
5528   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5529 }
5530 
5531 /// Emit reduction combiner. If the combiner is a simple expression emit it as
5532 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of
5533 /// UDR combiner function.
5534 static void emitReductionCombiner(CodeGenFunction &CGF,
5535                                   const Expr *ReductionOp) {
5536   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
5537     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
5538       if (const auto *DRE =
5539               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
5540         if (const auto *DRD =
5541                 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) {
5542           std::pair<llvm::Function *, llvm::Function *> Reduction =
5543               CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
5544           RValue Func = RValue::get(Reduction.first);
5545           CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
5546           CGF.EmitIgnoredExpr(ReductionOp);
5547           return;
5548         }
5549   CGF.EmitIgnoredExpr(ReductionOp);
5550 }
5551 
5552 llvm::Function *CGOpenMPRuntime::emitReductionFunction(
5553     SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates,
5554     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
5555     ArrayRef<const Expr *> ReductionOps) {
5556   ASTContext &C = CGM.getContext();
5557 
5558   // void reduction_func(void *LHSArg, void *RHSArg);
5559   FunctionArgList Args;
5560   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5561                            ImplicitParamDecl::Other);
5562   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5563                            ImplicitParamDecl::Other);
5564   Args.push_back(&LHSArg);
5565   Args.push_back(&RHSArg);
5566   const auto &CGFI =
5567       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5568   std::string Name = getName({"omp", "reduction", "reduction_func"});
5569   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
5570                                     llvm::GlobalValue::InternalLinkage, Name,
5571                                     &CGM.getModule());
5572   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
5573   Fn->setDoesNotRecurse();
5574   CodeGenFunction CGF(CGM);
5575   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
5576 
5577   // Dst = (void*[n])(LHSArg);
5578   // Src = (void*[n])(RHSArg);
5579   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5580       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
5581       ArgsType), CGF.getPointerAlign());
5582   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5583       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
5584       ArgsType), CGF.getPointerAlign());
5585 
5586   //  ...
5587   //  *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
5588   //  ...
5589   CodeGenFunction::OMPPrivateScope Scope(CGF);
5590   auto IPriv = Privates.begin();
5591   unsigned Idx = 0;
5592   for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) {
5593     const auto *RHSVar =
5594         cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl());
5595     Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() {
5596       return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar);
5597     });
5598     const auto *LHSVar =
5599         cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl());
5600     Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() {
5601       return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar);
5602     });
5603     QualType PrivTy = (*IPriv)->getType();
5604     if (PrivTy->isVariablyModifiedType()) {
5605       // Get array size and emit VLA type.
5606       ++Idx;
5607       Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx);
5608       llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem);
5609       const VariableArrayType *VLA =
5610           CGF.getContext().getAsVariableArrayType(PrivTy);
5611       const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr());
5612       CodeGenFunction::OpaqueValueMapping OpaqueMap(
5613           CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy)));
5614       CGF.EmitVariablyModifiedType(PrivTy);
5615     }
5616   }
5617   Scope.Privatize();
5618   IPriv = Privates.begin();
5619   auto ILHS = LHSExprs.begin();
5620   auto IRHS = RHSExprs.begin();
5621   for (const Expr *E : ReductionOps) {
5622     if ((*IPriv)->getType()->isArrayType()) {
5623       // Emit reduction for array section.
5624       const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5625       const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5626       EmitOMPAggregateReduction(
5627           CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5628           [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5629             emitReductionCombiner(CGF, E);
5630           });
5631     } else {
5632       // Emit reduction for array subscript or single variable.
5633       emitReductionCombiner(CGF, E);
5634     }
5635     ++IPriv;
5636     ++ILHS;
5637     ++IRHS;
5638   }
5639   Scope.ForceCleanup();
5640   CGF.FinishFunction();
5641   return Fn;
5642 }
5643 
5644 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF,
5645                                                   const Expr *ReductionOp,
5646                                                   const Expr *PrivateRef,
5647                                                   const DeclRefExpr *LHS,
5648                                                   const DeclRefExpr *RHS) {
5649   if (PrivateRef->getType()->isArrayType()) {
5650     // Emit reduction for array section.
5651     const auto *LHSVar = cast<VarDecl>(LHS->getDecl());
5652     const auto *RHSVar = cast<VarDecl>(RHS->getDecl());
5653     EmitOMPAggregateReduction(
5654         CGF, PrivateRef->getType(), LHSVar, RHSVar,
5655         [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5656           emitReductionCombiner(CGF, ReductionOp);
5657         });
5658   } else {
5659     // Emit reduction for array subscript or single variable.
5660     emitReductionCombiner(CGF, ReductionOp);
5661   }
5662 }
5663 
5664 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc,
5665                                     ArrayRef<const Expr *> Privates,
5666                                     ArrayRef<const Expr *> LHSExprs,
5667                                     ArrayRef<const Expr *> RHSExprs,
5668                                     ArrayRef<const Expr *> ReductionOps,
5669                                     ReductionOptionsTy Options) {
5670   if (!CGF.HaveInsertPoint())
5671     return;
5672 
5673   bool WithNowait = Options.WithNowait;
5674   bool SimpleReduction = Options.SimpleReduction;
5675 
5676   // Next code should be emitted for reduction:
5677   //
5678   // static kmp_critical_name lock = { 0 };
5679   //
5680   // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
5681   //  *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
5682   //  ...
5683   //  *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
5684   //  *(Type<n>-1*)rhs[<n>-1]);
5685   // }
5686   //
5687   // ...
5688   // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
5689   // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5690   // RedList, reduce_func, &<lock>)) {
5691   // case 1:
5692   //  ...
5693   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5694   //  ...
5695   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5696   // break;
5697   // case 2:
5698   //  ...
5699   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5700   //  ...
5701   // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
5702   // break;
5703   // default:;
5704   // }
5705   //
5706   // if SimpleReduction is true, only the next code is generated:
5707   //  ...
5708   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5709   //  ...
5710 
5711   ASTContext &C = CGM.getContext();
5712 
5713   if (SimpleReduction) {
5714     CodeGenFunction::RunCleanupsScope Scope(CGF);
5715     auto IPriv = Privates.begin();
5716     auto ILHS = LHSExprs.begin();
5717     auto IRHS = RHSExprs.begin();
5718     for (const Expr *E : ReductionOps) {
5719       emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5720                                   cast<DeclRefExpr>(*IRHS));
5721       ++IPriv;
5722       ++ILHS;
5723       ++IRHS;
5724     }
5725     return;
5726   }
5727 
5728   // 1. Build a list of reduction variables.
5729   // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
5730   auto Size = RHSExprs.size();
5731   for (const Expr *E : Privates) {
5732     if (E->getType()->isVariablyModifiedType())
5733       // Reserve place for array size.
5734       ++Size;
5735   }
5736   llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size);
5737   QualType ReductionArrayTy =
5738       C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
5739                              /*IndexTypeQuals=*/0);
5740   Address ReductionList =
5741       CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list");
5742   auto IPriv = Privates.begin();
5743   unsigned Idx = 0;
5744   for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) {
5745     Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5746     CGF.Builder.CreateStore(
5747         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5748             CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy),
5749         Elem);
5750     if ((*IPriv)->getType()->isVariablyModifiedType()) {
5751       // Store array size.
5752       ++Idx;
5753       Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5754       llvm::Value *Size = CGF.Builder.CreateIntCast(
5755           CGF.getVLASize(
5756                  CGF.getContext().getAsVariableArrayType((*IPriv)->getType()))
5757               .NumElts,
5758           CGF.SizeTy, /*isSigned=*/false);
5759       CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy),
5760                               Elem);
5761     }
5762   }
5763 
5764   // 2. Emit reduce_func().
5765   llvm::Function *ReductionFn = emitReductionFunction(
5766       Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates,
5767       LHSExprs, RHSExprs, ReductionOps);
5768 
5769   // 3. Create static kmp_critical_name lock = { 0 };
5770   std::string Name = getName({"reduction"});
5771   llvm::Value *Lock = getCriticalRegionLock(Name);
5772 
5773   // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5774   // RedList, reduce_func, &<lock>);
5775   llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE);
5776   llvm::Value *ThreadId = getThreadID(CGF, Loc);
5777   llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy);
5778   llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5779       ReductionList.getPointer(), CGF.VoidPtrTy);
5780   llvm::Value *Args[] = {
5781       IdentTLoc,                             // ident_t *<loc>
5782       ThreadId,                              // i32 <gtid>
5783       CGF.Builder.getInt32(RHSExprs.size()), // i32 <n>
5784       ReductionArrayTySize,                  // size_type sizeof(RedList)
5785       RL,                                    // void *RedList
5786       ReductionFn, // void (*) (void *, void *) <reduce_func>
5787       Lock         // kmp_critical_name *&<lock>
5788   };
5789   llvm::Value *Res = CGF.EmitRuntimeCall(
5790       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait
5791                                        : OMPRTL__kmpc_reduce),
5792       Args);
5793 
5794   // 5. Build switch(res)
5795   llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default");
5796   llvm::SwitchInst *SwInst =
5797       CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2);
5798 
5799   // 6. Build case 1:
5800   //  ...
5801   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5802   //  ...
5803   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5804   // break;
5805   llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1");
5806   SwInst->addCase(CGF.Builder.getInt32(1), Case1BB);
5807   CGF.EmitBlock(Case1BB);
5808 
5809   // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5810   llvm::Value *EndArgs[] = {
5811       IdentTLoc, // ident_t *<loc>
5812       ThreadId,  // i32 <gtid>
5813       Lock       // kmp_critical_name *&<lock>
5814   };
5815   auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps](
5816                        CodeGenFunction &CGF, PrePostActionTy &Action) {
5817     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5818     auto IPriv = Privates.begin();
5819     auto ILHS = LHSExprs.begin();
5820     auto IRHS = RHSExprs.begin();
5821     for (const Expr *E : ReductionOps) {
5822       RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5823                                      cast<DeclRefExpr>(*IRHS));
5824       ++IPriv;
5825       ++ILHS;
5826       ++IRHS;
5827     }
5828   };
5829   RegionCodeGenTy RCG(CodeGen);
5830   CommonActionTy Action(
5831       nullptr, llvm::None,
5832       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait
5833                                        : OMPRTL__kmpc_end_reduce),
5834       EndArgs);
5835   RCG.setAction(Action);
5836   RCG(CGF);
5837 
5838   CGF.EmitBranch(DefaultBB);
5839 
5840   // 7. Build case 2:
5841   //  ...
5842   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5843   //  ...
5844   // break;
5845   llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2");
5846   SwInst->addCase(CGF.Builder.getInt32(2), Case2BB);
5847   CGF.EmitBlock(Case2BB);
5848 
5849   auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps](
5850                              CodeGenFunction &CGF, PrePostActionTy &Action) {
5851     auto ILHS = LHSExprs.begin();
5852     auto IRHS = RHSExprs.begin();
5853     auto IPriv = Privates.begin();
5854     for (const Expr *E : ReductionOps) {
5855       const Expr *XExpr = nullptr;
5856       const Expr *EExpr = nullptr;
5857       const Expr *UpExpr = nullptr;
5858       BinaryOperatorKind BO = BO_Comma;
5859       if (const auto *BO = dyn_cast<BinaryOperator>(E)) {
5860         if (BO->getOpcode() == BO_Assign) {
5861           XExpr = BO->getLHS();
5862           UpExpr = BO->getRHS();
5863         }
5864       }
5865       // Try to emit update expression as a simple atomic.
5866       const Expr *RHSExpr = UpExpr;
5867       if (RHSExpr) {
5868         // Analyze RHS part of the whole expression.
5869         if (const auto *ACO = dyn_cast<AbstractConditionalOperator>(
5870                 RHSExpr->IgnoreParenImpCasts())) {
5871           // If this is a conditional operator, analyze its condition for
5872           // min/max reduction operator.
5873           RHSExpr = ACO->getCond();
5874         }
5875         if (const auto *BORHS =
5876                 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) {
5877           EExpr = BORHS->getRHS();
5878           BO = BORHS->getOpcode();
5879         }
5880       }
5881       if (XExpr) {
5882         const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5883         auto &&AtomicRedGen = [BO, VD,
5884                                Loc](CodeGenFunction &CGF, const Expr *XExpr,
5885                                     const Expr *EExpr, const Expr *UpExpr) {
5886           LValue X = CGF.EmitLValue(XExpr);
5887           RValue E;
5888           if (EExpr)
5889             E = CGF.EmitAnyExpr(EExpr);
5890           CGF.EmitOMPAtomicSimpleUpdateExpr(
5891               X, E, BO, /*IsXLHSInRHSPart=*/true,
5892               llvm::AtomicOrdering::Monotonic, Loc,
5893               [&CGF, UpExpr, VD, Loc](RValue XRValue) {
5894                 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
5895                 PrivateScope.addPrivate(
5896                     VD, [&CGF, VD, XRValue, Loc]() {
5897                       Address LHSTemp = CGF.CreateMemTemp(VD->getType());
5898                       CGF.emitOMPSimpleStore(
5899                           CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue,
5900                           VD->getType().getNonReferenceType(), Loc);
5901                       return LHSTemp;
5902                     });
5903                 (void)PrivateScope.Privatize();
5904                 return CGF.EmitAnyExpr(UpExpr);
5905               });
5906         };
5907         if ((*IPriv)->getType()->isArrayType()) {
5908           // Emit atomic reduction for array section.
5909           const auto *RHSVar =
5910               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5911           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar,
5912                                     AtomicRedGen, XExpr, EExpr, UpExpr);
5913         } else {
5914           // Emit atomic reduction for array subscript or single variable.
5915           AtomicRedGen(CGF, XExpr, EExpr, UpExpr);
5916         }
5917       } else {
5918         // Emit as a critical region.
5919         auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *,
5920                                            const Expr *, const Expr *) {
5921           CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5922           std::string Name = RT.getName({"atomic_reduction"});
5923           RT.emitCriticalRegion(
5924               CGF, Name,
5925               [=](CodeGenFunction &CGF, PrePostActionTy &Action) {
5926                 Action.Enter(CGF);
5927                 emitReductionCombiner(CGF, E);
5928               },
5929               Loc);
5930         };
5931         if ((*IPriv)->getType()->isArrayType()) {
5932           const auto *LHSVar =
5933               cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5934           const auto *RHSVar =
5935               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5936           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5937                                     CritRedGen);
5938         } else {
5939           CritRedGen(CGF, nullptr, nullptr, nullptr);
5940         }
5941       }
5942       ++ILHS;
5943       ++IRHS;
5944       ++IPriv;
5945     }
5946   };
5947   RegionCodeGenTy AtomicRCG(AtomicCodeGen);
5948   if (!WithNowait) {
5949     // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
5950     llvm::Value *EndArgs[] = {
5951         IdentTLoc, // ident_t *<loc>
5952         ThreadId,  // i32 <gtid>
5953         Lock       // kmp_critical_name *&<lock>
5954     };
5955     CommonActionTy Action(nullptr, llvm::None,
5956                           createRuntimeFunction(OMPRTL__kmpc_end_reduce),
5957                           EndArgs);
5958     AtomicRCG.setAction(Action);
5959     AtomicRCG(CGF);
5960   } else {
5961     AtomicRCG(CGF);
5962   }
5963 
5964   CGF.EmitBranch(DefaultBB);
5965   CGF.EmitBlock(DefaultBB, /*IsFinished=*/true);
5966 }
5967 
5968 /// Generates unique name for artificial threadprivate variables.
5969 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>"
5970 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix,
5971                                       const Expr *Ref) {
5972   SmallString<256> Buffer;
5973   llvm::raw_svector_ostream Out(Buffer);
5974   const clang::DeclRefExpr *DE;
5975   const VarDecl *D = ::getBaseDecl(Ref, DE);
5976   if (!D)
5977     D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl());
5978   D = D->getCanonicalDecl();
5979   std::string Name = CGM.getOpenMPRuntime().getName(
5980       {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)});
5981   Out << Prefix << Name << "_"
5982       << D->getCanonicalDecl()->getBeginLoc().getRawEncoding();
5983   return std::string(Out.str());
5984 }
5985 
5986 /// Emits reduction initializer function:
5987 /// \code
5988 /// void @.red_init(void* %arg) {
5989 /// %0 = bitcast void* %arg to <type>*
5990 /// store <type> <init>, <type>* %0
5991 /// ret void
5992 /// }
5993 /// \endcode
5994 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM,
5995                                            SourceLocation Loc,
5996                                            ReductionCodeGen &RCG, unsigned N) {
5997   ASTContext &C = CGM.getContext();
5998   FunctionArgList Args;
5999   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6000                           ImplicitParamDecl::Other);
6001   Args.emplace_back(&Param);
6002   const auto &FnInfo =
6003       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6004   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6005   std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""});
6006   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6007                                     Name, &CGM.getModule());
6008   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6009   Fn->setDoesNotRecurse();
6010   CodeGenFunction CGF(CGM);
6011   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6012   Address PrivateAddr = CGF.EmitLoadOfPointer(
6013       CGF.GetAddrOfLocalVar(&Param),
6014       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6015   llvm::Value *Size = nullptr;
6016   // If the size of the reduction item is non-constant, load it from global
6017   // threadprivate variable.
6018   if (RCG.getSizes(N).second) {
6019     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6020         CGF, CGM.getContext().getSizeType(),
6021         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6022     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6023                                 CGM.getContext().getSizeType(), Loc);
6024   }
6025   RCG.emitAggregateType(CGF, N, Size);
6026   LValue SharedLVal;
6027   // If initializer uses initializer from declare reduction construct, emit a
6028   // pointer to the address of the original reduction item (reuired by reduction
6029   // initializer)
6030   if (RCG.usesReductionInitializer(N)) {
6031     Address SharedAddr =
6032         CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6033             CGF, CGM.getContext().VoidPtrTy,
6034             generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6035     SharedAddr = CGF.EmitLoadOfPointer(
6036         SharedAddr,
6037         CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr());
6038     SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy);
6039   } else {
6040     SharedLVal = CGF.MakeNaturalAlignAddrLValue(
6041         llvm::ConstantPointerNull::get(CGM.VoidPtrTy),
6042         CGM.getContext().VoidPtrTy);
6043   }
6044   // Emit the initializer:
6045   // %0 = bitcast void* %arg to <type>*
6046   // store <type> <init>, <type>* %0
6047   RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal,
6048                          [](CodeGenFunction &) { return false; });
6049   CGF.FinishFunction();
6050   return Fn;
6051 }
6052 
6053 /// Emits reduction combiner function:
6054 /// \code
6055 /// void @.red_comb(void* %arg0, void* %arg1) {
6056 /// %lhs = bitcast void* %arg0 to <type>*
6057 /// %rhs = bitcast void* %arg1 to <type>*
6058 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs)
6059 /// store <type> %2, <type>* %lhs
6060 /// ret void
6061 /// }
6062 /// \endcode
6063 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM,
6064                                            SourceLocation Loc,
6065                                            ReductionCodeGen &RCG, unsigned N,
6066                                            const Expr *ReductionOp,
6067                                            const Expr *LHS, const Expr *RHS,
6068                                            const Expr *PrivateRef) {
6069   ASTContext &C = CGM.getContext();
6070   const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl());
6071   const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl());
6072   FunctionArgList Args;
6073   ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
6074                                C.VoidPtrTy, ImplicitParamDecl::Other);
6075   ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6076                             ImplicitParamDecl::Other);
6077   Args.emplace_back(&ParamInOut);
6078   Args.emplace_back(&ParamIn);
6079   const auto &FnInfo =
6080       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6081   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6082   std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""});
6083   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6084                                     Name, &CGM.getModule());
6085   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6086   Fn->setDoesNotRecurse();
6087   CodeGenFunction CGF(CGM);
6088   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6089   llvm::Value *Size = nullptr;
6090   // If the size of the reduction item is non-constant, load it from global
6091   // threadprivate variable.
6092   if (RCG.getSizes(N).second) {
6093     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6094         CGF, CGM.getContext().getSizeType(),
6095         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6096     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6097                                 CGM.getContext().getSizeType(), Loc);
6098   }
6099   RCG.emitAggregateType(CGF, N, Size);
6100   // Remap lhs and rhs variables to the addresses of the function arguments.
6101   // %lhs = bitcast void* %arg0 to <type>*
6102   // %rhs = bitcast void* %arg1 to <type>*
6103   CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
6104   PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() {
6105     // Pull out the pointer to the variable.
6106     Address PtrAddr = CGF.EmitLoadOfPointer(
6107         CGF.GetAddrOfLocalVar(&ParamInOut),
6108         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6109     return CGF.Builder.CreateElementBitCast(
6110         PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType()));
6111   });
6112   PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() {
6113     // Pull out the pointer to the variable.
6114     Address PtrAddr = CGF.EmitLoadOfPointer(
6115         CGF.GetAddrOfLocalVar(&ParamIn),
6116         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6117     return CGF.Builder.CreateElementBitCast(
6118         PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType()));
6119   });
6120   PrivateScope.Privatize();
6121   // Emit the combiner body:
6122   // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs)
6123   // store <type> %2, <type>* %lhs
6124   CGM.getOpenMPRuntime().emitSingleReductionCombiner(
6125       CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS),
6126       cast<DeclRefExpr>(RHS));
6127   CGF.FinishFunction();
6128   return Fn;
6129 }
6130 
6131 /// Emits reduction finalizer function:
6132 /// \code
6133 /// void @.red_fini(void* %arg) {
6134 /// %0 = bitcast void* %arg to <type>*
6135 /// <destroy>(<type>* %0)
6136 /// ret void
6137 /// }
6138 /// \endcode
6139 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM,
6140                                            SourceLocation Loc,
6141                                            ReductionCodeGen &RCG, unsigned N) {
6142   if (!RCG.needCleanups(N))
6143     return nullptr;
6144   ASTContext &C = CGM.getContext();
6145   FunctionArgList Args;
6146   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6147                           ImplicitParamDecl::Other);
6148   Args.emplace_back(&Param);
6149   const auto &FnInfo =
6150       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6151   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6152   std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""});
6153   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6154                                     Name, &CGM.getModule());
6155   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6156   Fn->setDoesNotRecurse();
6157   CodeGenFunction CGF(CGM);
6158   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6159   Address PrivateAddr = CGF.EmitLoadOfPointer(
6160       CGF.GetAddrOfLocalVar(&Param),
6161       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6162   llvm::Value *Size = nullptr;
6163   // If the size of the reduction item is non-constant, load it from global
6164   // threadprivate variable.
6165   if (RCG.getSizes(N).second) {
6166     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6167         CGF, CGM.getContext().getSizeType(),
6168         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6169     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6170                                 CGM.getContext().getSizeType(), Loc);
6171   }
6172   RCG.emitAggregateType(CGF, N, Size);
6173   // Emit the finalizer body:
6174   // <destroy>(<type>* %0)
6175   RCG.emitCleanups(CGF, N, PrivateAddr);
6176   CGF.FinishFunction(Loc);
6177   return Fn;
6178 }
6179 
6180 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit(
6181     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
6182     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
6183   if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty())
6184     return nullptr;
6185 
6186   // Build typedef struct:
6187   // kmp_task_red_input {
6188   //   void *reduce_shar; // shared reduction item
6189   //   size_t reduce_size; // size of data item
6190   //   void *reduce_init; // data initialization routine
6191   //   void *reduce_fini; // data finalization routine
6192   //   void *reduce_comb; // data combiner routine
6193   //   kmp_task_red_flags_t flags; // flags for additional info from compiler
6194   // } kmp_task_red_input_t;
6195   ASTContext &C = CGM.getContext();
6196   RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t");
6197   RD->startDefinition();
6198   const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6199   const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType());
6200   const FieldDecl *InitFD  = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6201   const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6202   const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6203   const FieldDecl *FlagsFD = addFieldToRecordDecl(
6204       C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false));
6205   RD->completeDefinition();
6206   QualType RDType = C.getRecordType(RD);
6207   unsigned Size = Data.ReductionVars.size();
6208   llvm::APInt ArraySize(/*numBits=*/64, Size);
6209   QualType ArrayRDType = C.getConstantArrayType(
6210       RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
6211   // kmp_task_red_input_t .rd_input.[Size];
6212   Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input.");
6213   ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies,
6214                        Data.ReductionOps);
6215   for (unsigned Cnt = 0; Cnt < Size; ++Cnt) {
6216     // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt];
6217     llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0),
6218                            llvm::ConstantInt::get(CGM.SizeTy, Cnt)};
6219     llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP(
6220         TaskRedInput.getPointer(), Idxs,
6221         /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc,
6222         ".rd_input.gep.");
6223     LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType);
6224     // ElemLVal.reduce_shar = &Shareds[Cnt];
6225     LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD);
6226     RCG.emitSharedLValue(CGF, Cnt);
6227     llvm::Value *CastedShared =
6228         CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF));
6229     CGF.EmitStoreOfScalar(CastedShared, SharedLVal);
6230     RCG.emitAggregateType(CGF, Cnt);
6231     llvm::Value *SizeValInChars;
6232     llvm::Value *SizeVal;
6233     std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt);
6234     // We use delayed creation/initialization for VLAs, array sections and
6235     // custom reduction initializations. It is required because runtime does not
6236     // provide the way to pass the sizes of VLAs/array sections to
6237     // initializer/combiner/finalizer functions and does not pass the pointer to
6238     // original reduction item to the initializer. Instead threadprivate global
6239     // variables are used to store these values and use them in the functions.
6240     bool DelayedCreation = !!SizeVal;
6241     SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy,
6242                                                /*isSigned=*/false);
6243     LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD);
6244     CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal);
6245     // ElemLVal.reduce_init = init;
6246     LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD);
6247     llvm::Value *InitAddr =
6248         CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt));
6249     CGF.EmitStoreOfScalar(InitAddr, InitLVal);
6250     DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt);
6251     // ElemLVal.reduce_fini = fini;
6252     LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD);
6253     llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt);
6254     llvm::Value *FiniAddr = Fini
6255                                 ? CGF.EmitCastToVoidPtr(Fini)
6256                                 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy);
6257     CGF.EmitStoreOfScalar(FiniAddr, FiniLVal);
6258     // ElemLVal.reduce_comb = comb;
6259     LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD);
6260     llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction(
6261         CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt],
6262         RHSExprs[Cnt], Data.ReductionCopies[Cnt]));
6263     CGF.EmitStoreOfScalar(CombAddr, CombLVal);
6264     // ElemLVal.flags = 0;
6265     LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD);
6266     if (DelayedCreation) {
6267       CGF.EmitStoreOfScalar(
6268           llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true),
6269           FlagsLVal);
6270     } else
6271       CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF),
6272                                  FlagsLVal.getType());
6273   }
6274   // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void
6275   // *data);
6276   llvm::Value *Args[] = {
6277       CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy,
6278                                 /*isSigned=*/true),
6279       llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6280       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(),
6281                                                       CGM.VoidPtrTy)};
6282   return CGF.EmitRuntimeCall(
6283       createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args);
6284 }
6285 
6286 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
6287                                               SourceLocation Loc,
6288                                               ReductionCodeGen &RCG,
6289                                               unsigned N) {
6290   auto Sizes = RCG.getSizes(N);
6291   // Emit threadprivate global variable if the type is non-constant
6292   // (Sizes.second = nullptr).
6293   if (Sizes.second) {
6294     llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy,
6295                                                      /*isSigned=*/false);
6296     Address SizeAddr = getAddrOfArtificialThreadPrivate(
6297         CGF, CGM.getContext().getSizeType(),
6298         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6299     CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false);
6300   }
6301   // Store address of the original reduction item if custom initializer is used.
6302   if (RCG.usesReductionInitializer(N)) {
6303     Address SharedAddr = getAddrOfArtificialThreadPrivate(
6304         CGF, CGM.getContext().VoidPtrTy,
6305         generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6306     CGF.Builder.CreateStore(
6307         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6308             RCG.getSharedLValue(N).getPointer(CGF), CGM.VoidPtrTy),
6309         SharedAddr, /*IsVolatile=*/false);
6310   }
6311 }
6312 
6313 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF,
6314                                               SourceLocation Loc,
6315                                               llvm::Value *ReductionsPtr,
6316                                               LValue SharedLVal) {
6317   // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
6318   // *d);
6319   llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6320                                                    CGM.IntTy,
6321                                                    /*isSigned=*/true),
6322                          ReductionsPtr,
6323                          CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6324                              SharedLVal.getPointer(CGF), CGM.VoidPtrTy)};
6325   return Address(
6326       CGF.EmitRuntimeCall(
6327           createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args),
6328       SharedLVal.getAlignment());
6329 }
6330 
6331 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
6332                                        SourceLocation Loc) {
6333   if (!CGF.HaveInsertPoint())
6334     return;
6335   // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
6336   // global_tid);
6337   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
6338   // Ignore return result until untied tasks are supported.
6339   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args);
6340   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
6341     Region->emitUntiedSwitch(CGF);
6342 }
6343 
6344 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF,
6345                                            OpenMPDirectiveKind InnerKind,
6346                                            const RegionCodeGenTy &CodeGen,
6347                                            bool HasCancel) {
6348   if (!CGF.HaveInsertPoint())
6349     return;
6350   InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel);
6351   CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
6352 }
6353 
6354 namespace {
6355 enum RTCancelKind {
6356   CancelNoreq = 0,
6357   CancelParallel = 1,
6358   CancelLoop = 2,
6359   CancelSections = 3,
6360   CancelTaskgroup = 4
6361 };
6362 } // anonymous namespace
6363 
6364 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) {
6365   RTCancelKind CancelKind = CancelNoreq;
6366   if (CancelRegion == OMPD_parallel)
6367     CancelKind = CancelParallel;
6368   else if (CancelRegion == OMPD_for)
6369     CancelKind = CancelLoop;
6370   else if (CancelRegion == OMPD_sections)
6371     CancelKind = CancelSections;
6372   else {
6373     assert(CancelRegion == OMPD_taskgroup);
6374     CancelKind = CancelTaskgroup;
6375   }
6376   return CancelKind;
6377 }
6378 
6379 void CGOpenMPRuntime::emitCancellationPointCall(
6380     CodeGenFunction &CGF, SourceLocation Loc,
6381     OpenMPDirectiveKind CancelRegion) {
6382   if (!CGF.HaveInsertPoint())
6383     return;
6384   // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
6385   // global_tid, kmp_int32 cncl_kind);
6386   if (auto *OMPRegionInfo =
6387           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6388     // For 'cancellation point taskgroup', the task region info may not have a
6389     // cancel. This may instead happen in another adjacent task.
6390     if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) {
6391       llvm::Value *Args[] = {
6392           emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
6393           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6394       // Ignore return result until untied tasks are supported.
6395       llvm::Value *Result = CGF.EmitRuntimeCall(
6396           createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args);
6397       // if (__kmpc_cancellationpoint()) {
6398       //   exit from construct;
6399       // }
6400       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6401       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6402       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6403       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6404       CGF.EmitBlock(ExitBB);
6405       // exit from construct;
6406       CodeGenFunction::JumpDest CancelDest =
6407           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6408       CGF.EmitBranchThroughCleanup(CancelDest);
6409       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6410     }
6411   }
6412 }
6413 
6414 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc,
6415                                      const Expr *IfCond,
6416                                      OpenMPDirectiveKind CancelRegion) {
6417   if (!CGF.HaveInsertPoint())
6418     return;
6419   // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
6420   // kmp_int32 cncl_kind);
6421   if (auto *OMPRegionInfo =
6422           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6423     auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF,
6424                                                         PrePostActionTy &) {
6425       CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6426       llvm::Value *Args[] = {
6427           RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc),
6428           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6429       // Ignore return result until untied tasks are supported.
6430       llvm::Value *Result = CGF.EmitRuntimeCall(
6431           RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args);
6432       // if (__kmpc_cancel()) {
6433       //   exit from construct;
6434       // }
6435       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6436       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6437       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6438       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6439       CGF.EmitBlock(ExitBB);
6440       // exit from construct;
6441       CodeGenFunction::JumpDest CancelDest =
6442           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6443       CGF.EmitBranchThroughCleanup(CancelDest);
6444       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6445     };
6446     if (IfCond) {
6447       emitIfClause(CGF, IfCond, ThenGen,
6448                    [](CodeGenFunction &, PrePostActionTy &) {});
6449     } else {
6450       RegionCodeGenTy ThenRCG(ThenGen);
6451       ThenRCG(CGF);
6452     }
6453   }
6454 }
6455 
6456 void CGOpenMPRuntime::emitTargetOutlinedFunction(
6457     const OMPExecutableDirective &D, StringRef ParentName,
6458     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6459     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6460   assert(!ParentName.empty() && "Invalid target region parent name!");
6461   HasEmittedTargetRegion = true;
6462   emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID,
6463                                    IsOffloadEntry, CodeGen);
6464 }
6465 
6466 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper(
6467     const OMPExecutableDirective &D, StringRef ParentName,
6468     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6469     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6470   // Create a unique name for the entry function using the source location
6471   // information of the current target region. The name will be something like:
6472   //
6473   // __omp_offloading_DD_FFFF_PP_lBB
6474   //
6475   // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the
6476   // mangled name of the function that encloses the target region and BB is the
6477   // line number of the target region.
6478 
6479   unsigned DeviceID;
6480   unsigned FileID;
6481   unsigned Line;
6482   getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID,
6483                            Line);
6484   SmallString<64> EntryFnName;
6485   {
6486     llvm::raw_svector_ostream OS(EntryFnName);
6487     OS << "__omp_offloading" << llvm::format("_%x", DeviceID)
6488        << llvm::format("_%x_", FileID) << ParentName << "_l" << Line;
6489   }
6490 
6491   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
6492 
6493   CodeGenFunction CGF(CGM, true);
6494   CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName);
6495   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6496 
6497   OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc());
6498 
6499   // If this target outline function is not an offload entry, we don't need to
6500   // register it.
6501   if (!IsOffloadEntry)
6502     return;
6503 
6504   // The target region ID is used by the runtime library to identify the current
6505   // target region, so it only has to be unique and not necessarily point to
6506   // anything. It could be the pointer to the outlined function that implements
6507   // the target region, but we aren't using that so that the compiler doesn't
6508   // need to keep that, and could therefore inline the host function if proven
6509   // worthwhile during optimization. In the other hand, if emitting code for the
6510   // device, the ID has to be the function address so that it can retrieved from
6511   // the offloading entry and launched by the runtime library. We also mark the
6512   // outlined function to have external linkage in case we are emitting code for
6513   // the device, because these functions will be entry points to the device.
6514 
6515   if (CGM.getLangOpts().OpenMPIsDevice) {
6516     OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy);
6517     OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
6518     OutlinedFn->setDSOLocal(false);
6519   } else {
6520     std::string Name = getName({EntryFnName, "region_id"});
6521     OutlinedFnID = new llvm::GlobalVariable(
6522         CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
6523         llvm::GlobalValue::WeakAnyLinkage,
6524         llvm::Constant::getNullValue(CGM.Int8Ty), Name);
6525   }
6526 
6527   // Register the information for the entry associated with this target region.
6528   OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
6529       DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID,
6530       OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion);
6531 }
6532 
6533 /// Checks if the expression is constant or does not have non-trivial function
6534 /// calls.
6535 static bool isTrivial(ASTContext &Ctx, const Expr * E) {
6536   // We can skip constant expressions.
6537   // We can skip expressions with trivial calls or simple expressions.
6538   return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) ||
6539           !E->hasNonTrivialCall(Ctx)) &&
6540          !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true);
6541 }
6542 
6543 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx,
6544                                                     const Stmt *Body) {
6545   const Stmt *Child = Body->IgnoreContainers();
6546   while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) {
6547     Child = nullptr;
6548     for (const Stmt *S : C->body()) {
6549       if (const auto *E = dyn_cast<Expr>(S)) {
6550         if (isTrivial(Ctx, E))
6551           continue;
6552       }
6553       // Some of the statements can be ignored.
6554       if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) ||
6555           isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S))
6556         continue;
6557       // Analyze declarations.
6558       if (const auto *DS = dyn_cast<DeclStmt>(S)) {
6559         if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) {
6560               if (isa<EmptyDecl>(D) || isa<DeclContext>(D) ||
6561                   isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) ||
6562                   isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) ||
6563                   isa<UsingDirectiveDecl>(D) ||
6564                   isa<OMPDeclareReductionDecl>(D) ||
6565                   isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D))
6566                 return true;
6567               const auto *VD = dyn_cast<VarDecl>(D);
6568               if (!VD)
6569                 return false;
6570               return VD->isConstexpr() ||
6571                      ((VD->getType().isTrivialType(Ctx) ||
6572                        VD->getType()->isReferenceType()) &&
6573                       (!VD->hasInit() || isTrivial(Ctx, VD->getInit())));
6574             }))
6575           continue;
6576       }
6577       // Found multiple children - cannot get the one child only.
6578       if (Child)
6579         return nullptr;
6580       Child = S;
6581     }
6582     if (Child)
6583       Child = Child->IgnoreContainers();
6584   }
6585   return Child;
6586 }
6587 
6588 /// Emit the number of teams for a target directive.  Inspect the num_teams
6589 /// clause associated with a teams construct combined or closely nested
6590 /// with the target directive.
6591 ///
6592 /// Emit a team of size one for directives such as 'target parallel' that
6593 /// have no associated teams construct.
6594 ///
6595 /// Otherwise, return nullptr.
6596 static llvm::Value *
6597 emitNumTeamsForTargetDirective(CodeGenFunction &CGF,
6598                                const OMPExecutableDirective &D) {
6599   assert(!CGF.getLangOpts().OpenMPIsDevice &&
6600          "Clauses associated with the teams directive expected to be emitted "
6601          "only for the host!");
6602   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6603   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6604          "Expected target-based executable directive.");
6605   CGBuilderTy &Bld = CGF.Builder;
6606   switch (DirectiveKind) {
6607   case OMPD_target: {
6608     const auto *CS = D.getInnermostCapturedStmt();
6609     const auto *Body =
6610         CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
6611     const Stmt *ChildStmt =
6612         CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body);
6613     if (const auto *NestedDir =
6614             dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
6615       if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) {
6616         if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) {
6617           CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6618           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6619           const Expr *NumTeams =
6620               NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6621           llvm::Value *NumTeamsVal =
6622               CGF.EmitScalarExpr(NumTeams,
6623                                  /*IgnoreResultAssign*/ true);
6624           return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6625                                    /*isSigned=*/true);
6626         }
6627         return Bld.getInt32(0);
6628       }
6629       if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) ||
6630           isOpenMPSimdDirective(NestedDir->getDirectiveKind()))
6631         return Bld.getInt32(1);
6632       return Bld.getInt32(0);
6633     }
6634     return nullptr;
6635   }
6636   case OMPD_target_teams:
6637   case OMPD_target_teams_distribute:
6638   case OMPD_target_teams_distribute_simd:
6639   case OMPD_target_teams_distribute_parallel_for:
6640   case OMPD_target_teams_distribute_parallel_for_simd: {
6641     if (D.hasClausesOfKind<OMPNumTeamsClause>()) {
6642       CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF);
6643       const Expr *NumTeams =
6644           D.getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6645       llvm::Value *NumTeamsVal =
6646           CGF.EmitScalarExpr(NumTeams,
6647                              /*IgnoreResultAssign*/ true);
6648       return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6649                                /*isSigned=*/true);
6650     }
6651     return Bld.getInt32(0);
6652   }
6653   case OMPD_target_parallel:
6654   case OMPD_target_parallel_for:
6655   case OMPD_target_parallel_for_simd:
6656   case OMPD_target_simd:
6657     return Bld.getInt32(1);
6658   case OMPD_parallel:
6659   case OMPD_for:
6660   case OMPD_parallel_for:
6661   case OMPD_parallel_master:
6662   case OMPD_parallel_sections:
6663   case OMPD_for_simd:
6664   case OMPD_parallel_for_simd:
6665   case OMPD_cancel:
6666   case OMPD_cancellation_point:
6667   case OMPD_ordered:
6668   case OMPD_threadprivate:
6669   case OMPD_allocate:
6670   case OMPD_task:
6671   case OMPD_simd:
6672   case OMPD_sections:
6673   case OMPD_section:
6674   case OMPD_single:
6675   case OMPD_master:
6676   case OMPD_critical:
6677   case OMPD_taskyield:
6678   case OMPD_barrier:
6679   case OMPD_taskwait:
6680   case OMPD_taskgroup:
6681   case OMPD_atomic:
6682   case OMPD_flush:
6683   case OMPD_teams:
6684   case OMPD_target_data:
6685   case OMPD_target_exit_data:
6686   case OMPD_target_enter_data:
6687   case OMPD_distribute:
6688   case OMPD_distribute_simd:
6689   case OMPD_distribute_parallel_for:
6690   case OMPD_distribute_parallel_for_simd:
6691   case OMPD_teams_distribute:
6692   case OMPD_teams_distribute_simd:
6693   case OMPD_teams_distribute_parallel_for:
6694   case OMPD_teams_distribute_parallel_for_simd:
6695   case OMPD_target_update:
6696   case OMPD_declare_simd:
6697   case OMPD_declare_variant:
6698   case OMPD_declare_target:
6699   case OMPD_end_declare_target:
6700   case OMPD_declare_reduction:
6701   case OMPD_declare_mapper:
6702   case OMPD_taskloop:
6703   case OMPD_taskloop_simd:
6704   case OMPD_master_taskloop:
6705   case OMPD_master_taskloop_simd:
6706   case OMPD_parallel_master_taskloop:
6707   case OMPD_parallel_master_taskloop_simd:
6708   case OMPD_requires:
6709   case OMPD_unknown:
6710     break;
6711   }
6712   llvm_unreachable("Unexpected directive kind.");
6713 }
6714 
6715 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS,
6716                                   llvm::Value *DefaultThreadLimitVal) {
6717   const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6718       CGF.getContext(), CS->getCapturedStmt());
6719   if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6720     if (isOpenMPParallelDirective(Dir->getDirectiveKind())) {
6721       llvm::Value *NumThreads = nullptr;
6722       llvm::Value *CondVal = nullptr;
6723       // Handle if clause. If if clause present, the number of threads is
6724       // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
6725       if (Dir->hasClausesOfKind<OMPIfClause>()) {
6726         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6727         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6728         const OMPIfClause *IfClause = nullptr;
6729         for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) {
6730           if (C->getNameModifier() == OMPD_unknown ||
6731               C->getNameModifier() == OMPD_parallel) {
6732             IfClause = C;
6733             break;
6734           }
6735         }
6736         if (IfClause) {
6737           const Expr *Cond = IfClause->getCondition();
6738           bool Result;
6739           if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
6740             if (!Result)
6741               return CGF.Builder.getInt32(1);
6742           } else {
6743             CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange());
6744             if (const auto *PreInit =
6745                     cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) {
6746               for (const auto *I : PreInit->decls()) {
6747                 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6748                   CGF.EmitVarDecl(cast<VarDecl>(*I));
6749                 } else {
6750                   CodeGenFunction::AutoVarEmission Emission =
6751                       CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6752                   CGF.EmitAutoVarCleanups(Emission);
6753                 }
6754               }
6755             }
6756             CondVal = CGF.EvaluateExprAsBool(Cond);
6757           }
6758         }
6759       }
6760       // Check the value of num_threads clause iff if clause was not specified
6761       // or is not evaluated to false.
6762       if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) {
6763         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6764         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6765         const auto *NumThreadsClause =
6766             Dir->getSingleClause<OMPNumThreadsClause>();
6767         CodeGenFunction::LexicalScope Scope(
6768             CGF, NumThreadsClause->getNumThreads()->getSourceRange());
6769         if (const auto *PreInit =
6770                 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) {
6771           for (const auto *I : PreInit->decls()) {
6772             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6773               CGF.EmitVarDecl(cast<VarDecl>(*I));
6774             } else {
6775               CodeGenFunction::AutoVarEmission Emission =
6776                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6777               CGF.EmitAutoVarCleanups(Emission);
6778             }
6779           }
6780         }
6781         NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads());
6782         NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty,
6783                                                /*isSigned=*/false);
6784         if (DefaultThreadLimitVal)
6785           NumThreads = CGF.Builder.CreateSelect(
6786               CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads),
6787               DefaultThreadLimitVal, NumThreads);
6788       } else {
6789         NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal
6790                                            : CGF.Builder.getInt32(0);
6791       }
6792       // Process condition of the if clause.
6793       if (CondVal) {
6794         NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads,
6795                                               CGF.Builder.getInt32(1));
6796       }
6797       return NumThreads;
6798     }
6799     if (isOpenMPSimdDirective(Dir->getDirectiveKind()))
6800       return CGF.Builder.getInt32(1);
6801     return DefaultThreadLimitVal;
6802   }
6803   return DefaultThreadLimitVal ? DefaultThreadLimitVal
6804                                : CGF.Builder.getInt32(0);
6805 }
6806 
6807 /// Emit the number of threads for a target directive.  Inspect the
6808 /// thread_limit clause associated with a teams construct combined or closely
6809 /// nested with the target directive.
6810 ///
6811 /// Emit the num_threads clause for directives such as 'target parallel' that
6812 /// have no associated teams construct.
6813 ///
6814 /// Otherwise, return nullptr.
6815 static llvm::Value *
6816 emitNumThreadsForTargetDirective(CodeGenFunction &CGF,
6817                                  const OMPExecutableDirective &D) {
6818   assert(!CGF.getLangOpts().OpenMPIsDevice &&
6819          "Clauses associated with the teams directive expected to be emitted "
6820          "only for the host!");
6821   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6822   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6823          "Expected target-based executable directive.");
6824   CGBuilderTy &Bld = CGF.Builder;
6825   llvm::Value *ThreadLimitVal = nullptr;
6826   llvm::Value *NumThreadsVal = nullptr;
6827   switch (DirectiveKind) {
6828   case OMPD_target: {
6829     const CapturedStmt *CS = D.getInnermostCapturedStmt();
6830     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
6831       return NumThreads;
6832     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6833         CGF.getContext(), CS->getCapturedStmt());
6834     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6835       if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) {
6836         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6837         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6838         const auto *ThreadLimitClause =
6839             Dir->getSingleClause<OMPThreadLimitClause>();
6840         CodeGenFunction::LexicalScope Scope(
6841             CGF, ThreadLimitClause->getThreadLimit()->getSourceRange());
6842         if (const auto *PreInit =
6843                 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) {
6844           for (const auto *I : PreInit->decls()) {
6845             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6846               CGF.EmitVarDecl(cast<VarDecl>(*I));
6847             } else {
6848               CodeGenFunction::AutoVarEmission Emission =
6849                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6850               CGF.EmitAutoVarCleanups(Emission);
6851             }
6852           }
6853         }
6854         llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
6855             ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
6856         ThreadLimitVal =
6857             Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
6858       }
6859       if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) &&
6860           !isOpenMPDistributeDirective(Dir->getDirectiveKind())) {
6861         CS = Dir->getInnermostCapturedStmt();
6862         const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6863             CGF.getContext(), CS->getCapturedStmt());
6864         Dir = dyn_cast_or_null<OMPExecutableDirective>(Child);
6865       }
6866       if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) &&
6867           !isOpenMPSimdDirective(Dir->getDirectiveKind())) {
6868         CS = Dir->getInnermostCapturedStmt();
6869         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
6870           return NumThreads;
6871       }
6872       if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind()))
6873         return Bld.getInt32(1);
6874     }
6875     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
6876   }
6877   case OMPD_target_teams: {
6878     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
6879       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
6880       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
6881       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
6882           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
6883       ThreadLimitVal =
6884           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
6885     }
6886     const CapturedStmt *CS = D.getInnermostCapturedStmt();
6887     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
6888       return NumThreads;
6889     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6890         CGF.getContext(), CS->getCapturedStmt());
6891     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6892       if (Dir->getDirectiveKind() == OMPD_distribute) {
6893         CS = Dir->getInnermostCapturedStmt();
6894         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
6895           return NumThreads;
6896       }
6897     }
6898     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
6899   }
6900   case OMPD_target_teams_distribute:
6901     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
6902       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
6903       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
6904       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
6905           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
6906       ThreadLimitVal =
6907           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
6908     }
6909     return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal);
6910   case OMPD_target_parallel:
6911   case OMPD_target_parallel_for:
6912   case OMPD_target_parallel_for_simd:
6913   case OMPD_target_teams_distribute_parallel_for:
6914   case OMPD_target_teams_distribute_parallel_for_simd: {
6915     llvm::Value *CondVal = nullptr;
6916     // Handle if clause. If if clause present, the number of threads is
6917     // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
6918     if (D.hasClausesOfKind<OMPIfClause>()) {
6919       const OMPIfClause *IfClause = nullptr;
6920       for (const auto *C : D.getClausesOfKind<OMPIfClause>()) {
6921         if (C->getNameModifier() == OMPD_unknown ||
6922             C->getNameModifier() == OMPD_parallel) {
6923           IfClause = C;
6924           break;
6925         }
6926       }
6927       if (IfClause) {
6928         const Expr *Cond = IfClause->getCondition();
6929         bool Result;
6930         if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
6931           if (!Result)
6932             return Bld.getInt32(1);
6933         } else {
6934           CodeGenFunction::RunCleanupsScope Scope(CGF);
6935           CondVal = CGF.EvaluateExprAsBool(Cond);
6936         }
6937       }
6938     }
6939     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
6940       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
6941       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
6942       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
6943           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
6944       ThreadLimitVal =
6945           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
6946     }
6947     if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
6948       CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
6949       const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
6950       llvm::Value *NumThreads = CGF.EmitScalarExpr(
6951           NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true);
6952       NumThreadsVal =
6953           Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false);
6954       ThreadLimitVal = ThreadLimitVal
6955                            ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal,
6956                                                                 ThreadLimitVal),
6957                                               NumThreadsVal, ThreadLimitVal)
6958                            : NumThreadsVal;
6959     }
6960     if (!ThreadLimitVal)
6961       ThreadLimitVal = Bld.getInt32(0);
6962     if (CondVal)
6963       return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1));
6964     return ThreadLimitVal;
6965   }
6966   case OMPD_target_teams_distribute_simd:
6967   case OMPD_target_simd:
6968     return Bld.getInt32(1);
6969   case OMPD_parallel:
6970   case OMPD_for:
6971   case OMPD_parallel_for:
6972   case OMPD_parallel_master:
6973   case OMPD_parallel_sections:
6974   case OMPD_for_simd:
6975   case OMPD_parallel_for_simd:
6976   case OMPD_cancel:
6977   case OMPD_cancellation_point:
6978   case OMPD_ordered:
6979   case OMPD_threadprivate:
6980   case OMPD_allocate:
6981   case OMPD_task:
6982   case OMPD_simd:
6983   case OMPD_sections:
6984   case OMPD_section:
6985   case OMPD_single:
6986   case OMPD_master:
6987   case OMPD_critical:
6988   case OMPD_taskyield:
6989   case OMPD_barrier:
6990   case OMPD_taskwait:
6991   case OMPD_taskgroup:
6992   case OMPD_atomic:
6993   case OMPD_flush:
6994   case OMPD_teams:
6995   case OMPD_target_data:
6996   case OMPD_target_exit_data:
6997   case OMPD_target_enter_data:
6998   case OMPD_distribute:
6999   case OMPD_distribute_simd:
7000   case OMPD_distribute_parallel_for:
7001   case OMPD_distribute_parallel_for_simd:
7002   case OMPD_teams_distribute:
7003   case OMPD_teams_distribute_simd:
7004   case OMPD_teams_distribute_parallel_for:
7005   case OMPD_teams_distribute_parallel_for_simd:
7006   case OMPD_target_update:
7007   case OMPD_declare_simd:
7008   case OMPD_declare_variant:
7009   case OMPD_declare_target:
7010   case OMPD_end_declare_target:
7011   case OMPD_declare_reduction:
7012   case OMPD_declare_mapper:
7013   case OMPD_taskloop:
7014   case OMPD_taskloop_simd:
7015   case OMPD_master_taskloop:
7016   case OMPD_master_taskloop_simd:
7017   case OMPD_parallel_master_taskloop:
7018   case OMPD_parallel_master_taskloop_simd:
7019   case OMPD_requires:
7020   case OMPD_unknown:
7021     break;
7022   }
7023   llvm_unreachable("Unsupported directive kind.");
7024 }
7025 
7026 namespace {
7027 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
7028 
7029 // Utility to handle information from clauses associated with a given
7030 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause).
7031 // It provides a convenient interface to obtain the information and generate
7032 // code for that information.
7033 class MappableExprsHandler {
7034 public:
7035   /// Values for bit flags used to specify the mapping type for
7036   /// offloading.
7037   enum OpenMPOffloadMappingFlags : uint64_t {
7038     /// No flags
7039     OMP_MAP_NONE = 0x0,
7040     /// Allocate memory on the device and move data from host to device.
7041     OMP_MAP_TO = 0x01,
7042     /// Allocate memory on the device and move data from device to host.
7043     OMP_MAP_FROM = 0x02,
7044     /// Always perform the requested mapping action on the element, even
7045     /// if it was already mapped before.
7046     OMP_MAP_ALWAYS = 0x04,
7047     /// Delete the element from the device environment, ignoring the
7048     /// current reference count associated with the element.
7049     OMP_MAP_DELETE = 0x08,
7050     /// The element being mapped is a pointer-pointee pair; both the
7051     /// pointer and the pointee should be mapped.
7052     OMP_MAP_PTR_AND_OBJ = 0x10,
7053     /// This flags signals that the base address of an entry should be
7054     /// passed to the target kernel as an argument.
7055     OMP_MAP_TARGET_PARAM = 0x20,
7056     /// Signal that the runtime library has to return the device pointer
7057     /// in the current position for the data being mapped. Used when we have the
7058     /// use_device_ptr clause.
7059     OMP_MAP_RETURN_PARAM = 0x40,
7060     /// This flag signals that the reference being passed is a pointer to
7061     /// private data.
7062     OMP_MAP_PRIVATE = 0x80,
7063     /// Pass the element to the device by value.
7064     OMP_MAP_LITERAL = 0x100,
7065     /// Implicit map
7066     OMP_MAP_IMPLICIT = 0x200,
7067     /// Close is a hint to the runtime to allocate memory close to
7068     /// the target device.
7069     OMP_MAP_CLOSE = 0x400,
7070     /// The 16 MSBs of the flags indicate whether the entry is member of some
7071     /// struct/class.
7072     OMP_MAP_MEMBER_OF = 0xffff000000000000,
7073     LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF),
7074   };
7075 
7076   /// Get the offset of the OMP_MAP_MEMBER_OF field.
7077   static unsigned getFlagMemberOffset() {
7078     unsigned Offset = 0;
7079     for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1);
7080          Remain = Remain >> 1)
7081       Offset++;
7082     return Offset;
7083   }
7084 
7085   /// Class that associates information with a base pointer to be passed to the
7086   /// runtime library.
7087   class BasePointerInfo {
7088     /// The base pointer.
7089     llvm::Value *Ptr = nullptr;
7090     /// The base declaration that refers to this device pointer, or null if
7091     /// there is none.
7092     const ValueDecl *DevPtrDecl = nullptr;
7093 
7094   public:
7095     BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr)
7096         : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {}
7097     llvm::Value *operator*() const { return Ptr; }
7098     const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; }
7099     void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; }
7100   };
7101 
7102   using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>;
7103   using MapValuesArrayTy = SmallVector<llvm::Value *, 4>;
7104   using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>;
7105 
7106   /// Map between a struct and the its lowest & highest elements which have been
7107   /// mapped.
7108   /// [ValueDecl *] --> {LE(FieldIndex, Pointer),
7109   ///                    HE(FieldIndex, Pointer)}
7110   struct StructRangeInfoTy {
7111     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = {
7112         0, Address::invalid()};
7113     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = {
7114         0, Address::invalid()};
7115     Address Base = Address::invalid();
7116   };
7117 
7118 private:
7119   /// Kind that defines how a device pointer has to be returned.
7120   struct MapInfo {
7121     OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
7122     OpenMPMapClauseKind MapType = OMPC_MAP_unknown;
7123     ArrayRef<OpenMPMapModifierKind> MapModifiers;
7124     bool ReturnDevicePointer = false;
7125     bool IsImplicit = false;
7126 
7127     MapInfo() = default;
7128     MapInfo(
7129         OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7130         OpenMPMapClauseKind MapType,
7131         ArrayRef<OpenMPMapModifierKind> MapModifiers,
7132         bool ReturnDevicePointer, bool IsImplicit)
7133         : Components(Components), MapType(MapType), MapModifiers(MapModifiers),
7134           ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {}
7135   };
7136 
7137   /// If use_device_ptr is used on a pointer which is a struct member and there
7138   /// is no map information about it, then emission of that entry is deferred
7139   /// until the whole struct has been processed.
7140   struct DeferredDevicePtrEntryTy {
7141     const Expr *IE = nullptr;
7142     const ValueDecl *VD = nullptr;
7143 
7144     DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD)
7145         : IE(IE), VD(VD) {}
7146   };
7147 
7148   /// The target directive from where the mappable clauses were extracted. It
7149   /// is either a executable directive or a user-defined mapper directive.
7150   llvm::PointerUnion<const OMPExecutableDirective *,
7151                      const OMPDeclareMapperDecl *>
7152       CurDir;
7153 
7154   /// Function the directive is being generated for.
7155   CodeGenFunction &CGF;
7156 
7157   /// Set of all first private variables in the current directive.
7158   /// bool data is set to true if the variable is implicitly marked as
7159   /// firstprivate, false otherwise.
7160   llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls;
7161 
7162   /// Map between device pointer declarations and their expression components.
7163   /// The key value for declarations in 'this' is null.
7164   llvm::DenseMap<
7165       const ValueDecl *,
7166       SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7167       DevPointersMap;
7168 
7169   llvm::Value *getExprTypeSize(const Expr *E) const {
7170     QualType ExprTy = E->getType().getCanonicalType();
7171 
7172     // Reference types are ignored for mapping purposes.
7173     if (const auto *RefTy = ExprTy->getAs<ReferenceType>())
7174       ExprTy = RefTy->getPointeeType().getCanonicalType();
7175 
7176     // Given that an array section is considered a built-in type, we need to
7177     // do the calculation based on the length of the section instead of relying
7178     // on CGF.getTypeSize(E->getType()).
7179     if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) {
7180       QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType(
7181                             OAE->getBase()->IgnoreParenImpCasts())
7182                             .getCanonicalType();
7183 
7184       // If there is no length associated with the expression and lower bound is
7185       // not specified too, that means we are using the whole length of the
7186       // base.
7187       if (!OAE->getLength() && OAE->getColonLoc().isValid() &&
7188           !OAE->getLowerBound())
7189         return CGF.getTypeSize(BaseTy);
7190 
7191       llvm::Value *ElemSize;
7192       if (const auto *PTy = BaseTy->getAs<PointerType>()) {
7193         ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType());
7194       } else {
7195         const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr());
7196         assert(ATy && "Expecting array type if not a pointer type.");
7197         ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType());
7198       }
7199 
7200       // If we don't have a length at this point, that is because we have an
7201       // array section with a single element.
7202       if (!OAE->getLength() && OAE->getColonLoc().isInvalid())
7203         return ElemSize;
7204 
7205       if (const Expr *LenExpr = OAE->getLength()) {
7206         llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr);
7207         LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(),
7208                                              CGF.getContext().getSizeType(),
7209                                              LenExpr->getExprLoc());
7210         return CGF.Builder.CreateNUWMul(LengthVal, ElemSize);
7211       }
7212       assert(!OAE->getLength() && OAE->getColonLoc().isValid() &&
7213              OAE->getLowerBound() && "expected array_section[lb:].");
7214       // Size = sizetype - lb * elemtype;
7215       llvm::Value *LengthVal = CGF.getTypeSize(BaseTy);
7216       llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound());
7217       LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(),
7218                                        CGF.getContext().getSizeType(),
7219                                        OAE->getLowerBound()->getExprLoc());
7220       LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize);
7221       llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal);
7222       llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal);
7223       LengthVal = CGF.Builder.CreateSelect(
7224           Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0));
7225       return LengthVal;
7226     }
7227     return CGF.getTypeSize(ExprTy);
7228   }
7229 
7230   /// Return the corresponding bits for a given map clause modifier. Add
7231   /// a flag marking the map as a pointer if requested. Add a flag marking the
7232   /// map as the first one of a series of maps that relate to the same map
7233   /// expression.
7234   OpenMPOffloadMappingFlags getMapTypeBits(
7235       OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7236       bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const {
7237     OpenMPOffloadMappingFlags Bits =
7238         IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE;
7239     switch (MapType) {
7240     case OMPC_MAP_alloc:
7241     case OMPC_MAP_release:
7242       // alloc and release is the default behavior in the runtime library,  i.e.
7243       // if we don't pass any bits alloc/release that is what the runtime is
7244       // going to do. Therefore, we don't need to signal anything for these two
7245       // type modifiers.
7246       break;
7247     case OMPC_MAP_to:
7248       Bits |= OMP_MAP_TO;
7249       break;
7250     case OMPC_MAP_from:
7251       Bits |= OMP_MAP_FROM;
7252       break;
7253     case OMPC_MAP_tofrom:
7254       Bits |= OMP_MAP_TO | OMP_MAP_FROM;
7255       break;
7256     case OMPC_MAP_delete:
7257       Bits |= OMP_MAP_DELETE;
7258       break;
7259     case OMPC_MAP_unknown:
7260       llvm_unreachable("Unexpected map type!");
7261     }
7262     if (AddPtrFlag)
7263       Bits |= OMP_MAP_PTR_AND_OBJ;
7264     if (AddIsTargetParamFlag)
7265       Bits |= OMP_MAP_TARGET_PARAM;
7266     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always)
7267         != MapModifiers.end())
7268       Bits |= OMP_MAP_ALWAYS;
7269     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close)
7270         != MapModifiers.end())
7271       Bits |= OMP_MAP_CLOSE;
7272     return Bits;
7273   }
7274 
7275   /// Return true if the provided expression is a final array section. A
7276   /// final array section, is one whose length can't be proved to be one.
7277   bool isFinalArraySectionExpression(const Expr *E) const {
7278     const auto *OASE = dyn_cast<OMPArraySectionExpr>(E);
7279 
7280     // It is not an array section and therefore not a unity-size one.
7281     if (!OASE)
7282       return false;
7283 
7284     // An array section with no colon always refer to a single element.
7285     if (OASE->getColonLoc().isInvalid())
7286       return false;
7287 
7288     const Expr *Length = OASE->getLength();
7289 
7290     // If we don't have a length we have to check if the array has size 1
7291     // for this dimension. Also, we should always expect a length if the
7292     // base type is pointer.
7293     if (!Length) {
7294       QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType(
7295                              OASE->getBase()->IgnoreParenImpCasts())
7296                              .getCanonicalType();
7297       if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr()))
7298         return ATy->getSize().getSExtValue() != 1;
7299       // If we don't have a constant dimension length, we have to consider
7300       // the current section as having any size, so it is not necessarily
7301       // unitary. If it happen to be unity size, that's user fault.
7302       return true;
7303     }
7304 
7305     // Check if the length evaluates to 1.
7306     Expr::EvalResult Result;
7307     if (!Length->EvaluateAsInt(Result, CGF.getContext()))
7308       return true; // Can have more that size 1.
7309 
7310     llvm::APSInt ConstLength = Result.Val.getInt();
7311     return ConstLength.getSExtValue() != 1;
7312   }
7313 
7314   /// Generate the base pointers, section pointers, sizes and map type
7315   /// bits for the provided map type, map modifier, and expression components.
7316   /// \a IsFirstComponent should be set to true if the provided set of
7317   /// components is the first associated with a capture.
7318   void generateInfoForComponentList(
7319       OpenMPMapClauseKind MapType,
7320       ArrayRef<OpenMPMapModifierKind> MapModifiers,
7321       OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7322       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
7323       MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
7324       StructRangeInfoTy &PartialStruct, bool IsFirstComponentList,
7325       bool IsImplicit,
7326       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
7327           OverlappedElements = llvm::None) const {
7328     // The following summarizes what has to be generated for each map and the
7329     // types below. The generated information is expressed in this order:
7330     // base pointer, section pointer, size, flags
7331     // (to add to the ones that come from the map type and modifier).
7332     //
7333     // double d;
7334     // int i[100];
7335     // float *p;
7336     //
7337     // struct S1 {
7338     //   int i;
7339     //   float f[50];
7340     // }
7341     // struct S2 {
7342     //   int i;
7343     //   float f[50];
7344     //   S1 s;
7345     //   double *p;
7346     //   struct S2 *ps;
7347     // }
7348     // S2 s;
7349     // S2 *ps;
7350     //
7351     // map(d)
7352     // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM
7353     //
7354     // map(i)
7355     // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM
7356     //
7357     // map(i[1:23])
7358     // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM
7359     //
7360     // map(p)
7361     // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
7362     //
7363     // map(p[1:24])
7364     // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM
7365     //
7366     // map(s)
7367     // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM
7368     //
7369     // map(s.i)
7370     // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM
7371     //
7372     // map(s.s.f)
7373     // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7374     //
7375     // map(s.p)
7376     // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM
7377     //
7378     // map(to: s.p[:22])
7379     // &s, &(s.p), sizeof(double*), TARGET_PARAM (*)
7380     // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**)
7381     // &(s.p), &(s.p[0]), 22*sizeof(double),
7382     //   MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
7383     // (*) alloc space for struct members, only this is a target parameter
7384     // (**) map the pointer (nothing to be mapped in this example) (the compiler
7385     //      optimizes this entry out, same in the examples below)
7386     // (***) map the pointee (map: to)
7387     //
7388     // map(s.ps)
7389     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7390     //
7391     // map(from: s.ps->s.i)
7392     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7393     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7394     // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ  | FROM
7395     //
7396     // map(to: s.ps->ps)
7397     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7398     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7399     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ  | TO
7400     //
7401     // map(s.ps->ps->ps)
7402     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7403     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7404     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7405     // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7406     //
7407     // map(to: s.ps->ps->s.f[:22])
7408     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7409     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7410     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7411     // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7412     //
7413     // map(ps)
7414     // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM
7415     //
7416     // map(ps->i)
7417     // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM
7418     //
7419     // map(ps->s.f)
7420     // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7421     //
7422     // map(from: ps->p)
7423     // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM
7424     //
7425     // map(to: ps->p[:22])
7426     // ps, &(ps->p), sizeof(double*), TARGET_PARAM
7427     // ps, &(ps->p), sizeof(double*), MEMBER_OF(1)
7428     // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO
7429     //
7430     // map(ps->ps)
7431     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7432     //
7433     // map(from: ps->ps->s.i)
7434     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7435     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7436     // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7437     //
7438     // map(from: ps->ps->ps)
7439     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7440     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7441     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7442     //
7443     // map(ps->ps->ps->ps)
7444     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7445     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7446     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7447     // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7448     //
7449     // map(to: ps->ps->ps->s.f[:22])
7450     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7451     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7452     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7453     // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7454     //
7455     // map(to: s.f[:22]) map(from: s.p[:33])
7456     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) +
7457     //     sizeof(double*) (**), TARGET_PARAM
7458     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO
7459     // &s, &(s.p), sizeof(double*), MEMBER_OF(1)
7460     // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7461     // (*) allocate contiguous space needed to fit all mapped members even if
7462     //     we allocate space for members not mapped (in this example,
7463     //     s.f[22..49] and s.s are not mapped, yet we must allocate space for
7464     //     them as well because they fall between &s.f[0] and &s.p)
7465     //
7466     // map(from: s.f[:22]) map(to: ps->p[:33])
7467     // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM
7468     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7469     // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*)
7470     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO
7471     // (*) the struct this entry pertains to is the 2nd element in the list of
7472     //     arguments, hence MEMBER_OF(2)
7473     //
7474     // map(from: s.f[:22], s.s) map(to: ps->p[:33])
7475     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM
7476     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM
7477     // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM
7478     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7479     // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*)
7480     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO
7481     // (*) the struct this entry pertains to is the 4th element in the list
7482     //     of arguments, hence MEMBER_OF(4)
7483 
7484     // Track if the map information being generated is the first for a capture.
7485     bool IsCaptureFirstInfo = IsFirstComponentList;
7486     // When the variable is on a declare target link or in a to clause with
7487     // unified memory, a reference is needed to hold the host/device address
7488     // of the variable.
7489     bool RequiresReference = false;
7490 
7491     // Scan the components from the base to the complete expression.
7492     auto CI = Components.rbegin();
7493     auto CE = Components.rend();
7494     auto I = CI;
7495 
7496     // Track if the map information being generated is the first for a list of
7497     // components.
7498     bool IsExpressionFirstInfo = true;
7499     Address BP = Address::invalid();
7500     const Expr *AssocExpr = I->getAssociatedExpression();
7501     const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr);
7502     const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr);
7503 
7504     if (isa<MemberExpr>(AssocExpr)) {
7505       // The base is the 'this' pointer. The content of the pointer is going
7506       // to be the base of the field being mapped.
7507       BP = CGF.LoadCXXThisAddress();
7508     } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) ||
7509                (OASE &&
7510                 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) {
7511       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7512     } else {
7513       // The base is the reference to the variable.
7514       // BP = &Var.
7515       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7516       if (const auto *VD =
7517               dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) {
7518         if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
7519                 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) {
7520           if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
7521               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
7522                CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) {
7523             RequiresReference = true;
7524             BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
7525           }
7526         }
7527       }
7528 
7529       // If the variable is a pointer and is being dereferenced (i.e. is not
7530       // the last component), the base has to be the pointer itself, not its
7531       // reference. References are ignored for mapping purposes.
7532       QualType Ty =
7533           I->getAssociatedDeclaration()->getType().getNonReferenceType();
7534       if (Ty->isAnyPointerType() && std::next(I) != CE) {
7535         BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
7536 
7537         // We do not need to generate individual map information for the
7538         // pointer, it can be associated with the combined storage.
7539         ++I;
7540       }
7541     }
7542 
7543     // Track whether a component of the list should be marked as MEMBER_OF some
7544     // combined entry (for partial structs). Only the first PTR_AND_OBJ entry
7545     // in a component list should be marked as MEMBER_OF, all subsequent entries
7546     // do not belong to the base struct. E.g.
7547     // struct S2 s;
7548     // s.ps->ps->ps->f[:]
7549     //   (1) (2) (3) (4)
7550     // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a
7551     // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3)
7552     // is the pointee of ps(2) which is not member of struct s, so it should not
7553     // be marked as such (it is still PTR_AND_OBJ).
7554     // The variable is initialized to false so that PTR_AND_OBJ entries which
7555     // are not struct members are not considered (e.g. array of pointers to
7556     // data).
7557     bool ShouldBeMemberOf = false;
7558 
7559     // Variable keeping track of whether or not we have encountered a component
7560     // in the component list which is a member expression. Useful when we have a
7561     // pointer or a final array section, in which case it is the previous
7562     // component in the list which tells us whether we have a member expression.
7563     // E.g. X.f[:]
7564     // While processing the final array section "[:]" it is "f" which tells us
7565     // whether we are dealing with a member of a declared struct.
7566     const MemberExpr *EncounteredME = nullptr;
7567 
7568     for (; I != CE; ++I) {
7569       // If the current component is member of a struct (parent struct) mark it.
7570       if (!EncounteredME) {
7571         EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression());
7572         // If we encounter a PTR_AND_OBJ entry from now on it should be marked
7573         // as MEMBER_OF the parent struct.
7574         if (EncounteredME)
7575           ShouldBeMemberOf = true;
7576       }
7577 
7578       auto Next = std::next(I);
7579 
7580       // We need to generate the addresses and sizes if this is the last
7581       // component, if the component is a pointer or if it is an array section
7582       // whose length can't be proved to be one. If this is a pointer, it
7583       // becomes the base address for the following components.
7584 
7585       // A final array section, is one whose length can't be proved to be one.
7586       bool IsFinalArraySection =
7587           isFinalArraySectionExpression(I->getAssociatedExpression());
7588 
7589       // Get information on whether the element is a pointer. Have to do a
7590       // special treatment for array sections given that they are built-in
7591       // types.
7592       const auto *OASE =
7593           dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression());
7594       bool IsPointer =
7595           (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE)
7596                        .getCanonicalType()
7597                        ->isAnyPointerType()) ||
7598           I->getAssociatedExpression()->getType()->isAnyPointerType();
7599 
7600       if (Next == CE || IsPointer || IsFinalArraySection) {
7601         // If this is not the last component, we expect the pointer to be
7602         // associated with an array expression or member expression.
7603         assert((Next == CE ||
7604                 isa<MemberExpr>(Next->getAssociatedExpression()) ||
7605                 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) ||
7606                 isa<OMPArraySectionExpr>(Next->getAssociatedExpression())) &&
7607                "Unexpected expression");
7608 
7609         Address LB = CGF.EmitOMPSharedLValue(I->getAssociatedExpression())
7610                          .getAddress(CGF);
7611 
7612         // If this component is a pointer inside the base struct then we don't
7613         // need to create any entry for it - it will be combined with the object
7614         // it is pointing to into a single PTR_AND_OBJ entry.
7615         bool IsMemberPointer =
7616             IsPointer && EncounteredME &&
7617             (dyn_cast<MemberExpr>(I->getAssociatedExpression()) ==
7618              EncounteredME);
7619         if (!OverlappedElements.empty()) {
7620           // Handle base element with the info for overlapped elements.
7621           assert(!PartialStruct.Base.isValid() && "The base element is set.");
7622           assert(Next == CE &&
7623                  "Expected last element for the overlapped elements.");
7624           assert(!IsPointer &&
7625                  "Unexpected base element with the pointer type.");
7626           // Mark the whole struct as the struct that requires allocation on the
7627           // device.
7628           PartialStruct.LowestElem = {0, LB};
7629           CharUnits TypeSize = CGF.getContext().getTypeSizeInChars(
7630               I->getAssociatedExpression()->getType());
7631           Address HB = CGF.Builder.CreateConstGEP(
7632               CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB,
7633                                                               CGF.VoidPtrTy),
7634               TypeSize.getQuantity() - 1);
7635           PartialStruct.HighestElem = {
7636               std::numeric_limits<decltype(
7637                   PartialStruct.HighestElem.first)>::max(),
7638               HB};
7639           PartialStruct.Base = BP;
7640           // Emit data for non-overlapped data.
7641           OpenMPOffloadMappingFlags Flags =
7642               OMP_MAP_MEMBER_OF |
7643               getMapTypeBits(MapType, MapModifiers, IsImplicit,
7644                              /*AddPtrFlag=*/false,
7645                              /*AddIsTargetParamFlag=*/false);
7646           LB = BP;
7647           llvm::Value *Size = nullptr;
7648           // Do bitcopy of all non-overlapped structure elements.
7649           for (OMPClauseMappableExprCommon::MappableExprComponentListRef
7650                    Component : OverlappedElements) {
7651             Address ComponentLB = Address::invalid();
7652             for (const OMPClauseMappableExprCommon::MappableComponent &MC :
7653                  Component) {
7654               if (MC.getAssociatedDeclaration()) {
7655                 ComponentLB =
7656                     CGF.EmitOMPSharedLValue(MC.getAssociatedExpression())
7657                         .getAddress(CGF);
7658                 Size = CGF.Builder.CreatePtrDiff(
7659                     CGF.EmitCastToVoidPtr(ComponentLB.getPointer()),
7660                     CGF.EmitCastToVoidPtr(LB.getPointer()));
7661                 break;
7662               }
7663             }
7664             BasePointers.push_back(BP.getPointer());
7665             Pointers.push_back(LB.getPointer());
7666             Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty,
7667                                                       /*isSigned=*/true));
7668             Types.push_back(Flags);
7669             LB = CGF.Builder.CreateConstGEP(ComponentLB, 1);
7670           }
7671           BasePointers.push_back(BP.getPointer());
7672           Pointers.push_back(LB.getPointer());
7673           Size = CGF.Builder.CreatePtrDiff(
7674               CGF.EmitCastToVoidPtr(
7675                   CGF.Builder.CreateConstGEP(HB, 1).getPointer()),
7676               CGF.EmitCastToVoidPtr(LB.getPointer()));
7677           Sizes.push_back(
7678               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7679           Types.push_back(Flags);
7680           break;
7681         }
7682         llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression());
7683         if (!IsMemberPointer) {
7684           BasePointers.push_back(BP.getPointer());
7685           Pointers.push_back(LB.getPointer());
7686           Sizes.push_back(
7687               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7688 
7689           // We need to add a pointer flag for each map that comes from the
7690           // same expression except for the first one. We also need to signal
7691           // this map is the first one that relates with the current capture
7692           // (there is a set of entries for each capture).
7693           OpenMPOffloadMappingFlags Flags = getMapTypeBits(
7694               MapType, MapModifiers, IsImplicit,
7695               !IsExpressionFirstInfo || RequiresReference,
7696               IsCaptureFirstInfo && !RequiresReference);
7697 
7698           if (!IsExpressionFirstInfo) {
7699             // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well,
7700             // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags.
7701             if (IsPointer)
7702               Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS |
7703                          OMP_MAP_DELETE | OMP_MAP_CLOSE);
7704 
7705             if (ShouldBeMemberOf) {
7706               // Set placeholder value MEMBER_OF=FFFF to indicate that the flag
7707               // should be later updated with the correct value of MEMBER_OF.
7708               Flags |= OMP_MAP_MEMBER_OF;
7709               // From now on, all subsequent PTR_AND_OBJ entries should not be
7710               // marked as MEMBER_OF.
7711               ShouldBeMemberOf = false;
7712             }
7713           }
7714 
7715           Types.push_back(Flags);
7716         }
7717 
7718         // If we have encountered a member expression so far, keep track of the
7719         // mapped member. If the parent is "*this", then the value declaration
7720         // is nullptr.
7721         if (EncounteredME) {
7722           const auto *FD = dyn_cast<FieldDecl>(EncounteredME->getMemberDecl());
7723           unsigned FieldIndex = FD->getFieldIndex();
7724 
7725           // Update info about the lowest and highest elements for this struct
7726           if (!PartialStruct.Base.isValid()) {
7727             PartialStruct.LowestElem = {FieldIndex, LB};
7728             PartialStruct.HighestElem = {FieldIndex, LB};
7729             PartialStruct.Base = BP;
7730           } else if (FieldIndex < PartialStruct.LowestElem.first) {
7731             PartialStruct.LowestElem = {FieldIndex, LB};
7732           } else if (FieldIndex > PartialStruct.HighestElem.first) {
7733             PartialStruct.HighestElem = {FieldIndex, LB};
7734           }
7735         }
7736 
7737         // If we have a final array section, we are done with this expression.
7738         if (IsFinalArraySection)
7739           break;
7740 
7741         // The pointer becomes the base for the next element.
7742         if (Next != CE)
7743           BP = LB;
7744 
7745         IsExpressionFirstInfo = false;
7746         IsCaptureFirstInfo = false;
7747       }
7748     }
7749   }
7750 
7751   /// Return the adjusted map modifiers if the declaration a capture refers to
7752   /// appears in a first-private clause. This is expected to be used only with
7753   /// directives that start with 'target'.
7754   MappableExprsHandler::OpenMPOffloadMappingFlags
7755   getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const {
7756     assert(Cap.capturesVariable() && "Expected capture by reference only!");
7757 
7758     // A first private variable captured by reference will use only the
7759     // 'private ptr' and 'map to' flag. Return the right flags if the captured
7760     // declaration is known as first-private in this handler.
7761     if (FirstPrivateDecls.count(Cap.getCapturedVar())) {
7762       if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) &&
7763           Cap.getCaptureKind() == CapturedStmt::VCK_ByRef)
7764         return MappableExprsHandler::OMP_MAP_ALWAYS |
7765                MappableExprsHandler::OMP_MAP_TO;
7766       if (Cap.getCapturedVar()->getType()->isAnyPointerType())
7767         return MappableExprsHandler::OMP_MAP_TO |
7768                MappableExprsHandler::OMP_MAP_PTR_AND_OBJ;
7769       return MappableExprsHandler::OMP_MAP_PRIVATE |
7770              MappableExprsHandler::OMP_MAP_TO;
7771     }
7772     return MappableExprsHandler::OMP_MAP_TO |
7773            MappableExprsHandler::OMP_MAP_FROM;
7774   }
7775 
7776   static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) {
7777     // Rotate by getFlagMemberOffset() bits.
7778     return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1)
7779                                                   << getFlagMemberOffset());
7780   }
7781 
7782   static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags,
7783                                      OpenMPOffloadMappingFlags MemberOfFlag) {
7784     // If the entry is PTR_AND_OBJ but has not been marked with the special
7785     // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be
7786     // marked as MEMBER_OF.
7787     if ((Flags & OMP_MAP_PTR_AND_OBJ) &&
7788         ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF))
7789       return;
7790 
7791     // Reset the placeholder value to prepare the flag for the assignment of the
7792     // proper MEMBER_OF value.
7793     Flags &= ~OMP_MAP_MEMBER_OF;
7794     Flags |= MemberOfFlag;
7795   }
7796 
7797   void getPlainLayout(const CXXRecordDecl *RD,
7798                       llvm::SmallVectorImpl<const FieldDecl *> &Layout,
7799                       bool AsBase) const {
7800     const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD);
7801 
7802     llvm::StructType *St =
7803         AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType();
7804 
7805     unsigned NumElements = St->getNumElements();
7806     llvm::SmallVector<
7807         llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4>
7808         RecordLayout(NumElements);
7809 
7810     // Fill bases.
7811     for (const auto &I : RD->bases()) {
7812       if (I.isVirtual())
7813         continue;
7814       const auto *Base = I.getType()->getAsCXXRecordDecl();
7815       // Ignore empty bases.
7816       if (Base->isEmpty() || CGF.getContext()
7817                                  .getASTRecordLayout(Base)
7818                                  .getNonVirtualSize()
7819                                  .isZero())
7820         continue;
7821 
7822       unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base);
7823       RecordLayout[FieldIndex] = Base;
7824     }
7825     // Fill in virtual bases.
7826     for (const auto &I : RD->vbases()) {
7827       const auto *Base = I.getType()->getAsCXXRecordDecl();
7828       // Ignore empty bases.
7829       if (Base->isEmpty())
7830         continue;
7831       unsigned FieldIndex = RL.getVirtualBaseIndex(Base);
7832       if (RecordLayout[FieldIndex])
7833         continue;
7834       RecordLayout[FieldIndex] = Base;
7835     }
7836     // Fill in all the fields.
7837     assert(!RD->isUnion() && "Unexpected union.");
7838     for (const auto *Field : RD->fields()) {
7839       // Fill in non-bitfields. (Bitfields always use a zero pattern, which we
7840       // will fill in later.)
7841       if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) {
7842         unsigned FieldIndex = RL.getLLVMFieldNo(Field);
7843         RecordLayout[FieldIndex] = Field;
7844       }
7845     }
7846     for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>
7847              &Data : RecordLayout) {
7848       if (Data.isNull())
7849         continue;
7850       if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>())
7851         getPlainLayout(Base, Layout, /*AsBase=*/true);
7852       else
7853         Layout.push_back(Data.get<const FieldDecl *>());
7854     }
7855   }
7856 
7857 public:
7858   MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF)
7859       : CurDir(&Dir), CGF(CGF) {
7860     // Extract firstprivate clause information.
7861     for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>())
7862       for (const auto *D : C->varlists())
7863         FirstPrivateDecls.try_emplace(
7864             cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit());
7865     // Extract device pointer clause information.
7866     for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
7867       for (auto L : C->component_lists())
7868         DevPointersMap[L.first].push_back(L.second);
7869   }
7870 
7871   /// Constructor for the declare mapper directive.
7872   MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF)
7873       : CurDir(&Dir), CGF(CGF) {}
7874 
7875   /// Generate code for the combined entry if we have a partially mapped struct
7876   /// and take care of the mapping flags of the arguments corresponding to
7877   /// individual struct members.
7878   void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers,
7879                          MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
7880                          MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes,
7881                          const StructRangeInfoTy &PartialStruct) const {
7882     // Base is the base of the struct
7883     BasePointers.push_back(PartialStruct.Base.getPointer());
7884     // Pointer is the address of the lowest element
7885     llvm::Value *LB = PartialStruct.LowestElem.second.getPointer();
7886     Pointers.push_back(LB);
7887     // Size is (addr of {highest+1} element) - (addr of lowest element)
7888     llvm::Value *HB = PartialStruct.HighestElem.second.getPointer();
7889     llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1);
7890     llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy);
7891     llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy);
7892     llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr);
7893     llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty,
7894                                                   /*isSigned=*/false);
7895     Sizes.push_back(Size);
7896     // Map type is always TARGET_PARAM
7897     Types.push_back(OMP_MAP_TARGET_PARAM);
7898     // Remove TARGET_PARAM flag from the first element
7899     (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM;
7900 
7901     // All other current entries will be MEMBER_OF the combined entry
7902     // (except for PTR_AND_OBJ entries which do not have a placeholder value
7903     // 0xFFFF in the MEMBER_OF field).
7904     OpenMPOffloadMappingFlags MemberOfFlag =
7905         getMemberOfFlag(BasePointers.size() - 1);
7906     for (auto &M : CurTypes)
7907       setCorrectMemberOfFlag(M, MemberOfFlag);
7908   }
7909 
7910   /// Generate all the base pointers, section pointers, sizes and map
7911   /// types for the extracted mappable expressions. Also, for each item that
7912   /// relates with a device pointer, a pair of the relevant declaration and
7913   /// index where it occurs is appended to the device pointers info array.
7914   void generateAllInfo(MapBaseValuesArrayTy &BasePointers,
7915                        MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
7916                        MapFlagsArrayTy &Types) const {
7917     // We have to process the component lists that relate with the same
7918     // declaration in a single chunk so that we can generate the map flags
7919     // correctly. Therefore, we organize all lists in a map.
7920     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
7921 
7922     // Helper function to fill the information map for the different supported
7923     // clauses.
7924     auto &&InfoGen = [&Info](
7925         const ValueDecl *D,
7926         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
7927         OpenMPMapClauseKind MapType,
7928         ArrayRef<OpenMPMapModifierKind> MapModifiers,
7929         bool ReturnDevicePointer, bool IsImplicit) {
7930       const ValueDecl *VD =
7931           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
7932       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
7933                             IsImplicit);
7934     };
7935 
7936     assert(CurDir.is<const OMPExecutableDirective *>() &&
7937            "Expect a executable directive");
7938     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
7939     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>())
7940       for (const auto L : C->component_lists()) {
7941         InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(),
7942             /*ReturnDevicePointer=*/false, C->isImplicit());
7943       }
7944     for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>())
7945       for (const auto L : C->component_lists()) {
7946         InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None,
7947             /*ReturnDevicePointer=*/false, C->isImplicit());
7948       }
7949     for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>())
7950       for (const auto L : C->component_lists()) {
7951         InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None,
7952             /*ReturnDevicePointer=*/false, C->isImplicit());
7953       }
7954 
7955     // Look at the use_device_ptr clause information and mark the existing map
7956     // entries as such. If there is no map information for an entry in the
7957     // use_device_ptr list, we create one with map type 'alloc' and zero size
7958     // section. It is the user fault if that was not mapped before. If there is
7959     // no map information and the pointer is a struct member, then we defer the
7960     // emission of that entry until the whole struct has been processed.
7961     llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>>
7962         DeferredInfo;
7963 
7964     for (const auto *C :
7965          CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) {
7966       for (const auto L : C->component_lists()) {
7967         assert(!L.second.empty() && "Not expecting empty list of components!");
7968         const ValueDecl *VD = L.second.back().getAssociatedDeclaration();
7969         VD = cast<ValueDecl>(VD->getCanonicalDecl());
7970         const Expr *IE = L.second.back().getAssociatedExpression();
7971         // If the first component is a member expression, we have to look into
7972         // 'this', which maps to null in the map of map information. Otherwise
7973         // look directly for the information.
7974         auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD);
7975 
7976         // We potentially have map information for this declaration already.
7977         // Look for the first set of components that refer to it.
7978         if (It != Info.end()) {
7979           auto CI = std::find_if(
7980               It->second.begin(), It->second.end(), [VD](const MapInfo &MI) {
7981                 return MI.Components.back().getAssociatedDeclaration() == VD;
7982               });
7983           // If we found a map entry, signal that the pointer has to be returned
7984           // and move on to the next declaration.
7985           if (CI != It->second.end()) {
7986             CI->ReturnDevicePointer = true;
7987             continue;
7988           }
7989         }
7990 
7991         // We didn't find any match in our map information - generate a zero
7992         // size array section - if the pointer is a struct member we defer this
7993         // action until the whole struct has been processed.
7994         if (isa<MemberExpr>(IE)) {
7995           // Insert the pointer into Info to be processed by
7996           // generateInfoForComponentList. Because it is a member pointer
7997           // without a pointee, no entry will be generated for it, therefore
7998           // we need to generate one after the whole struct has been processed.
7999           // Nonetheless, generateInfoForComponentList must be called to take
8000           // the pointer into account for the calculation of the range of the
8001           // partial struct.
8002           InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None,
8003                   /*ReturnDevicePointer=*/false, C->isImplicit());
8004           DeferredInfo[nullptr].emplace_back(IE, VD);
8005         } else {
8006           llvm::Value *Ptr =
8007               CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc());
8008           BasePointers.emplace_back(Ptr, VD);
8009           Pointers.push_back(Ptr);
8010           Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8011           Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM);
8012         }
8013       }
8014     }
8015 
8016     for (const auto &M : Info) {
8017       // We need to know when we generate information for the first component
8018       // associated with a capture, because the mapping flags depend on it.
8019       bool IsFirstComponentList = true;
8020 
8021       // Temporary versions of arrays
8022       MapBaseValuesArrayTy CurBasePointers;
8023       MapValuesArrayTy CurPointers;
8024       MapValuesArrayTy CurSizes;
8025       MapFlagsArrayTy CurTypes;
8026       StructRangeInfoTy PartialStruct;
8027 
8028       for (const MapInfo &L : M.second) {
8029         assert(!L.Components.empty() &&
8030                "Not expecting declaration with no component lists.");
8031 
8032         // Remember the current base pointer index.
8033         unsigned CurrentBasePointersIdx = CurBasePointers.size();
8034         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8035                                      CurBasePointers, CurPointers, CurSizes,
8036                                      CurTypes, PartialStruct,
8037                                      IsFirstComponentList, L.IsImplicit);
8038 
8039         // If this entry relates with a device pointer, set the relevant
8040         // declaration and add the 'return pointer' flag.
8041         if (L.ReturnDevicePointer) {
8042           assert(CurBasePointers.size() > CurrentBasePointersIdx &&
8043                  "Unexpected number of mapped base pointers.");
8044 
8045           const ValueDecl *RelevantVD =
8046               L.Components.back().getAssociatedDeclaration();
8047           assert(RelevantVD &&
8048                  "No relevant declaration related with device pointer??");
8049 
8050           CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD);
8051           CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM;
8052         }
8053         IsFirstComponentList = false;
8054       }
8055 
8056       // Append any pending zero-length pointers which are struct members and
8057       // used with use_device_ptr.
8058       auto CI = DeferredInfo.find(M.first);
8059       if (CI != DeferredInfo.end()) {
8060         for (const DeferredDevicePtrEntryTy &L : CI->second) {
8061           llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF);
8062           llvm::Value *Ptr = this->CGF.EmitLoadOfScalar(
8063               this->CGF.EmitLValue(L.IE), L.IE->getExprLoc());
8064           CurBasePointers.emplace_back(BasePtr, L.VD);
8065           CurPointers.push_back(Ptr);
8066           CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty));
8067           // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder
8068           // value MEMBER_OF=FFFF so that the entry is later updated with the
8069           // correct value of MEMBER_OF.
8070           CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM |
8071                              OMP_MAP_MEMBER_OF);
8072         }
8073       }
8074 
8075       // If there is an entry in PartialStruct it means we have a struct with
8076       // individual members mapped. Emit an extra combined entry.
8077       if (PartialStruct.Base.isValid())
8078         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8079                           PartialStruct);
8080 
8081       // We need to append the results of this capture to what we already have.
8082       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8083       Pointers.append(CurPointers.begin(), CurPointers.end());
8084       Sizes.append(CurSizes.begin(), CurSizes.end());
8085       Types.append(CurTypes.begin(), CurTypes.end());
8086     }
8087   }
8088 
8089   /// Generate all the base pointers, section pointers, sizes and map types for
8090   /// the extracted map clauses of user-defined mapper.
8091   void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers,
8092                                 MapValuesArrayTy &Pointers,
8093                                 MapValuesArrayTy &Sizes,
8094                                 MapFlagsArrayTy &Types) const {
8095     assert(CurDir.is<const OMPDeclareMapperDecl *>() &&
8096            "Expect a declare mapper directive");
8097     const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>();
8098     // We have to process the component lists that relate with the same
8099     // declaration in a single chunk so that we can generate the map flags
8100     // correctly. Therefore, we organize all lists in a map.
8101     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
8102 
8103     // Helper function to fill the information map for the different supported
8104     // clauses.
8105     auto &&InfoGen = [&Info](
8106         const ValueDecl *D,
8107         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8108         OpenMPMapClauseKind MapType,
8109         ArrayRef<OpenMPMapModifierKind> MapModifiers,
8110         bool ReturnDevicePointer, bool IsImplicit) {
8111       const ValueDecl *VD =
8112           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
8113       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
8114                             IsImplicit);
8115     };
8116 
8117     for (const auto *C : CurMapperDir->clauselists()) {
8118       const auto *MC = cast<OMPMapClause>(C);
8119       for (const auto L : MC->component_lists()) {
8120         InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(),
8121                 /*ReturnDevicePointer=*/false, MC->isImplicit());
8122       }
8123     }
8124 
8125     for (const auto &M : Info) {
8126       // We need to know when we generate information for the first component
8127       // associated with a capture, because the mapping flags depend on it.
8128       bool IsFirstComponentList = true;
8129 
8130       // Temporary versions of arrays
8131       MapBaseValuesArrayTy CurBasePointers;
8132       MapValuesArrayTy CurPointers;
8133       MapValuesArrayTy CurSizes;
8134       MapFlagsArrayTy CurTypes;
8135       StructRangeInfoTy PartialStruct;
8136 
8137       for (const MapInfo &L : M.second) {
8138         assert(!L.Components.empty() &&
8139                "Not expecting declaration with no component lists.");
8140         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8141                                      CurBasePointers, CurPointers, CurSizes,
8142                                      CurTypes, PartialStruct,
8143                                      IsFirstComponentList, L.IsImplicit);
8144         IsFirstComponentList = false;
8145       }
8146 
8147       // If there is an entry in PartialStruct it means we have a struct with
8148       // individual members mapped. Emit an extra combined entry.
8149       if (PartialStruct.Base.isValid())
8150         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8151                           PartialStruct);
8152 
8153       // We need to append the results of this capture to what we already have.
8154       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8155       Pointers.append(CurPointers.begin(), CurPointers.end());
8156       Sizes.append(CurSizes.begin(), CurSizes.end());
8157       Types.append(CurTypes.begin(), CurTypes.end());
8158     }
8159   }
8160 
8161   /// Emit capture info for lambdas for variables captured by reference.
8162   void generateInfoForLambdaCaptures(
8163       const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers,
8164       MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8165       MapFlagsArrayTy &Types,
8166       llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const {
8167     const auto *RD = VD->getType()
8168                          .getCanonicalType()
8169                          .getNonReferenceType()
8170                          ->getAsCXXRecordDecl();
8171     if (!RD || !RD->isLambda())
8172       return;
8173     Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD));
8174     LValue VDLVal = CGF.MakeAddrLValue(
8175         VDAddr, VD->getType().getCanonicalType().getNonReferenceType());
8176     llvm::DenseMap<const VarDecl *, FieldDecl *> Captures;
8177     FieldDecl *ThisCapture = nullptr;
8178     RD->getCaptureFields(Captures, ThisCapture);
8179     if (ThisCapture) {
8180       LValue ThisLVal =
8181           CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture);
8182       LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture);
8183       LambdaPointers.try_emplace(ThisLVal.getPointer(CGF),
8184                                  VDLVal.getPointer(CGF));
8185       BasePointers.push_back(ThisLVal.getPointer(CGF));
8186       Pointers.push_back(ThisLValVal.getPointer(CGF));
8187       Sizes.push_back(
8188           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8189                                     CGF.Int64Ty, /*isSigned=*/true));
8190       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8191                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8192     }
8193     for (const LambdaCapture &LC : RD->captures()) {
8194       if (!LC.capturesVariable())
8195         continue;
8196       const VarDecl *VD = LC.getCapturedVar();
8197       if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType())
8198         continue;
8199       auto It = Captures.find(VD);
8200       assert(It != Captures.end() && "Found lambda capture without field.");
8201       LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second);
8202       if (LC.getCaptureKind() == LCK_ByRef) {
8203         LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second);
8204         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
8205                                    VDLVal.getPointer(CGF));
8206         BasePointers.push_back(VarLVal.getPointer(CGF));
8207         Pointers.push_back(VarLValVal.getPointer(CGF));
8208         Sizes.push_back(CGF.Builder.CreateIntCast(
8209             CGF.getTypeSize(
8210                 VD->getType().getCanonicalType().getNonReferenceType()),
8211             CGF.Int64Ty, /*isSigned=*/true));
8212       } else {
8213         RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation());
8214         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
8215                                    VDLVal.getPointer(CGF));
8216         BasePointers.push_back(VarLVal.getPointer(CGF));
8217         Pointers.push_back(VarRVal.getScalarVal());
8218         Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0));
8219       }
8220       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8221                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8222     }
8223   }
8224 
8225   /// Set correct indices for lambdas captures.
8226   void adjustMemberOfForLambdaCaptures(
8227       const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers,
8228       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
8229       MapFlagsArrayTy &Types) const {
8230     for (unsigned I = 0, E = Types.size(); I < E; ++I) {
8231       // Set correct member_of idx for all implicit lambda captures.
8232       if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8233                        OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT))
8234         continue;
8235       llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]);
8236       assert(BasePtr && "Unable to find base lambda address.");
8237       int TgtIdx = -1;
8238       for (unsigned J = I; J > 0; --J) {
8239         unsigned Idx = J - 1;
8240         if (Pointers[Idx] != BasePtr)
8241           continue;
8242         TgtIdx = Idx;
8243         break;
8244       }
8245       assert(TgtIdx != -1 && "Unable to find parent lambda.");
8246       // All other current entries will be MEMBER_OF the combined entry
8247       // (except for PTR_AND_OBJ entries which do not have a placeholder value
8248       // 0xFFFF in the MEMBER_OF field).
8249       OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx);
8250       setCorrectMemberOfFlag(Types[I], MemberOfFlag);
8251     }
8252   }
8253 
8254   /// Generate the base pointers, section pointers, sizes and map types
8255   /// associated to a given capture.
8256   void generateInfoForCapture(const CapturedStmt::Capture *Cap,
8257                               llvm::Value *Arg,
8258                               MapBaseValuesArrayTy &BasePointers,
8259                               MapValuesArrayTy &Pointers,
8260                               MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
8261                               StructRangeInfoTy &PartialStruct) const {
8262     assert(!Cap->capturesVariableArrayType() &&
8263            "Not expecting to generate map info for a variable array type!");
8264 
8265     // We need to know when we generating information for the first component
8266     const ValueDecl *VD = Cap->capturesThis()
8267                               ? nullptr
8268                               : Cap->getCapturedVar()->getCanonicalDecl();
8269 
8270     // If this declaration appears in a is_device_ptr clause we just have to
8271     // pass the pointer by value. If it is a reference to a declaration, we just
8272     // pass its value.
8273     if (DevPointersMap.count(VD)) {
8274       BasePointers.emplace_back(Arg, VD);
8275       Pointers.push_back(Arg);
8276       Sizes.push_back(
8277           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8278                                     CGF.Int64Ty, /*isSigned=*/true));
8279       Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM);
8280       return;
8281     }
8282 
8283     using MapData =
8284         std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef,
8285                    OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>;
8286     SmallVector<MapData, 4> DeclComponentLists;
8287     assert(CurDir.is<const OMPExecutableDirective *>() &&
8288            "Expect a executable directive");
8289     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8290     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8291       for (const auto L : C->decl_component_lists(VD)) {
8292         assert(L.first == VD &&
8293                "We got information for the wrong declaration??");
8294         assert(!L.second.empty() &&
8295                "Not expecting declaration with no component lists.");
8296         DeclComponentLists.emplace_back(L.second, C->getMapType(),
8297                                         C->getMapTypeModifiers(),
8298                                         C->isImplicit());
8299       }
8300     }
8301 
8302     // Find overlapping elements (including the offset from the base element).
8303     llvm::SmallDenseMap<
8304         const MapData *,
8305         llvm::SmallVector<
8306             OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>,
8307         4>
8308         OverlappedData;
8309     size_t Count = 0;
8310     for (const MapData &L : DeclComponentLists) {
8311       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8312       OpenMPMapClauseKind MapType;
8313       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8314       bool IsImplicit;
8315       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8316       ++Count;
8317       for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) {
8318         OMPClauseMappableExprCommon::MappableExprComponentListRef Components1;
8319         std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1;
8320         auto CI = Components.rbegin();
8321         auto CE = Components.rend();
8322         auto SI = Components1.rbegin();
8323         auto SE = Components1.rend();
8324         for (; CI != CE && SI != SE; ++CI, ++SI) {
8325           if (CI->getAssociatedExpression()->getStmtClass() !=
8326               SI->getAssociatedExpression()->getStmtClass())
8327             break;
8328           // Are we dealing with different variables/fields?
8329           if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration())
8330             break;
8331         }
8332         // Found overlapping if, at least for one component, reached the head of
8333         // the components list.
8334         if (CI == CE || SI == SE) {
8335           assert((CI != CE || SI != SE) &&
8336                  "Unexpected full match of the mapping components.");
8337           const MapData &BaseData = CI == CE ? L : L1;
8338           OMPClauseMappableExprCommon::MappableExprComponentListRef SubData =
8339               SI == SE ? Components : Components1;
8340           auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData);
8341           OverlappedElements.getSecond().push_back(SubData);
8342         }
8343       }
8344     }
8345     // Sort the overlapped elements for each item.
8346     llvm::SmallVector<const FieldDecl *, 4> Layout;
8347     if (!OverlappedData.empty()) {
8348       if (const auto *CRD =
8349               VD->getType().getCanonicalType()->getAsCXXRecordDecl())
8350         getPlainLayout(CRD, Layout, /*AsBase=*/false);
8351       else {
8352         const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl();
8353         Layout.append(RD->field_begin(), RD->field_end());
8354       }
8355     }
8356     for (auto &Pair : OverlappedData) {
8357       llvm::sort(
8358           Pair.getSecond(),
8359           [&Layout](
8360               OMPClauseMappableExprCommon::MappableExprComponentListRef First,
8361               OMPClauseMappableExprCommon::MappableExprComponentListRef
8362                   Second) {
8363             auto CI = First.rbegin();
8364             auto CE = First.rend();
8365             auto SI = Second.rbegin();
8366             auto SE = Second.rend();
8367             for (; CI != CE && SI != SE; ++CI, ++SI) {
8368               if (CI->getAssociatedExpression()->getStmtClass() !=
8369                   SI->getAssociatedExpression()->getStmtClass())
8370                 break;
8371               // Are we dealing with different variables/fields?
8372               if (CI->getAssociatedDeclaration() !=
8373                   SI->getAssociatedDeclaration())
8374                 break;
8375             }
8376 
8377             // Lists contain the same elements.
8378             if (CI == CE && SI == SE)
8379               return false;
8380 
8381             // List with less elements is less than list with more elements.
8382             if (CI == CE || SI == SE)
8383               return CI == CE;
8384 
8385             const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration());
8386             const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration());
8387             if (FD1->getParent() == FD2->getParent())
8388               return FD1->getFieldIndex() < FD2->getFieldIndex();
8389             const auto It =
8390                 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) {
8391                   return FD == FD1 || FD == FD2;
8392                 });
8393             return *It == FD1;
8394           });
8395     }
8396 
8397     // Associated with a capture, because the mapping flags depend on it.
8398     // Go through all of the elements with the overlapped elements.
8399     for (const auto &Pair : OverlappedData) {
8400       const MapData &L = *Pair.getFirst();
8401       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8402       OpenMPMapClauseKind MapType;
8403       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8404       bool IsImplicit;
8405       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8406       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
8407           OverlappedComponents = Pair.getSecond();
8408       bool IsFirstComponentList = true;
8409       generateInfoForComponentList(MapType, MapModifiers, Components,
8410                                    BasePointers, Pointers, Sizes, Types,
8411                                    PartialStruct, IsFirstComponentList,
8412                                    IsImplicit, OverlappedComponents);
8413     }
8414     // Go through other elements without overlapped elements.
8415     bool IsFirstComponentList = OverlappedData.empty();
8416     for (const MapData &L : DeclComponentLists) {
8417       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8418       OpenMPMapClauseKind MapType;
8419       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8420       bool IsImplicit;
8421       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8422       auto It = OverlappedData.find(&L);
8423       if (It == OverlappedData.end())
8424         generateInfoForComponentList(MapType, MapModifiers, Components,
8425                                      BasePointers, Pointers, Sizes, Types,
8426                                      PartialStruct, IsFirstComponentList,
8427                                      IsImplicit);
8428       IsFirstComponentList = false;
8429     }
8430   }
8431 
8432   /// Generate the base pointers, section pointers, sizes and map types
8433   /// associated with the declare target link variables.
8434   void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers,
8435                                         MapValuesArrayTy &Pointers,
8436                                         MapValuesArrayTy &Sizes,
8437                                         MapFlagsArrayTy &Types) const {
8438     assert(CurDir.is<const OMPExecutableDirective *>() &&
8439            "Expect a executable directive");
8440     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8441     // Map other list items in the map clause which are not captured variables
8442     // but "declare target link" global variables.
8443     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8444       for (const auto L : C->component_lists()) {
8445         if (!L.first)
8446           continue;
8447         const auto *VD = dyn_cast<VarDecl>(L.first);
8448         if (!VD)
8449           continue;
8450         llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
8451             OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
8452         if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
8453             !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link)
8454           continue;
8455         StructRangeInfoTy PartialStruct;
8456         generateInfoForComponentList(
8457             C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers,
8458             Pointers, Sizes, Types, PartialStruct,
8459             /*IsFirstComponentList=*/true, C->isImplicit());
8460         assert(!PartialStruct.Base.isValid() &&
8461                "No partial structs for declare target link expected.");
8462       }
8463     }
8464   }
8465 
8466   /// Generate the default map information for a given capture \a CI,
8467   /// record field declaration \a RI and captured value \a CV.
8468   void generateDefaultMapInfo(const CapturedStmt::Capture &CI,
8469                               const FieldDecl &RI, llvm::Value *CV,
8470                               MapBaseValuesArrayTy &CurBasePointers,
8471                               MapValuesArrayTy &CurPointers,
8472                               MapValuesArrayTy &CurSizes,
8473                               MapFlagsArrayTy &CurMapTypes) const {
8474     bool IsImplicit = true;
8475     // Do the default mapping.
8476     if (CI.capturesThis()) {
8477       CurBasePointers.push_back(CV);
8478       CurPointers.push_back(CV);
8479       const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr());
8480       CurSizes.push_back(
8481           CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()),
8482                                     CGF.Int64Ty, /*isSigned=*/true));
8483       // Default map type.
8484       CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM);
8485     } else if (CI.capturesVariableByCopy()) {
8486       CurBasePointers.push_back(CV);
8487       CurPointers.push_back(CV);
8488       if (!RI.getType()->isAnyPointerType()) {
8489         // We have to signal to the runtime captures passed by value that are
8490         // not pointers.
8491         CurMapTypes.push_back(OMP_MAP_LITERAL);
8492         CurSizes.push_back(CGF.Builder.CreateIntCast(
8493             CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true));
8494       } else {
8495         // Pointers are implicitly mapped with a zero size and no flags
8496         // (other than first map that is added for all implicit maps).
8497         CurMapTypes.push_back(OMP_MAP_NONE);
8498         CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8499       }
8500       const VarDecl *VD = CI.getCapturedVar();
8501       auto I = FirstPrivateDecls.find(VD);
8502       if (I != FirstPrivateDecls.end())
8503         IsImplicit = I->getSecond();
8504     } else {
8505       assert(CI.capturesVariable() && "Expected captured reference.");
8506       const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr());
8507       QualType ElementType = PtrTy->getPointeeType();
8508       CurSizes.push_back(CGF.Builder.CreateIntCast(
8509           CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true));
8510       // The default map type for a scalar/complex type is 'to' because by
8511       // default the value doesn't have to be retrieved. For an aggregate
8512       // type, the default is 'tofrom'.
8513       CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI));
8514       const VarDecl *VD = CI.getCapturedVar();
8515       auto I = FirstPrivateDecls.find(VD);
8516       if (I != FirstPrivateDecls.end() &&
8517           VD->getType().isConstant(CGF.getContext())) {
8518         llvm::Constant *Addr =
8519             CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD);
8520         // Copy the value of the original variable to the new global copy.
8521         CGF.Builder.CreateMemCpy(
8522             CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(CGF),
8523             Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)),
8524             CurSizes.back(), /*IsVolatile=*/false);
8525         // Use new global variable as the base pointers.
8526         CurBasePointers.push_back(Addr);
8527         CurPointers.push_back(Addr);
8528       } else {
8529         CurBasePointers.push_back(CV);
8530         if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) {
8531           Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue(
8532               CV, ElementType, CGF.getContext().getDeclAlign(VD),
8533               AlignmentSource::Decl));
8534           CurPointers.push_back(PtrAddr.getPointer());
8535         } else {
8536           CurPointers.push_back(CV);
8537         }
8538       }
8539       if (I != FirstPrivateDecls.end())
8540         IsImplicit = I->getSecond();
8541     }
8542     // Every default map produces a single argument which is a target parameter.
8543     CurMapTypes.back() |= OMP_MAP_TARGET_PARAM;
8544 
8545     // Add flag stating this is an implicit map.
8546     if (IsImplicit)
8547       CurMapTypes.back() |= OMP_MAP_IMPLICIT;
8548   }
8549 };
8550 } // anonymous namespace
8551 
8552 /// Emit the arrays used to pass the captures and map information to the
8553 /// offloading runtime library. If there is no map or capture information,
8554 /// return nullptr by reference.
8555 static void
8556 emitOffloadingArrays(CodeGenFunction &CGF,
8557                      MappableExprsHandler::MapBaseValuesArrayTy &BasePointers,
8558                      MappableExprsHandler::MapValuesArrayTy &Pointers,
8559                      MappableExprsHandler::MapValuesArrayTy &Sizes,
8560                      MappableExprsHandler::MapFlagsArrayTy &MapTypes,
8561                      CGOpenMPRuntime::TargetDataInfo &Info) {
8562   CodeGenModule &CGM = CGF.CGM;
8563   ASTContext &Ctx = CGF.getContext();
8564 
8565   // Reset the array information.
8566   Info.clearArrayInfo();
8567   Info.NumberOfPtrs = BasePointers.size();
8568 
8569   if (Info.NumberOfPtrs) {
8570     // Detect if we have any capture size requiring runtime evaluation of the
8571     // size so that a constant array could be eventually used.
8572     bool hasRuntimeEvaluationCaptureSize = false;
8573     for (llvm::Value *S : Sizes)
8574       if (!isa<llvm::Constant>(S)) {
8575         hasRuntimeEvaluationCaptureSize = true;
8576         break;
8577       }
8578 
8579     llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true);
8580     QualType PointerArrayType = Ctx.getConstantArrayType(
8581         Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal,
8582         /*IndexTypeQuals=*/0);
8583 
8584     Info.BasePointersArray =
8585         CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer();
8586     Info.PointersArray =
8587         CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer();
8588 
8589     // If we don't have any VLA types or other types that require runtime
8590     // evaluation, we can use a constant array for the map sizes, otherwise we
8591     // need to fill up the arrays as we do for the pointers.
8592     QualType Int64Ty =
8593         Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
8594     if (hasRuntimeEvaluationCaptureSize) {
8595       QualType SizeArrayType = Ctx.getConstantArrayType(
8596           Int64Ty, PointerNumAP, nullptr, ArrayType::Normal,
8597           /*IndexTypeQuals=*/0);
8598       Info.SizesArray =
8599           CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer();
8600     } else {
8601       // We expect all the sizes to be constant, so we collect them to create
8602       // a constant array.
8603       SmallVector<llvm::Constant *, 16> ConstSizes;
8604       for (llvm::Value *S : Sizes)
8605         ConstSizes.push_back(cast<llvm::Constant>(S));
8606 
8607       auto *SizesArrayInit = llvm::ConstantArray::get(
8608           llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes);
8609       std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"});
8610       auto *SizesArrayGbl = new llvm::GlobalVariable(
8611           CGM.getModule(), SizesArrayInit->getType(),
8612           /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8613           SizesArrayInit, Name);
8614       SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8615       Info.SizesArray = SizesArrayGbl;
8616     }
8617 
8618     // The map types are always constant so we don't need to generate code to
8619     // fill arrays. Instead, we create an array constant.
8620     SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0);
8621     llvm::copy(MapTypes, Mapping.begin());
8622     llvm::Constant *MapTypesArrayInit =
8623         llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping);
8624     std::string MaptypesName =
8625         CGM.getOpenMPRuntime().getName({"offload_maptypes"});
8626     auto *MapTypesArrayGbl = new llvm::GlobalVariable(
8627         CGM.getModule(), MapTypesArrayInit->getType(),
8628         /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8629         MapTypesArrayInit, MaptypesName);
8630     MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8631     Info.MapTypesArray = MapTypesArrayGbl;
8632 
8633     for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) {
8634       llvm::Value *BPVal = *BasePointers[I];
8635       llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32(
8636           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8637           Info.BasePointersArray, 0, I);
8638       BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8639           BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0));
8640       Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8641       CGF.Builder.CreateStore(BPVal, BPAddr);
8642 
8643       if (Info.requiresDevicePointerInfo())
8644         if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl())
8645           Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr);
8646 
8647       llvm::Value *PVal = Pointers[I];
8648       llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32(
8649           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8650           Info.PointersArray, 0, I);
8651       P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8652           P, PVal->getType()->getPointerTo(/*AddrSpace=*/0));
8653       Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8654       CGF.Builder.CreateStore(PVal, PAddr);
8655 
8656       if (hasRuntimeEvaluationCaptureSize) {
8657         llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32(
8658             llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8659             Info.SizesArray,
8660             /*Idx0=*/0,
8661             /*Idx1=*/I);
8662         Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty));
8663         CGF.Builder.CreateStore(
8664             CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true),
8665             SAddr);
8666       }
8667     }
8668   }
8669 }
8670 
8671 /// Emit the arguments to be passed to the runtime library based on the
8672 /// arrays of pointers, sizes and map types.
8673 static void emitOffloadingArraysArgument(
8674     CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg,
8675     llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg,
8676     llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) {
8677   CodeGenModule &CGM = CGF.CGM;
8678   if (Info.NumberOfPtrs) {
8679     BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8680         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8681         Info.BasePointersArray,
8682         /*Idx0=*/0, /*Idx1=*/0);
8683     PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8684         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8685         Info.PointersArray,
8686         /*Idx0=*/0,
8687         /*Idx1=*/0);
8688     SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8689         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray,
8690         /*Idx0=*/0, /*Idx1=*/0);
8691     MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8692         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8693         Info.MapTypesArray,
8694         /*Idx0=*/0,
8695         /*Idx1=*/0);
8696   } else {
8697     BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8698     PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8699     SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8700     MapTypesArrayArg =
8701         llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8702   }
8703 }
8704 
8705 /// Check for inner distribute directive.
8706 static const OMPExecutableDirective *
8707 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) {
8708   const auto *CS = D.getInnermostCapturedStmt();
8709   const auto *Body =
8710       CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
8711   const Stmt *ChildStmt =
8712       CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
8713 
8714   if (const auto *NestedDir =
8715           dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
8716     OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind();
8717     switch (D.getDirectiveKind()) {
8718     case OMPD_target:
8719       if (isOpenMPDistributeDirective(DKind))
8720         return NestedDir;
8721       if (DKind == OMPD_teams) {
8722         Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers(
8723             /*IgnoreCaptured=*/true);
8724         if (!Body)
8725           return nullptr;
8726         ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
8727         if (const auto *NND =
8728                 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
8729           DKind = NND->getDirectiveKind();
8730           if (isOpenMPDistributeDirective(DKind))
8731             return NND;
8732         }
8733       }
8734       return nullptr;
8735     case OMPD_target_teams:
8736       if (isOpenMPDistributeDirective(DKind))
8737         return NestedDir;
8738       return nullptr;
8739     case OMPD_target_parallel:
8740     case OMPD_target_simd:
8741     case OMPD_target_parallel_for:
8742     case OMPD_target_parallel_for_simd:
8743       return nullptr;
8744     case OMPD_target_teams_distribute:
8745     case OMPD_target_teams_distribute_simd:
8746     case OMPD_target_teams_distribute_parallel_for:
8747     case OMPD_target_teams_distribute_parallel_for_simd:
8748     case OMPD_parallel:
8749     case OMPD_for:
8750     case OMPD_parallel_for:
8751     case OMPD_parallel_master:
8752     case OMPD_parallel_sections:
8753     case OMPD_for_simd:
8754     case OMPD_parallel_for_simd:
8755     case OMPD_cancel:
8756     case OMPD_cancellation_point:
8757     case OMPD_ordered:
8758     case OMPD_threadprivate:
8759     case OMPD_allocate:
8760     case OMPD_task:
8761     case OMPD_simd:
8762     case OMPD_sections:
8763     case OMPD_section:
8764     case OMPD_single:
8765     case OMPD_master:
8766     case OMPD_critical:
8767     case OMPD_taskyield:
8768     case OMPD_barrier:
8769     case OMPD_taskwait:
8770     case OMPD_taskgroup:
8771     case OMPD_atomic:
8772     case OMPD_flush:
8773     case OMPD_teams:
8774     case OMPD_target_data:
8775     case OMPD_target_exit_data:
8776     case OMPD_target_enter_data:
8777     case OMPD_distribute:
8778     case OMPD_distribute_simd:
8779     case OMPD_distribute_parallel_for:
8780     case OMPD_distribute_parallel_for_simd:
8781     case OMPD_teams_distribute:
8782     case OMPD_teams_distribute_simd:
8783     case OMPD_teams_distribute_parallel_for:
8784     case OMPD_teams_distribute_parallel_for_simd:
8785     case OMPD_target_update:
8786     case OMPD_declare_simd:
8787     case OMPD_declare_variant:
8788     case OMPD_declare_target:
8789     case OMPD_end_declare_target:
8790     case OMPD_declare_reduction:
8791     case OMPD_declare_mapper:
8792     case OMPD_taskloop:
8793     case OMPD_taskloop_simd:
8794     case OMPD_master_taskloop:
8795     case OMPD_master_taskloop_simd:
8796     case OMPD_parallel_master_taskloop:
8797     case OMPD_parallel_master_taskloop_simd:
8798     case OMPD_requires:
8799     case OMPD_unknown:
8800       llvm_unreachable("Unexpected directive.");
8801     }
8802   }
8803 
8804   return nullptr;
8805 }
8806 
8807 /// Emit the user-defined mapper function. The code generation follows the
8808 /// pattern in the example below.
8809 /// \code
8810 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle,
8811 ///                                           void *base, void *begin,
8812 ///                                           int64_t size, int64_t type) {
8813 ///   // Allocate space for an array section first.
8814 ///   if (size > 1 && !maptype.IsDelete)
8815 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
8816 ///                                 size*sizeof(Ty), clearToFrom(type));
8817 ///   // Map members.
8818 ///   for (unsigned i = 0; i < size; i++) {
8819 ///     // For each component specified by this mapper:
8820 ///     for (auto c : all_components) {
8821 ///       if (c.hasMapper())
8822 ///         (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size,
8823 ///                       c.arg_type);
8824 ///       else
8825 ///         __tgt_push_mapper_component(rt_mapper_handle, c.arg_base,
8826 ///                                     c.arg_begin, c.arg_size, c.arg_type);
8827 ///     }
8828 ///   }
8829 ///   // Delete the array section.
8830 ///   if (size > 1 && maptype.IsDelete)
8831 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
8832 ///                                 size*sizeof(Ty), clearToFrom(type));
8833 /// }
8834 /// \endcode
8835 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D,
8836                                             CodeGenFunction *CGF) {
8837   if (UDMMap.count(D) > 0)
8838     return;
8839   ASTContext &C = CGM.getContext();
8840   QualType Ty = D->getType();
8841   QualType PtrTy = C.getPointerType(Ty).withRestrict();
8842   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
8843   auto *MapperVarDecl =
8844       cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl());
8845   SourceLocation Loc = D->getLocation();
8846   CharUnits ElementSize = C.getTypeSizeInChars(Ty);
8847 
8848   // Prepare mapper function arguments and attributes.
8849   ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
8850                               C.VoidPtrTy, ImplicitParamDecl::Other);
8851   ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
8852                             ImplicitParamDecl::Other);
8853   ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
8854                              C.VoidPtrTy, ImplicitParamDecl::Other);
8855   ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
8856                             ImplicitParamDecl::Other);
8857   ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
8858                             ImplicitParamDecl::Other);
8859   FunctionArgList Args;
8860   Args.push_back(&HandleArg);
8861   Args.push_back(&BaseArg);
8862   Args.push_back(&BeginArg);
8863   Args.push_back(&SizeArg);
8864   Args.push_back(&TypeArg);
8865   const CGFunctionInfo &FnInfo =
8866       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
8867   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
8868   SmallString<64> TyStr;
8869   llvm::raw_svector_ostream Out(TyStr);
8870   CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out);
8871   std::string Name = getName({"omp_mapper", TyStr, D->getName()});
8872   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
8873                                     Name, &CGM.getModule());
8874   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
8875   Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
8876   // Start the mapper function code generation.
8877   CodeGenFunction MapperCGF(CGM);
8878   MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
8879   // Compute the starting and end addreses of array elements.
8880   llvm::Value *Size = MapperCGF.EmitLoadOfScalar(
8881       MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false,
8882       C.getPointerType(Int64Ty), Loc);
8883   llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast(
8884       MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(),
8885       CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy)));
8886   llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size);
8887   llvm::Value *MapType = MapperCGF.EmitLoadOfScalar(
8888       MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false,
8889       C.getPointerType(Int64Ty), Loc);
8890   // Prepare common arguments for array initiation and deletion.
8891   llvm::Value *Handle = MapperCGF.EmitLoadOfScalar(
8892       MapperCGF.GetAddrOfLocalVar(&HandleArg),
8893       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
8894   llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar(
8895       MapperCGF.GetAddrOfLocalVar(&BaseArg),
8896       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
8897   llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar(
8898       MapperCGF.GetAddrOfLocalVar(&BeginArg),
8899       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
8900 
8901   // Emit array initiation if this is an array section and \p MapType indicates
8902   // that memory allocation is required.
8903   llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head");
8904   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
8905                              ElementSize, HeadBB, /*IsInit=*/true);
8906 
8907   // Emit a for loop to iterate through SizeArg of elements and map all of them.
8908 
8909   // Emit the loop header block.
8910   MapperCGF.EmitBlock(HeadBB);
8911   llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body");
8912   llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done");
8913   // Evaluate whether the initial condition is satisfied.
8914   llvm::Value *IsEmpty =
8915       MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty");
8916   MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
8917   llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock();
8918 
8919   // Emit the loop body block.
8920   MapperCGF.EmitBlock(BodyBB);
8921   llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI(
8922       PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent");
8923   PtrPHI->addIncoming(PtrBegin, EntryBB);
8924   Address PtrCurrent =
8925       Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg)
8926                           .getAlignment()
8927                           .alignmentOfArrayElement(ElementSize));
8928   // Privatize the declared variable of mapper to be the current array element.
8929   CodeGenFunction::OMPPrivateScope Scope(MapperCGF);
8930   Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() {
8931     return MapperCGF
8932         .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>())
8933         .getAddress(MapperCGF);
8934   });
8935   (void)Scope.Privatize();
8936 
8937   // Get map clause information. Fill up the arrays with all mapped variables.
8938   MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
8939   MappableExprsHandler::MapValuesArrayTy Pointers;
8940   MappableExprsHandler::MapValuesArrayTy Sizes;
8941   MappableExprsHandler::MapFlagsArrayTy MapTypes;
8942   MappableExprsHandler MEHandler(*D, MapperCGF);
8943   MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes);
8944 
8945   // Call the runtime API __tgt_mapper_num_components to get the number of
8946   // pre-existing components.
8947   llvm::Value *OffloadingArgs[] = {Handle};
8948   llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall(
8949       createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs);
8950   llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl(
8951       PreviousSize,
8952       MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset()));
8953 
8954   // Fill up the runtime mapper handle for all components.
8955   for (unsigned I = 0; I < BasePointers.size(); ++I) {
8956     llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast(
8957         *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
8958     llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast(
8959         Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
8960     llvm::Value *CurSizeArg = Sizes[I];
8961 
8962     // Extract the MEMBER_OF field from the map type.
8963     llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member");
8964     MapperCGF.EmitBlock(MemberBB);
8965     llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]);
8966     llvm::Value *Member = MapperCGF.Builder.CreateAnd(
8967         OriMapType,
8968         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF));
8969     llvm::BasicBlock *MemberCombineBB =
8970         MapperCGF.createBasicBlock("omp.member.combine");
8971     llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type");
8972     llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member);
8973     MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB);
8974     // Add the number of pre-existing components to the MEMBER_OF field if it
8975     // is valid.
8976     MapperCGF.EmitBlock(MemberCombineBB);
8977     llvm::Value *CombinedMember =
8978         MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
8979     // Do nothing if it is not a member of previous components.
8980     MapperCGF.EmitBlock(TypeBB);
8981     llvm::PHINode *MemberMapType =
8982         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype");
8983     MemberMapType->addIncoming(OriMapType, MemberBB);
8984     MemberMapType->addIncoming(CombinedMember, MemberCombineBB);
8985 
8986     // Combine the map type inherited from user-defined mapper with that
8987     // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM
8988     // bits of the \a MapType, which is the input argument of the mapper
8989     // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM
8990     // bits of MemberMapType.
8991     // [OpenMP 5.0], 1.2.6. map-type decay.
8992     //        | alloc |  to   | from  | tofrom | release | delete
8993     // ----------------------------------------------------------
8994     // alloc  | alloc | alloc | alloc | alloc  | release | delete
8995     // to     | alloc |  to   | alloc |   to   | release | delete
8996     // from   | alloc | alloc | from  |  from  | release | delete
8997     // tofrom | alloc |  to   | from  | tofrom | release | delete
8998     llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd(
8999         MapType,
9000         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO |
9001                                    MappableExprsHandler::OMP_MAP_FROM));
9002     llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc");
9003     llvm::BasicBlock *AllocElseBB =
9004         MapperCGF.createBasicBlock("omp.type.alloc.else");
9005     llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to");
9006     llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else");
9007     llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from");
9008     llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end");
9009     llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom);
9010     MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
9011     // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM.
9012     MapperCGF.EmitBlock(AllocBB);
9013     llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd(
9014         MemberMapType,
9015         MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9016                                      MappableExprsHandler::OMP_MAP_FROM)));
9017     MapperCGF.Builder.CreateBr(EndBB);
9018     MapperCGF.EmitBlock(AllocElseBB);
9019     llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ(
9020         LeftToFrom,
9021         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO));
9022     MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
9023     // In case of to, clear OMP_MAP_FROM.
9024     MapperCGF.EmitBlock(ToBB);
9025     llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd(
9026         MemberMapType,
9027         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM));
9028     MapperCGF.Builder.CreateBr(EndBB);
9029     MapperCGF.EmitBlock(ToElseBB);
9030     llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ(
9031         LeftToFrom,
9032         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM));
9033     MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB);
9034     // In case of from, clear OMP_MAP_TO.
9035     MapperCGF.EmitBlock(FromBB);
9036     llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd(
9037         MemberMapType,
9038         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO));
9039     // In case of tofrom, do nothing.
9040     MapperCGF.EmitBlock(EndBB);
9041     llvm::PHINode *CurMapType =
9042         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype");
9043     CurMapType->addIncoming(AllocMapType, AllocBB);
9044     CurMapType->addIncoming(ToMapType, ToBB);
9045     CurMapType->addIncoming(FromMapType, FromBB);
9046     CurMapType->addIncoming(MemberMapType, ToElseBB);
9047 
9048     // TODO: call the corresponding mapper function if a user-defined mapper is
9049     // associated with this map clause.
9050     // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9051     // data structure.
9052     llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg,
9053                                      CurSizeArg, CurMapType};
9054     MapperCGF.EmitRuntimeCall(
9055         createRuntimeFunction(OMPRTL__tgt_push_mapper_component),
9056         OffloadingArgs);
9057   }
9058 
9059   // Update the pointer to point to the next element that needs to be mapped,
9060   // and check whether we have mapped all elements.
9061   llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32(
9062       PtrPHI, /*Idx0=*/1, "omp.arraymap.next");
9063   PtrPHI->addIncoming(PtrNext, BodyBB);
9064   llvm::Value *IsDone =
9065       MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone");
9066   llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit");
9067   MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
9068 
9069   MapperCGF.EmitBlock(ExitBB);
9070   // Emit array deletion if this is an array section and \p MapType indicates
9071   // that deletion is required.
9072   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
9073                              ElementSize, DoneBB, /*IsInit=*/false);
9074 
9075   // Emit the function exit block.
9076   MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true);
9077   MapperCGF.FinishFunction();
9078   UDMMap.try_emplace(D, Fn);
9079   if (CGF) {
9080     auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn);
9081     Decls.second.push_back(D);
9082   }
9083 }
9084 
9085 /// Emit the array initialization or deletion portion for user-defined mapper
9086 /// code generation. First, it evaluates whether an array section is mapped and
9087 /// whether the \a MapType instructs to delete this section. If \a IsInit is
9088 /// true, and \a MapType indicates to not delete this array, array
9089 /// initialization code is generated. If \a IsInit is false, and \a MapType
9090 /// indicates to not this array, array deletion code is generated.
9091 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel(
9092     CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base,
9093     llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType,
9094     CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) {
9095   StringRef Prefix = IsInit ? ".init" : ".del";
9096 
9097   // Evaluate if this is an array section.
9098   llvm::BasicBlock *IsDeleteBB =
9099       MapperCGF.createBasicBlock(getName({"omp.array", Prefix, ".evaldelete"}));
9100   llvm::BasicBlock *BodyBB =
9101       MapperCGF.createBasicBlock(getName({"omp.array", Prefix}));
9102   llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE(
9103       Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray");
9104   MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB);
9105 
9106   // Evaluate if we are going to delete this section.
9107   MapperCGF.EmitBlock(IsDeleteBB);
9108   llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd(
9109       MapType,
9110       MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE));
9111   llvm::Value *DeleteCond;
9112   if (IsInit) {
9113     DeleteCond = MapperCGF.Builder.CreateIsNull(
9114         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
9115   } else {
9116     DeleteCond = MapperCGF.Builder.CreateIsNotNull(
9117         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
9118   }
9119   MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB);
9120 
9121   MapperCGF.EmitBlock(BodyBB);
9122   // Get the array size by multiplying element size and element number (i.e., \p
9123   // Size).
9124   llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul(
9125       Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity()));
9126   // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves
9127   // memory allocation/deletion purpose only.
9128   llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd(
9129       MapType,
9130       MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9131                                    MappableExprsHandler::OMP_MAP_FROM)));
9132   // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9133   // data structure.
9134   llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg};
9135   MapperCGF.EmitRuntimeCall(
9136       createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs);
9137 }
9138 
9139 void CGOpenMPRuntime::emitTargetNumIterationsCall(
9140     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9141     llvm::Value *DeviceID,
9142     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9143                                      const OMPLoopDirective &D)>
9144         SizeEmitter) {
9145   OpenMPDirectiveKind Kind = D.getDirectiveKind();
9146   const OMPExecutableDirective *TD = &D;
9147   // Get nested teams distribute kind directive, if any.
9148   if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind))
9149     TD = getNestedDistributeDirective(CGM.getContext(), D);
9150   if (!TD)
9151     return;
9152   const auto *LD = cast<OMPLoopDirective>(TD);
9153   auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF,
9154                                                      PrePostActionTy &) {
9155     if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) {
9156       llvm::Value *Args[] = {DeviceID, NumIterations};
9157       CGF.EmitRuntimeCall(
9158           createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args);
9159     }
9160   };
9161   emitInlinedDirective(CGF, OMPD_unknown, CodeGen);
9162 }
9163 
9164 void CGOpenMPRuntime::emitTargetCall(
9165     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9166     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
9167     const Expr *Device,
9168     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9169                                      const OMPLoopDirective &D)>
9170         SizeEmitter) {
9171   if (!CGF.HaveInsertPoint())
9172     return;
9173 
9174   assert(OutlinedFn && "Invalid outlined function!");
9175 
9176   const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>();
9177   llvm::SmallVector<llvm::Value *, 16> CapturedVars;
9178   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
9179   auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF,
9180                                             PrePostActionTy &) {
9181     CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9182   };
9183   emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen);
9184 
9185   CodeGenFunction::OMPTargetDataInfo InputInfo;
9186   llvm::Value *MapTypesArray = nullptr;
9187   // Fill up the pointer arrays and transfer execution to the device.
9188   auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo,
9189                     &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars,
9190                     SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) {
9191     // On top of the arrays that were filled up, the target offloading call
9192     // takes as arguments the device id as well as the host pointer. The host
9193     // pointer is used by the runtime library to identify the current target
9194     // region, so it only has to be unique and not necessarily point to
9195     // anything. It could be the pointer to the outlined function that
9196     // implements the target region, but we aren't using that so that the
9197     // compiler doesn't need to keep that, and could therefore inline the host
9198     // function if proven worthwhile during optimization.
9199 
9200     // From this point on, we need to have an ID of the target region defined.
9201     assert(OutlinedFnID && "Invalid outlined function ID!");
9202 
9203     // Emit device ID if any.
9204     llvm::Value *DeviceID;
9205     if (Device) {
9206       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
9207                                            CGF.Int64Ty, /*isSigned=*/true);
9208     } else {
9209       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
9210     }
9211 
9212     // Emit the number of elements in the offloading arrays.
9213     llvm::Value *PointerNum =
9214         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
9215 
9216     // Return value of the runtime offloading call.
9217     llvm::Value *Return;
9218 
9219     llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D);
9220     llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D);
9221 
9222     // Emit tripcount for the target loop-based directive.
9223     emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter);
9224 
9225     bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
9226     // The target region is an outlined function launched by the runtime
9227     // via calls __tgt_target() or __tgt_target_teams().
9228     //
9229     // __tgt_target() launches a target region with one team and one thread,
9230     // executing a serial region.  This master thread may in turn launch
9231     // more threads within its team upon encountering a parallel region,
9232     // however, no additional teams can be launched on the device.
9233     //
9234     // __tgt_target_teams() launches a target region with one or more teams,
9235     // each with one or more threads.  This call is required for target
9236     // constructs such as:
9237     //  'target teams'
9238     //  'target' / 'teams'
9239     //  'target teams distribute parallel for'
9240     //  'target parallel'
9241     // and so on.
9242     //
9243     // Note that on the host and CPU targets, the runtime implementation of
9244     // these calls simply call the outlined function without forking threads.
9245     // The outlined functions themselves have runtime calls to
9246     // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by
9247     // the compiler in emitTeamsCall() and emitParallelCall().
9248     //
9249     // In contrast, on the NVPTX target, the implementation of
9250     // __tgt_target_teams() launches a GPU kernel with the requested number
9251     // of teams and threads so no additional calls to the runtime are required.
9252     if (NumTeams) {
9253       // If we have NumTeams defined this means that we have an enclosed teams
9254       // region. Therefore we also expect to have NumThreads defined. These two
9255       // values should be defined in the presence of a teams directive,
9256       // regardless of having any clauses associated. If the user is using teams
9257       // but no clauses, these two values will be the default that should be
9258       // passed to the runtime library - a 32-bit integer with the value zero.
9259       assert(NumThreads && "Thread limit expression should be available along "
9260                            "with number of teams.");
9261       llvm::Value *OffloadingArgs[] = {DeviceID,
9262                                        OutlinedFnID,
9263                                        PointerNum,
9264                                        InputInfo.BasePointersArray.getPointer(),
9265                                        InputInfo.PointersArray.getPointer(),
9266                                        InputInfo.SizesArray.getPointer(),
9267                                        MapTypesArray,
9268                                        NumTeams,
9269                                        NumThreads};
9270       Return = CGF.EmitRuntimeCall(
9271           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait
9272                                           : OMPRTL__tgt_target_teams),
9273           OffloadingArgs);
9274     } else {
9275       llvm::Value *OffloadingArgs[] = {DeviceID,
9276                                        OutlinedFnID,
9277                                        PointerNum,
9278                                        InputInfo.BasePointersArray.getPointer(),
9279                                        InputInfo.PointersArray.getPointer(),
9280                                        InputInfo.SizesArray.getPointer(),
9281                                        MapTypesArray};
9282       Return = CGF.EmitRuntimeCall(
9283           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait
9284                                           : OMPRTL__tgt_target),
9285           OffloadingArgs);
9286     }
9287 
9288     // Check the error code and execute the host version if required.
9289     llvm::BasicBlock *OffloadFailedBlock =
9290         CGF.createBasicBlock("omp_offload.failed");
9291     llvm::BasicBlock *OffloadContBlock =
9292         CGF.createBasicBlock("omp_offload.cont");
9293     llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return);
9294     CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock);
9295 
9296     CGF.EmitBlock(OffloadFailedBlock);
9297     if (RequiresOuterTask) {
9298       CapturedVars.clear();
9299       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9300     }
9301     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9302     CGF.EmitBranch(OffloadContBlock);
9303 
9304     CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true);
9305   };
9306 
9307   // Notify that the host version must be executed.
9308   auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars,
9309                     RequiresOuterTask](CodeGenFunction &CGF,
9310                                        PrePostActionTy &) {
9311     if (RequiresOuterTask) {
9312       CapturedVars.clear();
9313       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9314     }
9315     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9316   };
9317 
9318   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
9319                           &CapturedVars, RequiresOuterTask,
9320                           &CS](CodeGenFunction &CGF, PrePostActionTy &) {
9321     // Fill up the arrays with all the captured variables.
9322     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9323     MappableExprsHandler::MapValuesArrayTy Pointers;
9324     MappableExprsHandler::MapValuesArrayTy Sizes;
9325     MappableExprsHandler::MapFlagsArrayTy MapTypes;
9326 
9327     // Get mappable expression information.
9328     MappableExprsHandler MEHandler(D, CGF);
9329     llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers;
9330 
9331     auto RI = CS.getCapturedRecordDecl()->field_begin();
9332     auto CV = CapturedVars.begin();
9333     for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(),
9334                                               CE = CS.capture_end();
9335          CI != CE; ++CI, ++RI, ++CV) {
9336       MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers;
9337       MappableExprsHandler::MapValuesArrayTy CurPointers;
9338       MappableExprsHandler::MapValuesArrayTy CurSizes;
9339       MappableExprsHandler::MapFlagsArrayTy CurMapTypes;
9340       MappableExprsHandler::StructRangeInfoTy PartialStruct;
9341 
9342       // VLA sizes are passed to the outlined region by copy and do not have map
9343       // information associated.
9344       if (CI->capturesVariableArrayType()) {
9345         CurBasePointers.push_back(*CV);
9346         CurPointers.push_back(*CV);
9347         CurSizes.push_back(CGF.Builder.CreateIntCast(
9348             CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true));
9349         // Copy to the device as an argument. No need to retrieve it.
9350         CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL |
9351                               MappableExprsHandler::OMP_MAP_TARGET_PARAM |
9352                               MappableExprsHandler::OMP_MAP_IMPLICIT);
9353       } else {
9354         // If we have any information in the map clause, we use it, otherwise we
9355         // just do a default mapping.
9356         MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers,
9357                                          CurSizes, CurMapTypes, PartialStruct);
9358         if (CurBasePointers.empty())
9359           MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers,
9360                                            CurPointers, CurSizes, CurMapTypes);
9361         // Generate correct mapping for variables captured by reference in
9362         // lambdas.
9363         if (CI->capturesVariable())
9364           MEHandler.generateInfoForLambdaCaptures(
9365               CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes,
9366               CurMapTypes, LambdaPointers);
9367       }
9368       // We expect to have at least an element of information for this capture.
9369       assert(!CurBasePointers.empty() &&
9370              "Non-existing map pointer for capture!");
9371       assert(CurBasePointers.size() == CurPointers.size() &&
9372              CurBasePointers.size() == CurSizes.size() &&
9373              CurBasePointers.size() == CurMapTypes.size() &&
9374              "Inconsistent map information sizes!");
9375 
9376       // If there is an entry in PartialStruct it means we have a struct with
9377       // individual members mapped. Emit an extra combined entry.
9378       if (PartialStruct.Base.isValid())
9379         MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes,
9380                                     CurMapTypes, PartialStruct);
9381 
9382       // We need to append the results of this capture to what we already have.
9383       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
9384       Pointers.append(CurPointers.begin(), CurPointers.end());
9385       Sizes.append(CurSizes.begin(), CurSizes.end());
9386       MapTypes.append(CurMapTypes.begin(), CurMapTypes.end());
9387     }
9388     // Adjust MEMBER_OF flags for the lambdas captures.
9389     MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers,
9390                                               Pointers, MapTypes);
9391     // Map other list items in the map clause which are not captured variables
9392     // but "declare target link" global variables.
9393     MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes,
9394                                                MapTypes);
9395 
9396     TargetDataInfo Info;
9397     // Fill up the arrays and create the arguments.
9398     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
9399     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
9400                                  Info.PointersArray, Info.SizesArray,
9401                                  Info.MapTypesArray, Info);
9402     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
9403     InputInfo.BasePointersArray =
9404         Address(Info.BasePointersArray, CGM.getPointerAlign());
9405     InputInfo.PointersArray =
9406         Address(Info.PointersArray, CGM.getPointerAlign());
9407     InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign());
9408     MapTypesArray = Info.MapTypesArray;
9409     if (RequiresOuterTask)
9410       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
9411     else
9412       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
9413   };
9414 
9415   auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask](
9416                              CodeGenFunction &CGF, PrePostActionTy &) {
9417     if (RequiresOuterTask) {
9418       CodeGenFunction::OMPTargetDataInfo InputInfo;
9419       CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo);
9420     } else {
9421       emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen);
9422     }
9423   };
9424 
9425   // If we have a target function ID it means that we need to support
9426   // offloading, otherwise, just execute on the host. We need to execute on host
9427   // regardless of the conditional in the if clause if, e.g., the user do not
9428   // specify target triples.
9429   if (OutlinedFnID) {
9430     if (IfCond) {
9431       emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen);
9432     } else {
9433       RegionCodeGenTy ThenRCG(TargetThenGen);
9434       ThenRCG(CGF);
9435     }
9436   } else {
9437     RegionCodeGenTy ElseRCG(TargetElseGen);
9438     ElseRCG(CGF);
9439   }
9440 }
9441 
9442 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S,
9443                                                     StringRef ParentName) {
9444   if (!S)
9445     return;
9446 
9447   // Codegen OMP target directives that offload compute to the device.
9448   bool RequiresDeviceCodegen =
9449       isa<OMPExecutableDirective>(S) &&
9450       isOpenMPTargetExecutionDirective(
9451           cast<OMPExecutableDirective>(S)->getDirectiveKind());
9452 
9453   if (RequiresDeviceCodegen) {
9454     const auto &E = *cast<OMPExecutableDirective>(S);
9455     unsigned DeviceID;
9456     unsigned FileID;
9457     unsigned Line;
9458     getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID,
9459                              FileID, Line);
9460 
9461     // Is this a target region that should not be emitted as an entry point? If
9462     // so just signal we are done with this target region.
9463     if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID,
9464                                                             ParentName, Line))
9465       return;
9466 
9467     switch (E.getDirectiveKind()) {
9468     case OMPD_target:
9469       CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName,
9470                                                    cast<OMPTargetDirective>(E));
9471       break;
9472     case OMPD_target_parallel:
9473       CodeGenFunction::EmitOMPTargetParallelDeviceFunction(
9474           CGM, ParentName, cast<OMPTargetParallelDirective>(E));
9475       break;
9476     case OMPD_target_teams:
9477       CodeGenFunction::EmitOMPTargetTeamsDeviceFunction(
9478           CGM, ParentName, cast<OMPTargetTeamsDirective>(E));
9479       break;
9480     case OMPD_target_teams_distribute:
9481       CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction(
9482           CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E));
9483       break;
9484     case OMPD_target_teams_distribute_simd:
9485       CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction(
9486           CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E));
9487       break;
9488     case OMPD_target_parallel_for:
9489       CodeGenFunction::EmitOMPTargetParallelForDeviceFunction(
9490           CGM, ParentName, cast<OMPTargetParallelForDirective>(E));
9491       break;
9492     case OMPD_target_parallel_for_simd:
9493       CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction(
9494           CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E));
9495       break;
9496     case OMPD_target_simd:
9497       CodeGenFunction::EmitOMPTargetSimdDeviceFunction(
9498           CGM, ParentName, cast<OMPTargetSimdDirective>(E));
9499       break;
9500     case OMPD_target_teams_distribute_parallel_for:
9501       CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction(
9502           CGM, ParentName,
9503           cast<OMPTargetTeamsDistributeParallelForDirective>(E));
9504       break;
9505     case OMPD_target_teams_distribute_parallel_for_simd:
9506       CodeGenFunction::
9507           EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(
9508               CGM, ParentName,
9509               cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E));
9510       break;
9511     case OMPD_parallel:
9512     case OMPD_for:
9513     case OMPD_parallel_for:
9514     case OMPD_parallel_master:
9515     case OMPD_parallel_sections:
9516     case OMPD_for_simd:
9517     case OMPD_parallel_for_simd:
9518     case OMPD_cancel:
9519     case OMPD_cancellation_point:
9520     case OMPD_ordered:
9521     case OMPD_threadprivate:
9522     case OMPD_allocate:
9523     case OMPD_task:
9524     case OMPD_simd:
9525     case OMPD_sections:
9526     case OMPD_section:
9527     case OMPD_single:
9528     case OMPD_master:
9529     case OMPD_critical:
9530     case OMPD_taskyield:
9531     case OMPD_barrier:
9532     case OMPD_taskwait:
9533     case OMPD_taskgroup:
9534     case OMPD_atomic:
9535     case OMPD_flush:
9536     case OMPD_teams:
9537     case OMPD_target_data:
9538     case OMPD_target_exit_data:
9539     case OMPD_target_enter_data:
9540     case OMPD_distribute:
9541     case OMPD_distribute_simd:
9542     case OMPD_distribute_parallel_for:
9543     case OMPD_distribute_parallel_for_simd:
9544     case OMPD_teams_distribute:
9545     case OMPD_teams_distribute_simd:
9546     case OMPD_teams_distribute_parallel_for:
9547     case OMPD_teams_distribute_parallel_for_simd:
9548     case OMPD_target_update:
9549     case OMPD_declare_simd:
9550     case OMPD_declare_variant:
9551     case OMPD_declare_target:
9552     case OMPD_end_declare_target:
9553     case OMPD_declare_reduction:
9554     case OMPD_declare_mapper:
9555     case OMPD_taskloop:
9556     case OMPD_taskloop_simd:
9557     case OMPD_master_taskloop:
9558     case OMPD_master_taskloop_simd:
9559     case OMPD_parallel_master_taskloop:
9560     case OMPD_parallel_master_taskloop_simd:
9561     case OMPD_requires:
9562     case OMPD_unknown:
9563       llvm_unreachable("Unknown target directive for OpenMP device codegen.");
9564     }
9565     return;
9566   }
9567 
9568   if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) {
9569     if (!E->hasAssociatedStmt() || !E->getAssociatedStmt())
9570       return;
9571 
9572     scanForTargetRegionsFunctions(
9573         E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName);
9574     return;
9575   }
9576 
9577   // If this is a lambda function, look into its body.
9578   if (const auto *L = dyn_cast<LambdaExpr>(S))
9579     S = L->getBody();
9580 
9581   // Keep looking for target regions recursively.
9582   for (const Stmt *II : S->children())
9583     scanForTargetRegionsFunctions(II, ParentName);
9584 }
9585 
9586 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) {
9587   // If emitting code for the host, we do not process FD here. Instead we do
9588   // the normal code generation.
9589   if (!CGM.getLangOpts().OpenMPIsDevice) {
9590     if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) {
9591       Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9592           OMPDeclareTargetDeclAttr::getDeviceType(FD);
9593       // Do not emit device_type(nohost) functions for the host.
9594       if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost)
9595         return true;
9596     }
9597     return false;
9598   }
9599 
9600   const ValueDecl *VD = cast<ValueDecl>(GD.getDecl());
9601   // Try to detect target regions in the function.
9602   if (const auto *FD = dyn_cast<FunctionDecl>(VD)) {
9603     StringRef Name = CGM.getMangledName(GD);
9604     scanForTargetRegionsFunctions(FD->getBody(), Name);
9605     Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9606         OMPDeclareTargetDeclAttr::getDeviceType(FD);
9607     // Do not emit device_type(nohost) functions for the host.
9608     if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host)
9609       return true;
9610   }
9611 
9612   // Do not to emit function if it is not marked as declare target.
9613   return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) &&
9614          AlreadyEmittedTargetDecls.count(VD) == 0;
9615 }
9616 
9617 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
9618   if (!CGM.getLangOpts().OpenMPIsDevice)
9619     return false;
9620 
9621   // Check if there are Ctors/Dtors in this declaration and look for target
9622   // regions in it. We use the complete variant to produce the kernel name
9623   // mangling.
9624   QualType RDTy = cast<VarDecl>(GD.getDecl())->getType();
9625   if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) {
9626     for (const CXXConstructorDecl *Ctor : RD->ctors()) {
9627       StringRef ParentName =
9628           CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete));
9629       scanForTargetRegionsFunctions(Ctor->getBody(), ParentName);
9630     }
9631     if (const CXXDestructorDecl *Dtor = RD->getDestructor()) {
9632       StringRef ParentName =
9633           CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete));
9634       scanForTargetRegionsFunctions(Dtor->getBody(), ParentName);
9635     }
9636   }
9637 
9638   // Do not to emit variable if it is not marked as declare target.
9639   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9640       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(
9641           cast<VarDecl>(GD.getDecl()));
9642   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
9643       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9644        HasRequiresUnifiedSharedMemory)) {
9645     DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl()));
9646     return true;
9647   }
9648   return false;
9649 }
9650 
9651 llvm::Constant *
9652 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF,
9653                                                 const VarDecl *VD) {
9654   assert(VD->getType().isConstant(CGM.getContext()) &&
9655          "Expected constant variable.");
9656   StringRef VarName;
9657   llvm::Constant *Addr;
9658   llvm::GlobalValue::LinkageTypes Linkage;
9659   QualType Ty = VD->getType();
9660   SmallString<128> Buffer;
9661   {
9662     unsigned DeviceID;
9663     unsigned FileID;
9664     unsigned Line;
9665     getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID,
9666                              FileID, Line);
9667     llvm::raw_svector_ostream OS(Buffer);
9668     OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID)
9669        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
9670     VarName = OS.str();
9671   }
9672   Linkage = llvm::GlobalValue::InternalLinkage;
9673   Addr =
9674       getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName,
9675                                   getDefaultFirstprivateAddressSpace());
9676   cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage);
9677   CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty);
9678   CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr));
9679   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
9680       VarName, Addr, VarSize,
9681       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage);
9682   return Addr;
9683 }
9684 
9685 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD,
9686                                                    llvm::Constant *Addr) {
9687   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
9688       !CGM.getLangOpts().OpenMPIsDevice)
9689     return;
9690   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9691       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
9692   if (!Res) {
9693     if (CGM.getLangOpts().OpenMPIsDevice) {
9694       // Register non-target variables being emitted in device code (debug info
9695       // may cause this).
9696       StringRef VarName = CGM.getMangledName(VD);
9697       EmittedNonTargetVariables.try_emplace(VarName, Addr);
9698     }
9699     return;
9700   }
9701   // Register declare target variables.
9702   OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags;
9703   StringRef VarName;
9704   CharUnits VarSize;
9705   llvm::GlobalValue::LinkageTypes Linkage;
9706 
9707   if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9708       !HasRequiresUnifiedSharedMemory) {
9709     Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
9710     VarName = CGM.getMangledName(VD);
9711     if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) {
9712       VarSize = CGM.getContext().getTypeSizeInChars(VD->getType());
9713       assert(!VarSize.isZero() && "Expected non-zero size of the variable");
9714     } else {
9715       VarSize = CharUnits::Zero();
9716     }
9717     Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false);
9718     // Temp solution to prevent optimizations of the internal variables.
9719     if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) {
9720       std::string RefName = getName({VarName, "ref"});
9721       if (!CGM.GetGlobalValue(RefName)) {
9722         llvm::Constant *AddrRef =
9723             getOrCreateInternalVariable(Addr->getType(), RefName);
9724         auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef);
9725         GVAddrRef->setConstant(/*Val=*/true);
9726         GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage);
9727         GVAddrRef->setInitializer(Addr);
9728         CGM.addCompilerUsedGlobal(GVAddrRef);
9729       }
9730     }
9731   } else {
9732     assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
9733             (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9734              HasRequiresUnifiedSharedMemory)) &&
9735            "Declare target attribute must link or to with unified memory.");
9736     if (*Res == OMPDeclareTargetDeclAttr::MT_Link)
9737       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink;
9738     else
9739       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
9740 
9741     if (CGM.getLangOpts().OpenMPIsDevice) {
9742       VarName = Addr->getName();
9743       Addr = nullptr;
9744     } else {
9745       VarName = getAddrOfDeclareTargetVar(VD).getName();
9746       Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer());
9747     }
9748     VarSize = CGM.getPointerSize();
9749     Linkage = llvm::GlobalValue::WeakAnyLinkage;
9750   }
9751 
9752   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
9753       VarName, Addr, VarSize, Flags, Linkage);
9754 }
9755 
9756 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) {
9757   if (isa<FunctionDecl>(GD.getDecl()) ||
9758       isa<OMPDeclareReductionDecl>(GD.getDecl()))
9759     return emitTargetFunctions(GD);
9760 
9761   return emitTargetGlobalVariable(GD);
9762 }
9763 
9764 void CGOpenMPRuntime::emitDeferredTargetDecls() const {
9765   for (const VarDecl *VD : DeferredGlobalVariables) {
9766     llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9767         OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
9768     if (!Res)
9769       continue;
9770     if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9771         !HasRequiresUnifiedSharedMemory) {
9772       CGM.EmitGlobal(VD);
9773     } else {
9774       assert((*Res == OMPDeclareTargetDeclAttr::MT_Link ||
9775               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9776                HasRequiresUnifiedSharedMemory)) &&
9777              "Expected link clause or to clause with unified memory.");
9778       (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
9779     }
9780   }
9781 }
9782 
9783 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas(
9784     CodeGenFunction &CGF, const OMPExecutableDirective &D) const {
9785   assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) &&
9786          " Expected target-based directive.");
9787 }
9788 
9789 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) {
9790   for (const OMPClause *Clause : D->clauselists()) {
9791     if (Clause->getClauseKind() == OMPC_unified_shared_memory) {
9792       HasRequiresUnifiedSharedMemory = true;
9793     } else if (const auto *AC =
9794                    dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) {
9795       switch (AC->getAtomicDefaultMemOrderKind()) {
9796       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel:
9797         RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease;
9798         break;
9799       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst:
9800         RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent;
9801         break;
9802       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed:
9803         RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic;
9804         break;
9805       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown:
9806         break;
9807       }
9808     }
9809   }
9810 }
9811 
9812 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const {
9813   return RequiresAtomicOrdering;
9814 }
9815 
9816 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD,
9817                                                        LangAS &AS) {
9818   if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())
9819     return false;
9820   const auto *A = VD->getAttr<OMPAllocateDeclAttr>();
9821   switch(A->getAllocatorType()) {
9822   case OMPAllocateDeclAttr::OMPDefaultMemAlloc:
9823   // Not supported, fallback to the default mem space.
9824   case OMPAllocateDeclAttr::OMPLargeCapMemAlloc:
9825   case OMPAllocateDeclAttr::OMPCGroupMemAlloc:
9826   case OMPAllocateDeclAttr::OMPHighBWMemAlloc:
9827   case OMPAllocateDeclAttr::OMPLowLatMemAlloc:
9828   case OMPAllocateDeclAttr::OMPThreadMemAlloc:
9829   case OMPAllocateDeclAttr::OMPConstMemAlloc:
9830   case OMPAllocateDeclAttr::OMPPTeamMemAlloc:
9831     AS = LangAS::Default;
9832     return true;
9833   case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc:
9834     llvm_unreachable("Expected predefined allocator for the variables with the "
9835                      "static storage.");
9836   }
9837   return false;
9838 }
9839 
9840 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const {
9841   return HasRequiresUnifiedSharedMemory;
9842 }
9843 
9844 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII(
9845     CodeGenModule &CGM)
9846     : CGM(CGM) {
9847   if (CGM.getLangOpts().OpenMPIsDevice) {
9848     SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal;
9849     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false;
9850   }
9851 }
9852 
9853 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() {
9854   if (CGM.getLangOpts().OpenMPIsDevice)
9855     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal;
9856 }
9857 
9858 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) {
9859   if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal)
9860     return true;
9861 
9862   const auto *D = cast<FunctionDecl>(GD.getDecl());
9863   // Do not to emit function if it is marked as declare target as it was already
9864   // emitted.
9865   if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) {
9866     if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) {
9867       if (auto *F = dyn_cast_or_null<llvm::Function>(
9868               CGM.GetGlobalValue(CGM.getMangledName(GD))))
9869         return !F->isDeclaration();
9870       return false;
9871     }
9872     return true;
9873   }
9874 
9875   return !AlreadyEmittedTargetDecls.insert(D).second;
9876 }
9877 
9878 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() {
9879   // If we don't have entries or if we are emitting code for the device, we
9880   // don't need to do anything.
9881   if (CGM.getLangOpts().OMPTargetTriples.empty() ||
9882       CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice ||
9883       (OffloadEntriesInfoManager.empty() &&
9884        !HasEmittedDeclareTargetRegion &&
9885        !HasEmittedTargetRegion))
9886     return nullptr;
9887 
9888   // Create and register the function that handles the requires directives.
9889   ASTContext &C = CGM.getContext();
9890 
9891   llvm::Function *RequiresRegFn;
9892   {
9893     CodeGenFunction CGF(CGM);
9894     const auto &FI = CGM.getTypes().arrangeNullaryFunction();
9895     llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
9896     std::string ReqName = getName({"omp_offloading", "requires_reg"});
9897     RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI);
9898     CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {});
9899     OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE;
9900     // TODO: check for other requires clauses.
9901     // The requires directive takes effect only when a target region is
9902     // present in the compilation unit. Otherwise it is ignored and not
9903     // passed to the runtime. This avoids the runtime from throwing an error
9904     // for mismatching requires clauses across compilation units that don't
9905     // contain at least 1 target region.
9906     assert((HasEmittedTargetRegion ||
9907             HasEmittedDeclareTargetRegion ||
9908             !OffloadEntriesInfoManager.empty()) &&
9909            "Target or declare target region expected.");
9910     if (HasRequiresUnifiedSharedMemory)
9911       Flags = OMP_REQ_UNIFIED_SHARED_MEMORY;
9912     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires),
9913         llvm::ConstantInt::get(CGM.Int64Ty, Flags));
9914     CGF.FinishFunction();
9915   }
9916   return RequiresRegFn;
9917 }
9918 
9919 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF,
9920                                     const OMPExecutableDirective &D,
9921                                     SourceLocation Loc,
9922                                     llvm::Function *OutlinedFn,
9923                                     ArrayRef<llvm::Value *> CapturedVars) {
9924   if (!CGF.HaveInsertPoint())
9925     return;
9926 
9927   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
9928   CodeGenFunction::RunCleanupsScope Scope(CGF);
9929 
9930   // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn);
9931   llvm::Value *Args[] = {
9932       RTLoc,
9933       CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
9934       CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())};
9935   llvm::SmallVector<llvm::Value *, 16> RealArgs;
9936   RealArgs.append(std::begin(Args), std::end(Args));
9937   RealArgs.append(CapturedVars.begin(), CapturedVars.end());
9938 
9939   llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams);
9940   CGF.EmitRuntimeCall(RTLFn, RealArgs);
9941 }
9942 
9943 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
9944                                          const Expr *NumTeams,
9945                                          const Expr *ThreadLimit,
9946                                          SourceLocation Loc) {
9947   if (!CGF.HaveInsertPoint())
9948     return;
9949 
9950   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
9951 
9952   llvm::Value *NumTeamsVal =
9953       NumTeams
9954           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams),
9955                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
9956           : CGF.Builder.getInt32(0);
9957 
9958   llvm::Value *ThreadLimitVal =
9959       ThreadLimit
9960           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit),
9961                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
9962           : CGF.Builder.getInt32(0);
9963 
9964   // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit)
9965   llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal,
9966                                      ThreadLimitVal};
9967   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams),
9968                       PushNumTeamsArgs);
9969 }
9970 
9971 void CGOpenMPRuntime::emitTargetDataCalls(
9972     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
9973     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
9974   if (!CGF.HaveInsertPoint())
9975     return;
9976 
9977   // Action used to replace the default codegen action and turn privatization
9978   // off.
9979   PrePostActionTy NoPrivAction;
9980 
9981   // Generate the code for the opening of the data environment. Capture all the
9982   // arguments of the runtime call by reference because they are used in the
9983   // closing of the region.
9984   auto &&BeginThenGen = [this, &D, Device, &Info,
9985                          &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) {
9986     // Fill up the arrays with all the mapped variables.
9987     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9988     MappableExprsHandler::MapValuesArrayTy Pointers;
9989     MappableExprsHandler::MapValuesArrayTy Sizes;
9990     MappableExprsHandler::MapFlagsArrayTy MapTypes;
9991 
9992     // Get map clause information.
9993     MappableExprsHandler MCHandler(D, CGF);
9994     MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
9995 
9996     // Fill up the arrays and create the arguments.
9997     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
9998 
9999     llvm::Value *BasePointersArrayArg = nullptr;
10000     llvm::Value *PointersArrayArg = nullptr;
10001     llvm::Value *SizesArrayArg = nullptr;
10002     llvm::Value *MapTypesArrayArg = nullptr;
10003     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10004                                  SizesArrayArg, MapTypesArrayArg, Info);
10005 
10006     // Emit device ID if any.
10007     llvm::Value *DeviceID = nullptr;
10008     if (Device) {
10009       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10010                                            CGF.Int64Ty, /*isSigned=*/true);
10011     } else {
10012       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10013     }
10014 
10015     // Emit the number of elements in the offloading arrays.
10016     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10017 
10018     llvm::Value *OffloadingArgs[] = {
10019         DeviceID,         PointerNum,    BasePointersArrayArg,
10020         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10021     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin),
10022                         OffloadingArgs);
10023 
10024     // If device pointer privatization is required, emit the body of the region
10025     // here. It will have to be duplicated: with and without privatization.
10026     if (!Info.CaptureDeviceAddrMap.empty())
10027       CodeGen(CGF);
10028   };
10029 
10030   // Generate code for the closing of the data region.
10031   auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF,
10032                                             PrePostActionTy &) {
10033     assert(Info.isValid() && "Invalid data environment closing arguments.");
10034 
10035     llvm::Value *BasePointersArrayArg = nullptr;
10036     llvm::Value *PointersArrayArg = nullptr;
10037     llvm::Value *SizesArrayArg = nullptr;
10038     llvm::Value *MapTypesArrayArg = nullptr;
10039     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10040                                  SizesArrayArg, MapTypesArrayArg, Info);
10041 
10042     // Emit device ID if any.
10043     llvm::Value *DeviceID = nullptr;
10044     if (Device) {
10045       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10046                                            CGF.Int64Ty, /*isSigned=*/true);
10047     } else {
10048       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10049     }
10050 
10051     // Emit the number of elements in the offloading arrays.
10052     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10053 
10054     llvm::Value *OffloadingArgs[] = {
10055         DeviceID,         PointerNum,    BasePointersArrayArg,
10056         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10057     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end),
10058                         OffloadingArgs);
10059   };
10060 
10061   // If we need device pointer privatization, we need to emit the body of the
10062   // region with no privatization in the 'else' branch of the conditional.
10063   // Otherwise, we don't have to do anything.
10064   auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF,
10065                                                          PrePostActionTy &) {
10066     if (!Info.CaptureDeviceAddrMap.empty()) {
10067       CodeGen.setAction(NoPrivAction);
10068       CodeGen(CGF);
10069     }
10070   };
10071 
10072   // We don't have to do anything to close the region if the if clause evaluates
10073   // to false.
10074   auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {};
10075 
10076   if (IfCond) {
10077     emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen);
10078   } else {
10079     RegionCodeGenTy RCG(BeginThenGen);
10080     RCG(CGF);
10081   }
10082 
10083   // If we don't require privatization of device pointers, we emit the body in
10084   // between the runtime calls. This avoids duplicating the body code.
10085   if (Info.CaptureDeviceAddrMap.empty()) {
10086     CodeGen.setAction(NoPrivAction);
10087     CodeGen(CGF);
10088   }
10089 
10090   if (IfCond) {
10091     emitIfClause(CGF, IfCond, EndThenGen, EndElseGen);
10092   } else {
10093     RegionCodeGenTy RCG(EndThenGen);
10094     RCG(CGF);
10095   }
10096 }
10097 
10098 void CGOpenMPRuntime::emitTargetDataStandAloneCall(
10099     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
10100     const Expr *Device) {
10101   if (!CGF.HaveInsertPoint())
10102     return;
10103 
10104   assert((isa<OMPTargetEnterDataDirective>(D) ||
10105           isa<OMPTargetExitDataDirective>(D) ||
10106           isa<OMPTargetUpdateDirective>(D)) &&
10107          "Expecting either target enter, exit data, or update directives.");
10108 
10109   CodeGenFunction::OMPTargetDataInfo InputInfo;
10110   llvm::Value *MapTypesArray = nullptr;
10111   // Generate the code for the opening of the data environment.
10112   auto &&ThenGen = [this, &D, Device, &InputInfo,
10113                     &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) {
10114     // Emit device ID if any.
10115     llvm::Value *DeviceID = nullptr;
10116     if (Device) {
10117       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10118                                            CGF.Int64Ty, /*isSigned=*/true);
10119     } else {
10120       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10121     }
10122 
10123     // Emit the number of elements in the offloading arrays.
10124     llvm::Constant *PointerNum =
10125         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
10126 
10127     llvm::Value *OffloadingArgs[] = {DeviceID,
10128                                      PointerNum,
10129                                      InputInfo.BasePointersArray.getPointer(),
10130                                      InputInfo.PointersArray.getPointer(),
10131                                      InputInfo.SizesArray.getPointer(),
10132                                      MapTypesArray};
10133 
10134     // Select the right runtime function call for each expected standalone
10135     // directive.
10136     const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
10137     OpenMPRTLFunction RTLFn;
10138     switch (D.getDirectiveKind()) {
10139     case OMPD_target_enter_data:
10140       RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait
10141                         : OMPRTL__tgt_target_data_begin;
10142       break;
10143     case OMPD_target_exit_data:
10144       RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait
10145                         : OMPRTL__tgt_target_data_end;
10146       break;
10147     case OMPD_target_update:
10148       RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait
10149                         : OMPRTL__tgt_target_data_update;
10150       break;
10151     case OMPD_parallel:
10152     case OMPD_for:
10153     case OMPD_parallel_for:
10154     case OMPD_parallel_master:
10155     case OMPD_parallel_sections:
10156     case OMPD_for_simd:
10157     case OMPD_parallel_for_simd:
10158     case OMPD_cancel:
10159     case OMPD_cancellation_point:
10160     case OMPD_ordered:
10161     case OMPD_threadprivate:
10162     case OMPD_allocate:
10163     case OMPD_task:
10164     case OMPD_simd:
10165     case OMPD_sections:
10166     case OMPD_section:
10167     case OMPD_single:
10168     case OMPD_master:
10169     case OMPD_critical:
10170     case OMPD_taskyield:
10171     case OMPD_barrier:
10172     case OMPD_taskwait:
10173     case OMPD_taskgroup:
10174     case OMPD_atomic:
10175     case OMPD_flush:
10176     case OMPD_teams:
10177     case OMPD_target_data:
10178     case OMPD_distribute:
10179     case OMPD_distribute_simd:
10180     case OMPD_distribute_parallel_for:
10181     case OMPD_distribute_parallel_for_simd:
10182     case OMPD_teams_distribute:
10183     case OMPD_teams_distribute_simd:
10184     case OMPD_teams_distribute_parallel_for:
10185     case OMPD_teams_distribute_parallel_for_simd:
10186     case OMPD_declare_simd:
10187     case OMPD_declare_variant:
10188     case OMPD_declare_target:
10189     case OMPD_end_declare_target:
10190     case OMPD_declare_reduction:
10191     case OMPD_declare_mapper:
10192     case OMPD_taskloop:
10193     case OMPD_taskloop_simd:
10194     case OMPD_master_taskloop:
10195     case OMPD_master_taskloop_simd:
10196     case OMPD_parallel_master_taskloop:
10197     case OMPD_parallel_master_taskloop_simd:
10198     case OMPD_target:
10199     case OMPD_target_simd:
10200     case OMPD_target_teams_distribute:
10201     case OMPD_target_teams_distribute_simd:
10202     case OMPD_target_teams_distribute_parallel_for:
10203     case OMPD_target_teams_distribute_parallel_for_simd:
10204     case OMPD_target_teams:
10205     case OMPD_target_parallel:
10206     case OMPD_target_parallel_for:
10207     case OMPD_target_parallel_for_simd:
10208     case OMPD_requires:
10209     case OMPD_unknown:
10210       llvm_unreachable("Unexpected standalone target data directive.");
10211       break;
10212     }
10213     CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs);
10214   };
10215 
10216   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray](
10217                              CodeGenFunction &CGF, PrePostActionTy &) {
10218     // Fill up the arrays with all the mapped variables.
10219     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
10220     MappableExprsHandler::MapValuesArrayTy Pointers;
10221     MappableExprsHandler::MapValuesArrayTy Sizes;
10222     MappableExprsHandler::MapFlagsArrayTy MapTypes;
10223 
10224     // Get map clause information.
10225     MappableExprsHandler MEHandler(D, CGF);
10226     MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
10227 
10228     TargetDataInfo Info;
10229     // Fill up the arrays and create the arguments.
10230     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
10231     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
10232                                  Info.PointersArray, Info.SizesArray,
10233                                  Info.MapTypesArray, Info);
10234     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
10235     InputInfo.BasePointersArray =
10236         Address(Info.BasePointersArray, CGM.getPointerAlign());
10237     InputInfo.PointersArray =
10238         Address(Info.PointersArray, CGM.getPointerAlign());
10239     InputInfo.SizesArray =
10240         Address(Info.SizesArray, CGM.getPointerAlign());
10241     MapTypesArray = Info.MapTypesArray;
10242     if (D.hasClausesOfKind<OMPDependClause>())
10243       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
10244     else
10245       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
10246   };
10247 
10248   if (IfCond) {
10249     emitIfClause(CGF, IfCond, TargetThenGen,
10250                  [](CodeGenFunction &CGF, PrePostActionTy &) {});
10251   } else {
10252     RegionCodeGenTy ThenRCG(TargetThenGen);
10253     ThenRCG(CGF);
10254   }
10255 }
10256 
10257 namespace {
10258   /// Kind of parameter in a function with 'declare simd' directive.
10259   enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector };
10260   /// Attribute set of the parameter.
10261   struct ParamAttrTy {
10262     ParamKindTy Kind = Vector;
10263     llvm::APSInt StrideOrArg;
10264     llvm::APSInt Alignment;
10265   };
10266 } // namespace
10267 
10268 static unsigned evaluateCDTSize(const FunctionDecl *FD,
10269                                 ArrayRef<ParamAttrTy> ParamAttrs) {
10270   // Every vector variant of a SIMD-enabled function has a vector length (VLEN).
10271   // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument
10272   // of that clause. The VLEN value must be power of 2.
10273   // In other case the notion of the function`s "characteristic data type" (CDT)
10274   // is used to compute the vector length.
10275   // CDT is defined in the following order:
10276   //   a) For non-void function, the CDT is the return type.
10277   //   b) If the function has any non-uniform, non-linear parameters, then the
10278   //   CDT is the type of the first such parameter.
10279   //   c) If the CDT determined by a) or b) above is struct, union, or class
10280   //   type which is pass-by-value (except for the type that maps to the
10281   //   built-in complex data type), the characteristic data type is int.
10282   //   d) If none of the above three cases is applicable, the CDT is int.
10283   // The VLEN is then determined based on the CDT and the size of vector
10284   // register of that ISA for which current vector version is generated. The
10285   // VLEN is computed using the formula below:
10286   //   VLEN  = sizeof(vector_register) / sizeof(CDT),
10287   // where vector register size specified in section 3.2.1 Registers and the
10288   // Stack Frame of original AMD64 ABI document.
10289   QualType RetType = FD->getReturnType();
10290   if (RetType.isNull())
10291     return 0;
10292   ASTContext &C = FD->getASTContext();
10293   QualType CDT;
10294   if (!RetType.isNull() && !RetType->isVoidType()) {
10295     CDT = RetType;
10296   } else {
10297     unsigned Offset = 0;
10298     if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) {
10299       if (ParamAttrs[Offset].Kind == Vector)
10300         CDT = C.getPointerType(C.getRecordType(MD->getParent()));
10301       ++Offset;
10302     }
10303     if (CDT.isNull()) {
10304       for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10305         if (ParamAttrs[I + Offset].Kind == Vector) {
10306           CDT = FD->getParamDecl(I)->getType();
10307           break;
10308         }
10309       }
10310     }
10311   }
10312   if (CDT.isNull())
10313     CDT = C.IntTy;
10314   CDT = CDT->getCanonicalTypeUnqualified();
10315   if (CDT->isRecordType() || CDT->isUnionType())
10316     CDT = C.IntTy;
10317   return C.getTypeSize(CDT);
10318 }
10319 
10320 static void
10321 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn,
10322                            const llvm::APSInt &VLENVal,
10323                            ArrayRef<ParamAttrTy> ParamAttrs,
10324                            OMPDeclareSimdDeclAttr::BranchStateTy State) {
10325   struct ISADataTy {
10326     char ISA;
10327     unsigned VecRegSize;
10328   };
10329   ISADataTy ISAData[] = {
10330       {
10331           'b', 128
10332       }, // SSE
10333       {
10334           'c', 256
10335       }, // AVX
10336       {
10337           'd', 256
10338       }, // AVX2
10339       {
10340           'e', 512
10341       }, // AVX512
10342   };
10343   llvm::SmallVector<char, 2> Masked;
10344   switch (State) {
10345   case OMPDeclareSimdDeclAttr::BS_Undefined:
10346     Masked.push_back('N');
10347     Masked.push_back('M');
10348     break;
10349   case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10350     Masked.push_back('N');
10351     break;
10352   case OMPDeclareSimdDeclAttr::BS_Inbranch:
10353     Masked.push_back('M');
10354     break;
10355   }
10356   for (char Mask : Masked) {
10357     for (const ISADataTy &Data : ISAData) {
10358       SmallString<256> Buffer;
10359       llvm::raw_svector_ostream Out(Buffer);
10360       Out << "_ZGV" << Data.ISA << Mask;
10361       if (!VLENVal) {
10362         unsigned NumElts = evaluateCDTSize(FD, ParamAttrs);
10363         assert(NumElts && "Non-zero simdlen/cdtsize expected");
10364         Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts);
10365       } else {
10366         Out << VLENVal;
10367       }
10368       for (const ParamAttrTy &ParamAttr : ParamAttrs) {
10369         switch (ParamAttr.Kind){
10370         case LinearWithVarStride:
10371           Out << 's' << ParamAttr.StrideOrArg;
10372           break;
10373         case Linear:
10374           Out << 'l';
10375           if (!!ParamAttr.StrideOrArg)
10376             Out << ParamAttr.StrideOrArg;
10377           break;
10378         case Uniform:
10379           Out << 'u';
10380           break;
10381         case Vector:
10382           Out << 'v';
10383           break;
10384         }
10385         if (!!ParamAttr.Alignment)
10386           Out << 'a' << ParamAttr.Alignment;
10387       }
10388       Out << '_' << Fn->getName();
10389       Fn->addFnAttr(Out.str());
10390     }
10391   }
10392 }
10393 
10394 // This are the Functions that are needed to mangle the name of the
10395 // vector functions generated by the compiler, according to the rules
10396 // defined in the "Vector Function ABI specifications for AArch64",
10397 // available at
10398 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi.
10399 
10400 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI.
10401 ///
10402 /// TODO: Need to implement the behavior for reference marked with a
10403 /// var or no linear modifiers (1.b in the section). For this, we
10404 /// need to extend ParamKindTy to support the linear modifiers.
10405 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) {
10406   QT = QT.getCanonicalType();
10407 
10408   if (QT->isVoidType())
10409     return false;
10410 
10411   if (Kind == ParamKindTy::Uniform)
10412     return false;
10413 
10414   if (Kind == ParamKindTy::Linear)
10415     return false;
10416 
10417   // TODO: Handle linear references with modifiers
10418 
10419   if (Kind == ParamKindTy::LinearWithVarStride)
10420     return false;
10421 
10422   return true;
10423 }
10424 
10425 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
10426 static bool getAArch64PBV(QualType QT, ASTContext &C) {
10427   QT = QT.getCanonicalType();
10428   unsigned Size = C.getTypeSize(QT);
10429 
10430   // Only scalars and complex within 16 bytes wide set PVB to true.
10431   if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128)
10432     return false;
10433 
10434   if (QT->isFloatingType())
10435     return true;
10436 
10437   if (QT->isIntegerType())
10438     return true;
10439 
10440   if (QT->isPointerType())
10441     return true;
10442 
10443   // TODO: Add support for complex types (section 3.1.2, item 2).
10444 
10445   return false;
10446 }
10447 
10448 /// Computes the lane size (LS) of a return type or of an input parameter,
10449 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI.
10450 /// TODO: Add support for references, section 3.2.1, item 1.
10451 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) {
10452   if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) {
10453     QualType PTy = QT.getCanonicalType()->getPointeeType();
10454     if (getAArch64PBV(PTy, C))
10455       return C.getTypeSize(PTy);
10456   }
10457   if (getAArch64PBV(QT, C))
10458     return C.getTypeSize(QT);
10459 
10460   return C.getTypeSize(C.getUIntPtrType());
10461 }
10462 
10463 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the
10464 // signature of the scalar function, as defined in 3.2.2 of the
10465 // AAVFABI.
10466 static std::tuple<unsigned, unsigned, bool>
10467 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) {
10468   QualType RetType = FD->getReturnType().getCanonicalType();
10469 
10470   ASTContext &C = FD->getASTContext();
10471 
10472   bool OutputBecomesInput = false;
10473 
10474   llvm::SmallVector<unsigned, 8> Sizes;
10475   if (!RetType->isVoidType()) {
10476     Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C));
10477     if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {}))
10478       OutputBecomesInput = true;
10479   }
10480   for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10481     QualType QT = FD->getParamDecl(I)->getType().getCanonicalType();
10482     Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C));
10483   }
10484 
10485   assert(!Sizes.empty() && "Unable to determine NDS and WDS.");
10486   // The LS of a function parameter / return value can only be a power
10487   // of 2, starting from 8 bits, up to 128.
10488   assert(std::all_of(Sizes.begin(), Sizes.end(),
10489                      [](unsigned Size) {
10490                        return Size == 8 || Size == 16 || Size == 32 ||
10491                               Size == 64 || Size == 128;
10492                      }) &&
10493          "Invalid size");
10494 
10495   return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)),
10496                          *std::max_element(std::begin(Sizes), std::end(Sizes)),
10497                          OutputBecomesInput);
10498 }
10499 
10500 /// Mangle the parameter part of the vector function name according to
10501 /// their OpenMP classification. The mangling function is defined in
10502 /// section 3.5 of the AAVFABI.
10503 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) {
10504   SmallString<256> Buffer;
10505   llvm::raw_svector_ostream Out(Buffer);
10506   for (const auto &ParamAttr : ParamAttrs) {
10507     switch (ParamAttr.Kind) {
10508     case LinearWithVarStride:
10509       Out << "ls" << ParamAttr.StrideOrArg;
10510       break;
10511     case Linear:
10512       Out << 'l';
10513       // Don't print the step value if it is not present or if it is
10514       // equal to 1.
10515       if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1)
10516         Out << ParamAttr.StrideOrArg;
10517       break;
10518     case Uniform:
10519       Out << 'u';
10520       break;
10521     case Vector:
10522       Out << 'v';
10523       break;
10524     }
10525 
10526     if (!!ParamAttr.Alignment)
10527       Out << 'a' << ParamAttr.Alignment;
10528   }
10529 
10530   return std::string(Out.str());
10531 }
10532 
10533 // Function used to add the attribute. The parameter `VLEN` is
10534 // templated to allow the use of "x" when targeting scalable functions
10535 // for SVE.
10536 template <typename T>
10537 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix,
10538                                  char ISA, StringRef ParSeq,
10539                                  StringRef MangledName, bool OutputBecomesInput,
10540                                  llvm::Function *Fn) {
10541   SmallString<256> Buffer;
10542   llvm::raw_svector_ostream Out(Buffer);
10543   Out << Prefix << ISA << LMask << VLEN;
10544   if (OutputBecomesInput)
10545     Out << "v";
10546   Out << ParSeq << "_" << MangledName;
10547   Fn->addFnAttr(Out.str());
10548 }
10549 
10550 // Helper function to generate the Advanced SIMD names depending on
10551 // the value of the NDS when simdlen is not present.
10552 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask,
10553                                       StringRef Prefix, char ISA,
10554                                       StringRef ParSeq, StringRef MangledName,
10555                                       bool OutputBecomesInput,
10556                                       llvm::Function *Fn) {
10557   switch (NDS) {
10558   case 8:
10559     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10560                          OutputBecomesInput, Fn);
10561     addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName,
10562                          OutputBecomesInput, Fn);
10563     break;
10564   case 16:
10565     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10566                          OutputBecomesInput, Fn);
10567     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10568                          OutputBecomesInput, Fn);
10569     break;
10570   case 32:
10571     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10572                          OutputBecomesInput, Fn);
10573     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10574                          OutputBecomesInput, Fn);
10575     break;
10576   case 64:
10577   case 128:
10578     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10579                          OutputBecomesInput, Fn);
10580     break;
10581   default:
10582     llvm_unreachable("Scalar type is too wide.");
10583   }
10584 }
10585 
10586 /// Emit vector function attributes for AArch64, as defined in the AAVFABI.
10587 static void emitAArch64DeclareSimdFunction(
10588     CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN,
10589     ArrayRef<ParamAttrTy> ParamAttrs,
10590     OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName,
10591     char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) {
10592 
10593   // Get basic data for building the vector signature.
10594   const auto Data = getNDSWDS(FD, ParamAttrs);
10595   const unsigned NDS = std::get<0>(Data);
10596   const unsigned WDS = std::get<1>(Data);
10597   const bool OutputBecomesInput = std::get<2>(Data);
10598 
10599   // Check the values provided via `simdlen` by the user.
10600   // 1. A `simdlen(1)` doesn't produce vector signatures,
10601   if (UserVLEN == 1) {
10602     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10603         DiagnosticsEngine::Warning,
10604         "The clause simdlen(1) has no effect when targeting aarch64.");
10605     CGM.getDiags().Report(SLoc, DiagID);
10606     return;
10607   }
10608 
10609   // 2. Section 3.3.1, item 1: user input must be a power of 2 for
10610   // Advanced SIMD output.
10611   if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) {
10612     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10613         DiagnosticsEngine::Warning, "The value specified in simdlen must be a "
10614                                     "power of 2 when targeting Advanced SIMD.");
10615     CGM.getDiags().Report(SLoc, DiagID);
10616     return;
10617   }
10618 
10619   // 3. Section 3.4.1. SVE fixed lengh must obey the architectural
10620   // limits.
10621   if (ISA == 's' && UserVLEN != 0) {
10622     if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) {
10623       unsigned DiagID = CGM.getDiags().getCustomDiagID(
10624           DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit "
10625                                       "lanes in the architectural constraints "
10626                                       "for SVE (min is 128-bit, max is "
10627                                       "2048-bit, by steps of 128-bit)");
10628       CGM.getDiags().Report(SLoc, DiagID) << WDS;
10629       return;
10630     }
10631   }
10632 
10633   // Sort out parameter sequence.
10634   const std::string ParSeq = mangleVectorParameters(ParamAttrs);
10635   StringRef Prefix = "_ZGV";
10636   // Generate simdlen from user input (if any).
10637   if (UserVLEN) {
10638     if (ISA == 's') {
10639       // SVE generates only a masked function.
10640       addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10641                            OutputBecomesInput, Fn);
10642     } else {
10643       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10644       // Advanced SIMD generates one or two functions, depending on
10645       // the `[not]inbranch` clause.
10646       switch (State) {
10647       case OMPDeclareSimdDeclAttr::BS_Undefined:
10648         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10649                              OutputBecomesInput, Fn);
10650         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10651                              OutputBecomesInput, Fn);
10652         break;
10653       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10654         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10655                              OutputBecomesInput, Fn);
10656         break;
10657       case OMPDeclareSimdDeclAttr::BS_Inbranch:
10658         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10659                              OutputBecomesInput, Fn);
10660         break;
10661       }
10662     }
10663   } else {
10664     // If no user simdlen is provided, follow the AAVFABI rules for
10665     // generating the vector length.
10666     if (ISA == 's') {
10667       // SVE, section 3.4.1, item 1.
10668       addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName,
10669                            OutputBecomesInput, Fn);
10670     } else {
10671       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10672       // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or
10673       // two vector names depending on the use of the clause
10674       // `[not]inbranch`.
10675       switch (State) {
10676       case OMPDeclareSimdDeclAttr::BS_Undefined:
10677         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10678                                   OutputBecomesInput, Fn);
10679         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
10680                                   OutputBecomesInput, Fn);
10681         break;
10682       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10683         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10684                                   OutputBecomesInput, Fn);
10685         break;
10686       case OMPDeclareSimdDeclAttr::BS_Inbranch:
10687         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
10688                                   OutputBecomesInput, Fn);
10689         break;
10690       }
10691     }
10692   }
10693 }
10694 
10695 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD,
10696                                               llvm::Function *Fn) {
10697   ASTContext &C = CGM.getContext();
10698   FD = FD->getMostRecentDecl();
10699   // Map params to their positions in function decl.
10700   llvm::DenseMap<const Decl *, unsigned> ParamPositions;
10701   if (isa<CXXMethodDecl>(FD))
10702     ParamPositions.try_emplace(FD, 0);
10703   unsigned ParamPos = ParamPositions.size();
10704   for (const ParmVarDecl *P : FD->parameters()) {
10705     ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos);
10706     ++ParamPos;
10707   }
10708   while (FD) {
10709     for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) {
10710       llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size());
10711       // Mark uniform parameters.
10712       for (const Expr *E : Attr->uniforms()) {
10713         E = E->IgnoreParenImpCasts();
10714         unsigned Pos;
10715         if (isa<CXXThisExpr>(E)) {
10716           Pos = ParamPositions[FD];
10717         } else {
10718           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
10719                                 ->getCanonicalDecl();
10720           Pos = ParamPositions[PVD];
10721         }
10722         ParamAttrs[Pos].Kind = Uniform;
10723       }
10724       // Get alignment info.
10725       auto NI = Attr->alignments_begin();
10726       for (const Expr *E : Attr->aligneds()) {
10727         E = E->IgnoreParenImpCasts();
10728         unsigned Pos;
10729         QualType ParmTy;
10730         if (isa<CXXThisExpr>(E)) {
10731           Pos = ParamPositions[FD];
10732           ParmTy = E->getType();
10733         } else {
10734           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
10735                                 ->getCanonicalDecl();
10736           Pos = ParamPositions[PVD];
10737           ParmTy = PVD->getType();
10738         }
10739         ParamAttrs[Pos].Alignment =
10740             (*NI)
10741                 ? (*NI)->EvaluateKnownConstInt(C)
10742                 : llvm::APSInt::getUnsigned(
10743                       C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy))
10744                           .getQuantity());
10745         ++NI;
10746       }
10747       // Mark linear parameters.
10748       auto SI = Attr->steps_begin();
10749       auto MI = Attr->modifiers_begin();
10750       for (const Expr *E : Attr->linears()) {
10751         E = E->IgnoreParenImpCasts();
10752         unsigned Pos;
10753         if (isa<CXXThisExpr>(E)) {
10754           Pos = ParamPositions[FD];
10755         } else {
10756           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
10757                                 ->getCanonicalDecl();
10758           Pos = ParamPositions[PVD];
10759         }
10760         ParamAttrTy &ParamAttr = ParamAttrs[Pos];
10761         ParamAttr.Kind = Linear;
10762         if (*SI) {
10763           Expr::EvalResult Result;
10764           if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) {
10765             if (const auto *DRE =
10766                     cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) {
10767               if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) {
10768                 ParamAttr.Kind = LinearWithVarStride;
10769                 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(
10770                     ParamPositions[StridePVD->getCanonicalDecl()]);
10771               }
10772             }
10773           } else {
10774             ParamAttr.StrideOrArg = Result.Val.getInt();
10775           }
10776         }
10777         ++SI;
10778         ++MI;
10779       }
10780       llvm::APSInt VLENVal;
10781       SourceLocation ExprLoc;
10782       const Expr *VLENExpr = Attr->getSimdlen();
10783       if (VLENExpr) {
10784         VLENVal = VLENExpr->EvaluateKnownConstInt(C);
10785         ExprLoc = VLENExpr->getExprLoc();
10786       }
10787       OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState();
10788       if (CGM.getTriple().isX86()) {
10789         emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State);
10790       } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) {
10791         unsigned VLEN = VLENVal.getExtValue();
10792         StringRef MangledName = Fn->getName();
10793         if (CGM.getTarget().hasFeature("sve"))
10794           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
10795                                          MangledName, 's', 128, Fn, ExprLoc);
10796         if (CGM.getTarget().hasFeature("neon"))
10797           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
10798                                          MangledName, 'n', 128, Fn, ExprLoc);
10799       }
10800     }
10801     FD = FD->getPreviousDecl();
10802   }
10803 }
10804 
10805 namespace {
10806 /// Cleanup action for doacross support.
10807 class DoacrossCleanupTy final : public EHScopeStack::Cleanup {
10808 public:
10809   static const int DoacrossFinArgs = 2;
10810 
10811 private:
10812   llvm::FunctionCallee RTLFn;
10813   llvm::Value *Args[DoacrossFinArgs];
10814 
10815 public:
10816   DoacrossCleanupTy(llvm::FunctionCallee RTLFn,
10817                     ArrayRef<llvm::Value *> CallArgs)
10818       : RTLFn(RTLFn) {
10819     assert(CallArgs.size() == DoacrossFinArgs);
10820     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
10821   }
10822   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
10823     if (!CGF.HaveInsertPoint())
10824       return;
10825     CGF.EmitRuntimeCall(RTLFn, Args);
10826   }
10827 };
10828 } // namespace
10829 
10830 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF,
10831                                        const OMPLoopDirective &D,
10832                                        ArrayRef<Expr *> NumIterations) {
10833   if (!CGF.HaveInsertPoint())
10834     return;
10835 
10836   ASTContext &C = CGM.getContext();
10837   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
10838   RecordDecl *RD;
10839   if (KmpDimTy.isNull()) {
10840     // Build struct kmp_dim {  // loop bounds info casted to kmp_int64
10841     //  kmp_int64 lo; // lower
10842     //  kmp_int64 up; // upper
10843     //  kmp_int64 st; // stride
10844     // };
10845     RD = C.buildImplicitRecord("kmp_dim");
10846     RD->startDefinition();
10847     addFieldToRecordDecl(C, RD, Int64Ty);
10848     addFieldToRecordDecl(C, RD, Int64Ty);
10849     addFieldToRecordDecl(C, RD, Int64Ty);
10850     RD->completeDefinition();
10851     KmpDimTy = C.getRecordType(RD);
10852   } else {
10853     RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl());
10854   }
10855   llvm::APInt Size(/*numBits=*/32, NumIterations.size());
10856   QualType ArrayTy =
10857       C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0);
10858 
10859   Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims");
10860   CGF.EmitNullInitialization(DimsAddr, ArrayTy);
10861   enum { LowerFD = 0, UpperFD, StrideFD };
10862   // Fill dims with data.
10863   for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) {
10864     LValue DimsLVal = CGF.MakeAddrLValue(
10865         CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy);
10866     // dims.upper = num_iterations;
10867     LValue UpperLVal = CGF.EmitLValueForField(
10868         DimsLVal, *std::next(RD->field_begin(), UpperFD));
10869     llvm::Value *NumIterVal =
10870         CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]),
10871                                  D.getNumIterations()->getType(), Int64Ty,
10872                                  D.getNumIterations()->getExprLoc());
10873     CGF.EmitStoreOfScalar(NumIterVal, UpperLVal);
10874     // dims.stride = 1;
10875     LValue StrideLVal = CGF.EmitLValueForField(
10876         DimsLVal, *std::next(RD->field_begin(), StrideFD));
10877     CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1),
10878                           StrideLVal);
10879   }
10880 
10881   // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid,
10882   // kmp_int32 num_dims, struct kmp_dim * dims);
10883   llvm::Value *Args[] = {
10884       emitUpdateLocation(CGF, D.getBeginLoc()),
10885       getThreadID(CGF, D.getBeginLoc()),
10886       llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()),
10887       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
10888           CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(),
10889           CGM.VoidPtrTy)};
10890 
10891   llvm::FunctionCallee RTLFn =
10892       createRuntimeFunction(OMPRTL__kmpc_doacross_init);
10893   CGF.EmitRuntimeCall(RTLFn, Args);
10894   llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = {
10895       emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())};
10896   llvm::FunctionCallee FiniRTLFn =
10897       createRuntimeFunction(OMPRTL__kmpc_doacross_fini);
10898   CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
10899                                              llvm::makeArrayRef(FiniArgs));
10900 }
10901 
10902 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
10903                                           const OMPDependClause *C) {
10904   QualType Int64Ty =
10905       CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
10906   llvm::APInt Size(/*numBits=*/32, C->getNumLoops());
10907   QualType ArrayTy = CGM.getContext().getConstantArrayType(
10908       Int64Ty, Size, nullptr, ArrayType::Normal, 0);
10909   Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr");
10910   for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) {
10911     const Expr *CounterVal = C->getLoopData(I);
10912     assert(CounterVal);
10913     llvm::Value *CntVal = CGF.EmitScalarConversion(
10914         CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty,
10915         CounterVal->getExprLoc());
10916     CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I),
10917                           /*Volatile=*/false, Int64Ty);
10918   }
10919   llvm::Value *Args[] = {
10920       emitUpdateLocation(CGF, C->getBeginLoc()),
10921       getThreadID(CGF, C->getBeginLoc()),
10922       CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()};
10923   llvm::FunctionCallee RTLFn;
10924   if (C->getDependencyKind() == OMPC_DEPEND_source) {
10925     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post);
10926   } else {
10927     assert(C->getDependencyKind() == OMPC_DEPEND_sink);
10928     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait);
10929   }
10930   CGF.EmitRuntimeCall(RTLFn, Args);
10931 }
10932 
10933 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc,
10934                                llvm::FunctionCallee Callee,
10935                                ArrayRef<llvm::Value *> Args) const {
10936   assert(Loc.isValid() && "Outlined function call location must be valid.");
10937   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
10938 
10939   if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) {
10940     if (Fn->doesNotThrow()) {
10941       CGF.EmitNounwindRuntimeCall(Fn, Args);
10942       return;
10943     }
10944   }
10945   CGF.EmitRuntimeCall(Callee, Args);
10946 }
10947 
10948 void CGOpenMPRuntime::emitOutlinedFunctionCall(
10949     CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn,
10950     ArrayRef<llvm::Value *> Args) const {
10951   emitCall(CGF, Loc, OutlinedFn, Args);
10952 }
10953 
10954 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) {
10955   if (const auto *FD = dyn_cast<FunctionDecl>(D))
10956     if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD))
10957       HasEmittedDeclareTargetRegion = true;
10958 }
10959 
10960 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF,
10961                                              const VarDecl *NativeParam,
10962                                              const VarDecl *TargetParam) const {
10963   return CGF.GetAddrOfLocalVar(NativeParam);
10964 }
10965 
10966 namespace {
10967 /// Cleanup action for allocate support.
10968 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup {
10969 public:
10970   static const int CleanupArgs = 3;
10971 
10972 private:
10973   llvm::FunctionCallee RTLFn;
10974   llvm::Value *Args[CleanupArgs];
10975 
10976 public:
10977   OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn,
10978                        ArrayRef<llvm::Value *> CallArgs)
10979       : RTLFn(RTLFn) {
10980     assert(CallArgs.size() == CleanupArgs &&
10981            "Size of arguments does not match.");
10982     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
10983   }
10984   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
10985     if (!CGF.HaveInsertPoint())
10986       return;
10987     CGF.EmitRuntimeCall(RTLFn, Args);
10988   }
10989 };
10990 } // namespace
10991 
10992 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF,
10993                                                    const VarDecl *VD) {
10994   if (!VD)
10995     return Address::invalid();
10996   const VarDecl *CVD = VD->getCanonicalDecl();
10997   if (!CVD->hasAttr<OMPAllocateDeclAttr>())
10998     return Address::invalid();
10999   const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
11000   // Use the default allocation.
11001   if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc &&
11002       !AA->getAllocator())
11003     return Address::invalid();
11004   llvm::Value *Size;
11005   CharUnits Align = CGM.getContext().getDeclAlign(CVD);
11006   if (CVD->getType()->isVariablyModifiedType()) {
11007     Size = CGF.getTypeSize(CVD->getType());
11008     // Align the size: ((size + align - 1) / align) * align
11009     Size = CGF.Builder.CreateNUWAdd(
11010         Size, CGM.getSize(Align - CharUnits::fromQuantity(1)));
11011     Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align));
11012     Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align));
11013   } else {
11014     CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType());
11015     Size = CGM.getSize(Sz.alignTo(Align));
11016   }
11017   llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc());
11018   assert(AA->getAllocator() &&
11019          "Expected allocator expression for non-default allocator.");
11020   llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator());
11021   // According to the standard, the original allocator type is a enum (integer).
11022   // Convert to pointer type, if required.
11023   if (Allocator->getType()->isIntegerTy())
11024     Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy);
11025   else if (Allocator->getType()->isPointerTy())
11026     Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator,
11027                                                                 CGM.VoidPtrTy);
11028   llvm::Value *Args[] = {ThreadID, Size, Allocator};
11029 
11030   llvm::Value *Addr =
11031       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args,
11032                           getName({CVD->getName(), ".void.addr"}));
11033   llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr,
11034                                                               Allocator};
11035   llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free);
11036 
11037   CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
11038                                                 llvm::makeArrayRef(FiniArgs));
11039   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11040       Addr,
11041       CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())),
11042       getName({CVD->getName(), ".addr"}));
11043   return Address(Addr, Align);
11044 }
11045 
11046 namespace {
11047 using OMPContextSelectorData =
11048     OpenMPCtxSelectorData<ArrayRef<StringRef>, llvm::APSInt>;
11049 using CompleteOMPContextSelectorData = SmallVector<OMPContextSelectorData, 4>;
11050 } // anonymous namespace
11051 
11052 /// Checks current context and returns true if it matches the context selector.
11053 template <OpenMPContextSelectorSetKind CtxSet, OpenMPContextSelectorKind Ctx,
11054           typename... Arguments>
11055 static bool checkContext(const OMPContextSelectorData &Data,
11056                          Arguments... Params) {
11057   assert(Data.CtxSet != OMP_CTX_SET_unknown && Data.Ctx != OMP_CTX_unknown &&
11058          "Unknown context selector or context selector set.");
11059   return false;
11060 }
11061 
11062 /// Checks for implementation={vendor(<vendor>)} context selector.
11063 /// \returns true iff <vendor>="llvm", false otherwise.
11064 template <>
11065 bool checkContext<OMP_CTX_SET_implementation, OMP_CTX_vendor>(
11066     const OMPContextSelectorData &Data) {
11067   return llvm::all_of(Data.Names,
11068                       [](StringRef S) { return !S.compare_lower("llvm"); });
11069 }
11070 
11071 /// Checks for device={kind(<kind>)} context selector.
11072 /// \returns true if <kind>="host" and compilation is for host.
11073 /// true if <kind>="nohost" and compilation is for device.
11074 /// true if <kind>="cpu" and compilation is for Arm, X86 or PPC CPU.
11075 /// true if <kind>="gpu" and compilation is for NVPTX or AMDGCN.
11076 /// false otherwise.
11077 template <>
11078 bool checkContext<OMP_CTX_SET_device, OMP_CTX_kind, CodeGenModule &>(
11079     const OMPContextSelectorData &Data, CodeGenModule &CGM) {
11080   for (StringRef Name : Data.Names) {
11081     if (!Name.compare_lower("host")) {
11082       if (CGM.getLangOpts().OpenMPIsDevice)
11083         return false;
11084       continue;
11085     }
11086     if (!Name.compare_lower("nohost")) {
11087       if (!CGM.getLangOpts().OpenMPIsDevice)
11088         return false;
11089       continue;
11090     }
11091     switch (CGM.getTriple().getArch()) {
11092     case llvm::Triple::arm:
11093     case llvm::Triple::armeb:
11094     case llvm::Triple::aarch64:
11095     case llvm::Triple::aarch64_be:
11096     case llvm::Triple::aarch64_32:
11097     case llvm::Triple::ppc:
11098     case llvm::Triple::ppc64:
11099     case llvm::Triple::ppc64le:
11100     case llvm::Triple::x86:
11101     case llvm::Triple::x86_64:
11102       if (Name.compare_lower("cpu"))
11103         return false;
11104       break;
11105     case llvm::Triple::amdgcn:
11106     case llvm::Triple::nvptx:
11107     case llvm::Triple::nvptx64:
11108       if (Name.compare_lower("gpu"))
11109         return false;
11110       break;
11111     case llvm::Triple::UnknownArch:
11112     case llvm::Triple::arc:
11113     case llvm::Triple::avr:
11114     case llvm::Triple::bpfel:
11115     case llvm::Triple::bpfeb:
11116     case llvm::Triple::hexagon:
11117     case llvm::Triple::mips:
11118     case llvm::Triple::mipsel:
11119     case llvm::Triple::mips64:
11120     case llvm::Triple::mips64el:
11121     case llvm::Triple::msp430:
11122     case llvm::Triple::r600:
11123     case llvm::Triple::riscv32:
11124     case llvm::Triple::riscv64:
11125     case llvm::Triple::sparc:
11126     case llvm::Triple::sparcv9:
11127     case llvm::Triple::sparcel:
11128     case llvm::Triple::systemz:
11129     case llvm::Triple::tce:
11130     case llvm::Triple::tcele:
11131     case llvm::Triple::thumb:
11132     case llvm::Triple::thumbeb:
11133     case llvm::Triple::xcore:
11134     case llvm::Triple::le32:
11135     case llvm::Triple::le64:
11136     case llvm::Triple::amdil:
11137     case llvm::Triple::amdil64:
11138     case llvm::Triple::hsail:
11139     case llvm::Triple::hsail64:
11140     case llvm::Triple::spir:
11141     case llvm::Triple::spir64:
11142     case llvm::Triple::kalimba:
11143     case llvm::Triple::shave:
11144     case llvm::Triple::lanai:
11145     case llvm::Triple::wasm32:
11146     case llvm::Triple::wasm64:
11147     case llvm::Triple::renderscript32:
11148     case llvm::Triple::renderscript64:
11149     case llvm::Triple::ve:
11150       return false;
11151     }
11152   }
11153   return true;
11154 }
11155 
11156 static bool matchesContext(CodeGenModule &CGM,
11157                            const CompleteOMPContextSelectorData &ContextData) {
11158   for (const OMPContextSelectorData &Data : ContextData) {
11159     switch (Data.Ctx) {
11160     case OMP_CTX_vendor:
11161       assert(Data.CtxSet == OMP_CTX_SET_implementation &&
11162              "Expected implementation context selector set.");
11163       if (!checkContext<OMP_CTX_SET_implementation, OMP_CTX_vendor>(Data))
11164         return false;
11165       break;
11166     case OMP_CTX_kind:
11167       assert(Data.CtxSet == OMP_CTX_SET_device &&
11168              "Expected device context selector set.");
11169       if (!checkContext<OMP_CTX_SET_device, OMP_CTX_kind, CodeGenModule &>(Data,
11170                                                                            CGM))
11171         return false;
11172       break;
11173     case OMP_CTX_unknown:
11174       llvm_unreachable("Unknown context selector kind.");
11175     }
11176   }
11177   return true;
11178 }
11179 
11180 static CompleteOMPContextSelectorData
11181 translateAttrToContextSelectorData(ASTContext &C,
11182                                    const OMPDeclareVariantAttr *A) {
11183   CompleteOMPContextSelectorData Data;
11184   for (unsigned I = 0, E = A->scores_size(); I < E; ++I) {
11185     Data.emplace_back();
11186     auto CtxSet = static_cast<OpenMPContextSelectorSetKind>(
11187         *std::next(A->ctxSelectorSets_begin(), I));
11188     auto Ctx = static_cast<OpenMPContextSelectorKind>(
11189         *std::next(A->ctxSelectors_begin(), I));
11190     Data.back().CtxSet = CtxSet;
11191     Data.back().Ctx = Ctx;
11192     const Expr *Score = *std::next(A->scores_begin(), I);
11193     Data.back().Score = Score->EvaluateKnownConstInt(C);
11194     switch (Ctx) {
11195     case OMP_CTX_vendor:
11196       assert(CtxSet == OMP_CTX_SET_implementation &&
11197              "Expected implementation context selector set.");
11198       Data.back().Names =
11199           llvm::makeArrayRef(A->implVendors_begin(), A->implVendors_end());
11200       break;
11201     case OMP_CTX_kind:
11202       assert(CtxSet == OMP_CTX_SET_device &&
11203              "Expected device context selector set.");
11204       Data.back().Names =
11205           llvm::makeArrayRef(A->deviceKinds_begin(), A->deviceKinds_end());
11206       break;
11207     case OMP_CTX_unknown:
11208       llvm_unreachable("Unknown context selector kind.");
11209     }
11210   }
11211   return Data;
11212 }
11213 
11214 static bool isStrictSubset(const CompleteOMPContextSelectorData &LHS,
11215                            const CompleteOMPContextSelectorData &RHS) {
11216   llvm::SmallDenseMap<std::pair<int, int>, llvm::StringSet<>, 4> RHSData;
11217   for (const OMPContextSelectorData &D : RHS) {
11218     auto &Pair = RHSData.FindAndConstruct(std::make_pair(D.CtxSet, D.Ctx));
11219     Pair.getSecond().insert(D.Names.begin(), D.Names.end());
11220   }
11221   bool AllSetsAreEqual = true;
11222   for (const OMPContextSelectorData &D : LHS) {
11223     auto It = RHSData.find(std::make_pair(D.CtxSet, D.Ctx));
11224     if (It == RHSData.end())
11225       return false;
11226     if (D.Names.size() > It->getSecond().size())
11227       return false;
11228     if (llvm::set_union(It->getSecond(), D.Names))
11229       return false;
11230     AllSetsAreEqual =
11231         AllSetsAreEqual && (D.Names.size() == It->getSecond().size());
11232   }
11233 
11234   return LHS.size() != RHS.size() || !AllSetsAreEqual;
11235 }
11236 
11237 static bool greaterCtxScore(const CompleteOMPContextSelectorData &LHS,
11238                             const CompleteOMPContextSelectorData &RHS) {
11239   // Score is calculated as sum of all scores + 1.
11240   llvm::APSInt LHSScore(llvm::APInt(64, 1), /*isUnsigned=*/false);
11241   bool RHSIsSubsetOfLHS = isStrictSubset(RHS, LHS);
11242   if (RHSIsSubsetOfLHS) {
11243     LHSScore = llvm::APSInt::get(0);
11244   } else {
11245     for (const OMPContextSelectorData &Data : LHS) {
11246       if (Data.Score.getBitWidth() > LHSScore.getBitWidth()) {
11247         LHSScore = LHSScore.extend(Data.Score.getBitWidth()) + Data.Score;
11248       } else if (Data.Score.getBitWidth() < LHSScore.getBitWidth()) {
11249         LHSScore += Data.Score.extend(LHSScore.getBitWidth());
11250       } else {
11251         LHSScore += Data.Score;
11252       }
11253     }
11254   }
11255   llvm::APSInt RHSScore(llvm::APInt(64, 1), /*isUnsigned=*/false);
11256   if (!RHSIsSubsetOfLHS && isStrictSubset(LHS, RHS)) {
11257     RHSScore = llvm::APSInt::get(0);
11258   } else {
11259     for (const OMPContextSelectorData &Data : RHS) {
11260       if (Data.Score.getBitWidth() > RHSScore.getBitWidth()) {
11261         RHSScore = RHSScore.extend(Data.Score.getBitWidth()) + Data.Score;
11262       } else if (Data.Score.getBitWidth() < RHSScore.getBitWidth()) {
11263         RHSScore += Data.Score.extend(RHSScore.getBitWidth());
11264       } else {
11265         RHSScore += Data.Score;
11266       }
11267     }
11268   }
11269   return llvm::APSInt::compareValues(LHSScore, RHSScore) >= 0;
11270 }
11271 
11272 /// Finds the variant function that matches current context with its context
11273 /// selector.
11274 static const FunctionDecl *getDeclareVariantFunction(CodeGenModule &CGM,
11275                                                      const FunctionDecl *FD) {
11276   if (!FD->hasAttrs() || !FD->hasAttr<OMPDeclareVariantAttr>())
11277     return FD;
11278   // Iterate through all DeclareVariant attributes and check context selectors.
11279   const OMPDeclareVariantAttr *TopMostAttr = nullptr;
11280   CompleteOMPContextSelectorData TopMostData;
11281   for (const auto *A : FD->specific_attrs<OMPDeclareVariantAttr>()) {
11282     CompleteOMPContextSelectorData Data =
11283         translateAttrToContextSelectorData(CGM.getContext(), A);
11284     if (!matchesContext(CGM, Data))
11285       continue;
11286     // If the attribute matches the context, find the attribute with the highest
11287     // score.
11288     if (!TopMostAttr || !greaterCtxScore(TopMostData, Data)) {
11289       TopMostAttr = A;
11290       TopMostData.swap(Data);
11291     }
11292   }
11293   if (!TopMostAttr)
11294     return FD;
11295   return cast<FunctionDecl>(
11296       cast<DeclRefExpr>(TopMostAttr->getVariantFuncRef()->IgnoreParenImpCasts())
11297           ->getDecl());
11298 }
11299 
11300 bool CGOpenMPRuntime::emitDeclareVariant(GlobalDecl GD, bool IsForDefinition) {
11301   const auto *D = cast<FunctionDecl>(GD.getDecl());
11302   // If the original function is defined already, use its definition.
11303   StringRef MangledName = CGM.getMangledName(GD);
11304   llvm::GlobalValue *Orig = CGM.GetGlobalValue(MangledName);
11305   if (Orig && !Orig->isDeclaration())
11306     return false;
11307   const FunctionDecl *NewFD = getDeclareVariantFunction(CGM, D);
11308   // Emit original function if it does not have declare variant attribute or the
11309   // context does not match.
11310   if (NewFD == D)
11311     return false;
11312   GlobalDecl NewGD = GD.getWithDecl(NewFD);
11313   if (tryEmitDeclareVariant(NewGD, GD, Orig, IsForDefinition)) {
11314     DeferredVariantFunction.erase(D);
11315     return true;
11316   }
11317   DeferredVariantFunction.insert(std::make_pair(D, std::make_pair(NewGD, GD)));
11318   return true;
11319 }
11320 
11321 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII(
11322     CodeGenModule &CGM, const OMPLoopDirective &S)
11323     : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) {
11324   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11325   if (!NeedToPush)
11326     return;
11327   NontemporalDeclsSet &DS =
11328       CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back();
11329   for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) {
11330     for (const Stmt *Ref : C->private_refs()) {
11331       const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts();
11332       const ValueDecl *VD;
11333       if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) {
11334         VD = DRE->getDecl();
11335       } else {
11336         const auto *ME = cast<MemberExpr>(SimpleRefExpr);
11337         assert((ME->isImplicitCXXThis() ||
11338                 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) &&
11339                "Expected member of current class.");
11340         VD = ME->getMemberDecl();
11341       }
11342       DS.insert(VD);
11343     }
11344   }
11345 }
11346 
11347 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() {
11348   if (!NeedToPush)
11349     return;
11350   CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back();
11351 }
11352 
11353 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const {
11354   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11355 
11356   return llvm::any_of(
11357       CGM.getOpenMPRuntime().NontemporalDeclsStack,
11358       [VD](const NontemporalDeclsSet &Set) { return Set.count(VD) > 0; });
11359 }
11360 
11361 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis(
11362     const OMPExecutableDirective &S,
11363     llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled)
11364     const {
11365   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs;
11366   // Vars in target/task regions must be excluded completely.
11367   if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) ||
11368       isOpenMPTaskingDirective(S.getDirectiveKind())) {
11369     SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
11370     getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind());
11371     const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front());
11372     for (const CapturedStmt::Capture &Cap : CS->captures()) {
11373       if (Cap.capturesVariable() || Cap.capturesVariableByCopy())
11374         NeedToCheckForLPCs.insert(Cap.getCapturedVar());
11375     }
11376   }
11377   // Exclude vars in private clauses.
11378   for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) {
11379     for (const Expr *Ref : C->varlists()) {
11380       if (!Ref->getType()->isScalarType())
11381         continue;
11382       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11383       if (!DRE)
11384         continue;
11385       NeedToCheckForLPCs.insert(DRE->getDecl());
11386     }
11387   }
11388   for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
11389     for (const Expr *Ref : C->varlists()) {
11390       if (!Ref->getType()->isScalarType())
11391         continue;
11392       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11393       if (!DRE)
11394         continue;
11395       NeedToCheckForLPCs.insert(DRE->getDecl());
11396     }
11397   }
11398   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
11399     for (const Expr *Ref : C->varlists()) {
11400       if (!Ref->getType()->isScalarType())
11401         continue;
11402       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11403       if (!DRE)
11404         continue;
11405       NeedToCheckForLPCs.insert(DRE->getDecl());
11406     }
11407   }
11408   for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
11409     for (const Expr *Ref : C->varlists()) {
11410       if (!Ref->getType()->isScalarType())
11411         continue;
11412       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11413       if (!DRE)
11414         continue;
11415       NeedToCheckForLPCs.insert(DRE->getDecl());
11416     }
11417   }
11418   for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) {
11419     for (const Expr *Ref : C->varlists()) {
11420       if (!Ref->getType()->isScalarType())
11421         continue;
11422       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11423       if (!DRE)
11424         continue;
11425       NeedToCheckForLPCs.insert(DRE->getDecl());
11426     }
11427   }
11428   for (const Decl *VD : NeedToCheckForLPCs) {
11429     for (const LastprivateConditionalData &Data :
11430          llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) {
11431       if (Data.DeclToUniqueName.count(VD) > 0) {
11432         if (!Data.Disabled)
11433           NeedToAddForLPCsAsDisabled.insert(VD);
11434         break;
11435       }
11436     }
11437   }
11438 }
11439 
11440 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
11441     CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal)
11442     : CGM(CGF.CGM),
11443       Action((CGM.getLangOpts().OpenMP >= 50 &&
11444               llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(),
11445                            [](const OMPLastprivateClause *C) {
11446                              return C->getKind() ==
11447                                     OMPC_LASTPRIVATE_conditional;
11448                            }))
11449                  ? ActionToDo::PushAsLastprivateConditional
11450                  : ActionToDo::DoNotPush) {
11451   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11452   if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush)
11453     return;
11454   assert(Action == ActionToDo::PushAsLastprivateConditional &&
11455          "Expected a push action.");
11456   LastprivateConditionalData &Data =
11457       CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
11458   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
11459     if (C->getKind() != OMPC_LASTPRIVATE_conditional)
11460       continue;
11461 
11462     for (const Expr *Ref : C->varlists()) {
11463       Data.DeclToUniqueName.insert(std::make_pair(
11464           cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(),
11465           SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref))));
11466     }
11467   }
11468   Data.IVLVal = IVLVal;
11469   Data.Fn = CGF.CurFn;
11470 }
11471 
11472 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
11473     CodeGenFunction &CGF, const OMPExecutableDirective &S)
11474     : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) {
11475   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11476   if (CGM.getLangOpts().OpenMP < 50)
11477     return;
11478   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled;
11479   tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled);
11480   if (!NeedToAddForLPCsAsDisabled.empty()) {
11481     Action = ActionToDo::DisableLastprivateConditional;
11482     LastprivateConditionalData &Data =
11483         CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
11484     for (const Decl *VD : NeedToAddForLPCsAsDisabled)
11485       Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>()));
11486     Data.Fn = CGF.CurFn;
11487     Data.Disabled = true;
11488   }
11489 }
11490 
11491 CGOpenMPRuntime::LastprivateConditionalRAII
11492 CGOpenMPRuntime::LastprivateConditionalRAII::disable(
11493     CodeGenFunction &CGF, const OMPExecutableDirective &S) {
11494   return LastprivateConditionalRAII(CGF, S);
11495 }
11496 
11497 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() {
11498   if (CGM.getLangOpts().OpenMP < 50)
11499     return;
11500   if (Action == ActionToDo::DisableLastprivateConditional) {
11501     assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
11502            "Expected list of disabled private vars.");
11503     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
11504   }
11505   if (Action == ActionToDo::PushAsLastprivateConditional) {
11506     assert(
11507         !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
11508         "Expected list of lastprivate conditional vars.");
11509     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
11510   }
11511 }
11512 
11513 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF,
11514                                                         const VarDecl *VD) {
11515   ASTContext &C = CGM.getContext();
11516   auto I = LastprivateConditionalToTypes.find(CGF.CurFn);
11517   if (I == LastprivateConditionalToTypes.end())
11518     I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first;
11519   QualType NewType;
11520   const FieldDecl *VDField;
11521   const FieldDecl *FiredField;
11522   LValue BaseLVal;
11523   auto VI = I->getSecond().find(VD);
11524   if (VI == I->getSecond().end()) {
11525     RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional");
11526     RD->startDefinition();
11527     VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType());
11528     FiredField = addFieldToRecordDecl(C, RD, C.CharTy);
11529     RD->completeDefinition();
11530     NewType = C.getRecordType(RD);
11531     Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName());
11532     BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl);
11533     I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal);
11534   } else {
11535     NewType = std::get<0>(VI->getSecond());
11536     VDField = std::get<1>(VI->getSecond());
11537     FiredField = std::get<2>(VI->getSecond());
11538     BaseLVal = std::get<3>(VI->getSecond());
11539   }
11540   LValue FiredLVal =
11541       CGF.EmitLValueForField(BaseLVal, FiredField);
11542   CGF.EmitStoreOfScalar(
11543       llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)),
11544       FiredLVal);
11545   return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF);
11546 }
11547 
11548 namespace {
11549 /// Checks if the lastprivate conditional variable is referenced in LHS.
11550 class LastprivateConditionalRefChecker final
11551     : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> {
11552   ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM;
11553   const Expr *FoundE = nullptr;
11554   const Decl *FoundD = nullptr;
11555   StringRef UniqueDeclName;
11556   LValue IVLVal;
11557   llvm::Function *FoundFn = nullptr;
11558   SourceLocation Loc;
11559 
11560 public:
11561   bool VisitDeclRefExpr(const DeclRefExpr *E) {
11562     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
11563          llvm::reverse(LPM)) {
11564       auto It = D.DeclToUniqueName.find(E->getDecl());
11565       if (It == D.DeclToUniqueName.end())
11566         continue;
11567       if (D.Disabled)
11568         return false;
11569       FoundE = E;
11570       FoundD = E->getDecl()->getCanonicalDecl();
11571       UniqueDeclName = It->second;
11572       IVLVal = D.IVLVal;
11573       FoundFn = D.Fn;
11574       break;
11575     }
11576     return FoundE == E;
11577   }
11578   bool VisitMemberExpr(const MemberExpr *E) {
11579     if (!CodeGenFunction::IsWrappedCXXThis(E->getBase()))
11580       return false;
11581     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
11582          llvm::reverse(LPM)) {
11583       auto It = D.DeclToUniqueName.find(E->getMemberDecl());
11584       if (It == D.DeclToUniqueName.end())
11585         continue;
11586       if (D.Disabled)
11587         return false;
11588       FoundE = E;
11589       FoundD = E->getMemberDecl()->getCanonicalDecl();
11590       UniqueDeclName = It->second;
11591       IVLVal = D.IVLVal;
11592       FoundFn = D.Fn;
11593       break;
11594     }
11595     return FoundE == E;
11596   }
11597   bool VisitStmt(const Stmt *S) {
11598     for (const Stmt *Child : S->children()) {
11599       if (!Child)
11600         continue;
11601       if (const auto *E = dyn_cast<Expr>(Child))
11602         if (!E->isGLValue())
11603           continue;
11604       if (Visit(Child))
11605         return true;
11606     }
11607     return false;
11608   }
11609   explicit LastprivateConditionalRefChecker(
11610       ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM)
11611       : LPM(LPM) {}
11612   std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *>
11613   getFoundData() const {
11614     return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn);
11615   }
11616 };
11617 } // namespace
11618 
11619 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF,
11620                                                        LValue IVLVal,
11621                                                        StringRef UniqueDeclName,
11622                                                        LValue LVal,
11623                                                        SourceLocation Loc) {
11624   // Last updated loop counter for the lastprivate conditional var.
11625   // int<xx> last_iv = 0;
11626   llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType());
11627   llvm::Constant *LastIV =
11628       getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"}));
11629   cast<llvm::GlobalVariable>(LastIV)->setAlignment(
11630       IVLVal.getAlignment().getAsAlign());
11631   LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType());
11632 
11633   // Last value of the lastprivate conditional.
11634   // decltype(priv_a) last_a;
11635   llvm::Constant *Last = getOrCreateInternalVariable(
11636       CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName);
11637   cast<llvm::GlobalVariable>(Last)->setAlignment(
11638       LVal.getAlignment().getAsAlign());
11639   LValue LastLVal =
11640       CGF.MakeAddrLValue(Last, LVal.getType(), LVal.getAlignment());
11641 
11642   // Global loop counter. Required to handle inner parallel-for regions.
11643   // iv
11644   llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc);
11645 
11646   // #pragma omp critical(a)
11647   // if (last_iv <= iv) {
11648   //   last_iv = iv;
11649   //   last_a = priv_a;
11650   // }
11651   auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal,
11652                     Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
11653     Action.Enter(CGF);
11654     llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc);
11655     // (last_iv <= iv) ? Check if the variable is updated and store new
11656     // value in global var.
11657     llvm::Value *CmpRes;
11658     if (IVLVal.getType()->isSignedIntegerType()) {
11659       CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal);
11660     } else {
11661       assert(IVLVal.getType()->isUnsignedIntegerType() &&
11662              "Loop iteration variable must be integer.");
11663       CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal);
11664     }
11665     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then");
11666     llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit");
11667     CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB);
11668     // {
11669     CGF.EmitBlock(ThenBB);
11670 
11671     //   last_iv = iv;
11672     CGF.EmitStoreOfScalar(IVVal, LastIVLVal);
11673 
11674     //   last_a = priv_a;
11675     switch (CGF.getEvaluationKind(LVal.getType())) {
11676     case TEK_Scalar: {
11677       llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc);
11678       CGF.EmitStoreOfScalar(PrivVal, LastLVal);
11679       break;
11680     }
11681     case TEK_Complex: {
11682       CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc);
11683       CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false);
11684       break;
11685     }
11686     case TEK_Aggregate:
11687       llvm_unreachable(
11688           "Aggregates are not supported in lastprivate conditional.");
11689     }
11690     // }
11691     CGF.EmitBranch(ExitBB);
11692     // There is no need to emit line number for unconditional branch.
11693     (void)ApplyDebugLocation::CreateEmpty(CGF);
11694     CGF.EmitBlock(ExitBB, /*IsFinished=*/true);
11695   };
11696 
11697   if (CGM.getLangOpts().OpenMPSimd) {
11698     // Do not emit as a critical region as no parallel region could be emitted.
11699     RegionCodeGenTy ThenRCG(CodeGen);
11700     ThenRCG(CGF);
11701   } else {
11702     emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc);
11703   }
11704 }
11705 
11706 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF,
11707                                                          const Expr *LHS) {
11708   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
11709     return;
11710   LastprivateConditionalRefChecker Checker(LastprivateConditionalStack);
11711   if (!Checker.Visit(LHS))
11712     return;
11713   const Expr *FoundE;
11714   const Decl *FoundD;
11715   StringRef UniqueDeclName;
11716   LValue IVLVal;
11717   llvm::Function *FoundFn;
11718   std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) =
11719       Checker.getFoundData();
11720   if (FoundFn != CGF.CurFn) {
11721     // Special codegen for inner parallel regions.
11722     // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1;
11723     auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD);
11724     assert(It != LastprivateConditionalToTypes[FoundFn].end() &&
11725            "Lastprivate conditional is not found in outer region.");
11726     QualType StructTy = std::get<0>(It->getSecond());
11727     const FieldDecl* FiredDecl = std::get<2>(It->getSecond());
11728     LValue PrivLVal = CGF.EmitLValue(FoundE);
11729     Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11730         PrivLVal.getAddress(CGF),
11731         CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy)));
11732     LValue BaseLVal =
11733         CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl);
11734     LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl);
11735     CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get(
11736                             CGF.ConvertTypeForMem(FiredDecl->getType()), 1)),
11737                         FiredLVal, llvm::AtomicOrdering::Unordered,
11738                         /*IsVolatile=*/true, /*isInit=*/false);
11739     return;
11740   }
11741 
11742   // Private address of the lastprivate conditional in the current context.
11743   // priv_a
11744   LValue LVal = CGF.EmitLValue(FoundE);
11745   emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal,
11746                                    FoundE->getExprLoc());
11747 }
11748 
11749 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional(
11750     CodeGenFunction &CGF, const OMPExecutableDirective &D,
11751     const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) {
11752   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
11753     return;
11754   auto Range = llvm::reverse(LastprivateConditionalStack);
11755   auto It = llvm::find_if(
11756       Range, [](const LastprivateConditionalData &D) { return !D.Disabled; });
11757   if (It == Range.end() || It->Fn != CGF.CurFn)
11758     return;
11759   auto LPCI = LastprivateConditionalToTypes.find(It->Fn);
11760   assert(LPCI != LastprivateConditionalToTypes.end() &&
11761          "Lastprivates must be registered already.");
11762   SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
11763   getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind());
11764   const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back());
11765   for (const auto &Pair : It->DeclToUniqueName) {
11766     const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl());
11767     if (!CS->capturesVariable(VD) || IgnoredDecls.count(VD) > 0)
11768       continue;
11769     auto I = LPCI->getSecond().find(Pair.first);
11770     assert(I != LPCI->getSecond().end() &&
11771            "Lastprivate must be rehistered already.");
11772     // bool Cmp = priv_a.Fired != 0;
11773     LValue BaseLVal = std::get<3>(I->getSecond());
11774     LValue FiredLVal =
11775         CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond()));
11776     llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc());
11777     llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res);
11778     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then");
11779     llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done");
11780     // if (Cmp) {
11781     CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB);
11782     CGF.EmitBlock(ThenBB);
11783     Address Addr = CGF.GetAddrOfLocalVar(VD);
11784     LValue LVal;
11785     if (VD->getType()->isReferenceType())
11786       LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(),
11787                                            AlignmentSource::Decl);
11788     else
11789       LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(),
11790                                 AlignmentSource::Decl);
11791     emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal,
11792                                      D.getBeginLoc());
11793     auto AL = ApplyDebugLocation::CreateArtificial(CGF);
11794     CGF.EmitBlock(DoneBB, /*IsFinal=*/true);
11795     // }
11796   }
11797 }
11798 
11799 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate(
11800     CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD,
11801     SourceLocation Loc) {
11802   if (CGF.getLangOpts().OpenMP < 50)
11803     return;
11804   auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD);
11805   assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() &&
11806          "Unknown lastprivate conditional variable.");
11807   StringRef UniqueName = It->second;
11808   llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName);
11809   // The variable was not updated in the region - exit.
11810   if (!GV)
11811     return;
11812   LValue LPLVal = CGF.MakeAddrLValue(
11813       GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment());
11814   llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc);
11815   CGF.EmitStoreOfScalar(Res, PrivLVal);
11816 }
11817 
11818 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction(
11819     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11820     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11821   llvm_unreachable("Not supported in SIMD-only mode");
11822 }
11823 
11824 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction(
11825     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11826     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11827   llvm_unreachable("Not supported in SIMD-only mode");
11828 }
11829 
11830 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction(
11831     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11832     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
11833     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
11834     bool Tied, unsigned &NumberOfParts) {
11835   llvm_unreachable("Not supported in SIMD-only mode");
11836 }
11837 
11838 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF,
11839                                            SourceLocation Loc,
11840                                            llvm::Function *OutlinedFn,
11841                                            ArrayRef<llvm::Value *> CapturedVars,
11842                                            const Expr *IfCond) {
11843   llvm_unreachable("Not supported in SIMD-only mode");
11844 }
11845 
11846 void CGOpenMPSIMDRuntime::emitCriticalRegion(
11847     CodeGenFunction &CGF, StringRef CriticalName,
11848     const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc,
11849     const Expr *Hint) {
11850   llvm_unreachable("Not supported in SIMD-only mode");
11851 }
11852 
11853 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF,
11854                                            const RegionCodeGenTy &MasterOpGen,
11855                                            SourceLocation Loc) {
11856   llvm_unreachable("Not supported in SIMD-only mode");
11857 }
11858 
11859 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
11860                                             SourceLocation Loc) {
11861   llvm_unreachable("Not supported in SIMD-only mode");
11862 }
11863 
11864 void CGOpenMPSIMDRuntime::emitTaskgroupRegion(
11865     CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen,
11866     SourceLocation Loc) {
11867   llvm_unreachable("Not supported in SIMD-only mode");
11868 }
11869 
11870 void CGOpenMPSIMDRuntime::emitSingleRegion(
11871     CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen,
11872     SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars,
11873     ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs,
11874     ArrayRef<const Expr *> AssignmentOps) {
11875   llvm_unreachable("Not supported in SIMD-only mode");
11876 }
11877 
11878 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF,
11879                                             const RegionCodeGenTy &OrderedOpGen,
11880                                             SourceLocation Loc,
11881                                             bool IsThreads) {
11882   llvm_unreachable("Not supported in SIMD-only mode");
11883 }
11884 
11885 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF,
11886                                           SourceLocation Loc,
11887                                           OpenMPDirectiveKind Kind,
11888                                           bool EmitChecks,
11889                                           bool ForceSimpleCall) {
11890   llvm_unreachable("Not supported in SIMD-only mode");
11891 }
11892 
11893 void CGOpenMPSIMDRuntime::emitForDispatchInit(
11894     CodeGenFunction &CGF, SourceLocation Loc,
11895     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
11896     bool Ordered, const DispatchRTInput &DispatchValues) {
11897   llvm_unreachable("Not supported in SIMD-only mode");
11898 }
11899 
11900 void CGOpenMPSIMDRuntime::emitForStaticInit(
11901     CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind,
11902     const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) {
11903   llvm_unreachable("Not supported in SIMD-only mode");
11904 }
11905 
11906 void CGOpenMPSIMDRuntime::emitDistributeStaticInit(
11907     CodeGenFunction &CGF, SourceLocation Loc,
11908     OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) {
11909   llvm_unreachable("Not supported in SIMD-only mode");
11910 }
11911 
11912 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
11913                                                      SourceLocation Loc,
11914                                                      unsigned IVSize,
11915                                                      bool IVSigned) {
11916   llvm_unreachable("Not supported in SIMD-only mode");
11917 }
11918 
11919 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF,
11920                                               SourceLocation Loc,
11921                                               OpenMPDirectiveKind DKind) {
11922   llvm_unreachable("Not supported in SIMD-only mode");
11923 }
11924 
11925 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF,
11926                                               SourceLocation Loc,
11927                                               unsigned IVSize, bool IVSigned,
11928                                               Address IL, Address LB,
11929                                               Address UB, Address ST) {
11930   llvm_unreachable("Not supported in SIMD-only mode");
11931 }
11932 
11933 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
11934                                                llvm::Value *NumThreads,
11935                                                SourceLocation Loc) {
11936   llvm_unreachable("Not supported in SIMD-only mode");
11937 }
11938 
11939 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF,
11940                                              ProcBindKind ProcBind,
11941                                              SourceLocation Loc) {
11942   llvm_unreachable("Not supported in SIMD-only mode");
11943 }
11944 
11945 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
11946                                                     const VarDecl *VD,
11947                                                     Address VDAddr,
11948                                                     SourceLocation Loc) {
11949   llvm_unreachable("Not supported in SIMD-only mode");
11950 }
11951 
11952 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition(
11953     const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit,
11954     CodeGenFunction *CGF) {
11955   llvm_unreachable("Not supported in SIMD-only mode");
11956 }
11957 
11958 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate(
11959     CodeGenFunction &CGF, QualType VarType, StringRef Name) {
11960   llvm_unreachable("Not supported in SIMD-only mode");
11961 }
11962 
11963 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF,
11964                                     ArrayRef<const Expr *> Vars,
11965                                     SourceLocation Loc,
11966                                     llvm::AtomicOrdering AO) {
11967   llvm_unreachable("Not supported in SIMD-only mode");
11968 }
11969 
11970 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
11971                                        const OMPExecutableDirective &D,
11972                                        llvm::Function *TaskFunction,
11973                                        QualType SharedsTy, Address Shareds,
11974                                        const Expr *IfCond,
11975                                        const OMPTaskDataTy &Data) {
11976   llvm_unreachable("Not supported in SIMD-only mode");
11977 }
11978 
11979 void CGOpenMPSIMDRuntime::emitTaskLoopCall(
11980     CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D,
11981     llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds,
11982     const Expr *IfCond, const OMPTaskDataTy &Data) {
11983   llvm_unreachable("Not supported in SIMD-only mode");
11984 }
11985 
11986 void CGOpenMPSIMDRuntime::emitReduction(
11987     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates,
11988     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
11989     ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) {
11990   assert(Options.SimpleReduction && "Only simple reduction is expected.");
11991   CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs,
11992                                  ReductionOps, Options);
11993 }
11994 
11995 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit(
11996     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
11997     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
11998   llvm_unreachable("Not supported in SIMD-only mode");
11999 }
12000 
12001 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
12002                                                   SourceLocation Loc,
12003                                                   ReductionCodeGen &RCG,
12004                                                   unsigned N) {
12005   llvm_unreachable("Not supported in SIMD-only mode");
12006 }
12007 
12008 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF,
12009                                                   SourceLocation Loc,
12010                                                   llvm::Value *ReductionsPtr,
12011                                                   LValue SharedLVal) {
12012   llvm_unreachable("Not supported in SIMD-only mode");
12013 }
12014 
12015 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
12016                                            SourceLocation Loc) {
12017   llvm_unreachable("Not supported in SIMD-only mode");
12018 }
12019 
12020 void CGOpenMPSIMDRuntime::emitCancellationPointCall(
12021     CodeGenFunction &CGF, SourceLocation Loc,
12022     OpenMPDirectiveKind CancelRegion) {
12023   llvm_unreachable("Not supported in SIMD-only mode");
12024 }
12025 
12026 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF,
12027                                          SourceLocation Loc, const Expr *IfCond,
12028                                          OpenMPDirectiveKind CancelRegion) {
12029   llvm_unreachable("Not supported in SIMD-only mode");
12030 }
12031 
12032 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction(
12033     const OMPExecutableDirective &D, StringRef ParentName,
12034     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
12035     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
12036   llvm_unreachable("Not supported in SIMD-only mode");
12037 }
12038 
12039 void CGOpenMPSIMDRuntime::emitTargetCall(
12040     CodeGenFunction &CGF, const OMPExecutableDirective &D,
12041     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
12042     const Expr *Device,
12043     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
12044                                      const OMPLoopDirective &D)>
12045         SizeEmitter) {
12046   llvm_unreachable("Not supported in SIMD-only mode");
12047 }
12048 
12049 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) {
12050   llvm_unreachable("Not supported in SIMD-only mode");
12051 }
12052 
12053 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
12054   llvm_unreachable("Not supported in SIMD-only mode");
12055 }
12056 
12057 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) {
12058   return false;
12059 }
12060 
12061 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF,
12062                                         const OMPExecutableDirective &D,
12063                                         SourceLocation Loc,
12064                                         llvm::Function *OutlinedFn,
12065                                         ArrayRef<llvm::Value *> CapturedVars) {
12066   llvm_unreachable("Not supported in SIMD-only mode");
12067 }
12068 
12069 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
12070                                              const Expr *NumTeams,
12071                                              const Expr *ThreadLimit,
12072                                              SourceLocation Loc) {
12073   llvm_unreachable("Not supported in SIMD-only mode");
12074 }
12075 
12076 void CGOpenMPSIMDRuntime::emitTargetDataCalls(
12077     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12078     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
12079   llvm_unreachable("Not supported in SIMD-only mode");
12080 }
12081 
12082 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall(
12083     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12084     const Expr *Device) {
12085   llvm_unreachable("Not supported in SIMD-only mode");
12086 }
12087 
12088 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF,
12089                                            const OMPLoopDirective &D,
12090                                            ArrayRef<Expr *> NumIterations) {
12091   llvm_unreachable("Not supported in SIMD-only mode");
12092 }
12093 
12094 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
12095                                               const OMPDependClause *C) {
12096   llvm_unreachable("Not supported in SIMD-only mode");
12097 }
12098 
12099 const VarDecl *
12100 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD,
12101                                         const VarDecl *NativeParam) const {
12102   llvm_unreachable("Not supported in SIMD-only mode");
12103 }
12104 
12105 Address
12106 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF,
12107                                          const VarDecl *NativeParam,
12108                                          const VarDecl *TargetParam) const {
12109   llvm_unreachable("Not supported in SIMD-only mode");
12110 }
12111