1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This provides a class for OpenMP runtime code generation.
10 //
11 //===----------------------------------------------------------------------===//
12 
13 #include "CGOpenMPRuntime.h"
14 #include "CGCXXABI.h"
15 #include "CGCleanup.h"
16 #include "CGRecordLayout.h"
17 #include "CodeGenFunction.h"
18 #include "clang/AST/Attr.h"
19 #include "clang/AST/Decl.h"
20 #include "clang/AST/OpenMPClause.h"
21 #include "clang/AST/StmtOpenMP.h"
22 #include "clang/AST/StmtVisitor.h"
23 #include "clang/Basic/BitmaskEnum.h"
24 #include "clang/Basic/FileManager.h"
25 #include "clang/Basic/OpenMPKinds.h"
26 #include "clang/Basic/SourceManager.h"
27 #include "clang/CodeGen/ConstantInitBuilder.h"
28 #include "llvm/ADT/ArrayRef.h"
29 #include "llvm/ADT/SetOperations.h"
30 #include "llvm/ADT/StringExtras.h"
31 #include "llvm/Bitcode/BitcodeReader.h"
32 #include "llvm/Frontend/OpenMP/OMPIRBuilder.h"
33 #include "llvm/IR/Constants.h"
34 #include "llvm/IR/DerivedTypes.h"
35 #include "llvm/IR/GlobalValue.h"
36 #include "llvm/IR/Value.h"
37 #include "llvm/Support/AtomicOrdering.h"
38 #include "llvm/Support/Format.h"
39 #include "llvm/Support/raw_ostream.h"
40 #include <cassert>
41 
42 using namespace clang;
43 using namespace CodeGen;
44 using namespace llvm::omp;
45 
46 namespace {
47 /// Base class for handling code generation inside OpenMP regions.
48 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
49 public:
50   /// Kinds of OpenMP regions used in codegen.
51   enum CGOpenMPRegionKind {
52     /// Region with outlined function for standalone 'parallel'
53     /// directive.
54     ParallelOutlinedRegion,
55     /// Region with outlined function for standalone 'task' directive.
56     TaskOutlinedRegion,
57     /// Region for constructs that do not require function outlining,
58     /// like 'for', 'sections', 'atomic' etc. directives.
59     InlinedRegion,
60     /// Region with outlined function for standalone 'target' directive.
61     TargetRegion,
62   };
63 
64   CGOpenMPRegionInfo(const CapturedStmt &CS,
65                      const CGOpenMPRegionKind RegionKind,
66                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
67                      bool HasCancel)
68       : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
69         CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {}
70 
71   CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
72                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
73                      bool HasCancel)
74       : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen),
75         Kind(Kind), HasCancel(HasCancel) {}
76 
77   /// Get a variable or parameter for storing global thread id
78   /// inside OpenMP construct.
79   virtual const VarDecl *getThreadIDVariable() const = 0;
80 
81   /// Emit the captured statement body.
82   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
83 
84   /// Get an LValue for the current ThreadID variable.
85   /// \return LValue for thread id variable. This LValue always has type int32*.
86   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
87 
88   virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {}
89 
90   CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
91 
92   OpenMPDirectiveKind getDirectiveKind() const { return Kind; }
93 
94   bool hasCancel() const { return HasCancel; }
95 
96   static bool classof(const CGCapturedStmtInfo *Info) {
97     return Info->getKind() == CR_OpenMP;
98   }
99 
100   ~CGOpenMPRegionInfo() override = default;
101 
102 protected:
103   CGOpenMPRegionKind RegionKind;
104   RegionCodeGenTy CodeGen;
105   OpenMPDirectiveKind Kind;
106   bool HasCancel;
107 };
108 
109 /// API for captured statement code generation in OpenMP constructs.
110 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo {
111 public:
112   CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
113                              const RegionCodeGenTy &CodeGen,
114                              OpenMPDirectiveKind Kind, bool HasCancel,
115                              StringRef HelperName)
116       : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind,
117                            HasCancel),
118         ThreadIDVar(ThreadIDVar), HelperName(HelperName) {
119     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
120   }
121 
122   /// Get a variable or parameter for storing global thread id
123   /// inside OpenMP construct.
124   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
125 
126   /// Get the name of the capture helper.
127   StringRef getHelperName() const override { return HelperName; }
128 
129   static bool classof(const CGCapturedStmtInfo *Info) {
130     return CGOpenMPRegionInfo::classof(Info) &&
131            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
132                ParallelOutlinedRegion;
133   }
134 
135 private:
136   /// A variable or parameter storing global thread id for OpenMP
137   /// constructs.
138   const VarDecl *ThreadIDVar;
139   StringRef HelperName;
140 };
141 
142 /// API for captured statement code generation in OpenMP constructs.
143 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo {
144 public:
145   class UntiedTaskActionTy final : public PrePostActionTy {
146     bool Untied;
147     const VarDecl *PartIDVar;
148     const RegionCodeGenTy UntiedCodeGen;
149     llvm::SwitchInst *UntiedSwitch = nullptr;
150 
151   public:
152     UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar,
153                        const RegionCodeGenTy &UntiedCodeGen)
154         : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {}
155     void Enter(CodeGenFunction &CGF) override {
156       if (Untied) {
157         // Emit task switching point.
158         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
159             CGF.GetAddrOfLocalVar(PartIDVar),
160             PartIDVar->getType()->castAs<PointerType>());
161         llvm::Value *Res =
162             CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation());
163         llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done.");
164         UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB);
165         CGF.EmitBlock(DoneBB);
166         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
167         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
168         UntiedSwitch->addCase(CGF.Builder.getInt32(0),
169                               CGF.Builder.GetInsertBlock());
170         emitUntiedSwitch(CGF);
171       }
172     }
173     void emitUntiedSwitch(CodeGenFunction &CGF) const {
174       if (Untied) {
175         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
176             CGF.GetAddrOfLocalVar(PartIDVar),
177             PartIDVar->getType()->castAs<PointerType>());
178         CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
179                               PartIdLVal);
180         UntiedCodeGen(CGF);
181         CodeGenFunction::JumpDest CurPoint =
182             CGF.getJumpDestInCurrentScope(".untied.next.");
183         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
184         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
185         UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
186                               CGF.Builder.GetInsertBlock());
187         CGF.EmitBranchThroughCleanup(CurPoint);
188         CGF.EmitBlock(CurPoint.getBlock());
189       }
190     }
191     unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); }
192   };
193   CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
194                                  const VarDecl *ThreadIDVar,
195                                  const RegionCodeGenTy &CodeGen,
196                                  OpenMPDirectiveKind Kind, bool HasCancel,
197                                  const UntiedTaskActionTy &Action)
198       : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel),
199         ThreadIDVar(ThreadIDVar), Action(Action) {
200     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
201   }
202 
203   /// Get a variable or parameter for storing global thread id
204   /// inside OpenMP construct.
205   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
206 
207   /// Get an LValue for the current ThreadID variable.
208   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
209 
210   /// Get the name of the capture helper.
211   StringRef getHelperName() const override { return ".omp_outlined."; }
212 
213   void emitUntiedSwitch(CodeGenFunction &CGF) override {
214     Action.emitUntiedSwitch(CGF);
215   }
216 
217   static bool classof(const CGCapturedStmtInfo *Info) {
218     return CGOpenMPRegionInfo::classof(Info) &&
219            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
220                TaskOutlinedRegion;
221   }
222 
223 private:
224   /// A variable or parameter storing global thread id for OpenMP
225   /// constructs.
226   const VarDecl *ThreadIDVar;
227   /// Action for emitting code for untied tasks.
228   const UntiedTaskActionTy &Action;
229 };
230 
231 /// API for inlined captured statement code generation in OpenMP
232 /// constructs.
233 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
234 public:
235   CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
236                             const RegionCodeGenTy &CodeGen,
237                             OpenMPDirectiveKind Kind, bool HasCancel)
238       : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel),
239         OldCSI(OldCSI),
240         OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
241 
242   // Retrieve the value of the context parameter.
243   llvm::Value *getContextValue() const override {
244     if (OuterRegionInfo)
245       return OuterRegionInfo->getContextValue();
246     llvm_unreachable("No context value for inlined OpenMP region");
247   }
248 
249   void setContextValue(llvm::Value *V) override {
250     if (OuterRegionInfo) {
251       OuterRegionInfo->setContextValue(V);
252       return;
253     }
254     llvm_unreachable("No context value for inlined OpenMP region");
255   }
256 
257   /// Lookup the captured field decl for a variable.
258   const FieldDecl *lookup(const VarDecl *VD) const override {
259     if (OuterRegionInfo)
260       return OuterRegionInfo->lookup(VD);
261     // If there is no outer outlined region,no need to lookup in a list of
262     // captured variables, we can use the original one.
263     return nullptr;
264   }
265 
266   FieldDecl *getThisFieldDecl() const override {
267     if (OuterRegionInfo)
268       return OuterRegionInfo->getThisFieldDecl();
269     return nullptr;
270   }
271 
272   /// Get a variable or parameter for storing global thread id
273   /// inside OpenMP construct.
274   const VarDecl *getThreadIDVariable() const override {
275     if (OuterRegionInfo)
276       return OuterRegionInfo->getThreadIDVariable();
277     return nullptr;
278   }
279 
280   /// Get an LValue for the current ThreadID variable.
281   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override {
282     if (OuterRegionInfo)
283       return OuterRegionInfo->getThreadIDVariableLValue(CGF);
284     llvm_unreachable("No LValue for inlined OpenMP construct");
285   }
286 
287   /// Get the name of the capture helper.
288   StringRef getHelperName() const override {
289     if (auto *OuterRegionInfo = getOldCSI())
290       return OuterRegionInfo->getHelperName();
291     llvm_unreachable("No helper name for inlined OpenMP construct");
292   }
293 
294   void emitUntiedSwitch(CodeGenFunction &CGF) override {
295     if (OuterRegionInfo)
296       OuterRegionInfo->emitUntiedSwitch(CGF);
297   }
298 
299   CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
300 
301   static bool classof(const CGCapturedStmtInfo *Info) {
302     return CGOpenMPRegionInfo::classof(Info) &&
303            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion;
304   }
305 
306   ~CGOpenMPInlinedRegionInfo() override = default;
307 
308 private:
309   /// CodeGen info about outer OpenMP region.
310   CodeGenFunction::CGCapturedStmtInfo *OldCSI;
311   CGOpenMPRegionInfo *OuterRegionInfo;
312 };
313 
314 /// API for captured statement code generation in OpenMP target
315 /// constructs. For this captures, implicit parameters are used instead of the
316 /// captured fields. The name of the target region has to be unique in a given
317 /// application so it is provided by the client, because only the client has
318 /// the information to generate that.
319 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo {
320 public:
321   CGOpenMPTargetRegionInfo(const CapturedStmt &CS,
322                            const RegionCodeGenTy &CodeGen, StringRef HelperName)
323       : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target,
324                            /*HasCancel=*/false),
325         HelperName(HelperName) {}
326 
327   /// This is unused for target regions because each starts executing
328   /// with a single thread.
329   const VarDecl *getThreadIDVariable() const override { return nullptr; }
330 
331   /// Get the name of the capture helper.
332   StringRef getHelperName() const override { return HelperName; }
333 
334   static bool classof(const CGCapturedStmtInfo *Info) {
335     return CGOpenMPRegionInfo::classof(Info) &&
336            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion;
337   }
338 
339 private:
340   StringRef HelperName;
341 };
342 
343 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) {
344   llvm_unreachable("No codegen for expressions");
345 }
346 /// API for generation of expressions captured in a innermost OpenMP
347 /// region.
348 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo {
349 public:
350   CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS)
351       : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen,
352                                   OMPD_unknown,
353                                   /*HasCancel=*/false),
354         PrivScope(CGF) {
355     // Make sure the globals captured in the provided statement are local by
356     // using the privatization logic. We assume the same variable is not
357     // captured more than once.
358     for (const auto &C : CS.captures()) {
359       if (!C.capturesVariable() && !C.capturesVariableByCopy())
360         continue;
361 
362       const VarDecl *VD = C.getCapturedVar();
363       if (VD->isLocalVarDeclOrParm())
364         continue;
365 
366       DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
367                       /*RefersToEnclosingVariableOrCapture=*/false,
368                       VD->getType().getNonReferenceType(), VK_LValue,
369                       C.getLocation());
370       PrivScope.addPrivate(
371           VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(CGF); });
372     }
373     (void)PrivScope.Privatize();
374   }
375 
376   /// Lookup the captured field decl for a variable.
377   const FieldDecl *lookup(const VarDecl *VD) const override {
378     if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD))
379       return FD;
380     return nullptr;
381   }
382 
383   /// Emit the captured statement body.
384   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override {
385     llvm_unreachable("No body for expressions");
386   }
387 
388   /// Get a variable or parameter for storing global thread id
389   /// inside OpenMP construct.
390   const VarDecl *getThreadIDVariable() const override {
391     llvm_unreachable("No thread id for expressions");
392   }
393 
394   /// Get the name of the capture helper.
395   StringRef getHelperName() const override {
396     llvm_unreachable("No helper name for expressions");
397   }
398 
399   static bool classof(const CGCapturedStmtInfo *Info) { return false; }
400 
401 private:
402   /// Private scope to capture global variables.
403   CodeGenFunction::OMPPrivateScope PrivScope;
404 };
405 
406 /// RAII for emitting code of OpenMP constructs.
407 class InlinedOpenMPRegionRAII {
408   CodeGenFunction &CGF;
409   llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields;
410   FieldDecl *LambdaThisCaptureField = nullptr;
411   const CodeGen::CGBlockInfo *BlockInfo = nullptr;
412 
413 public:
414   /// Constructs region for combined constructs.
415   /// \param CodeGen Code generation sequence for combined directives. Includes
416   /// a list of functions used for code generation of implicitly inlined
417   /// regions.
418   InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen,
419                           OpenMPDirectiveKind Kind, bool HasCancel)
420       : CGF(CGF) {
421     // Start emission for the construct.
422     CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(
423         CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel);
424     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
425     LambdaThisCaptureField = CGF.LambdaThisCaptureField;
426     CGF.LambdaThisCaptureField = nullptr;
427     BlockInfo = CGF.BlockInfo;
428     CGF.BlockInfo = nullptr;
429   }
430 
431   ~InlinedOpenMPRegionRAII() {
432     // Restore original CapturedStmtInfo only if we're done with code emission.
433     auto *OldCSI =
434         cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
435     delete CGF.CapturedStmtInfo;
436     CGF.CapturedStmtInfo = OldCSI;
437     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
438     CGF.LambdaThisCaptureField = LambdaThisCaptureField;
439     CGF.BlockInfo = BlockInfo;
440   }
441 };
442 
443 /// Values for bit flags used in the ident_t to describe the fields.
444 /// All enumeric elements are named and described in accordance with the code
445 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
446 enum OpenMPLocationFlags : unsigned {
447   /// Use trampoline for internal microtask.
448   OMP_IDENT_IMD = 0x01,
449   /// Use c-style ident structure.
450   OMP_IDENT_KMPC = 0x02,
451   /// Atomic reduction option for kmpc_reduce.
452   OMP_ATOMIC_REDUCE = 0x10,
453   /// Explicit 'barrier' directive.
454   OMP_IDENT_BARRIER_EXPL = 0x20,
455   /// Implicit barrier in code.
456   OMP_IDENT_BARRIER_IMPL = 0x40,
457   /// Implicit barrier in 'for' directive.
458   OMP_IDENT_BARRIER_IMPL_FOR = 0x40,
459   /// Implicit barrier in 'sections' directive.
460   OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0,
461   /// Implicit barrier in 'single' directive.
462   OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140,
463   /// Call of __kmp_for_static_init for static loop.
464   OMP_IDENT_WORK_LOOP = 0x200,
465   /// Call of __kmp_for_static_init for sections.
466   OMP_IDENT_WORK_SECTIONS = 0x400,
467   /// Call of __kmp_for_static_init for distribute.
468   OMP_IDENT_WORK_DISTRIBUTE = 0x800,
469   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE)
470 };
471 
472 namespace {
473 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
474 /// Values for bit flags for marking which requires clauses have been used.
475 enum OpenMPOffloadingRequiresDirFlags : int64_t {
476   /// flag undefined.
477   OMP_REQ_UNDEFINED               = 0x000,
478   /// no requires clause present.
479   OMP_REQ_NONE                    = 0x001,
480   /// reverse_offload clause.
481   OMP_REQ_REVERSE_OFFLOAD         = 0x002,
482   /// unified_address clause.
483   OMP_REQ_UNIFIED_ADDRESS         = 0x004,
484   /// unified_shared_memory clause.
485   OMP_REQ_UNIFIED_SHARED_MEMORY   = 0x008,
486   /// dynamic_allocators clause.
487   OMP_REQ_DYNAMIC_ALLOCATORS      = 0x010,
488   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS)
489 };
490 
491 enum OpenMPOffloadingReservedDeviceIDs {
492   /// Device ID if the device was not defined, runtime should get it
493   /// from environment variables in the spec.
494   OMP_DEVICEID_UNDEF = -1,
495 };
496 } // anonymous namespace
497 
498 /// Describes ident structure that describes a source location.
499 /// All descriptions are taken from
500 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
501 /// Original structure:
502 /// typedef struct ident {
503 ///    kmp_int32 reserved_1;   /**<  might be used in Fortran;
504 ///                                  see above  */
505 ///    kmp_int32 flags;        /**<  also f.flags; KMP_IDENT_xxx flags;
506 ///                                  KMP_IDENT_KMPC identifies this union
507 ///                                  member  */
508 ///    kmp_int32 reserved_2;   /**<  not really used in Fortran any more;
509 ///                                  see above */
510 ///#if USE_ITT_BUILD
511 ///                            /*  but currently used for storing
512 ///                                region-specific ITT */
513 ///                            /*  contextual information. */
514 ///#endif /* USE_ITT_BUILD */
515 ///    kmp_int32 reserved_3;   /**< source[4] in Fortran, do not use for
516 ///                                 C++  */
517 ///    char const *psource;    /**< String describing the source location.
518 ///                            The string is composed of semi-colon separated
519 //                             fields which describe the source file,
520 ///                            the function and a pair of line numbers that
521 ///                            delimit the construct.
522 ///                             */
523 /// } ident_t;
524 enum IdentFieldIndex {
525   /// might be used in Fortran
526   IdentField_Reserved_1,
527   /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member.
528   IdentField_Flags,
529   /// Not really used in Fortran any more
530   IdentField_Reserved_2,
531   /// Source[4] in Fortran, do not use for C++
532   IdentField_Reserved_3,
533   /// String describing the source location. The string is composed of
534   /// semi-colon separated fields which describe the source file, the function
535   /// and a pair of line numbers that delimit the construct.
536   IdentField_PSource
537 };
538 
539 /// Schedule types for 'omp for' loops (these enumerators are taken from
540 /// the enum sched_type in kmp.h).
541 enum OpenMPSchedType {
542   /// Lower bound for default (unordered) versions.
543   OMP_sch_lower = 32,
544   OMP_sch_static_chunked = 33,
545   OMP_sch_static = 34,
546   OMP_sch_dynamic_chunked = 35,
547   OMP_sch_guided_chunked = 36,
548   OMP_sch_runtime = 37,
549   OMP_sch_auto = 38,
550   /// static with chunk adjustment (e.g., simd)
551   OMP_sch_static_balanced_chunked = 45,
552   /// Lower bound for 'ordered' versions.
553   OMP_ord_lower = 64,
554   OMP_ord_static_chunked = 65,
555   OMP_ord_static = 66,
556   OMP_ord_dynamic_chunked = 67,
557   OMP_ord_guided_chunked = 68,
558   OMP_ord_runtime = 69,
559   OMP_ord_auto = 70,
560   OMP_sch_default = OMP_sch_static,
561   /// dist_schedule types
562   OMP_dist_sch_static_chunked = 91,
563   OMP_dist_sch_static = 92,
564   /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers.
565   /// Set if the monotonic schedule modifier was present.
566   OMP_sch_modifier_monotonic = (1 << 29),
567   /// Set if the nonmonotonic schedule modifier was present.
568   OMP_sch_modifier_nonmonotonic = (1 << 30),
569 };
570 
571 enum OpenMPRTLFunction {
572   /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc,
573   /// kmpc_micro microtask, ...);
574   OMPRTL__kmpc_fork_call,
575   /// Call to void *__kmpc_threadprivate_cached(ident_t *loc,
576   /// kmp_int32 global_tid, void *data, size_t size, void ***cache);
577   OMPRTL__kmpc_threadprivate_cached,
578   /// Call to void __kmpc_threadprivate_register( ident_t *,
579   /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
580   OMPRTL__kmpc_threadprivate_register,
581   // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc);
582   OMPRTL__kmpc_global_thread_num,
583   // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
584   // kmp_critical_name *crit);
585   OMPRTL__kmpc_critical,
586   // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32
587   // global_tid, kmp_critical_name *crit, uintptr_t hint);
588   OMPRTL__kmpc_critical_with_hint,
589   // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
590   // kmp_critical_name *crit);
591   OMPRTL__kmpc_end_critical,
592   // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
593   // global_tid);
594   OMPRTL__kmpc_cancel_barrier,
595   // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
596   OMPRTL__kmpc_barrier,
597   // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
598   OMPRTL__kmpc_for_static_fini,
599   // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
600   // global_tid);
601   OMPRTL__kmpc_serialized_parallel,
602   // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
603   // global_tid);
604   OMPRTL__kmpc_end_serialized_parallel,
605   // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
606   // kmp_int32 num_threads);
607   OMPRTL__kmpc_push_num_threads,
608   // Call to void __kmpc_flush(ident_t *loc);
609   OMPRTL__kmpc_flush,
610   // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid);
611   OMPRTL__kmpc_master,
612   // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid);
613   OMPRTL__kmpc_end_master,
614   // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
615   // int end_part);
616   OMPRTL__kmpc_omp_taskyield,
617   // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid);
618   OMPRTL__kmpc_single,
619   // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid);
620   OMPRTL__kmpc_end_single,
621   // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
622   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
623   // kmp_routine_entry_t *task_entry);
624   OMPRTL__kmpc_omp_task_alloc,
625   // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *,
626   // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t,
627   // size_t sizeof_shareds, kmp_routine_entry_t *task_entry,
628   // kmp_int64 device_id);
629   OMPRTL__kmpc_omp_target_task_alloc,
630   // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t *
631   // new_task);
632   OMPRTL__kmpc_omp_task,
633   // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
634   // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
635   // kmp_int32 didit);
636   OMPRTL__kmpc_copyprivate,
637   // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
638   // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
639   // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
640   OMPRTL__kmpc_reduce,
641   // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
642   // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
643   // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
644   // *lck);
645   OMPRTL__kmpc_reduce_nowait,
646   // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
647   // kmp_critical_name *lck);
648   OMPRTL__kmpc_end_reduce,
649   // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
650   // kmp_critical_name *lck);
651   OMPRTL__kmpc_end_reduce_nowait,
652   // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
653   // kmp_task_t * new_task);
654   OMPRTL__kmpc_omp_task_begin_if0,
655   // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
656   // kmp_task_t * new_task);
657   OMPRTL__kmpc_omp_task_complete_if0,
658   // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
659   OMPRTL__kmpc_ordered,
660   // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
661   OMPRTL__kmpc_end_ordered,
662   // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
663   // global_tid);
664   OMPRTL__kmpc_omp_taskwait,
665   // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
666   OMPRTL__kmpc_taskgroup,
667   // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
668   OMPRTL__kmpc_end_taskgroup,
669   // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
670   // int proc_bind);
671   OMPRTL__kmpc_push_proc_bind,
672   // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32
673   // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t
674   // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
675   OMPRTL__kmpc_omp_task_with_deps,
676   // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32
677   // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
678   // ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
679   OMPRTL__kmpc_omp_wait_deps,
680   // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
681   // global_tid, kmp_int32 cncl_kind);
682   OMPRTL__kmpc_cancellationpoint,
683   // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
684   // kmp_int32 cncl_kind);
685   OMPRTL__kmpc_cancel,
686   // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid,
687   // kmp_int32 num_teams, kmp_int32 thread_limit);
688   OMPRTL__kmpc_push_num_teams,
689   // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
690   // microtask, ...);
691   OMPRTL__kmpc_fork_teams,
692   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
693   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
694   // sched, kmp_uint64 grainsize, void *task_dup);
695   OMPRTL__kmpc_taskloop,
696   // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
697   // num_dims, struct kmp_dim *dims);
698   OMPRTL__kmpc_doacross_init,
699   // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
700   OMPRTL__kmpc_doacross_fini,
701   // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
702   // *vec);
703   OMPRTL__kmpc_doacross_post,
704   // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
705   // *vec);
706   OMPRTL__kmpc_doacross_wait,
707   // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void
708   // *data);
709   OMPRTL__kmpc_task_reduction_init,
710   // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
711   // *d);
712   OMPRTL__kmpc_task_reduction_get_th_data,
713   // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al);
714   OMPRTL__kmpc_alloc,
715   // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al);
716   OMPRTL__kmpc_free,
717 
718   //
719   // Offloading related calls
720   //
721   // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
722   // size);
723   OMPRTL__kmpc_push_target_tripcount,
724   // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
725   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
726   // *arg_types);
727   OMPRTL__tgt_target,
728   // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
729   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
730   // *arg_types);
731   OMPRTL__tgt_target_nowait,
732   // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
733   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
734   // *arg_types, int32_t num_teams, int32_t thread_limit);
735   OMPRTL__tgt_target_teams,
736   // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void
737   // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
738   // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
739   OMPRTL__tgt_target_teams_nowait,
740   // Call to void __tgt_register_requires(int64_t flags);
741   OMPRTL__tgt_register_requires,
742   // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
743   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
744   OMPRTL__tgt_target_data_begin,
745   // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
746   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
747   // *arg_types);
748   OMPRTL__tgt_target_data_begin_nowait,
749   // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
750   // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types);
751   OMPRTL__tgt_target_data_end,
752   // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t
753   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
754   // *arg_types);
755   OMPRTL__tgt_target_data_end_nowait,
756   // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
757   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
758   OMPRTL__tgt_target_data_update,
759   // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t
760   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
761   // *arg_types);
762   OMPRTL__tgt_target_data_update_nowait,
763   // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
764   OMPRTL__tgt_mapper_num_components,
765   // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void
766   // *base, void *begin, int64_t size, int64_t type);
767   OMPRTL__tgt_push_mapper_component,
768   // Call to kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
769   // int gtid, kmp_task_t *task);
770   OMPRTL__kmpc_task_allow_completion_event,
771 };
772 
773 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP
774 /// region.
775 class CleanupTy final : public EHScopeStack::Cleanup {
776   PrePostActionTy *Action;
777 
778 public:
779   explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {}
780   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
781     if (!CGF.HaveInsertPoint())
782       return;
783     Action->Exit(CGF);
784   }
785 };
786 
787 } // anonymous namespace
788 
789 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const {
790   CodeGenFunction::RunCleanupsScope Scope(CGF);
791   if (PrePostAction) {
792     CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction);
793     Callback(CodeGen, CGF, *PrePostAction);
794   } else {
795     PrePostActionTy Action;
796     Callback(CodeGen, CGF, Action);
797   }
798 }
799 
800 /// Check if the combiner is a call to UDR combiner and if it is so return the
801 /// UDR decl used for reduction.
802 static const OMPDeclareReductionDecl *
803 getReductionInit(const Expr *ReductionOp) {
804   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
805     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
806       if (const auto *DRE =
807               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
808         if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl()))
809           return DRD;
810   return nullptr;
811 }
812 
813 static void emitInitWithReductionInitializer(CodeGenFunction &CGF,
814                                              const OMPDeclareReductionDecl *DRD,
815                                              const Expr *InitOp,
816                                              Address Private, Address Original,
817                                              QualType Ty) {
818   if (DRD->getInitializer()) {
819     std::pair<llvm::Function *, llvm::Function *> Reduction =
820         CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
821     const auto *CE = cast<CallExpr>(InitOp);
822     const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee());
823     const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts();
824     const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts();
825     const auto *LHSDRE =
826         cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr());
827     const auto *RHSDRE =
828         cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr());
829     CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
830     PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()),
831                             [=]() { return Private; });
832     PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()),
833                             [=]() { return Original; });
834     (void)PrivateScope.Privatize();
835     RValue Func = RValue::get(Reduction.second);
836     CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
837     CGF.EmitIgnoredExpr(InitOp);
838   } else {
839     llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty);
840     std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"});
841     auto *GV = new llvm::GlobalVariable(
842         CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true,
843         llvm::GlobalValue::PrivateLinkage, Init, Name);
844     LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty);
845     RValue InitRVal;
846     switch (CGF.getEvaluationKind(Ty)) {
847     case TEK_Scalar:
848       InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation());
849       break;
850     case TEK_Complex:
851       InitRVal =
852           RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation()));
853       break;
854     case TEK_Aggregate:
855       InitRVal = RValue::getAggregate(LV.getAddress(CGF));
856       break;
857     }
858     OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue);
859     CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal);
860     CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
861                          /*IsInitializer=*/false);
862   }
863 }
864 
865 /// Emit initialization of arrays of complex types.
866 /// \param DestAddr Address of the array.
867 /// \param Type Type of array.
868 /// \param Init Initial expression of array.
869 /// \param SrcAddr Address of the original array.
870 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr,
871                                  QualType Type, bool EmitDeclareReductionInit,
872                                  const Expr *Init,
873                                  const OMPDeclareReductionDecl *DRD,
874                                  Address SrcAddr = Address::invalid()) {
875   // Perform element-by-element initialization.
876   QualType ElementTy;
877 
878   // Drill down to the base element type on both arrays.
879   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
880   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr);
881   DestAddr =
882       CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType());
883   if (DRD)
884     SrcAddr =
885         CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType());
886 
887   llvm::Value *SrcBegin = nullptr;
888   if (DRD)
889     SrcBegin = SrcAddr.getPointer();
890   llvm::Value *DestBegin = DestAddr.getPointer();
891   // Cast from pointer to array type to pointer to single element.
892   llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements);
893   // The basic structure here is a while-do loop.
894   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body");
895   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done");
896   llvm::Value *IsEmpty =
897       CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty");
898   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
899 
900   // Enter the loop body, making that address the current address.
901   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
902   CGF.EmitBlock(BodyBB);
903 
904   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
905 
906   llvm::PHINode *SrcElementPHI = nullptr;
907   Address SrcElementCurrent = Address::invalid();
908   if (DRD) {
909     SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2,
910                                           "omp.arraycpy.srcElementPast");
911     SrcElementPHI->addIncoming(SrcBegin, EntryBB);
912     SrcElementCurrent =
913         Address(SrcElementPHI,
914                 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize));
915   }
916   llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI(
917       DestBegin->getType(), 2, "omp.arraycpy.destElementPast");
918   DestElementPHI->addIncoming(DestBegin, EntryBB);
919   Address DestElementCurrent =
920       Address(DestElementPHI,
921               DestAddr.getAlignment().alignmentOfArrayElement(ElementSize));
922 
923   // Emit copy.
924   {
925     CodeGenFunction::RunCleanupsScope InitScope(CGF);
926     if (EmitDeclareReductionInit) {
927       emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent,
928                                        SrcElementCurrent, ElementTy);
929     } else
930       CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(),
931                            /*IsInitializer=*/false);
932   }
933 
934   if (DRD) {
935     // Shift the address forward by one element.
936     llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32(
937         SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
938     SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock());
939   }
940 
941   // Shift the address forward by one element.
942   llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32(
943       DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
944   // Check whether we've reached the end.
945   llvm::Value *Done =
946       CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done");
947   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
948   DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock());
949 
950   // Done.
951   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
952 }
953 
954 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) {
955   return CGF.EmitOMPSharedLValue(E);
956 }
957 
958 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF,
959                                             const Expr *E) {
960   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E))
961     return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false);
962   return LValue();
963 }
964 
965 void ReductionCodeGen::emitAggregateInitialization(
966     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
967     const OMPDeclareReductionDecl *DRD) {
968   // Emit VarDecl with copy init for arrays.
969   // Get the address of the original variable captured in current
970   // captured region.
971   const auto *PrivateVD =
972       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
973   bool EmitDeclareReductionInit =
974       DRD && (DRD->getInitializer() || !PrivateVD->hasInit());
975   EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(),
976                        EmitDeclareReductionInit,
977                        EmitDeclareReductionInit ? ClausesData[N].ReductionOp
978                                                 : PrivateVD->getInit(),
979                        DRD, SharedLVal.getAddress(CGF));
980 }
981 
982 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds,
983                                    ArrayRef<const Expr *> Privates,
984                                    ArrayRef<const Expr *> ReductionOps) {
985   ClausesData.reserve(Shareds.size());
986   SharedAddresses.reserve(Shareds.size());
987   Sizes.reserve(Shareds.size());
988   BaseDecls.reserve(Shareds.size());
989   auto IPriv = Privates.begin();
990   auto IRed = ReductionOps.begin();
991   for (const Expr *Ref : Shareds) {
992     ClausesData.emplace_back(Ref, *IPriv, *IRed);
993     std::advance(IPriv, 1);
994     std::advance(IRed, 1);
995   }
996 }
997 
998 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) {
999   assert(SharedAddresses.size() == N &&
1000          "Number of generated lvalues must be exactly N.");
1001   LValue First = emitSharedLValue(CGF, ClausesData[N].Ref);
1002   LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref);
1003   SharedAddresses.emplace_back(First, Second);
1004 }
1005 
1006 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) {
1007   const auto *PrivateVD =
1008       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1009   QualType PrivateType = PrivateVD->getType();
1010   bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref);
1011   if (!PrivateType->isVariablyModifiedType()) {
1012     Sizes.emplace_back(
1013         CGF.getTypeSize(
1014             SharedAddresses[N].first.getType().getNonReferenceType()),
1015         nullptr);
1016     return;
1017   }
1018   llvm::Value *Size;
1019   llvm::Value *SizeInChars;
1020   auto *ElemType = cast<llvm::PointerType>(
1021                        SharedAddresses[N].first.getPointer(CGF)->getType())
1022                        ->getElementType();
1023   auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType);
1024   if (AsArraySection) {
1025     Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(CGF),
1026                                      SharedAddresses[N].first.getPointer(CGF));
1027     Size = CGF.Builder.CreateNUWAdd(
1028         Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1));
1029     SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf);
1030   } else {
1031     SizeInChars = CGF.getTypeSize(
1032         SharedAddresses[N].first.getType().getNonReferenceType());
1033     Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf);
1034   }
1035   Sizes.emplace_back(SizeInChars, Size);
1036   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1037       CGF,
1038       cast<OpaqueValueExpr>(
1039           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1040       RValue::get(Size));
1041   CGF.EmitVariablyModifiedType(PrivateType);
1042 }
1043 
1044 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N,
1045                                          llvm::Value *Size) {
1046   const auto *PrivateVD =
1047       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1048   QualType PrivateType = PrivateVD->getType();
1049   if (!PrivateType->isVariablyModifiedType()) {
1050     assert(!Size && !Sizes[N].second &&
1051            "Size should be nullptr for non-variably modified reduction "
1052            "items.");
1053     return;
1054   }
1055   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1056       CGF,
1057       cast<OpaqueValueExpr>(
1058           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1059       RValue::get(Size));
1060   CGF.EmitVariablyModifiedType(PrivateType);
1061 }
1062 
1063 void ReductionCodeGen::emitInitialization(
1064     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
1065     llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) {
1066   assert(SharedAddresses.size() > N && "No variable was generated");
1067   const auto *PrivateVD =
1068       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1069   const OMPDeclareReductionDecl *DRD =
1070       getReductionInit(ClausesData[N].ReductionOp);
1071   QualType PrivateType = PrivateVD->getType();
1072   PrivateAddr = CGF.Builder.CreateElementBitCast(
1073       PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1074   QualType SharedType = SharedAddresses[N].first.getType();
1075   SharedLVal = CGF.MakeAddrLValue(
1076       CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(CGF),
1077                                        CGF.ConvertTypeForMem(SharedType)),
1078       SharedType, SharedAddresses[N].first.getBaseInfo(),
1079       CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType));
1080   if (CGF.getContext().getAsArrayType(PrivateVD->getType())) {
1081     emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD);
1082   } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) {
1083     emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp,
1084                                      PrivateAddr, SharedLVal.getAddress(CGF),
1085                                      SharedLVal.getType());
1086   } else if (!DefaultInit(CGF) && PrivateVD->hasInit() &&
1087              !CGF.isTrivialInitializer(PrivateVD->getInit())) {
1088     CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr,
1089                          PrivateVD->getType().getQualifiers(),
1090                          /*IsInitializer=*/false);
1091   }
1092 }
1093 
1094 bool ReductionCodeGen::needCleanups(unsigned N) {
1095   const auto *PrivateVD =
1096       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1097   QualType PrivateType = PrivateVD->getType();
1098   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1099   return DTorKind != QualType::DK_none;
1100 }
1101 
1102 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N,
1103                                     Address PrivateAddr) {
1104   const auto *PrivateVD =
1105       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1106   QualType PrivateType = PrivateVD->getType();
1107   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1108   if (needCleanups(N)) {
1109     PrivateAddr = CGF.Builder.CreateElementBitCast(
1110         PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1111     CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType);
1112   }
1113 }
1114 
1115 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1116                           LValue BaseLV) {
1117   BaseTy = BaseTy.getNonReferenceType();
1118   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1119          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1120     if (const auto *PtrTy = BaseTy->getAs<PointerType>()) {
1121       BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy);
1122     } else {
1123       LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy);
1124       BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal);
1125     }
1126     BaseTy = BaseTy->getPointeeType();
1127   }
1128   return CGF.MakeAddrLValue(
1129       CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF),
1130                                        CGF.ConvertTypeForMem(ElTy)),
1131       BaseLV.getType(), BaseLV.getBaseInfo(),
1132       CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType()));
1133 }
1134 
1135 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1136                           llvm::Type *BaseLVType, CharUnits BaseLVAlignment,
1137                           llvm::Value *Addr) {
1138   Address Tmp = Address::invalid();
1139   Address TopTmp = Address::invalid();
1140   Address MostTopTmp = Address::invalid();
1141   BaseTy = BaseTy.getNonReferenceType();
1142   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1143          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1144     Tmp = CGF.CreateMemTemp(BaseTy);
1145     if (TopTmp.isValid())
1146       CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp);
1147     else
1148       MostTopTmp = Tmp;
1149     TopTmp = Tmp;
1150     BaseTy = BaseTy->getPointeeType();
1151   }
1152   llvm::Type *Ty = BaseLVType;
1153   if (Tmp.isValid())
1154     Ty = Tmp.getElementType();
1155   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty);
1156   if (Tmp.isValid()) {
1157     CGF.Builder.CreateStore(Addr, Tmp);
1158     return MostTopTmp;
1159   }
1160   return Address(Addr, BaseLVAlignment);
1161 }
1162 
1163 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) {
1164   const VarDecl *OrigVD = nullptr;
1165   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) {
1166     const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
1167     while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base))
1168       Base = TempOASE->getBase()->IgnoreParenImpCasts();
1169     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1170       Base = TempASE->getBase()->IgnoreParenImpCasts();
1171     DE = cast<DeclRefExpr>(Base);
1172     OrigVD = cast<VarDecl>(DE->getDecl());
1173   } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) {
1174     const Expr *Base = ASE->getBase()->IgnoreParenImpCasts();
1175     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1176       Base = TempASE->getBase()->IgnoreParenImpCasts();
1177     DE = cast<DeclRefExpr>(Base);
1178     OrigVD = cast<VarDecl>(DE->getDecl());
1179   }
1180   return OrigVD;
1181 }
1182 
1183 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N,
1184                                                Address PrivateAddr) {
1185   const DeclRefExpr *DE;
1186   if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) {
1187     BaseDecls.emplace_back(OrigVD);
1188     LValue OriginalBaseLValue = CGF.EmitLValue(DE);
1189     LValue BaseLValue =
1190         loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(),
1191                     OriginalBaseLValue);
1192     llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff(
1193         BaseLValue.getPointer(CGF), SharedAddresses[N].first.getPointer(CGF));
1194     llvm::Value *PrivatePointer =
1195         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1196             PrivateAddr.getPointer(),
1197             SharedAddresses[N].first.getAddress(CGF).getType());
1198     llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment);
1199     return castToBase(CGF, OrigVD->getType(),
1200                       SharedAddresses[N].first.getType(),
1201                       OriginalBaseLValue.getAddress(CGF).getType(),
1202                       OriginalBaseLValue.getAlignment(), Ptr);
1203   }
1204   BaseDecls.emplace_back(
1205       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl()));
1206   return PrivateAddr;
1207 }
1208 
1209 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const {
1210   const OMPDeclareReductionDecl *DRD =
1211       getReductionInit(ClausesData[N].ReductionOp);
1212   return DRD && DRD->getInitializer();
1213 }
1214 
1215 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
1216   return CGF.EmitLoadOfPointerLValue(
1217       CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1218       getThreadIDVariable()->getType()->castAs<PointerType>());
1219 }
1220 
1221 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) {
1222   if (!CGF.HaveInsertPoint())
1223     return;
1224   // 1.2.2 OpenMP Language Terminology
1225   // Structured block - An executable statement with a single entry at the
1226   // top and a single exit at the bottom.
1227   // The point of exit cannot be a branch out of the structured block.
1228   // longjmp() and throw() must not violate the entry/exit criteria.
1229   CGF.EHStack.pushTerminate();
1230   CodeGen(CGF);
1231   CGF.EHStack.popTerminate();
1232 }
1233 
1234 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
1235     CodeGenFunction &CGF) {
1236   return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1237                             getThreadIDVariable()->getType(),
1238                             AlignmentSource::Decl);
1239 }
1240 
1241 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1242                                        QualType FieldTy) {
1243   auto *Field = FieldDecl::Create(
1244       C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1245       C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1246       /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1247   Field->setAccess(AS_public);
1248   DC->addDecl(Field);
1249   return Field;
1250 }
1251 
1252 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator,
1253                                  StringRef Separator)
1254     : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator),
1255       OffloadEntriesInfoManager(CGM) {
1256   ASTContext &C = CGM.getContext();
1257   RecordDecl *RD = C.buildImplicitRecord("ident_t");
1258   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
1259   RD->startDefinition();
1260   // reserved_1
1261   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1262   // flags
1263   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1264   // reserved_2
1265   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1266   // reserved_3
1267   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1268   // psource
1269   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
1270   RD->completeDefinition();
1271   IdentQTy = C.getRecordType(RD);
1272   IdentTy = CGM.getTypes().ConvertRecordDeclType(RD);
1273   KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
1274 
1275   loadOffloadInfoMetadata();
1276 }
1277 
1278 void CGOpenMPRuntime::clear() {
1279   InternalVars.clear();
1280   // Clean non-target variable declarations possibly used only in debug info.
1281   for (const auto &Data : EmittedNonTargetVariables) {
1282     if (!Data.getValue().pointsToAliveValue())
1283       continue;
1284     auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue());
1285     if (!GV)
1286       continue;
1287     if (!GV->isDeclaration() || GV->getNumUses() > 0)
1288       continue;
1289     GV->eraseFromParent();
1290   }
1291 }
1292 
1293 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const {
1294   SmallString<128> Buffer;
1295   llvm::raw_svector_ostream OS(Buffer);
1296   StringRef Sep = FirstSeparator;
1297   for (StringRef Part : Parts) {
1298     OS << Sep << Part;
1299     Sep = Separator;
1300   }
1301   return std::string(OS.str());
1302 }
1303 
1304 static llvm::Function *
1305 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty,
1306                           const Expr *CombinerInitializer, const VarDecl *In,
1307                           const VarDecl *Out, bool IsCombiner) {
1308   // void .omp_combiner.(Ty *in, Ty *out);
1309   ASTContext &C = CGM.getContext();
1310   QualType PtrTy = C.getPointerType(Ty).withRestrict();
1311   FunctionArgList Args;
1312   ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(),
1313                                /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1314   ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(),
1315                               /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1316   Args.push_back(&OmpOutParm);
1317   Args.push_back(&OmpInParm);
1318   const CGFunctionInfo &FnInfo =
1319       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
1320   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
1321   std::string Name = CGM.getOpenMPRuntime().getName(
1322       {IsCombiner ? "omp_combiner" : "omp_initializer", ""});
1323   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
1324                                     Name, &CGM.getModule());
1325   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
1326   if (CGM.getLangOpts().Optimize) {
1327     Fn->removeFnAttr(llvm::Attribute::NoInline);
1328     Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
1329     Fn->addFnAttr(llvm::Attribute::AlwaysInline);
1330   }
1331   CodeGenFunction CGF(CGM);
1332   // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions.
1333   // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions.
1334   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(),
1335                     Out->getLocation());
1336   CodeGenFunction::OMPPrivateScope Scope(CGF);
1337   Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm);
1338   Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() {
1339     return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>())
1340         .getAddress(CGF);
1341   });
1342   Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm);
1343   Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() {
1344     return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>())
1345         .getAddress(CGF);
1346   });
1347   (void)Scope.Privatize();
1348   if (!IsCombiner && Out->hasInit() &&
1349       !CGF.isTrivialInitializer(Out->getInit())) {
1350     CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out),
1351                          Out->getType().getQualifiers(),
1352                          /*IsInitializer=*/true);
1353   }
1354   if (CombinerInitializer)
1355     CGF.EmitIgnoredExpr(CombinerInitializer);
1356   Scope.ForceCleanup();
1357   CGF.FinishFunction();
1358   return Fn;
1359 }
1360 
1361 void CGOpenMPRuntime::emitUserDefinedReduction(
1362     CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) {
1363   if (UDRMap.count(D) > 0)
1364     return;
1365   llvm::Function *Combiner = emitCombinerOrInitializer(
1366       CGM, D->getType(), D->getCombiner(),
1367       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()),
1368       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()),
1369       /*IsCombiner=*/true);
1370   llvm::Function *Initializer = nullptr;
1371   if (const Expr *Init = D->getInitializer()) {
1372     Initializer = emitCombinerOrInitializer(
1373         CGM, D->getType(),
1374         D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init
1375                                                                      : nullptr,
1376         cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()),
1377         cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()),
1378         /*IsCombiner=*/false);
1379   }
1380   UDRMap.try_emplace(D, Combiner, Initializer);
1381   if (CGF) {
1382     auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn);
1383     Decls.second.push_back(D);
1384   }
1385 }
1386 
1387 std::pair<llvm::Function *, llvm::Function *>
1388 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) {
1389   auto I = UDRMap.find(D);
1390   if (I != UDRMap.end())
1391     return I->second;
1392   emitUserDefinedReduction(/*CGF=*/nullptr, D);
1393   return UDRMap.lookup(D);
1394 }
1395 
1396 namespace {
1397 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR
1398 // Builder if one is present.
1399 struct PushAndPopStackRAII {
1400   PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF,
1401                       bool HasCancel)
1402       : OMPBuilder(OMPBuilder) {
1403     if (!OMPBuilder)
1404       return;
1405 
1406     // The following callback is the crucial part of clangs cleanup process.
1407     //
1408     // NOTE:
1409     // Once the OpenMPIRBuilder is used to create parallel regions (and
1410     // similar), the cancellation destination (Dest below) is determined via
1411     // IP. That means if we have variables to finalize we split the block at IP,
1412     // use the new block (=BB) as destination to build a JumpDest (via
1413     // getJumpDestInCurrentScope(BB)) which then is fed to
1414     // EmitBranchThroughCleanup. Furthermore, there will not be the need
1415     // to push & pop an FinalizationInfo object.
1416     // The FiniCB will still be needed but at the point where the
1417     // OpenMPIRBuilder is asked to construct a parallel (or similar) construct.
1418     auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) {
1419       assert(IP.getBlock()->end() == IP.getPoint() &&
1420              "Clang CG should cause non-terminated block!");
1421       CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1422       CGF.Builder.restoreIP(IP);
1423       CodeGenFunction::JumpDest Dest =
1424           CGF.getOMPCancelDestination(OMPD_parallel);
1425       CGF.EmitBranchThroughCleanup(Dest);
1426     };
1427 
1428     // TODO: Remove this once we emit parallel regions through the
1429     //       OpenMPIRBuilder as it can do this setup internally.
1430     llvm::OpenMPIRBuilder::FinalizationInfo FI(
1431         {FiniCB, OMPD_parallel, HasCancel});
1432     OMPBuilder->pushFinalizationCB(std::move(FI));
1433   }
1434   ~PushAndPopStackRAII() {
1435     if (OMPBuilder)
1436       OMPBuilder->popFinalizationCB();
1437   }
1438   llvm::OpenMPIRBuilder *OMPBuilder;
1439 };
1440 } // namespace
1441 
1442 static llvm::Function *emitParallelOrTeamsOutlinedFunction(
1443     CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS,
1444     const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1445     const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) {
1446   assert(ThreadIDVar->getType()->isPointerType() &&
1447          "thread id variable must be of type kmp_int32 *");
1448   CodeGenFunction CGF(CGM, true);
1449   bool HasCancel = false;
1450   if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D))
1451     HasCancel = OPD->hasCancel();
1452   else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D))
1453     HasCancel = OPSD->hasCancel();
1454   else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D))
1455     HasCancel = OPFD->hasCancel();
1456   else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D))
1457     HasCancel = OPFD->hasCancel();
1458   else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D))
1459     HasCancel = OPFD->hasCancel();
1460   else if (const auto *OPFD =
1461                dyn_cast<OMPTeamsDistributeParallelForDirective>(&D))
1462     HasCancel = OPFD->hasCancel();
1463   else if (const auto *OPFD =
1464                dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D))
1465     HasCancel = OPFD->hasCancel();
1466 
1467   // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new
1468   //       parallel region to make cancellation barriers work properly.
1469   llvm::OpenMPIRBuilder *OMPBuilder = CGM.getOpenMPIRBuilder();
1470   PushAndPopStackRAII PSR(OMPBuilder, CGF, HasCancel);
1471   CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind,
1472                                     HasCancel, OutlinedHelperName);
1473   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1474   return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc());
1475 }
1476 
1477 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction(
1478     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1479     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1480   const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel);
1481   return emitParallelOrTeamsOutlinedFunction(
1482       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1483 }
1484 
1485 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction(
1486     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1487     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1488   const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams);
1489   return emitParallelOrTeamsOutlinedFunction(
1490       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1491 }
1492 
1493 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction(
1494     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1495     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
1496     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
1497     bool Tied, unsigned &NumberOfParts) {
1498   auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF,
1499                                               PrePostActionTy &) {
1500     llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc());
1501     llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc());
1502     llvm::Value *TaskArgs[] = {
1503         UpLoc, ThreadID,
1504         CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar),
1505                                     TaskTVar->getType()->castAs<PointerType>())
1506             .getPointer(CGF)};
1507     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs);
1508   };
1509   CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar,
1510                                                             UntiedCodeGen);
1511   CodeGen.setAction(Action);
1512   assert(!ThreadIDVar->getType()->isPointerType() &&
1513          "thread id variable must be of type kmp_int32 for tasks");
1514   const OpenMPDirectiveKind Region =
1515       isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop
1516                                                       : OMPD_task;
1517   const CapturedStmt *CS = D.getCapturedStmt(Region);
1518   bool HasCancel = false;
1519   if (const auto *TD = dyn_cast<OMPTaskDirective>(&D))
1520     HasCancel = TD->hasCancel();
1521   else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D))
1522     HasCancel = TD->hasCancel();
1523   else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D))
1524     HasCancel = TD->hasCancel();
1525   else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D))
1526     HasCancel = TD->hasCancel();
1527 
1528   CodeGenFunction CGF(CGM, true);
1529   CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen,
1530                                         InnermostKind, HasCancel, Action);
1531   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1532   llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS);
1533   if (!Tied)
1534     NumberOfParts = Action.getNumberOfParts();
1535   return Res;
1536 }
1537 
1538 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM,
1539                              const RecordDecl *RD, const CGRecordLayout &RL,
1540                              ArrayRef<llvm::Constant *> Data) {
1541   llvm::StructType *StructTy = RL.getLLVMType();
1542   unsigned PrevIdx = 0;
1543   ConstantInitBuilder CIBuilder(CGM);
1544   auto DI = Data.begin();
1545   for (const FieldDecl *FD : RD->fields()) {
1546     unsigned Idx = RL.getLLVMFieldNo(FD);
1547     // Fill the alignment.
1548     for (unsigned I = PrevIdx; I < Idx; ++I)
1549       Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I)));
1550     PrevIdx = Idx + 1;
1551     Fields.add(*DI);
1552     ++DI;
1553   }
1554 }
1555 
1556 template <class... As>
1557 static llvm::GlobalVariable *
1558 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant,
1559                    ArrayRef<llvm::Constant *> Data, const Twine &Name,
1560                    As &&... Args) {
1561   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1562   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1563   ConstantInitBuilder CIBuilder(CGM);
1564   ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType());
1565   buildStructValue(Fields, CGM, RD, RL, Data);
1566   return Fields.finishAndCreateGlobal(
1567       Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant,
1568       std::forward<As>(Args)...);
1569 }
1570 
1571 template <typename T>
1572 static void
1573 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty,
1574                                          ArrayRef<llvm::Constant *> Data,
1575                                          T &Parent) {
1576   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1577   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1578   ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType());
1579   buildStructValue(Fields, CGM, RD, RL, Data);
1580   Fields.finishAndAddTo(Parent);
1581 }
1582 
1583 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) {
1584   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1585   unsigned Reserved2Flags = getDefaultLocationReserved2Flags();
1586   FlagsTy FlagsKey(Flags, Reserved2Flags);
1587   llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey);
1588   if (!Entry) {
1589     if (!DefaultOpenMPPSource) {
1590       // Initialize default location for psource field of ident_t structure of
1591       // all ident_t objects. Format is ";file;function;line;column;;".
1592       // Taken from
1593       // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp
1594       DefaultOpenMPPSource =
1595           CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer();
1596       DefaultOpenMPPSource =
1597           llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy);
1598     }
1599 
1600     llvm::Constant *Data[] = {
1601         llvm::ConstantInt::getNullValue(CGM.Int32Ty),
1602         llvm::ConstantInt::get(CGM.Int32Ty, Flags),
1603         llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags),
1604         llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource};
1605     llvm::GlobalValue *DefaultOpenMPLocation =
1606         createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "",
1607                            llvm::GlobalValue::PrivateLinkage);
1608     DefaultOpenMPLocation->setUnnamedAddr(
1609         llvm::GlobalValue::UnnamedAddr::Global);
1610 
1611     OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation;
1612   }
1613   return Address(Entry, Align);
1614 }
1615 
1616 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF,
1617                                              bool AtCurrentPoint) {
1618   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1619   assert(!Elem.second.ServiceInsertPt && "Insert point is set already.");
1620 
1621   llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty);
1622   if (AtCurrentPoint) {
1623     Elem.second.ServiceInsertPt = new llvm::BitCastInst(
1624         Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock());
1625   } else {
1626     Elem.second.ServiceInsertPt =
1627         new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt");
1628     Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt);
1629   }
1630 }
1631 
1632 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) {
1633   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1634   if (Elem.second.ServiceInsertPt) {
1635     llvm::Instruction *Ptr = Elem.second.ServiceInsertPt;
1636     Elem.second.ServiceInsertPt = nullptr;
1637     Ptr->eraseFromParent();
1638   }
1639 }
1640 
1641 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
1642                                                  SourceLocation Loc,
1643                                                  unsigned Flags) {
1644   Flags |= OMP_IDENT_KMPC;
1645   // If no debug info is generated - return global default location.
1646   if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo ||
1647       Loc.isInvalid())
1648     return getOrCreateDefaultLocation(Flags).getPointer();
1649 
1650   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1651 
1652   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1653   Address LocValue = Address::invalid();
1654   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1655   if (I != OpenMPLocThreadIDMap.end())
1656     LocValue = Address(I->second.DebugLoc, Align);
1657 
1658   // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if
1659   // GetOpenMPThreadID was called before this routine.
1660   if (!LocValue.isValid()) {
1661     // Generate "ident_t .kmpc_loc.addr;"
1662     Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr");
1663     auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1664     Elem.second.DebugLoc = AI.getPointer();
1665     LocValue = AI;
1666 
1667     if (!Elem.second.ServiceInsertPt)
1668       setLocThreadIdInsertPt(CGF);
1669     CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1670     CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1671     CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags),
1672                              CGF.getTypeSize(IdentQTy));
1673   }
1674 
1675   // char **psource = &.kmpc_loc_<flags>.addr.psource;
1676   LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy);
1677   auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin();
1678   LValue PSource =
1679       CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource));
1680 
1681   llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding());
1682   if (OMPDebugLoc == nullptr) {
1683     SmallString<128> Buffer2;
1684     llvm::raw_svector_ostream OS2(Buffer2);
1685     // Build debug location
1686     PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1687     OS2 << ";" << PLoc.getFilename() << ";";
1688     if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1689       OS2 << FD->getQualifiedNameAsString();
1690     OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
1691     OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str());
1692     OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc;
1693   }
1694   // *psource = ";<File>;<Function>;<Line>;<Column>;;";
1695   CGF.EmitStoreOfScalar(OMPDebugLoc, PSource);
1696 
1697   // Our callers always pass this to a runtime function, so for
1698   // convenience, go ahead and return a naked pointer.
1699   return LocValue.getPointer();
1700 }
1701 
1702 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
1703                                           SourceLocation Loc) {
1704   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1705 
1706   llvm::Value *ThreadID = nullptr;
1707   // Check whether we've already cached a load of the thread id in this
1708   // function.
1709   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1710   if (I != OpenMPLocThreadIDMap.end()) {
1711     ThreadID = I->second.ThreadID;
1712     if (ThreadID != nullptr)
1713       return ThreadID;
1714   }
1715   // If exceptions are enabled, do not use parameter to avoid possible crash.
1716   if (auto *OMPRegionInfo =
1717           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
1718     if (OMPRegionInfo->getThreadIDVariable()) {
1719       // Check if this an outlined function with thread id passed as argument.
1720       LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
1721       llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent();
1722       if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions ||
1723           !CGF.getLangOpts().CXXExceptions ||
1724           CGF.Builder.GetInsertBlock() == TopBlock ||
1725           !isa<llvm::Instruction>(LVal.getPointer(CGF)) ||
1726           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1727               TopBlock ||
1728           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1729               CGF.Builder.GetInsertBlock()) {
1730         ThreadID = CGF.EmitLoadOfScalar(LVal, Loc);
1731         // If value loaded in entry block, cache it and use it everywhere in
1732         // function.
1733         if (CGF.Builder.GetInsertBlock() == TopBlock) {
1734           auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1735           Elem.second.ThreadID = ThreadID;
1736         }
1737         return ThreadID;
1738       }
1739     }
1740   }
1741 
1742   // This is not an outlined function region - need to call __kmpc_int32
1743   // kmpc_global_thread_num(ident_t *loc).
1744   // Generate thread id value and cache this value for use across the
1745   // function.
1746   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1747   if (!Elem.second.ServiceInsertPt)
1748     setLocThreadIdInsertPt(CGF);
1749   CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1750   CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1751   llvm::CallInst *Call = CGF.Builder.CreateCall(
1752       createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
1753       emitUpdateLocation(CGF, Loc));
1754   Call->setCallingConv(CGF.getRuntimeCC());
1755   Elem.second.ThreadID = Call;
1756   return Call;
1757 }
1758 
1759 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
1760   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1761   if (OpenMPLocThreadIDMap.count(CGF.CurFn)) {
1762     clearLocThreadIdInsertPt(CGF);
1763     OpenMPLocThreadIDMap.erase(CGF.CurFn);
1764   }
1765   if (FunctionUDRMap.count(CGF.CurFn) > 0) {
1766     for(const auto *D : FunctionUDRMap[CGF.CurFn])
1767       UDRMap.erase(D);
1768     FunctionUDRMap.erase(CGF.CurFn);
1769   }
1770   auto I = FunctionUDMMap.find(CGF.CurFn);
1771   if (I != FunctionUDMMap.end()) {
1772     for(const auto *D : I->second)
1773       UDMMap.erase(D);
1774     FunctionUDMMap.erase(I);
1775   }
1776   LastprivateConditionalToTypes.erase(CGF.CurFn);
1777 }
1778 
1779 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
1780   return IdentTy->getPointerTo();
1781 }
1782 
1783 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() {
1784   if (!Kmpc_MicroTy) {
1785     // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...)
1786     llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty),
1787                                  llvm::PointerType::getUnqual(CGM.Int32Ty)};
1788     Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true);
1789   }
1790   return llvm::PointerType::getUnqual(Kmpc_MicroTy);
1791 }
1792 
1793 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) {
1794   llvm::FunctionCallee RTLFn = nullptr;
1795   switch (static_cast<OpenMPRTLFunction>(Function)) {
1796   case OMPRTL__kmpc_fork_call: {
1797     // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro
1798     // microtask, ...);
1799     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1800                                 getKmpc_MicroPointerTy()};
1801     auto *FnTy =
1802         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
1803     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call");
1804     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
1805       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
1806         llvm::LLVMContext &Ctx = F->getContext();
1807         llvm::MDBuilder MDB(Ctx);
1808         // Annotate the callback behavior of the __kmpc_fork_call:
1809         //  - The callback callee is argument number 2 (microtask).
1810         //  - The first two arguments of the callback callee are unknown (-1).
1811         //  - All variadic arguments to the __kmpc_fork_call are passed to the
1812         //    callback callee.
1813         F->addMetadata(
1814             llvm::LLVMContext::MD_callback,
1815             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
1816                                         2, {-1, -1},
1817                                         /* VarArgsArePassed */ true)}));
1818       }
1819     }
1820     break;
1821   }
1822   case OMPRTL__kmpc_global_thread_num: {
1823     // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc);
1824     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1825     auto *FnTy =
1826         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1827     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num");
1828     break;
1829   }
1830   case OMPRTL__kmpc_threadprivate_cached: {
1831     // Build void *__kmpc_threadprivate_cached(ident_t *loc,
1832     // kmp_int32 global_tid, void *data, size_t size, void ***cache);
1833     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1834                                 CGM.VoidPtrTy, CGM.SizeTy,
1835                                 CGM.VoidPtrTy->getPointerTo()->getPointerTo()};
1836     auto *FnTy =
1837         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false);
1838     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached");
1839     break;
1840   }
1841   case OMPRTL__kmpc_critical: {
1842     // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
1843     // kmp_critical_name *crit);
1844     llvm::Type *TypeParams[] = {
1845         getIdentTyPointerTy(), CGM.Int32Ty,
1846         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1847     auto *FnTy =
1848         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1849     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical");
1850     break;
1851   }
1852   case OMPRTL__kmpc_critical_with_hint: {
1853     // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid,
1854     // kmp_critical_name *crit, uintptr_t hint);
1855     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1856                                 llvm::PointerType::getUnqual(KmpCriticalNameTy),
1857                                 CGM.IntPtrTy};
1858     auto *FnTy =
1859         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1860     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint");
1861     break;
1862   }
1863   case OMPRTL__kmpc_threadprivate_register: {
1864     // Build void __kmpc_threadprivate_register(ident_t *, void *data,
1865     // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
1866     // typedef void *(*kmpc_ctor)(void *);
1867     auto *KmpcCtorTy =
1868         llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
1869                                 /*isVarArg*/ false)->getPointerTo();
1870     // typedef void *(*kmpc_cctor)(void *, void *);
1871     llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
1872     auto *KmpcCopyCtorTy =
1873         llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs,
1874                                 /*isVarArg*/ false)
1875             ->getPointerTo();
1876     // typedef void (*kmpc_dtor)(void *);
1877     auto *KmpcDtorTy =
1878         llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false)
1879             ->getPointerTo();
1880     llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy,
1881                               KmpcCopyCtorTy, KmpcDtorTy};
1882     auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs,
1883                                         /*isVarArg*/ false);
1884     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register");
1885     break;
1886   }
1887   case OMPRTL__kmpc_end_critical: {
1888     // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
1889     // kmp_critical_name *crit);
1890     llvm::Type *TypeParams[] = {
1891         getIdentTyPointerTy(), CGM.Int32Ty,
1892         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1893     auto *FnTy =
1894         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1895     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical");
1896     break;
1897   }
1898   case OMPRTL__kmpc_cancel_barrier: {
1899     // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
1900     // global_tid);
1901     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1902     auto *FnTy =
1903         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1904     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier");
1905     break;
1906   }
1907   case OMPRTL__kmpc_barrier: {
1908     // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
1909     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1910     auto *FnTy =
1911         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1912     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier");
1913     break;
1914   }
1915   case OMPRTL__kmpc_for_static_fini: {
1916     // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
1917     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1918     auto *FnTy =
1919         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1920     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini");
1921     break;
1922   }
1923   case OMPRTL__kmpc_push_num_threads: {
1924     // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
1925     // kmp_int32 num_threads)
1926     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1927                                 CGM.Int32Ty};
1928     auto *FnTy =
1929         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1930     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads");
1931     break;
1932   }
1933   case OMPRTL__kmpc_serialized_parallel: {
1934     // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
1935     // global_tid);
1936     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1937     auto *FnTy =
1938         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1939     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel");
1940     break;
1941   }
1942   case OMPRTL__kmpc_end_serialized_parallel: {
1943     // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
1944     // global_tid);
1945     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1946     auto *FnTy =
1947         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1948     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel");
1949     break;
1950   }
1951   case OMPRTL__kmpc_flush: {
1952     // Build void __kmpc_flush(ident_t *loc);
1953     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1954     auto *FnTy =
1955         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1956     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush");
1957     break;
1958   }
1959   case OMPRTL__kmpc_master: {
1960     // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid);
1961     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1962     auto *FnTy =
1963         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1964     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master");
1965     break;
1966   }
1967   case OMPRTL__kmpc_end_master: {
1968     // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid);
1969     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1970     auto *FnTy =
1971         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
1972     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master");
1973     break;
1974   }
1975   case OMPRTL__kmpc_omp_taskyield: {
1976     // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
1977     // int end_part);
1978     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
1979     auto *FnTy =
1980         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1981     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield");
1982     break;
1983   }
1984   case OMPRTL__kmpc_single: {
1985     // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid);
1986     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1987     auto *FnTy =
1988         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1989     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single");
1990     break;
1991   }
1992   case OMPRTL__kmpc_end_single: {
1993     // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid);
1994     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1995     auto *FnTy =
1996         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
1997     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single");
1998     break;
1999   }
2000   case OMPRTL__kmpc_omp_task_alloc: {
2001     // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
2002     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
2003     // kmp_routine_entry_t *task_entry);
2004     assert(KmpRoutineEntryPtrTy != nullptr &&
2005            "Type kmp_routine_entry_t must be created.");
2006     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2007                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy};
2008     // Return void * and then cast to particular kmp_task_t type.
2009     auto *FnTy =
2010         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2011     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc");
2012     break;
2013   }
2014   case OMPRTL__kmpc_omp_target_task_alloc: {
2015     // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid,
2016     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
2017     // kmp_routine_entry_t *task_entry, kmp_int64 device_id);
2018     assert(KmpRoutineEntryPtrTy != nullptr &&
2019            "Type kmp_routine_entry_t must be created.");
2020     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2021                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy,
2022                                 CGM.Int64Ty};
2023     // Return void * and then cast to particular kmp_task_t type.
2024     auto *FnTy =
2025         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2026     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc");
2027     break;
2028   }
2029   case OMPRTL__kmpc_omp_task: {
2030     // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2031     // *new_task);
2032     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2033                                 CGM.VoidPtrTy};
2034     auto *FnTy =
2035         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2036     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task");
2037     break;
2038   }
2039   case OMPRTL__kmpc_copyprivate: {
2040     // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
2041     // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
2042     // kmp_int32 didit);
2043     llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2044     auto *CpyFnTy =
2045         llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false);
2046     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy,
2047                                 CGM.VoidPtrTy, CpyFnTy->getPointerTo(),
2048                                 CGM.Int32Ty};
2049     auto *FnTy =
2050         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2051     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate");
2052     break;
2053   }
2054   case OMPRTL__kmpc_reduce: {
2055     // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
2056     // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
2057     // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
2058     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2059     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2060                                                /*isVarArg=*/false);
2061     llvm::Type *TypeParams[] = {
2062         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2063         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2064         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2065     auto *FnTy =
2066         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2067     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce");
2068     break;
2069   }
2070   case OMPRTL__kmpc_reduce_nowait: {
2071     // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
2072     // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
2073     // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
2074     // *lck);
2075     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2076     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2077                                                /*isVarArg=*/false);
2078     llvm::Type *TypeParams[] = {
2079         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2080         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2081         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2082     auto *FnTy =
2083         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2084     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait");
2085     break;
2086   }
2087   case OMPRTL__kmpc_end_reduce: {
2088     // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
2089     // kmp_critical_name *lck);
2090     llvm::Type *TypeParams[] = {
2091         getIdentTyPointerTy(), CGM.Int32Ty,
2092         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2093     auto *FnTy =
2094         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2095     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce");
2096     break;
2097   }
2098   case OMPRTL__kmpc_end_reduce_nowait: {
2099     // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
2100     // kmp_critical_name *lck);
2101     llvm::Type *TypeParams[] = {
2102         getIdentTyPointerTy(), CGM.Int32Ty,
2103         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2104     auto *FnTy =
2105         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2106     RTLFn =
2107         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait");
2108     break;
2109   }
2110   case OMPRTL__kmpc_omp_task_begin_if0: {
2111     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2112     // *new_task);
2113     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2114                                 CGM.VoidPtrTy};
2115     auto *FnTy =
2116         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2117     RTLFn =
2118         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0");
2119     break;
2120   }
2121   case OMPRTL__kmpc_omp_task_complete_if0: {
2122     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2123     // *new_task);
2124     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2125                                 CGM.VoidPtrTy};
2126     auto *FnTy =
2127         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2128     RTLFn = CGM.CreateRuntimeFunction(FnTy,
2129                                       /*Name=*/"__kmpc_omp_task_complete_if0");
2130     break;
2131   }
2132   case OMPRTL__kmpc_ordered: {
2133     // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
2134     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2135     auto *FnTy =
2136         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2137     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered");
2138     break;
2139   }
2140   case OMPRTL__kmpc_end_ordered: {
2141     // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
2142     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2143     auto *FnTy =
2144         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2145     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered");
2146     break;
2147   }
2148   case OMPRTL__kmpc_omp_taskwait: {
2149     // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid);
2150     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2151     auto *FnTy =
2152         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2153     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait");
2154     break;
2155   }
2156   case OMPRTL__kmpc_taskgroup: {
2157     // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
2158     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2159     auto *FnTy =
2160         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2161     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup");
2162     break;
2163   }
2164   case OMPRTL__kmpc_end_taskgroup: {
2165     // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
2166     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2167     auto *FnTy =
2168         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2169     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup");
2170     break;
2171   }
2172   case OMPRTL__kmpc_push_proc_bind: {
2173     // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
2174     // int proc_bind)
2175     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2176     auto *FnTy =
2177         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2178     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind");
2179     break;
2180   }
2181   case OMPRTL__kmpc_omp_task_with_deps: {
2182     // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
2183     // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
2184     // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
2185     llvm::Type *TypeParams[] = {
2186         getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty,
2187         CGM.VoidPtrTy,         CGM.Int32Ty, CGM.VoidPtrTy};
2188     auto *FnTy =
2189         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2190     RTLFn =
2191         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps");
2192     break;
2193   }
2194   case OMPRTL__kmpc_omp_wait_deps: {
2195     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
2196     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias,
2197     // kmp_depend_info_t *noalias_dep_list);
2198     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2199                                 CGM.Int32Ty,           CGM.VoidPtrTy,
2200                                 CGM.Int32Ty,           CGM.VoidPtrTy};
2201     auto *FnTy =
2202         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2203     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps");
2204     break;
2205   }
2206   case OMPRTL__kmpc_cancellationpoint: {
2207     // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
2208     // global_tid, kmp_int32 cncl_kind)
2209     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2210     auto *FnTy =
2211         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2212     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint");
2213     break;
2214   }
2215   case OMPRTL__kmpc_cancel: {
2216     // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
2217     // kmp_int32 cncl_kind)
2218     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2219     auto *FnTy =
2220         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2221     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel");
2222     break;
2223   }
2224   case OMPRTL__kmpc_push_num_teams: {
2225     // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid,
2226     // kmp_int32 num_teams, kmp_int32 num_threads)
2227     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2228         CGM.Int32Ty};
2229     auto *FnTy =
2230         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2231     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams");
2232     break;
2233   }
2234   case OMPRTL__kmpc_fork_teams: {
2235     // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
2236     // microtask, ...);
2237     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2238                                 getKmpc_MicroPointerTy()};
2239     auto *FnTy =
2240         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
2241     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams");
2242     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
2243       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
2244         llvm::LLVMContext &Ctx = F->getContext();
2245         llvm::MDBuilder MDB(Ctx);
2246         // Annotate the callback behavior of the __kmpc_fork_teams:
2247         //  - The callback callee is argument number 2 (microtask).
2248         //  - The first two arguments of the callback callee are unknown (-1).
2249         //  - All variadic arguments to the __kmpc_fork_teams are passed to the
2250         //    callback callee.
2251         F->addMetadata(
2252             llvm::LLVMContext::MD_callback,
2253             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
2254                                         2, {-1, -1},
2255                                         /* VarArgsArePassed */ true)}));
2256       }
2257     }
2258     break;
2259   }
2260   case OMPRTL__kmpc_taskloop: {
2261     // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
2262     // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
2263     // sched, kmp_uint64 grainsize, void *task_dup);
2264     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2265                                 CGM.IntTy,
2266                                 CGM.VoidPtrTy,
2267                                 CGM.IntTy,
2268                                 CGM.Int64Ty->getPointerTo(),
2269                                 CGM.Int64Ty->getPointerTo(),
2270                                 CGM.Int64Ty,
2271                                 CGM.IntTy,
2272                                 CGM.IntTy,
2273                                 CGM.Int64Ty,
2274                                 CGM.VoidPtrTy};
2275     auto *FnTy =
2276         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2277     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop");
2278     break;
2279   }
2280   case OMPRTL__kmpc_doacross_init: {
2281     // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
2282     // num_dims, struct kmp_dim *dims);
2283     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2284                                 CGM.Int32Ty,
2285                                 CGM.Int32Ty,
2286                                 CGM.VoidPtrTy};
2287     auto *FnTy =
2288         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2289     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init");
2290     break;
2291   }
2292   case OMPRTL__kmpc_doacross_fini: {
2293     // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
2294     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2295     auto *FnTy =
2296         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2297     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini");
2298     break;
2299   }
2300   case OMPRTL__kmpc_doacross_post: {
2301     // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
2302     // *vec);
2303     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2304                                 CGM.Int64Ty->getPointerTo()};
2305     auto *FnTy =
2306         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2307     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post");
2308     break;
2309   }
2310   case OMPRTL__kmpc_doacross_wait: {
2311     // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
2312     // *vec);
2313     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2314                                 CGM.Int64Ty->getPointerTo()};
2315     auto *FnTy =
2316         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2317     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait");
2318     break;
2319   }
2320   case OMPRTL__kmpc_task_reduction_init: {
2321     // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void
2322     // *data);
2323     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy};
2324     auto *FnTy =
2325         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2326     RTLFn =
2327         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init");
2328     break;
2329   }
2330   case OMPRTL__kmpc_task_reduction_get_th_data: {
2331     // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
2332     // *d);
2333     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2334     auto *FnTy =
2335         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2336     RTLFn = CGM.CreateRuntimeFunction(
2337         FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data");
2338     break;
2339   }
2340   case OMPRTL__kmpc_alloc: {
2341     // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t
2342     // al); omp_allocator_handle_t type is void *.
2343     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy};
2344     auto *FnTy =
2345         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2346     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc");
2347     break;
2348   }
2349   case OMPRTL__kmpc_free: {
2350     // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t
2351     // al); omp_allocator_handle_t type is void *.
2352     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2353     auto *FnTy =
2354         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2355     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free");
2356     break;
2357   }
2358   case OMPRTL__kmpc_push_target_tripcount: {
2359     // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
2360     // size);
2361     llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty};
2362     llvm::FunctionType *FnTy =
2363         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2364     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount");
2365     break;
2366   }
2367   case OMPRTL__tgt_target: {
2368     // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
2369     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2370     // *arg_types);
2371     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2372                                 CGM.VoidPtrTy,
2373                                 CGM.Int32Ty,
2374                                 CGM.VoidPtrPtrTy,
2375                                 CGM.VoidPtrPtrTy,
2376                                 CGM.Int64Ty->getPointerTo(),
2377                                 CGM.Int64Ty->getPointerTo()};
2378     auto *FnTy =
2379         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2380     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target");
2381     break;
2382   }
2383   case OMPRTL__tgt_target_nowait: {
2384     // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
2385     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2386     // int64_t *arg_types);
2387     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2388                                 CGM.VoidPtrTy,
2389                                 CGM.Int32Ty,
2390                                 CGM.VoidPtrPtrTy,
2391                                 CGM.VoidPtrPtrTy,
2392                                 CGM.Int64Ty->getPointerTo(),
2393                                 CGM.Int64Ty->getPointerTo()};
2394     auto *FnTy =
2395         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2396     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait");
2397     break;
2398   }
2399   case OMPRTL__tgt_target_teams: {
2400     // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
2401     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2402     // int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2403     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2404                                 CGM.VoidPtrTy,
2405                                 CGM.Int32Ty,
2406                                 CGM.VoidPtrPtrTy,
2407                                 CGM.VoidPtrPtrTy,
2408                                 CGM.Int64Ty->getPointerTo(),
2409                                 CGM.Int64Ty->getPointerTo(),
2410                                 CGM.Int32Ty,
2411                                 CGM.Int32Ty};
2412     auto *FnTy =
2413         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2414     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams");
2415     break;
2416   }
2417   case OMPRTL__tgt_target_teams_nowait: {
2418     // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void
2419     // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
2420     // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2421     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2422                                 CGM.VoidPtrTy,
2423                                 CGM.Int32Ty,
2424                                 CGM.VoidPtrPtrTy,
2425                                 CGM.VoidPtrPtrTy,
2426                                 CGM.Int64Ty->getPointerTo(),
2427                                 CGM.Int64Ty->getPointerTo(),
2428                                 CGM.Int32Ty,
2429                                 CGM.Int32Ty};
2430     auto *FnTy =
2431         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2432     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait");
2433     break;
2434   }
2435   case OMPRTL__tgt_register_requires: {
2436     // Build void __tgt_register_requires(int64_t flags);
2437     llvm::Type *TypeParams[] = {CGM.Int64Ty};
2438     auto *FnTy =
2439         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2440     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires");
2441     break;
2442   }
2443   case OMPRTL__tgt_target_data_begin: {
2444     // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
2445     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2446     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2447                                 CGM.Int32Ty,
2448                                 CGM.VoidPtrPtrTy,
2449                                 CGM.VoidPtrPtrTy,
2450                                 CGM.Int64Ty->getPointerTo(),
2451                                 CGM.Int64Ty->getPointerTo()};
2452     auto *FnTy =
2453         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2454     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin");
2455     break;
2456   }
2457   case OMPRTL__tgt_target_data_begin_nowait: {
2458     // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
2459     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2460     // *arg_types);
2461     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2462                                 CGM.Int32Ty,
2463                                 CGM.VoidPtrPtrTy,
2464                                 CGM.VoidPtrPtrTy,
2465                                 CGM.Int64Ty->getPointerTo(),
2466                                 CGM.Int64Ty->getPointerTo()};
2467     auto *FnTy =
2468         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2469     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait");
2470     break;
2471   }
2472   case OMPRTL__tgt_target_data_end: {
2473     // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
2474     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2475     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2476                                 CGM.Int32Ty,
2477                                 CGM.VoidPtrPtrTy,
2478                                 CGM.VoidPtrPtrTy,
2479                                 CGM.Int64Ty->getPointerTo(),
2480                                 CGM.Int64Ty->getPointerTo()};
2481     auto *FnTy =
2482         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2483     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end");
2484     break;
2485   }
2486   case OMPRTL__tgt_target_data_end_nowait: {
2487     // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t
2488     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2489     // *arg_types);
2490     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2491                                 CGM.Int32Ty,
2492                                 CGM.VoidPtrPtrTy,
2493                                 CGM.VoidPtrPtrTy,
2494                                 CGM.Int64Ty->getPointerTo(),
2495                                 CGM.Int64Ty->getPointerTo()};
2496     auto *FnTy =
2497         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2498     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait");
2499     break;
2500   }
2501   case OMPRTL__tgt_target_data_update: {
2502     // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
2503     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2504     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2505                                 CGM.Int32Ty,
2506                                 CGM.VoidPtrPtrTy,
2507                                 CGM.VoidPtrPtrTy,
2508                                 CGM.Int64Ty->getPointerTo(),
2509                                 CGM.Int64Ty->getPointerTo()};
2510     auto *FnTy =
2511         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2512     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update");
2513     break;
2514   }
2515   case OMPRTL__tgt_target_data_update_nowait: {
2516     // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t
2517     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2518     // *arg_types);
2519     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2520                                 CGM.Int32Ty,
2521                                 CGM.VoidPtrPtrTy,
2522                                 CGM.VoidPtrPtrTy,
2523                                 CGM.Int64Ty->getPointerTo(),
2524                                 CGM.Int64Ty->getPointerTo()};
2525     auto *FnTy =
2526         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2527     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait");
2528     break;
2529   }
2530   case OMPRTL__tgt_mapper_num_components: {
2531     // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
2532     llvm::Type *TypeParams[] = {CGM.VoidPtrTy};
2533     auto *FnTy =
2534         llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false);
2535     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components");
2536     break;
2537   }
2538   case OMPRTL__tgt_push_mapper_component: {
2539     // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void
2540     // *base, void *begin, int64_t size, int64_t type);
2541     llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy,
2542                                 CGM.Int64Ty, CGM.Int64Ty};
2543     auto *FnTy =
2544         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2545     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component");
2546     break;
2547   }
2548   case OMPRTL__kmpc_task_allow_completion_event: {
2549     // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
2550     // int gtid, kmp_task_t *task);
2551     auto *FnTy = llvm::FunctionType::get(
2552         CGM.VoidPtrTy, {getIdentTyPointerTy(), CGM.IntTy, CGM.VoidPtrTy},
2553         /*isVarArg=*/false);
2554     RTLFn =
2555         CGM.CreateRuntimeFunction(FnTy, "__kmpc_task_allow_completion_event");
2556     break;
2557   }
2558   }
2559   assert(RTLFn && "Unable to find OpenMP runtime function");
2560   return RTLFn;
2561 }
2562 
2563 llvm::FunctionCallee
2564 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) {
2565   assert((IVSize == 32 || IVSize == 64) &&
2566          "IV size is not compatible with the omp runtime");
2567   StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4"
2568                                             : "__kmpc_for_static_init_4u")
2569                                 : (IVSigned ? "__kmpc_for_static_init_8"
2570                                             : "__kmpc_for_static_init_8u");
2571   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2572   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2573   llvm::Type *TypeParams[] = {
2574     getIdentTyPointerTy(),                     // loc
2575     CGM.Int32Ty,                               // tid
2576     CGM.Int32Ty,                               // schedtype
2577     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2578     PtrTy,                                     // p_lower
2579     PtrTy,                                     // p_upper
2580     PtrTy,                                     // p_stride
2581     ITy,                                       // incr
2582     ITy                                        // chunk
2583   };
2584   auto *FnTy =
2585       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2586   return CGM.CreateRuntimeFunction(FnTy, Name);
2587 }
2588 
2589 llvm::FunctionCallee
2590 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) {
2591   assert((IVSize == 32 || IVSize == 64) &&
2592          "IV size is not compatible with the omp runtime");
2593   StringRef Name =
2594       IVSize == 32
2595           ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u")
2596           : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u");
2597   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2598   llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc
2599                                CGM.Int32Ty,           // tid
2600                                CGM.Int32Ty,           // schedtype
2601                                ITy,                   // lower
2602                                ITy,                   // upper
2603                                ITy,                   // stride
2604                                ITy                    // chunk
2605   };
2606   auto *FnTy =
2607       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2608   return CGM.CreateRuntimeFunction(FnTy, Name);
2609 }
2610 
2611 llvm::FunctionCallee
2612 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) {
2613   assert((IVSize == 32 || IVSize == 64) &&
2614          "IV size is not compatible with the omp runtime");
2615   StringRef Name =
2616       IVSize == 32
2617           ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u")
2618           : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u");
2619   llvm::Type *TypeParams[] = {
2620       getIdentTyPointerTy(), // loc
2621       CGM.Int32Ty,           // tid
2622   };
2623   auto *FnTy =
2624       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2625   return CGM.CreateRuntimeFunction(FnTy, Name);
2626 }
2627 
2628 llvm::FunctionCallee
2629 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) {
2630   assert((IVSize == 32 || IVSize == 64) &&
2631          "IV size is not compatible with the omp runtime");
2632   StringRef Name =
2633       IVSize == 32
2634           ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u")
2635           : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u");
2636   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2637   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2638   llvm::Type *TypeParams[] = {
2639     getIdentTyPointerTy(),                     // loc
2640     CGM.Int32Ty,                               // tid
2641     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2642     PtrTy,                                     // p_lower
2643     PtrTy,                                     // p_upper
2644     PtrTy                                      // p_stride
2645   };
2646   auto *FnTy =
2647       llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2648   return CGM.CreateRuntimeFunction(FnTy, Name);
2649 }
2650 
2651 /// Obtain information that uniquely identifies a target entry. This
2652 /// consists of the file and device IDs as well as line number associated with
2653 /// the relevant entry source location.
2654 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc,
2655                                      unsigned &DeviceID, unsigned &FileID,
2656                                      unsigned &LineNum) {
2657   SourceManager &SM = C.getSourceManager();
2658 
2659   // The loc should be always valid and have a file ID (the user cannot use
2660   // #pragma directives in macros)
2661 
2662   assert(Loc.isValid() && "Source location is expected to be always valid.");
2663 
2664   PresumedLoc PLoc = SM.getPresumedLoc(Loc);
2665   assert(PLoc.isValid() && "Source location is expected to be always valid.");
2666 
2667   llvm::sys::fs::UniqueID ID;
2668   if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID))
2669     SM.getDiagnostics().Report(diag::err_cannot_open_file)
2670         << PLoc.getFilename() << EC.message();
2671 
2672   DeviceID = ID.getDevice();
2673   FileID = ID.getFile();
2674   LineNum = PLoc.getLine();
2675 }
2676 
2677 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) {
2678   if (CGM.getLangOpts().OpenMPSimd)
2679     return Address::invalid();
2680   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2681       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2682   if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link ||
2683               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2684                HasRequiresUnifiedSharedMemory))) {
2685     SmallString<64> PtrName;
2686     {
2687       llvm::raw_svector_ostream OS(PtrName);
2688       OS << CGM.getMangledName(GlobalDecl(VD));
2689       if (!VD->isExternallyVisible()) {
2690         unsigned DeviceID, FileID, Line;
2691         getTargetEntryUniqueInfo(CGM.getContext(),
2692                                  VD->getCanonicalDecl()->getBeginLoc(),
2693                                  DeviceID, FileID, Line);
2694         OS << llvm::format("_%x", FileID);
2695       }
2696       OS << "_decl_tgt_ref_ptr";
2697     }
2698     llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName);
2699     if (!Ptr) {
2700       QualType PtrTy = CGM.getContext().getPointerType(VD->getType());
2701       Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy),
2702                                         PtrName);
2703 
2704       auto *GV = cast<llvm::GlobalVariable>(Ptr);
2705       GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
2706 
2707       if (!CGM.getLangOpts().OpenMPIsDevice)
2708         GV->setInitializer(CGM.GetAddrOfGlobal(VD));
2709       registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr));
2710     }
2711     return Address(Ptr, CGM.getContext().getDeclAlign(VD));
2712   }
2713   return Address::invalid();
2714 }
2715 
2716 llvm::Constant *
2717 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
2718   assert(!CGM.getLangOpts().OpenMPUseTLS ||
2719          !CGM.getContext().getTargetInfo().isTLSSupported());
2720   // Lookup the entry, lazily creating it if necessary.
2721   std::string Suffix = getName({"cache", ""});
2722   return getOrCreateInternalVariable(
2723       CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix));
2724 }
2725 
2726 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
2727                                                 const VarDecl *VD,
2728                                                 Address VDAddr,
2729                                                 SourceLocation Loc) {
2730   if (CGM.getLangOpts().OpenMPUseTLS &&
2731       CGM.getContext().getTargetInfo().isTLSSupported())
2732     return VDAddr;
2733 
2734   llvm::Type *VarTy = VDAddr.getElementType();
2735   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2736                          CGF.Builder.CreatePointerCast(VDAddr.getPointer(),
2737                                                        CGM.Int8PtrTy),
2738                          CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
2739                          getOrCreateThreadPrivateCache(VD)};
2740   return Address(CGF.EmitRuntimeCall(
2741       createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
2742                  VDAddr.getAlignment());
2743 }
2744 
2745 void CGOpenMPRuntime::emitThreadPrivateVarInit(
2746     CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor,
2747     llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
2748   // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
2749   // library.
2750   llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc);
2751   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
2752                       OMPLoc);
2753   // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
2754   // to register constructor/destructor for variable.
2755   llvm::Value *Args[] = {
2756       OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy),
2757       Ctor, CopyCtor, Dtor};
2758   CGF.EmitRuntimeCall(
2759       createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args);
2760 }
2761 
2762 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
2763     const VarDecl *VD, Address VDAddr, SourceLocation Loc,
2764     bool PerformInit, CodeGenFunction *CGF) {
2765   if (CGM.getLangOpts().OpenMPUseTLS &&
2766       CGM.getContext().getTargetInfo().isTLSSupported())
2767     return nullptr;
2768 
2769   VD = VD->getDefinition(CGM.getContext());
2770   if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) {
2771     QualType ASTTy = VD->getType();
2772 
2773     llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
2774     const Expr *Init = VD->getAnyInitializer();
2775     if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2776       // Generate function that re-emits the declaration's initializer into the
2777       // threadprivate copy of the variable VD
2778       CodeGenFunction CtorCGF(CGM);
2779       FunctionArgList Args;
2780       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2781                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2782                             ImplicitParamDecl::Other);
2783       Args.push_back(&Dst);
2784 
2785       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2786           CGM.getContext().VoidPtrTy, Args);
2787       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2788       std::string Name = getName({"__kmpc_global_ctor_", ""});
2789       llvm::Function *Fn =
2790           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2791       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
2792                             Args, Loc, Loc);
2793       llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar(
2794           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2795           CGM.getContext().VoidPtrTy, Dst.getLocation());
2796       Address Arg = Address(ArgVal, VDAddr.getAlignment());
2797       Arg = CtorCGF.Builder.CreateElementBitCast(
2798           Arg, CtorCGF.ConvertTypeForMem(ASTTy));
2799       CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
2800                                /*IsInitializer=*/true);
2801       ArgVal = CtorCGF.EmitLoadOfScalar(
2802           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2803           CGM.getContext().VoidPtrTy, Dst.getLocation());
2804       CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
2805       CtorCGF.FinishFunction();
2806       Ctor = Fn;
2807     }
2808     if (VD->getType().isDestructedType() != QualType::DK_none) {
2809       // Generate function that emits destructor call for the threadprivate copy
2810       // of the variable VD
2811       CodeGenFunction DtorCGF(CGM);
2812       FunctionArgList Args;
2813       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2814                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2815                             ImplicitParamDecl::Other);
2816       Args.push_back(&Dst);
2817 
2818       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2819           CGM.getContext().VoidTy, Args);
2820       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2821       std::string Name = getName({"__kmpc_global_dtor_", ""});
2822       llvm::Function *Fn =
2823           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2824       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2825       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
2826                             Loc, Loc);
2827       // Create a scope with an artificial location for the body of this function.
2828       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
2829       llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar(
2830           DtorCGF.GetAddrOfLocalVar(&Dst),
2831           /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation());
2832       DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy,
2833                           DtorCGF.getDestroyer(ASTTy.isDestructedType()),
2834                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
2835       DtorCGF.FinishFunction();
2836       Dtor = Fn;
2837     }
2838     // Do not emit init function if it is not required.
2839     if (!Ctor && !Dtor)
2840       return nullptr;
2841 
2842     llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2843     auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs,
2844                                                /*isVarArg=*/false)
2845                            ->getPointerTo();
2846     // Copying constructor for the threadprivate variable.
2847     // Must be NULL - reserved by runtime, but currently it requires that this
2848     // parameter is always NULL. Otherwise it fires assertion.
2849     CopyCtor = llvm::Constant::getNullValue(CopyCtorTy);
2850     if (Ctor == nullptr) {
2851       auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
2852                                              /*isVarArg=*/false)
2853                          ->getPointerTo();
2854       Ctor = llvm::Constant::getNullValue(CtorTy);
2855     }
2856     if (Dtor == nullptr) {
2857       auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy,
2858                                              /*isVarArg=*/false)
2859                          ->getPointerTo();
2860       Dtor = llvm::Constant::getNullValue(DtorTy);
2861     }
2862     if (!CGF) {
2863       auto *InitFunctionTy =
2864           llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
2865       std::string Name = getName({"__omp_threadprivate_init_", ""});
2866       llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction(
2867           InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction());
2868       CodeGenFunction InitCGF(CGM);
2869       FunctionArgList ArgList;
2870       InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
2871                             CGM.getTypes().arrangeNullaryFunction(), ArgList,
2872                             Loc, Loc);
2873       emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2874       InitCGF.FinishFunction();
2875       return InitFunction;
2876     }
2877     emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2878   }
2879   return nullptr;
2880 }
2881 
2882 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD,
2883                                                      llvm::GlobalVariable *Addr,
2884                                                      bool PerformInit) {
2885   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
2886       !CGM.getLangOpts().OpenMPIsDevice)
2887     return false;
2888   Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2889       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2890   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
2891       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2892        HasRequiresUnifiedSharedMemory))
2893     return CGM.getLangOpts().OpenMPIsDevice;
2894   VD = VD->getDefinition(CGM.getContext());
2895   assert(VD && "Unknown VarDecl");
2896 
2897   if (!DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second)
2898     return CGM.getLangOpts().OpenMPIsDevice;
2899 
2900   QualType ASTTy = VD->getType();
2901   SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc();
2902 
2903   // Produce the unique prefix to identify the new target regions. We use
2904   // the source location of the variable declaration which we know to not
2905   // conflict with any target region.
2906   unsigned DeviceID;
2907   unsigned FileID;
2908   unsigned Line;
2909   getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line);
2910   SmallString<128> Buffer, Out;
2911   {
2912     llvm::raw_svector_ostream OS(Buffer);
2913     OS << "__omp_offloading_" << llvm::format("_%x", DeviceID)
2914        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
2915   }
2916 
2917   const Expr *Init = VD->getAnyInitializer();
2918   if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2919     llvm::Constant *Ctor;
2920     llvm::Constant *ID;
2921     if (CGM.getLangOpts().OpenMPIsDevice) {
2922       // Generate function that re-emits the declaration's initializer into
2923       // the threadprivate copy of the variable VD
2924       CodeGenFunction CtorCGF(CGM);
2925 
2926       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2927       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2928       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2929           FTy, Twine(Buffer, "_ctor"), FI, Loc);
2930       auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF);
2931       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2932                             FunctionArgList(), Loc, Loc);
2933       auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF);
2934       CtorCGF.EmitAnyExprToMem(Init,
2935                                Address(Addr, CGM.getContext().getDeclAlign(VD)),
2936                                Init->getType().getQualifiers(),
2937                                /*IsInitializer=*/true);
2938       CtorCGF.FinishFunction();
2939       Ctor = Fn;
2940       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
2941       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor));
2942     } else {
2943       Ctor = new llvm::GlobalVariable(
2944           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
2945           llvm::GlobalValue::PrivateLinkage,
2946           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor"));
2947       ID = Ctor;
2948     }
2949 
2950     // Register the information for the entry associated with the constructor.
2951     Out.clear();
2952     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2953         DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor,
2954         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor);
2955   }
2956   if (VD->getType().isDestructedType() != QualType::DK_none) {
2957     llvm::Constant *Dtor;
2958     llvm::Constant *ID;
2959     if (CGM.getLangOpts().OpenMPIsDevice) {
2960       // Generate function that emits destructor call for the threadprivate
2961       // copy of the variable VD
2962       CodeGenFunction DtorCGF(CGM);
2963 
2964       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2965       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2966       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2967           FTy, Twine(Buffer, "_dtor"), FI, Loc);
2968       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2969       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2970                             FunctionArgList(), Loc, Loc);
2971       // Create a scope with an artificial location for the body of this
2972       // function.
2973       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
2974       DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)),
2975                           ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()),
2976                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
2977       DtorCGF.FinishFunction();
2978       Dtor = Fn;
2979       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
2980       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor));
2981     } else {
2982       Dtor = new llvm::GlobalVariable(
2983           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
2984           llvm::GlobalValue::PrivateLinkage,
2985           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor"));
2986       ID = Dtor;
2987     }
2988     // Register the information for the entry associated with the destructor.
2989     Out.clear();
2990     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2991         DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor,
2992         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor);
2993   }
2994   return CGM.getLangOpts().OpenMPIsDevice;
2995 }
2996 
2997 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF,
2998                                                           QualType VarType,
2999                                                           StringRef Name) {
3000   std::string Suffix = getName({"artificial", ""});
3001   llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType);
3002   llvm::Value *GAddr =
3003       getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix));
3004   if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS &&
3005       CGM.getTarget().isTLSSupported()) {
3006     cast<llvm::GlobalVariable>(GAddr)->setThreadLocal(/*Val=*/true);
3007     return Address(GAddr, CGM.getContext().getTypeAlignInChars(VarType));
3008   }
3009   std::string CacheSuffix = getName({"cache", ""});
3010   llvm::Value *Args[] = {
3011       emitUpdateLocation(CGF, SourceLocation()),
3012       getThreadID(CGF, SourceLocation()),
3013       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy),
3014       CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy,
3015                                 /*isSigned=*/false),
3016       getOrCreateInternalVariable(
3017           CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))};
3018   return Address(
3019       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3020           CGF.EmitRuntimeCall(
3021               createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
3022           VarLVType->getPointerTo(/*AddrSpace=*/0)),
3023       CGM.getContext().getTypeAlignInChars(VarType));
3024 }
3025 
3026 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond,
3027                                    const RegionCodeGenTy &ThenGen,
3028                                    const RegionCodeGenTy &ElseGen) {
3029   CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
3030 
3031   // If the condition constant folds and can be elided, try to avoid emitting
3032   // the condition and the dead arm of the if/else.
3033   bool CondConstant;
3034   if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
3035     if (CondConstant)
3036       ThenGen(CGF);
3037     else
3038       ElseGen(CGF);
3039     return;
3040   }
3041 
3042   // Otherwise, the condition did not fold, or we couldn't elide it.  Just
3043   // emit the conditional branch.
3044   llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then");
3045   llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else");
3046   llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end");
3047   CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0);
3048 
3049   // Emit the 'then' code.
3050   CGF.EmitBlock(ThenBlock);
3051   ThenGen(CGF);
3052   CGF.EmitBranch(ContBlock);
3053   // Emit the 'else' code if present.
3054   // There is no need to emit line number for unconditional branch.
3055   (void)ApplyDebugLocation::CreateEmpty(CGF);
3056   CGF.EmitBlock(ElseBlock);
3057   ElseGen(CGF);
3058   // There is no need to emit line number for unconditional branch.
3059   (void)ApplyDebugLocation::CreateEmpty(CGF);
3060   CGF.EmitBranch(ContBlock);
3061   // Emit the continuation block for code after the if.
3062   CGF.EmitBlock(ContBlock, /*IsFinished=*/true);
3063 }
3064 
3065 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc,
3066                                        llvm::Function *OutlinedFn,
3067                                        ArrayRef<llvm::Value *> CapturedVars,
3068                                        const Expr *IfCond) {
3069   if (!CGF.HaveInsertPoint())
3070     return;
3071   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
3072   auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF,
3073                                                      PrePostActionTy &) {
3074     // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn);
3075     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3076     llvm::Value *Args[] = {
3077         RTLoc,
3078         CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
3079         CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())};
3080     llvm::SmallVector<llvm::Value *, 16> RealArgs;
3081     RealArgs.append(std::begin(Args), std::end(Args));
3082     RealArgs.append(CapturedVars.begin(), CapturedVars.end());
3083 
3084     llvm::FunctionCallee RTLFn =
3085         RT.createRuntimeFunction(OMPRTL__kmpc_fork_call);
3086     CGF.EmitRuntimeCall(RTLFn, RealArgs);
3087   };
3088   auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF,
3089                                                           PrePostActionTy &) {
3090     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3091     llvm::Value *ThreadID = RT.getThreadID(CGF, Loc);
3092     // Build calls:
3093     // __kmpc_serialized_parallel(&Loc, GTid);
3094     llvm::Value *Args[] = {RTLoc, ThreadID};
3095     CGF.EmitRuntimeCall(
3096         RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args);
3097 
3098     // OutlinedFn(&GTid, &zero_bound, CapturedStruct);
3099     Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc);
3100     Address ZeroAddrBound =
3101         CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty,
3102                                          /*Name=*/".bound.zero.addr");
3103     CGF.InitTempAlloca(ZeroAddrBound, CGF.Builder.getInt32(/*C*/ 0));
3104     llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs;
3105     // ThreadId for serialized parallels is 0.
3106     OutlinedFnArgs.push_back(ThreadIDAddr.getPointer());
3107     OutlinedFnArgs.push_back(ZeroAddrBound.getPointer());
3108     OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end());
3109     RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs);
3110 
3111     // __kmpc_end_serialized_parallel(&Loc, GTid);
3112     llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID};
3113     CGF.EmitRuntimeCall(
3114         RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel),
3115         EndArgs);
3116   };
3117   if (IfCond) {
3118     emitIfClause(CGF, IfCond, ThenGen, ElseGen);
3119   } else {
3120     RegionCodeGenTy ThenRCG(ThenGen);
3121     ThenRCG(CGF);
3122   }
3123 }
3124 
3125 // If we're inside an (outlined) parallel region, use the region info's
3126 // thread-ID variable (it is passed in a first argument of the outlined function
3127 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
3128 // regular serial code region, get thread ID by calling kmp_int32
3129 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
3130 // return the address of that temp.
3131 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
3132                                              SourceLocation Loc) {
3133   if (auto *OMPRegionInfo =
3134           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3135     if (OMPRegionInfo->getThreadIDVariable())
3136       return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF);
3137 
3138   llvm::Value *ThreadID = getThreadID(CGF, Loc);
3139   QualType Int32Ty =
3140       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
3141   Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp.");
3142   CGF.EmitStoreOfScalar(ThreadID,
3143                         CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty));
3144 
3145   return ThreadIDTemp;
3146 }
3147 
3148 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable(
3149     llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) {
3150   SmallString<256> Buffer;
3151   llvm::raw_svector_ostream Out(Buffer);
3152   Out << Name;
3153   StringRef RuntimeName = Out.str();
3154   auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first;
3155   if (Elem.second) {
3156     assert(Elem.second->getType()->getPointerElementType() == Ty &&
3157            "OMP internal variable has different type than requested");
3158     return &*Elem.second;
3159   }
3160 
3161   return Elem.second = new llvm::GlobalVariable(
3162              CGM.getModule(), Ty, /*IsConstant*/ false,
3163              llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty),
3164              Elem.first(), /*InsertBefore=*/nullptr,
3165              llvm::GlobalValue::NotThreadLocal, AddressSpace);
3166 }
3167 
3168 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
3169   std::string Prefix = Twine("gomp_critical_user_", CriticalName).str();
3170   std::string Name = getName({Prefix, "var"});
3171   return getOrCreateInternalVariable(KmpCriticalNameTy, Name);
3172 }
3173 
3174 namespace {
3175 /// Common pre(post)-action for different OpenMP constructs.
3176 class CommonActionTy final : public PrePostActionTy {
3177   llvm::FunctionCallee EnterCallee;
3178   ArrayRef<llvm::Value *> EnterArgs;
3179   llvm::FunctionCallee ExitCallee;
3180   ArrayRef<llvm::Value *> ExitArgs;
3181   bool Conditional;
3182   llvm::BasicBlock *ContBlock = nullptr;
3183 
3184 public:
3185   CommonActionTy(llvm::FunctionCallee EnterCallee,
3186                  ArrayRef<llvm::Value *> EnterArgs,
3187                  llvm::FunctionCallee ExitCallee,
3188                  ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false)
3189       : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee),
3190         ExitArgs(ExitArgs), Conditional(Conditional) {}
3191   void Enter(CodeGenFunction &CGF) override {
3192     llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs);
3193     if (Conditional) {
3194       llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes);
3195       auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
3196       ContBlock = CGF.createBasicBlock("omp_if.end");
3197       // Generate the branch (If-stmt)
3198       CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
3199       CGF.EmitBlock(ThenBlock);
3200     }
3201   }
3202   void Done(CodeGenFunction &CGF) {
3203     // Emit the rest of blocks/branches
3204     CGF.EmitBranch(ContBlock);
3205     CGF.EmitBlock(ContBlock, true);
3206   }
3207   void Exit(CodeGenFunction &CGF) override {
3208     CGF.EmitRuntimeCall(ExitCallee, ExitArgs);
3209   }
3210 };
3211 } // anonymous namespace
3212 
3213 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF,
3214                                          StringRef CriticalName,
3215                                          const RegionCodeGenTy &CriticalOpGen,
3216                                          SourceLocation Loc, const Expr *Hint) {
3217   // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]);
3218   // CriticalOpGen();
3219   // __kmpc_end_critical(ident_t *, gtid, Lock);
3220   // Prepare arguments and build a call to __kmpc_critical
3221   if (!CGF.HaveInsertPoint())
3222     return;
3223   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3224                          getCriticalRegionLock(CriticalName)};
3225   llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args),
3226                                                 std::end(Args));
3227   if (Hint) {
3228     EnterArgs.push_back(CGF.Builder.CreateIntCast(
3229         CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false));
3230   }
3231   CommonActionTy Action(
3232       createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint
3233                                  : OMPRTL__kmpc_critical),
3234       EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args);
3235   CriticalOpGen.setAction(Action);
3236   emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen);
3237 }
3238 
3239 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
3240                                        const RegionCodeGenTy &MasterOpGen,
3241                                        SourceLocation Loc) {
3242   if (!CGF.HaveInsertPoint())
3243     return;
3244   // if(__kmpc_master(ident_t *, gtid)) {
3245   //   MasterOpGen();
3246   //   __kmpc_end_master(ident_t *, gtid);
3247   // }
3248   // Prepare arguments and build a call to __kmpc_master
3249   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3250   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args,
3251                         createRuntimeFunction(OMPRTL__kmpc_end_master), Args,
3252                         /*Conditional=*/true);
3253   MasterOpGen.setAction(Action);
3254   emitInlinedDirective(CGF, OMPD_master, MasterOpGen);
3255   Action.Done(CGF);
3256 }
3257 
3258 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
3259                                         SourceLocation Loc) {
3260   if (!CGF.HaveInsertPoint())
3261     return;
3262   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3263   if (OMPBuilder) {
3264     OMPBuilder->CreateTaskyield(CGF.Builder);
3265   } else {
3266     // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
3267     llvm::Value *Args[] = {
3268         emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3269         llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
3270     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield),
3271                         Args);
3272   }
3273 
3274   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3275     Region->emitUntiedSwitch(CGF);
3276 }
3277 
3278 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF,
3279                                           const RegionCodeGenTy &TaskgroupOpGen,
3280                                           SourceLocation Loc) {
3281   if (!CGF.HaveInsertPoint())
3282     return;
3283   // __kmpc_taskgroup(ident_t *, gtid);
3284   // TaskgroupOpGen();
3285   // __kmpc_end_taskgroup(ident_t *, gtid);
3286   // Prepare arguments and build a call to __kmpc_taskgroup
3287   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3288   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args,
3289                         createRuntimeFunction(OMPRTL__kmpc_end_taskgroup),
3290                         Args);
3291   TaskgroupOpGen.setAction(Action);
3292   emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen);
3293 }
3294 
3295 /// Given an array of pointers to variables, project the address of a
3296 /// given variable.
3297 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array,
3298                                       unsigned Index, const VarDecl *Var) {
3299   // Pull out the pointer to the variable.
3300   Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index);
3301   llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr);
3302 
3303   Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var));
3304   Addr = CGF.Builder.CreateElementBitCast(
3305       Addr, CGF.ConvertTypeForMem(Var->getType()));
3306   return Addr;
3307 }
3308 
3309 static llvm::Value *emitCopyprivateCopyFunction(
3310     CodeGenModule &CGM, llvm::Type *ArgsType,
3311     ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
3312     ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps,
3313     SourceLocation Loc) {
3314   ASTContext &C = CGM.getContext();
3315   // void copy_func(void *LHSArg, void *RHSArg);
3316   FunctionArgList Args;
3317   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3318                            ImplicitParamDecl::Other);
3319   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3320                            ImplicitParamDecl::Other);
3321   Args.push_back(&LHSArg);
3322   Args.push_back(&RHSArg);
3323   const auto &CGFI =
3324       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
3325   std::string Name =
3326       CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"});
3327   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
3328                                     llvm::GlobalValue::InternalLinkage, Name,
3329                                     &CGM.getModule());
3330   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
3331   Fn->setDoesNotRecurse();
3332   CodeGenFunction CGF(CGM);
3333   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
3334   // Dest = (void*[n])(LHSArg);
3335   // Src = (void*[n])(RHSArg);
3336   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3337       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
3338       ArgsType), CGF.getPointerAlign());
3339   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3340       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
3341       ArgsType), CGF.getPointerAlign());
3342   // *(Type0*)Dst[0] = *(Type0*)Src[0];
3343   // *(Type1*)Dst[1] = *(Type1*)Src[1];
3344   // ...
3345   // *(Typen*)Dst[n] = *(Typen*)Src[n];
3346   for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
3347     const auto *DestVar =
3348         cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl());
3349     Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar);
3350 
3351     const auto *SrcVar =
3352         cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl());
3353     Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar);
3354 
3355     const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl();
3356     QualType Type = VD->getType();
3357     CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]);
3358   }
3359   CGF.FinishFunction();
3360   return Fn;
3361 }
3362 
3363 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
3364                                        const RegionCodeGenTy &SingleOpGen,
3365                                        SourceLocation Loc,
3366                                        ArrayRef<const Expr *> CopyprivateVars,
3367                                        ArrayRef<const Expr *> SrcExprs,
3368                                        ArrayRef<const Expr *> DstExprs,
3369                                        ArrayRef<const Expr *> AssignmentOps) {
3370   if (!CGF.HaveInsertPoint())
3371     return;
3372   assert(CopyprivateVars.size() == SrcExprs.size() &&
3373          CopyprivateVars.size() == DstExprs.size() &&
3374          CopyprivateVars.size() == AssignmentOps.size());
3375   ASTContext &C = CGM.getContext();
3376   // int32 did_it = 0;
3377   // if(__kmpc_single(ident_t *, gtid)) {
3378   //   SingleOpGen();
3379   //   __kmpc_end_single(ident_t *, gtid);
3380   //   did_it = 1;
3381   // }
3382   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3383   // <copy_func>, did_it);
3384 
3385   Address DidIt = Address::invalid();
3386   if (!CopyprivateVars.empty()) {
3387     // int32 did_it = 0;
3388     QualType KmpInt32Ty =
3389         C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
3390     DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it");
3391     CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt);
3392   }
3393   // Prepare arguments and build a call to __kmpc_single
3394   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3395   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args,
3396                         createRuntimeFunction(OMPRTL__kmpc_end_single), Args,
3397                         /*Conditional=*/true);
3398   SingleOpGen.setAction(Action);
3399   emitInlinedDirective(CGF, OMPD_single, SingleOpGen);
3400   if (DidIt.isValid()) {
3401     // did_it = 1;
3402     CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt);
3403   }
3404   Action.Done(CGF);
3405   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3406   // <copy_func>, did_it);
3407   if (DidIt.isValid()) {
3408     llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
3409     QualType CopyprivateArrayTy = C.getConstantArrayType(
3410         C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
3411         /*IndexTypeQuals=*/0);
3412     // Create a list of all private variables for copyprivate.
3413     Address CopyprivateList =
3414         CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list");
3415     for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
3416       Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I);
3417       CGF.Builder.CreateStore(
3418           CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3419               CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF),
3420               CGF.VoidPtrTy),
3421           Elem);
3422     }
3423     // Build function that copies private values from single region to all other
3424     // threads in the corresponding parallel region.
3425     llvm::Value *CpyFn = emitCopyprivateCopyFunction(
3426         CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(),
3427         CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc);
3428     llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy);
3429     Address CL =
3430       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList,
3431                                                       CGF.VoidPtrTy);
3432     llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt);
3433     llvm::Value *Args[] = {
3434         emitUpdateLocation(CGF, Loc), // ident_t *<loc>
3435         getThreadID(CGF, Loc),        // i32 <gtid>
3436         BufSize,                      // size_t <buf_size>
3437         CL.getPointer(),              // void *<copyprivate list>
3438         CpyFn,                        // void (*) (void *, void *) <copy_func>
3439         DidItVal                      // i32 did_it
3440     };
3441     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args);
3442   }
3443 }
3444 
3445 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF,
3446                                         const RegionCodeGenTy &OrderedOpGen,
3447                                         SourceLocation Loc, bool IsThreads) {
3448   if (!CGF.HaveInsertPoint())
3449     return;
3450   // __kmpc_ordered(ident_t *, gtid);
3451   // OrderedOpGen();
3452   // __kmpc_end_ordered(ident_t *, gtid);
3453   // Prepare arguments and build a call to __kmpc_ordered
3454   if (IsThreads) {
3455     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3456     CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args,
3457                           createRuntimeFunction(OMPRTL__kmpc_end_ordered),
3458                           Args);
3459     OrderedOpGen.setAction(Action);
3460     emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3461     return;
3462   }
3463   emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3464 }
3465 
3466 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) {
3467   unsigned Flags;
3468   if (Kind == OMPD_for)
3469     Flags = OMP_IDENT_BARRIER_IMPL_FOR;
3470   else if (Kind == OMPD_sections)
3471     Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS;
3472   else if (Kind == OMPD_single)
3473     Flags = OMP_IDENT_BARRIER_IMPL_SINGLE;
3474   else if (Kind == OMPD_barrier)
3475     Flags = OMP_IDENT_BARRIER_EXPL;
3476   else
3477     Flags = OMP_IDENT_BARRIER_IMPL;
3478   return Flags;
3479 }
3480 
3481 void CGOpenMPRuntime::getDefaultScheduleAndChunk(
3482     CodeGenFunction &CGF, const OMPLoopDirective &S,
3483     OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const {
3484   // Check if the loop directive is actually a doacross loop directive. In this
3485   // case choose static, 1 schedule.
3486   if (llvm::any_of(
3487           S.getClausesOfKind<OMPOrderedClause>(),
3488           [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) {
3489     ScheduleKind = OMPC_SCHEDULE_static;
3490     // Chunk size is 1 in this case.
3491     llvm::APInt ChunkSize(32, 1);
3492     ChunkExpr = IntegerLiteral::Create(
3493         CGF.getContext(), ChunkSize,
3494         CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0),
3495         SourceLocation());
3496   }
3497 }
3498 
3499 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
3500                                       OpenMPDirectiveKind Kind, bool EmitChecks,
3501                                       bool ForceSimpleCall) {
3502   // Check if we should use the OMPBuilder
3503   auto *OMPRegionInfo =
3504       dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo);
3505   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3506   if (OMPBuilder) {
3507     CGF.Builder.restoreIP(OMPBuilder->CreateBarrier(
3508         CGF.Builder, Kind, ForceSimpleCall, EmitChecks));
3509     return;
3510   }
3511 
3512   if (!CGF.HaveInsertPoint())
3513     return;
3514   // Build call __kmpc_cancel_barrier(loc, thread_id);
3515   // Build call __kmpc_barrier(loc, thread_id);
3516   unsigned Flags = getDefaultFlagsForBarriers(Kind);
3517   // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc,
3518   // thread_id);
3519   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
3520                          getThreadID(CGF, Loc)};
3521   if (OMPRegionInfo) {
3522     if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) {
3523       llvm::Value *Result = CGF.EmitRuntimeCall(
3524           createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args);
3525       if (EmitChecks) {
3526         // if (__kmpc_cancel_barrier()) {
3527         //   exit from construct;
3528         // }
3529         llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
3530         llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
3531         llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
3532         CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
3533         CGF.EmitBlock(ExitBB);
3534         //   exit from construct;
3535         CodeGenFunction::JumpDest CancelDestination =
3536             CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
3537         CGF.EmitBranchThroughCleanup(CancelDestination);
3538         CGF.EmitBlock(ContBB, /*IsFinished=*/true);
3539       }
3540       return;
3541     }
3542   }
3543   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args);
3544 }
3545 
3546 /// Map the OpenMP loop schedule to the runtime enumeration.
3547 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
3548                                           bool Chunked, bool Ordered) {
3549   switch (ScheduleKind) {
3550   case OMPC_SCHEDULE_static:
3551     return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked)
3552                    : (Ordered ? OMP_ord_static : OMP_sch_static);
3553   case OMPC_SCHEDULE_dynamic:
3554     return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked;
3555   case OMPC_SCHEDULE_guided:
3556     return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked;
3557   case OMPC_SCHEDULE_runtime:
3558     return Ordered ? OMP_ord_runtime : OMP_sch_runtime;
3559   case OMPC_SCHEDULE_auto:
3560     return Ordered ? OMP_ord_auto : OMP_sch_auto;
3561   case OMPC_SCHEDULE_unknown:
3562     assert(!Chunked && "chunk was specified but schedule kind not known");
3563     return Ordered ? OMP_ord_static : OMP_sch_static;
3564   }
3565   llvm_unreachable("Unexpected runtime schedule");
3566 }
3567 
3568 /// Map the OpenMP distribute schedule to the runtime enumeration.
3569 static OpenMPSchedType
3570 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) {
3571   // only static is allowed for dist_schedule
3572   return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static;
3573 }
3574 
3575 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
3576                                          bool Chunked) const {
3577   OpenMPSchedType Schedule =
3578       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3579   return Schedule == OMP_sch_static;
3580 }
3581 
3582 bool CGOpenMPRuntime::isStaticNonchunked(
3583     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3584   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3585   return Schedule == OMP_dist_sch_static;
3586 }
3587 
3588 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind,
3589                                       bool Chunked) const {
3590   OpenMPSchedType Schedule =
3591       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3592   return Schedule == OMP_sch_static_chunked;
3593 }
3594 
3595 bool CGOpenMPRuntime::isStaticChunked(
3596     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3597   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3598   return Schedule == OMP_dist_sch_static_chunked;
3599 }
3600 
3601 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
3602   OpenMPSchedType Schedule =
3603       getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false);
3604   assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
3605   return Schedule != OMP_sch_static;
3606 }
3607 
3608 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule,
3609                                   OpenMPScheduleClauseModifier M1,
3610                                   OpenMPScheduleClauseModifier M2) {
3611   int Modifier = 0;
3612   switch (M1) {
3613   case OMPC_SCHEDULE_MODIFIER_monotonic:
3614     Modifier = OMP_sch_modifier_monotonic;
3615     break;
3616   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3617     Modifier = OMP_sch_modifier_nonmonotonic;
3618     break;
3619   case OMPC_SCHEDULE_MODIFIER_simd:
3620     if (Schedule == OMP_sch_static_chunked)
3621       Schedule = OMP_sch_static_balanced_chunked;
3622     break;
3623   case OMPC_SCHEDULE_MODIFIER_last:
3624   case OMPC_SCHEDULE_MODIFIER_unknown:
3625     break;
3626   }
3627   switch (M2) {
3628   case OMPC_SCHEDULE_MODIFIER_monotonic:
3629     Modifier = OMP_sch_modifier_monotonic;
3630     break;
3631   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3632     Modifier = OMP_sch_modifier_nonmonotonic;
3633     break;
3634   case OMPC_SCHEDULE_MODIFIER_simd:
3635     if (Schedule == OMP_sch_static_chunked)
3636       Schedule = OMP_sch_static_balanced_chunked;
3637     break;
3638   case OMPC_SCHEDULE_MODIFIER_last:
3639   case OMPC_SCHEDULE_MODIFIER_unknown:
3640     break;
3641   }
3642   // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription.
3643   // If the static schedule kind is specified or if the ordered clause is
3644   // specified, and if the nonmonotonic modifier is not specified, the effect is
3645   // as if the monotonic modifier is specified. Otherwise, unless the monotonic
3646   // modifier is specified, the effect is as if the nonmonotonic modifier is
3647   // specified.
3648   if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) {
3649     if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static ||
3650           Schedule == OMP_sch_static_balanced_chunked ||
3651           Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static ||
3652           Schedule == OMP_dist_sch_static_chunked ||
3653           Schedule == OMP_dist_sch_static))
3654       Modifier = OMP_sch_modifier_nonmonotonic;
3655   }
3656   return Schedule | Modifier;
3657 }
3658 
3659 void CGOpenMPRuntime::emitForDispatchInit(
3660     CodeGenFunction &CGF, SourceLocation Loc,
3661     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
3662     bool Ordered, const DispatchRTInput &DispatchValues) {
3663   if (!CGF.HaveInsertPoint())
3664     return;
3665   OpenMPSchedType Schedule = getRuntimeSchedule(
3666       ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered);
3667   assert(Ordered ||
3668          (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked &&
3669           Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked &&
3670           Schedule != OMP_sch_static_balanced_chunked));
3671   // Call __kmpc_dispatch_init(
3672   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
3673   //          kmp_int[32|64] lower, kmp_int[32|64] upper,
3674   //          kmp_int[32|64] stride, kmp_int[32|64] chunk);
3675 
3676   // If the Chunk was not specified in the clause - use default value 1.
3677   llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk
3678                                             : CGF.Builder.getIntN(IVSize, 1);
3679   llvm::Value *Args[] = {
3680       emitUpdateLocation(CGF, Loc),
3681       getThreadID(CGF, Loc),
3682       CGF.Builder.getInt32(addMonoNonMonoModifier(
3683           CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type
3684       DispatchValues.LB,                                     // Lower
3685       DispatchValues.UB,                                     // Upper
3686       CGF.Builder.getIntN(IVSize, 1),                        // Stride
3687       Chunk                                                  // Chunk
3688   };
3689   CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args);
3690 }
3691 
3692 static void emitForStaticInitCall(
3693     CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId,
3694     llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule,
3695     OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2,
3696     const CGOpenMPRuntime::StaticRTInput &Values) {
3697   if (!CGF.HaveInsertPoint())
3698     return;
3699 
3700   assert(!Values.Ordered);
3701   assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked ||
3702          Schedule == OMP_sch_static_balanced_chunked ||
3703          Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked ||
3704          Schedule == OMP_dist_sch_static ||
3705          Schedule == OMP_dist_sch_static_chunked);
3706 
3707   // Call __kmpc_for_static_init(
3708   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
3709   //          kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
3710   //          kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
3711   //          kmp_int[32|64] incr, kmp_int[32|64] chunk);
3712   llvm::Value *Chunk = Values.Chunk;
3713   if (Chunk == nullptr) {
3714     assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static ||
3715             Schedule == OMP_dist_sch_static) &&
3716            "expected static non-chunked schedule");
3717     // If the Chunk was not specified in the clause - use default value 1.
3718     Chunk = CGF.Builder.getIntN(Values.IVSize, 1);
3719   } else {
3720     assert((Schedule == OMP_sch_static_chunked ||
3721             Schedule == OMP_sch_static_balanced_chunked ||
3722             Schedule == OMP_ord_static_chunked ||
3723             Schedule == OMP_dist_sch_static_chunked) &&
3724            "expected static chunked schedule");
3725   }
3726   llvm::Value *Args[] = {
3727       UpdateLocation,
3728       ThreadId,
3729       CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1,
3730                                                   M2)), // Schedule type
3731       Values.IL.getPointer(),                           // &isLastIter
3732       Values.LB.getPointer(),                           // &LB
3733       Values.UB.getPointer(),                           // &UB
3734       Values.ST.getPointer(),                           // &Stride
3735       CGF.Builder.getIntN(Values.IVSize, 1),            // Incr
3736       Chunk                                             // Chunk
3737   };
3738   CGF.EmitRuntimeCall(ForStaticInitFunction, Args);
3739 }
3740 
3741 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF,
3742                                         SourceLocation Loc,
3743                                         OpenMPDirectiveKind DKind,
3744                                         const OpenMPScheduleTy &ScheduleKind,
3745                                         const StaticRTInput &Values) {
3746   OpenMPSchedType ScheduleNum = getRuntimeSchedule(
3747       ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered);
3748   assert(isOpenMPWorksharingDirective(DKind) &&
3749          "Expected loop-based or sections-based directive.");
3750   llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc,
3751                                              isOpenMPLoopDirective(DKind)
3752                                                  ? OMP_IDENT_WORK_LOOP
3753                                                  : OMP_IDENT_WORK_SECTIONS);
3754   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3755   llvm::FunctionCallee StaticInitFunction =
3756       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3757   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
3758   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3759                         ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values);
3760 }
3761 
3762 void CGOpenMPRuntime::emitDistributeStaticInit(
3763     CodeGenFunction &CGF, SourceLocation Loc,
3764     OpenMPDistScheduleClauseKind SchedKind,
3765     const CGOpenMPRuntime::StaticRTInput &Values) {
3766   OpenMPSchedType ScheduleNum =
3767       getRuntimeSchedule(SchedKind, Values.Chunk != nullptr);
3768   llvm::Value *UpdatedLocation =
3769       emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE);
3770   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3771   llvm::FunctionCallee StaticInitFunction =
3772       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3773   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3774                         ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown,
3775                         OMPC_SCHEDULE_MODIFIER_unknown, Values);
3776 }
3777 
3778 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF,
3779                                           SourceLocation Loc,
3780                                           OpenMPDirectiveKind DKind) {
3781   if (!CGF.HaveInsertPoint())
3782     return;
3783   // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
3784   llvm::Value *Args[] = {
3785       emitUpdateLocation(CGF, Loc,
3786                          isOpenMPDistributeDirective(DKind)
3787                              ? OMP_IDENT_WORK_DISTRIBUTE
3788                              : isOpenMPLoopDirective(DKind)
3789                                    ? OMP_IDENT_WORK_LOOP
3790                                    : OMP_IDENT_WORK_SECTIONS),
3791       getThreadID(CGF, Loc)};
3792   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
3793   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini),
3794                       Args);
3795 }
3796 
3797 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
3798                                                  SourceLocation Loc,
3799                                                  unsigned IVSize,
3800                                                  bool IVSigned) {
3801   if (!CGF.HaveInsertPoint())
3802     return;
3803   // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
3804   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3805   CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args);
3806 }
3807 
3808 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
3809                                           SourceLocation Loc, unsigned IVSize,
3810                                           bool IVSigned, Address IL,
3811                                           Address LB, Address UB,
3812                                           Address ST) {
3813   // Call __kmpc_dispatch_next(
3814   //          ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
3815   //          kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
3816   //          kmp_int[32|64] *p_stride);
3817   llvm::Value *Args[] = {
3818       emitUpdateLocation(CGF, Loc),
3819       getThreadID(CGF, Loc),
3820       IL.getPointer(), // &isLastIter
3821       LB.getPointer(), // &Lower
3822       UB.getPointer(), // &Upper
3823       ST.getPointer()  // &Stride
3824   };
3825   llvm::Value *Call =
3826       CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args);
3827   return CGF.EmitScalarConversion(
3828       Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1),
3829       CGF.getContext().BoolTy, Loc);
3830 }
3831 
3832 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
3833                                            llvm::Value *NumThreads,
3834                                            SourceLocation Loc) {
3835   if (!CGF.HaveInsertPoint())
3836     return;
3837   // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
3838   llvm::Value *Args[] = {
3839       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3840       CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)};
3841   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads),
3842                       Args);
3843 }
3844 
3845 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF,
3846                                          ProcBindKind ProcBind,
3847                                          SourceLocation Loc) {
3848   if (!CGF.HaveInsertPoint())
3849     return;
3850   assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value.");
3851   // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind)
3852   llvm::Value *Args[] = {
3853       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3854       llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)};
3855   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args);
3856 }
3857 
3858 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
3859                                 SourceLocation Loc, llvm::AtomicOrdering AO) {
3860   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3861   if (OMPBuilder) {
3862     OMPBuilder->CreateFlush(CGF.Builder);
3863   } else {
3864     if (!CGF.HaveInsertPoint())
3865       return;
3866     // Build call void __kmpc_flush(ident_t *loc)
3867     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush),
3868                         emitUpdateLocation(CGF, Loc));
3869   }
3870 }
3871 
3872 namespace {
3873 /// Indexes of fields for type kmp_task_t.
3874 enum KmpTaskTFields {
3875   /// List of shared variables.
3876   KmpTaskTShareds,
3877   /// Task routine.
3878   KmpTaskTRoutine,
3879   /// Partition id for the untied tasks.
3880   KmpTaskTPartId,
3881   /// Function with call of destructors for private variables.
3882   Data1,
3883   /// Task priority.
3884   Data2,
3885   /// (Taskloops only) Lower bound.
3886   KmpTaskTLowerBound,
3887   /// (Taskloops only) Upper bound.
3888   KmpTaskTUpperBound,
3889   /// (Taskloops only) Stride.
3890   KmpTaskTStride,
3891   /// (Taskloops only) Is last iteration flag.
3892   KmpTaskTLastIter,
3893   /// (Taskloops only) Reduction data.
3894   KmpTaskTReductions,
3895 };
3896 } // anonymous namespace
3897 
3898 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const {
3899   return OffloadEntriesTargetRegion.empty() &&
3900          OffloadEntriesDeviceGlobalVar.empty();
3901 }
3902 
3903 /// Initialize target region entry.
3904 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3905     initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3906                                     StringRef ParentName, unsigned LineNum,
3907                                     unsigned Order) {
3908   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3909                                              "only required for the device "
3910                                              "code generation.");
3911   OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] =
3912       OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr,
3913                                    OMPTargetRegionEntryTargetRegion);
3914   ++OffloadingEntriesNum;
3915 }
3916 
3917 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3918     registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3919                                   StringRef ParentName, unsigned LineNum,
3920                                   llvm::Constant *Addr, llvm::Constant *ID,
3921                                   OMPTargetRegionEntryKind Flags) {
3922   // If we are emitting code for a target, the entry is already initialized,
3923   // only has to be registered.
3924   if (CGM.getLangOpts().OpenMPIsDevice) {
3925     if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) {
3926       unsigned DiagID = CGM.getDiags().getCustomDiagID(
3927           DiagnosticsEngine::Error,
3928           "Unable to find target region on line '%0' in the device code.");
3929       CGM.getDiags().Report(DiagID) << LineNum;
3930       return;
3931     }
3932     auto &Entry =
3933         OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum];
3934     assert(Entry.isValid() && "Entry not initialized!");
3935     Entry.setAddress(Addr);
3936     Entry.setID(ID);
3937     Entry.setFlags(Flags);
3938   } else {
3939     OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags);
3940     OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry;
3941     ++OffloadingEntriesNum;
3942   }
3943 }
3944 
3945 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo(
3946     unsigned DeviceID, unsigned FileID, StringRef ParentName,
3947     unsigned LineNum) const {
3948   auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID);
3949   if (PerDevice == OffloadEntriesTargetRegion.end())
3950     return false;
3951   auto PerFile = PerDevice->second.find(FileID);
3952   if (PerFile == PerDevice->second.end())
3953     return false;
3954   auto PerParentName = PerFile->second.find(ParentName);
3955   if (PerParentName == PerFile->second.end())
3956     return false;
3957   auto PerLine = PerParentName->second.find(LineNum);
3958   if (PerLine == PerParentName->second.end())
3959     return false;
3960   // Fail if this entry is already registered.
3961   if (PerLine->second.getAddress() || PerLine->second.getID())
3962     return false;
3963   return true;
3964 }
3965 
3966 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo(
3967     const OffloadTargetRegionEntryInfoActTy &Action) {
3968   // Scan all target region entries and perform the provided action.
3969   for (const auto &D : OffloadEntriesTargetRegion)
3970     for (const auto &F : D.second)
3971       for (const auto &P : F.second)
3972         for (const auto &L : P.second)
3973           Action(D.first, F.first, P.first(), L.first, L.second);
3974 }
3975 
3976 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3977     initializeDeviceGlobalVarEntryInfo(StringRef Name,
3978                                        OMPTargetGlobalVarEntryKind Flags,
3979                                        unsigned Order) {
3980   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3981                                              "only required for the device "
3982                                              "code generation.");
3983   OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
3984   ++OffloadingEntriesNum;
3985 }
3986 
3987 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3988     registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr,
3989                                      CharUnits VarSize,
3990                                      OMPTargetGlobalVarEntryKind Flags,
3991                                      llvm::GlobalValue::LinkageTypes Linkage) {
3992   if (CGM.getLangOpts().OpenMPIsDevice) {
3993     auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
3994     assert(Entry.isValid() && Entry.getFlags() == Flags &&
3995            "Entry not initialized!");
3996     assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
3997            "Resetting with the new address.");
3998     if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) {
3999       if (Entry.getVarSize().isZero()) {
4000         Entry.setVarSize(VarSize);
4001         Entry.setLinkage(Linkage);
4002       }
4003       return;
4004     }
4005     Entry.setVarSize(VarSize);
4006     Entry.setLinkage(Linkage);
4007     Entry.setAddress(Addr);
4008   } else {
4009     if (hasDeviceGlobalVarEntryInfo(VarName)) {
4010       auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
4011       assert(Entry.isValid() && Entry.getFlags() == Flags &&
4012              "Entry not initialized!");
4013       assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
4014              "Resetting with the new address.");
4015       if (Entry.getVarSize().isZero()) {
4016         Entry.setVarSize(VarSize);
4017         Entry.setLinkage(Linkage);
4018       }
4019       return;
4020     }
4021     OffloadEntriesDeviceGlobalVar.try_emplace(
4022         VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage);
4023     ++OffloadingEntriesNum;
4024   }
4025 }
4026 
4027 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
4028     actOnDeviceGlobalVarEntriesInfo(
4029         const OffloadDeviceGlobalVarEntryInfoActTy &Action) {
4030   // Scan all target region entries and perform the provided action.
4031   for (const auto &E : OffloadEntriesDeviceGlobalVar)
4032     Action(E.getKey(), E.getValue());
4033 }
4034 
4035 void CGOpenMPRuntime::createOffloadEntry(
4036     llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags,
4037     llvm::GlobalValue::LinkageTypes Linkage) {
4038   StringRef Name = Addr->getName();
4039   llvm::Module &M = CGM.getModule();
4040   llvm::LLVMContext &C = M.getContext();
4041 
4042   // Create constant string with the name.
4043   llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name);
4044 
4045   std::string StringName = getName({"omp_offloading", "entry_name"});
4046   auto *Str = new llvm::GlobalVariable(
4047       M, StrPtrInit->getType(), /*isConstant=*/true,
4048       llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName);
4049   Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
4050 
4051   llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy),
4052                             llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy),
4053                             llvm::ConstantInt::get(CGM.SizeTy, Size),
4054                             llvm::ConstantInt::get(CGM.Int32Ty, Flags),
4055                             llvm::ConstantInt::get(CGM.Int32Ty, 0)};
4056   std::string EntryName = getName({"omp_offloading", "entry", ""});
4057   llvm::GlobalVariable *Entry = createGlobalStruct(
4058       CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data,
4059       Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage);
4060 
4061   // The entry has to be created in the section the linker expects it to be.
4062   Entry->setSection("omp_offloading_entries");
4063 }
4064 
4065 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() {
4066   // Emit the offloading entries and metadata so that the device codegen side
4067   // can easily figure out what to emit. The produced metadata looks like
4068   // this:
4069   //
4070   // !omp_offload.info = !{!1, ...}
4071   //
4072   // Right now we only generate metadata for function that contain target
4073   // regions.
4074 
4075   // If we are in simd mode or there are no entries, we don't need to do
4076   // anything.
4077   if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty())
4078     return;
4079 
4080   llvm::Module &M = CGM.getModule();
4081   llvm::LLVMContext &C = M.getContext();
4082   SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *,
4083                          SourceLocation, StringRef>,
4084               16>
4085       OrderedEntries(OffloadEntriesInfoManager.size());
4086   llvm::SmallVector<StringRef, 16> ParentFunctions(
4087       OffloadEntriesInfoManager.size());
4088 
4089   // Auxiliary methods to create metadata values and strings.
4090   auto &&GetMDInt = [this](unsigned V) {
4091     return llvm::ConstantAsMetadata::get(
4092         llvm::ConstantInt::get(CGM.Int32Ty, V));
4093   };
4094 
4095   auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); };
4096 
4097   // Create the offloading info metadata node.
4098   llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info");
4099 
4100   // Create function that emits metadata for each target region entry;
4101   auto &&TargetRegionMetadataEmitter =
4102       [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt,
4103        &GetMDString](
4104           unsigned DeviceID, unsigned FileID, StringRef ParentName,
4105           unsigned Line,
4106           const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) {
4107         // Generate metadata for target regions. Each entry of this metadata
4108         // contains:
4109         // - Entry 0 -> Kind of this type of metadata (0).
4110         // - Entry 1 -> Device ID of the file where the entry was identified.
4111         // - Entry 2 -> File ID of the file where the entry was identified.
4112         // - Entry 3 -> Mangled name of the function where the entry was
4113         // identified.
4114         // - Entry 4 -> Line in the file where the entry was identified.
4115         // - Entry 5 -> Order the entry was created.
4116         // The first element of the metadata node is the kind.
4117         llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID),
4118                                  GetMDInt(FileID),      GetMDString(ParentName),
4119                                  GetMDInt(Line),        GetMDInt(E.getOrder())};
4120 
4121         SourceLocation Loc;
4122         for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(),
4123                   E = CGM.getContext().getSourceManager().fileinfo_end();
4124              I != E; ++I) {
4125           if (I->getFirst()->getUniqueID().getDevice() == DeviceID &&
4126               I->getFirst()->getUniqueID().getFile() == FileID) {
4127             Loc = CGM.getContext().getSourceManager().translateFileLineCol(
4128                 I->getFirst(), Line, 1);
4129             break;
4130           }
4131         }
4132         // Save this entry in the right position of the ordered entries array.
4133         OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName);
4134         ParentFunctions[E.getOrder()] = ParentName;
4135 
4136         // Add metadata to the named metadata node.
4137         MD->addOperand(llvm::MDNode::get(C, Ops));
4138       };
4139 
4140   OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo(
4141       TargetRegionMetadataEmitter);
4142 
4143   // Create function that emits metadata for each device global variable entry;
4144   auto &&DeviceGlobalVarMetadataEmitter =
4145       [&C, &OrderedEntries, &GetMDInt, &GetMDString,
4146        MD](StringRef MangledName,
4147            const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar
4148                &E) {
4149         // Generate metadata for global variables. Each entry of this metadata
4150         // contains:
4151         // - Entry 0 -> Kind of this type of metadata (1).
4152         // - Entry 1 -> Mangled name of the variable.
4153         // - Entry 2 -> Declare target kind.
4154         // - Entry 3 -> Order the entry was created.
4155         // The first element of the metadata node is the kind.
4156         llvm::Metadata *Ops[] = {
4157             GetMDInt(E.getKind()), GetMDString(MangledName),
4158             GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
4159 
4160         // Save this entry in the right position of the ordered entries array.
4161         OrderedEntries[E.getOrder()] =
4162             std::make_tuple(&E, SourceLocation(), MangledName);
4163 
4164         // Add metadata to the named metadata node.
4165         MD->addOperand(llvm::MDNode::get(C, Ops));
4166       };
4167 
4168   OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo(
4169       DeviceGlobalVarMetadataEmitter);
4170 
4171   for (const auto &E : OrderedEntries) {
4172     assert(std::get<0>(E) && "All ordered entries must exist!");
4173     if (const auto *CE =
4174             dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>(
4175                 std::get<0>(E))) {
4176       if (!CE->getID() || !CE->getAddress()) {
4177         // Do not blame the entry if the parent funtion is not emitted.
4178         StringRef FnName = ParentFunctions[CE->getOrder()];
4179         if (!CGM.GetGlobalValue(FnName))
4180           continue;
4181         unsigned DiagID = CGM.getDiags().getCustomDiagID(
4182             DiagnosticsEngine::Error,
4183             "Offloading entry for target region in %0 is incorrect: either the "
4184             "address or the ID is invalid.");
4185         CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName;
4186         continue;
4187       }
4188       createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0,
4189                          CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage);
4190     } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy::
4191                                              OffloadEntryInfoDeviceGlobalVar>(
4192                    std::get<0>(E))) {
4193       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags =
4194           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4195               CE->getFlags());
4196       switch (Flags) {
4197       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: {
4198         if (CGM.getLangOpts().OpenMPIsDevice &&
4199             CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())
4200           continue;
4201         if (!CE->getAddress()) {
4202           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4203               DiagnosticsEngine::Error, "Offloading entry for declare target "
4204                                         "variable %0 is incorrect: the "
4205                                         "address is invalid.");
4206           CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E);
4207           continue;
4208         }
4209         // The vaiable has no definition - no need to add the entry.
4210         if (CE->getVarSize().isZero())
4211           continue;
4212         break;
4213       }
4214       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink:
4215         assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) ||
4216                 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) &&
4217                "Declaret target link address is set.");
4218         if (CGM.getLangOpts().OpenMPIsDevice)
4219           continue;
4220         if (!CE->getAddress()) {
4221           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4222               DiagnosticsEngine::Error,
4223               "Offloading entry for declare target variable is incorrect: the "
4224               "address is invalid.");
4225           CGM.getDiags().Report(DiagID);
4226           continue;
4227         }
4228         break;
4229       }
4230       createOffloadEntry(CE->getAddress(), CE->getAddress(),
4231                          CE->getVarSize().getQuantity(), Flags,
4232                          CE->getLinkage());
4233     } else {
4234       llvm_unreachable("Unsupported entry kind.");
4235     }
4236   }
4237 }
4238 
4239 /// Loads all the offload entries information from the host IR
4240 /// metadata.
4241 void CGOpenMPRuntime::loadOffloadInfoMetadata() {
4242   // If we are in target mode, load the metadata from the host IR. This code has
4243   // to match the metadaata creation in createOffloadEntriesAndInfoMetadata().
4244 
4245   if (!CGM.getLangOpts().OpenMPIsDevice)
4246     return;
4247 
4248   if (CGM.getLangOpts().OMPHostIRFile.empty())
4249     return;
4250 
4251   auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile);
4252   if (auto EC = Buf.getError()) {
4253     CGM.getDiags().Report(diag::err_cannot_open_file)
4254         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4255     return;
4256   }
4257 
4258   llvm::LLVMContext C;
4259   auto ME = expectedToErrorOrAndEmitErrors(
4260       C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C));
4261 
4262   if (auto EC = ME.getError()) {
4263     unsigned DiagID = CGM.getDiags().getCustomDiagID(
4264         DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'");
4265     CGM.getDiags().Report(DiagID)
4266         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4267     return;
4268   }
4269 
4270   llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info");
4271   if (!MD)
4272     return;
4273 
4274   for (llvm::MDNode *MN : MD->operands()) {
4275     auto &&GetMDInt = [MN](unsigned Idx) {
4276       auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx));
4277       return cast<llvm::ConstantInt>(V->getValue())->getZExtValue();
4278     };
4279 
4280     auto &&GetMDString = [MN](unsigned Idx) {
4281       auto *V = cast<llvm::MDString>(MN->getOperand(Idx));
4282       return V->getString();
4283     };
4284 
4285     switch (GetMDInt(0)) {
4286     default:
4287       llvm_unreachable("Unexpected metadata!");
4288       break;
4289     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4290         OffloadingEntryInfoTargetRegion:
4291       OffloadEntriesInfoManager.initializeTargetRegionEntryInfo(
4292           /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2),
4293           /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4),
4294           /*Order=*/GetMDInt(5));
4295       break;
4296     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4297         OffloadingEntryInfoDeviceGlobalVar:
4298       OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo(
4299           /*MangledName=*/GetMDString(1),
4300           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4301               /*Flags=*/GetMDInt(2)),
4302           /*Order=*/GetMDInt(3));
4303       break;
4304     }
4305   }
4306 }
4307 
4308 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
4309   if (!KmpRoutineEntryPtrTy) {
4310     // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
4311     ASTContext &C = CGM.getContext();
4312     QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
4313     FunctionProtoType::ExtProtoInfo EPI;
4314     KmpRoutineEntryPtrQTy = C.getPointerType(
4315         C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
4316     KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
4317   }
4318 }
4319 
4320 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() {
4321   // Make sure the type of the entry is already created. This is the type we
4322   // have to create:
4323   // struct __tgt_offload_entry{
4324   //   void      *addr;       // Pointer to the offload entry info.
4325   //                          // (function or global)
4326   //   char      *name;       // Name of the function or global.
4327   //   size_t     size;       // Size of the entry info (0 if it a function).
4328   //   int32_t    flags;      // Flags associated with the entry, e.g. 'link'.
4329   //   int32_t    reserved;   // Reserved, to use by the runtime library.
4330   // };
4331   if (TgtOffloadEntryQTy.isNull()) {
4332     ASTContext &C = CGM.getContext();
4333     RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry");
4334     RD->startDefinition();
4335     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4336     addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy));
4337     addFieldToRecordDecl(C, RD, C.getSizeType());
4338     addFieldToRecordDecl(
4339         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4340     addFieldToRecordDecl(
4341         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4342     RD->completeDefinition();
4343     RD->addAttr(PackedAttr::CreateImplicit(C));
4344     TgtOffloadEntryQTy = C.getRecordType(RD);
4345   }
4346   return TgtOffloadEntryQTy;
4347 }
4348 
4349 namespace {
4350 struct PrivateHelpersTy {
4351   PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original,
4352                    const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit)
4353       : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy),
4354         PrivateElemInit(PrivateElemInit) {}
4355   const Expr *OriginalRef = nullptr;
4356   const VarDecl *Original = nullptr;
4357   const VarDecl *PrivateCopy = nullptr;
4358   const VarDecl *PrivateElemInit = nullptr;
4359 };
4360 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
4361 } // anonymous namespace
4362 
4363 static RecordDecl *
4364 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) {
4365   if (!Privates.empty()) {
4366     ASTContext &C = CGM.getContext();
4367     // Build struct .kmp_privates_t. {
4368     //         /*  private vars  */
4369     //       };
4370     RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t");
4371     RD->startDefinition();
4372     for (const auto &Pair : Privates) {
4373       const VarDecl *VD = Pair.second.Original;
4374       QualType Type = VD->getType().getNonReferenceType();
4375       FieldDecl *FD = addFieldToRecordDecl(C, RD, Type);
4376       if (VD->hasAttrs()) {
4377         for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()),
4378              E(VD->getAttrs().end());
4379              I != E; ++I)
4380           FD->addAttr(*I);
4381       }
4382     }
4383     RD->completeDefinition();
4384     return RD;
4385   }
4386   return nullptr;
4387 }
4388 
4389 static RecordDecl *
4390 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind,
4391                          QualType KmpInt32Ty,
4392                          QualType KmpRoutineEntryPointerQTy) {
4393   ASTContext &C = CGM.getContext();
4394   // Build struct kmp_task_t {
4395   //         void *              shareds;
4396   //         kmp_routine_entry_t routine;
4397   //         kmp_int32           part_id;
4398   //         kmp_cmplrdata_t data1;
4399   //         kmp_cmplrdata_t data2;
4400   // For taskloops additional fields:
4401   //         kmp_uint64          lb;
4402   //         kmp_uint64          ub;
4403   //         kmp_int64           st;
4404   //         kmp_int32           liter;
4405   //         void *              reductions;
4406   //       };
4407   RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union);
4408   UD->startDefinition();
4409   addFieldToRecordDecl(C, UD, KmpInt32Ty);
4410   addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy);
4411   UD->completeDefinition();
4412   QualType KmpCmplrdataTy = C.getRecordType(UD);
4413   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t");
4414   RD->startDefinition();
4415   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4416   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
4417   addFieldToRecordDecl(C, RD, KmpInt32Ty);
4418   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4419   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4420   if (isOpenMPTaskLoopDirective(Kind)) {
4421     QualType KmpUInt64Ty =
4422         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
4423     QualType KmpInt64Ty =
4424         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
4425     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4426     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4427     addFieldToRecordDecl(C, RD, KmpInt64Ty);
4428     addFieldToRecordDecl(C, RD, KmpInt32Ty);
4429     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4430   }
4431   RD->completeDefinition();
4432   return RD;
4433 }
4434 
4435 static RecordDecl *
4436 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy,
4437                                      ArrayRef<PrivateDataTy> Privates) {
4438   ASTContext &C = CGM.getContext();
4439   // Build struct kmp_task_t_with_privates {
4440   //         kmp_task_t task_data;
4441   //         .kmp_privates_t. privates;
4442   //       };
4443   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates");
4444   RD->startDefinition();
4445   addFieldToRecordDecl(C, RD, KmpTaskTQTy);
4446   if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates))
4447     addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD));
4448   RD->completeDefinition();
4449   return RD;
4450 }
4451 
4452 /// Emit a proxy function which accepts kmp_task_t as the second
4453 /// argument.
4454 /// \code
4455 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
4456 ///   TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt,
4457 ///   For taskloops:
4458 ///   tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4459 ///   tt->reductions, tt->shareds);
4460 ///   return 0;
4461 /// }
4462 /// \endcode
4463 static llvm::Function *
4464 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
4465                       OpenMPDirectiveKind Kind, QualType KmpInt32Ty,
4466                       QualType KmpTaskTWithPrivatesPtrQTy,
4467                       QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
4468                       QualType SharedsPtrTy, llvm::Function *TaskFunction,
4469                       llvm::Value *TaskPrivatesMap) {
4470   ASTContext &C = CGM.getContext();
4471   FunctionArgList Args;
4472   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4473                             ImplicitParamDecl::Other);
4474   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4475                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4476                                 ImplicitParamDecl::Other);
4477   Args.push_back(&GtidArg);
4478   Args.push_back(&TaskTypeArg);
4479   const auto &TaskEntryFnInfo =
4480       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4481   llvm::FunctionType *TaskEntryTy =
4482       CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
4483   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""});
4484   auto *TaskEntry = llvm::Function::Create(
4485       TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4486   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo);
4487   TaskEntry->setDoesNotRecurse();
4488   CodeGenFunction CGF(CGM);
4489   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args,
4490                     Loc, Loc);
4491 
4492   // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map,
4493   // tt,
4494   // For taskloops:
4495   // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4496   // tt->task_data.shareds);
4497   llvm::Value *GtidParam = CGF.EmitLoadOfScalar(
4498       CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc);
4499   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4500       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4501       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4502   const auto *KmpTaskTWithPrivatesQTyRD =
4503       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4504   LValue Base =
4505       CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4506   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
4507   auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
4508   LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI);
4509   llvm::Value *PartidParam = PartIdLVal.getPointer(CGF);
4510 
4511   auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds);
4512   LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI);
4513   llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4514       CGF.EmitLoadOfScalar(SharedsLVal, Loc),
4515       CGF.ConvertTypeForMem(SharedsPtrTy));
4516 
4517   auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
4518   llvm::Value *PrivatesParam;
4519   if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) {
4520     LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI);
4521     PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4522         PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy);
4523   } else {
4524     PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4525   }
4526 
4527   llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam,
4528                                TaskPrivatesMap,
4529                                CGF.Builder
4530                                    .CreatePointerBitCastOrAddrSpaceCast(
4531                                        TDBase.getAddress(CGF), CGF.VoidPtrTy)
4532                                    .getPointer()};
4533   SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs),
4534                                           std::end(CommonArgs));
4535   if (isOpenMPTaskLoopDirective(Kind)) {
4536     auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound);
4537     LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI);
4538     llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc);
4539     auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound);
4540     LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI);
4541     llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc);
4542     auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride);
4543     LValue StLVal = CGF.EmitLValueForField(Base, *StFI);
4544     llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc);
4545     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4546     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4547     llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc);
4548     auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions);
4549     LValue RLVal = CGF.EmitLValueForField(Base, *RFI);
4550     llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc);
4551     CallArgs.push_back(LBParam);
4552     CallArgs.push_back(UBParam);
4553     CallArgs.push_back(StParam);
4554     CallArgs.push_back(LIParam);
4555     CallArgs.push_back(RParam);
4556   }
4557   CallArgs.push_back(SharedsParam);
4558 
4559   CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction,
4560                                                   CallArgs);
4561   CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)),
4562                              CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty));
4563   CGF.FinishFunction();
4564   return TaskEntry;
4565 }
4566 
4567 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM,
4568                                             SourceLocation Loc,
4569                                             QualType KmpInt32Ty,
4570                                             QualType KmpTaskTWithPrivatesPtrQTy,
4571                                             QualType KmpTaskTWithPrivatesQTy) {
4572   ASTContext &C = CGM.getContext();
4573   FunctionArgList Args;
4574   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4575                             ImplicitParamDecl::Other);
4576   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4577                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4578                                 ImplicitParamDecl::Other);
4579   Args.push_back(&GtidArg);
4580   Args.push_back(&TaskTypeArg);
4581   const auto &DestructorFnInfo =
4582       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4583   llvm::FunctionType *DestructorFnTy =
4584       CGM.getTypes().GetFunctionType(DestructorFnInfo);
4585   std::string Name =
4586       CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""});
4587   auto *DestructorFn =
4588       llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage,
4589                              Name, &CGM.getModule());
4590   CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn,
4591                                     DestructorFnInfo);
4592   DestructorFn->setDoesNotRecurse();
4593   CodeGenFunction CGF(CGM);
4594   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo,
4595                     Args, Loc, Loc);
4596 
4597   LValue Base = CGF.EmitLoadOfPointerLValue(
4598       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4599       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4600   const auto *KmpTaskTWithPrivatesQTyRD =
4601       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4602   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4603   Base = CGF.EmitLValueForField(Base, *FI);
4604   for (const auto *Field :
4605        cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) {
4606     if (QualType::DestructionKind DtorKind =
4607             Field->getType().isDestructedType()) {
4608       LValue FieldLValue = CGF.EmitLValueForField(Base, Field);
4609       CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType());
4610     }
4611   }
4612   CGF.FinishFunction();
4613   return DestructorFn;
4614 }
4615 
4616 /// Emit a privates mapping function for correct handling of private and
4617 /// firstprivate variables.
4618 /// \code
4619 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1>
4620 /// **noalias priv1,...,  <tyn> **noalias privn) {
4621 ///   *priv1 = &.privates.priv1;
4622 ///   ...;
4623 ///   *privn = &.privates.privn;
4624 /// }
4625 /// \endcode
4626 static llvm::Value *
4627 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc,
4628                                ArrayRef<const Expr *> PrivateVars,
4629                                ArrayRef<const Expr *> FirstprivateVars,
4630                                ArrayRef<const Expr *> LastprivateVars,
4631                                QualType PrivatesQTy,
4632                                ArrayRef<PrivateDataTy> Privates) {
4633   ASTContext &C = CGM.getContext();
4634   FunctionArgList Args;
4635   ImplicitParamDecl TaskPrivatesArg(
4636       C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4637       C.getPointerType(PrivatesQTy).withConst().withRestrict(),
4638       ImplicitParamDecl::Other);
4639   Args.push_back(&TaskPrivatesArg);
4640   llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos;
4641   unsigned Counter = 1;
4642   for (const Expr *E : PrivateVars) {
4643     Args.push_back(ImplicitParamDecl::Create(
4644         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4645         C.getPointerType(C.getPointerType(E->getType()))
4646             .withConst()
4647             .withRestrict(),
4648         ImplicitParamDecl::Other));
4649     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4650     PrivateVarsPos[VD] = Counter;
4651     ++Counter;
4652   }
4653   for (const Expr *E : FirstprivateVars) {
4654     Args.push_back(ImplicitParamDecl::Create(
4655         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4656         C.getPointerType(C.getPointerType(E->getType()))
4657             .withConst()
4658             .withRestrict(),
4659         ImplicitParamDecl::Other));
4660     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4661     PrivateVarsPos[VD] = Counter;
4662     ++Counter;
4663   }
4664   for (const Expr *E : LastprivateVars) {
4665     Args.push_back(ImplicitParamDecl::Create(
4666         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4667         C.getPointerType(C.getPointerType(E->getType()))
4668             .withConst()
4669             .withRestrict(),
4670         ImplicitParamDecl::Other));
4671     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4672     PrivateVarsPos[VD] = Counter;
4673     ++Counter;
4674   }
4675   const auto &TaskPrivatesMapFnInfo =
4676       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4677   llvm::FunctionType *TaskPrivatesMapTy =
4678       CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo);
4679   std::string Name =
4680       CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""});
4681   auto *TaskPrivatesMap = llvm::Function::Create(
4682       TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name,
4683       &CGM.getModule());
4684   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap,
4685                                     TaskPrivatesMapFnInfo);
4686   if (CGM.getLangOpts().Optimize) {
4687     TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline);
4688     TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone);
4689     TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline);
4690   }
4691   CodeGenFunction CGF(CGM);
4692   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap,
4693                     TaskPrivatesMapFnInfo, Args, Loc, Loc);
4694 
4695   // *privi = &.privates.privi;
4696   LValue Base = CGF.EmitLoadOfPointerLValue(
4697       CGF.GetAddrOfLocalVar(&TaskPrivatesArg),
4698       TaskPrivatesArg.getType()->castAs<PointerType>());
4699   const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl());
4700   Counter = 0;
4701   for (const FieldDecl *Field : PrivatesQTyRD->fields()) {
4702     LValue FieldLVal = CGF.EmitLValueForField(Base, Field);
4703     const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]];
4704     LValue RefLVal =
4705         CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType());
4706     LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue(
4707         RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>());
4708     CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal);
4709     ++Counter;
4710   }
4711   CGF.FinishFunction();
4712   return TaskPrivatesMap;
4713 }
4714 
4715 /// Emit initialization for private variables in task-based directives.
4716 static void emitPrivatesInit(CodeGenFunction &CGF,
4717                              const OMPExecutableDirective &D,
4718                              Address KmpTaskSharedsPtr, LValue TDBase,
4719                              const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4720                              QualType SharedsTy, QualType SharedsPtrTy,
4721                              const OMPTaskDataTy &Data,
4722                              ArrayRef<PrivateDataTy> Privates, bool ForDup) {
4723   ASTContext &C = CGF.getContext();
4724   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4725   LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI);
4726   OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind())
4727                                  ? OMPD_taskloop
4728                                  : OMPD_task;
4729   const CapturedStmt &CS = *D.getCapturedStmt(Kind);
4730   CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS);
4731   LValue SrcBase;
4732   bool IsTargetTask =
4733       isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) ||
4734       isOpenMPTargetExecutionDirective(D.getDirectiveKind());
4735   // For target-based directives skip 3 firstprivate arrays BasePointersArray,
4736   // PointersArray and SizesArray. The original variables for these arrays are
4737   // not captured and we get their addresses explicitly.
4738   if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) ||
4739       (IsTargetTask && KmpTaskSharedsPtr.isValid())) {
4740     SrcBase = CGF.MakeAddrLValue(
4741         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4742             KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)),
4743         SharedsTy);
4744   }
4745   FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin();
4746   for (const PrivateDataTy &Pair : Privates) {
4747     const VarDecl *VD = Pair.second.PrivateCopy;
4748     const Expr *Init = VD->getAnyInitializer();
4749     if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) &&
4750                              !CGF.isTrivialInitializer(Init)))) {
4751       LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
4752       if (const VarDecl *Elem = Pair.second.PrivateElemInit) {
4753         const VarDecl *OriginalVD = Pair.second.Original;
4754         // Check if the variable is the target-based BasePointersArray,
4755         // PointersArray or SizesArray.
4756         LValue SharedRefLValue;
4757         QualType Type = PrivateLValue.getType();
4758         const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD);
4759         if (IsTargetTask && !SharedField) {
4760           assert(isa<ImplicitParamDecl>(OriginalVD) &&
4761                  isa<CapturedDecl>(OriginalVD->getDeclContext()) &&
4762                  cast<CapturedDecl>(OriginalVD->getDeclContext())
4763                          ->getNumParams() == 0 &&
4764                  isa<TranslationUnitDecl>(
4765                      cast<CapturedDecl>(OriginalVD->getDeclContext())
4766                          ->getDeclContext()) &&
4767                  "Expected artificial target data variable.");
4768           SharedRefLValue =
4769               CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type);
4770         } else if (ForDup) {
4771           SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField);
4772           SharedRefLValue = CGF.MakeAddrLValue(
4773               Address(SharedRefLValue.getPointer(CGF),
4774                       C.getDeclAlign(OriginalVD)),
4775               SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl),
4776               SharedRefLValue.getTBAAInfo());
4777         } else {
4778           InlinedOpenMPRegionRAII Region(
4779               CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown,
4780               /*HasCancel=*/false);
4781           SharedRefLValue =  CGF.EmitLValue(Pair.second.OriginalRef);
4782         }
4783         if (Type->isArrayType()) {
4784           // Initialize firstprivate array.
4785           if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) {
4786             // Perform simple memcpy.
4787             CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type);
4788           } else {
4789             // Initialize firstprivate array using element-by-element
4790             // initialization.
4791             CGF.EmitOMPAggregateAssign(
4792                 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF),
4793                 Type,
4794                 [&CGF, Elem, Init, &CapturesInfo](Address DestElement,
4795                                                   Address SrcElement) {
4796                   // Clean up any temporaries needed by the initialization.
4797                   CodeGenFunction::OMPPrivateScope InitScope(CGF);
4798                   InitScope.addPrivate(
4799                       Elem, [SrcElement]() -> Address { return SrcElement; });
4800                   (void)InitScope.Privatize();
4801                   // Emit initialization for single element.
4802                   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(
4803                       CGF, &CapturesInfo);
4804                   CGF.EmitAnyExprToMem(Init, DestElement,
4805                                        Init->getType().getQualifiers(),
4806                                        /*IsInitializer=*/false);
4807                 });
4808           }
4809         } else {
4810           CodeGenFunction::OMPPrivateScope InitScope(CGF);
4811           InitScope.addPrivate(Elem, [SharedRefLValue, &CGF]() -> Address {
4812             return SharedRefLValue.getAddress(CGF);
4813           });
4814           (void)InitScope.Privatize();
4815           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo);
4816           CGF.EmitExprAsInit(Init, VD, PrivateLValue,
4817                              /*capturedByInit=*/false);
4818         }
4819       } else {
4820         CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false);
4821       }
4822     }
4823     ++FI;
4824   }
4825 }
4826 
4827 /// Check if duplication function is required for taskloops.
4828 static bool checkInitIsRequired(CodeGenFunction &CGF,
4829                                 ArrayRef<PrivateDataTy> Privates) {
4830   bool InitRequired = false;
4831   for (const PrivateDataTy &Pair : Privates) {
4832     const VarDecl *VD = Pair.second.PrivateCopy;
4833     const Expr *Init = VD->getAnyInitializer();
4834     InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) &&
4835                                     !CGF.isTrivialInitializer(Init));
4836     if (InitRequired)
4837       break;
4838   }
4839   return InitRequired;
4840 }
4841 
4842 
4843 /// Emit task_dup function (for initialization of
4844 /// private/firstprivate/lastprivate vars and last_iter flag)
4845 /// \code
4846 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int
4847 /// lastpriv) {
4848 /// // setup lastprivate flag
4849 ///    task_dst->last = lastpriv;
4850 /// // could be constructor calls here...
4851 /// }
4852 /// \endcode
4853 static llvm::Value *
4854 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc,
4855                     const OMPExecutableDirective &D,
4856                     QualType KmpTaskTWithPrivatesPtrQTy,
4857                     const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4858                     const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy,
4859                     QualType SharedsPtrTy, const OMPTaskDataTy &Data,
4860                     ArrayRef<PrivateDataTy> Privates, bool WithLastIter) {
4861   ASTContext &C = CGM.getContext();
4862   FunctionArgList Args;
4863   ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4864                            KmpTaskTWithPrivatesPtrQTy,
4865                            ImplicitParamDecl::Other);
4866   ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4867                            KmpTaskTWithPrivatesPtrQTy,
4868                            ImplicitParamDecl::Other);
4869   ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy,
4870                                 ImplicitParamDecl::Other);
4871   Args.push_back(&DstArg);
4872   Args.push_back(&SrcArg);
4873   Args.push_back(&LastprivArg);
4874   const auto &TaskDupFnInfo =
4875       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4876   llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo);
4877   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""});
4878   auto *TaskDup = llvm::Function::Create(
4879       TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4880   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo);
4881   TaskDup->setDoesNotRecurse();
4882   CodeGenFunction CGF(CGM);
4883   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc,
4884                     Loc);
4885 
4886   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4887       CGF.GetAddrOfLocalVar(&DstArg),
4888       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4889   // task_dst->liter = lastpriv;
4890   if (WithLastIter) {
4891     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4892     LValue Base = CGF.EmitLValueForField(
4893         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4894     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4895     llvm::Value *Lastpriv = CGF.EmitLoadOfScalar(
4896         CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc);
4897     CGF.EmitStoreOfScalar(Lastpriv, LILVal);
4898   }
4899 
4900   // Emit initial values for private copies (if any).
4901   assert(!Privates.empty());
4902   Address KmpTaskSharedsPtr = Address::invalid();
4903   if (!Data.FirstprivateVars.empty()) {
4904     LValue TDBase = CGF.EmitLoadOfPointerLValue(
4905         CGF.GetAddrOfLocalVar(&SrcArg),
4906         KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4907     LValue Base = CGF.EmitLValueForField(
4908         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4909     KmpTaskSharedsPtr = Address(
4910         CGF.EmitLoadOfScalar(CGF.EmitLValueForField(
4911                                  Base, *std::next(KmpTaskTQTyRD->field_begin(),
4912                                                   KmpTaskTShareds)),
4913                              Loc),
4914         CGF.getNaturalTypeAlignment(SharedsTy));
4915   }
4916   emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD,
4917                    SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true);
4918   CGF.FinishFunction();
4919   return TaskDup;
4920 }
4921 
4922 /// Checks if destructor function is required to be generated.
4923 /// \return true if cleanups are required, false otherwise.
4924 static bool
4925 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) {
4926   bool NeedsCleanup = false;
4927   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
4928   const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl());
4929   for (const FieldDecl *FD : PrivateRD->fields()) {
4930     NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType();
4931     if (NeedsCleanup)
4932       break;
4933   }
4934   return NeedsCleanup;
4935 }
4936 
4937 CGOpenMPRuntime::TaskResultTy
4938 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc,
4939                               const OMPExecutableDirective &D,
4940                               llvm::Function *TaskFunction, QualType SharedsTy,
4941                               Address Shareds, const OMPTaskDataTy &Data) {
4942   ASTContext &C = CGM.getContext();
4943   llvm::SmallVector<PrivateDataTy, 4> Privates;
4944   // Aggregate privates and sort them by the alignment.
4945   const auto *I = Data.PrivateCopies.begin();
4946   for (const Expr *E : Data.PrivateVars) {
4947     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4948     Privates.emplace_back(
4949         C.getDeclAlign(VD),
4950         PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4951                          /*PrivateElemInit=*/nullptr));
4952     ++I;
4953   }
4954   I = Data.FirstprivateCopies.begin();
4955   const auto *IElemInitRef = Data.FirstprivateInits.begin();
4956   for (const Expr *E : Data.FirstprivateVars) {
4957     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4958     Privates.emplace_back(
4959         C.getDeclAlign(VD),
4960         PrivateHelpersTy(
4961             E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4962             cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl())));
4963     ++I;
4964     ++IElemInitRef;
4965   }
4966   I = Data.LastprivateCopies.begin();
4967   for (const Expr *E : Data.LastprivateVars) {
4968     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4969     Privates.emplace_back(
4970         C.getDeclAlign(VD),
4971         PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4972                          /*PrivateElemInit=*/nullptr));
4973     ++I;
4974   }
4975   llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) {
4976     return L.first > R.first;
4977   });
4978   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
4979   // Build type kmp_routine_entry_t (if not built yet).
4980   emitKmpRoutineEntryT(KmpInt32Ty);
4981   // Build type kmp_task_t (if not built yet).
4982   if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) {
4983     if (SavedKmpTaskloopTQTy.isNull()) {
4984       SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl(
4985           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
4986     }
4987     KmpTaskTQTy = SavedKmpTaskloopTQTy;
4988   } else {
4989     assert((D.getDirectiveKind() == OMPD_task ||
4990             isOpenMPTargetExecutionDirective(D.getDirectiveKind()) ||
4991             isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) &&
4992            "Expected taskloop, task or target directive");
4993     if (SavedKmpTaskTQTy.isNull()) {
4994       SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl(
4995           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
4996     }
4997     KmpTaskTQTy = SavedKmpTaskTQTy;
4998   }
4999   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
5000   // Build particular struct kmp_task_t for the given task.
5001   const RecordDecl *KmpTaskTWithPrivatesQTyRD =
5002       createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates);
5003   QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD);
5004   QualType KmpTaskTWithPrivatesPtrQTy =
5005       C.getPointerType(KmpTaskTWithPrivatesQTy);
5006   llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy);
5007   llvm::Type *KmpTaskTWithPrivatesPtrTy =
5008       KmpTaskTWithPrivatesTy->getPointerTo();
5009   llvm::Value *KmpTaskTWithPrivatesTySize =
5010       CGF.getTypeSize(KmpTaskTWithPrivatesQTy);
5011   QualType SharedsPtrTy = C.getPointerType(SharedsTy);
5012 
5013   // Emit initial values for private copies (if any).
5014   llvm::Value *TaskPrivatesMap = nullptr;
5015   llvm::Type *TaskPrivatesMapTy =
5016       std::next(TaskFunction->arg_begin(), 3)->getType();
5017   if (!Privates.empty()) {
5018     auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
5019     TaskPrivatesMap = emitTaskPrivateMappingFunction(
5020         CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars,
5021         FI->getType(), Privates);
5022     TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5023         TaskPrivatesMap, TaskPrivatesMapTy);
5024   } else {
5025     TaskPrivatesMap = llvm::ConstantPointerNull::get(
5026         cast<llvm::PointerType>(TaskPrivatesMapTy));
5027   }
5028   // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
5029   // kmp_task_t *tt);
5030   llvm::Function *TaskEntry = emitProxyTaskFunction(
5031       CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5032       KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction,
5033       TaskPrivatesMap);
5034 
5035   // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
5036   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
5037   // kmp_routine_entry_t *task_entry);
5038   // Task flags. Format is taken from
5039   // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h,
5040   // description of kmp_tasking_flags struct.
5041   enum {
5042     TiedFlag = 0x1,
5043     FinalFlag = 0x2,
5044     DestructorsFlag = 0x8,
5045     PriorityFlag = 0x20,
5046     DetachableFlag = 0x40,
5047   };
5048   unsigned Flags = Data.Tied ? TiedFlag : 0;
5049   bool NeedsCleanup = false;
5050   if (!Privates.empty()) {
5051     NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD);
5052     if (NeedsCleanup)
5053       Flags = Flags | DestructorsFlag;
5054   }
5055   if (Data.Priority.getInt())
5056     Flags = Flags | PriorityFlag;
5057   if (D.hasClausesOfKind<OMPDetachClause>())
5058     Flags = Flags | DetachableFlag;
5059   llvm::Value *TaskFlags =
5060       Data.Final.getPointer()
5061           ? CGF.Builder.CreateSelect(Data.Final.getPointer(),
5062                                      CGF.Builder.getInt32(FinalFlag),
5063                                      CGF.Builder.getInt32(/*C=*/0))
5064           : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0);
5065   TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
5066   llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy));
5067   SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc),
5068       getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize,
5069       SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5070           TaskEntry, KmpRoutineEntryPtrTy)};
5071   llvm::Value *NewTask;
5072   if (D.hasClausesOfKind<OMPNowaitClause>()) {
5073     // Check if we have any device clause associated with the directive.
5074     const Expr *Device = nullptr;
5075     if (auto *C = D.getSingleClause<OMPDeviceClause>())
5076       Device = C->getDevice();
5077     // Emit device ID if any otherwise use default value.
5078     llvm::Value *DeviceID;
5079     if (Device)
5080       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
5081                                            CGF.Int64Ty, /*isSigned=*/true);
5082     else
5083       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
5084     AllocArgs.push_back(DeviceID);
5085     NewTask = CGF.EmitRuntimeCall(
5086       createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs);
5087   } else {
5088     NewTask = CGF.EmitRuntimeCall(
5089       createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs);
5090   }
5091   // Emit detach clause initialization.
5092   // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid,
5093   // task_descriptor);
5094   if (const auto *DC = D.getSingleClause<OMPDetachClause>()) {
5095     const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts();
5096     LValue EvtLVal = CGF.EmitLValue(Evt);
5097 
5098     // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
5099     // int gtid, kmp_task_t *task);
5100     llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc());
5101     llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc());
5102     Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false);
5103     llvm::Value *EvtVal = CGF.EmitRuntimeCall(
5104         createRuntimeFunction(OMPRTL__kmpc_task_allow_completion_event),
5105         {Loc, Tid, NewTask});
5106     EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(),
5107                                       Evt->getExprLoc());
5108     CGF.EmitStoreOfScalar(EvtVal, EvtLVal);
5109   }
5110   llvm::Value *NewTaskNewTaskTTy =
5111       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5112           NewTask, KmpTaskTWithPrivatesPtrTy);
5113   LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy,
5114                                                KmpTaskTWithPrivatesQTy);
5115   LValue TDBase =
5116       CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
5117   // Fill the data in the resulting kmp_task_t record.
5118   // Copy shareds if there are any.
5119   Address KmpTaskSharedsPtr = Address::invalid();
5120   if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) {
5121     KmpTaskSharedsPtr =
5122         Address(CGF.EmitLoadOfScalar(
5123                     CGF.EmitLValueForField(
5124                         TDBase, *std::next(KmpTaskTQTyRD->field_begin(),
5125                                            KmpTaskTShareds)),
5126                     Loc),
5127                 CGF.getNaturalTypeAlignment(SharedsTy));
5128     LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy);
5129     LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy);
5130     CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap);
5131   }
5132   // Emit initial values for private copies (if any).
5133   TaskResultTy Result;
5134   if (!Privates.empty()) {
5135     emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD,
5136                      SharedsTy, SharedsPtrTy, Data, Privates,
5137                      /*ForDup=*/false);
5138     if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) &&
5139         (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) {
5140       Result.TaskDupFn = emitTaskDupFunction(
5141           CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD,
5142           KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates,
5143           /*WithLastIter=*/!Data.LastprivateVars.empty());
5144     }
5145   }
5146   // Fields of union "kmp_cmplrdata_t" for destructors and priority.
5147   enum { Priority = 0, Destructors = 1 };
5148   // Provide pointer to function with destructors for privates.
5149   auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1);
5150   const RecordDecl *KmpCmplrdataUD =
5151       (*FI)->getType()->getAsUnionType()->getDecl();
5152   if (NeedsCleanup) {
5153     llvm::Value *DestructorFn = emitDestructorsFunction(
5154         CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5155         KmpTaskTWithPrivatesQTy);
5156     LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI);
5157     LValue DestructorsLV = CGF.EmitLValueForField(
5158         Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors));
5159     CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5160                               DestructorFn, KmpRoutineEntryPtrTy),
5161                           DestructorsLV);
5162   }
5163   // Set priority.
5164   if (Data.Priority.getInt()) {
5165     LValue Data2LV = CGF.EmitLValueForField(
5166         TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2));
5167     LValue PriorityLV = CGF.EmitLValueForField(
5168         Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority));
5169     CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV);
5170   }
5171   Result.NewTask = NewTask;
5172   Result.TaskEntry = TaskEntry;
5173   Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy;
5174   Result.TDBase = TDBase;
5175   Result.KmpTaskTQTyRD = KmpTaskTQTyRD;
5176   return Result;
5177 }
5178 
5179 namespace {
5180 /// Dependence kind for RTL.
5181 enum RTLDependenceKindTy {
5182   DepIn = 0x01,
5183   DepInOut = 0x3,
5184   DepMutexInOutSet = 0x4
5185 };
5186 /// Fields ids in kmp_depend_info record.
5187 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags };
5188 } // namespace
5189 
5190 /// Translates internal dependency kind into the runtime kind.
5191 static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) {
5192   RTLDependenceKindTy DepKind;
5193   switch (K) {
5194   case OMPC_DEPEND_in:
5195     DepKind = DepIn;
5196     break;
5197   // Out and InOut dependencies must use the same code.
5198   case OMPC_DEPEND_out:
5199   case OMPC_DEPEND_inout:
5200     DepKind = DepInOut;
5201     break;
5202   case OMPC_DEPEND_mutexinoutset:
5203     DepKind = DepMutexInOutSet;
5204     break;
5205   case OMPC_DEPEND_source:
5206   case OMPC_DEPEND_sink:
5207   case OMPC_DEPEND_depobj:
5208   case OMPC_DEPEND_unknown:
5209     llvm_unreachable("Unknown task dependence type");
5210   }
5211   return DepKind;
5212 }
5213 
5214 /// Builds kmp_depend_info, if it is not built yet, and builds flags type.
5215 static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy,
5216                            QualType &FlagsTy) {
5217   FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false);
5218   if (KmpDependInfoTy.isNull()) {
5219     RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info");
5220     KmpDependInfoRD->startDefinition();
5221     addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType());
5222     addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType());
5223     addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy);
5224     KmpDependInfoRD->completeDefinition();
5225     KmpDependInfoTy = C.getRecordType(KmpDependInfoRD);
5226   }
5227 }
5228 
5229 std::pair<llvm::Value *, LValue>
5230 CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal,
5231                                    SourceLocation Loc) {
5232   ASTContext &C = CGM.getContext();
5233   QualType FlagsTy;
5234   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5235   RecordDecl *KmpDependInfoRD =
5236       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5237   LValue Base = CGF.EmitLoadOfPointerLValue(
5238       DepobjLVal.getAddress(CGF),
5239       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
5240   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
5241   Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5242           Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy));
5243   Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(),
5244                             Base.getTBAAInfo());
5245   llvm::Value *DepObjAddr = CGF.Builder.CreateGEP(
5246       Addr.getPointer(),
5247       llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
5248   LValue NumDepsBase = CGF.MakeAddrLValue(
5249       Address(DepObjAddr, Addr.getAlignment()), KmpDependInfoTy,
5250       Base.getBaseInfo(), Base.getTBAAInfo());
5251   // NumDeps = deps[i].base_addr;
5252   LValue BaseAddrLVal = CGF.EmitLValueForField(
5253       NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5254   llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc);
5255   return std::make_pair(NumDeps, Base);
5256 }
5257 
5258 std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause(
5259     CodeGenFunction &CGF,
5260     ArrayRef<std::pair<OpenMPDependClauseKind, const Expr *>> Dependencies,
5261     bool ForDepobj, SourceLocation Loc) {
5262   // Process list of dependencies.
5263   ASTContext &C = CGM.getContext();
5264   Address DependenciesArray = Address::invalid();
5265   unsigned NumDependencies = Dependencies.size();
5266   llvm::Value *NumOfElements = nullptr;
5267   if (NumDependencies) {
5268     QualType FlagsTy;
5269     getDependTypes(C, KmpDependInfoTy, FlagsTy);
5270     RecordDecl *KmpDependInfoRD =
5271         cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5272     llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
5273     unsigned NumDepobjDependecies = 0;
5274     SmallVector<std::pair<llvm::Value *, LValue>, 4> Depobjs;
5275     llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0);
5276     // Calculate number of depobj dependecies.
5277     for (const std::pair<OpenMPDependClauseKind, const Expr *> &Pair :
5278          Dependencies) {
5279       if (Pair.first != OMPC_DEPEND_depobj)
5280         continue;
5281       LValue DepobjLVal = CGF.EmitLValue(Pair.second);
5282       llvm::Value *NumDeps;
5283       LValue Base;
5284       std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc);
5285       NumOfDepobjElements =
5286           CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumDeps);
5287       Depobjs.emplace_back(NumDeps, Base);
5288       ++NumDepobjDependecies;
5289     }
5290 
5291     QualType KmpDependInfoArrayTy;
5292     // Define type kmp_depend_info[<Dependencies.size()>];
5293     // For depobj reserve one extra element to store the number of elements.
5294     // It is required to handle depobj(x) update(in) construct.
5295     // kmp_depend_info[<Dependencies.size()>] deps;
5296     if (ForDepobj) {
5297       assert(NumDepobjDependecies == 0 &&
5298              "depobj dependency kind is not expected in depobj directive.");
5299       KmpDependInfoArrayTy = C.getConstantArrayType(
5300           KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1),
5301           nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
5302       // Need to allocate on the dynamic memory.
5303       llvm::Value *ThreadID = getThreadID(CGF, Loc);
5304       // Use default allocator.
5305       llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5306       CharUnits Align = C.getTypeAlignInChars(KmpDependInfoArrayTy);
5307       CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy);
5308       llvm::Value *Size = CGF.CGM.getSize(Sz.alignTo(Align));
5309       llvm::Value *Args[] = {ThreadID, Size, Allocator};
5310 
5311       llvm::Value *Addr = CGF.EmitRuntimeCall(
5312           createRuntimeFunction(OMPRTL__kmpc_alloc), Args, ".dep.arr.addr");
5313       Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5314           Addr, CGF.ConvertTypeForMem(KmpDependInfoArrayTy)->getPointerTo());
5315       DependenciesArray = Address(Addr, Align);
5316       NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies,
5317                                              /*isSigned=*/false);
5318     } else if (NumDepobjDependecies > 0) {
5319       NumOfElements = CGF.Builder.CreateNUWAdd(
5320           NumOfDepobjElements,
5321           llvm::ConstantInt::get(CGM.IntPtrTy,
5322                                  NumDependencies - NumDepobjDependecies,
5323                                  /*isSigned=*/false));
5324       NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty,
5325                                                 /*isSigned=*/false);
5326       OpaqueValueExpr OVE(
5327           Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0),
5328           VK_RValue);
5329       CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE,
5330                                                     RValue::get(NumOfElements));
5331       KmpDependInfoArrayTy =
5332           C.getVariableArrayType(KmpDependInfoTy, &OVE, ArrayType::Normal,
5333                                  /*IndexTypeQuals=*/0, SourceRange(Loc, Loc));
5334       // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy);
5335       // Properly emit variable-sized array.
5336       auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy,
5337                                            ImplicitParamDecl::Other);
5338       CGF.EmitVarDecl(*PD);
5339       DependenciesArray = CGF.GetAddrOfLocalVar(PD);
5340     } else {
5341       KmpDependInfoArrayTy = C.getConstantArrayType(
5342           KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies),
5343           nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
5344       DependenciesArray =
5345           CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr");
5346       NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies,
5347                                              /*isSigned=*/false);
5348     }
5349     if (ForDepobj) {
5350       // Write number of elements in the first element of array for depobj.
5351       llvm::Value *NumVal =
5352           llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies);
5353       LValue Base = CGF.MakeAddrLValue(
5354           CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0),
5355           KmpDependInfoTy);
5356       // deps[i].base_addr = NumDependencies;
5357       LValue BaseAddrLVal = CGF.EmitLValueForField(
5358           Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5359       CGF.EmitStoreOfScalar(NumVal, BaseAddrLVal);
5360     }
5361     unsigned Pos = ForDepobj ? 1 : 0;
5362     for (unsigned I = 0; I < NumDependencies; ++I) {
5363       if (Dependencies[I].first == OMPC_DEPEND_depobj)
5364         continue;
5365       const Expr *E = Dependencies[I].second;
5366       const auto *OASE = dyn_cast<OMPArrayShapingExpr>(E);
5367       LValue Addr;
5368       if (OASE) {
5369         const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
5370         Addr =
5371             CGF.EmitLoadOfPointerLValue(CGF.EmitLValue(Base).getAddress(CGF),
5372                                         Base->getType()->castAs<PointerType>());
5373       } else {
5374         Addr = CGF.EmitLValue(E);
5375       }
5376       llvm::Value *Size;
5377       QualType Ty = E->getType();
5378       if (OASE) {
5379         Size = llvm::ConstantInt::get(CGF.SizeTy,/*V=*/1);
5380         for (const Expr *SE : OASE->getDimensions()) {
5381            llvm::Value *Sz = CGF.EmitScalarExpr(SE);
5382            Sz = CGF.EmitScalarConversion(Sz, SE->getType(),
5383                                     CGF.getContext().getSizeType(),
5384                                     SE->getExprLoc());
5385            Size = CGF.Builder.CreateNUWMul(Size, Sz);
5386         }
5387       } else if (const auto *ASE =
5388                      dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) {
5389         LValue UpAddrLVal =
5390             CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false);
5391         llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32(
5392             UpAddrLVal.getPointer(CGF), /*Idx0=*/1);
5393         llvm::Value *LowIntPtr =
5394             CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGM.SizeTy);
5395         llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy);
5396         Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr);
5397       } else {
5398         Size = CGF.getTypeSize(Ty);
5399       }
5400       LValue Base;
5401       if (NumDepobjDependecies > 0) {
5402         Base = CGF.MakeAddrLValue(
5403             CGF.Builder.CreateConstGEP(DependenciesArray, Pos),
5404             KmpDependInfoTy);
5405       } else {
5406         Base = CGF.MakeAddrLValue(
5407             CGF.Builder.CreateConstArrayGEP(DependenciesArray, Pos),
5408             KmpDependInfoTy);
5409       }
5410       // deps[i].base_addr = &<Dependencies[i].second>;
5411       LValue BaseAddrLVal = CGF.EmitLValueForField(
5412           Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5413       CGF.EmitStoreOfScalar(
5414           CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGF.IntPtrTy),
5415           BaseAddrLVal);
5416       // deps[i].len = sizeof(<Dependencies[i].second>);
5417       LValue LenLVal = CGF.EmitLValueForField(
5418           Base, *std::next(KmpDependInfoRD->field_begin(), Len));
5419       CGF.EmitStoreOfScalar(Size, LenLVal);
5420       // deps[i].flags = <Dependencies[i].first>;
5421       RTLDependenceKindTy DepKind =
5422           translateDependencyKind(Dependencies[I].first);
5423       LValue FlagsLVal = CGF.EmitLValueForField(
5424           Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
5425       CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
5426                             FlagsLVal);
5427       ++Pos;
5428     }
5429     // Copy final depobj arrays.
5430     if (NumDepobjDependecies > 0) {
5431       llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy);
5432       Address Addr = CGF.Builder.CreateConstGEP(DependenciesArray, Pos);
5433       for (const std::pair<llvm::Value *, LValue> &Pair : Depobjs) {
5434         llvm::Value *Size = CGF.Builder.CreateNUWMul(ElSize, Pair.first);
5435         CGF.Builder.CreateMemCpy(Addr, Pair.second.getAddress(CGF), Size);
5436         Addr =
5437             Address(CGF.Builder.CreateGEP(
5438                         Addr.getElementType(), Addr.getPointer(), Pair.first),
5439                     DependenciesArray.getAlignment().alignmentOfArrayElement(
5440                         C.getTypeSizeInChars(KmpDependInfoTy)));
5441       }
5442       DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5443           DependenciesArray, CGF.VoidPtrTy);
5444     } else {
5445       DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5446           CGF.Builder.CreateConstArrayGEP(DependenciesArray, ForDepobj ? 1 : 0),
5447           CGF.VoidPtrTy);
5448     }
5449   }
5450   return std::make_pair(NumOfElements, DependenciesArray);
5451 }
5452 
5453 void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal,
5454                                         SourceLocation Loc) {
5455   ASTContext &C = CGM.getContext();
5456   QualType FlagsTy;
5457   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5458   LValue Base = CGF.EmitLoadOfPointerLValue(
5459       DepobjLVal.getAddress(CGF),
5460       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
5461   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
5462   Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5463       Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy));
5464   llvm::Value *DepObjAddr = CGF.Builder.CreateGEP(
5465       Addr.getPointer(),
5466       llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
5467   DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr,
5468                                                                CGF.VoidPtrTy);
5469   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5470   // Use default allocator.
5471   llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5472   llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator};
5473 
5474   // _kmpc_free(gtid, addr, nullptr);
5475   (void)CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_free), Args);
5476 }
5477 
5478 void CGOpenMPRuntime::emitUpdateClause(CodeGenFunction &CGF, LValue DepobjLVal,
5479                                        OpenMPDependClauseKind NewDepKind,
5480                                        SourceLocation Loc) {
5481   ASTContext &C = CGM.getContext();
5482   QualType FlagsTy;
5483   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5484   RecordDecl *KmpDependInfoRD =
5485       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5486   llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
5487   llvm::Value *NumDeps;
5488   LValue Base;
5489   std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc);
5490 
5491   Address Begin = Base.getAddress(CGF);
5492   // Cast from pointer to array type to pointer to single element.
5493   llvm::Value *End = CGF.Builder.CreateGEP(Begin.getPointer(), NumDeps);
5494   // The basic structure here is a while-do loop.
5495   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body");
5496   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done");
5497   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5498   CGF.EmitBlock(BodyBB);
5499   llvm::PHINode *ElementPHI =
5500       CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast");
5501   ElementPHI->addIncoming(Begin.getPointer(), EntryBB);
5502   Begin = Address(ElementPHI, Begin.getAlignment());
5503   Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(),
5504                             Base.getTBAAInfo());
5505   // deps[i].flags = NewDepKind;
5506   RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind);
5507   LValue FlagsLVal = CGF.EmitLValueForField(
5508       Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
5509   CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
5510                         FlagsLVal);
5511 
5512   // Shift the address forward by one element.
5513   Address ElementNext =
5514       CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext");
5515   ElementPHI->addIncoming(ElementNext.getPointer(),
5516                           CGF.Builder.GetInsertBlock());
5517   llvm::Value *IsEmpty =
5518       CGF.Builder.CreateICmpEQ(ElementNext.getPointer(), End, "omp.isempty");
5519   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5520   // Done.
5521   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5522 }
5523 
5524 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
5525                                    const OMPExecutableDirective &D,
5526                                    llvm::Function *TaskFunction,
5527                                    QualType SharedsTy, Address Shareds,
5528                                    const Expr *IfCond,
5529                                    const OMPTaskDataTy &Data) {
5530   if (!CGF.HaveInsertPoint())
5531     return;
5532 
5533   TaskResultTy Result =
5534       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5535   llvm::Value *NewTask = Result.NewTask;
5536   llvm::Function *TaskEntry = Result.TaskEntry;
5537   llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy;
5538   LValue TDBase = Result.TDBase;
5539   const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD;
5540   // Process list of dependences.
5541   Address DependenciesArray = Address::invalid();
5542   llvm::Value *NumOfElements;
5543   std::tie(NumOfElements, DependenciesArray) =
5544       emitDependClause(CGF, Data.Dependences, /*ForDepobj=*/false, Loc);
5545 
5546   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5547   // libcall.
5548   // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
5549   // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
5550   // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence
5551   // list is not empty
5552   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5553   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5554   llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask };
5555   llvm::Value *DepTaskArgs[7];
5556   if (!Data.Dependences.empty()) {
5557     DepTaskArgs[0] = UpLoc;
5558     DepTaskArgs[1] = ThreadID;
5559     DepTaskArgs[2] = NewTask;
5560     DepTaskArgs[3] = NumOfElements;
5561     DepTaskArgs[4] = DependenciesArray.getPointer();
5562     DepTaskArgs[5] = CGF.Builder.getInt32(0);
5563     DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5564   }
5565   auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs,
5566                         &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) {
5567     if (!Data.Tied) {
5568       auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
5569       LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI);
5570       CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal);
5571     }
5572     if (!Data.Dependences.empty()) {
5573       CGF.EmitRuntimeCall(
5574           createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs);
5575     } else {
5576       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task),
5577                           TaskArgs);
5578     }
5579     // Check if parent region is untied and build return for untied task;
5580     if (auto *Region =
5581             dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
5582       Region->emitUntiedSwitch(CGF);
5583   };
5584 
5585   llvm::Value *DepWaitTaskArgs[6];
5586   if (!Data.Dependences.empty()) {
5587     DepWaitTaskArgs[0] = UpLoc;
5588     DepWaitTaskArgs[1] = ThreadID;
5589     DepWaitTaskArgs[2] = NumOfElements;
5590     DepWaitTaskArgs[3] = DependenciesArray.getPointer();
5591     DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
5592     DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5593   }
5594   auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry,
5595                         &Data, &DepWaitTaskArgs,
5596                         Loc](CodeGenFunction &CGF, PrePostActionTy &) {
5597     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5598     CodeGenFunction::RunCleanupsScope LocalScope(CGF);
5599     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
5600     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
5601     // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
5602     // is specified.
5603     if (!Data.Dependences.empty())
5604       CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps),
5605                           DepWaitTaskArgs);
5606     // Call proxy_task_entry(gtid, new_task);
5607     auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy,
5608                       Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
5609       Action.Enter(CGF);
5610       llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
5611       CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry,
5612                                                           OutlinedFnArgs);
5613     };
5614 
5615     // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
5616     // kmp_task_t *new_task);
5617     // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
5618     // kmp_task_t *new_task);
5619     RegionCodeGenTy RCG(CodeGen);
5620     CommonActionTy Action(
5621         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs,
5622         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs);
5623     RCG.setAction(Action);
5624     RCG(CGF);
5625   };
5626 
5627   if (IfCond) {
5628     emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen);
5629   } else {
5630     RegionCodeGenTy ThenRCG(ThenCodeGen);
5631     ThenRCG(CGF);
5632   }
5633 }
5634 
5635 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc,
5636                                        const OMPLoopDirective &D,
5637                                        llvm::Function *TaskFunction,
5638                                        QualType SharedsTy, Address Shareds,
5639                                        const Expr *IfCond,
5640                                        const OMPTaskDataTy &Data) {
5641   if (!CGF.HaveInsertPoint())
5642     return;
5643   TaskResultTy Result =
5644       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5645   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5646   // libcall.
5647   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
5648   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
5649   // sched, kmp_uint64 grainsize, void *task_dup);
5650   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5651   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5652   llvm::Value *IfVal;
5653   if (IfCond) {
5654     IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy,
5655                                       /*isSigned=*/true);
5656   } else {
5657     IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1);
5658   }
5659 
5660   LValue LBLVal = CGF.EmitLValueForField(
5661       Result.TDBase,
5662       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound));
5663   const auto *LBVar =
5664       cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl());
5665   CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF),
5666                        LBLVal.getQuals(),
5667                        /*IsInitializer=*/true);
5668   LValue UBLVal = CGF.EmitLValueForField(
5669       Result.TDBase,
5670       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound));
5671   const auto *UBVar =
5672       cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl());
5673   CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF),
5674                        UBLVal.getQuals(),
5675                        /*IsInitializer=*/true);
5676   LValue StLVal = CGF.EmitLValueForField(
5677       Result.TDBase,
5678       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride));
5679   const auto *StVar =
5680       cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl());
5681   CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF),
5682                        StLVal.getQuals(),
5683                        /*IsInitializer=*/true);
5684   // Store reductions address.
5685   LValue RedLVal = CGF.EmitLValueForField(
5686       Result.TDBase,
5687       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions));
5688   if (Data.Reductions) {
5689     CGF.EmitStoreOfScalar(Data.Reductions, RedLVal);
5690   } else {
5691     CGF.EmitNullInitialization(RedLVal.getAddress(CGF),
5692                                CGF.getContext().VoidPtrTy);
5693   }
5694   enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 };
5695   llvm::Value *TaskArgs[] = {
5696       UpLoc,
5697       ThreadID,
5698       Result.NewTask,
5699       IfVal,
5700       LBLVal.getPointer(CGF),
5701       UBLVal.getPointer(CGF),
5702       CGF.EmitLoadOfScalar(StLVal, Loc),
5703       llvm::ConstantInt::getSigned(
5704           CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler
5705       llvm::ConstantInt::getSigned(
5706           CGF.IntTy, Data.Schedule.getPointer()
5707                          ? Data.Schedule.getInt() ? NumTasks : Grainsize
5708                          : NoSchedule),
5709       Data.Schedule.getPointer()
5710           ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty,
5711                                       /*isSigned=*/false)
5712           : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0),
5713       Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5714                              Result.TaskDupFn, CGF.VoidPtrTy)
5715                        : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)};
5716   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs);
5717 }
5718 
5719 /// Emit reduction operation for each element of array (required for
5720 /// array sections) LHS op = RHS.
5721 /// \param Type Type of array.
5722 /// \param LHSVar Variable on the left side of the reduction operation
5723 /// (references element of array in original variable).
5724 /// \param RHSVar Variable on the right side of the reduction operation
5725 /// (references element of array in original variable).
5726 /// \param RedOpGen Generator of reduction operation with use of LHSVar and
5727 /// RHSVar.
5728 static void EmitOMPAggregateReduction(
5729     CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar,
5730     const VarDecl *RHSVar,
5731     const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *,
5732                                   const Expr *, const Expr *)> &RedOpGen,
5733     const Expr *XExpr = nullptr, const Expr *EExpr = nullptr,
5734     const Expr *UpExpr = nullptr) {
5735   // Perform element-by-element initialization.
5736   QualType ElementTy;
5737   Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar);
5738   Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar);
5739 
5740   // Drill down to the base element type on both arrays.
5741   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
5742   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr);
5743 
5744   llvm::Value *RHSBegin = RHSAddr.getPointer();
5745   llvm::Value *LHSBegin = LHSAddr.getPointer();
5746   // Cast from pointer to array type to pointer to single element.
5747   llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements);
5748   // The basic structure here is a while-do loop.
5749   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body");
5750   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done");
5751   llvm::Value *IsEmpty =
5752       CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty");
5753   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5754 
5755   // Enter the loop body, making that address the current address.
5756   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5757   CGF.EmitBlock(BodyBB);
5758 
5759   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
5760 
5761   llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI(
5762       RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast");
5763   RHSElementPHI->addIncoming(RHSBegin, EntryBB);
5764   Address RHSElementCurrent =
5765       Address(RHSElementPHI,
5766               RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5767 
5768   llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI(
5769       LHSBegin->getType(), 2, "omp.arraycpy.destElementPast");
5770   LHSElementPHI->addIncoming(LHSBegin, EntryBB);
5771   Address LHSElementCurrent =
5772       Address(LHSElementPHI,
5773               LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5774 
5775   // Emit copy.
5776   CodeGenFunction::OMPPrivateScope Scope(CGF);
5777   Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; });
5778   Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; });
5779   Scope.Privatize();
5780   RedOpGen(CGF, XExpr, EExpr, UpExpr);
5781   Scope.ForceCleanup();
5782 
5783   // Shift the address forward by one element.
5784   llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32(
5785       LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
5786   llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32(
5787       RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element");
5788   // Check whether we've reached the end.
5789   llvm::Value *Done =
5790       CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done");
5791   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
5792   LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock());
5793   RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock());
5794 
5795   // Done.
5796   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5797 }
5798 
5799 /// Emit reduction combiner. If the combiner is a simple expression emit it as
5800 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of
5801 /// UDR combiner function.
5802 static void emitReductionCombiner(CodeGenFunction &CGF,
5803                                   const Expr *ReductionOp) {
5804   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
5805     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
5806       if (const auto *DRE =
5807               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
5808         if (const auto *DRD =
5809                 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) {
5810           std::pair<llvm::Function *, llvm::Function *> Reduction =
5811               CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
5812           RValue Func = RValue::get(Reduction.first);
5813           CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
5814           CGF.EmitIgnoredExpr(ReductionOp);
5815           return;
5816         }
5817   CGF.EmitIgnoredExpr(ReductionOp);
5818 }
5819 
5820 llvm::Function *CGOpenMPRuntime::emitReductionFunction(
5821     SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates,
5822     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
5823     ArrayRef<const Expr *> ReductionOps) {
5824   ASTContext &C = CGM.getContext();
5825 
5826   // void reduction_func(void *LHSArg, void *RHSArg);
5827   FunctionArgList Args;
5828   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5829                            ImplicitParamDecl::Other);
5830   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5831                            ImplicitParamDecl::Other);
5832   Args.push_back(&LHSArg);
5833   Args.push_back(&RHSArg);
5834   const auto &CGFI =
5835       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5836   std::string Name = getName({"omp", "reduction", "reduction_func"});
5837   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
5838                                     llvm::GlobalValue::InternalLinkage, Name,
5839                                     &CGM.getModule());
5840   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
5841   Fn->setDoesNotRecurse();
5842   CodeGenFunction CGF(CGM);
5843   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
5844 
5845   // Dst = (void*[n])(LHSArg);
5846   // Src = (void*[n])(RHSArg);
5847   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5848       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
5849       ArgsType), CGF.getPointerAlign());
5850   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5851       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
5852       ArgsType), CGF.getPointerAlign());
5853 
5854   //  ...
5855   //  *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
5856   //  ...
5857   CodeGenFunction::OMPPrivateScope Scope(CGF);
5858   auto IPriv = Privates.begin();
5859   unsigned Idx = 0;
5860   for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) {
5861     const auto *RHSVar =
5862         cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl());
5863     Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() {
5864       return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar);
5865     });
5866     const auto *LHSVar =
5867         cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl());
5868     Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() {
5869       return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar);
5870     });
5871     QualType PrivTy = (*IPriv)->getType();
5872     if (PrivTy->isVariablyModifiedType()) {
5873       // Get array size and emit VLA type.
5874       ++Idx;
5875       Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx);
5876       llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem);
5877       const VariableArrayType *VLA =
5878           CGF.getContext().getAsVariableArrayType(PrivTy);
5879       const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr());
5880       CodeGenFunction::OpaqueValueMapping OpaqueMap(
5881           CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy)));
5882       CGF.EmitVariablyModifiedType(PrivTy);
5883     }
5884   }
5885   Scope.Privatize();
5886   IPriv = Privates.begin();
5887   auto ILHS = LHSExprs.begin();
5888   auto IRHS = RHSExprs.begin();
5889   for (const Expr *E : ReductionOps) {
5890     if ((*IPriv)->getType()->isArrayType()) {
5891       // Emit reduction for array section.
5892       const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5893       const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5894       EmitOMPAggregateReduction(
5895           CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5896           [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5897             emitReductionCombiner(CGF, E);
5898           });
5899     } else {
5900       // Emit reduction for array subscript or single variable.
5901       emitReductionCombiner(CGF, E);
5902     }
5903     ++IPriv;
5904     ++ILHS;
5905     ++IRHS;
5906   }
5907   Scope.ForceCleanup();
5908   CGF.FinishFunction();
5909   return Fn;
5910 }
5911 
5912 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF,
5913                                                   const Expr *ReductionOp,
5914                                                   const Expr *PrivateRef,
5915                                                   const DeclRefExpr *LHS,
5916                                                   const DeclRefExpr *RHS) {
5917   if (PrivateRef->getType()->isArrayType()) {
5918     // Emit reduction for array section.
5919     const auto *LHSVar = cast<VarDecl>(LHS->getDecl());
5920     const auto *RHSVar = cast<VarDecl>(RHS->getDecl());
5921     EmitOMPAggregateReduction(
5922         CGF, PrivateRef->getType(), LHSVar, RHSVar,
5923         [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5924           emitReductionCombiner(CGF, ReductionOp);
5925         });
5926   } else {
5927     // Emit reduction for array subscript or single variable.
5928     emitReductionCombiner(CGF, ReductionOp);
5929   }
5930 }
5931 
5932 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc,
5933                                     ArrayRef<const Expr *> Privates,
5934                                     ArrayRef<const Expr *> LHSExprs,
5935                                     ArrayRef<const Expr *> RHSExprs,
5936                                     ArrayRef<const Expr *> ReductionOps,
5937                                     ReductionOptionsTy Options) {
5938   if (!CGF.HaveInsertPoint())
5939     return;
5940 
5941   bool WithNowait = Options.WithNowait;
5942   bool SimpleReduction = Options.SimpleReduction;
5943 
5944   // Next code should be emitted for reduction:
5945   //
5946   // static kmp_critical_name lock = { 0 };
5947   //
5948   // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
5949   //  *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
5950   //  ...
5951   //  *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
5952   //  *(Type<n>-1*)rhs[<n>-1]);
5953   // }
5954   //
5955   // ...
5956   // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
5957   // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5958   // RedList, reduce_func, &<lock>)) {
5959   // case 1:
5960   //  ...
5961   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5962   //  ...
5963   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5964   // break;
5965   // case 2:
5966   //  ...
5967   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5968   //  ...
5969   // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
5970   // break;
5971   // default:;
5972   // }
5973   //
5974   // if SimpleReduction is true, only the next code is generated:
5975   //  ...
5976   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5977   //  ...
5978 
5979   ASTContext &C = CGM.getContext();
5980 
5981   if (SimpleReduction) {
5982     CodeGenFunction::RunCleanupsScope Scope(CGF);
5983     auto IPriv = Privates.begin();
5984     auto ILHS = LHSExprs.begin();
5985     auto IRHS = RHSExprs.begin();
5986     for (const Expr *E : ReductionOps) {
5987       emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5988                                   cast<DeclRefExpr>(*IRHS));
5989       ++IPriv;
5990       ++ILHS;
5991       ++IRHS;
5992     }
5993     return;
5994   }
5995 
5996   // 1. Build a list of reduction variables.
5997   // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
5998   auto Size = RHSExprs.size();
5999   for (const Expr *E : Privates) {
6000     if (E->getType()->isVariablyModifiedType())
6001       // Reserve place for array size.
6002       ++Size;
6003   }
6004   llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size);
6005   QualType ReductionArrayTy =
6006       C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
6007                              /*IndexTypeQuals=*/0);
6008   Address ReductionList =
6009       CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list");
6010   auto IPriv = Privates.begin();
6011   unsigned Idx = 0;
6012   for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) {
6013     Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
6014     CGF.Builder.CreateStore(
6015         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6016             CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy),
6017         Elem);
6018     if ((*IPriv)->getType()->isVariablyModifiedType()) {
6019       // Store array size.
6020       ++Idx;
6021       Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
6022       llvm::Value *Size = CGF.Builder.CreateIntCast(
6023           CGF.getVLASize(
6024                  CGF.getContext().getAsVariableArrayType((*IPriv)->getType()))
6025               .NumElts,
6026           CGF.SizeTy, /*isSigned=*/false);
6027       CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy),
6028                               Elem);
6029     }
6030   }
6031 
6032   // 2. Emit reduce_func().
6033   llvm::Function *ReductionFn = emitReductionFunction(
6034       Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates,
6035       LHSExprs, RHSExprs, ReductionOps);
6036 
6037   // 3. Create static kmp_critical_name lock = { 0 };
6038   std::string Name = getName({"reduction"});
6039   llvm::Value *Lock = getCriticalRegionLock(Name);
6040 
6041   // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
6042   // RedList, reduce_func, &<lock>);
6043   llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE);
6044   llvm::Value *ThreadId = getThreadID(CGF, Loc);
6045   llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy);
6046   llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6047       ReductionList.getPointer(), CGF.VoidPtrTy);
6048   llvm::Value *Args[] = {
6049       IdentTLoc,                             // ident_t *<loc>
6050       ThreadId,                              // i32 <gtid>
6051       CGF.Builder.getInt32(RHSExprs.size()), // i32 <n>
6052       ReductionArrayTySize,                  // size_type sizeof(RedList)
6053       RL,                                    // void *RedList
6054       ReductionFn, // void (*) (void *, void *) <reduce_func>
6055       Lock         // kmp_critical_name *&<lock>
6056   };
6057   llvm::Value *Res = CGF.EmitRuntimeCall(
6058       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait
6059                                        : OMPRTL__kmpc_reduce),
6060       Args);
6061 
6062   // 5. Build switch(res)
6063   llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default");
6064   llvm::SwitchInst *SwInst =
6065       CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2);
6066 
6067   // 6. Build case 1:
6068   //  ...
6069   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
6070   //  ...
6071   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
6072   // break;
6073   llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1");
6074   SwInst->addCase(CGF.Builder.getInt32(1), Case1BB);
6075   CGF.EmitBlock(Case1BB);
6076 
6077   // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
6078   llvm::Value *EndArgs[] = {
6079       IdentTLoc, // ident_t *<loc>
6080       ThreadId,  // i32 <gtid>
6081       Lock       // kmp_critical_name *&<lock>
6082   };
6083   auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps](
6084                        CodeGenFunction &CGF, PrePostActionTy &Action) {
6085     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6086     auto IPriv = Privates.begin();
6087     auto ILHS = LHSExprs.begin();
6088     auto IRHS = RHSExprs.begin();
6089     for (const Expr *E : ReductionOps) {
6090       RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
6091                                      cast<DeclRefExpr>(*IRHS));
6092       ++IPriv;
6093       ++ILHS;
6094       ++IRHS;
6095     }
6096   };
6097   RegionCodeGenTy RCG(CodeGen);
6098   CommonActionTy Action(
6099       nullptr, llvm::None,
6100       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait
6101                                        : OMPRTL__kmpc_end_reduce),
6102       EndArgs);
6103   RCG.setAction(Action);
6104   RCG(CGF);
6105 
6106   CGF.EmitBranch(DefaultBB);
6107 
6108   // 7. Build case 2:
6109   //  ...
6110   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
6111   //  ...
6112   // break;
6113   llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2");
6114   SwInst->addCase(CGF.Builder.getInt32(2), Case2BB);
6115   CGF.EmitBlock(Case2BB);
6116 
6117   auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps](
6118                              CodeGenFunction &CGF, PrePostActionTy &Action) {
6119     auto ILHS = LHSExprs.begin();
6120     auto IRHS = RHSExprs.begin();
6121     auto IPriv = Privates.begin();
6122     for (const Expr *E : ReductionOps) {
6123       const Expr *XExpr = nullptr;
6124       const Expr *EExpr = nullptr;
6125       const Expr *UpExpr = nullptr;
6126       BinaryOperatorKind BO = BO_Comma;
6127       if (const auto *BO = dyn_cast<BinaryOperator>(E)) {
6128         if (BO->getOpcode() == BO_Assign) {
6129           XExpr = BO->getLHS();
6130           UpExpr = BO->getRHS();
6131         }
6132       }
6133       // Try to emit update expression as a simple atomic.
6134       const Expr *RHSExpr = UpExpr;
6135       if (RHSExpr) {
6136         // Analyze RHS part of the whole expression.
6137         if (const auto *ACO = dyn_cast<AbstractConditionalOperator>(
6138                 RHSExpr->IgnoreParenImpCasts())) {
6139           // If this is a conditional operator, analyze its condition for
6140           // min/max reduction operator.
6141           RHSExpr = ACO->getCond();
6142         }
6143         if (const auto *BORHS =
6144                 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) {
6145           EExpr = BORHS->getRHS();
6146           BO = BORHS->getOpcode();
6147         }
6148       }
6149       if (XExpr) {
6150         const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
6151         auto &&AtomicRedGen = [BO, VD,
6152                                Loc](CodeGenFunction &CGF, const Expr *XExpr,
6153                                     const Expr *EExpr, const Expr *UpExpr) {
6154           LValue X = CGF.EmitLValue(XExpr);
6155           RValue E;
6156           if (EExpr)
6157             E = CGF.EmitAnyExpr(EExpr);
6158           CGF.EmitOMPAtomicSimpleUpdateExpr(
6159               X, E, BO, /*IsXLHSInRHSPart=*/true,
6160               llvm::AtomicOrdering::Monotonic, Loc,
6161               [&CGF, UpExpr, VD, Loc](RValue XRValue) {
6162                 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
6163                 PrivateScope.addPrivate(
6164                     VD, [&CGF, VD, XRValue, Loc]() {
6165                       Address LHSTemp = CGF.CreateMemTemp(VD->getType());
6166                       CGF.emitOMPSimpleStore(
6167                           CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue,
6168                           VD->getType().getNonReferenceType(), Loc);
6169                       return LHSTemp;
6170                     });
6171                 (void)PrivateScope.Privatize();
6172                 return CGF.EmitAnyExpr(UpExpr);
6173               });
6174         };
6175         if ((*IPriv)->getType()->isArrayType()) {
6176           // Emit atomic reduction for array section.
6177           const auto *RHSVar =
6178               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
6179           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar,
6180                                     AtomicRedGen, XExpr, EExpr, UpExpr);
6181         } else {
6182           // Emit atomic reduction for array subscript or single variable.
6183           AtomicRedGen(CGF, XExpr, EExpr, UpExpr);
6184         }
6185       } else {
6186         // Emit as a critical region.
6187         auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *,
6188                                            const Expr *, const Expr *) {
6189           CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6190           std::string Name = RT.getName({"atomic_reduction"});
6191           RT.emitCriticalRegion(
6192               CGF, Name,
6193               [=](CodeGenFunction &CGF, PrePostActionTy &Action) {
6194                 Action.Enter(CGF);
6195                 emitReductionCombiner(CGF, E);
6196               },
6197               Loc);
6198         };
6199         if ((*IPriv)->getType()->isArrayType()) {
6200           const auto *LHSVar =
6201               cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
6202           const auto *RHSVar =
6203               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
6204           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar,
6205                                     CritRedGen);
6206         } else {
6207           CritRedGen(CGF, nullptr, nullptr, nullptr);
6208         }
6209       }
6210       ++ILHS;
6211       ++IRHS;
6212       ++IPriv;
6213     }
6214   };
6215   RegionCodeGenTy AtomicRCG(AtomicCodeGen);
6216   if (!WithNowait) {
6217     // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
6218     llvm::Value *EndArgs[] = {
6219         IdentTLoc, // ident_t *<loc>
6220         ThreadId,  // i32 <gtid>
6221         Lock       // kmp_critical_name *&<lock>
6222     };
6223     CommonActionTy Action(nullptr, llvm::None,
6224                           createRuntimeFunction(OMPRTL__kmpc_end_reduce),
6225                           EndArgs);
6226     AtomicRCG.setAction(Action);
6227     AtomicRCG(CGF);
6228   } else {
6229     AtomicRCG(CGF);
6230   }
6231 
6232   CGF.EmitBranch(DefaultBB);
6233   CGF.EmitBlock(DefaultBB, /*IsFinished=*/true);
6234 }
6235 
6236 /// Generates unique name for artificial threadprivate variables.
6237 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>"
6238 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix,
6239                                       const Expr *Ref) {
6240   SmallString<256> Buffer;
6241   llvm::raw_svector_ostream Out(Buffer);
6242   const clang::DeclRefExpr *DE;
6243   const VarDecl *D = ::getBaseDecl(Ref, DE);
6244   if (!D)
6245     D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl());
6246   D = D->getCanonicalDecl();
6247   std::string Name = CGM.getOpenMPRuntime().getName(
6248       {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)});
6249   Out << Prefix << Name << "_"
6250       << D->getCanonicalDecl()->getBeginLoc().getRawEncoding();
6251   return std::string(Out.str());
6252 }
6253 
6254 /// Emits reduction initializer function:
6255 /// \code
6256 /// void @.red_init(void* %arg) {
6257 /// %0 = bitcast void* %arg to <type>*
6258 /// store <type> <init>, <type>* %0
6259 /// ret void
6260 /// }
6261 /// \endcode
6262 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM,
6263                                            SourceLocation Loc,
6264                                            ReductionCodeGen &RCG, unsigned N) {
6265   ASTContext &C = CGM.getContext();
6266   FunctionArgList Args;
6267   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6268                           ImplicitParamDecl::Other);
6269   Args.emplace_back(&Param);
6270   const auto &FnInfo =
6271       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6272   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6273   std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""});
6274   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6275                                     Name, &CGM.getModule());
6276   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6277   Fn->setDoesNotRecurse();
6278   CodeGenFunction CGF(CGM);
6279   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6280   Address PrivateAddr = CGF.EmitLoadOfPointer(
6281       CGF.GetAddrOfLocalVar(&Param),
6282       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6283   llvm::Value *Size = nullptr;
6284   // If the size of the reduction item is non-constant, load it from global
6285   // threadprivate variable.
6286   if (RCG.getSizes(N).second) {
6287     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6288         CGF, CGM.getContext().getSizeType(),
6289         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6290     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6291                                 CGM.getContext().getSizeType(), Loc);
6292   }
6293   RCG.emitAggregateType(CGF, N, Size);
6294   LValue SharedLVal;
6295   // If initializer uses initializer from declare reduction construct, emit a
6296   // pointer to the address of the original reduction item (reuired by reduction
6297   // initializer)
6298   if (RCG.usesReductionInitializer(N)) {
6299     Address SharedAddr =
6300         CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6301             CGF, CGM.getContext().VoidPtrTy,
6302             generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6303     SharedAddr = CGF.EmitLoadOfPointer(
6304         SharedAddr,
6305         CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr());
6306     SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy);
6307   } else {
6308     SharedLVal = CGF.MakeNaturalAlignAddrLValue(
6309         llvm::ConstantPointerNull::get(CGM.VoidPtrTy),
6310         CGM.getContext().VoidPtrTy);
6311   }
6312   // Emit the initializer:
6313   // %0 = bitcast void* %arg to <type>*
6314   // store <type> <init>, <type>* %0
6315   RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal,
6316                          [](CodeGenFunction &) { return false; });
6317   CGF.FinishFunction();
6318   return Fn;
6319 }
6320 
6321 /// Emits reduction combiner function:
6322 /// \code
6323 /// void @.red_comb(void* %arg0, void* %arg1) {
6324 /// %lhs = bitcast void* %arg0 to <type>*
6325 /// %rhs = bitcast void* %arg1 to <type>*
6326 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs)
6327 /// store <type> %2, <type>* %lhs
6328 /// ret void
6329 /// }
6330 /// \endcode
6331 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM,
6332                                            SourceLocation Loc,
6333                                            ReductionCodeGen &RCG, unsigned N,
6334                                            const Expr *ReductionOp,
6335                                            const Expr *LHS, const Expr *RHS,
6336                                            const Expr *PrivateRef) {
6337   ASTContext &C = CGM.getContext();
6338   const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl());
6339   const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl());
6340   FunctionArgList Args;
6341   ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
6342                                C.VoidPtrTy, ImplicitParamDecl::Other);
6343   ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6344                             ImplicitParamDecl::Other);
6345   Args.emplace_back(&ParamInOut);
6346   Args.emplace_back(&ParamIn);
6347   const auto &FnInfo =
6348       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6349   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6350   std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""});
6351   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6352                                     Name, &CGM.getModule());
6353   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6354   Fn->setDoesNotRecurse();
6355   CodeGenFunction CGF(CGM);
6356   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6357   llvm::Value *Size = nullptr;
6358   // If the size of the reduction item is non-constant, load it from global
6359   // threadprivate variable.
6360   if (RCG.getSizes(N).second) {
6361     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6362         CGF, CGM.getContext().getSizeType(),
6363         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6364     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6365                                 CGM.getContext().getSizeType(), Loc);
6366   }
6367   RCG.emitAggregateType(CGF, N, Size);
6368   // Remap lhs and rhs variables to the addresses of the function arguments.
6369   // %lhs = bitcast void* %arg0 to <type>*
6370   // %rhs = bitcast void* %arg1 to <type>*
6371   CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
6372   PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() {
6373     // Pull out the pointer to the variable.
6374     Address PtrAddr = CGF.EmitLoadOfPointer(
6375         CGF.GetAddrOfLocalVar(&ParamInOut),
6376         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6377     return CGF.Builder.CreateElementBitCast(
6378         PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType()));
6379   });
6380   PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() {
6381     // Pull out the pointer to the variable.
6382     Address PtrAddr = CGF.EmitLoadOfPointer(
6383         CGF.GetAddrOfLocalVar(&ParamIn),
6384         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6385     return CGF.Builder.CreateElementBitCast(
6386         PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType()));
6387   });
6388   PrivateScope.Privatize();
6389   // Emit the combiner body:
6390   // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs)
6391   // store <type> %2, <type>* %lhs
6392   CGM.getOpenMPRuntime().emitSingleReductionCombiner(
6393       CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS),
6394       cast<DeclRefExpr>(RHS));
6395   CGF.FinishFunction();
6396   return Fn;
6397 }
6398 
6399 /// Emits reduction finalizer function:
6400 /// \code
6401 /// void @.red_fini(void* %arg) {
6402 /// %0 = bitcast void* %arg to <type>*
6403 /// <destroy>(<type>* %0)
6404 /// ret void
6405 /// }
6406 /// \endcode
6407 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM,
6408                                            SourceLocation Loc,
6409                                            ReductionCodeGen &RCG, unsigned N) {
6410   if (!RCG.needCleanups(N))
6411     return nullptr;
6412   ASTContext &C = CGM.getContext();
6413   FunctionArgList Args;
6414   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6415                           ImplicitParamDecl::Other);
6416   Args.emplace_back(&Param);
6417   const auto &FnInfo =
6418       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6419   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6420   std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""});
6421   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6422                                     Name, &CGM.getModule());
6423   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6424   Fn->setDoesNotRecurse();
6425   CodeGenFunction CGF(CGM);
6426   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6427   Address PrivateAddr = CGF.EmitLoadOfPointer(
6428       CGF.GetAddrOfLocalVar(&Param),
6429       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6430   llvm::Value *Size = nullptr;
6431   // If the size of the reduction item is non-constant, load it from global
6432   // threadprivate variable.
6433   if (RCG.getSizes(N).second) {
6434     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6435         CGF, CGM.getContext().getSizeType(),
6436         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6437     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6438                                 CGM.getContext().getSizeType(), Loc);
6439   }
6440   RCG.emitAggregateType(CGF, N, Size);
6441   // Emit the finalizer body:
6442   // <destroy>(<type>* %0)
6443   RCG.emitCleanups(CGF, N, PrivateAddr);
6444   CGF.FinishFunction(Loc);
6445   return Fn;
6446 }
6447 
6448 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit(
6449     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
6450     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
6451   if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty())
6452     return nullptr;
6453 
6454   // Build typedef struct:
6455   // kmp_task_red_input {
6456   //   void *reduce_shar; // shared reduction item
6457   //   size_t reduce_size; // size of data item
6458   //   void *reduce_init; // data initialization routine
6459   //   void *reduce_fini; // data finalization routine
6460   //   void *reduce_comb; // data combiner routine
6461   //   kmp_task_red_flags_t flags; // flags for additional info from compiler
6462   // } kmp_task_red_input_t;
6463   ASTContext &C = CGM.getContext();
6464   RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t");
6465   RD->startDefinition();
6466   const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6467   const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType());
6468   const FieldDecl *InitFD  = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6469   const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6470   const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6471   const FieldDecl *FlagsFD = addFieldToRecordDecl(
6472       C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false));
6473   RD->completeDefinition();
6474   QualType RDType = C.getRecordType(RD);
6475   unsigned Size = Data.ReductionVars.size();
6476   llvm::APInt ArraySize(/*numBits=*/64, Size);
6477   QualType ArrayRDType = C.getConstantArrayType(
6478       RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
6479   // kmp_task_red_input_t .rd_input.[Size];
6480   Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input.");
6481   ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies,
6482                        Data.ReductionOps);
6483   for (unsigned Cnt = 0; Cnt < Size; ++Cnt) {
6484     // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt];
6485     llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0),
6486                            llvm::ConstantInt::get(CGM.SizeTy, Cnt)};
6487     llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP(
6488         TaskRedInput.getPointer(), Idxs,
6489         /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc,
6490         ".rd_input.gep.");
6491     LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType);
6492     // ElemLVal.reduce_shar = &Shareds[Cnt];
6493     LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD);
6494     RCG.emitSharedLValue(CGF, Cnt);
6495     llvm::Value *CastedShared =
6496         CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF));
6497     CGF.EmitStoreOfScalar(CastedShared, SharedLVal);
6498     RCG.emitAggregateType(CGF, Cnt);
6499     llvm::Value *SizeValInChars;
6500     llvm::Value *SizeVal;
6501     std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt);
6502     // We use delayed creation/initialization for VLAs, array sections and
6503     // custom reduction initializations. It is required because runtime does not
6504     // provide the way to pass the sizes of VLAs/array sections to
6505     // initializer/combiner/finalizer functions and does not pass the pointer to
6506     // original reduction item to the initializer. Instead threadprivate global
6507     // variables are used to store these values and use them in the functions.
6508     bool DelayedCreation = !!SizeVal;
6509     SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy,
6510                                                /*isSigned=*/false);
6511     LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD);
6512     CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal);
6513     // ElemLVal.reduce_init = init;
6514     LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD);
6515     llvm::Value *InitAddr =
6516         CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt));
6517     CGF.EmitStoreOfScalar(InitAddr, InitLVal);
6518     DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt);
6519     // ElemLVal.reduce_fini = fini;
6520     LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD);
6521     llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt);
6522     llvm::Value *FiniAddr = Fini
6523                                 ? CGF.EmitCastToVoidPtr(Fini)
6524                                 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy);
6525     CGF.EmitStoreOfScalar(FiniAddr, FiniLVal);
6526     // ElemLVal.reduce_comb = comb;
6527     LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD);
6528     llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction(
6529         CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt],
6530         RHSExprs[Cnt], Data.ReductionCopies[Cnt]));
6531     CGF.EmitStoreOfScalar(CombAddr, CombLVal);
6532     // ElemLVal.flags = 0;
6533     LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD);
6534     if (DelayedCreation) {
6535       CGF.EmitStoreOfScalar(
6536           llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true),
6537           FlagsLVal);
6538     } else
6539       CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF),
6540                                  FlagsLVal.getType());
6541   }
6542   // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void
6543   // *data);
6544   llvm::Value *Args[] = {
6545       CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy,
6546                                 /*isSigned=*/true),
6547       llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6548       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(),
6549                                                       CGM.VoidPtrTy)};
6550   return CGF.EmitRuntimeCall(
6551       createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args);
6552 }
6553 
6554 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
6555                                               SourceLocation Loc,
6556                                               ReductionCodeGen &RCG,
6557                                               unsigned N) {
6558   auto Sizes = RCG.getSizes(N);
6559   // Emit threadprivate global variable if the type is non-constant
6560   // (Sizes.second = nullptr).
6561   if (Sizes.second) {
6562     llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy,
6563                                                      /*isSigned=*/false);
6564     Address SizeAddr = getAddrOfArtificialThreadPrivate(
6565         CGF, CGM.getContext().getSizeType(),
6566         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6567     CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false);
6568   }
6569   // Store address of the original reduction item if custom initializer is used.
6570   if (RCG.usesReductionInitializer(N)) {
6571     Address SharedAddr = getAddrOfArtificialThreadPrivate(
6572         CGF, CGM.getContext().VoidPtrTy,
6573         generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6574     CGF.Builder.CreateStore(
6575         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6576             RCG.getSharedLValue(N).getPointer(CGF), CGM.VoidPtrTy),
6577         SharedAddr, /*IsVolatile=*/false);
6578   }
6579 }
6580 
6581 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF,
6582                                               SourceLocation Loc,
6583                                               llvm::Value *ReductionsPtr,
6584                                               LValue SharedLVal) {
6585   // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
6586   // *d);
6587   llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6588                                                    CGM.IntTy,
6589                                                    /*isSigned=*/true),
6590                          ReductionsPtr,
6591                          CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6592                              SharedLVal.getPointer(CGF), CGM.VoidPtrTy)};
6593   return Address(
6594       CGF.EmitRuntimeCall(
6595           createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args),
6596       SharedLVal.getAlignment());
6597 }
6598 
6599 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
6600                                        SourceLocation Loc) {
6601   if (!CGF.HaveInsertPoint())
6602     return;
6603 
6604   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
6605   if (OMPBuilder) {
6606     OMPBuilder->CreateTaskwait(CGF.Builder);
6607   } else {
6608     // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
6609     // global_tid);
6610     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
6611     // Ignore return result until untied tasks are supported.
6612     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args);
6613   }
6614 
6615   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
6616     Region->emitUntiedSwitch(CGF);
6617 }
6618 
6619 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF,
6620                                            OpenMPDirectiveKind InnerKind,
6621                                            const RegionCodeGenTy &CodeGen,
6622                                            bool HasCancel) {
6623   if (!CGF.HaveInsertPoint())
6624     return;
6625   InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel);
6626   CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
6627 }
6628 
6629 namespace {
6630 enum RTCancelKind {
6631   CancelNoreq = 0,
6632   CancelParallel = 1,
6633   CancelLoop = 2,
6634   CancelSections = 3,
6635   CancelTaskgroup = 4
6636 };
6637 } // anonymous namespace
6638 
6639 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) {
6640   RTCancelKind CancelKind = CancelNoreq;
6641   if (CancelRegion == OMPD_parallel)
6642     CancelKind = CancelParallel;
6643   else if (CancelRegion == OMPD_for)
6644     CancelKind = CancelLoop;
6645   else if (CancelRegion == OMPD_sections)
6646     CancelKind = CancelSections;
6647   else {
6648     assert(CancelRegion == OMPD_taskgroup);
6649     CancelKind = CancelTaskgroup;
6650   }
6651   return CancelKind;
6652 }
6653 
6654 void CGOpenMPRuntime::emitCancellationPointCall(
6655     CodeGenFunction &CGF, SourceLocation Loc,
6656     OpenMPDirectiveKind CancelRegion) {
6657   if (!CGF.HaveInsertPoint())
6658     return;
6659   // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
6660   // global_tid, kmp_int32 cncl_kind);
6661   if (auto *OMPRegionInfo =
6662           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6663     // For 'cancellation point taskgroup', the task region info may not have a
6664     // cancel. This may instead happen in another adjacent task.
6665     if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) {
6666       llvm::Value *Args[] = {
6667           emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
6668           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6669       // Ignore return result until untied tasks are supported.
6670       llvm::Value *Result = CGF.EmitRuntimeCall(
6671           createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args);
6672       // if (__kmpc_cancellationpoint()) {
6673       //   exit from construct;
6674       // }
6675       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6676       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6677       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6678       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6679       CGF.EmitBlock(ExitBB);
6680       // exit from construct;
6681       CodeGenFunction::JumpDest CancelDest =
6682           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6683       CGF.EmitBranchThroughCleanup(CancelDest);
6684       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6685     }
6686   }
6687 }
6688 
6689 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc,
6690                                      const Expr *IfCond,
6691                                      OpenMPDirectiveKind CancelRegion) {
6692   if (!CGF.HaveInsertPoint())
6693     return;
6694   // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
6695   // kmp_int32 cncl_kind);
6696   if (auto *OMPRegionInfo =
6697           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6698     auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF,
6699                                                         PrePostActionTy &) {
6700       CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6701       llvm::Value *Args[] = {
6702           RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc),
6703           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6704       // Ignore return result until untied tasks are supported.
6705       llvm::Value *Result = CGF.EmitRuntimeCall(
6706           RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args);
6707       // if (__kmpc_cancel()) {
6708       //   exit from construct;
6709       // }
6710       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6711       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6712       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6713       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6714       CGF.EmitBlock(ExitBB);
6715       // exit from construct;
6716       CodeGenFunction::JumpDest CancelDest =
6717           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6718       CGF.EmitBranchThroughCleanup(CancelDest);
6719       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6720     };
6721     if (IfCond) {
6722       emitIfClause(CGF, IfCond, ThenGen,
6723                    [](CodeGenFunction &, PrePostActionTy &) {});
6724     } else {
6725       RegionCodeGenTy ThenRCG(ThenGen);
6726       ThenRCG(CGF);
6727     }
6728   }
6729 }
6730 
6731 void CGOpenMPRuntime::emitTargetOutlinedFunction(
6732     const OMPExecutableDirective &D, StringRef ParentName,
6733     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6734     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6735   assert(!ParentName.empty() && "Invalid target region parent name!");
6736   HasEmittedTargetRegion = true;
6737   emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID,
6738                                    IsOffloadEntry, CodeGen);
6739 }
6740 
6741 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper(
6742     const OMPExecutableDirective &D, StringRef ParentName,
6743     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6744     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6745   // Create a unique name for the entry function using the source location
6746   // information of the current target region. The name will be something like:
6747   //
6748   // __omp_offloading_DD_FFFF_PP_lBB
6749   //
6750   // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the
6751   // mangled name of the function that encloses the target region and BB is the
6752   // line number of the target region.
6753 
6754   unsigned DeviceID;
6755   unsigned FileID;
6756   unsigned Line;
6757   getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID,
6758                            Line);
6759   SmallString<64> EntryFnName;
6760   {
6761     llvm::raw_svector_ostream OS(EntryFnName);
6762     OS << "__omp_offloading" << llvm::format("_%x", DeviceID)
6763        << llvm::format("_%x_", FileID) << ParentName << "_l" << Line;
6764   }
6765 
6766   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
6767 
6768   CodeGenFunction CGF(CGM, true);
6769   CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName);
6770   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6771 
6772   OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc());
6773 
6774   // If this target outline function is not an offload entry, we don't need to
6775   // register it.
6776   if (!IsOffloadEntry)
6777     return;
6778 
6779   // The target region ID is used by the runtime library to identify the current
6780   // target region, so it only has to be unique and not necessarily point to
6781   // anything. It could be the pointer to the outlined function that implements
6782   // the target region, but we aren't using that so that the compiler doesn't
6783   // need to keep that, and could therefore inline the host function if proven
6784   // worthwhile during optimization. In the other hand, if emitting code for the
6785   // device, the ID has to be the function address so that it can retrieved from
6786   // the offloading entry and launched by the runtime library. We also mark the
6787   // outlined function to have external linkage in case we are emitting code for
6788   // the device, because these functions will be entry points to the device.
6789 
6790   if (CGM.getLangOpts().OpenMPIsDevice) {
6791     OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy);
6792     OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
6793     OutlinedFn->setDSOLocal(false);
6794   } else {
6795     std::string Name = getName({EntryFnName, "region_id"});
6796     OutlinedFnID = new llvm::GlobalVariable(
6797         CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
6798         llvm::GlobalValue::WeakAnyLinkage,
6799         llvm::Constant::getNullValue(CGM.Int8Ty), Name);
6800   }
6801 
6802   // Register the information for the entry associated with this target region.
6803   OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
6804       DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID,
6805       OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion);
6806 }
6807 
6808 /// Checks if the expression is constant or does not have non-trivial function
6809 /// calls.
6810 static bool isTrivial(ASTContext &Ctx, const Expr * E) {
6811   // We can skip constant expressions.
6812   // We can skip expressions with trivial calls or simple expressions.
6813   return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) ||
6814           !E->hasNonTrivialCall(Ctx)) &&
6815          !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true);
6816 }
6817 
6818 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx,
6819                                                     const Stmt *Body) {
6820   const Stmt *Child = Body->IgnoreContainers();
6821   while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) {
6822     Child = nullptr;
6823     for (const Stmt *S : C->body()) {
6824       if (const auto *E = dyn_cast<Expr>(S)) {
6825         if (isTrivial(Ctx, E))
6826           continue;
6827       }
6828       // Some of the statements can be ignored.
6829       if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) ||
6830           isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S))
6831         continue;
6832       // Analyze declarations.
6833       if (const auto *DS = dyn_cast<DeclStmt>(S)) {
6834         if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) {
6835               if (isa<EmptyDecl>(D) || isa<DeclContext>(D) ||
6836                   isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) ||
6837                   isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) ||
6838                   isa<UsingDirectiveDecl>(D) ||
6839                   isa<OMPDeclareReductionDecl>(D) ||
6840                   isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D))
6841                 return true;
6842               const auto *VD = dyn_cast<VarDecl>(D);
6843               if (!VD)
6844                 return false;
6845               return VD->isConstexpr() ||
6846                      ((VD->getType().isTrivialType(Ctx) ||
6847                        VD->getType()->isReferenceType()) &&
6848                       (!VD->hasInit() || isTrivial(Ctx, VD->getInit())));
6849             }))
6850           continue;
6851       }
6852       // Found multiple children - cannot get the one child only.
6853       if (Child)
6854         return nullptr;
6855       Child = S;
6856     }
6857     if (Child)
6858       Child = Child->IgnoreContainers();
6859   }
6860   return Child;
6861 }
6862 
6863 /// Emit the number of teams for a target directive.  Inspect the num_teams
6864 /// clause associated with a teams construct combined or closely nested
6865 /// with the target directive.
6866 ///
6867 /// Emit a team of size one for directives such as 'target parallel' that
6868 /// have no associated teams construct.
6869 ///
6870 /// Otherwise, return nullptr.
6871 static llvm::Value *
6872 emitNumTeamsForTargetDirective(CodeGenFunction &CGF,
6873                                const OMPExecutableDirective &D) {
6874   assert(!CGF.getLangOpts().OpenMPIsDevice &&
6875          "Clauses associated with the teams directive expected to be emitted "
6876          "only for the host!");
6877   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6878   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6879          "Expected target-based executable directive.");
6880   CGBuilderTy &Bld = CGF.Builder;
6881   switch (DirectiveKind) {
6882   case OMPD_target: {
6883     const auto *CS = D.getInnermostCapturedStmt();
6884     const auto *Body =
6885         CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
6886     const Stmt *ChildStmt =
6887         CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body);
6888     if (const auto *NestedDir =
6889             dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
6890       if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) {
6891         if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) {
6892           CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6893           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6894           const Expr *NumTeams =
6895               NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6896           llvm::Value *NumTeamsVal =
6897               CGF.EmitScalarExpr(NumTeams,
6898                                  /*IgnoreResultAssign*/ true);
6899           return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6900                                    /*isSigned=*/true);
6901         }
6902         return Bld.getInt32(0);
6903       }
6904       if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) ||
6905           isOpenMPSimdDirective(NestedDir->getDirectiveKind()))
6906         return Bld.getInt32(1);
6907       return Bld.getInt32(0);
6908     }
6909     return nullptr;
6910   }
6911   case OMPD_target_teams:
6912   case OMPD_target_teams_distribute:
6913   case OMPD_target_teams_distribute_simd:
6914   case OMPD_target_teams_distribute_parallel_for:
6915   case OMPD_target_teams_distribute_parallel_for_simd: {
6916     if (D.hasClausesOfKind<OMPNumTeamsClause>()) {
6917       CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF);
6918       const Expr *NumTeams =
6919           D.getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6920       llvm::Value *NumTeamsVal =
6921           CGF.EmitScalarExpr(NumTeams,
6922                              /*IgnoreResultAssign*/ true);
6923       return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6924                                /*isSigned=*/true);
6925     }
6926     return Bld.getInt32(0);
6927   }
6928   case OMPD_target_parallel:
6929   case OMPD_target_parallel_for:
6930   case OMPD_target_parallel_for_simd:
6931   case OMPD_target_simd:
6932     return Bld.getInt32(1);
6933   case OMPD_parallel:
6934   case OMPD_for:
6935   case OMPD_parallel_for:
6936   case OMPD_parallel_master:
6937   case OMPD_parallel_sections:
6938   case OMPD_for_simd:
6939   case OMPD_parallel_for_simd:
6940   case OMPD_cancel:
6941   case OMPD_cancellation_point:
6942   case OMPD_ordered:
6943   case OMPD_threadprivate:
6944   case OMPD_allocate:
6945   case OMPD_task:
6946   case OMPD_simd:
6947   case OMPD_sections:
6948   case OMPD_section:
6949   case OMPD_single:
6950   case OMPD_master:
6951   case OMPD_critical:
6952   case OMPD_taskyield:
6953   case OMPD_barrier:
6954   case OMPD_taskwait:
6955   case OMPD_taskgroup:
6956   case OMPD_atomic:
6957   case OMPD_flush:
6958   case OMPD_depobj:
6959   case OMPD_scan:
6960   case OMPD_teams:
6961   case OMPD_target_data:
6962   case OMPD_target_exit_data:
6963   case OMPD_target_enter_data:
6964   case OMPD_distribute:
6965   case OMPD_distribute_simd:
6966   case OMPD_distribute_parallel_for:
6967   case OMPD_distribute_parallel_for_simd:
6968   case OMPD_teams_distribute:
6969   case OMPD_teams_distribute_simd:
6970   case OMPD_teams_distribute_parallel_for:
6971   case OMPD_teams_distribute_parallel_for_simd:
6972   case OMPD_target_update:
6973   case OMPD_declare_simd:
6974   case OMPD_declare_variant:
6975   case OMPD_begin_declare_variant:
6976   case OMPD_end_declare_variant:
6977   case OMPD_declare_target:
6978   case OMPD_end_declare_target:
6979   case OMPD_declare_reduction:
6980   case OMPD_declare_mapper:
6981   case OMPD_taskloop:
6982   case OMPD_taskloop_simd:
6983   case OMPD_master_taskloop:
6984   case OMPD_master_taskloop_simd:
6985   case OMPD_parallel_master_taskloop:
6986   case OMPD_parallel_master_taskloop_simd:
6987   case OMPD_requires:
6988   case OMPD_unknown:
6989     break;
6990   }
6991   llvm_unreachable("Unexpected directive kind.");
6992 }
6993 
6994 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS,
6995                                   llvm::Value *DefaultThreadLimitVal) {
6996   const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6997       CGF.getContext(), CS->getCapturedStmt());
6998   if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6999     if (isOpenMPParallelDirective(Dir->getDirectiveKind())) {
7000       llvm::Value *NumThreads = nullptr;
7001       llvm::Value *CondVal = nullptr;
7002       // Handle if clause. If if clause present, the number of threads is
7003       // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
7004       if (Dir->hasClausesOfKind<OMPIfClause>()) {
7005         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
7006         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
7007         const OMPIfClause *IfClause = nullptr;
7008         for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) {
7009           if (C->getNameModifier() == OMPD_unknown ||
7010               C->getNameModifier() == OMPD_parallel) {
7011             IfClause = C;
7012             break;
7013           }
7014         }
7015         if (IfClause) {
7016           const Expr *Cond = IfClause->getCondition();
7017           bool Result;
7018           if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
7019             if (!Result)
7020               return CGF.Builder.getInt32(1);
7021           } else {
7022             CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange());
7023             if (const auto *PreInit =
7024                     cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) {
7025               for (const auto *I : PreInit->decls()) {
7026                 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7027                   CGF.EmitVarDecl(cast<VarDecl>(*I));
7028                 } else {
7029                   CodeGenFunction::AutoVarEmission Emission =
7030                       CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
7031                   CGF.EmitAutoVarCleanups(Emission);
7032                 }
7033               }
7034             }
7035             CondVal = CGF.EvaluateExprAsBool(Cond);
7036           }
7037         }
7038       }
7039       // Check the value of num_threads clause iff if clause was not specified
7040       // or is not evaluated to false.
7041       if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) {
7042         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
7043         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
7044         const auto *NumThreadsClause =
7045             Dir->getSingleClause<OMPNumThreadsClause>();
7046         CodeGenFunction::LexicalScope Scope(
7047             CGF, NumThreadsClause->getNumThreads()->getSourceRange());
7048         if (const auto *PreInit =
7049                 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) {
7050           for (const auto *I : PreInit->decls()) {
7051             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7052               CGF.EmitVarDecl(cast<VarDecl>(*I));
7053             } else {
7054               CodeGenFunction::AutoVarEmission Emission =
7055                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
7056               CGF.EmitAutoVarCleanups(Emission);
7057             }
7058           }
7059         }
7060         NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads());
7061         NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty,
7062                                                /*isSigned=*/false);
7063         if (DefaultThreadLimitVal)
7064           NumThreads = CGF.Builder.CreateSelect(
7065               CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads),
7066               DefaultThreadLimitVal, NumThreads);
7067       } else {
7068         NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal
7069                                            : CGF.Builder.getInt32(0);
7070       }
7071       // Process condition of the if clause.
7072       if (CondVal) {
7073         NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads,
7074                                               CGF.Builder.getInt32(1));
7075       }
7076       return NumThreads;
7077     }
7078     if (isOpenMPSimdDirective(Dir->getDirectiveKind()))
7079       return CGF.Builder.getInt32(1);
7080     return DefaultThreadLimitVal;
7081   }
7082   return DefaultThreadLimitVal ? DefaultThreadLimitVal
7083                                : CGF.Builder.getInt32(0);
7084 }
7085 
7086 /// Emit the number of threads for a target directive.  Inspect the
7087 /// thread_limit clause associated with a teams construct combined or closely
7088 /// nested with the target directive.
7089 ///
7090 /// Emit the num_threads clause for directives such as 'target parallel' that
7091 /// have no associated teams construct.
7092 ///
7093 /// Otherwise, return nullptr.
7094 static llvm::Value *
7095 emitNumThreadsForTargetDirective(CodeGenFunction &CGF,
7096                                  const OMPExecutableDirective &D) {
7097   assert(!CGF.getLangOpts().OpenMPIsDevice &&
7098          "Clauses associated with the teams directive expected to be emitted "
7099          "only for the host!");
7100   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
7101   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
7102          "Expected target-based executable directive.");
7103   CGBuilderTy &Bld = CGF.Builder;
7104   llvm::Value *ThreadLimitVal = nullptr;
7105   llvm::Value *NumThreadsVal = nullptr;
7106   switch (DirectiveKind) {
7107   case OMPD_target: {
7108     const CapturedStmt *CS = D.getInnermostCapturedStmt();
7109     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7110       return NumThreads;
7111     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7112         CGF.getContext(), CS->getCapturedStmt());
7113     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7114       if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) {
7115         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
7116         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
7117         const auto *ThreadLimitClause =
7118             Dir->getSingleClause<OMPThreadLimitClause>();
7119         CodeGenFunction::LexicalScope Scope(
7120             CGF, ThreadLimitClause->getThreadLimit()->getSourceRange());
7121         if (const auto *PreInit =
7122                 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) {
7123           for (const auto *I : PreInit->decls()) {
7124             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7125               CGF.EmitVarDecl(cast<VarDecl>(*I));
7126             } else {
7127               CodeGenFunction::AutoVarEmission Emission =
7128                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
7129               CGF.EmitAutoVarCleanups(Emission);
7130             }
7131           }
7132         }
7133         llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7134             ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7135         ThreadLimitVal =
7136             Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7137       }
7138       if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) &&
7139           !isOpenMPDistributeDirective(Dir->getDirectiveKind())) {
7140         CS = Dir->getInnermostCapturedStmt();
7141         const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7142             CGF.getContext(), CS->getCapturedStmt());
7143         Dir = dyn_cast_or_null<OMPExecutableDirective>(Child);
7144       }
7145       if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) &&
7146           !isOpenMPSimdDirective(Dir->getDirectiveKind())) {
7147         CS = Dir->getInnermostCapturedStmt();
7148         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7149           return NumThreads;
7150       }
7151       if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind()))
7152         return Bld.getInt32(1);
7153     }
7154     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7155   }
7156   case OMPD_target_teams: {
7157     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7158       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7159       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7160       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7161           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7162       ThreadLimitVal =
7163           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7164     }
7165     const CapturedStmt *CS = D.getInnermostCapturedStmt();
7166     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7167       return NumThreads;
7168     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7169         CGF.getContext(), CS->getCapturedStmt());
7170     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7171       if (Dir->getDirectiveKind() == OMPD_distribute) {
7172         CS = Dir->getInnermostCapturedStmt();
7173         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7174           return NumThreads;
7175       }
7176     }
7177     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7178   }
7179   case OMPD_target_teams_distribute:
7180     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7181       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7182       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7183       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7184           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7185       ThreadLimitVal =
7186           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7187     }
7188     return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal);
7189   case OMPD_target_parallel:
7190   case OMPD_target_parallel_for:
7191   case OMPD_target_parallel_for_simd:
7192   case OMPD_target_teams_distribute_parallel_for:
7193   case OMPD_target_teams_distribute_parallel_for_simd: {
7194     llvm::Value *CondVal = nullptr;
7195     // Handle if clause. If if clause present, the number of threads is
7196     // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
7197     if (D.hasClausesOfKind<OMPIfClause>()) {
7198       const OMPIfClause *IfClause = nullptr;
7199       for (const auto *C : D.getClausesOfKind<OMPIfClause>()) {
7200         if (C->getNameModifier() == OMPD_unknown ||
7201             C->getNameModifier() == OMPD_parallel) {
7202           IfClause = C;
7203           break;
7204         }
7205       }
7206       if (IfClause) {
7207         const Expr *Cond = IfClause->getCondition();
7208         bool Result;
7209         if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
7210           if (!Result)
7211             return Bld.getInt32(1);
7212         } else {
7213           CodeGenFunction::RunCleanupsScope Scope(CGF);
7214           CondVal = CGF.EvaluateExprAsBool(Cond);
7215         }
7216       }
7217     }
7218     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7219       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7220       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7221       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7222           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7223       ThreadLimitVal =
7224           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7225     }
7226     if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
7227       CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
7228       const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
7229       llvm::Value *NumThreads = CGF.EmitScalarExpr(
7230           NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true);
7231       NumThreadsVal =
7232           Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false);
7233       ThreadLimitVal = ThreadLimitVal
7234                            ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal,
7235                                                                 ThreadLimitVal),
7236                                               NumThreadsVal, ThreadLimitVal)
7237                            : NumThreadsVal;
7238     }
7239     if (!ThreadLimitVal)
7240       ThreadLimitVal = Bld.getInt32(0);
7241     if (CondVal)
7242       return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1));
7243     return ThreadLimitVal;
7244   }
7245   case OMPD_target_teams_distribute_simd:
7246   case OMPD_target_simd:
7247     return Bld.getInt32(1);
7248   case OMPD_parallel:
7249   case OMPD_for:
7250   case OMPD_parallel_for:
7251   case OMPD_parallel_master:
7252   case OMPD_parallel_sections:
7253   case OMPD_for_simd:
7254   case OMPD_parallel_for_simd:
7255   case OMPD_cancel:
7256   case OMPD_cancellation_point:
7257   case OMPD_ordered:
7258   case OMPD_threadprivate:
7259   case OMPD_allocate:
7260   case OMPD_task:
7261   case OMPD_simd:
7262   case OMPD_sections:
7263   case OMPD_section:
7264   case OMPD_single:
7265   case OMPD_master:
7266   case OMPD_critical:
7267   case OMPD_taskyield:
7268   case OMPD_barrier:
7269   case OMPD_taskwait:
7270   case OMPD_taskgroup:
7271   case OMPD_atomic:
7272   case OMPD_flush:
7273   case OMPD_depobj:
7274   case OMPD_scan:
7275   case OMPD_teams:
7276   case OMPD_target_data:
7277   case OMPD_target_exit_data:
7278   case OMPD_target_enter_data:
7279   case OMPD_distribute:
7280   case OMPD_distribute_simd:
7281   case OMPD_distribute_parallel_for:
7282   case OMPD_distribute_parallel_for_simd:
7283   case OMPD_teams_distribute:
7284   case OMPD_teams_distribute_simd:
7285   case OMPD_teams_distribute_parallel_for:
7286   case OMPD_teams_distribute_parallel_for_simd:
7287   case OMPD_target_update:
7288   case OMPD_declare_simd:
7289   case OMPD_declare_variant:
7290   case OMPD_begin_declare_variant:
7291   case OMPD_end_declare_variant:
7292   case OMPD_declare_target:
7293   case OMPD_end_declare_target:
7294   case OMPD_declare_reduction:
7295   case OMPD_declare_mapper:
7296   case OMPD_taskloop:
7297   case OMPD_taskloop_simd:
7298   case OMPD_master_taskloop:
7299   case OMPD_master_taskloop_simd:
7300   case OMPD_parallel_master_taskloop:
7301   case OMPD_parallel_master_taskloop_simd:
7302   case OMPD_requires:
7303   case OMPD_unknown:
7304     break;
7305   }
7306   llvm_unreachable("Unsupported directive kind.");
7307 }
7308 
7309 namespace {
7310 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
7311 
7312 // Utility to handle information from clauses associated with a given
7313 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause).
7314 // It provides a convenient interface to obtain the information and generate
7315 // code for that information.
7316 class MappableExprsHandler {
7317 public:
7318   /// Values for bit flags used to specify the mapping type for
7319   /// offloading.
7320   enum OpenMPOffloadMappingFlags : uint64_t {
7321     /// No flags
7322     OMP_MAP_NONE = 0x0,
7323     /// Allocate memory on the device and move data from host to device.
7324     OMP_MAP_TO = 0x01,
7325     /// Allocate memory on the device and move data from device to host.
7326     OMP_MAP_FROM = 0x02,
7327     /// Always perform the requested mapping action on the element, even
7328     /// if it was already mapped before.
7329     OMP_MAP_ALWAYS = 0x04,
7330     /// Delete the element from the device environment, ignoring the
7331     /// current reference count associated with the element.
7332     OMP_MAP_DELETE = 0x08,
7333     /// The element being mapped is a pointer-pointee pair; both the
7334     /// pointer and the pointee should be mapped.
7335     OMP_MAP_PTR_AND_OBJ = 0x10,
7336     /// This flags signals that the base address of an entry should be
7337     /// passed to the target kernel as an argument.
7338     OMP_MAP_TARGET_PARAM = 0x20,
7339     /// Signal that the runtime library has to return the device pointer
7340     /// in the current position for the data being mapped. Used when we have the
7341     /// use_device_ptr clause.
7342     OMP_MAP_RETURN_PARAM = 0x40,
7343     /// This flag signals that the reference being passed is a pointer to
7344     /// private data.
7345     OMP_MAP_PRIVATE = 0x80,
7346     /// Pass the element to the device by value.
7347     OMP_MAP_LITERAL = 0x100,
7348     /// Implicit map
7349     OMP_MAP_IMPLICIT = 0x200,
7350     /// Close is a hint to the runtime to allocate memory close to
7351     /// the target device.
7352     OMP_MAP_CLOSE = 0x400,
7353     /// The 16 MSBs of the flags indicate whether the entry is member of some
7354     /// struct/class.
7355     OMP_MAP_MEMBER_OF = 0xffff000000000000,
7356     LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF),
7357   };
7358 
7359   /// Get the offset of the OMP_MAP_MEMBER_OF field.
7360   static unsigned getFlagMemberOffset() {
7361     unsigned Offset = 0;
7362     for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1);
7363          Remain = Remain >> 1)
7364       Offset++;
7365     return Offset;
7366   }
7367 
7368   /// Class that associates information with a base pointer to be passed to the
7369   /// runtime library.
7370   class BasePointerInfo {
7371     /// The base pointer.
7372     llvm::Value *Ptr = nullptr;
7373     /// The base declaration that refers to this device pointer, or null if
7374     /// there is none.
7375     const ValueDecl *DevPtrDecl = nullptr;
7376 
7377   public:
7378     BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr)
7379         : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {}
7380     llvm::Value *operator*() const { return Ptr; }
7381     const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; }
7382     void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; }
7383   };
7384 
7385   using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>;
7386   using MapValuesArrayTy = SmallVector<llvm::Value *, 4>;
7387   using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>;
7388 
7389   /// Map between a struct and the its lowest & highest elements which have been
7390   /// mapped.
7391   /// [ValueDecl *] --> {LE(FieldIndex, Pointer),
7392   ///                    HE(FieldIndex, Pointer)}
7393   struct StructRangeInfoTy {
7394     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = {
7395         0, Address::invalid()};
7396     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = {
7397         0, Address::invalid()};
7398     Address Base = Address::invalid();
7399   };
7400 
7401 private:
7402   /// Kind that defines how a device pointer has to be returned.
7403   struct MapInfo {
7404     OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
7405     OpenMPMapClauseKind MapType = OMPC_MAP_unknown;
7406     ArrayRef<OpenMPMapModifierKind> MapModifiers;
7407     bool ReturnDevicePointer = false;
7408     bool IsImplicit = false;
7409 
7410     MapInfo() = default;
7411     MapInfo(
7412         OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7413         OpenMPMapClauseKind MapType,
7414         ArrayRef<OpenMPMapModifierKind> MapModifiers,
7415         bool ReturnDevicePointer, bool IsImplicit)
7416         : Components(Components), MapType(MapType), MapModifiers(MapModifiers),
7417           ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {}
7418   };
7419 
7420   /// If use_device_ptr is used on a pointer which is a struct member and there
7421   /// is no map information about it, then emission of that entry is deferred
7422   /// until the whole struct has been processed.
7423   struct DeferredDevicePtrEntryTy {
7424     const Expr *IE = nullptr;
7425     const ValueDecl *VD = nullptr;
7426 
7427     DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD)
7428         : IE(IE), VD(VD) {}
7429   };
7430 
7431   /// The target directive from where the mappable clauses were extracted. It
7432   /// is either a executable directive or a user-defined mapper directive.
7433   llvm::PointerUnion<const OMPExecutableDirective *,
7434                      const OMPDeclareMapperDecl *>
7435       CurDir;
7436 
7437   /// Function the directive is being generated for.
7438   CodeGenFunction &CGF;
7439 
7440   /// Set of all first private variables in the current directive.
7441   /// bool data is set to true if the variable is implicitly marked as
7442   /// firstprivate, false otherwise.
7443   llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls;
7444 
7445   /// Map between device pointer declarations and their expression components.
7446   /// The key value for declarations in 'this' is null.
7447   llvm::DenseMap<
7448       const ValueDecl *,
7449       SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7450       DevPointersMap;
7451 
7452   llvm::Value *getExprTypeSize(const Expr *E) const {
7453     QualType ExprTy = E->getType().getCanonicalType();
7454 
7455     // Reference types are ignored for mapping purposes.
7456     if (const auto *RefTy = ExprTy->getAs<ReferenceType>())
7457       ExprTy = RefTy->getPointeeType().getCanonicalType();
7458 
7459     // Given that an array section is considered a built-in type, we need to
7460     // do the calculation based on the length of the section instead of relying
7461     // on CGF.getTypeSize(E->getType()).
7462     if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) {
7463       QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType(
7464                             OAE->getBase()->IgnoreParenImpCasts())
7465                             .getCanonicalType();
7466 
7467       // If there is no length associated with the expression and lower bound is
7468       // not specified too, that means we are using the whole length of the
7469       // base.
7470       if (!OAE->getLength() && OAE->getColonLoc().isValid() &&
7471           !OAE->getLowerBound())
7472         return CGF.getTypeSize(BaseTy);
7473 
7474       llvm::Value *ElemSize;
7475       if (const auto *PTy = BaseTy->getAs<PointerType>()) {
7476         ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType());
7477       } else {
7478         const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr());
7479         assert(ATy && "Expecting array type if not a pointer type.");
7480         ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType());
7481       }
7482 
7483       // If we don't have a length at this point, that is because we have an
7484       // array section with a single element.
7485       if (!OAE->getLength() && OAE->getColonLoc().isInvalid())
7486         return ElemSize;
7487 
7488       if (const Expr *LenExpr = OAE->getLength()) {
7489         llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr);
7490         LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(),
7491                                              CGF.getContext().getSizeType(),
7492                                              LenExpr->getExprLoc());
7493         return CGF.Builder.CreateNUWMul(LengthVal, ElemSize);
7494       }
7495       assert(!OAE->getLength() && OAE->getColonLoc().isValid() &&
7496              OAE->getLowerBound() && "expected array_section[lb:].");
7497       // Size = sizetype - lb * elemtype;
7498       llvm::Value *LengthVal = CGF.getTypeSize(BaseTy);
7499       llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound());
7500       LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(),
7501                                        CGF.getContext().getSizeType(),
7502                                        OAE->getLowerBound()->getExprLoc());
7503       LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize);
7504       llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal);
7505       llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal);
7506       LengthVal = CGF.Builder.CreateSelect(
7507           Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0));
7508       return LengthVal;
7509     }
7510     return CGF.getTypeSize(ExprTy);
7511   }
7512 
7513   /// Return the corresponding bits for a given map clause modifier. Add
7514   /// a flag marking the map as a pointer if requested. Add a flag marking the
7515   /// map as the first one of a series of maps that relate to the same map
7516   /// expression.
7517   OpenMPOffloadMappingFlags getMapTypeBits(
7518       OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7519       bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const {
7520     OpenMPOffloadMappingFlags Bits =
7521         IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE;
7522     switch (MapType) {
7523     case OMPC_MAP_alloc:
7524     case OMPC_MAP_release:
7525       // alloc and release is the default behavior in the runtime library,  i.e.
7526       // if we don't pass any bits alloc/release that is what the runtime is
7527       // going to do. Therefore, we don't need to signal anything for these two
7528       // type modifiers.
7529       break;
7530     case OMPC_MAP_to:
7531       Bits |= OMP_MAP_TO;
7532       break;
7533     case OMPC_MAP_from:
7534       Bits |= OMP_MAP_FROM;
7535       break;
7536     case OMPC_MAP_tofrom:
7537       Bits |= OMP_MAP_TO | OMP_MAP_FROM;
7538       break;
7539     case OMPC_MAP_delete:
7540       Bits |= OMP_MAP_DELETE;
7541       break;
7542     case OMPC_MAP_unknown:
7543       llvm_unreachable("Unexpected map type!");
7544     }
7545     if (AddPtrFlag)
7546       Bits |= OMP_MAP_PTR_AND_OBJ;
7547     if (AddIsTargetParamFlag)
7548       Bits |= OMP_MAP_TARGET_PARAM;
7549     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always)
7550         != MapModifiers.end())
7551       Bits |= OMP_MAP_ALWAYS;
7552     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close)
7553         != MapModifiers.end())
7554       Bits |= OMP_MAP_CLOSE;
7555     return Bits;
7556   }
7557 
7558   /// Return true if the provided expression is a final array section. A
7559   /// final array section, is one whose length can't be proved to be one.
7560   bool isFinalArraySectionExpression(const Expr *E) const {
7561     const auto *OASE = dyn_cast<OMPArraySectionExpr>(E);
7562 
7563     // It is not an array section and therefore not a unity-size one.
7564     if (!OASE)
7565       return false;
7566 
7567     // An array section with no colon always refer to a single element.
7568     if (OASE->getColonLoc().isInvalid())
7569       return false;
7570 
7571     const Expr *Length = OASE->getLength();
7572 
7573     // If we don't have a length we have to check if the array has size 1
7574     // for this dimension. Also, we should always expect a length if the
7575     // base type is pointer.
7576     if (!Length) {
7577       QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType(
7578                              OASE->getBase()->IgnoreParenImpCasts())
7579                              .getCanonicalType();
7580       if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr()))
7581         return ATy->getSize().getSExtValue() != 1;
7582       // If we don't have a constant dimension length, we have to consider
7583       // the current section as having any size, so it is not necessarily
7584       // unitary. If it happen to be unity size, that's user fault.
7585       return true;
7586     }
7587 
7588     // Check if the length evaluates to 1.
7589     Expr::EvalResult Result;
7590     if (!Length->EvaluateAsInt(Result, CGF.getContext()))
7591       return true; // Can have more that size 1.
7592 
7593     llvm::APSInt ConstLength = Result.Val.getInt();
7594     return ConstLength.getSExtValue() != 1;
7595   }
7596 
7597   /// Generate the base pointers, section pointers, sizes and map type
7598   /// bits for the provided map type, map modifier, and expression components.
7599   /// \a IsFirstComponent should be set to true if the provided set of
7600   /// components is the first associated with a capture.
7601   void generateInfoForComponentList(
7602       OpenMPMapClauseKind MapType,
7603       ArrayRef<OpenMPMapModifierKind> MapModifiers,
7604       OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7605       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
7606       MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
7607       StructRangeInfoTy &PartialStruct, bool IsFirstComponentList,
7608       bool IsImplicit,
7609       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
7610           OverlappedElements = llvm::None) const {
7611     // The following summarizes what has to be generated for each map and the
7612     // types below. The generated information is expressed in this order:
7613     // base pointer, section pointer, size, flags
7614     // (to add to the ones that come from the map type and modifier).
7615     //
7616     // double d;
7617     // int i[100];
7618     // float *p;
7619     //
7620     // struct S1 {
7621     //   int i;
7622     //   float f[50];
7623     // }
7624     // struct S2 {
7625     //   int i;
7626     //   float f[50];
7627     //   S1 s;
7628     //   double *p;
7629     //   struct S2 *ps;
7630     // }
7631     // S2 s;
7632     // S2 *ps;
7633     //
7634     // map(d)
7635     // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM
7636     //
7637     // map(i)
7638     // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM
7639     //
7640     // map(i[1:23])
7641     // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM
7642     //
7643     // map(p)
7644     // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
7645     //
7646     // map(p[1:24])
7647     // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM
7648     //
7649     // map(s)
7650     // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM
7651     //
7652     // map(s.i)
7653     // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM
7654     //
7655     // map(s.s.f)
7656     // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7657     //
7658     // map(s.p)
7659     // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM
7660     //
7661     // map(to: s.p[:22])
7662     // &s, &(s.p), sizeof(double*), TARGET_PARAM (*)
7663     // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**)
7664     // &(s.p), &(s.p[0]), 22*sizeof(double),
7665     //   MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
7666     // (*) alloc space for struct members, only this is a target parameter
7667     // (**) map the pointer (nothing to be mapped in this example) (the compiler
7668     //      optimizes this entry out, same in the examples below)
7669     // (***) map the pointee (map: to)
7670     //
7671     // map(s.ps)
7672     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7673     //
7674     // map(from: s.ps->s.i)
7675     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7676     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7677     // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ  | FROM
7678     //
7679     // map(to: s.ps->ps)
7680     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7681     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7682     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ  | TO
7683     //
7684     // map(s.ps->ps->ps)
7685     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7686     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7687     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7688     // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7689     //
7690     // map(to: s.ps->ps->s.f[:22])
7691     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7692     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7693     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7694     // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7695     //
7696     // map(ps)
7697     // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM
7698     //
7699     // map(ps->i)
7700     // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM
7701     //
7702     // map(ps->s.f)
7703     // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7704     //
7705     // map(from: ps->p)
7706     // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM
7707     //
7708     // map(to: ps->p[:22])
7709     // ps, &(ps->p), sizeof(double*), TARGET_PARAM
7710     // ps, &(ps->p), sizeof(double*), MEMBER_OF(1)
7711     // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO
7712     //
7713     // map(ps->ps)
7714     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7715     //
7716     // map(from: ps->ps->s.i)
7717     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7718     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7719     // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7720     //
7721     // map(from: ps->ps->ps)
7722     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7723     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7724     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7725     //
7726     // map(ps->ps->ps->ps)
7727     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7728     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7729     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7730     // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7731     //
7732     // map(to: ps->ps->ps->s.f[:22])
7733     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7734     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7735     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7736     // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7737     //
7738     // map(to: s.f[:22]) map(from: s.p[:33])
7739     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) +
7740     //     sizeof(double*) (**), TARGET_PARAM
7741     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO
7742     // &s, &(s.p), sizeof(double*), MEMBER_OF(1)
7743     // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7744     // (*) allocate contiguous space needed to fit all mapped members even if
7745     //     we allocate space for members not mapped (in this example,
7746     //     s.f[22..49] and s.s are not mapped, yet we must allocate space for
7747     //     them as well because they fall between &s.f[0] and &s.p)
7748     //
7749     // map(from: s.f[:22]) map(to: ps->p[:33])
7750     // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM
7751     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7752     // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*)
7753     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO
7754     // (*) the struct this entry pertains to is the 2nd element in the list of
7755     //     arguments, hence MEMBER_OF(2)
7756     //
7757     // map(from: s.f[:22], s.s) map(to: ps->p[:33])
7758     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM
7759     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM
7760     // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM
7761     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7762     // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*)
7763     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO
7764     // (*) the struct this entry pertains to is the 4th element in the list
7765     //     of arguments, hence MEMBER_OF(4)
7766 
7767     // Track if the map information being generated is the first for a capture.
7768     bool IsCaptureFirstInfo = IsFirstComponentList;
7769     // When the variable is on a declare target link or in a to clause with
7770     // unified memory, a reference is needed to hold the host/device address
7771     // of the variable.
7772     bool RequiresReference = false;
7773 
7774     // Scan the components from the base to the complete expression.
7775     auto CI = Components.rbegin();
7776     auto CE = Components.rend();
7777     auto I = CI;
7778 
7779     // Track if the map information being generated is the first for a list of
7780     // components.
7781     bool IsExpressionFirstInfo = true;
7782     Address BP = Address::invalid();
7783     const Expr *AssocExpr = I->getAssociatedExpression();
7784     const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr);
7785     const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr);
7786 
7787     if (isa<MemberExpr>(AssocExpr)) {
7788       // The base is the 'this' pointer. The content of the pointer is going
7789       // to be the base of the field being mapped.
7790       BP = CGF.LoadCXXThisAddress();
7791     } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) ||
7792                (OASE &&
7793                 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) {
7794       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7795     } else {
7796       // The base is the reference to the variable.
7797       // BP = &Var.
7798       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7799       if (const auto *VD =
7800               dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) {
7801         if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
7802                 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) {
7803           if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
7804               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
7805                CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) {
7806             RequiresReference = true;
7807             BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
7808           }
7809         }
7810       }
7811 
7812       // If the variable is a pointer and is being dereferenced (i.e. is not
7813       // the last component), the base has to be the pointer itself, not its
7814       // reference. References are ignored for mapping purposes.
7815       QualType Ty =
7816           I->getAssociatedDeclaration()->getType().getNonReferenceType();
7817       if (Ty->isAnyPointerType() && std::next(I) != CE) {
7818         BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
7819 
7820         // We do not need to generate individual map information for the
7821         // pointer, it can be associated with the combined storage.
7822         ++I;
7823       }
7824     }
7825 
7826     // Track whether a component of the list should be marked as MEMBER_OF some
7827     // combined entry (for partial structs). Only the first PTR_AND_OBJ entry
7828     // in a component list should be marked as MEMBER_OF, all subsequent entries
7829     // do not belong to the base struct. E.g.
7830     // struct S2 s;
7831     // s.ps->ps->ps->f[:]
7832     //   (1) (2) (3) (4)
7833     // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a
7834     // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3)
7835     // is the pointee of ps(2) which is not member of struct s, so it should not
7836     // be marked as such (it is still PTR_AND_OBJ).
7837     // The variable is initialized to false so that PTR_AND_OBJ entries which
7838     // are not struct members are not considered (e.g. array of pointers to
7839     // data).
7840     bool ShouldBeMemberOf = false;
7841 
7842     // Variable keeping track of whether or not we have encountered a component
7843     // in the component list which is a member expression. Useful when we have a
7844     // pointer or a final array section, in which case it is the previous
7845     // component in the list which tells us whether we have a member expression.
7846     // E.g. X.f[:]
7847     // While processing the final array section "[:]" it is "f" which tells us
7848     // whether we are dealing with a member of a declared struct.
7849     const MemberExpr *EncounteredME = nullptr;
7850 
7851     for (; I != CE; ++I) {
7852       // If the current component is member of a struct (parent struct) mark it.
7853       if (!EncounteredME) {
7854         EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression());
7855         // If we encounter a PTR_AND_OBJ entry from now on it should be marked
7856         // as MEMBER_OF the parent struct.
7857         if (EncounteredME)
7858           ShouldBeMemberOf = true;
7859       }
7860 
7861       auto Next = std::next(I);
7862 
7863       // We need to generate the addresses and sizes if this is the last
7864       // component, if the component is a pointer or if it is an array section
7865       // whose length can't be proved to be one. If this is a pointer, it
7866       // becomes the base address for the following components.
7867 
7868       // A final array section, is one whose length can't be proved to be one.
7869       bool IsFinalArraySection =
7870           isFinalArraySectionExpression(I->getAssociatedExpression());
7871 
7872       // Get information on whether the element is a pointer. Have to do a
7873       // special treatment for array sections given that they are built-in
7874       // types.
7875       const auto *OASE =
7876           dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression());
7877       const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression());
7878       const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression());
7879       bool IsPointer =
7880           (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE)
7881                        .getCanonicalType()
7882                        ->isAnyPointerType()) ||
7883           I->getAssociatedExpression()->getType()->isAnyPointerType();
7884       bool IsNonDerefPointer = IsPointer && !UO && !BO;
7885 
7886       if (Next == CE || IsNonDerefPointer || IsFinalArraySection) {
7887         // If this is not the last component, we expect the pointer to be
7888         // associated with an array expression or member expression.
7889         assert((Next == CE ||
7890                 isa<MemberExpr>(Next->getAssociatedExpression()) ||
7891                 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) ||
7892                 isa<OMPArraySectionExpr>(Next->getAssociatedExpression()) ||
7893                 isa<UnaryOperator>(Next->getAssociatedExpression()) ||
7894                 isa<BinaryOperator>(Next->getAssociatedExpression())) &&
7895                "Unexpected expression");
7896 
7897         Address LB = CGF.EmitOMPSharedLValue(I->getAssociatedExpression())
7898                          .getAddress(CGF);
7899 
7900         // If this component is a pointer inside the base struct then we don't
7901         // need to create any entry for it - it will be combined with the object
7902         // it is pointing to into a single PTR_AND_OBJ entry.
7903         bool IsMemberPointer =
7904             IsPointer && EncounteredME &&
7905             (dyn_cast<MemberExpr>(I->getAssociatedExpression()) ==
7906              EncounteredME);
7907         if (!OverlappedElements.empty()) {
7908           // Handle base element with the info for overlapped elements.
7909           assert(!PartialStruct.Base.isValid() && "The base element is set.");
7910           assert(Next == CE &&
7911                  "Expected last element for the overlapped elements.");
7912           assert(!IsPointer &&
7913                  "Unexpected base element with the pointer type.");
7914           // Mark the whole struct as the struct that requires allocation on the
7915           // device.
7916           PartialStruct.LowestElem = {0, LB};
7917           CharUnits TypeSize = CGF.getContext().getTypeSizeInChars(
7918               I->getAssociatedExpression()->getType());
7919           Address HB = CGF.Builder.CreateConstGEP(
7920               CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB,
7921                                                               CGF.VoidPtrTy),
7922               TypeSize.getQuantity() - 1);
7923           PartialStruct.HighestElem = {
7924               std::numeric_limits<decltype(
7925                   PartialStruct.HighestElem.first)>::max(),
7926               HB};
7927           PartialStruct.Base = BP;
7928           // Emit data for non-overlapped data.
7929           OpenMPOffloadMappingFlags Flags =
7930               OMP_MAP_MEMBER_OF |
7931               getMapTypeBits(MapType, MapModifiers, IsImplicit,
7932                              /*AddPtrFlag=*/false,
7933                              /*AddIsTargetParamFlag=*/false);
7934           LB = BP;
7935           llvm::Value *Size = nullptr;
7936           // Do bitcopy of all non-overlapped structure elements.
7937           for (OMPClauseMappableExprCommon::MappableExprComponentListRef
7938                    Component : OverlappedElements) {
7939             Address ComponentLB = Address::invalid();
7940             for (const OMPClauseMappableExprCommon::MappableComponent &MC :
7941                  Component) {
7942               if (MC.getAssociatedDeclaration()) {
7943                 ComponentLB =
7944                     CGF.EmitOMPSharedLValue(MC.getAssociatedExpression())
7945                         .getAddress(CGF);
7946                 Size = CGF.Builder.CreatePtrDiff(
7947                     CGF.EmitCastToVoidPtr(ComponentLB.getPointer()),
7948                     CGF.EmitCastToVoidPtr(LB.getPointer()));
7949                 break;
7950               }
7951             }
7952             BasePointers.push_back(BP.getPointer());
7953             Pointers.push_back(LB.getPointer());
7954             Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty,
7955                                                       /*isSigned=*/true));
7956             Types.push_back(Flags);
7957             LB = CGF.Builder.CreateConstGEP(ComponentLB, 1);
7958           }
7959           BasePointers.push_back(BP.getPointer());
7960           Pointers.push_back(LB.getPointer());
7961           Size = CGF.Builder.CreatePtrDiff(
7962               CGF.EmitCastToVoidPtr(
7963                   CGF.Builder.CreateConstGEP(HB, 1).getPointer()),
7964               CGF.EmitCastToVoidPtr(LB.getPointer()));
7965           Sizes.push_back(
7966               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7967           Types.push_back(Flags);
7968           break;
7969         }
7970         llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression());
7971         if (!IsMemberPointer) {
7972           BasePointers.push_back(BP.getPointer());
7973           Pointers.push_back(LB.getPointer());
7974           Sizes.push_back(
7975               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7976 
7977           // We need to add a pointer flag for each map that comes from the
7978           // same expression except for the first one. We also need to signal
7979           // this map is the first one that relates with the current capture
7980           // (there is a set of entries for each capture).
7981           OpenMPOffloadMappingFlags Flags = getMapTypeBits(
7982               MapType, MapModifiers, IsImplicit,
7983               !IsExpressionFirstInfo || RequiresReference,
7984               IsCaptureFirstInfo && !RequiresReference);
7985 
7986           if (!IsExpressionFirstInfo) {
7987             // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well,
7988             // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags.
7989             if (IsPointer)
7990               Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS |
7991                          OMP_MAP_DELETE | OMP_MAP_CLOSE);
7992 
7993             if (ShouldBeMemberOf) {
7994               // Set placeholder value MEMBER_OF=FFFF to indicate that the flag
7995               // should be later updated with the correct value of MEMBER_OF.
7996               Flags |= OMP_MAP_MEMBER_OF;
7997               // From now on, all subsequent PTR_AND_OBJ entries should not be
7998               // marked as MEMBER_OF.
7999               ShouldBeMemberOf = false;
8000             }
8001           }
8002 
8003           Types.push_back(Flags);
8004         }
8005 
8006         // If we have encountered a member expression so far, keep track of the
8007         // mapped member. If the parent is "*this", then the value declaration
8008         // is nullptr.
8009         if (EncounteredME) {
8010           const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl());
8011           unsigned FieldIndex = FD->getFieldIndex();
8012 
8013           // Update info about the lowest and highest elements for this struct
8014           if (!PartialStruct.Base.isValid()) {
8015             PartialStruct.LowestElem = {FieldIndex, LB};
8016             PartialStruct.HighestElem = {FieldIndex, LB};
8017             PartialStruct.Base = BP;
8018           } else if (FieldIndex < PartialStruct.LowestElem.first) {
8019             PartialStruct.LowestElem = {FieldIndex, LB};
8020           } else if (FieldIndex > PartialStruct.HighestElem.first) {
8021             PartialStruct.HighestElem = {FieldIndex, LB};
8022           }
8023         }
8024 
8025         // If we have a final array section, we are done with this expression.
8026         if (IsFinalArraySection)
8027           break;
8028 
8029         // The pointer becomes the base for the next element.
8030         if (Next != CE)
8031           BP = LB;
8032 
8033         IsExpressionFirstInfo = false;
8034         IsCaptureFirstInfo = false;
8035       }
8036     }
8037   }
8038 
8039   /// Return the adjusted map modifiers if the declaration a capture refers to
8040   /// appears in a first-private clause. This is expected to be used only with
8041   /// directives that start with 'target'.
8042   MappableExprsHandler::OpenMPOffloadMappingFlags
8043   getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const {
8044     assert(Cap.capturesVariable() && "Expected capture by reference only!");
8045 
8046     // A first private variable captured by reference will use only the
8047     // 'private ptr' and 'map to' flag. Return the right flags if the captured
8048     // declaration is known as first-private in this handler.
8049     if (FirstPrivateDecls.count(Cap.getCapturedVar())) {
8050       if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) &&
8051           Cap.getCaptureKind() == CapturedStmt::VCK_ByRef)
8052         return MappableExprsHandler::OMP_MAP_ALWAYS |
8053                MappableExprsHandler::OMP_MAP_TO;
8054       if (Cap.getCapturedVar()->getType()->isAnyPointerType())
8055         return MappableExprsHandler::OMP_MAP_TO |
8056                MappableExprsHandler::OMP_MAP_PTR_AND_OBJ;
8057       return MappableExprsHandler::OMP_MAP_PRIVATE |
8058              MappableExprsHandler::OMP_MAP_TO;
8059     }
8060     return MappableExprsHandler::OMP_MAP_TO |
8061            MappableExprsHandler::OMP_MAP_FROM;
8062   }
8063 
8064   static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) {
8065     // Rotate by getFlagMemberOffset() bits.
8066     return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1)
8067                                                   << getFlagMemberOffset());
8068   }
8069 
8070   static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags,
8071                                      OpenMPOffloadMappingFlags MemberOfFlag) {
8072     // If the entry is PTR_AND_OBJ but has not been marked with the special
8073     // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be
8074     // marked as MEMBER_OF.
8075     if ((Flags & OMP_MAP_PTR_AND_OBJ) &&
8076         ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF))
8077       return;
8078 
8079     // Reset the placeholder value to prepare the flag for the assignment of the
8080     // proper MEMBER_OF value.
8081     Flags &= ~OMP_MAP_MEMBER_OF;
8082     Flags |= MemberOfFlag;
8083   }
8084 
8085   void getPlainLayout(const CXXRecordDecl *RD,
8086                       llvm::SmallVectorImpl<const FieldDecl *> &Layout,
8087                       bool AsBase) const {
8088     const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD);
8089 
8090     llvm::StructType *St =
8091         AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType();
8092 
8093     unsigned NumElements = St->getNumElements();
8094     llvm::SmallVector<
8095         llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4>
8096         RecordLayout(NumElements);
8097 
8098     // Fill bases.
8099     for (const auto &I : RD->bases()) {
8100       if (I.isVirtual())
8101         continue;
8102       const auto *Base = I.getType()->getAsCXXRecordDecl();
8103       // Ignore empty bases.
8104       if (Base->isEmpty() || CGF.getContext()
8105                                  .getASTRecordLayout(Base)
8106                                  .getNonVirtualSize()
8107                                  .isZero())
8108         continue;
8109 
8110       unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base);
8111       RecordLayout[FieldIndex] = Base;
8112     }
8113     // Fill in virtual bases.
8114     for (const auto &I : RD->vbases()) {
8115       const auto *Base = I.getType()->getAsCXXRecordDecl();
8116       // Ignore empty bases.
8117       if (Base->isEmpty())
8118         continue;
8119       unsigned FieldIndex = RL.getVirtualBaseIndex(Base);
8120       if (RecordLayout[FieldIndex])
8121         continue;
8122       RecordLayout[FieldIndex] = Base;
8123     }
8124     // Fill in all the fields.
8125     assert(!RD->isUnion() && "Unexpected union.");
8126     for (const auto *Field : RD->fields()) {
8127       // Fill in non-bitfields. (Bitfields always use a zero pattern, which we
8128       // will fill in later.)
8129       if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) {
8130         unsigned FieldIndex = RL.getLLVMFieldNo(Field);
8131         RecordLayout[FieldIndex] = Field;
8132       }
8133     }
8134     for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>
8135              &Data : RecordLayout) {
8136       if (Data.isNull())
8137         continue;
8138       if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>())
8139         getPlainLayout(Base, Layout, /*AsBase=*/true);
8140       else
8141         Layout.push_back(Data.get<const FieldDecl *>());
8142     }
8143   }
8144 
8145 public:
8146   MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF)
8147       : CurDir(&Dir), CGF(CGF) {
8148     // Extract firstprivate clause information.
8149     for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>())
8150       for (const auto *D : C->varlists())
8151         FirstPrivateDecls.try_emplace(
8152             cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit());
8153     // Extract device pointer clause information.
8154     for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
8155       for (auto L : C->component_lists())
8156         DevPointersMap[L.first].push_back(L.second);
8157   }
8158 
8159   /// Constructor for the declare mapper directive.
8160   MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF)
8161       : CurDir(&Dir), CGF(CGF) {}
8162 
8163   /// Generate code for the combined entry if we have a partially mapped struct
8164   /// and take care of the mapping flags of the arguments corresponding to
8165   /// individual struct members.
8166   void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers,
8167                          MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8168                          MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes,
8169                          const StructRangeInfoTy &PartialStruct) const {
8170     // Base is the base of the struct
8171     BasePointers.push_back(PartialStruct.Base.getPointer());
8172     // Pointer is the address of the lowest element
8173     llvm::Value *LB = PartialStruct.LowestElem.second.getPointer();
8174     Pointers.push_back(LB);
8175     // Size is (addr of {highest+1} element) - (addr of lowest element)
8176     llvm::Value *HB = PartialStruct.HighestElem.second.getPointer();
8177     llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1);
8178     llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy);
8179     llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy);
8180     llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr);
8181     llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty,
8182                                                   /*isSigned=*/false);
8183     Sizes.push_back(Size);
8184     // Map type is always TARGET_PARAM
8185     Types.push_back(OMP_MAP_TARGET_PARAM);
8186     // Remove TARGET_PARAM flag from the first element
8187     (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM;
8188 
8189     // All other current entries will be MEMBER_OF the combined entry
8190     // (except for PTR_AND_OBJ entries which do not have a placeholder value
8191     // 0xFFFF in the MEMBER_OF field).
8192     OpenMPOffloadMappingFlags MemberOfFlag =
8193         getMemberOfFlag(BasePointers.size() - 1);
8194     for (auto &M : CurTypes)
8195       setCorrectMemberOfFlag(M, MemberOfFlag);
8196   }
8197 
8198   /// Generate all the base pointers, section pointers, sizes and map
8199   /// types for the extracted mappable expressions. Also, for each item that
8200   /// relates with a device pointer, a pair of the relevant declaration and
8201   /// index where it occurs is appended to the device pointers info array.
8202   void generateAllInfo(MapBaseValuesArrayTy &BasePointers,
8203                        MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8204                        MapFlagsArrayTy &Types) const {
8205     // We have to process the component lists that relate with the same
8206     // declaration in a single chunk so that we can generate the map flags
8207     // correctly. Therefore, we organize all lists in a map.
8208     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
8209 
8210     // Helper function to fill the information map for the different supported
8211     // clauses.
8212     auto &&InfoGen = [&Info](
8213         const ValueDecl *D,
8214         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8215         OpenMPMapClauseKind MapType,
8216         ArrayRef<OpenMPMapModifierKind> MapModifiers,
8217         bool ReturnDevicePointer, bool IsImplicit) {
8218       const ValueDecl *VD =
8219           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
8220       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
8221                             IsImplicit);
8222     };
8223 
8224     assert(CurDir.is<const OMPExecutableDirective *>() &&
8225            "Expect a executable directive");
8226     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8227     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>())
8228       for (const auto L : C->component_lists()) {
8229         InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(),
8230             /*ReturnDevicePointer=*/false, C->isImplicit());
8231       }
8232     for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>())
8233       for (const auto L : C->component_lists()) {
8234         InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None,
8235             /*ReturnDevicePointer=*/false, C->isImplicit());
8236       }
8237     for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>())
8238       for (const auto L : C->component_lists()) {
8239         InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None,
8240             /*ReturnDevicePointer=*/false, C->isImplicit());
8241       }
8242 
8243     // Look at the use_device_ptr clause information and mark the existing map
8244     // entries as such. If there is no map information for an entry in the
8245     // use_device_ptr list, we create one with map type 'alloc' and zero size
8246     // section. It is the user fault if that was not mapped before. If there is
8247     // no map information and the pointer is a struct member, then we defer the
8248     // emission of that entry until the whole struct has been processed.
8249     llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>>
8250         DeferredInfo;
8251 
8252     for (const auto *C :
8253          CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) {
8254       for (const auto L : C->component_lists()) {
8255         assert(!L.second.empty() && "Not expecting empty list of components!");
8256         const ValueDecl *VD = L.second.back().getAssociatedDeclaration();
8257         VD = cast<ValueDecl>(VD->getCanonicalDecl());
8258         const Expr *IE = L.second.back().getAssociatedExpression();
8259         // If the first component is a member expression, we have to look into
8260         // 'this', which maps to null in the map of map information. Otherwise
8261         // look directly for the information.
8262         auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD);
8263 
8264         // We potentially have map information for this declaration already.
8265         // Look for the first set of components that refer to it.
8266         if (It != Info.end()) {
8267           auto CI = std::find_if(
8268               It->second.begin(), It->second.end(), [VD](const MapInfo &MI) {
8269                 return MI.Components.back().getAssociatedDeclaration() == VD;
8270               });
8271           // If we found a map entry, signal that the pointer has to be returned
8272           // and move on to the next declaration.
8273           if (CI != It->second.end()) {
8274             CI->ReturnDevicePointer = true;
8275             continue;
8276           }
8277         }
8278 
8279         // We didn't find any match in our map information - generate a zero
8280         // size array section - if the pointer is a struct member we defer this
8281         // action until the whole struct has been processed.
8282         if (isa<MemberExpr>(IE)) {
8283           // Insert the pointer into Info to be processed by
8284           // generateInfoForComponentList. Because it is a member pointer
8285           // without a pointee, no entry will be generated for it, therefore
8286           // we need to generate one after the whole struct has been processed.
8287           // Nonetheless, generateInfoForComponentList must be called to take
8288           // the pointer into account for the calculation of the range of the
8289           // partial struct.
8290           InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None,
8291                   /*ReturnDevicePointer=*/false, C->isImplicit());
8292           DeferredInfo[nullptr].emplace_back(IE, VD);
8293         } else {
8294           llvm::Value *Ptr =
8295               CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc());
8296           BasePointers.emplace_back(Ptr, VD);
8297           Pointers.push_back(Ptr);
8298           Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8299           Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM);
8300         }
8301       }
8302     }
8303 
8304     for (const auto &M : Info) {
8305       // We need to know when we generate information for the first component
8306       // associated with a capture, because the mapping flags depend on it.
8307       bool IsFirstComponentList = true;
8308 
8309       // Temporary versions of arrays
8310       MapBaseValuesArrayTy CurBasePointers;
8311       MapValuesArrayTy CurPointers;
8312       MapValuesArrayTy CurSizes;
8313       MapFlagsArrayTy CurTypes;
8314       StructRangeInfoTy PartialStruct;
8315 
8316       for (const MapInfo &L : M.second) {
8317         assert(!L.Components.empty() &&
8318                "Not expecting declaration with no component lists.");
8319 
8320         // Remember the current base pointer index.
8321         unsigned CurrentBasePointersIdx = CurBasePointers.size();
8322         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8323                                      CurBasePointers, CurPointers, CurSizes,
8324                                      CurTypes, PartialStruct,
8325                                      IsFirstComponentList, L.IsImplicit);
8326 
8327         // If this entry relates with a device pointer, set the relevant
8328         // declaration and add the 'return pointer' flag.
8329         if (L.ReturnDevicePointer) {
8330           assert(CurBasePointers.size() > CurrentBasePointersIdx &&
8331                  "Unexpected number of mapped base pointers.");
8332 
8333           const ValueDecl *RelevantVD =
8334               L.Components.back().getAssociatedDeclaration();
8335           assert(RelevantVD &&
8336                  "No relevant declaration related with device pointer??");
8337 
8338           CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD);
8339           CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM;
8340         }
8341         IsFirstComponentList = false;
8342       }
8343 
8344       // Append any pending zero-length pointers which are struct members and
8345       // used with use_device_ptr.
8346       auto CI = DeferredInfo.find(M.first);
8347       if (CI != DeferredInfo.end()) {
8348         for (const DeferredDevicePtrEntryTy &L : CI->second) {
8349           llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF);
8350           llvm::Value *Ptr = this->CGF.EmitLoadOfScalar(
8351               this->CGF.EmitLValue(L.IE), L.IE->getExprLoc());
8352           CurBasePointers.emplace_back(BasePtr, L.VD);
8353           CurPointers.push_back(Ptr);
8354           CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty));
8355           // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder
8356           // value MEMBER_OF=FFFF so that the entry is later updated with the
8357           // correct value of MEMBER_OF.
8358           CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM |
8359                              OMP_MAP_MEMBER_OF);
8360         }
8361       }
8362 
8363       // If there is an entry in PartialStruct it means we have a struct with
8364       // individual members mapped. Emit an extra combined entry.
8365       if (PartialStruct.Base.isValid())
8366         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8367                           PartialStruct);
8368 
8369       // We need to append the results of this capture to what we already have.
8370       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8371       Pointers.append(CurPointers.begin(), CurPointers.end());
8372       Sizes.append(CurSizes.begin(), CurSizes.end());
8373       Types.append(CurTypes.begin(), CurTypes.end());
8374     }
8375   }
8376 
8377   /// Generate all the base pointers, section pointers, sizes and map types for
8378   /// the extracted map clauses of user-defined mapper.
8379   void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers,
8380                                 MapValuesArrayTy &Pointers,
8381                                 MapValuesArrayTy &Sizes,
8382                                 MapFlagsArrayTy &Types) const {
8383     assert(CurDir.is<const OMPDeclareMapperDecl *>() &&
8384            "Expect a declare mapper directive");
8385     const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>();
8386     // We have to process the component lists that relate with the same
8387     // declaration in a single chunk so that we can generate the map flags
8388     // correctly. Therefore, we organize all lists in a map.
8389     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
8390 
8391     // Helper function to fill the information map for the different supported
8392     // clauses.
8393     auto &&InfoGen = [&Info](
8394         const ValueDecl *D,
8395         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8396         OpenMPMapClauseKind MapType,
8397         ArrayRef<OpenMPMapModifierKind> MapModifiers,
8398         bool ReturnDevicePointer, bool IsImplicit) {
8399       const ValueDecl *VD =
8400           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
8401       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
8402                             IsImplicit);
8403     };
8404 
8405     for (const auto *C : CurMapperDir->clauselists()) {
8406       const auto *MC = cast<OMPMapClause>(C);
8407       for (const auto L : MC->component_lists()) {
8408         InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(),
8409                 /*ReturnDevicePointer=*/false, MC->isImplicit());
8410       }
8411     }
8412 
8413     for (const auto &M : Info) {
8414       // We need to know when we generate information for the first component
8415       // associated with a capture, because the mapping flags depend on it.
8416       bool IsFirstComponentList = true;
8417 
8418       // Temporary versions of arrays
8419       MapBaseValuesArrayTy CurBasePointers;
8420       MapValuesArrayTy CurPointers;
8421       MapValuesArrayTy CurSizes;
8422       MapFlagsArrayTy CurTypes;
8423       StructRangeInfoTy PartialStruct;
8424 
8425       for (const MapInfo &L : M.second) {
8426         assert(!L.Components.empty() &&
8427                "Not expecting declaration with no component lists.");
8428         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8429                                      CurBasePointers, CurPointers, CurSizes,
8430                                      CurTypes, PartialStruct,
8431                                      IsFirstComponentList, L.IsImplicit);
8432         IsFirstComponentList = false;
8433       }
8434 
8435       // If there is an entry in PartialStruct it means we have a struct with
8436       // individual members mapped. Emit an extra combined entry.
8437       if (PartialStruct.Base.isValid())
8438         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8439                           PartialStruct);
8440 
8441       // We need to append the results of this capture to what we already have.
8442       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8443       Pointers.append(CurPointers.begin(), CurPointers.end());
8444       Sizes.append(CurSizes.begin(), CurSizes.end());
8445       Types.append(CurTypes.begin(), CurTypes.end());
8446     }
8447   }
8448 
8449   /// Emit capture info for lambdas for variables captured by reference.
8450   void generateInfoForLambdaCaptures(
8451       const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers,
8452       MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8453       MapFlagsArrayTy &Types,
8454       llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const {
8455     const auto *RD = VD->getType()
8456                          .getCanonicalType()
8457                          .getNonReferenceType()
8458                          ->getAsCXXRecordDecl();
8459     if (!RD || !RD->isLambda())
8460       return;
8461     Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD));
8462     LValue VDLVal = CGF.MakeAddrLValue(
8463         VDAddr, VD->getType().getCanonicalType().getNonReferenceType());
8464     llvm::DenseMap<const VarDecl *, FieldDecl *> Captures;
8465     FieldDecl *ThisCapture = nullptr;
8466     RD->getCaptureFields(Captures, ThisCapture);
8467     if (ThisCapture) {
8468       LValue ThisLVal =
8469           CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture);
8470       LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture);
8471       LambdaPointers.try_emplace(ThisLVal.getPointer(CGF),
8472                                  VDLVal.getPointer(CGF));
8473       BasePointers.push_back(ThisLVal.getPointer(CGF));
8474       Pointers.push_back(ThisLValVal.getPointer(CGF));
8475       Sizes.push_back(
8476           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8477                                     CGF.Int64Ty, /*isSigned=*/true));
8478       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8479                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8480     }
8481     for (const LambdaCapture &LC : RD->captures()) {
8482       if (!LC.capturesVariable())
8483         continue;
8484       const VarDecl *VD = LC.getCapturedVar();
8485       if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType())
8486         continue;
8487       auto It = Captures.find(VD);
8488       assert(It != Captures.end() && "Found lambda capture without field.");
8489       LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second);
8490       if (LC.getCaptureKind() == LCK_ByRef) {
8491         LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second);
8492         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
8493                                    VDLVal.getPointer(CGF));
8494         BasePointers.push_back(VarLVal.getPointer(CGF));
8495         Pointers.push_back(VarLValVal.getPointer(CGF));
8496         Sizes.push_back(CGF.Builder.CreateIntCast(
8497             CGF.getTypeSize(
8498                 VD->getType().getCanonicalType().getNonReferenceType()),
8499             CGF.Int64Ty, /*isSigned=*/true));
8500       } else {
8501         RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation());
8502         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
8503                                    VDLVal.getPointer(CGF));
8504         BasePointers.push_back(VarLVal.getPointer(CGF));
8505         Pointers.push_back(VarRVal.getScalarVal());
8506         Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0));
8507       }
8508       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8509                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8510     }
8511   }
8512 
8513   /// Set correct indices for lambdas captures.
8514   void adjustMemberOfForLambdaCaptures(
8515       const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers,
8516       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
8517       MapFlagsArrayTy &Types) const {
8518     for (unsigned I = 0, E = Types.size(); I < E; ++I) {
8519       // Set correct member_of idx for all implicit lambda captures.
8520       if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8521                        OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT))
8522         continue;
8523       llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]);
8524       assert(BasePtr && "Unable to find base lambda address.");
8525       int TgtIdx = -1;
8526       for (unsigned J = I; J > 0; --J) {
8527         unsigned Idx = J - 1;
8528         if (Pointers[Idx] != BasePtr)
8529           continue;
8530         TgtIdx = Idx;
8531         break;
8532       }
8533       assert(TgtIdx != -1 && "Unable to find parent lambda.");
8534       // All other current entries will be MEMBER_OF the combined entry
8535       // (except for PTR_AND_OBJ entries which do not have a placeholder value
8536       // 0xFFFF in the MEMBER_OF field).
8537       OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx);
8538       setCorrectMemberOfFlag(Types[I], MemberOfFlag);
8539     }
8540   }
8541 
8542   /// Generate the base pointers, section pointers, sizes and map types
8543   /// associated to a given capture.
8544   void generateInfoForCapture(const CapturedStmt::Capture *Cap,
8545                               llvm::Value *Arg,
8546                               MapBaseValuesArrayTy &BasePointers,
8547                               MapValuesArrayTy &Pointers,
8548                               MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
8549                               StructRangeInfoTy &PartialStruct) const {
8550     assert(!Cap->capturesVariableArrayType() &&
8551            "Not expecting to generate map info for a variable array type!");
8552 
8553     // We need to know when we generating information for the first component
8554     const ValueDecl *VD = Cap->capturesThis()
8555                               ? nullptr
8556                               : Cap->getCapturedVar()->getCanonicalDecl();
8557 
8558     // If this declaration appears in a is_device_ptr clause we just have to
8559     // pass the pointer by value. If it is a reference to a declaration, we just
8560     // pass its value.
8561     if (DevPointersMap.count(VD)) {
8562       BasePointers.emplace_back(Arg, VD);
8563       Pointers.push_back(Arg);
8564       Sizes.push_back(
8565           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8566                                     CGF.Int64Ty, /*isSigned=*/true));
8567       Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM);
8568       return;
8569     }
8570 
8571     using MapData =
8572         std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef,
8573                    OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>;
8574     SmallVector<MapData, 4> DeclComponentLists;
8575     assert(CurDir.is<const OMPExecutableDirective *>() &&
8576            "Expect a executable directive");
8577     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8578     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8579       for (const auto L : C->decl_component_lists(VD)) {
8580         assert(L.first == VD &&
8581                "We got information for the wrong declaration??");
8582         assert(!L.second.empty() &&
8583                "Not expecting declaration with no component lists.");
8584         DeclComponentLists.emplace_back(L.second, C->getMapType(),
8585                                         C->getMapTypeModifiers(),
8586                                         C->isImplicit());
8587       }
8588     }
8589 
8590     // Find overlapping elements (including the offset from the base element).
8591     llvm::SmallDenseMap<
8592         const MapData *,
8593         llvm::SmallVector<
8594             OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>,
8595         4>
8596         OverlappedData;
8597     size_t Count = 0;
8598     for (const MapData &L : DeclComponentLists) {
8599       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8600       OpenMPMapClauseKind MapType;
8601       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8602       bool IsImplicit;
8603       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8604       ++Count;
8605       for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) {
8606         OMPClauseMappableExprCommon::MappableExprComponentListRef Components1;
8607         std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1;
8608         auto CI = Components.rbegin();
8609         auto CE = Components.rend();
8610         auto SI = Components1.rbegin();
8611         auto SE = Components1.rend();
8612         for (; CI != CE && SI != SE; ++CI, ++SI) {
8613           if (CI->getAssociatedExpression()->getStmtClass() !=
8614               SI->getAssociatedExpression()->getStmtClass())
8615             break;
8616           // Are we dealing with different variables/fields?
8617           if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration())
8618             break;
8619         }
8620         // Found overlapping if, at least for one component, reached the head of
8621         // the components list.
8622         if (CI == CE || SI == SE) {
8623           assert((CI != CE || SI != SE) &&
8624                  "Unexpected full match of the mapping components.");
8625           const MapData &BaseData = CI == CE ? L : L1;
8626           OMPClauseMappableExprCommon::MappableExprComponentListRef SubData =
8627               SI == SE ? Components : Components1;
8628           auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData);
8629           OverlappedElements.getSecond().push_back(SubData);
8630         }
8631       }
8632     }
8633     // Sort the overlapped elements for each item.
8634     llvm::SmallVector<const FieldDecl *, 4> Layout;
8635     if (!OverlappedData.empty()) {
8636       if (const auto *CRD =
8637               VD->getType().getCanonicalType()->getAsCXXRecordDecl())
8638         getPlainLayout(CRD, Layout, /*AsBase=*/false);
8639       else {
8640         const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl();
8641         Layout.append(RD->field_begin(), RD->field_end());
8642       }
8643     }
8644     for (auto &Pair : OverlappedData) {
8645       llvm::sort(
8646           Pair.getSecond(),
8647           [&Layout](
8648               OMPClauseMappableExprCommon::MappableExprComponentListRef First,
8649               OMPClauseMappableExprCommon::MappableExprComponentListRef
8650                   Second) {
8651             auto CI = First.rbegin();
8652             auto CE = First.rend();
8653             auto SI = Second.rbegin();
8654             auto SE = Second.rend();
8655             for (; CI != CE && SI != SE; ++CI, ++SI) {
8656               if (CI->getAssociatedExpression()->getStmtClass() !=
8657                   SI->getAssociatedExpression()->getStmtClass())
8658                 break;
8659               // Are we dealing with different variables/fields?
8660               if (CI->getAssociatedDeclaration() !=
8661                   SI->getAssociatedDeclaration())
8662                 break;
8663             }
8664 
8665             // Lists contain the same elements.
8666             if (CI == CE && SI == SE)
8667               return false;
8668 
8669             // List with less elements is less than list with more elements.
8670             if (CI == CE || SI == SE)
8671               return CI == CE;
8672 
8673             const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration());
8674             const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration());
8675             if (FD1->getParent() == FD2->getParent())
8676               return FD1->getFieldIndex() < FD2->getFieldIndex();
8677             const auto It =
8678                 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) {
8679                   return FD == FD1 || FD == FD2;
8680                 });
8681             return *It == FD1;
8682           });
8683     }
8684 
8685     // Associated with a capture, because the mapping flags depend on it.
8686     // Go through all of the elements with the overlapped elements.
8687     for (const auto &Pair : OverlappedData) {
8688       const MapData &L = *Pair.getFirst();
8689       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8690       OpenMPMapClauseKind MapType;
8691       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8692       bool IsImplicit;
8693       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8694       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
8695           OverlappedComponents = Pair.getSecond();
8696       bool IsFirstComponentList = true;
8697       generateInfoForComponentList(MapType, MapModifiers, Components,
8698                                    BasePointers, Pointers, Sizes, Types,
8699                                    PartialStruct, IsFirstComponentList,
8700                                    IsImplicit, OverlappedComponents);
8701     }
8702     // Go through other elements without overlapped elements.
8703     bool IsFirstComponentList = OverlappedData.empty();
8704     for (const MapData &L : DeclComponentLists) {
8705       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8706       OpenMPMapClauseKind MapType;
8707       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8708       bool IsImplicit;
8709       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8710       auto It = OverlappedData.find(&L);
8711       if (It == OverlappedData.end())
8712         generateInfoForComponentList(MapType, MapModifiers, Components,
8713                                      BasePointers, Pointers, Sizes, Types,
8714                                      PartialStruct, IsFirstComponentList,
8715                                      IsImplicit);
8716       IsFirstComponentList = false;
8717     }
8718   }
8719 
8720   /// Generate the base pointers, section pointers, sizes and map types
8721   /// associated with the declare target link variables.
8722   void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers,
8723                                         MapValuesArrayTy &Pointers,
8724                                         MapValuesArrayTy &Sizes,
8725                                         MapFlagsArrayTy &Types) const {
8726     assert(CurDir.is<const OMPExecutableDirective *>() &&
8727            "Expect a executable directive");
8728     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8729     // Map other list items in the map clause which are not captured variables
8730     // but "declare target link" global variables.
8731     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8732       for (const auto L : C->component_lists()) {
8733         if (!L.first)
8734           continue;
8735         const auto *VD = dyn_cast<VarDecl>(L.first);
8736         if (!VD)
8737           continue;
8738         llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
8739             OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
8740         if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
8741             !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link)
8742           continue;
8743         StructRangeInfoTy PartialStruct;
8744         generateInfoForComponentList(
8745             C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers,
8746             Pointers, Sizes, Types, PartialStruct,
8747             /*IsFirstComponentList=*/true, C->isImplicit());
8748         assert(!PartialStruct.Base.isValid() &&
8749                "No partial structs for declare target link expected.");
8750       }
8751     }
8752   }
8753 
8754   /// Generate the default map information for a given capture \a CI,
8755   /// record field declaration \a RI and captured value \a CV.
8756   void generateDefaultMapInfo(const CapturedStmt::Capture &CI,
8757                               const FieldDecl &RI, llvm::Value *CV,
8758                               MapBaseValuesArrayTy &CurBasePointers,
8759                               MapValuesArrayTy &CurPointers,
8760                               MapValuesArrayTy &CurSizes,
8761                               MapFlagsArrayTy &CurMapTypes) const {
8762     bool IsImplicit = true;
8763     // Do the default mapping.
8764     if (CI.capturesThis()) {
8765       CurBasePointers.push_back(CV);
8766       CurPointers.push_back(CV);
8767       const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr());
8768       CurSizes.push_back(
8769           CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()),
8770                                     CGF.Int64Ty, /*isSigned=*/true));
8771       // Default map type.
8772       CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM);
8773     } else if (CI.capturesVariableByCopy()) {
8774       CurBasePointers.push_back(CV);
8775       CurPointers.push_back(CV);
8776       if (!RI.getType()->isAnyPointerType()) {
8777         // We have to signal to the runtime captures passed by value that are
8778         // not pointers.
8779         CurMapTypes.push_back(OMP_MAP_LITERAL);
8780         CurSizes.push_back(CGF.Builder.CreateIntCast(
8781             CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true));
8782       } else {
8783         // Pointers are implicitly mapped with a zero size and no flags
8784         // (other than first map that is added for all implicit maps).
8785         CurMapTypes.push_back(OMP_MAP_NONE);
8786         CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8787       }
8788       const VarDecl *VD = CI.getCapturedVar();
8789       auto I = FirstPrivateDecls.find(VD);
8790       if (I != FirstPrivateDecls.end())
8791         IsImplicit = I->getSecond();
8792     } else {
8793       assert(CI.capturesVariable() && "Expected captured reference.");
8794       const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr());
8795       QualType ElementType = PtrTy->getPointeeType();
8796       CurSizes.push_back(CGF.Builder.CreateIntCast(
8797           CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true));
8798       // The default map type for a scalar/complex type is 'to' because by
8799       // default the value doesn't have to be retrieved. For an aggregate
8800       // type, the default is 'tofrom'.
8801       CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI));
8802       const VarDecl *VD = CI.getCapturedVar();
8803       auto I = FirstPrivateDecls.find(VD);
8804       if (I != FirstPrivateDecls.end() &&
8805           VD->getType().isConstant(CGF.getContext())) {
8806         llvm::Constant *Addr =
8807             CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD);
8808         // Copy the value of the original variable to the new global copy.
8809         CGF.Builder.CreateMemCpy(
8810             CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(CGF),
8811             Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)),
8812             CurSizes.back(), /*IsVolatile=*/false);
8813         // Use new global variable as the base pointers.
8814         CurBasePointers.push_back(Addr);
8815         CurPointers.push_back(Addr);
8816       } else {
8817         CurBasePointers.push_back(CV);
8818         if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) {
8819           Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue(
8820               CV, ElementType, CGF.getContext().getDeclAlign(VD),
8821               AlignmentSource::Decl));
8822           CurPointers.push_back(PtrAddr.getPointer());
8823         } else {
8824           CurPointers.push_back(CV);
8825         }
8826       }
8827       if (I != FirstPrivateDecls.end())
8828         IsImplicit = I->getSecond();
8829     }
8830     // Every default map produces a single argument which is a target parameter.
8831     CurMapTypes.back() |= OMP_MAP_TARGET_PARAM;
8832 
8833     // Add flag stating this is an implicit map.
8834     if (IsImplicit)
8835       CurMapTypes.back() |= OMP_MAP_IMPLICIT;
8836   }
8837 };
8838 } // anonymous namespace
8839 
8840 /// Emit the arrays used to pass the captures and map information to the
8841 /// offloading runtime library. If there is no map or capture information,
8842 /// return nullptr by reference.
8843 static void
8844 emitOffloadingArrays(CodeGenFunction &CGF,
8845                      MappableExprsHandler::MapBaseValuesArrayTy &BasePointers,
8846                      MappableExprsHandler::MapValuesArrayTy &Pointers,
8847                      MappableExprsHandler::MapValuesArrayTy &Sizes,
8848                      MappableExprsHandler::MapFlagsArrayTy &MapTypes,
8849                      CGOpenMPRuntime::TargetDataInfo &Info) {
8850   CodeGenModule &CGM = CGF.CGM;
8851   ASTContext &Ctx = CGF.getContext();
8852 
8853   // Reset the array information.
8854   Info.clearArrayInfo();
8855   Info.NumberOfPtrs = BasePointers.size();
8856 
8857   if (Info.NumberOfPtrs) {
8858     // Detect if we have any capture size requiring runtime evaluation of the
8859     // size so that a constant array could be eventually used.
8860     bool hasRuntimeEvaluationCaptureSize = false;
8861     for (llvm::Value *S : Sizes)
8862       if (!isa<llvm::Constant>(S)) {
8863         hasRuntimeEvaluationCaptureSize = true;
8864         break;
8865       }
8866 
8867     llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true);
8868     QualType PointerArrayType = Ctx.getConstantArrayType(
8869         Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal,
8870         /*IndexTypeQuals=*/0);
8871 
8872     Info.BasePointersArray =
8873         CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer();
8874     Info.PointersArray =
8875         CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer();
8876 
8877     // If we don't have any VLA types or other types that require runtime
8878     // evaluation, we can use a constant array for the map sizes, otherwise we
8879     // need to fill up the arrays as we do for the pointers.
8880     QualType Int64Ty =
8881         Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
8882     if (hasRuntimeEvaluationCaptureSize) {
8883       QualType SizeArrayType = Ctx.getConstantArrayType(
8884           Int64Ty, PointerNumAP, nullptr, ArrayType::Normal,
8885           /*IndexTypeQuals=*/0);
8886       Info.SizesArray =
8887           CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer();
8888     } else {
8889       // We expect all the sizes to be constant, so we collect them to create
8890       // a constant array.
8891       SmallVector<llvm::Constant *, 16> ConstSizes;
8892       for (llvm::Value *S : Sizes)
8893         ConstSizes.push_back(cast<llvm::Constant>(S));
8894 
8895       auto *SizesArrayInit = llvm::ConstantArray::get(
8896           llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes);
8897       std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"});
8898       auto *SizesArrayGbl = new llvm::GlobalVariable(
8899           CGM.getModule(), SizesArrayInit->getType(),
8900           /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8901           SizesArrayInit, Name);
8902       SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8903       Info.SizesArray = SizesArrayGbl;
8904     }
8905 
8906     // The map types are always constant so we don't need to generate code to
8907     // fill arrays. Instead, we create an array constant.
8908     SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0);
8909     llvm::copy(MapTypes, Mapping.begin());
8910     llvm::Constant *MapTypesArrayInit =
8911         llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping);
8912     std::string MaptypesName =
8913         CGM.getOpenMPRuntime().getName({"offload_maptypes"});
8914     auto *MapTypesArrayGbl = new llvm::GlobalVariable(
8915         CGM.getModule(), MapTypesArrayInit->getType(),
8916         /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8917         MapTypesArrayInit, MaptypesName);
8918     MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8919     Info.MapTypesArray = MapTypesArrayGbl;
8920 
8921     for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) {
8922       llvm::Value *BPVal = *BasePointers[I];
8923       llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32(
8924           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8925           Info.BasePointersArray, 0, I);
8926       BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8927           BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0));
8928       Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8929       CGF.Builder.CreateStore(BPVal, BPAddr);
8930 
8931       if (Info.requiresDevicePointerInfo())
8932         if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl())
8933           Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr);
8934 
8935       llvm::Value *PVal = Pointers[I];
8936       llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32(
8937           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8938           Info.PointersArray, 0, I);
8939       P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8940           P, PVal->getType()->getPointerTo(/*AddrSpace=*/0));
8941       Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8942       CGF.Builder.CreateStore(PVal, PAddr);
8943 
8944       if (hasRuntimeEvaluationCaptureSize) {
8945         llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32(
8946             llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8947             Info.SizesArray,
8948             /*Idx0=*/0,
8949             /*Idx1=*/I);
8950         Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty));
8951         CGF.Builder.CreateStore(
8952             CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true),
8953             SAddr);
8954       }
8955     }
8956   }
8957 }
8958 
8959 /// Emit the arguments to be passed to the runtime library based on the
8960 /// arrays of pointers, sizes and map types.
8961 static void emitOffloadingArraysArgument(
8962     CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg,
8963     llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg,
8964     llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) {
8965   CodeGenModule &CGM = CGF.CGM;
8966   if (Info.NumberOfPtrs) {
8967     BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8968         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8969         Info.BasePointersArray,
8970         /*Idx0=*/0, /*Idx1=*/0);
8971     PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8972         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8973         Info.PointersArray,
8974         /*Idx0=*/0,
8975         /*Idx1=*/0);
8976     SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8977         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray,
8978         /*Idx0=*/0, /*Idx1=*/0);
8979     MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8980         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8981         Info.MapTypesArray,
8982         /*Idx0=*/0,
8983         /*Idx1=*/0);
8984   } else {
8985     BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8986     PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8987     SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8988     MapTypesArrayArg =
8989         llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8990   }
8991 }
8992 
8993 /// Check for inner distribute directive.
8994 static const OMPExecutableDirective *
8995 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) {
8996   const auto *CS = D.getInnermostCapturedStmt();
8997   const auto *Body =
8998       CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
8999   const Stmt *ChildStmt =
9000       CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
9001 
9002   if (const auto *NestedDir =
9003           dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
9004     OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind();
9005     switch (D.getDirectiveKind()) {
9006     case OMPD_target:
9007       if (isOpenMPDistributeDirective(DKind))
9008         return NestedDir;
9009       if (DKind == OMPD_teams) {
9010         Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers(
9011             /*IgnoreCaptured=*/true);
9012         if (!Body)
9013           return nullptr;
9014         ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
9015         if (const auto *NND =
9016                 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
9017           DKind = NND->getDirectiveKind();
9018           if (isOpenMPDistributeDirective(DKind))
9019             return NND;
9020         }
9021       }
9022       return nullptr;
9023     case OMPD_target_teams:
9024       if (isOpenMPDistributeDirective(DKind))
9025         return NestedDir;
9026       return nullptr;
9027     case OMPD_target_parallel:
9028     case OMPD_target_simd:
9029     case OMPD_target_parallel_for:
9030     case OMPD_target_parallel_for_simd:
9031       return nullptr;
9032     case OMPD_target_teams_distribute:
9033     case OMPD_target_teams_distribute_simd:
9034     case OMPD_target_teams_distribute_parallel_for:
9035     case OMPD_target_teams_distribute_parallel_for_simd:
9036     case OMPD_parallel:
9037     case OMPD_for:
9038     case OMPD_parallel_for:
9039     case OMPD_parallel_master:
9040     case OMPD_parallel_sections:
9041     case OMPD_for_simd:
9042     case OMPD_parallel_for_simd:
9043     case OMPD_cancel:
9044     case OMPD_cancellation_point:
9045     case OMPD_ordered:
9046     case OMPD_threadprivate:
9047     case OMPD_allocate:
9048     case OMPD_task:
9049     case OMPD_simd:
9050     case OMPD_sections:
9051     case OMPD_section:
9052     case OMPD_single:
9053     case OMPD_master:
9054     case OMPD_critical:
9055     case OMPD_taskyield:
9056     case OMPD_barrier:
9057     case OMPD_taskwait:
9058     case OMPD_taskgroup:
9059     case OMPD_atomic:
9060     case OMPD_flush:
9061     case OMPD_depobj:
9062     case OMPD_scan:
9063     case OMPD_teams:
9064     case OMPD_target_data:
9065     case OMPD_target_exit_data:
9066     case OMPD_target_enter_data:
9067     case OMPD_distribute:
9068     case OMPD_distribute_simd:
9069     case OMPD_distribute_parallel_for:
9070     case OMPD_distribute_parallel_for_simd:
9071     case OMPD_teams_distribute:
9072     case OMPD_teams_distribute_simd:
9073     case OMPD_teams_distribute_parallel_for:
9074     case OMPD_teams_distribute_parallel_for_simd:
9075     case OMPD_target_update:
9076     case OMPD_declare_simd:
9077     case OMPD_declare_variant:
9078     case OMPD_begin_declare_variant:
9079     case OMPD_end_declare_variant:
9080     case OMPD_declare_target:
9081     case OMPD_end_declare_target:
9082     case OMPD_declare_reduction:
9083     case OMPD_declare_mapper:
9084     case OMPD_taskloop:
9085     case OMPD_taskloop_simd:
9086     case OMPD_master_taskloop:
9087     case OMPD_master_taskloop_simd:
9088     case OMPD_parallel_master_taskloop:
9089     case OMPD_parallel_master_taskloop_simd:
9090     case OMPD_requires:
9091     case OMPD_unknown:
9092       llvm_unreachable("Unexpected directive.");
9093     }
9094   }
9095 
9096   return nullptr;
9097 }
9098 
9099 /// Emit the user-defined mapper function. The code generation follows the
9100 /// pattern in the example below.
9101 /// \code
9102 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle,
9103 ///                                           void *base, void *begin,
9104 ///                                           int64_t size, int64_t type) {
9105 ///   // Allocate space for an array section first.
9106 ///   if (size > 1 && !maptype.IsDelete)
9107 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
9108 ///                                 size*sizeof(Ty), clearToFrom(type));
9109 ///   // Map members.
9110 ///   for (unsigned i = 0; i < size; i++) {
9111 ///     // For each component specified by this mapper:
9112 ///     for (auto c : all_components) {
9113 ///       if (c.hasMapper())
9114 ///         (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size,
9115 ///                       c.arg_type);
9116 ///       else
9117 ///         __tgt_push_mapper_component(rt_mapper_handle, c.arg_base,
9118 ///                                     c.arg_begin, c.arg_size, c.arg_type);
9119 ///     }
9120 ///   }
9121 ///   // Delete the array section.
9122 ///   if (size > 1 && maptype.IsDelete)
9123 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
9124 ///                                 size*sizeof(Ty), clearToFrom(type));
9125 /// }
9126 /// \endcode
9127 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D,
9128                                             CodeGenFunction *CGF) {
9129   if (UDMMap.count(D) > 0)
9130     return;
9131   ASTContext &C = CGM.getContext();
9132   QualType Ty = D->getType();
9133   QualType PtrTy = C.getPointerType(Ty).withRestrict();
9134   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
9135   auto *MapperVarDecl =
9136       cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl());
9137   SourceLocation Loc = D->getLocation();
9138   CharUnits ElementSize = C.getTypeSizeInChars(Ty);
9139 
9140   // Prepare mapper function arguments and attributes.
9141   ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
9142                               C.VoidPtrTy, ImplicitParamDecl::Other);
9143   ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
9144                             ImplicitParamDecl::Other);
9145   ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
9146                              C.VoidPtrTy, ImplicitParamDecl::Other);
9147   ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
9148                             ImplicitParamDecl::Other);
9149   ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
9150                             ImplicitParamDecl::Other);
9151   FunctionArgList Args;
9152   Args.push_back(&HandleArg);
9153   Args.push_back(&BaseArg);
9154   Args.push_back(&BeginArg);
9155   Args.push_back(&SizeArg);
9156   Args.push_back(&TypeArg);
9157   const CGFunctionInfo &FnInfo =
9158       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
9159   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
9160   SmallString<64> TyStr;
9161   llvm::raw_svector_ostream Out(TyStr);
9162   CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out);
9163   std::string Name = getName({"omp_mapper", TyStr, D->getName()});
9164   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
9165                                     Name, &CGM.getModule());
9166   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
9167   Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
9168   // Start the mapper function code generation.
9169   CodeGenFunction MapperCGF(CGM);
9170   MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
9171   // Compute the starting and end addreses of array elements.
9172   llvm::Value *Size = MapperCGF.EmitLoadOfScalar(
9173       MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false,
9174       C.getPointerType(Int64Ty), Loc);
9175   llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast(
9176       MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(),
9177       CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy)));
9178   llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size);
9179   llvm::Value *MapType = MapperCGF.EmitLoadOfScalar(
9180       MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false,
9181       C.getPointerType(Int64Ty), Loc);
9182   // Prepare common arguments for array initiation and deletion.
9183   llvm::Value *Handle = MapperCGF.EmitLoadOfScalar(
9184       MapperCGF.GetAddrOfLocalVar(&HandleArg),
9185       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9186   llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar(
9187       MapperCGF.GetAddrOfLocalVar(&BaseArg),
9188       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9189   llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar(
9190       MapperCGF.GetAddrOfLocalVar(&BeginArg),
9191       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9192 
9193   // Emit array initiation if this is an array section and \p MapType indicates
9194   // that memory allocation is required.
9195   llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head");
9196   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
9197                              ElementSize, HeadBB, /*IsInit=*/true);
9198 
9199   // Emit a for loop to iterate through SizeArg of elements and map all of them.
9200 
9201   // Emit the loop header block.
9202   MapperCGF.EmitBlock(HeadBB);
9203   llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body");
9204   llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done");
9205   // Evaluate whether the initial condition is satisfied.
9206   llvm::Value *IsEmpty =
9207       MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty");
9208   MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
9209   llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock();
9210 
9211   // Emit the loop body block.
9212   MapperCGF.EmitBlock(BodyBB);
9213   llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI(
9214       PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent");
9215   PtrPHI->addIncoming(PtrBegin, EntryBB);
9216   Address PtrCurrent =
9217       Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg)
9218                           .getAlignment()
9219                           .alignmentOfArrayElement(ElementSize));
9220   // Privatize the declared variable of mapper to be the current array element.
9221   CodeGenFunction::OMPPrivateScope Scope(MapperCGF);
9222   Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() {
9223     return MapperCGF
9224         .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>())
9225         .getAddress(MapperCGF);
9226   });
9227   (void)Scope.Privatize();
9228 
9229   // Get map clause information. Fill up the arrays with all mapped variables.
9230   MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9231   MappableExprsHandler::MapValuesArrayTy Pointers;
9232   MappableExprsHandler::MapValuesArrayTy Sizes;
9233   MappableExprsHandler::MapFlagsArrayTy MapTypes;
9234   MappableExprsHandler MEHandler(*D, MapperCGF);
9235   MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes);
9236 
9237   // Call the runtime API __tgt_mapper_num_components to get the number of
9238   // pre-existing components.
9239   llvm::Value *OffloadingArgs[] = {Handle};
9240   llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall(
9241       createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs);
9242   llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl(
9243       PreviousSize,
9244       MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset()));
9245 
9246   // Fill up the runtime mapper handle for all components.
9247   for (unsigned I = 0; I < BasePointers.size(); ++I) {
9248     llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast(
9249         *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
9250     llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast(
9251         Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
9252     llvm::Value *CurSizeArg = Sizes[I];
9253 
9254     // Extract the MEMBER_OF field from the map type.
9255     llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member");
9256     MapperCGF.EmitBlock(MemberBB);
9257     llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]);
9258     llvm::Value *Member = MapperCGF.Builder.CreateAnd(
9259         OriMapType,
9260         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF));
9261     llvm::BasicBlock *MemberCombineBB =
9262         MapperCGF.createBasicBlock("omp.member.combine");
9263     llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type");
9264     llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member);
9265     MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB);
9266     // Add the number of pre-existing components to the MEMBER_OF field if it
9267     // is valid.
9268     MapperCGF.EmitBlock(MemberCombineBB);
9269     llvm::Value *CombinedMember =
9270         MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
9271     // Do nothing if it is not a member of previous components.
9272     MapperCGF.EmitBlock(TypeBB);
9273     llvm::PHINode *MemberMapType =
9274         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype");
9275     MemberMapType->addIncoming(OriMapType, MemberBB);
9276     MemberMapType->addIncoming(CombinedMember, MemberCombineBB);
9277 
9278     // Combine the map type inherited from user-defined mapper with that
9279     // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM
9280     // bits of the \a MapType, which is the input argument of the mapper
9281     // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM
9282     // bits of MemberMapType.
9283     // [OpenMP 5.0], 1.2.6. map-type decay.
9284     //        | alloc |  to   | from  | tofrom | release | delete
9285     // ----------------------------------------------------------
9286     // alloc  | alloc | alloc | alloc | alloc  | release | delete
9287     // to     | alloc |  to   | alloc |   to   | release | delete
9288     // from   | alloc | alloc | from  |  from  | release | delete
9289     // tofrom | alloc |  to   | from  | tofrom | release | delete
9290     llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd(
9291         MapType,
9292         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO |
9293                                    MappableExprsHandler::OMP_MAP_FROM));
9294     llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc");
9295     llvm::BasicBlock *AllocElseBB =
9296         MapperCGF.createBasicBlock("omp.type.alloc.else");
9297     llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to");
9298     llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else");
9299     llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from");
9300     llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end");
9301     llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom);
9302     MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
9303     // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM.
9304     MapperCGF.EmitBlock(AllocBB);
9305     llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd(
9306         MemberMapType,
9307         MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9308                                      MappableExprsHandler::OMP_MAP_FROM)));
9309     MapperCGF.Builder.CreateBr(EndBB);
9310     MapperCGF.EmitBlock(AllocElseBB);
9311     llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ(
9312         LeftToFrom,
9313         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO));
9314     MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
9315     // In case of to, clear OMP_MAP_FROM.
9316     MapperCGF.EmitBlock(ToBB);
9317     llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd(
9318         MemberMapType,
9319         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM));
9320     MapperCGF.Builder.CreateBr(EndBB);
9321     MapperCGF.EmitBlock(ToElseBB);
9322     llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ(
9323         LeftToFrom,
9324         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM));
9325     MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB);
9326     // In case of from, clear OMP_MAP_TO.
9327     MapperCGF.EmitBlock(FromBB);
9328     llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd(
9329         MemberMapType,
9330         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO));
9331     // In case of tofrom, do nothing.
9332     MapperCGF.EmitBlock(EndBB);
9333     llvm::PHINode *CurMapType =
9334         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype");
9335     CurMapType->addIncoming(AllocMapType, AllocBB);
9336     CurMapType->addIncoming(ToMapType, ToBB);
9337     CurMapType->addIncoming(FromMapType, FromBB);
9338     CurMapType->addIncoming(MemberMapType, ToElseBB);
9339 
9340     // TODO: call the corresponding mapper function if a user-defined mapper is
9341     // associated with this map clause.
9342     // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9343     // data structure.
9344     llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg,
9345                                      CurSizeArg, CurMapType};
9346     MapperCGF.EmitRuntimeCall(
9347         createRuntimeFunction(OMPRTL__tgt_push_mapper_component),
9348         OffloadingArgs);
9349   }
9350 
9351   // Update the pointer to point to the next element that needs to be mapped,
9352   // and check whether we have mapped all elements.
9353   llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32(
9354       PtrPHI, /*Idx0=*/1, "omp.arraymap.next");
9355   PtrPHI->addIncoming(PtrNext, BodyBB);
9356   llvm::Value *IsDone =
9357       MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone");
9358   llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit");
9359   MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
9360 
9361   MapperCGF.EmitBlock(ExitBB);
9362   // Emit array deletion if this is an array section and \p MapType indicates
9363   // that deletion is required.
9364   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
9365                              ElementSize, DoneBB, /*IsInit=*/false);
9366 
9367   // Emit the function exit block.
9368   MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true);
9369   MapperCGF.FinishFunction();
9370   UDMMap.try_emplace(D, Fn);
9371   if (CGF) {
9372     auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn);
9373     Decls.second.push_back(D);
9374   }
9375 }
9376 
9377 /// Emit the array initialization or deletion portion for user-defined mapper
9378 /// code generation. First, it evaluates whether an array section is mapped and
9379 /// whether the \a MapType instructs to delete this section. If \a IsInit is
9380 /// true, and \a MapType indicates to not delete this array, array
9381 /// initialization code is generated. If \a IsInit is false, and \a MapType
9382 /// indicates to not this array, array deletion code is generated.
9383 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel(
9384     CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base,
9385     llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType,
9386     CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) {
9387   StringRef Prefix = IsInit ? ".init" : ".del";
9388 
9389   // Evaluate if this is an array section.
9390   llvm::BasicBlock *IsDeleteBB =
9391       MapperCGF.createBasicBlock(getName({"omp.array", Prefix, ".evaldelete"}));
9392   llvm::BasicBlock *BodyBB =
9393       MapperCGF.createBasicBlock(getName({"omp.array", Prefix}));
9394   llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE(
9395       Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray");
9396   MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB);
9397 
9398   // Evaluate if we are going to delete this section.
9399   MapperCGF.EmitBlock(IsDeleteBB);
9400   llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd(
9401       MapType,
9402       MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE));
9403   llvm::Value *DeleteCond;
9404   if (IsInit) {
9405     DeleteCond = MapperCGF.Builder.CreateIsNull(
9406         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
9407   } else {
9408     DeleteCond = MapperCGF.Builder.CreateIsNotNull(
9409         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
9410   }
9411   MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB);
9412 
9413   MapperCGF.EmitBlock(BodyBB);
9414   // Get the array size by multiplying element size and element number (i.e., \p
9415   // Size).
9416   llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul(
9417       Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity()));
9418   // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves
9419   // memory allocation/deletion purpose only.
9420   llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd(
9421       MapType,
9422       MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9423                                    MappableExprsHandler::OMP_MAP_FROM)));
9424   // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9425   // data structure.
9426   llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg};
9427   MapperCGF.EmitRuntimeCall(
9428       createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs);
9429 }
9430 
9431 void CGOpenMPRuntime::emitTargetNumIterationsCall(
9432     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9433     llvm::Value *DeviceID,
9434     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9435                                      const OMPLoopDirective &D)>
9436         SizeEmitter) {
9437   OpenMPDirectiveKind Kind = D.getDirectiveKind();
9438   const OMPExecutableDirective *TD = &D;
9439   // Get nested teams distribute kind directive, if any.
9440   if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind))
9441     TD = getNestedDistributeDirective(CGM.getContext(), D);
9442   if (!TD)
9443     return;
9444   const auto *LD = cast<OMPLoopDirective>(TD);
9445   auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF,
9446                                                      PrePostActionTy &) {
9447     if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) {
9448       llvm::Value *Args[] = {DeviceID, NumIterations};
9449       CGF.EmitRuntimeCall(
9450           createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args);
9451     }
9452   };
9453   emitInlinedDirective(CGF, OMPD_unknown, CodeGen);
9454 }
9455 
9456 void CGOpenMPRuntime::emitTargetCall(
9457     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9458     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
9459     llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
9460     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9461                                      const OMPLoopDirective &D)>
9462         SizeEmitter) {
9463   if (!CGF.HaveInsertPoint())
9464     return;
9465 
9466   assert(OutlinedFn && "Invalid outlined function!");
9467 
9468   const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>();
9469   llvm::SmallVector<llvm::Value *, 16> CapturedVars;
9470   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
9471   auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF,
9472                                             PrePostActionTy &) {
9473     CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9474   };
9475   emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen);
9476 
9477   CodeGenFunction::OMPTargetDataInfo InputInfo;
9478   llvm::Value *MapTypesArray = nullptr;
9479   // Fill up the pointer arrays and transfer execution to the device.
9480   auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo,
9481                     &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars,
9482                     SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) {
9483     if (Device.getInt() == OMPC_DEVICE_ancestor) {
9484       // Reverse offloading is not supported, so just execute on the host.
9485       if (RequiresOuterTask) {
9486         CapturedVars.clear();
9487         CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9488       }
9489       emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9490       return;
9491     }
9492 
9493     // On top of the arrays that were filled up, the target offloading call
9494     // takes as arguments the device id as well as the host pointer. The host
9495     // pointer is used by the runtime library to identify the current target
9496     // region, so it only has to be unique and not necessarily point to
9497     // anything. It could be the pointer to the outlined function that
9498     // implements the target region, but we aren't using that so that the
9499     // compiler doesn't need to keep that, and could therefore inline the host
9500     // function if proven worthwhile during optimization.
9501 
9502     // From this point on, we need to have an ID of the target region defined.
9503     assert(OutlinedFnID && "Invalid outlined function ID!");
9504 
9505     // Emit device ID if any.
9506     llvm::Value *DeviceID;
9507     if (Device.getPointer()) {
9508       assert((Device.getInt() == OMPC_DEVICE_unknown ||
9509               Device.getInt() == OMPC_DEVICE_device_num) &&
9510              "Expected device_num modifier.");
9511       llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer());
9512       DeviceID =
9513           CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true);
9514     } else {
9515       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
9516     }
9517 
9518     // Emit the number of elements in the offloading arrays.
9519     llvm::Value *PointerNum =
9520         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
9521 
9522     // Return value of the runtime offloading call.
9523     llvm::Value *Return;
9524 
9525     llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D);
9526     llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D);
9527 
9528     // Emit tripcount for the target loop-based directive.
9529     emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter);
9530 
9531     bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
9532     // The target region is an outlined function launched by the runtime
9533     // via calls __tgt_target() or __tgt_target_teams().
9534     //
9535     // __tgt_target() launches a target region with one team and one thread,
9536     // executing a serial region.  This master thread may in turn launch
9537     // more threads within its team upon encountering a parallel region,
9538     // however, no additional teams can be launched on the device.
9539     //
9540     // __tgt_target_teams() launches a target region with one or more teams,
9541     // each with one or more threads.  This call is required for target
9542     // constructs such as:
9543     //  'target teams'
9544     //  'target' / 'teams'
9545     //  'target teams distribute parallel for'
9546     //  'target parallel'
9547     // and so on.
9548     //
9549     // Note that on the host and CPU targets, the runtime implementation of
9550     // these calls simply call the outlined function without forking threads.
9551     // The outlined functions themselves have runtime calls to
9552     // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by
9553     // the compiler in emitTeamsCall() and emitParallelCall().
9554     //
9555     // In contrast, on the NVPTX target, the implementation of
9556     // __tgt_target_teams() launches a GPU kernel with the requested number
9557     // of teams and threads so no additional calls to the runtime are required.
9558     if (NumTeams) {
9559       // If we have NumTeams defined this means that we have an enclosed teams
9560       // region. Therefore we also expect to have NumThreads defined. These two
9561       // values should be defined in the presence of a teams directive,
9562       // regardless of having any clauses associated. If the user is using teams
9563       // but no clauses, these two values will be the default that should be
9564       // passed to the runtime library - a 32-bit integer with the value zero.
9565       assert(NumThreads && "Thread limit expression should be available along "
9566                            "with number of teams.");
9567       llvm::Value *OffloadingArgs[] = {DeviceID,
9568                                        OutlinedFnID,
9569                                        PointerNum,
9570                                        InputInfo.BasePointersArray.getPointer(),
9571                                        InputInfo.PointersArray.getPointer(),
9572                                        InputInfo.SizesArray.getPointer(),
9573                                        MapTypesArray,
9574                                        NumTeams,
9575                                        NumThreads};
9576       Return = CGF.EmitRuntimeCall(
9577           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait
9578                                           : OMPRTL__tgt_target_teams),
9579           OffloadingArgs);
9580     } else {
9581       llvm::Value *OffloadingArgs[] = {DeviceID,
9582                                        OutlinedFnID,
9583                                        PointerNum,
9584                                        InputInfo.BasePointersArray.getPointer(),
9585                                        InputInfo.PointersArray.getPointer(),
9586                                        InputInfo.SizesArray.getPointer(),
9587                                        MapTypesArray};
9588       Return = CGF.EmitRuntimeCall(
9589           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait
9590                                           : OMPRTL__tgt_target),
9591           OffloadingArgs);
9592     }
9593 
9594     // Check the error code and execute the host version if required.
9595     llvm::BasicBlock *OffloadFailedBlock =
9596         CGF.createBasicBlock("omp_offload.failed");
9597     llvm::BasicBlock *OffloadContBlock =
9598         CGF.createBasicBlock("omp_offload.cont");
9599     llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return);
9600     CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock);
9601 
9602     CGF.EmitBlock(OffloadFailedBlock);
9603     if (RequiresOuterTask) {
9604       CapturedVars.clear();
9605       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9606     }
9607     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9608     CGF.EmitBranch(OffloadContBlock);
9609 
9610     CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true);
9611   };
9612 
9613   // Notify that the host version must be executed.
9614   auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars,
9615                     RequiresOuterTask](CodeGenFunction &CGF,
9616                                        PrePostActionTy &) {
9617     if (RequiresOuterTask) {
9618       CapturedVars.clear();
9619       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9620     }
9621     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9622   };
9623 
9624   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
9625                           &CapturedVars, RequiresOuterTask,
9626                           &CS](CodeGenFunction &CGF, PrePostActionTy &) {
9627     // Fill up the arrays with all the captured variables.
9628     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9629     MappableExprsHandler::MapValuesArrayTy Pointers;
9630     MappableExprsHandler::MapValuesArrayTy Sizes;
9631     MappableExprsHandler::MapFlagsArrayTy MapTypes;
9632 
9633     // Get mappable expression information.
9634     MappableExprsHandler MEHandler(D, CGF);
9635     llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers;
9636 
9637     auto RI = CS.getCapturedRecordDecl()->field_begin();
9638     auto CV = CapturedVars.begin();
9639     for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(),
9640                                               CE = CS.capture_end();
9641          CI != CE; ++CI, ++RI, ++CV) {
9642       MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers;
9643       MappableExprsHandler::MapValuesArrayTy CurPointers;
9644       MappableExprsHandler::MapValuesArrayTy CurSizes;
9645       MappableExprsHandler::MapFlagsArrayTy CurMapTypes;
9646       MappableExprsHandler::StructRangeInfoTy PartialStruct;
9647 
9648       // VLA sizes are passed to the outlined region by copy and do not have map
9649       // information associated.
9650       if (CI->capturesVariableArrayType()) {
9651         CurBasePointers.push_back(*CV);
9652         CurPointers.push_back(*CV);
9653         CurSizes.push_back(CGF.Builder.CreateIntCast(
9654             CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true));
9655         // Copy to the device as an argument. No need to retrieve it.
9656         CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL |
9657                               MappableExprsHandler::OMP_MAP_TARGET_PARAM |
9658                               MappableExprsHandler::OMP_MAP_IMPLICIT);
9659       } else {
9660         // If we have any information in the map clause, we use it, otherwise we
9661         // just do a default mapping.
9662         MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers,
9663                                          CurSizes, CurMapTypes, PartialStruct);
9664         if (CurBasePointers.empty())
9665           MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers,
9666                                            CurPointers, CurSizes, CurMapTypes);
9667         // Generate correct mapping for variables captured by reference in
9668         // lambdas.
9669         if (CI->capturesVariable())
9670           MEHandler.generateInfoForLambdaCaptures(
9671               CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes,
9672               CurMapTypes, LambdaPointers);
9673       }
9674       // We expect to have at least an element of information for this capture.
9675       assert(!CurBasePointers.empty() &&
9676              "Non-existing map pointer for capture!");
9677       assert(CurBasePointers.size() == CurPointers.size() &&
9678              CurBasePointers.size() == CurSizes.size() &&
9679              CurBasePointers.size() == CurMapTypes.size() &&
9680              "Inconsistent map information sizes!");
9681 
9682       // If there is an entry in PartialStruct it means we have a struct with
9683       // individual members mapped. Emit an extra combined entry.
9684       if (PartialStruct.Base.isValid())
9685         MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes,
9686                                     CurMapTypes, PartialStruct);
9687 
9688       // We need to append the results of this capture to what we already have.
9689       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
9690       Pointers.append(CurPointers.begin(), CurPointers.end());
9691       Sizes.append(CurSizes.begin(), CurSizes.end());
9692       MapTypes.append(CurMapTypes.begin(), CurMapTypes.end());
9693     }
9694     // Adjust MEMBER_OF flags for the lambdas captures.
9695     MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers,
9696                                               Pointers, MapTypes);
9697     // Map other list items in the map clause which are not captured variables
9698     // but "declare target link" global variables.
9699     MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes,
9700                                                MapTypes);
9701 
9702     TargetDataInfo Info;
9703     // Fill up the arrays and create the arguments.
9704     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
9705     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
9706                                  Info.PointersArray, Info.SizesArray,
9707                                  Info.MapTypesArray, Info);
9708     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
9709     InputInfo.BasePointersArray =
9710         Address(Info.BasePointersArray, CGM.getPointerAlign());
9711     InputInfo.PointersArray =
9712         Address(Info.PointersArray, CGM.getPointerAlign());
9713     InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign());
9714     MapTypesArray = Info.MapTypesArray;
9715     if (RequiresOuterTask)
9716       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
9717     else
9718       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
9719   };
9720 
9721   auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask](
9722                              CodeGenFunction &CGF, PrePostActionTy &) {
9723     if (RequiresOuterTask) {
9724       CodeGenFunction::OMPTargetDataInfo InputInfo;
9725       CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo);
9726     } else {
9727       emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen);
9728     }
9729   };
9730 
9731   // If we have a target function ID it means that we need to support
9732   // offloading, otherwise, just execute on the host. We need to execute on host
9733   // regardless of the conditional in the if clause if, e.g., the user do not
9734   // specify target triples.
9735   if (OutlinedFnID) {
9736     if (IfCond) {
9737       emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen);
9738     } else {
9739       RegionCodeGenTy ThenRCG(TargetThenGen);
9740       ThenRCG(CGF);
9741     }
9742   } else {
9743     RegionCodeGenTy ElseRCG(TargetElseGen);
9744     ElseRCG(CGF);
9745   }
9746 }
9747 
9748 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S,
9749                                                     StringRef ParentName) {
9750   if (!S)
9751     return;
9752 
9753   // Codegen OMP target directives that offload compute to the device.
9754   bool RequiresDeviceCodegen =
9755       isa<OMPExecutableDirective>(S) &&
9756       isOpenMPTargetExecutionDirective(
9757           cast<OMPExecutableDirective>(S)->getDirectiveKind());
9758 
9759   if (RequiresDeviceCodegen) {
9760     const auto &E = *cast<OMPExecutableDirective>(S);
9761     unsigned DeviceID;
9762     unsigned FileID;
9763     unsigned Line;
9764     getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID,
9765                              FileID, Line);
9766 
9767     // Is this a target region that should not be emitted as an entry point? If
9768     // so just signal we are done with this target region.
9769     if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID,
9770                                                             ParentName, Line))
9771       return;
9772 
9773     switch (E.getDirectiveKind()) {
9774     case OMPD_target:
9775       CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName,
9776                                                    cast<OMPTargetDirective>(E));
9777       break;
9778     case OMPD_target_parallel:
9779       CodeGenFunction::EmitOMPTargetParallelDeviceFunction(
9780           CGM, ParentName, cast<OMPTargetParallelDirective>(E));
9781       break;
9782     case OMPD_target_teams:
9783       CodeGenFunction::EmitOMPTargetTeamsDeviceFunction(
9784           CGM, ParentName, cast<OMPTargetTeamsDirective>(E));
9785       break;
9786     case OMPD_target_teams_distribute:
9787       CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction(
9788           CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E));
9789       break;
9790     case OMPD_target_teams_distribute_simd:
9791       CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction(
9792           CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E));
9793       break;
9794     case OMPD_target_parallel_for:
9795       CodeGenFunction::EmitOMPTargetParallelForDeviceFunction(
9796           CGM, ParentName, cast<OMPTargetParallelForDirective>(E));
9797       break;
9798     case OMPD_target_parallel_for_simd:
9799       CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction(
9800           CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E));
9801       break;
9802     case OMPD_target_simd:
9803       CodeGenFunction::EmitOMPTargetSimdDeviceFunction(
9804           CGM, ParentName, cast<OMPTargetSimdDirective>(E));
9805       break;
9806     case OMPD_target_teams_distribute_parallel_for:
9807       CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction(
9808           CGM, ParentName,
9809           cast<OMPTargetTeamsDistributeParallelForDirective>(E));
9810       break;
9811     case OMPD_target_teams_distribute_parallel_for_simd:
9812       CodeGenFunction::
9813           EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(
9814               CGM, ParentName,
9815               cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E));
9816       break;
9817     case OMPD_parallel:
9818     case OMPD_for:
9819     case OMPD_parallel_for:
9820     case OMPD_parallel_master:
9821     case OMPD_parallel_sections:
9822     case OMPD_for_simd:
9823     case OMPD_parallel_for_simd:
9824     case OMPD_cancel:
9825     case OMPD_cancellation_point:
9826     case OMPD_ordered:
9827     case OMPD_threadprivate:
9828     case OMPD_allocate:
9829     case OMPD_task:
9830     case OMPD_simd:
9831     case OMPD_sections:
9832     case OMPD_section:
9833     case OMPD_single:
9834     case OMPD_master:
9835     case OMPD_critical:
9836     case OMPD_taskyield:
9837     case OMPD_barrier:
9838     case OMPD_taskwait:
9839     case OMPD_taskgroup:
9840     case OMPD_atomic:
9841     case OMPD_flush:
9842     case OMPD_depobj:
9843     case OMPD_scan:
9844     case OMPD_teams:
9845     case OMPD_target_data:
9846     case OMPD_target_exit_data:
9847     case OMPD_target_enter_data:
9848     case OMPD_distribute:
9849     case OMPD_distribute_simd:
9850     case OMPD_distribute_parallel_for:
9851     case OMPD_distribute_parallel_for_simd:
9852     case OMPD_teams_distribute:
9853     case OMPD_teams_distribute_simd:
9854     case OMPD_teams_distribute_parallel_for:
9855     case OMPD_teams_distribute_parallel_for_simd:
9856     case OMPD_target_update:
9857     case OMPD_declare_simd:
9858     case OMPD_declare_variant:
9859     case OMPD_begin_declare_variant:
9860     case OMPD_end_declare_variant:
9861     case OMPD_declare_target:
9862     case OMPD_end_declare_target:
9863     case OMPD_declare_reduction:
9864     case OMPD_declare_mapper:
9865     case OMPD_taskloop:
9866     case OMPD_taskloop_simd:
9867     case OMPD_master_taskloop:
9868     case OMPD_master_taskloop_simd:
9869     case OMPD_parallel_master_taskloop:
9870     case OMPD_parallel_master_taskloop_simd:
9871     case OMPD_requires:
9872     case OMPD_unknown:
9873       llvm_unreachable("Unknown target directive for OpenMP device codegen.");
9874     }
9875     return;
9876   }
9877 
9878   if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) {
9879     if (!E->hasAssociatedStmt() || !E->getAssociatedStmt())
9880       return;
9881 
9882     scanForTargetRegionsFunctions(
9883         E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName);
9884     return;
9885   }
9886 
9887   // If this is a lambda function, look into its body.
9888   if (const auto *L = dyn_cast<LambdaExpr>(S))
9889     S = L->getBody();
9890 
9891   // Keep looking for target regions recursively.
9892   for (const Stmt *II : S->children())
9893     scanForTargetRegionsFunctions(II, ParentName);
9894 }
9895 
9896 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) {
9897   // If emitting code for the host, we do not process FD here. Instead we do
9898   // the normal code generation.
9899   if (!CGM.getLangOpts().OpenMPIsDevice) {
9900     if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) {
9901       Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9902           OMPDeclareTargetDeclAttr::getDeviceType(FD);
9903       // Do not emit device_type(nohost) functions for the host.
9904       if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost)
9905         return true;
9906     }
9907     return false;
9908   }
9909 
9910   const ValueDecl *VD = cast<ValueDecl>(GD.getDecl());
9911   // Try to detect target regions in the function.
9912   if (const auto *FD = dyn_cast<FunctionDecl>(VD)) {
9913     StringRef Name = CGM.getMangledName(GD);
9914     scanForTargetRegionsFunctions(FD->getBody(), Name);
9915     Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9916         OMPDeclareTargetDeclAttr::getDeviceType(FD);
9917     // Do not emit device_type(nohost) functions for the host.
9918     if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host)
9919       return true;
9920   }
9921 
9922   // Do not to emit function if it is not marked as declare target.
9923   return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) &&
9924          AlreadyEmittedTargetDecls.count(VD) == 0;
9925 }
9926 
9927 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
9928   if (!CGM.getLangOpts().OpenMPIsDevice)
9929     return false;
9930 
9931   // Check if there are Ctors/Dtors in this declaration and look for target
9932   // regions in it. We use the complete variant to produce the kernel name
9933   // mangling.
9934   QualType RDTy = cast<VarDecl>(GD.getDecl())->getType();
9935   if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) {
9936     for (const CXXConstructorDecl *Ctor : RD->ctors()) {
9937       StringRef ParentName =
9938           CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete));
9939       scanForTargetRegionsFunctions(Ctor->getBody(), ParentName);
9940     }
9941     if (const CXXDestructorDecl *Dtor = RD->getDestructor()) {
9942       StringRef ParentName =
9943           CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete));
9944       scanForTargetRegionsFunctions(Dtor->getBody(), ParentName);
9945     }
9946   }
9947 
9948   // Do not to emit variable if it is not marked as declare target.
9949   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9950       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(
9951           cast<VarDecl>(GD.getDecl()));
9952   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
9953       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9954        HasRequiresUnifiedSharedMemory)) {
9955     DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl()));
9956     return true;
9957   }
9958   return false;
9959 }
9960 
9961 llvm::Constant *
9962 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF,
9963                                                 const VarDecl *VD) {
9964   assert(VD->getType().isConstant(CGM.getContext()) &&
9965          "Expected constant variable.");
9966   StringRef VarName;
9967   llvm::Constant *Addr;
9968   llvm::GlobalValue::LinkageTypes Linkage;
9969   QualType Ty = VD->getType();
9970   SmallString<128> Buffer;
9971   {
9972     unsigned DeviceID;
9973     unsigned FileID;
9974     unsigned Line;
9975     getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID,
9976                              FileID, Line);
9977     llvm::raw_svector_ostream OS(Buffer);
9978     OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID)
9979        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
9980     VarName = OS.str();
9981   }
9982   Linkage = llvm::GlobalValue::InternalLinkage;
9983   Addr =
9984       getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName,
9985                                   getDefaultFirstprivateAddressSpace());
9986   cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage);
9987   CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty);
9988   CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr));
9989   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
9990       VarName, Addr, VarSize,
9991       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage);
9992   return Addr;
9993 }
9994 
9995 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD,
9996                                                    llvm::Constant *Addr) {
9997   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
9998       !CGM.getLangOpts().OpenMPIsDevice)
9999     return;
10000   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
10001       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
10002   if (!Res) {
10003     if (CGM.getLangOpts().OpenMPIsDevice) {
10004       // Register non-target variables being emitted in device code (debug info
10005       // may cause this).
10006       StringRef VarName = CGM.getMangledName(VD);
10007       EmittedNonTargetVariables.try_emplace(VarName, Addr);
10008     }
10009     return;
10010   }
10011   // Register declare target variables.
10012   OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags;
10013   StringRef VarName;
10014   CharUnits VarSize;
10015   llvm::GlobalValue::LinkageTypes Linkage;
10016 
10017   if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10018       !HasRequiresUnifiedSharedMemory) {
10019     Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
10020     VarName = CGM.getMangledName(VD);
10021     if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) {
10022       VarSize = CGM.getContext().getTypeSizeInChars(VD->getType());
10023       assert(!VarSize.isZero() && "Expected non-zero size of the variable");
10024     } else {
10025       VarSize = CharUnits::Zero();
10026     }
10027     Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false);
10028     // Temp solution to prevent optimizations of the internal variables.
10029     if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) {
10030       std::string RefName = getName({VarName, "ref"});
10031       if (!CGM.GetGlobalValue(RefName)) {
10032         llvm::Constant *AddrRef =
10033             getOrCreateInternalVariable(Addr->getType(), RefName);
10034         auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef);
10035         GVAddrRef->setConstant(/*Val=*/true);
10036         GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage);
10037         GVAddrRef->setInitializer(Addr);
10038         CGM.addCompilerUsedGlobal(GVAddrRef);
10039       }
10040     }
10041   } else {
10042     assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
10043             (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10044              HasRequiresUnifiedSharedMemory)) &&
10045            "Declare target attribute must link or to with unified memory.");
10046     if (*Res == OMPDeclareTargetDeclAttr::MT_Link)
10047       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink;
10048     else
10049       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
10050 
10051     if (CGM.getLangOpts().OpenMPIsDevice) {
10052       VarName = Addr->getName();
10053       Addr = nullptr;
10054     } else {
10055       VarName = getAddrOfDeclareTargetVar(VD).getName();
10056       Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer());
10057     }
10058     VarSize = CGM.getPointerSize();
10059     Linkage = llvm::GlobalValue::WeakAnyLinkage;
10060   }
10061 
10062   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
10063       VarName, Addr, VarSize, Flags, Linkage);
10064 }
10065 
10066 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) {
10067   if (isa<FunctionDecl>(GD.getDecl()) ||
10068       isa<OMPDeclareReductionDecl>(GD.getDecl()))
10069     return emitTargetFunctions(GD);
10070 
10071   return emitTargetGlobalVariable(GD);
10072 }
10073 
10074 void CGOpenMPRuntime::emitDeferredTargetDecls() const {
10075   for (const VarDecl *VD : DeferredGlobalVariables) {
10076     llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
10077         OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
10078     if (!Res)
10079       continue;
10080     if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10081         !HasRequiresUnifiedSharedMemory) {
10082       CGM.EmitGlobal(VD);
10083     } else {
10084       assert((*Res == OMPDeclareTargetDeclAttr::MT_Link ||
10085               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10086                HasRequiresUnifiedSharedMemory)) &&
10087              "Expected link clause or to clause with unified memory.");
10088       (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
10089     }
10090   }
10091 }
10092 
10093 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas(
10094     CodeGenFunction &CGF, const OMPExecutableDirective &D) const {
10095   assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) &&
10096          " Expected target-based directive.");
10097 }
10098 
10099 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) {
10100   for (const OMPClause *Clause : D->clauselists()) {
10101     if (Clause->getClauseKind() == OMPC_unified_shared_memory) {
10102       HasRequiresUnifiedSharedMemory = true;
10103     } else if (const auto *AC =
10104                    dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) {
10105       switch (AC->getAtomicDefaultMemOrderKind()) {
10106       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel:
10107         RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease;
10108         break;
10109       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst:
10110         RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent;
10111         break;
10112       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed:
10113         RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic;
10114         break;
10115       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown:
10116         break;
10117       }
10118     }
10119   }
10120 }
10121 
10122 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const {
10123   return RequiresAtomicOrdering;
10124 }
10125 
10126 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD,
10127                                                        LangAS &AS) {
10128   if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())
10129     return false;
10130   const auto *A = VD->getAttr<OMPAllocateDeclAttr>();
10131   switch(A->getAllocatorType()) {
10132   case OMPAllocateDeclAttr::OMPDefaultMemAlloc:
10133   // Not supported, fallback to the default mem space.
10134   case OMPAllocateDeclAttr::OMPLargeCapMemAlloc:
10135   case OMPAllocateDeclAttr::OMPCGroupMemAlloc:
10136   case OMPAllocateDeclAttr::OMPHighBWMemAlloc:
10137   case OMPAllocateDeclAttr::OMPLowLatMemAlloc:
10138   case OMPAllocateDeclAttr::OMPThreadMemAlloc:
10139   case OMPAllocateDeclAttr::OMPConstMemAlloc:
10140   case OMPAllocateDeclAttr::OMPPTeamMemAlloc:
10141     AS = LangAS::Default;
10142     return true;
10143   case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc:
10144     llvm_unreachable("Expected predefined allocator for the variables with the "
10145                      "static storage.");
10146   }
10147   return false;
10148 }
10149 
10150 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const {
10151   return HasRequiresUnifiedSharedMemory;
10152 }
10153 
10154 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII(
10155     CodeGenModule &CGM)
10156     : CGM(CGM) {
10157   if (CGM.getLangOpts().OpenMPIsDevice) {
10158     SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal;
10159     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false;
10160   }
10161 }
10162 
10163 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() {
10164   if (CGM.getLangOpts().OpenMPIsDevice)
10165     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal;
10166 }
10167 
10168 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) {
10169   if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal)
10170     return true;
10171 
10172   const auto *D = cast<FunctionDecl>(GD.getDecl());
10173   // Do not to emit function if it is marked as declare target as it was already
10174   // emitted.
10175   if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) {
10176     if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) {
10177       if (auto *F = dyn_cast_or_null<llvm::Function>(
10178               CGM.GetGlobalValue(CGM.getMangledName(GD))))
10179         return !F->isDeclaration();
10180       return false;
10181     }
10182     return true;
10183   }
10184 
10185   return !AlreadyEmittedTargetDecls.insert(D).second;
10186 }
10187 
10188 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() {
10189   // If we don't have entries or if we are emitting code for the device, we
10190   // don't need to do anything.
10191   if (CGM.getLangOpts().OMPTargetTriples.empty() ||
10192       CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice ||
10193       (OffloadEntriesInfoManager.empty() &&
10194        !HasEmittedDeclareTargetRegion &&
10195        !HasEmittedTargetRegion))
10196     return nullptr;
10197 
10198   // Create and register the function that handles the requires directives.
10199   ASTContext &C = CGM.getContext();
10200 
10201   llvm::Function *RequiresRegFn;
10202   {
10203     CodeGenFunction CGF(CGM);
10204     const auto &FI = CGM.getTypes().arrangeNullaryFunction();
10205     llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
10206     std::string ReqName = getName({"omp_offloading", "requires_reg"});
10207     RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI);
10208     CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {});
10209     OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE;
10210     // TODO: check for other requires clauses.
10211     // The requires directive takes effect only when a target region is
10212     // present in the compilation unit. Otherwise it is ignored and not
10213     // passed to the runtime. This avoids the runtime from throwing an error
10214     // for mismatching requires clauses across compilation units that don't
10215     // contain at least 1 target region.
10216     assert((HasEmittedTargetRegion ||
10217             HasEmittedDeclareTargetRegion ||
10218             !OffloadEntriesInfoManager.empty()) &&
10219            "Target or declare target region expected.");
10220     if (HasRequiresUnifiedSharedMemory)
10221       Flags = OMP_REQ_UNIFIED_SHARED_MEMORY;
10222     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires),
10223         llvm::ConstantInt::get(CGM.Int64Ty, Flags));
10224     CGF.FinishFunction();
10225   }
10226   return RequiresRegFn;
10227 }
10228 
10229 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF,
10230                                     const OMPExecutableDirective &D,
10231                                     SourceLocation Loc,
10232                                     llvm::Function *OutlinedFn,
10233                                     ArrayRef<llvm::Value *> CapturedVars) {
10234   if (!CGF.HaveInsertPoint())
10235     return;
10236 
10237   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
10238   CodeGenFunction::RunCleanupsScope Scope(CGF);
10239 
10240   // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn);
10241   llvm::Value *Args[] = {
10242       RTLoc,
10243       CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
10244       CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())};
10245   llvm::SmallVector<llvm::Value *, 16> RealArgs;
10246   RealArgs.append(std::begin(Args), std::end(Args));
10247   RealArgs.append(CapturedVars.begin(), CapturedVars.end());
10248 
10249   llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams);
10250   CGF.EmitRuntimeCall(RTLFn, RealArgs);
10251 }
10252 
10253 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
10254                                          const Expr *NumTeams,
10255                                          const Expr *ThreadLimit,
10256                                          SourceLocation Loc) {
10257   if (!CGF.HaveInsertPoint())
10258     return;
10259 
10260   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
10261 
10262   llvm::Value *NumTeamsVal =
10263       NumTeams
10264           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams),
10265                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
10266           : CGF.Builder.getInt32(0);
10267 
10268   llvm::Value *ThreadLimitVal =
10269       ThreadLimit
10270           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit),
10271                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
10272           : CGF.Builder.getInt32(0);
10273 
10274   // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit)
10275   llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal,
10276                                      ThreadLimitVal};
10277   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams),
10278                       PushNumTeamsArgs);
10279 }
10280 
10281 void CGOpenMPRuntime::emitTargetDataCalls(
10282     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
10283     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
10284   if (!CGF.HaveInsertPoint())
10285     return;
10286 
10287   // Action used to replace the default codegen action and turn privatization
10288   // off.
10289   PrePostActionTy NoPrivAction;
10290 
10291   // Generate the code for the opening of the data environment. Capture all the
10292   // arguments of the runtime call by reference because they are used in the
10293   // closing of the region.
10294   auto &&BeginThenGen = [this, &D, Device, &Info,
10295                          &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) {
10296     // Fill up the arrays with all the mapped variables.
10297     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
10298     MappableExprsHandler::MapValuesArrayTy Pointers;
10299     MappableExprsHandler::MapValuesArrayTy Sizes;
10300     MappableExprsHandler::MapFlagsArrayTy MapTypes;
10301 
10302     // Get map clause information.
10303     MappableExprsHandler MCHandler(D, CGF);
10304     MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
10305 
10306     // Fill up the arrays and create the arguments.
10307     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
10308 
10309     llvm::Value *BasePointersArrayArg = nullptr;
10310     llvm::Value *PointersArrayArg = nullptr;
10311     llvm::Value *SizesArrayArg = nullptr;
10312     llvm::Value *MapTypesArrayArg = nullptr;
10313     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10314                                  SizesArrayArg, MapTypesArrayArg, Info);
10315 
10316     // Emit device ID if any.
10317     llvm::Value *DeviceID = nullptr;
10318     if (Device) {
10319       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10320                                            CGF.Int64Ty, /*isSigned=*/true);
10321     } else {
10322       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10323     }
10324 
10325     // Emit the number of elements in the offloading arrays.
10326     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10327 
10328     llvm::Value *OffloadingArgs[] = {
10329         DeviceID,         PointerNum,    BasePointersArrayArg,
10330         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10331     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin),
10332                         OffloadingArgs);
10333 
10334     // If device pointer privatization is required, emit the body of the region
10335     // here. It will have to be duplicated: with and without privatization.
10336     if (!Info.CaptureDeviceAddrMap.empty())
10337       CodeGen(CGF);
10338   };
10339 
10340   // Generate code for the closing of the data region.
10341   auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF,
10342                                             PrePostActionTy &) {
10343     assert(Info.isValid() && "Invalid data environment closing arguments.");
10344 
10345     llvm::Value *BasePointersArrayArg = nullptr;
10346     llvm::Value *PointersArrayArg = nullptr;
10347     llvm::Value *SizesArrayArg = nullptr;
10348     llvm::Value *MapTypesArrayArg = nullptr;
10349     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10350                                  SizesArrayArg, MapTypesArrayArg, Info);
10351 
10352     // Emit device ID if any.
10353     llvm::Value *DeviceID = nullptr;
10354     if (Device) {
10355       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10356                                            CGF.Int64Ty, /*isSigned=*/true);
10357     } else {
10358       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10359     }
10360 
10361     // Emit the number of elements in the offloading arrays.
10362     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10363 
10364     llvm::Value *OffloadingArgs[] = {
10365         DeviceID,         PointerNum,    BasePointersArrayArg,
10366         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10367     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end),
10368                         OffloadingArgs);
10369   };
10370 
10371   // If we need device pointer privatization, we need to emit the body of the
10372   // region with no privatization in the 'else' branch of the conditional.
10373   // Otherwise, we don't have to do anything.
10374   auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF,
10375                                                          PrePostActionTy &) {
10376     if (!Info.CaptureDeviceAddrMap.empty()) {
10377       CodeGen.setAction(NoPrivAction);
10378       CodeGen(CGF);
10379     }
10380   };
10381 
10382   // We don't have to do anything to close the region if the if clause evaluates
10383   // to false.
10384   auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {};
10385 
10386   if (IfCond) {
10387     emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen);
10388   } else {
10389     RegionCodeGenTy RCG(BeginThenGen);
10390     RCG(CGF);
10391   }
10392 
10393   // If we don't require privatization of device pointers, we emit the body in
10394   // between the runtime calls. This avoids duplicating the body code.
10395   if (Info.CaptureDeviceAddrMap.empty()) {
10396     CodeGen.setAction(NoPrivAction);
10397     CodeGen(CGF);
10398   }
10399 
10400   if (IfCond) {
10401     emitIfClause(CGF, IfCond, EndThenGen, EndElseGen);
10402   } else {
10403     RegionCodeGenTy RCG(EndThenGen);
10404     RCG(CGF);
10405   }
10406 }
10407 
10408 void CGOpenMPRuntime::emitTargetDataStandAloneCall(
10409     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
10410     const Expr *Device) {
10411   if (!CGF.HaveInsertPoint())
10412     return;
10413 
10414   assert((isa<OMPTargetEnterDataDirective>(D) ||
10415           isa<OMPTargetExitDataDirective>(D) ||
10416           isa<OMPTargetUpdateDirective>(D)) &&
10417          "Expecting either target enter, exit data, or update directives.");
10418 
10419   CodeGenFunction::OMPTargetDataInfo InputInfo;
10420   llvm::Value *MapTypesArray = nullptr;
10421   // Generate the code for the opening of the data environment.
10422   auto &&ThenGen = [this, &D, Device, &InputInfo,
10423                     &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) {
10424     // Emit device ID if any.
10425     llvm::Value *DeviceID = nullptr;
10426     if (Device) {
10427       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10428                                            CGF.Int64Ty, /*isSigned=*/true);
10429     } else {
10430       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10431     }
10432 
10433     // Emit the number of elements in the offloading arrays.
10434     llvm::Constant *PointerNum =
10435         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
10436 
10437     llvm::Value *OffloadingArgs[] = {DeviceID,
10438                                      PointerNum,
10439                                      InputInfo.BasePointersArray.getPointer(),
10440                                      InputInfo.PointersArray.getPointer(),
10441                                      InputInfo.SizesArray.getPointer(),
10442                                      MapTypesArray};
10443 
10444     // Select the right runtime function call for each expected standalone
10445     // directive.
10446     const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
10447     OpenMPRTLFunction RTLFn;
10448     switch (D.getDirectiveKind()) {
10449     case OMPD_target_enter_data:
10450       RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait
10451                         : OMPRTL__tgt_target_data_begin;
10452       break;
10453     case OMPD_target_exit_data:
10454       RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait
10455                         : OMPRTL__tgt_target_data_end;
10456       break;
10457     case OMPD_target_update:
10458       RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait
10459                         : OMPRTL__tgt_target_data_update;
10460       break;
10461     case OMPD_parallel:
10462     case OMPD_for:
10463     case OMPD_parallel_for:
10464     case OMPD_parallel_master:
10465     case OMPD_parallel_sections:
10466     case OMPD_for_simd:
10467     case OMPD_parallel_for_simd:
10468     case OMPD_cancel:
10469     case OMPD_cancellation_point:
10470     case OMPD_ordered:
10471     case OMPD_threadprivate:
10472     case OMPD_allocate:
10473     case OMPD_task:
10474     case OMPD_simd:
10475     case OMPD_sections:
10476     case OMPD_section:
10477     case OMPD_single:
10478     case OMPD_master:
10479     case OMPD_critical:
10480     case OMPD_taskyield:
10481     case OMPD_barrier:
10482     case OMPD_taskwait:
10483     case OMPD_taskgroup:
10484     case OMPD_atomic:
10485     case OMPD_flush:
10486     case OMPD_depobj:
10487     case OMPD_scan:
10488     case OMPD_teams:
10489     case OMPD_target_data:
10490     case OMPD_distribute:
10491     case OMPD_distribute_simd:
10492     case OMPD_distribute_parallel_for:
10493     case OMPD_distribute_parallel_for_simd:
10494     case OMPD_teams_distribute:
10495     case OMPD_teams_distribute_simd:
10496     case OMPD_teams_distribute_parallel_for:
10497     case OMPD_teams_distribute_parallel_for_simd:
10498     case OMPD_declare_simd:
10499     case OMPD_declare_variant:
10500     case OMPD_begin_declare_variant:
10501     case OMPD_end_declare_variant:
10502     case OMPD_declare_target:
10503     case OMPD_end_declare_target:
10504     case OMPD_declare_reduction:
10505     case OMPD_declare_mapper:
10506     case OMPD_taskloop:
10507     case OMPD_taskloop_simd:
10508     case OMPD_master_taskloop:
10509     case OMPD_master_taskloop_simd:
10510     case OMPD_parallel_master_taskloop:
10511     case OMPD_parallel_master_taskloop_simd:
10512     case OMPD_target:
10513     case OMPD_target_simd:
10514     case OMPD_target_teams_distribute:
10515     case OMPD_target_teams_distribute_simd:
10516     case OMPD_target_teams_distribute_parallel_for:
10517     case OMPD_target_teams_distribute_parallel_for_simd:
10518     case OMPD_target_teams:
10519     case OMPD_target_parallel:
10520     case OMPD_target_parallel_for:
10521     case OMPD_target_parallel_for_simd:
10522     case OMPD_requires:
10523     case OMPD_unknown:
10524       llvm_unreachable("Unexpected standalone target data directive.");
10525       break;
10526     }
10527     CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs);
10528   };
10529 
10530   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray](
10531                              CodeGenFunction &CGF, PrePostActionTy &) {
10532     // Fill up the arrays with all the mapped variables.
10533     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
10534     MappableExprsHandler::MapValuesArrayTy Pointers;
10535     MappableExprsHandler::MapValuesArrayTy Sizes;
10536     MappableExprsHandler::MapFlagsArrayTy MapTypes;
10537 
10538     // Get map clause information.
10539     MappableExprsHandler MEHandler(D, CGF);
10540     MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
10541 
10542     TargetDataInfo Info;
10543     // Fill up the arrays and create the arguments.
10544     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
10545     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
10546                                  Info.PointersArray, Info.SizesArray,
10547                                  Info.MapTypesArray, Info);
10548     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
10549     InputInfo.BasePointersArray =
10550         Address(Info.BasePointersArray, CGM.getPointerAlign());
10551     InputInfo.PointersArray =
10552         Address(Info.PointersArray, CGM.getPointerAlign());
10553     InputInfo.SizesArray =
10554         Address(Info.SizesArray, CGM.getPointerAlign());
10555     MapTypesArray = Info.MapTypesArray;
10556     if (D.hasClausesOfKind<OMPDependClause>())
10557       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
10558     else
10559       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
10560   };
10561 
10562   if (IfCond) {
10563     emitIfClause(CGF, IfCond, TargetThenGen,
10564                  [](CodeGenFunction &CGF, PrePostActionTy &) {});
10565   } else {
10566     RegionCodeGenTy ThenRCG(TargetThenGen);
10567     ThenRCG(CGF);
10568   }
10569 }
10570 
10571 namespace {
10572   /// Kind of parameter in a function with 'declare simd' directive.
10573   enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector };
10574   /// Attribute set of the parameter.
10575   struct ParamAttrTy {
10576     ParamKindTy Kind = Vector;
10577     llvm::APSInt StrideOrArg;
10578     llvm::APSInt Alignment;
10579   };
10580 } // namespace
10581 
10582 static unsigned evaluateCDTSize(const FunctionDecl *FD,
10583                                 ArrayRef<ParamAttrTy> ParamAttrs) {
10584   // Every vector variant of a SIMD-enabled function has a vector length (VLEN).
10585   // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument
10586   // of that clause. The VLEN value must be power of 2.
10587   // In other case the notion of the function`s "characteristic data type" (CDT)
10588   // is used to compute the vector length.
10589   // CDT is defined in the following order:
10590   //   a) For non-void function, the CDT is the return type.
10591   //   b) If the function has any non-uniform, non-linear parameters, then the
10592   //   CDT is the type of the first such parameter.
10593   //   c) If the CDT determined by a) or b) above is struct, union, or class
10594   //   type which is pass-by-value (except for the type that maps to the
10595   //   built-in complex data type), the characteristic data type is int.
10596   //   d) If none of the above three cases is applicable, the CDT is int.
10597   // The VLEN is then determined based on the CDT and the size of vector
10598   // register of that ISA for which current vector version is generated. The
10599   // VLEN is computed using the formula below:
10600   //   VLEN  = sizeof(vector_register) / sizeof(CDT),
10601   // where vector register size specified in section 3.2.1 Registers and the
10602   // Stack Frame of original AMD64 ABI document.
10603   QualType RetType = FD->getReturnType();
10604   if (RetType.isNull())
10605     return 0;
10606   ASTContext &C = FD->getASTContext();
10607   QualType CDT;
10608   if (!RetType.isNull() && !RetType->isVoidType()) {
10609     CDT = RetType;
10610   } else {
10611     unsigned Offset = 0;
10612     if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) {
10613       if (ParamAttrs[Offset].Kind == Vector)
10614         CDT = C.getPointerType(C.getRecordType(MD->getParent()));
10615       ++Offset;
10616     }
10617     if (CDT.isNull()) {
10618       for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10619         if (ParamAttrs[I + Offset].Kind == Vector) {
10620           CDT = FD->getParamDecl(I)->getType();
10621           break;
10622         }
10623       }
10624     }
10625   }
10626   if (CDT.isNull())
10627     CDT = C.IntTy;
10628   CDT = CDT->getCanonicalTypeUnqualified();
10629   if (CDT->isRecordType() || CDT->isUnionType())
10630     CDT = C.IntTy;
10631   return C.getTypeSize(CDT);
10632 }
10633 
10634 static void
10635 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn,
10636                            const llvm::APSInt &VLENVal,
10637                            ArrayRef<ParamAttrTy> ParamAttrs,
10638                            OMPDeclareSimdDeclAttr::BranchStateTy State) {
10639   struct ISADataTy {
10640     char ISA;
10641     unsigned VecRegSize;
10642   };
10643   ISADataTy ISAData[] = {
10644       {
10645           'b', 128
10646       }, // SSE
10647       {
10648           'c', 256
10649       }, // AVX
10650       {
10651           'd', 256
10652       }, // AVX2
10653       {
10654           'e', 512
10655       }, // AVX512
10656   };
10657   llvm::SmallVector<char, 2> Masked;
10658   switch (State) {
10659   case OMPDeclareSimdDeclAttr::BS_Undefined:
10660     Masked.push_back('N');
10661     Masked.push_back('M');
10662     break;
10663   case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10664     Masked.push_back('N');
10665     break;
10666   case OMPDeclareSimdDeclAttr::BS_Inbranch:
10667     Masked.push_back('M');
10668     break;
10669   }
10670   for (char Mask : Masked) {
10671     for (const ISADataTy &Data : ISAData) {
10672       SmallString<256> Buffer;
10673       llvm::raw_svector_ostream Out(Buffer);
10674       Out << "_ZGV" << Data.ISA << Mask;
10675       if (!VLENVal) {
10676         unsigned NumElts = evaluateCDTSize(FD, ParamAttrs);
10677         assert(NumElts && "Non-zero simdlen/cdtsize expected");
10678         Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts);
10679       } else {
10680         Out << VLENVal;
10681       }
10682       for (const ParamAttrTy &ParamAttr : ParamAttrs) {
10683         switch (ParamAttr.Kind){
10684         case LinearWithVarStride:
10685           Out << 's' << ParamAttr.StrideOrArg;
10686           break;
10687         case Linear:
10688           Out << 'l';
10689           if (!!ParamAttr.StrideOrArg)
10690             Out << ParamAttr.StrideOrArg;
10691           break;
10692         case Uniform:
10693           Out << 'u';
10694           break;
10695         case Vector:
10696           Out << 'v';
10697           break;
10698         }
10699         if (!!ParamAttr.Alignment)
10700           Out << 'a' << ParamAttr.Alignment;
10701       }
10702       Out << '_' << Fn->getName();
10703       Fn->addFnAttr(Out.str());
10704     }
10705   }
10706 }
10707 
10708 // This are the Functions that are needed to mangle the name of the
10709 // vector functions generated by the compiler, according to the rules
10710 // defined in the "Vector Function ABI specifications for AArch64",
10711 // available at
10712 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi.
10713 
10714 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI.
10715 ///
10716 /// TODO: Need to implement the behavior for reference marked with a
10717 /// var or no linear modifiers (1.b in the section). For this, we
10718 /// need to extend ParamKindTy to support the linear modifiers.
10719 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) {
10720   QT = QT.getCanonicalType();
10721 
10722   if (QT->isVoidType())
10723     return false;
10724 
10725   if (Kind == ParamKindTy::Uniform)
10726     return false;
10727 
10728   if (Kind == ParamKindTy::Linear)
10729     return false;
10730 
10731   // TODO: Handle linear references with modifiers
10732 
10733   if (Kind == ParamKindTy::LinearWithVarStride)
10734     return false;
10735 
10736   return true;
10737 }
10738 
10739 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
10740 static bool getAArch64PBV(QualType QT, ASTContext &C) {
10741   QT = QT.getCanonicalType();
10742   unsigned Size = C.getTypeSize(QT);
10743 
10744   // Only scalars and complex within 16 bytes wide set PVB to true.
10745   if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128)
10746     return false;
10747 
10748   if (QT->isFloatingType())
10749     return true;
10750 
10751   if (QT->isIntegerType())
10752     return true;
10753 
10754   if (QT->isPointerType())
10755     return true;
10756 
10757   // TODO: Add support for complex types (section 3.1.2, item 2).
10758 
10759   return false;
10760 }
10761 
10762 /// Computes the lane size (LS) of a return type or of an input parameter,
10763 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI.
10764 /// TODO: Add support for references, section 3.2.1, item 1.
10765 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) {
10766   if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) {
10767     QualType PTy = QT.getCanonicalType()->getPointeeType();
10768     if (getAArch64PBV(PTy, C))
10769       return C.getTypeSize(PTy);
10770   }
10771   if (getAArch64PBV(QT, C))
10772     return C.getTypeSize(QT);
10773 
10774   return C.getTypeSize(C.getUIntPtrType());
10775 }
10776 
10777 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the
10778 // signature of the scalar function, as defined in 3.2.2 of the
10779 // AAVFABI.
10780 static std::tuple<unsigned, unsigned, bool>
10781 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) {
10782   QualType RetType = FD->getReturnType().getCanonicalType();
10783 
10784   ASTContext &C = FD->getASTContext();
10785 
10786   bool OutputBecomesInput = false;
10787 
10788   llvm::SmallVector<unsigned, 8> Sizes;
10789   if (!RetType->isVoidType()) {
10790     Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C));
10791     if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {}))
10792       OutputBecomesInput = true;
10793   }
10794   for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10795     QualType QT = FD->getParamDecl(I)->getType().getCanonicalType();
10796     Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C));
10797   }
10798 
10799   assert(!Sizes.empty() && "Unable to determine NDS and WDS.");
10800   // The LS of a function parameter / return value can only be a power
10801   // of 2, starting from 8 bits, up to 128.
10802   assert(std::all_of(Sizes.begin(), Sizes.end(),
10803                      [](unsigned Size) {
10804                        return Size == 8 || Size == 16 || Size == 32 ||
10805                               Size == 64 || Size == 128;
10806                      }) &&
10807          "Invalid size");
10808 
10809   return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)),
10810                          *std::max_element(std::begin(Sizes), std::end(Sizes)),
10811                          OutputBecomesInput);
10812 }
10813 
10814 /// Mangle the parameter part of the vector function name according to
10815 /// their OpenMP classification. The mangling function is defined in
10816 /// section 3.5 of the AAVFABI.
10817 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) {
10818   SmallString<256> Buffer;
10819   llvm::raw_svector_ostream Out(Buffer);
10820   for (const auto &ParamAttr : ParamAttrs) {
10821     switch (ParamAttr.Kind) {
10822     case LinearWithVarStride:
10823       Out << "ls" << ParamAttr.StrideOrArg;
10824       break;
10825     case Linear:
10826       Out << 'l';
10827       // Don't print the step value if it is not present or if it is
10828       // equal to 1.
10829       if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1)
10830         Out << ParamAttr.StrideOrArg;
10831       break;
10832     case Uniform:
10833       Out << 'u';
10834       break;
10835     case Vector:
10836       Out << 'v';
10837       break;
10838     }
10839 
10840     if (!!ParamAttr.Alignment)
10841       Out << 'a' << ParamAttr.Alignment;
10842   }
10843 
10844   return std::string(Out.str());
10845 }
10846 
10847 // Function used to add the attribute. The parameter `VLEN` is
10848 // templated to allow the use of "x" when targeting scalable functions
10849 // for SVE.
10850 template <typename T>
10851 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix,
10852                                  char ISA, StringRef ParSeq,
10853                                  StringRef MangledName, bool OutputBecomesInput,
10854                                  llvm::Function *Fn) {
10855   SmallString<256> Buffer;
10856   llvm::raw_svector_ostream Out(Buffer);
10857   Out << Prefix << ISA << LMask << VLEN;
10858   if (OutputBecomesInput)
10859     Out << "v";
10860   Out << ParSeq << "_" << MangledName;
10861   Fn->addFnAttr(Out.str());
10862 }
10863 
10864 // Helper function to generate the Advanced SIMD names depending on
10865 // the value of the NDS when simdlen is not present.
10866 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask,
10867                                       StringRef Prefix, char ISA,
10868                                       StringRef ParSeq, StringRef MangledName,
10869                                       bool OutputBecomesInput,
10870                                       llvm::Function *Fn) {
10871   switch (NDS) {
10872   case 8:
10873     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10874                          OutputBecomesInput, Fn);
10875     addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName,
10876                          OutputBecomesInput, Fn);
10877     break;
10878   case 16:
10879     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10880                          OutputBecomesInput, Fn);
10881     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10882                          OutputBecomesInput, Fn);
10883     break;
10884   case 32:
10885     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10886                          OutputBecomesInput, Fn);
10887     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10888                          OutputBecomesInput, Fn);
10889     break;
10890   case 64:
10891   case 128:
10892     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10893                          OutputBecomesInput, Fn);
10894     break;
10895   default:
10896     llvm_unreachable("Scalar type is too wide.");
10897   }
10898 }
10899 
10900 /// Emit vector function attributes for AArch64, as defined in the AAVFABI.
10901 static void emitAArch64DeclareSimdFunction(
10902     CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN,
10903     ArrayRef<ParamAttrTy> ParamAttrs,
10904     OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName,
10905     char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) {
10906 
10907   // Get basic data for building the vector signature.
10908   const auto Data = getNDSWDS(FD, ParamAttrs);
10909   const unsigned NDS = std::get<0>(Data);
10910   const unsigned WDS = std::get<1>(Data);
10911   const bool OutputBecomesInput = std::get<2>(Data);
10912 
10913   // Check the values provided via `simdlen` by the user.
10914   // 1. A `simdlen(1)` doesn't produce vector signatures,
10915   if (UserVLEN == 1) {
10916     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10917         DiagnosticsEngine::Warning,
10918         "The clause simdlen(1) has no effect when targeting aarch64.");
10919     CGM.getDiags().Report(SLoc, DiagID);
10920     return;
10921   }
10922 
10923   // 2. Section 3.3.1, item 1: user input must be a power of 2 for
10924   // Advanced SIMD output.
10925   if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) {
10926     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10927         DiagnosticsEngine::Warning, "The value specified in simdlen must be a "
10928                                     "power of 2 when targeting Advanced SIMD.");
10929     CGM.getDiags().Report(SLoc, DiagID);
10930     return;
10931   }
10932 
10933   // 3. Section 3.4.1. SVE fixed lengh must obey the architectural
10934   // limits.
10935   if (ISA == 's' && UserVLEN != 0) {
10936     if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) {
10937       unsigned DiagID = CGM.getDiags().getCustomDiagID(
10938           DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit "
10939                                       "lanes in the architectural constraints "
10940                                       "for SVE (min is 128-bit, max is "
10941                                       "2048-bit, by steps of 128-bit)");
10942       CGM.getDiags().Report(SLoc, DiagID) << WDS;
10943       return;
10944     }
10945   }
10946 
10947   // Sort out parameter sequence.
10948   const std::string ParSeq = mangleVectorParameters(ParamAttrs);
10949   StringRef Prefix = "_ZGV";
10950   // Generate simdlen from user input (if any).
10951   if (UserVLEN) {
10952     if (ISA == 's') {
10953       // SVE generates only a masked function.
10954       addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10955                            OutputBecomesInput, Fn);
10956     } else {
10957       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10958       // Advanced SIMD generates one or two functions, depending on
10959       // the `[not]inbranch` clause.
10960       switch (State) {
10961       case OMPDeclareSimdDeclAttr::BS_Undefined:
10962         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10963                              OutputBecomesInput, Fn);
10964         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10965                              OutputBecomesInput, Fn);
10966         break;
10967       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10968         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10969                              OutputBecomesInput, Fn);
10970         break;
10971       case OMPDeclareSimdDeclAttr::BS_Inbranch:
10972         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10973                              OutputBecomesInput, Fn);
10974         break;
10975       }
10976     }
10977   } else {
10978     // If no user simdlen is provided, follow the AAVFABI rules for
10979     // generating the vector length.
10980     if (ISA == 's') {
10981       // SVE, section 3.4.1, item 1.
10982       addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName,
10983                            OutputBecomesInput, Fn);
10984     } else {
10985       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10986       // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or
10987       // two vector names depending on the use of the clause
10988       // `[not]inbranch`.
10989       switch (State) {
10990       case OMPDeclareSimdDeclAttr::BS_Undefined:
10991         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10992                                   OutputBecomesInput, Fn);
10993         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
10994                                   OutputBecomesInput, Fn);
10995         break;
10996       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10997         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10998                                   OutputBecomesInput, Fn);
10999         break;
11000       case OMPDeclareSimdDeclAttr::BS_Inbranch:
11001         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
11002                                   OutputBecomesInput, Fn);
11003         break;
11004       }
11005     }
11006   }
11007 }
11008 
11009 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD,
11010                                               llvm::Function *Fn) {
11011   ASTContext &C = CGM.getContext();
11012   FD = FD->getMostRecentDecl();
11013   // Map params to their positions in function decl.
11014   llvm::DenseMap<const Decl *, unsigned> ParamPositions;
11015   if (isa<CXXMethodDecl>(FD))
11016     ParamPositions.try_emplace(FD, 0);
11017   unsigned ParamPos = ParamPositions.size();
11018   for (const ParmVarDecl *P : FD->parameters()) {
11019     ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos);
11020     ++ParamPos;
11021   }
11022   while (FD) {
11023     for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) {
11024       llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size());
11025       // Mark uniform parameters.
11026       for (const Expr *E : Attr->uniforms()) {
11027         E = E->IgnoreParenImpCasts();
11028         unsigned Pos;
11029         if (isa<CXXThisExpr>(E)) {
11030           Pos = ParamPositions[FD];
11031         } else {
11032           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
11033                                 ->getCanonicalDecl();
11034           Pos = ParamPositions[PVD];
11035         }
11036         ParamAttrs[Pos].Kind = Uniform;
11037       }
11038       // Get alignment info.
11039       auto NI = Attr->alignments_begin();
11040       for (const Expr *E : Attr->aligneds()) {
11041         E = E->IgnoreParenImpCasts();
11042         unsigned Pos;
11043         QualType ParmTy;
11044         if (isa<CXXThisExpr>(E)) {
11045           Pos = ParamPositions[FD];
11046           ParmTy = E->getType();
11047         } else {
11048           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
11049                                 ->getCanonicalDecl();
11050           Pos = ParamPositions[PVD];
11051           ParmTy = PVD->getType();
11052         }
11053         ParamAttrs[Pos].Alignment =
11054             (*NI)
11055                 ? (*NI)->EvaluateKnownConstInt(C)
11056                 : llvm::APSInt::getUnsigned(
11057                       C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy))
11058                           .getQuantity());
11059         ++NI;
11060       }
11061       // Mark linear parameters.
11062       auto SI = Attr->steps_begin();
11063       auto MI = Attr->modifiers_begin();
11064       for (const Expr *E : Attr->linears()) {
11065         E = E->IgnoreParenImpCasts();
11066         unsigned Pos;
11067         if (isa<CXXThisExpr>(E)) {
11068           Pos = ParamPositions[FD];
11069         } else {
11070           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
11071                                 ->getCanonicalDecl();
11072           Pos = ParamPositions[PVD];
11073         }
11074         ParamAttrTy &ParamAttr = ParamAttrs[Pos];
11075         ParamAttr.Kind = Linear;
11076         if (*SI) {
11077           Expr::EvalResult Result;
11078           if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) {
11079             if (const auto *DRE =
11080                     cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) {
11081               if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) {
11082                 ParamAttr.Kind = LinearWithVarStride;
11083                 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(
11084                     ParamPositions[StridePVD->getCanonicalDecl()]);
11085               }
11086             }
11087           } else {
11088             ParamAttr.StrideOrArg = Result.Val.getInt();
11089           }
11090         }
11091         ++SI;
11092         ++MI;
11093       }
11094       llvm::APSInt VLENVal;
11095       SourceLocation ExprLoc;
11096       const Expr *VLENExpr = Attr->getSimdlen();
11097       if (VLENExpr) {
11098         VLENVal = VLENExpr->EvaluateKnownConstInt(C);
11099         ExprLoc = VLENExpr->getExprLoc();
11100       }
11101       OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState();
11102       if (CGM.getTriple().isX86()) {
11103         emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State);
11104       } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) {
11105         unsigned VLEN = VLENVal.getExtValue();
11106         StringRef MangledName = Fn->getName();
11107         if (CGM.getTarget().hasFeature("sve"))
11108           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
11109                                          MangledName, 's', 128, Fn, ExprLoc);
11110         if (CGM.getTarget().hasFeature("neon"))
11111           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
11112                                          MangledName, 'n', 128, Fn, ExprLoc);
11113       }
11114     }
11115     FD = FD->getPreviousDecl();
11116   }
11117 }
11118 
11119 namespace {
11120 /// Cleanup action for doacross support.
11121 class DoacrossCleanupTy final : public EHScopeStack::Cleanup {
11122 public:
11123   static const int DoacrossFinArgs = 2;
11124 
11125 private:
11126   llvm::FunctionCallee RTLFn;
11127   llvm::Value *Args[DoacrossFinArgs];
11128 
11129 public:
11130   DoacrossCleanupTy(llvm::FunctionCallee RTLFn,
11131                     ArrayRef<llvm::Value *> CallArgs)
11132       : RTLFn(RTLFn) {
11133     assert(CallArgs.size() == DoacrossFinArgs);
11134     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
11135   }
11136   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
11137     if (!CGF.HaveInsertPoint())
11138       return;
11139     CGF.EmitRuntimeCall(RTLFn, Args);
11140   }
11141 };
11142 } // namespace
11143 
11144 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF,
11145                                        const OMPLoopDirective &D,
11146                                        ArrayRef<Expr *> NumIterations) {
11147   if (!CGF.HaveInsertPoint())
11148     return;
11149 
11150   ASTContext &C = CGM.getContext();
11151   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
11152   RecordDecl *RD;
11153   if (KmpDimTy.isNull()) {
11154     // Build struct kmp_dim {  // loop bounds info casted to kmp_int64
11155     //  kmp_int64 lo; // lower
11156     //  kmp_int64 up; // upper
11157     //  kmp_int64 st; // stride
11158     // };
11159     RD = C.buildImplicitRecord("kmp_dim");
11160     RD->startDefinition();
11161     addFieldToRecordDecl(C, RD, Int64Ty);
11162     addFieldToRecordDecl(C, RD, Int64Ty);
11163     addFieldToRecordDecl(C, RD, Int64Ty);
11164     RD->completeDefinition();
11165     KmpDimTy = C.getRecordType(RD);
11166   } else {
11167     RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl());
11168   }
11169   llvm::APInt Size(/*numBits=*/32, NumIterations.size());
11170   QualType ArrayTy =
11171       C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0);
11172 
11173   Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims");
11174   CGF.EmitNullInitialization(DimsAddr, ArrayTy);
11175   enum { LowerFD = 0, UpperFD, StrideFD };
11176   // Fill dims with data.
11177   for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) {
11178     LValue DimsLVal = CGF.MakeAddrLValue(
11179         CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy);
11180     // dims.upper = num_iterations;
11181     LValue UpperLVal = CGF.EmitLValueForField(
11182         DimsLVal, *std::next(RD->field_begin(), UpperFD));
11183     llvm::Value *NumIterVal =
11184         CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]),
11185                                  D.getNumIterations()->getType(), Int64Ty,
11186                                  D.getNumIterations()->getExprLoc());
11187     CGF.EmitStoreOfScalar(NumIterVal, UpperLVal);
11188     // dims.stride = 1;
11189     LValue StrideLVal = CGF.EmitLValueForField(
11190         DimsLVal, *std::next(RD->field_begin(), StrideFD));
11191     CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1),
11192                           StrideLVal);
11193   }
11194 
11195   // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid,
11196   // kmp_int32 num_dims, struct kmp_dim * dims);
11197   llvm::Value *Args[] = {
11198       emitUpdateLocation(CGF, D.getBeginLoc()),
11199       getThreadID(CGF, D.getBeginLoc()),
11200       llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()),
11201       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11202           CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(),
11203           CGM.VoidPtrTy)};
11204 
11205   llvm::FunctionCallee RTLFn =
11206       createRuntimeFunction(OMPRTL__kmpc_doacross_init);
11207   CGF.EmitRuntimeCall(RTLFn, Args);
11208   llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = {
11209       emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())};
11210   llvm::FunctionCallee FiniRTLFn =
11211       createRuntimeFunction(OMPRTL__kmpc_doacross_fini);
11212   CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
11213                                              llvm::makeArrayRef(FiniArgs));
11214 }
11215 
11216 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
11217                                           const OMPDependClause *C) {
11218   QualType Int64Ty =
11219       CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
11220   llvm::APInt Size(/*numBits=*/32, C->getNumLoops());
11221   QualType ArrayTy = CGM.getContext().getConstantArrayType(
11222       Int64Ty, Size, nullptr, ArrayType::Normal, 0);
11223   Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr");
11224   for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) {
11225     const Expr *CounterVal = C->getLoopData(I);
11226     assert(CounterVal);
11227     llvm::Value *CntVal = CGF.EmitScalarConversion(
11228         CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty,
11229         CounterVal->getExprLoc());
11230     CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I),
11231                           /*Volatile=*/false, Int64Ty);
11232   }
11233   llvm::Value *Args[] = {
11234       emitUpdateLocation(CGF, C->getBeginLoc()),
11235       getThreadID(CGF, C->getBeginLoc()),
11236       CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()};
11237   llvm::FunctionCallee RTLFn;
11238   if (C->getDependencyKind() == OMPC_DEPEND_source) {
11239     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post);
11240   } else {
11241     assert(C->getDependencyKind() == OMPC_DEPEND_sink);
11242     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait);
11243   }
11244   CGF.EmitRuntimeCall(RTLFn, Args);
11245 }
11246 
11247 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc,
11248                                llvm::FunctionCallee Callee,
11249                                ArrayRef<llvm::Value *> Args) const {
11250   assert(Loc.isValid() && "Outlined function call location must be valid.");
11251   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
11252 
11253   if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) {
11254     if (Fn->doesNotThrow()) {
11255       CGF.EmitNounwindRuntimeCall(Fn, Args);
11256       return;
11257     }
11258   }
11259   CGF.EmitRuntimeCall(Callee, Args);
11260 }
11261 
11262 void CGOpenMPRuntime::emitOutlinedFunctionCall(
11263     CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn,
11264     ArrayRef<llvm::Value *> Args) const {
11265   emitCall(CGF, Loc, OutlinedFn, Args);
11266 }
11267 
11268 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) {
11269   if (const auto *FD = dyn_cast<FunctionDecl>(D))
11270     if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD))
11271       HasEmittedDeclareTargetRegion = true;
11272 }
11273 
11274 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF,
11275                                              const VarDecl *NativeParam,
11276                                              const VarDecl *TargetParam) const {
11277   return CGF.GetAddrOfLocalVar(NativeParam);
11278 }
11279 
11280 namespace {
11281 /// Cleanup action for allocate support.
11282 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup {
11283 public:
11284   static const int CleanupArgs = 3;
11285 
11286 private:
11287   llvm::FunctionCallee RTLFn;
11288   llvm::Value *Args[CleanupArgs];
11289 
11290 public:
11291   OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn,
11292                        ArrayRef<llvm::Value *> CallArgs)
11293       : RTLFn(RTLFn) {
11294     assert(CallArgs.size() == CleanupArgs &&
11295            "Size of arguments does not match.");
11296     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
11297   }
11298   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
11299     if (!CGF.HaveInsertPoint())
11300       return;
11301     CGF.EmitRuntimeCall(RTLFn, Args);
11302   }
11303 };
11304 } // namespace
11305 
11306 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF,
11307                                                    const VarDecl *VD) {
11308   if (!VD)
11309     return Address::invalid();
11310   const VarDecl *CVD = VD->getCanonicalDecl();
11311   if (!CVD->hasAttr<OMPAllocateDeclAttr>())
11312     return Address::invalid();
11313   const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
11314   // Use the default allocation.
11315   if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc &&
11316       !AA->getAllocator())
11317     return Address::invalid();
11318   llvm::Value *Size;
11319   CharUnits Align = CGM.getContext().getDeclAlign(CVD);
11320   if (CVD->getType()->isVariablyModifiedType()) {
11321     Size = CGF.getTypeSize(CVD->getType());
11322     // Align the size: ((size + align - 1) / align) * align
11323     Size = CGF.Builder.CreateNUWAdd(
11324         Size, CGM.getSize(Align - CharUnits::fromQuantity(1)));
11325     Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align));
11326     Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align));
11327   } else {
11328     CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType());
11329     Size = CGM.getSize(Sz.alignTo(Align));
11330   }
11331   llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc());
11332   assert(AA->getAllocator() &&
11333          "Expected allocator expression for non-default allocator.");
11334   llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator());
11335   // According to the standard, the original allocator type is a enum (integer).
11336   // Convert to pointer type, if required.
11337   if (Allocator->getType()->isIntegerTy())
11338     Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy);
11339   else if (Allocator->getType()->isPointerTy())
11340     Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator,
11341                                                                 CGM.VoidPtrTy);
11342   llvm::Value *Args[] = {ThreadID, Size, Allocator};
11343 
11344   llvm::Value *Addr =
11345       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args,
11346                           getName({CVD->getName(), ".void.addr"}));
11347   llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr,
11348                                                               Allocator};
11349   llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free);
11350 
11351   CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
11352                                                 llvm::makeArrayRef(FiniArgs));
11353   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11354       Addr,
11355       CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())),
11356       getName({CVD->getName(), ".addr"}));
11357   return Address(Addr, Align);
11358 }
11359 
11360 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII(
11361     CodeGenModule &CGM, const OMPLoopDirective &S)
11362     : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) {
11363   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11364   if (!NeedToPush)
11365     return;
11366   NontemporalDeclsSet &DS =
11367       CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back();
11368   for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) {
11369     for (const Stmt *Ref : C->private_refs()) {
11370       const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts();
11371       const ValueDecl *VD;
11372       if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) {
11373         VD = DRE->getDecl();
11374       } else {
11375         const auto *ME = cast<MemberExpr>(SimpleRefExpr);
11376         assert((ME->isImplicitCXXThis() ||
11377                 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) &&
11378                "Expected member of current class.");
11379         VD = ME->getMemberDecl();
11380       }
11381       DS.insert(VD);
11382     }
11383   }
11384 }
11385 
11386 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() {
11387   if (!NeedToPush)
11388     return;
11389   CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back();
11390 }
11391 
11392 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const {
11393   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11394 
11395   return llvm::any_of(
11396       CGM.getOpenMPRuntime().NontemporalDeclsStack,
11397       [VD](const NontemporalDeclsSet &Set) { return Set.count(VD) > 0; });
11398 }
11399 
11400 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis(
11401     const OMPExecutableDirective &S,
11402     llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled)
11403     const {
11404   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs;
11405   // Vars in target/task regions must be excluded completely.
11406   if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) ||
11407       isOpenMPTaskingDirective(S.getDirectiveKind())) {
11408     SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
11409     getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind());
11410     const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front());
11411     for (const CapturedStmt::Capture &Cap : CS->captures()) {
11412       if (Cap.capturesVariable() || Cap.capturesVariableByCopy())
11413         NeedToCheckForLPCs.insert(Cap.getCapturedVar());
11414     }
11415   }
11416   // Exclude vars in private clauses.
11417   for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) {
11418     for (const Expr *Ref : C->varlists()) {
11419       if (!Ref->getType()->isScalarType())
11420         continue;
11421       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11422       if (!DRE)
11423         continue;
11424       NeedToCheckForLPCs.insert(DRE->getDecl());
11425     }
11426   }
11427   for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
11428     for (const Expr *Ref : C->varlists()) {
11429       if (!Ref->getType()->isScalarType())
11430         continue;
11431       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11432       if (!DRE)
11433         continue;
11434       NeedToCheckForLPCs.insert(DRE->getDecl());
11435     }
11436   }
11437   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
11438     for (const Expr *Ref : C->varlists()) {
11439       if (!Ref->getType()->isScalarType())
11440         continue;
11441       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11442       if (!DRE)
11443         continue;
11444       NeedToCheckForLPCs.insert(DRE->getDecl());
11445     }
11446   }
11447   for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
11448     for (const Expr *Ref : C->varlists()) {
11449       if (!Ref->getType()->isScalarType())
11450         continue;
11451       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11452       if (!DRE)
11453         continue;
11454       NeedToCheckForLPCs.insert(DRE->getDecl());
11455     }
11456   }
11457   for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) {
11458     for (const Expr *Ref : C->varlists()) {
11459       if (!Ref->getType()->isScalarType())
11460         continue;
11461       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11462       if (!DRE)
11463         continue;
11464       NeedToCheckForLPCs.insert(DRE->getDecl());
11465     }
11466   }
11467   for (const Decl *VD : NeedToCheckForLPCs) {
11468     for (const LastprivateConditionalData &Data :
11469          llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) {
11470       if (Data.DeclToUniqueName.count(VD) > 0) {
11471         if (!Data.Disabled)
11472           NeedToAddForLPCsAsDisabled.insert(VD);
11473         break;
11474       }
11475     }
11476   }
11477 }
11478 
11479 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
11480     CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal)
11481     : CGM(CGF.CGM),
11482       Action((CGM.getLangOpts().OpenMP >= 50 &&
11483               llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(),
11484                            [](const OMPLastprivateClause *C) {
11485                              return C->getKind() ==
11486                                     OMPC_LASTPRIVATE_conditional;
11487                            }))
11488                  ? ActionToDo::PushAsLastprivateConditional
11489                  : ActionToDo::DoNotPush) {
11490   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11491   if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush)
11492     return;
11493   assert(Action == ActionToDo::PushAsLastprivateConditional &&
11494          "Expected a push action.");
11495   LastprivateConditionalData &Data =
11496       CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
11497   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
11498     if (C->getKind() != OMPC_LASTPRIVATE_conditional)
11499       continue;
11500 
11501     for (const Expr *Ref : C->varlists()) {
11502       Data.DeclToUniqueName.insert(std::make_pair(
11503           cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(),
11504           SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref))));
11505     }
11506   }
11507   Data.IVLVal = IVLVal;
11508   Data.Fn = CGF.CurFn;
11509 }
11510 
11511 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
11512     CodeGenFunction &CGF, const OMPExecutableDirective &S)
11513     : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) {
11514   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11515   if (CGM.getLangOpts().OpenMP < 50)
11516     return;
11517   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled;
11518   tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled);
11519   if (!NeedToAddForLPCsAsDisabled.empty()) {
11520     Action = ActionToDo::DisableLastprivateConditional;
11521     LastprivateConditionalData &Data =
11522         CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
11523     for (const Decl *VD : NeedToAddForLPCsAsDisabled)
11524       Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>()));
11525     Data.Fn = CGF.CurFn;
11526     Data.Disabled = true;
11527   }
11528 }
11529 
11530 CGOpenMPRuntime::LastprivateConditionalRAII
11531 CGOpenMPRuntime::LastprivateConditionalRAII::disable(
11532     CodeGenFunction &CGF, const OMPExecutableDirective &S) {
11533   return LastprivateConditionalRAII(CGF, S);
11534 }
11535 
11536 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() {
11537   if (CGM.getLangOpts().OpenMP < 50)
11538     return;
11539   if (Action == ActionToDo::DisableLastprivateConditional) {
11540     assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
11541            "Expected list of disabled private vars.");
11542     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
11543   }
11544   if (Action == ActionToDo::PushAsLastprivateConditional) {
11545     assert(
11546         !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
11547         "Expected list of lastprivate conditional vars.");
11548     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
11549   }
11550 }
11551 
11552 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF,
11553                                                         const VarDecl *VD) {
11554   ASTContext &C = CGM.getContext();
11555   auto I = LastprivateConditionalToTypes.find(CGF.CurFn);
11556   if (I == LastprivateConditionalToTypes.end())
11557     I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first;
11558   QualType NewType;
11559   const FieldDecl *VDField;
11560   const FieldDecl *FiredField;
11561   LValue BaseLVal;
11562   auto VI = I->getSecond().find(VD);
11563   if (VI == I->getSecond().end()) {
11564     RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional");
11565     RD->startDefinition();
11566     VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType());
11567     FiredField = addFieldToRecordDecl(C, RD, C.CharTy);
11568     RD->completeDefinition();
11569     NewType = C.getRecordType(RD);
11570     Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName());
11571     BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl);
11572     I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal);
11573   } else {
11574     NewType = std::get<0>(VI->getSecond());
11575     VDField = std::get<1>(VI->getSecond());
11576     FiredField = std::get<2>(VI->getSecond());
11577     BaseLVal = std::get<3>(VI->getSecond());
11578   }
11579   LValue FiredLVal =
11580       CGF.EmitLValueForField(BaseLVal, FiredField);
11581   CGF.EmitStoreOfScalar(
11582       llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)),
11583       FiredLVal);
11584   return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF);
11585 }
11586 
11587 namespace {
11588 /// Checks if the lastprivate conditional variable is referenced in LHS.
11589 class LastprivateConditionalRefChecker final
11590     : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> {
11591   ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM;
11592   const Expr *FoundE = nullptr;
11593   const Decl *FoundD = nullptr;
11594   StringRef UniqueDeclName;
11595   LValue IVLVal;
11596   llvm::Function *FoundFn = nullptr;
11597   SourceLocation Loc;
11598 
11599 public:
11600   bool VisitDeclRefExpr(const DeclRefExpr *E) {
11601     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
11602          llvm::reverse(LPM)) {
11603       auto It = D.DeclToUniqueName.find(E->getDecl());
11604       if (It == D.DeclToUniqueName.end())
11605         continue;
11606       if (D.Disabled)
11607         return false;
11608       FoundE = E;
11609       FoundD = E->getDecl()->getCanonicalDecl();
11610       UniqueDeclName = It->second;
11611       IVLVal = D.IVLVal;
11612       FoundFn = D.Fn;
11613       break;
11614     }
11615     return FoundE == E;
11616   }
11617   bool VisitMemberExpr(const MemberExpr *E) {
11618     if (!CodeGenFunction::IsWrappedCXXThis(E->getBase()))
11619       return false;
11620     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
11621          llvm::reverse(LPM)) {
11622       auto It = D.DeclToUniqueName.find(E->getMemberDecl());
11623       if (It == D.DeclToUniqueName.end())
11624         continue;
11625       if (D.Disabled)
11626         return false;
11627       FoundE = E;
11628       FoundD = E->getMemberDecl()->getCanonicalDecl();
11629       UniqueDeclName = It->second;
11630       IVLVal = D.IVLVal;
11631       FoundFn = D.Fn;
11632       break;
11633     }
11634     return FoundE == E;
11635   }
11636   bool VisitStmt(const Stmt *S) {
11637     for (const Stmt *Child : S->children()) {
11638       if (!Child)
11639         continue;
11640       if (const auto *E = dyn_cast<Expr>(Child))
11641         if (!E->isGLValue())
11642           continue;
11643       if (Visit(Child))
11644         return true;
11645     }
11646     return false;
11647   }
11648   explicit LastprivateConditionalRefChecker(
11649       ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM)
11650       : LPM(LPM) {}
11651   std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *>
11652   getFoundData() const {
11653     return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn);
11654   }
11655 };
11656 } // namespace
11657 
11658 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF,
11659                                                        LValue IVLVal,
11660                                                        StringRef UniqueDeclName,
11661                                                        LValue LVal,
11662                                                        SourceLocation Loc) {
11663   // Last updated loop counter for the lastprivate conditional var.
11664   // int<xx> last_iv = 0;
11665   llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType());
11666   llvm::Constant *LastIV =
11667       getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"}));
11668   cast<llvm::GlobalVariable>(LastIV)->setAlignment(
11669       IVLVal.getAlignment().getAsAlign());
11670   LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType());
11671 
11672   // Last value of the lastprivate conditional.
11673   // decltype(priv_a) last_a;
11674   llvm::Constant *Last = getOrCreateInternalVariable(
11675       CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName);
11676   cast<llvm::GlobalVariable>(Last)->setAlignment(
11677       LVal.getAlignment().getAsAlign());
11678   LValue LastLVal =
11679       CGF.MakeAddrLValue(Last, LVal.getType(), LVal.getAlignment());
11680 
11681   // Global loop counter. Required to handle inner parallel-for regions.
11682   // iv
11683   llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc);
11684 
11685   // #pragma omp critical(a)
11686   // if (last_iv <= iv) {
11687   //   last_iv = iv;
11688   //   last_a = priv_a;
11689   // }
11690   auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal,
11691                     Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
11692     Action.Enter(CGF);
11693     llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc);
11694     // (last_iv <= iv) ? Check if the variable is updated and store new
11695     // value in global var.
11696     llvm::Value *CmpRes;
11697     if (IVLVal.getType()->isSignedIntegerType()) {
11698       CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal);
11699     } else {
11700       assert(IVLVal.getType()->isUnsignedIntegerType() &&
11701              "Loop iteration variable must be integer.");
11702       CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal);
11703     }
11704     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then");
11705     llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit");
11706     CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB);
11707     // {
11708     CGF.EmitBlock(ThenBB);
11709 
11710     //   last_iv = iv;
11711     CGF.EmitStoreOfScalar(IVVal, LastIVLVal);
11712 
11713     //   last_a = priv_a;
11714     switch (CGF.getEvaluationKind(LVal.getType())) {
11715     case TEK_Scalar: {
11716       llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc);
11717       CGF.EmitStoreOfScalar(PrivVal, LastLVal);
11718       break;
11719     }
11720     case TEK_Complex: {
11721       CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc);
11722       CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false);
11723       break;
11724     }
11725     case TEK_Aggregate:
11726       llvm_unreachable(
11727           "Aggregates are not supported in lastprivate conditional.");
11728     }
11729     // }
11730     CGF.EmitBranch(ExitBB);
11731     // There is no need to emit line number for unconditional branch.
11732     (void)ApplyDebugLocation::CreateEmpty(CGF);
11733     CGF.EmitBlock(ExitBB, /*IsFinished=*/true);
11734   };
11735 
11736   if (CGM.getLangOpts().OpenMPSimd) {
11737     // Do not emit as a critical region as no parallel region could be emitted.
11738     RegionCodeGenTy ThenRCG(CodeGen);
11739     ThenRCG(CGF);
11740   } else {
11741     emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc);
11742   }
11743 }
11744 
11745 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF,
11746                                                          const Expr *LHS) {
11747   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
11748     return;
11749   LastprivateConditionalRefChecker Checker(LastprivateConditionalStack);
11750   if (!Checker.Visit(LHS))
11751     return;
11752   const Expr *FoundE;
11753   const Decl *FoundD;
11754   StringRef UniqueDeclName;
11755   LValue IVLVal;
11756   llvm::Function *FoundFn;
11757   std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) =
11758       Checker.getFoundData();
11759   if (FoundFn != CGF.CurFn) {
11760     // Special codegen for inner parallel regions.
11761     // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1;
11762     auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD);
11763     assert(It != LastprivateConditionalToTypes[FoundFn].end() &&
11764            "Lastprivate conditional is not found in outer region.");
11765     QualType StructTy = std::get<0>(It->getSecond());
11766     const FieldDecl* FiredDecl = std::get<2>(It->getSecond());
11767     LValue PrivLVal = CGF.EmitLValue(FoundE);
11768     Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11769         PrivLVal.getAddress(CGF),
11770         CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy)));
11771     LValue BaseLVal =
11772         CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl);
11773     LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl);
11774     CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get(
11775                             CGF.ConvertTypeForMem(FiredDecl->getType()), 1)),
11776                         FiredLVal, llvm::AtomicOrdering::Unordered,
11777                         /*IsVolatile=*/true, /*isInit=*/false);
11778     return;
11779   }
11780 
11781   // Private address of the lastprivate conditional in the current context.
11782   // priv_a
11783   LValue LVal = CGF.EmitLValue(FoundE);
11784   emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal,
11785                                    FoundE->getExprLoc());
11786 }
11787 
11788 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional(
11789     CodeGenFunction &CGF, const OMPExecutableDirective &D,
11790     const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) {
11791   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
11792     return;
11793   auto Range = llvm::reverse(LastprivateConditionalStack);
11794   auto It = llvm::find_if(
11795       Range, [](const LastprivateConditionalData &D) { return !D.Disabled; });
11796   if (It == Range.end() || It->Fn != CGF.CurFn)
11797     return;
11798   auto LPCI = LastprivateConditionalToTypes.find(It->Fn);
11799   assert(LPCI != LastprivateConditionalToTypes.end() &&
11800          "Lastprivates must be registered already.");
11801   SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
11802   getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind());
11803   const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back());
11804   for (const auto &Pair : It->DeclToUniqueName) {
11805     const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl());
11806     if (!CS->capturesVariable(VD) || IgnoredDecls.count(VD) > 0)
11807       continue;
11808     auto I = LPCI->getSecond().find(Pair.first);
11809     assert(I != LPCI->getSecond().end() &&
11810            "Lastprivate must be rehistered already.");
11811     // bool Cmp = priv_a.Fired != 0;
11812     LValue BaseLVal = std::get<3>(I->getSecond());
11813     LValue FiredLVal =
11814         CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond()));
11815     llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc());
11816     llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res);
11817     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then");
11818     llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done");
11819     // if (Cmp) {
11820     CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB);
11821     CGF.EmitBlock(ThenBB);
11822     Address Addr = CGF.GetAddrOfLocalVar(VD);
11823     LValue LVal;
11824     if (VD->getType()->isReferenceType())
11825       LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(),
11826                                            AlignmentSource::Decl);
11827     else
11828       LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(),
11829                                 AlignmentSource::Decl);
11830     emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal,
11831                                      D.getBeginLoc());
11832     auto AL = ApplyDebugLocation::CreateArtificial(CGF);
11833     CGF.EmitBlock(DoneBB, /*IsFinal=*/true);
11834     // }
11835   }
11836 }
11837 
11838 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate(
11839     CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD,
11840     SourceLocation Loc) {
11841   if (CGF.getLangOpts().OpenMP < 50)
11842     return;
11843   auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD);
11844   assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() &&
11845          "Unknown lastprivate conditional variable.");
11846   StringRef UniqueName = It->second;
11847   llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName);
11848   // The variable was not updated in the region - exit.
11849   if (!GV)
11850     return;
11851   LValue LPLVal = CGF.MakeAddrLValue(
11852       GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment());
11853   llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc);
11854   CGF.EmitStoreOfScalar(Res, PrivLVal);
11855 }
11856 
11857 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction(
11858     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11859     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11860   llvm_unreachable("Not supported in SIMD-only mode");
11861 }
11862 
11863 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction(
11864     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11865     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11866   llvm_unreachable("Not supported in SIMD-only mode");
11867 }
11868 
11869 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction(
11870     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11871     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
11872     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
11873     bool Tied, unsigned &NumberOfParts) {
11874   llvm_unreachable("Not supported in SIMD-only mode");
11875 }
11876 
11877 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF,
11878                                            SourceLocation Loc,
11879                                            llvm::Function *OutlinedFn,
11880                                            ArrayRef<llvm::Value *> CapturedVars,
11881                                            const Expr *IfCond) {
11882   llvm_unreachable("Not supported in SIMD-only mode");
11883 }
11884 
11885 void CGOpenMPSIMDRuntime::emitCriticalRegion(
11886     CodeGenFunction &CGF, StringRef CriticalName,
11887     const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc,
11888     const Expr *Hint) {
11889   llvm_unreachable("Not supported in SIMD-only mode");
11890 }
11891 
11892 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF,
11893                                            const RegionCodeGenTy &MasterOpGen,
11894                                            SourceLocation Loc) {
11895   llvm_unreachable("Not supported in SIMD-only mode");
11896 }
11897 
11898 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
11899                                             SourceLocation Loc) {
11900   llvm_unreachable("Not supported in SIMD-only mode");
11901 }
11902 
11903 void CGOpenMPSIMDRuntime::emitTaskgroupRegion(
11904     CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen,
11905     SourceLocation Loc) {
11906   llvm_unreachable("Not supported in SIMD-only mode");
11907 }
11908 
11909 void CGOpenMPSIMDRuntime::emitSingleRegion(
11910     CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen,
11911     SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars,
11912     ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs,
11913     ArrayRef<const Expr *> AssignmentOps) {
11914   llvm_unreachable("Not supported in SIMD-only mode");
11915 }
11916 
11917 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF,
11918                                             const RegionCodeGenTy &OrderedOpGen,
11919                                             SourceLocation Loc,
11920                                             bool IsThreads) {
11921   llvm_unreachable("Not supported in SIMD-only mode");
11922 }
11923 
11924 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF,
11925                                           SourceLocation Loc,
11926                                           OpenMPDirectiveKind Kind,
11927                                           bool EmitChecks,
11928                                           bool ForceSimpleCall) {
11929   llvm_unreachable("Not supported in SIMD-only mode");
11930 }
11931 
11932 void CGOpenMPSIMDRuntime::emitForDispatchInit(
11933     CodeGenFunction &CGF, SourceLocation Loc,
11934     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
11935     bool Ordered, const DispatchRTInput &DispatchValues) {
11936   llvm_unreachable("Not supported in SIMD-only mode");
11937 }
11938 
11939 void CGOpenMPSIMDRuntime::emitForStaticInit(
11940     CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind,
11941     const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) {
11942   llvm_unreachable("Not supported in SIMD-only mode");
11943 }
11944 
11945 void CGOpenMPSIMDRuntime::emitDistributeStaticInit(
11946     CodeGenFunction &CGF, SourceLocation Loc,
11947     OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) {
11948   llvm_unreachable("Not supported in SIMD-only mode");
11949 }
11950 
11951 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
11952                                                      SourceLocation Loc,
11953                                                      unsigned IVSize,
11954                                                      bool IVSigned) {
11955   llvm_unreachable("Not supported in SIMD-only mode");
11956 }
11957 
11958 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF,
11959                                               SourceLocation Loc,
11960                                               OpenMPDirectiveKind DKind) {
11961   llvm_unreachable("Not supported in SIMD-only mode");
11962 }
11963 
11964 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF,
11965                                               SourceLocation Loc,
11966                                               unsigned IVSize, bool IVSigned,
11967                                               Address IL, Address LB,
11968                                               Address UB, Address ST) {
11969   llvm_unreachable("Not supported in SIMD-only mode");
11970 }
11971 
11972 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
11973                                                llvm::Value *NumThreads,
11974                                                SourceLocation Loc) {
11975   llvm_unreachable("Not supported in SIMD-only mode");
11976 }
11977 
11978 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF,
11979                                              ProcBindKind ProcBind,
11980                                              SourceLocation Loc) {
11981   llvm_unreachable("Not supported in SIMD-only mode");
11982 }
11983 
11984 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
11985                                                     const VarDecl *VD,
11986                                                     Address VDAddr,
11987                                                     SourceLocation Loc) {
11988   llvm_unreachable("Not supported in SIMD-only mode");
11989 }
11990 
11991 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition(
11992     const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit,
11993     CodeGenFunction *CGF) {
11994   llvm_unreachable("Not supported in SIMD-only mode");
11995 }
11996 
11997 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate(
11998     CodeGenFunction &CGF, QualType VarType, StringRef Name) {
11999   llvm_unreachable("Not supported in SIMD-only mode");
12000 }
12001 
12002 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF,
12003                                     ArrayRef<const Expr *> Vars,
12004                                     SourceLocation Loc,
12005                                     llvm::AtomicOrdering AO) {
12006   llvm_unreachable("Not supported in SIMD-only mode");
12007 }
12008 
12009 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
12010                                        const OMPExecutableDirective &D,
12011                                        llvm::Function *TaskFunction,
12012                                        QualType SharedsTy, Address Shareds,
12013                                        const Expr *IfCond,
12014                                        const OMPTaskDataTy &Data) {
12015   llvm_unreachable("Not supported in SIMD-only mode");
12016 }
12017 
12018 void CGOpenMPSIMDRuntime::emitTaskLoopCall(
12019     CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D,
12020     llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds,
12021     const Expr *IfCond, const OMPTaskDataTy &Data) {
12022   llvm_unreachable("Not supported in SIMD-only mode");
12023 }
12024 
12025 void CGOpenMPSIMDRuntime::emitReduction(
12026     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates,
12027     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
12028     ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) {
12029   assert(Options.SimpleReduction && "Only simple reduction is expected.");
12030   CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs,
12031                                  ReductionOps, Options);
12032 }
12033 
12034 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit(
12035     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
12036     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
12037   llvm_unreachable("Not supported in SIMD-only mode");
12038 }
12039 
12040 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
12041                                                   SourceLocation Loc,
12042                                                   ReductionCodeGen &RCG,
12043                                                   unsigned N) {
12044   llvm_unreachable("Not supported in SIMD-only mode");
12045 }
12046 
12047 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF,
12048                                                   SourceLocation Loc,
12049                                                   llvm::Value *ReductionsPtr,
12050                                                   LValue SharedLVal) {
12051   llvm_unreachable("Not supported in SIMD-only mode");
12052 }
12053 
12054 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
12055                                            SourceLocation Loc) {
12056   llvm_unreachable("Not supported in SIMD-only mode");
12057 }
12058 
12059 void CGOpenMPSIMDRuntime::emitCancellationPointCall(
12060     CodeGenFunction &CGF, SourceLocation Loc,
12061     OpenMPDirectiveKind CancelRegion) {
12062   llvm_unreachable("Not supported in SIMD-only mode");
12063 }
12064 
12065 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF,
12066                                          SourceLocation Loc, const Expr *IfCond,
12067                                          OpenMPDirectiveKind CancelRegion) {
12068   llvm_unreachable("Not supported in SIMD-only mode");
12069 }
12070 
12071 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction(
12072     const OMPExecutableDirective &D, StringRef ParentName,
12073     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
12074     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
12075   llvm_unreachable("Not supported in SIMD-only mode");
12076 }
12077 
12078 void CGOpenMPSIMDRuntime::emitTargetCall(
12079     CodeGenFunction &CGF, const OMPExecutableDirective &D,
12080     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
12081     llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
12082     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
12083                                      const OMPLoopDirective &D)>
12084         SizeEmitter) {
12085   llvm_unreachable("Not supported in SIMD-only mode");
12086 }
12087 
12088 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) {
12089   llvm_unreachable("Not supported in SIMD-only mode");
12090 }
12091 
12092 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
12093   llvm_unreachable("Not supported in SIMD-only mode");
12094 }
12095 
12096 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) {
12097   return false;
12098 }
12099 
12100 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF,
12101                                         const OMPExecutableDirective &D,
12102                                         SourceLocation Loc,
12103                                         llvm::Function *OutlinedFn,
12104                                         ArrayRef<llvm::Value *> CapturedVars) {
12105   llvm_unreachable("Not supported in SIMD-only mode");
12106 }
12107 
12108 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
12109                                              const Expr *NumTeams,
12110                                              const Expr *ThreadLimit,
12111                                              SourceLocation Loc) {
12112   llvm_unreachable("Not supported in SIMD-only mode");
12113 }
12114 
12115 void CGOpenMPSIMDRuntime::emitTargetDataCalls(
12116     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12117     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
12118   llvm_unreachable("Not supported in SIMD-only mode");
12119 }
12120 
12121 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall(
12122     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12123     const Expr *Device) {
12124   llvm_unreachable("Not supported in SIMD-only mode");
12125 }
12126 
12127 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF,
12128                                            const OMPLoopDirective &D,
12129                                            ArrayRef<Expr *> NumIterations) {
12130   llvm_unreachable("Not supported in SIMD-only mode");
12131 }
12132 
12133 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
12134                                               const OMPDependClause *C) {
12135   llvm_unreachable("Not supported in SIMD-only mode");
12136 }
12137 
12138 const VarDecl *
12139 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD,
12140                                         const VarDecl *NativeParam) const {
12141   llvm_unreachable("Not supported in SIMD-only mode");
12142 }
12143 
12144 Address
12145 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF,
12146                                          const VarDecl *NativeParam,
12147                                          const VarDecl *TargetParam) const {
12148   llvm_unreachable("Not supported in SIMD-only mode");
12149 }
12150