1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This provides a class for OpenMP runtime code generation.
10 //
11 //===----------------------------------------------------------------------===//
12 
13 #include "CGOpenMPRuntime.h"
14 #include "CGCXXABI.h"
15 #include "CGCleanup.h"
16 #include "CGRecordLayout.h"
17 #include "CodeGenFunction.h"
18 #include "clang/AST/Attr.h"
19 #include "clang/AST/Decl.h"
20 #include "clang/AST/OpenMPClause.h"
21 #include "clang/AST/StmtOpenMP.h"
22 #include "clang/AST/StmtVisitor.h"
23 #include "clang/Basic/BitmaskEnum.h"
24 #include "clang/Basic/FileManager.h"
25 #include "clang/Basic/OpenMPKinds.h"
26 #include "clang/Basic/SourceManager.h"
27 #include "clang/CodeGen/ConstantInitBuilder.h"
28 #include "llvm/ADT/ArrayRef.h"
29 #include "llvm/ADT/SetOperations.h"
30 #include "llvm/ADT/StringExtras.h"
31 #include "llvm/Bitcode/BitcodeReader.h"
32 #include "llvm/Frontend/OpenMP/OMPIRBuilder.h"
33 #include "llvm/IR/Constants.h"
34 #include "llvm/IR/DerivedTypes.h"
35 #include "llvm/IR/GlobalValue.h"
36 #include "llvm/IR/Value.h"
37 #include "llvm/Support/AtomicOrdering.h"
38 #include "llvm/Support/Format.h"
39 #include "llvm/Support/raw_ostream.h"
40 #include <cassert>
41 
42 using namespace clang;
43 using namespace CodeGen;
44 using namespace llvm::omp;
45 
46 namespace {
47 /// Base class for handling code generation inside OpenMP regions.
48 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
49 public:
50   /// Kinds of OpenMP regions used in codegen.
51   enum CGOpenMPRegionKind {
52     /// Region with outlined function for standalone 'parallel'
53     /// directive.
54     ParallelOutlinedRegion,
55     /// Region with outlined function for standalone 'task' directive.
56     TaskOutlinedRegion,
57     /// Region for constructs that do not require function outlining,
58     /// like 'for', 'sections', 'atomic' etc. directives.
59     InlinedRegion,
60     /// Region with outlined function for standalone 'target' directive.
61     TargetRegion,
62   };
63 
64   CGOpenMPRegionInfo(const CapturedStmt &CS,
65                      const CGOpenMPRegionKind RegionKind,
66                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
67                      bool HasCancel)
68       : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
69         CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {}
70 
71   CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
72                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
73                      bool HasCancel)
74       : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen),
75         Kind(Kind), HasCancel(HasCancel) {}
76 
77   /// Get a variable or parameter for storing global thread id
78   /// inside OpenMP construct.
79   virtual const VarDecl *getThreadIDVariable() const = 0;
80 
81   /// Emit the captured statement body.
82   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
83 
84   /// Get an LValue for the current ThreadID variable.
85   /// \return LValue for thread id variable. This LValue always has type int32*.
86   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
87 
88   virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {}
89 
90   CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
91 
92   OpenMPDirectiveKind getDirectiveKind() const { return Kind; }
93 
94   bool hasCancel() const { return HasCancel; }
95 
96   static bool classof(const CGCapturedStmtInfo *Info) {
97     return Info->getKind() == CR_OpenMP;
98   }
99 
100   ~CGOpenMPRegionInfo() override = default;
101 
102 protected:
103   CGOpenMPRegionKind RegionKind;
104   RegionCodeGenTy CodeGen;
105   OpenMPDirectiveKind Kind;
106   bool HasCancel;
107 };
108 
109 /// API for captured statement code generation in OpenMP constructs.
110 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo {
111 public:
112   CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
113                              const RegionCodeGenTy &CodeGen,
114                              OpenMPDirectiveKind Kind, bool HasCancel,
115                              StringRef HelperName)
116       : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind,
117                            HasCancel),
118         ThreadIDVar(ThreadIDVar), HelperName(HelperName) {
119     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
120   }
121 
122   /// Get a variable or parameter for storing global thread id
123   /// inside OpenMP construct.
124   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
125 
126   /// Get the name of the capture helper.
127   StringRef getHelperName() const override { return HelperName; }
128 
129   static bool classof(const CGCapturedStmtInfo *Info) {
130     return CGOpenMPRegionInfo::classof(Info) &&
131            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
132                ParallelOutlinedRegion;
133   }
134 
135 private:
136   /// A variable or parameter storing global thread id for OpenMP
137   /// constructs.
138   const VarDecl *ThreadIDVar;
139   StringRef HelperName;
140 };
141 
142 /// API for captured statement code generation in OpenMP constructs.
143 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo {
144 public:
145   class UntiedTaskActionTy final : public PrePostActionTy {
146     bool Untied;
147     const VarDecl *PartIDVar;
148     const RegionCodeGenTy UntiedCodeGen;
149     llvm::SwitchInst *UntiedSwitch = nullptr;
150 
151   public:
152     UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar,
153                        const RegionCodeGenTy &UntiedCodeGen)
154         : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {}
155     void Enter(CodeGenFunction &CGF) override {
156       if (Untied) {
157         // Emit task switching point.
158         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
159             CGF.GetAddrOfLocalVar(PartIDVar),
160             PartIDVar->getType()->castAs<PointerType>());
161         llvm::Value *Res =
162             CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation());
163         llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done.");
164         UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB);
165         CGF.EmitBlock(DoneBB);
166         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
167         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
168         UntiedSwitch->addCase(CGF.Builder.getInt32(0),
169                               CGF.Builder.GetInsertBlock());
170         emitUntiedSwitch(CGF);
171       }
172     }
173     void emitUntiedSwitch(CodeGenFunction &CGF) const {
174       if (Untied) {
175         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
176             CGF.GetAddrOfLocalVar(PartIDVar),
177             PartIDVar->getType()->castAs<PointerType>());
178         CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
179                               PartIdLVal);
180         UntiedCodeGen(CGF);
181         CodeGenFunction::JumpDest CurPoint =
182             CGF.getJumpDestInCurrentScope(".untied.next.");
183         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
184         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
185         UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
186                               CGF.Builder.GetInsertBlock());
187         CGF.EmitBranchThroughCleanup(CurPoint);
188         CGF.EmitBlock(CurPoint.getBlock());
189       }
190     }
191     unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); }
192   };
193   CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
194                                  const VarDecl *ThreadIDVar,
195                                  const RegionCodeGenTy &CodeGen,
196                                  OpenMPDirectiveKind Kind, bool HasCancel,
197                                  const UntiedTaskActionTy &Action)
198       : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel),
199         ThreadIDVar(ThreadIDVar), Action(Action) {
200     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
201   }
202 
203   /// Get a variable or parameter for storing global thread id
204   /// inside OpenMP construct.
205   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
206 
207   /// Get an LValue for the current ThreadID variable.
208   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
209 
210   /// Get the name of the capture helper.
211   StringRef getHelperName() const override { return ".omp_outlined."; }
212 
213   void emitUntiedSwitch(CodeGenFunction &CGF) override {
214     Action.emitUntiedSwitch(CGF);
215   }
216 
217   static bool classof(const CGCapturedStmtInfo *Info) {
218     return CGOpenMPRegionInfo::classof(Info) &&
219            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
220                TaskOutlinedRegion;
221   }
222 
223 private:
224   /// A variable or parameter storing global thread id for OpenMP
225   /// constructs.
226   const VarDecl *ThreadIDVar;
227   /// Action for emitting code for untied tasks.
228   const UntiedTaskActionTy &Action;
229 };
230 
231 /// API for inlined captured statement code generation in OpenMP
232 /// constructs.
233 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
234 public:
235   CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
236                             const RegionCodeGenTy &CodeGen,
237                             OpenMPDirectiveKind Kind, bool HasCancel)
238       : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel),
239         OldCSI(OldCSI),
240         OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
241 
242   // Retrieve the value of the context parameter.
243   llvm::Value *getContextValue() const override {
244     if (OuterRegionInfo)
245       return OuterRegionInfo->getContextValue();
246     llvm_unreachable("No context value for inlined OpenMP region");
247   }
248 
249   void setContextValue(llvm::Value *V) override {
250     if (OuterRegionInfo) {
251       OuterRegionInfo->setContextValue(V);
252       return;
253     }
254     llvm_unreachable("No context value for inlined OpenMP region");
255   }
256 
257   /// Lookup the captured field decl for a variable.
258   const FieldDecl *lookup(const VarDecl *VD) const override {
259     if (OuterRegionInfo)
260       return OuterRegionInfo->lookup(VD);
261     // If there is no outer outlined region,no need to lookup in a list of
262     // captured variables, we can use the original one.
263     return nullptr;
264   }
265 
266   FieldDecl *getThisFieldDecl() const override {
267     if (OuterRegionInfo)
268       return OuterRegionInfo->getThisFieldDecl();
269     return nullptr;
270   }
271 
272   /// Get a variable or parameter for storing global thread id
273   /// inside OpenMP construct.
274   const VarDecl *getThreadIDVariable() const override {
275     if (OuterRegionInfo)
276       return OuterRegionInfo->getThreadIDVariable();
277     return nullptr;
278   }
279 
280   /// Get an LValue for the current ThreadID variable.
281   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override {
282     if (OuterRegionInfo)
283       return OuterRegionInfo->getThreadIDVariableLValue(CGF);
284     llvm_unreachable("No LValue for inlined OpenMP construct");
285   }
286 
287   /// Get the name of the capture helper.
288   StringRef getHelperName() const override {
289     if (auto *OuterRegionInfo = getOldCSI())
290       return OuterRegionInfo->getHelperName();
291     llvm_unreachable("No helper name for inlined OpenMP construct");
292   }
293 
294   void emitUntiedSwitch(CodeGenFunction &CGF) override {
295     if (OuterRegionInfo)
296       OuterRegionInfo->emitUntiedSwitch(CGF);
297   }
298 
299   CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
300 
301   static bool classof(const CGCapturedStmtInfo *Info) {
302     return CGOpenMPRegionInfo::classof(Info) &&
303            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion;
304   }
305 
306   ~CGOpenMPInlinedRegionInfo() override = default;
307 
308 private:
309   /// CodeGen info about outer OpenMP region.
310   CodeGenFunction::CGCapturedStmtInfo *OldCSI;
311   CGOpenMPRegionInfo *OuterRegionInfo;
312 };
313 
314 /// API for captured statement code generation in OpenMP target
315 /// constructs. For this captures, implicit parameters are used instead of the
316 /// captured fields. The name of the target region has to be unique in a given
317 /// application so it is provided by the client, because only the client has
318 /// the information to generate that.
319 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo {
320 public:
321   CGOpenMPTargetRegionInfo(const CapturedStmt &CS,
322                            const RegionCodeGenTy &CodeGen, StringRef HelperName)
323       : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target,
324                            /*HasCancel=*/false),
325         HelperName(HelperName) {}
326 
327   /// This is unused for target regions because each starts executing
328   /// with a single thread.
329   const VarDecl *getThreadIDVariable() const override { return nullptr; }
330 
331   /// Get the name of the capture helper.
332   StringRef getHelperName() const override { return HelperName; }
333 
334   static bool classof(const CGCapturedStmtInfo *Info) {
335     return CGOpenMPRegionInfo::classof(Info) &&
336            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion;
337   }
338 
339 private:
340   StringRef HelperName;
341 };
342 
343 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) {
344   llvm_unreachable("No codegen for expressions");
345 }
346 /// API for generation of expressions captured in a innermost OpenMP
347 /// region.
348 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo {
349 public:
350   CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS)
351       : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen,
352                                   OMPD_unknown,
353                                   /*HasCancel=*/false),
354         PrivScope(CGF) {
355     // Make sure the globals captured in the provided statement are local by
356     // using the privatization logic. We assume the same variable is not
357     // captured more than once.
358     for (const auto &C : CS.captures()) {
359       if (!C.capturesVariable() && !C.capturesVariableByCopy())
360         continue;
361 
362       const VarDecl *VD = C.getCapturedVar();
363       if (VD->isLocalVarDeclOrParm())
364         continue;
365 
366       DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
367                       /*RefersToEnclosingVariableOrCapture=*/false,
368                       VD->getType().getNonReferenceType(), VK_LValue,
369                       C.getLocation());
370       PrivScope.addPrivate(
371           VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(CGF); });
372     }
373     (void)PrivScope.Privatize();
374   }
375 
376   /// Lookup the captured field decl for a variable.
377   const FieldDecl *lookup(const VarDecl *VD) const override {
378     if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD))
379       return FD;
380     return nullptr;
381   }
382 
383   /// Emit the captured statement body.
384   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override {
385     llvm_unreachable("No body for expressions");
386   }
387 
388   /// Get a variable or parameter for storing global thread id
389   /// inside OpenMP construct.
390   const VarDecl *getThreadIDVariable() const override {
391     llvm_unreachable("No thread id for expressions");
392   }
393 
394   /// Get the name of the capture helper.
395   StringRef getHelperName() const override {
396     llvm_unreachable("No helper name for expressions");
397   }
398 
399   static bool classof(const CGCapturedStmtInfo *Info) { return false; }
400 
401 private:
402   /// Private scope to capture global variables.
403   CodeGenFunction::OMPPrivateScope PrivScope;
404 };
405 
406 /// RAII for emitting code of OpenMP constructs.
407 class InlinedOpenMPRegionRAII {
408   CodeGenFunction &CGF;
409   llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields;
410   FieldDecl *LambdaThisCaptureField = nullptr;
411   const CodeGen::CGBlockInfo *BlockInfo = nullptr;
412 
413 public:
414   /// Constructs region for combined constructs.
415   /// \param CodeGen Code generation sequence for combined directives. Includes
416   /// a list of functions used for code generation of implicitly inlined
417   /// regions.
418   InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen,
419                           OpenMPDirectiveKind Kind, bool HasCancel)
420       : CGF(CGF) {
421     // Start emission for the construct.
422     CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(
423         CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel);
424     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
425     LambdaThisCaptureField = CGF.LambdaThisCaptureField;
426     CGF.LambdaThisCaptureField = nullptr;
427     BlockInfo = CGF.BlockInfo;
428     CGF.BlockInfo = nullptr;
429   }
430 
431   ~InlinedOpenMPRegionRAII() {
432     // Restore original CapturedStmtInfo only if we're done with code emission.
433     auto *OldCSI =
434         cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
435     delete CGF.CapturedStmtInfo;
436     CGF.CapturedStmtInfo = OldCSI;
437     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
438     CGF.LambdaThisCaptureField = LambdaThisCaptureField;
439     CGF.BlockInfo = BlockInfo;
440   }
441 };
442 
443 /// Values for bit flags used in the ident_t to describe the fields.
444 /// All enumeric elements are named and described in accordance with the code
445 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
446 enum OpenMPLocationFlags : unsigned {
447   /// Use trampoline for internal microtask.
448   OMP_IDENT_IMD = 0x01,
449   /// Use c-style ident structure.
450   OMP_IDENT_KMPC = 0x02,
451   /// Atomic reduction option for kmpc_reduce.
452   OMP_ATOMIC_REDUCE = 0x10,
453   /// Explicit 'barrier' directive.
454   OMP_IDENT_BARRIER_EXPL = 0x20,
455   /// Implicit barrier in code.
456   OMP_IDENT_BARRIER_IMPL = 0x40,
457   /// Implicit barrier in 'for' directive.
458   OMP_IDENT_BARRIER_IMPL_FOR = 0x40,
459   /// Implicit barrier in 'sections' directive.
460   OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0,
461   /// Implicit barrier in 'single' directive.
462   OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140,
463   /// Call of __kmp_for_static_init for static loop.
464   OMP_IDENT_WORK_LOOP = 0x200,
465   /// Call of __kmp_for_static_init for sections.
466   OMP_IDENT_WORK_SECTIONS = 0x400,
467   /// Call of __kmp_for_static_init for distribute.
468   OMP_IDENT_WORK_DISTRIBUTE = 0x800,
469   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE)
470 };
471 
472 namespace {
473 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
474 /// Values for bit flags for marking which requires clauses have been used.
475 enum OpenMPOffloadingRequiresDirFlags : int64_t {
476   /// flag undefined.
477   OMP_REQ_UNDEFINED               = 0x000,
478   /// no requires clause present.
479   OMP_REQ_NONE                    = 0x001,
480   /// reverse_offload clause.
481   OMP_REQ_REVERSE_OFFLOAD         = 0x002,
482   /// unified_address clause.
483   OMP_REQ_UNIFIED_ADDRESS         = 0x004,
484   /// unified_shared_memory clause.
485   OMP_REQ_UNIFIED_SHARED_MEMORY   = 0x008,
486   /// dynamic_allocators clause.
487   OMP_REQ_DYNAMIC_ALLOCATORS      = 0x010,
488   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS)
489 };
490 
491 enum OpenMPOffloadingReservedDeviceIDs {
492   /// Device ID if the device was not defined, runtime should get it
493   /// from environment variables in the spec.
494   OMP_DEVICEID_UNDEF = -1,
495 };
496 } // anonymous namespace
497 
498 /// Describes ident structure that describes a source location.
499 /// All descriptions are taken from
500 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
501 /// Original structure:
502 /// typedef struct ident {
503 ///    kmp_int32 reserved_1;   /**<  might be used in Fortran;
504 ///                                  see above  */
505 ///    kmp_int32 flags;        /**<  also f.flags; KMP_IDENT_xxx flags;
506 ///                                  KMP_IDENT_KMPC identifies this union
507 ///                                  member  */
508 ///    kmp_int32 reserved_2;   /**<  not really used in Fortran any more;
509 ///                                  see above */
510 ///#if USE_ITT_BUILD
511 ///                            /*  but currently used for storing
512 ///                                region-specific ITT */
513 ///                            /*  contextual information. */
514 ///#endif /* USE_ITT_BUILD */
515 ///    kmp_int32 reserved_3;   /**< source[4] in Fortran, do not use for
516 ///                                 C++  */
517 ///    char const *psource;    /**< String describing the source location.
518 ///                            The string is composed of semi-colon separated
519 //                             fields which describe the source file,
520 ///                            the function and a pair of line numbers that
521 ///                            delimit the construct.
522 ///                             */
523 /// } ident_t;
524 enum IdentFieldIndex {
525   /// might be used in Fortran
526   IdentField_Reserved_1,
527   /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member.
528   IdentField_Flags,
529   /// Not really used in Fortran any more
530   IdentField_Reserved_2,
531   /// Source[4] in Fortran, do not use for C++
532   IdentField_Reserved_3,
533   /// String describing the source location. The string is composed of
534   /// semi-colon separated fields which describe the source file, the function
535   /// and a pair of line numbers that delimit the construct.
536   IdentField_PSource
537 };
538 
539 /// Schedule types for 'omp for' loops (these enumerators are taken from
540 /// the enum sched_type in kmp.h).
541 enum OpenMPSchedType {
542   /// Lower bound for default (unordered) versions.
543   OMP_sch_lower = 32,
544   OMP_sch_static_chunked = 33,
545   OMP_sch_static = 34,
546   OMP_sch_dynamic_chunked = 35,
547   OMP_sch_guided_chunked = 36,
548   OMP_sch_runtime = 37,
549   OMP_sch_auto = 38,
550   /// static with chunk adjustment (e.g., simd)
551   OMP_sch_static_balanced_chunked = 45,
552   /// Lower bound for 'ordered' versions.
553   OMP_ord_lower = 64,
554   OMP_ord_static_chunked = 65,
555   OMP_ord_static = 66,
556   OMP_ord_dynamic_chunked = 67,
557   OMP_ord_guided_chunked = 68,
558   OMP_ord_runtime = 69,
559   OMP_ord_auto = 70,
560   OMP_sch_default = OMP_sch_static,
561   /// dist_schedule types
562   OMP_dist_sch_static_chunked = 91,
563   OMP_dist_sch_static = 92,
564   /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers.
565   /// Set if the monotonic schedule modifier was present.
566   OMP_sch_modifier_monotonic = (1 << 29),
567   /// Set if the nonmonotonic schedule modifier was present.
568   OMP_sch_modifier_nonmonotonic = (1 << 30),
569 };
570 
571 enum OpenMPRTLFunction {
572   /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc,
573   /// kmpc_micro microtask, ...);
574   OMPRTL__kmpc_fork_call,
575   /// Call to void *__kmpc_threadprivate_cached(ident_t *loc,
576   /// kmp_int32 global_tid, void *data, size_t size, void ***cache);
577   OMPRTL__kmpc_threadprivate_cached,
578   /// Call to void __kmpc_threadprivate_register( ident_t *,
579   /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
580   OMPRTL__kmpc_threadprivate_register,
581   // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc);
582   OMPRTL__kmpc_global_thread_num,
583   // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
584   // kmp_critical_name *crit);
585   OMPRTL__kmpc_critical,
586   // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32
587   // global_tid, kmp_critical_name *crit, uintptr_t hint);
588   OMPRTL__kmpc_critical_with_hint,
589   // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
590   // kmp_critical_name *crit);
591   OMPRTL__kmpc_end_critical,
592   // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
593   // global_tid);
594   OMPRTL__kmpc_cancel_barrier,
595   // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
596   OMPRTL__kmpc_barrier,
597   // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
598   OMPRTL__kmpc_for_static_fini,
599   // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
600   // global_tid);
601   OMPRTL__kmpc_serialized_parallel,
602   // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
603   // global_tid);
604   OMPRTL__kmpc_end_serialized_parallel,
605   // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
606   // kmp_int32 num_threads);
607   OMPRTL__kmpc_push_num_threads,
608   // Call to void __kmpc_flush(ident_t *loc);
609   OMPRTL__kmpc_flush,
610   // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid);
611   OMPRTL__kmpc_master,
612   // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid);
613   OMPRTL__kmpc_end_master,
614   // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
615   // int end_part);
616   OMPRTL__kmpc_omp_taskyield,
617   // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid);
618   OMPRTL__kmpc_single,
619   // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid);
620   OMPRTL__kmpc_end_single,
621   // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
622   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
623   // kmp_routine_entry_t *task_entry);
624   OMPRTL__kmpc_omp_task_alloc,
625   // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *,
626   // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t,
627   // size_t sizeof_shareds, kmp_routine_entry_t *task_entry,
628   // kmp_int64 device_id);
629   OMPRTL__kmpc_omp_target_task_alloc,
630   // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t *
631   // new_task);
632   OMPRTL__kmpc_omp_task,
633   // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
634   // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
635   // kmp_int32 didit);
636   OMPRTL__kmpc_copyprivate,
637   // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
638   // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
639   // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
640   OMPRTL__kmpc_reduce,
641   // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
642   // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
643   // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
644   // *lck);
645   OMPRTL__kmpc_reduce_nowait,
646   // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
647   // kmp_critical_name *lck);
648   OMPRTL__kmpc_end_reduce,
649   // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
650   // kmp_critical_name *lck);
651   OMPRTL__kmpc_end_reduce_nowait,
652   // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
653   // kmp_task_t * new_task);
654   OMPRTL__kmpc_omp_task_begin_if0,
655   // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
656   // kmp_task_t * new_task);
657   OMPRTL__kmpc_omp_task_complete_if0,
658   // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
659   OMPRTL__kmpc_ordered,
660   // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
661   OMPRTL__kmpc_end_ordered,
662   // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
663   // global_tid);
664   OMPRTL__kmpc_omp_taskwait,
665   // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
666   OMPRTL__kmpc_taskgroup,
667   // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
668   OMPRTL__kmpc_end_taskgroup,
669   // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
670   // int proc_bind);
671   OMPRTL__kmpc_push_proc_bind,
672   // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32
673   // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t
674   // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
675   OMPRTL__kmpc_omp_task_with_deps,
676   // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32
677   // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
678   // ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
679   OMPRTL__kmpc_omp_wait_deps,
680   // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
681   // global_tid, kmp_int32 cncl_kind);
682   OMPRTL__kmpc_cancellationpoint,
683   // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
684   // kmp_int32 cncl_kind);
685   OMPRTL__kmpc_cancel,
686   // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid,
687   // kmp_int32 num_teams, kmp_int32 thread_limit);
688   OMPRTL__kmpc_push_num_teams,
689   // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
690   // microtask, ...);
691   OMPRTL__kmpc_fork_teams,
692   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
693   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
694   // sched, kmp_uint64 grainsize, void *task_dup);
695   OMPRTL__kmpc_taskloop,
696   // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
697   // num_dims, struct kmp_dim *dims);
698   OMPRTL__kmpc_doacross_init,
699   // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
700   OMPRTL__kmpc_doacross_fini,
701   // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
702   // *vec);
703   OMPRTL__kmpc_doacross_post,
704   // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
705   // *vec);
706   OMPRTL__kmpc_doacross_wait,
707   // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void
708   // *data);
709   OMPRTL__kmpc_task_reduction_init,
710   // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
711   // *d);
712   OMPRTL__kmpc_task_reduction_get_th_data,
713   // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al);
714   OMPRTL__kmpc_alloc,
715   // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al);
716   OMPRTL__kmpc_free,
717 
718   //
719   // Offloading related calls
720   //
721   // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
722   // size);
723   OMPRTL__kmpc_push_target_tripcount,
724   // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
725   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
726   // *arg_types);
727   OMPRTL__tgt_target,
728   // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
729   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
730   // *arg_types);
731   OMPRTL__tgt_target_nowait,
732   // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
733   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
734   // *arg_types, int32_t num_teams, int32_t thread_limit);
735   OMPRTL__tgt_target_teams,
736   // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void
737   // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
738   // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
739   OMPRTL__tgt_target_teams_nowait,
740   // Call to void __tgt_register_requires(int64_t flags);
741   OMPRTL__tgt_register_requires,
742   // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
743   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
744   OMPRTL__tgt_target_data_begin,
745   // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
746   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
747   // *arg_types);
748   OMPRTL__tgt_target_data_begin_nowait,
749   // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
750   // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types);
751   OMPRTL__tgt_target_data_end,
752   // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t
753   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
754   // *arg_types);
755   OMPRTL__tgt_target_data_end_nowait,
756   // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
757   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
758   OMPRTL__tgt_target_data_update,
759   // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t
760   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
761   // *arg_types);
762   OMPRTL__tgt_target_data_update_nowait,
763   // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
764   OMPRTL__tgt_mapper_num_components,
765   // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void
766   // *base, void *begin, int64_t size, int64_t type);
767   OMPRTL__tgt_push_mapper_component,
768   // Call to kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
769   // int gtid, kmp_task_t *task);
770   OMPRTL__kmpc_task_allow_completion_event,
771 };
772 
773 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP
774 /// region.
775 class CleanupTy final : public EHScopeStack::Cleanup {
776   PrePostActionTy *Action;
777 
778 public:
779   explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {}
780   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
781     if (!CGF.HaveInsertPoint())
782       return;
783     Action->Exit(CGF);
784   }
785 };
786 
787 } // anonymous namespace
788 
789 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const {
790   CodeGenFunction::RunCleanupsScope Scope(CGF);
791   if (PrePostAction) {
792     CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction);
793     Callback(CodeGen, CGF, *PrePostAction);
794   } else {
795     PrePostActionTy Action;
796     Callback(CodeGen, CGF, Action);
797   }
798 }
799 
800 /// Check if the combiner is a call to UDR combiner and if it is so return the
801 /// UDR decl used for reduction.
802 static const OMPDeclareReductionDecl *
803 getReductionInit(const Expr *ReductionOp) {
804   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
805     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
806       if (const auto *DRE =
807               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
808         if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl()))
809           return DRD;
810   return nullptr;
811 }
812 
813 static void emitInitWithReductionInitializer(CodeGenFunction &CGF,
814                                              const OMPDeclareReductionDecl *DRD,
815                                              const Expr *InitOp,
816                                              Address Private, Address Original,
817                                              QualType Ty) {
818   if (DRD->getInitializer()) {
819     std::pair<llvm::Function *, llvm::Function *> Reduction =
820         CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
821     const auto *CE = cast<CallExpr>(InitOp);
822     const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee());
823     const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts();
824     const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts();
825     const auto *LHSDRE =
826         cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr());
827     const auto *RHSDRE =
828         cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr());
829     CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
830     PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()),
831                             [=]() { return Private; });
832     PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()),
833                             [=]() { return Original; });
834     (void)PrivateScope.Privatize();
835     RValue Func = RValue::get(Reduction.second);
836     CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
837     CGF.EmitIgnoredExpr(InitOp);
838   } else {
839     llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty);
840     std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"});
841     auto *GV = new llvm::GlobalVariable(
842         CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true,
843         llvm::GlobalValue::PrivateLinkage, Init, Name);
844     LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty);
845     RValue InitRVal;
846     switch (CGF.getEvaluationKind(Ty)) {
847     case TEK_Scalar:
848       InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation());
849       break;
850     case TEK_Complex:
851       InitRVal =
852           RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation()));
853       break;
854     case TEK_Aggregate:
855       InitRVal = RValue::getAggregate(LV.getAddress(CGF));
856       break;
857     }
858     OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue);
859     CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal);
860     CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
861                          /*IsInitializer=*/false);
862   }
863 }
864 
865 /// Emit initialization of arrays of complex types.
866 /// \param DestAddr Address of the array.
867 /// \param Type Type of array.
868 /// \param Init Initial expression of array.
869 /// \param SrcAddr Address of the original array.
870 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr,
871                                  QualType Type, bool EmitDeclareReductionInit,
872                                  const Expr *Init,
873                                  const OMPDeclareReductionDecl *DRD,
874                                  Address SrcAddr = Address::invalid()) {
875   // Perform element-by-element initialization.
876   QualType ElementTy;
877 
878   // Drill down to the base element type on both arrays.
879   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
880   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr);
881   DestAddr =
882       CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType());
883   if (DRD)
884     SrcAddr =
885         CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType());
886 
887   llvm::Value *SrcBegin = nullptr;
888   if (DRD)
889     SrcBegin = SrcAddr.getPointer();
890   llvm::Value *DestBegin = DestAddr.getPointer();
891   // Cast from pointer to array type to pointer to single element.
892   llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements);
893   // The basic structure here is a while-do loop.
894   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body");
895   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done");
896   llvm::Value *IsEmpty =
897       CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty");
898   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
899 
900   // Enter the loop body, making that address the current address.
901   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
902   CGF.EmitBlock(BodyBB);
903 
904   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
905 
906   llvm::PHINode *SrcElementPHI = nullptr;
907   Address SrcElementCurrent = Address::invalid();
908   if (DRD) {
909     SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2,
910                                           "omp.arraycpy.srcElementPast");
911     SrcElementPHI->addIncoming(SrcBegin, EntryBB);
912     SrcElementCurrent =
913         Address(SrcElementPHI,
914                 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize));
915   }
916   llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI(
917       DestBegin->getType(), 2, "omp.arraycpy.destElementPast");
918   DestElementPHI->addIncoming(DestBegin, EntryBB);
919   Address DestElementCurrent =
920       Address(DestElementPHI,
921               DestAddr.getAlignment().alignmentOfArrayElement(ElementSize));
922 
923   // Emit copy.
924   {
925     CodeGenFunction::RunCleanupsScope InitScope(CGF);
926     if (EmitDeclareReductionInit) {
927       emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent,
928                                        SrcElementCurrent, ElementTy);
929     } else
930       CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(),
931                            /*IsInitializer=*/false);
932   }
933 
934   if (DRD) {
935     // Shift the address forward by one element.
936     llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32(
937         SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
938     SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock());
939   }
940 
941   // Shift the address forward by one element.
942   llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32(
943       DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
944   // Check whether we've reached the end.
945   llvm::Value *Done =
946       CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done");
947   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
948   DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock());
949 
950   // Done.
951   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
952 }
953 
954 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) {
955   return CGF.EmitOMPSharedLValue(E);
956 }
957 
958 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF,
959                                             const Expr *E) {
960   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E))
961     return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false);
962   return LValue();
963 }
964 
965 void ReductionCodeGen::emitAggregateInitialization(
966     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
967     const OMPDeclareReductionDecl *DRD) {
968   // Emit VarDecl with copy init for arrays.
969   // Get the address of the original variable captured in current
970   // captured region.
971   const auto *PrivateVD =
972       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
973   bool EmitDeclareReductionInit =
974       DRD && (DRD->getInitializer() || !PrivateVD->hasInit());
975   EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(),
976                        EmitDeclareReductionInit,
977                        EmitDeclareReductionInit ? ClausesData[N].ReductionOp
978                                                 : PrivateVD->getInit(),
979                        DRD, SharedLVal.getAddress(CGF));
980 }
981 
982 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds,
983                                    ArrayRef<const Expr *> Privates,
984                                    ArrayRef<const Expr *> ReductionOps) {
985   ClausesData.reserve(Shareds.size());
986   SharedAddresses.reserve(Shareds.size());
987   Sizes.reserve(Shareds.size());
988   BaseDecls.reserve(Shareds.size());
989   auto IPriv = Privates.begin();
990   auto IRed = ReductionOps.begin();
991   for (const Expr *Ref : Shareds) {
992     ClausesData.emplace_back(Ref, *IPriv, *IRed);
993     std::advance(IPriv, 1);
994     std::advance(IRed, 1);
995   }
996 }
997 
998 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) {
999   assert(SharedAddresses.size() == N &&
1000          "Number of generated lvalues must be exactly N.");
1001   LValue First = emitSharedLValue(CGF, ClausesData[N].Ref);
1002   LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref);
1003   SharedAddresses.emplace_back(First, Second);
1004 }
1005 
1006 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) {
1007   const auto *PrivateVD =
1008       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1009   QualType PrivateType = PrivateVD->getType();
1010   bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref);
1011   if (!PrivateType->isVariablyModifiedType()) {
1012     Sizes.emplace_back(
1013         CGF.getTypeSize(
1014             SharedAddresses[N].first.getType().getNonReferenceType()),
1015         nullptr);
1016     return;
1017   }
1018   llvm::Value *Size;
1019   llvm::Value *SizeInChars;
1020   auto *ElemType = cast<llvm::PointerType>(
1021                        SharedAddresses[N].first.getPointer(CGF)->getType())
1022                        ->getElementType();
1023   auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType);
1024   if (AsArraySection) {
1025     Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(CGF),
1026                                      SharedAddresses[N].first.getPointer(CGF));
1027     Size = CGF.Builder.CreateNUWAdd(
1028         Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1));
1029     SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf);
1030   } else {
1031     SizeInChars = CGF.getTypeSize(
1032         SharedAddresses[N].first.getType().getNonReferenceType());
1033     Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf);
1034   }
1035   Sizes.emplace_back(SizeInChars, Size);
1036   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1037       CGF,
1038       cast<OpaqueValueExpr>(
1039           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1040       RValue::get(Size));
1041   CGF.EmitVariablyModifiedType(PrivateType);
1042 }
1043 
1044 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N,
1045                                          llvm::Value *Size) {
1046   const auto *PrivateVD =
1047       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1048   QualType PrivateType = PrivateVD->getType();
1049   if (!PrivateType->isVariablyModifiedType()) {
1050     assert(!Size && !Sizes[N].second &&
1051            "Size should be nullptr for non-variably modified reduction "
1052            "items.");
1053     return;
1054   }
1055   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1056       CGF,
1057       cast<OpaqueValueExpr>(
1058           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1059       RValue::get(Size));
1060   CGF.EmitVariablyModifiedType(PrivateType);
1061 }
1062 
1063 void ReductionCodeGen::emitInitialization(
1064     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
1065     llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) {
1066   assert(SharedAddresses.size() > N && "No variable was generated");
1067   const auto *PrivateVD =
1068       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1069   const OMPDeclareReductionDecl *DRD =
1070       getReductionInit(ClausesData[N].ReductionOp);
1071   QualType PrivateType = PrivateVD->getType();
1072   PrivateAddr = CGF.Builder.CreateElementBitCast(
1073       PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1074   QualType SharedType = SharedAddresses[N].first.getType();
1075   SharedLVal = CGF.MakeAddrLValue(
1076       CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(CGF),
1077                                        CGF.ConvertTypeForMem(SharedType)),
1078       SharedType, SharedAddresses[N].first.getBaseInfo(),
1079       CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType));
1080   if (CGF.getContext().getAsArrayType(PrivateVD->getType())) {
1081     emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD);
1082   } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) {
1083     emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp,
1084                                      PrivateAddr, SharedLVal.getAddress(CGF),
1085                                      SharedLVal.getType());
1086   } else if (!DefaultInit(CGF) && PrivateVD->hasInit() &&
1087              !CGF.isTrivialInitializer(PrivateVD->getInit())) {
1088     CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr,
1089                          PrivateVD->getType().getQualifiers(),
1090                          /*IsInitializer=*/false);
1091   }
1092 }
1093 
1094 bool ReductionCodeGen::needCleanups(unsigned N) {
1095   const auto *PrivateVD =
1096       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1097   QualType PrivateType = PrivateVD->getType();
1098   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1099   return DTorKind != QualType::DK_none;
1100 }
1101 
1102 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N,
1103                                     Address PrivateAddr) {
1104   const auto *PrivateVD =
1105       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1106   QualType PrivateType = PrivateVD->getType();
1107   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1108   if (needCleanups(N)) {
1109     PrivateAddr = CGF.Builder.CreateElementBitCast(
1110         PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1111     CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType);
1112   }
1113 }
1114 
1115 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1116                           LValue BaseLV) {
1117   BaseTy = BaseTy.getNonReferenceType();
1118   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1119          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1120     if (const auto *PtrTy = BaseTy->getAs<PointerType>()) {
1121       BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy);
1122     } else {
1123       LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy);
1124       BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal);
1125     }
1126     BaseTy = BaseTy->getPointeeType();
1127   }
1128   return CGF.MakeAddrLValue(
1129       CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF),
1130                                        CGF.ConvertTypeForMem(ElTy)),
1131       BaseLV.getType(), BaseLV.getBaseInfo(),
1132       CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType()));
1133 }
1134 
1135 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1136                           llvm::Type *BaseLVType, CharUnits BaseLVAlignment,
1137                           llvm::Value *Addr) {
1138   Address Tmp = Address::invalid();
1139   Address TopTmp = Address::invalid();
1140   Address MostTopTmp = Address::invalid();
1141   BaseTy = BaseTy.getNonReferenceType();
1142   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1143          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1144     Tmp = CGF.CreateMemTemp(BaseTy);
1145     if (TopTmp.isValid())
1146       CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp);
1147     else
1148       MostTopTmp = Tmp;
1149     TopTmp = Tmp;
1150     BaseTy = BaseTy->getPointeeType();
1151   }
1152   llvm::Type *Ty = BaseLVType;
1153   if (Tmp.isValid())
1154     Ty = Tmp.getElementType();
1155   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty);
1156   if (Tmp.isValid()) {
1157     CGF.Builder.CreateStore(Addr, Tmp);
1158     return MostTopTmp;
1159   }
1160   return Address(Addr, BaseLVAlignment);
1161 }
1162 
1163 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) {
1164   const VarDecl *OrigVD = nullptr;
1165   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) {
1166     const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
1167     while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base))
1168       Base = TempOASE->getBase()->IgnoreParenImpCasts();
1169     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1170       Base = TempASE->getBase()->IgnoreParenImpCasts();
1171     DE = cast<DeclRefExpr>(Base);
1172     OrigVD = cast<VarDecl>(DE->getDecl());
1173   } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) {
1174     const Expr *Base = ASE->getBase()->IgnoreParenImpCasts();
1175     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1176       Base = TempASE->getBase()->IgnoreParenImpCasts();
1177     DE = cast<DeclRefExpr>(Base);
1178     OrigVD = cast<VarDecl>(DE->getDecl());
1179   }
1180   return OrigVD;
1181 }
1182 
1183 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N,
1184                                                Address PrivateAddr) {
1185   const DeclRefExpr *DE;
1186   if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) {
1187     BaseDecls.emplace_back(OrigVD);
1188     LValue OriginalBaseLValue = CGF.EmitLValue(DE);
1189     LValue BaseLValue =
1190         loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(),
1191                     OriginalBaseLValue);
1192     llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff(
1193         BaseLValue.getPointer(CGF), SharedAddresses[N].first.getPointer(CGF));
1194     llvm::Value *PrivatePointer =
1195         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1196             PrivateAddr.getPointer(),
1197             SharedAddresses[N].first.getAddress(CGF).getType());
1198     llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment);
1199     return castToBase(CGF, OrigVD->getType(),
1200                       SharedAddresses[N].first.getType(),
1201                       OriginalBaseLValue.getAddress(CGF).getType(),
1202                       OriginalBaseLValue.getAlignment(), Ptr);
1203   }
1204   BaseDecls.emplace_back(
1205       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl()));
1206   return PrivateAddr;
1207 }
1208 
1209 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const {
1210   const OMPDeclareReductionDecl *DRD =
1211       getReductionInit(ClausesData[N].ReductionOp);
1212   return DRD && DRD->getInitializer();
1213 }
1214 
1215 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
1216   return CGF.EmitLoadOfPointerLValue(
1217       CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1218       getThreadIDVariable()->getType()->castAs<PointerType>());
1219 }
1220 
1221 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) {
1222   if (!CGF.HaveInsertPoint())
1223     return;
1224   // 1.2.2 OpenMP Language Terminology
1225   // Structured block - An executable statement with a single entry at the
1226   // top and a single exit at the bottom.
1227   // The point of exit cannot be a branch out of the structured block.
1228   // longjmp() and throw() must not violate the entry/exit criteria.
1229   CGF.EHStack.pushTerminate();
1230   CodeGen(CGF);
1231   CGF.EHStack.popTerminate();
1232 }
1233 
1234 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
1235     CodeGenFunction &CGF) {
1236   return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1237                             getThreadIDVariable()->getType(),
1238                             AlignmentSource::Decl);
1239 }
1240 
1241 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1242                                        QualType FieldTy) {
1243   auto *Field = FieldDecl::Create(
1244       C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1245       C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1246       /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1247   Field->setAccess(AS_public);
1248   DC->addDecl(Field);
1249   return Field;
1250 }
1251 
1252 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator,
1253                                  StringRef Separator)
1254     : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator),
1255       OffloadEntriesInfoManager(CGM) {
1256   ASTContext &C = CGM.getContext();
1257   RecordDecl *RD = C.buildImplicitRecord("ident_t");
1258   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
1259   RD->startDefinition();
1260   // reserved_1
1261   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1262   // flags
1263   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1264   // reserved_2
1265   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1266   // reserved_3
1267   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1268   // psource
1269   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
1270   RD->completeDefinition();
1271   IdentQTy = C.getRecordType(RD);
1272   IdentTy = CGM.getTypes().ConvertRecordDeclType(RD);
1273   KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
1274 
1275   loadOffloadInfoMetadata();
1276 }
1277 
1278 void CGOpenMPRuntime::clear() {
1279   InternalVars.clear();
1280   // Clean non-target variable declarations possibly used only in debug info.
1281   for (const auto &Data : EmittedNonTargetVariables) {
1282     if (!Data.getValue().pointsToAliveValue())
1283       continue;
1284     auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue());
1285     if (!GV)
1286       continue;
1287     if (!GV->isDeclaration() || GV->getNumUses() > 0)
1288       continue;
1289     GV->eraseFromParent();
1290   }
1291 }
1292 
1293 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const {
1294   SmallString<128> Buffer;
1295   llvm::raw_svector_ostream OS(Buffer);
1296   StringRef Sep = FirstSeparator;
1297   for (StringRef Part : Parts) {
1298     OS << Sep << Part;
1299     Sep = Separator;
1300   }
1301   return std::string(OS.str());
1302 }
1303 
1304 static llvm::Function *
1305 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty,
1306                           const Expr *CombinerInitializer, const VarDecl *In,
1307                           const VarDecl *Out, bool IsCombiner) {
1308   // void .omp_combiner.(Ty *in, Ty *out);
1309   ASTContext &C = CGM.getContext();
1310   QualType PtrTy = C.getPointerType(Ty).withRestrict();
1311   FunctionArgList Args;
1312   ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(),
1313                                /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1314   ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(),
1315                               /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1316   Args.push_back(&OmpOutParm);
1317   Args.push_back(&OmpInParm);
1318   const CGFunctionInfo &FnInfo =
1319       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
1320   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
1321   std::string Name = CGM.getOpenMPRuntime().getName(
1322       {IsCombiner ? "omp_combiner" : "omp_initializer", ""});
1323   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
1324                                     Name, &CGM.getModule());
1325   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
1326   if (CGM.getLangOpts().Optimize) {
1327     Fn->removeFnAttr(llvm::Attribute::NoInline);
1328     Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
1329     Fn->addFnAttr(llvm::Attribute::AlwaysInline);
1330   }
1331   CodeGenFunction CGF(CGM);
1332   // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions.
1333   // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions.
1334   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(),
1335                     Out->getLocation());
1336   CodeGenFunction::OMPPrivateScope Scope(CGF);
1337   Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm);
1338   Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() {
1339     return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>())
1340         .getAddress(CGF);
1341   });
1342   Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm);
1343   Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() {
1344     return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>())
1345         .getAddress(CGF);
1346   });
1347   (void)Scope.Privatize();
1348   if (!IsCombiner && Out->hasInit() &&
1349       !CGF.isTrivialInitializer(Out->getInit())) {
1350     CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out),
1351                          Out->getType().getQualifiers(),
1352                          /*IsInitializer=*/true);
1353   }
1354   if (CombinerInitializer)
1355     CGF.EmitIgnoredExpr(CombinerInitializer);
1356   Scope.ForceCleanup();
1357   CGF.FinishFunction();
1358   return Fn;
1359 }
1360 
1361 void CGOpenMPRuntime::emitUserDefinedReduction(
1362     CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) {
1363   if (UDRMap.count(D) > 0)
1364     return;
1365   llvm::Function *Combiner = emitCombinerOrInitializer(
1366       CGM, D->getType(), D->getCombiner(),
1367       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()),
1368       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()),
1369       /*IsCombiner=*/true);
1370   llvm::Function *Initializer = nullptr;
1371   if (const Expr *Init = D->getInitializer()) {
1372     Initializer = emitCombinerOrInitializer(
1373         CGM, D->getType(),
1374         D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init
1375                                                                      : nullptr,
1376         cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()),
1377         cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()),
1378         /*IsCombiner=*/false);
1379   }
1380   UDRMap.try_emplace(D, Combiner, Initializer);
1381   if (CGF) {
1382     auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn);
1383     Decls.second.push_back(D);
1384   }
1385 }
1386 
1387 std::pair<llvm::Function *, llvm::Function *>
1388 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) {
1389   auto I = UDRMap.find(D);
1390   if (I != UDRMap.end())
1391     return I->second;
1392   emitUserDefinedReduction(/*CGF=*/nullptr, D);
1393   return UDRMap.lookup(D);
1394 }
1395 
1396 namespace {
1397 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR
1398 // Builder if one is present.
1399 struct PushAndPopStackRAII {
1400   PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF,
1401                       bool HasCancel)
1402       : OMPBuilder(OMPBuilder) {
1403     if (!OMPBuilder)
1404       return;
1405 
1406     // The following callback is the crucial part of clangs cleanup process.
1407     //
1408     // NOTE:
1409     // Once the OpenMPIRBuilder is used to create parallel regions (and
1410     // similar), the cancellation destination (Dest below) is determined via
1411     // IP. That means if we have variables to finalize we split the block at IP,
1412     // use the new block (=BB) as destination to build a JumpDest (via
1413     // getJumpDestInCurrentScope(BB)) which then is fed to
1414     // EmitBranchThroughCleanup. Furthermore, there will not be the need
1415     // to push & pop an FinalizationInfo object.
1416     // The FiniCB will still be needed but at the point where the
1417     // OpenMPIRBuilder is asked to construct a parallel (or similar) construct.
1418     auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) {
1419       assert(IP.getBlock()->end() == IP.getPoint() &&
1420              "Clang CG should cause non-terminated block!");
1421       CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1422       CGF.Builder.restoreIP(IP);
1423       CodeGenFunction::JumpDest Dest =
1424           CGF.getOMPCancelDestination(OMPD_parallel);
1425       CGF.EmitBranchThroughCleanup(Dest);
1426     };
1427 
1428     // TODO: Remove this once we emit parallel regions through the
1429     //       OpenMPIRBuilder as it can do this setup internally.
1430     llvm::OpenMPIRBuilder::FinalizationInfo FI(
1431         {FiniCB, OMPD_parallel, HasCancel});
1432     OMPBuilder->pushFinalizationCB(std::move(FI));
1433   }
1434   ~PushAndPopStackRAII() {
1435     if (OMPBuilder)
1436       OMPBuilder->popFinalizationCB();
1437   }
1438   llvm::OpenMPIRBuilder *OMPBuilder;
1439 };
1440 } // namespace
1441 
1442 static llvm::Function *emitParallelOrTeamsOutlinedFunction(
1443     CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS,
1444     const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1445     const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) {
1446   assert(ThreadIDVar->getType()->isPointerType() &&
1447          "thread id variable must be of type kmp_int32 *");
1448   CodeGenFunction CGF(CGM, true);
1449   bool HasCancel = false;
1450   if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D))
1451     HasCancel = OPD->hasCancel();
1452   else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D))
1453     HasCancel = OPSD->hasCancel();
1454   else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D))
1455     HasCancel = OPFD->hasCancel();
1456   else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D))
1457     HasCancel = OPFD->hasCancel();
1458   else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D))
1459     HasCancel = OPFD->hasCancel();
1460   else if (const auto *OPFD =
1461                dyn_cast<OMPTeamsDistributeParallelForDirective>(&D))
1462     HasCancel = OPFD->hasCancel();
1463   else if (const auto *OPFD =
1464                dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D))
1465     HasCancel = OPFD->hasCancel();
1466 
1467   // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new
1468   //       parallel region to make cancellation barriers work properly.
1469   llvm::OpenMPIRBuilder *OMPBuilder = CGM.getOpenMPIRBuilder();
1470   PushAndPopStackRAII PSR(OMPBuilder, CGF, HasCancel);
1471   CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind,
1472                                     HasCancel, OutlinedHelperName);
1473   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1474   return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc());
1475 }
1476 
1477 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction(
1478     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1479     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1480   const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel);
1481   return emitParallelOrTeamsOutlinedFunction(
1482       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1483 }
1484 
1485 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction(
1486     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1487     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1488   const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams);
1489   return emitParallelOrTeamsOutlinedFunction(
1490       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1491 }
1492 
1493 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction(
1494     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1495     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
1496     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
1497     bool Tied, unsigned &NumberOfParts) {
1498   auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF,
1499                                               PrePostActionTy &) {
1500     llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc());
1501     llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc());
1502     llvm::Value *TaskArgs[] = {
1503         UpLoc, ThreadID,
1504         CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar),
1505                                     TaskTVar->getType()->castAs<PointerType>())
1506             .getPointer(CGF)};
1507     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs);
1508   };
1509   CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar,
1510                                                             UntiedCodeGen);
1511   CodeGen.setAction(Action);
1512   assert(!ThreadIDVar->getType()->isPointerType() &&
1513          "thread id variable must be of type kmp_int32 for tasks");
1514   const OpenMPDirectiveKind Region =
1515       isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop
1516                                                       : OMPD_task;
1517   const CapturedStmt *CS = D.getCapturedStmt(Region);
1518   bool HasCancel = false;
1519   if (const auto *TD = dyn_cast<OMPTaskDirective>(&D))
1520     HasCancel = TD->hasCancel();
1521   else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D))
1522     HasCancel = TD->hasCancel();
1523   else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D))
1524     HasCancel = TD->hasCancel();
1525   else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D))
1526     HasCancel = TD->hasCancel();
1527 
1528   CodeGenFunction CGF(CGM, true);
1529   CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen,
1530                                         InnermostKind, HasCancel, Action);
1531   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1532   llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS);
1533   if (!Tied)
1534     NumberOfParts = Action.getNumberOfParts();
1535   return Res;
1536 }
1537 
1538 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM,
1539                              const RecordDecl *RD, const CGRecordLayout &RL,
1540                              ArrayRef<llvm::Constant *> Data) {
1541   llvm::StructType *StructTy = RL.getLLVMType();
1542   unsigned PrevIdx = 0;
1543   ConstantInitBuilder CIBuilder(CGM);
1544   auto DI = Data.begin();
1545   for (const FieldDecl *FD : RD->fields()) {
1546     unsigned Idx = RL.getLLVMFieldNo(FD);
1547     // Fill the alignment.
1548     for (unsigned I = PrevIdx; I < Idx; ++I)
1549       Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I)));
1550     PrevIdx = Idx + 1;
1551     Fields.add(*DI);
1552     ++DI;
1553   }
1554 }
1555 
1556 template <class... As>
1557 static llvm::GlobalVariable *
1558 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant,
1559                    ArrayRef<llvm::Constant *> Data, const Twine &Name,
1560                    As &&... Args) {
1561   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1562   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1563   ConstantInitBuilder CIBuilder(CGM);
1564   ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType());
1565   buildStructValue(Fields, CGM, RD, RL, Data);
1566   return Fields.finishAndCreateGlobal(
1567       Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant,
1568       std::forward<As>(Args)...);
1569 }
1570 
1571 template <typename T>
1572 static void
1573 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty,
1574                                          ArrayRef<llvm::Constant *> Data,
1575                                          T &Parent) {
1576   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1577   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1578   ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType());
1579   buildStructValue(Fields, CGM, RD, RL, Data);
1580   Fields.finishAndAddTo(Parent);
1581 }
1582 
1583 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) {
1584   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1585   unsigned Reserved2Flags = getDefaultLocationReserved2Flags();
1586   FlagsTy FlagsKey(Flags, Reserved2Flags);
1587   llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey);
1588   if (!Entry) {
1589     if (!DefaultOpenMPPSource) {
1590       // Initialize default location for psource field of ident_t structure of
1591       // all ident_t objects. Format is ";file;function;line;column;;".
1592       // Taken from
1593       // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp
1594       DefaultOpenMPPSource =
1595           CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer();
1596       DefaultOpenMPPSource =
1597           llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy);
1598     }
1599 
1600     llvm::Constant *Data[] = {
1601         llvm::ConstantInt::getNullValue(CGM.Int32Ty),
1602         llvm::ConstantInt::get(CGM.Int32Ty, Flags),
1603         llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags),
1604         llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource};
1605     llvm::GlobalValue *DefaultOpenMPLocation =
1606         createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "",
1607                            llvm::GlobalValue::PrivateLinkage);
1608     DefaultOpenMPLocation->setUnnamedAddr(
1609         llvm::GlobalValue::UnnamedAddr::Global);
1610 
1611     OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation;
1612   }
1613   return Address(Entry, Align);
1614 }
1615 
1616 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF,
1617                                              bool AtCurrentPoint) {
1618   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1619   assert(!Elem.second.ServiceInsertPt && "Insert point is set already.");
1620 
1621   llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty);
1622   if (AtCurrentPoint) {
1623     Elem.second.ServiceInsertPt = new llvm::BitCastInst(
1624         Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock());
1625   } else {
1626     Elem.second.ServiceInsertPt =
1627         new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt");
1628     Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt);
1629   }
1630 }
1631 
1632 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) {
1633   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1634   if (Elem.second.ServiceInsertPt) {
1635     llvm::Instruction *Ptr = Elem.second.ServiceInsertPt;
1636     Elem.second.ServiceInsertPt = nullptr;
1637     Ptr->eraseFromParent();
1638   }
1639 }
1640 
1641 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
1642                                                  SourceLocation Loc,
1643                                                  unsigned Flags) {
1644   Flags |= OMP_IDENT_KMPC;
1645   // If no debug info is generated - return global default location.
1646   if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo ||
1647       Loc.isInvalid())
1648     return getOrCreateDefaultLocation(Flags).getPointer();
1649 
1650   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1651 
1652   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1653   Address LocValue = Address::invalid();
1654   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1655   if (I != OpenMPLocThreadIDMap.end())
1656     LocValue = Address(I->second.DebugLoc, Align);
1657 
1658   // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if
1659   // GetOpenMPThreadID was called before this routine.
1660   if (!LocValue.isValid()) {
1661     // Generate "ident_t .kmpc_loc.addr;"
1662     Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr");
1663     auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1664     Elem.second.DebugLoc = AI.getPointer();
1665     LocValue = AI;
1666 
1667     if (!Elem.second.ServiceInsertPt)
1668       setLocThreadIdInsertPt(CGF);
1669     CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1670     CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1671     CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags),
1672                              CGF.getTypeSize(IdentQTy));
1673   }
1674 
1675   // char **psource = &.kmpc_loc_<flags>.addr.psource;
1676   LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy);
1677   auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin();
1678   LValue PSource =
1679       CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource));
1680 
1681   llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding());
1682   if (OMPDebugLoc == nullptr) {
1683     SmallString<128> Buffer2;
1684     llvm::raw_svector_ostream OS2(Buffer2);
1685     // Build debug location
1686     PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1687     OS2 << ";" << PLoc.getFilename() << ";";
1688     if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1689       OS2 << FD->getQualifiedNameAsString();
1690     OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
1691     OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str());
1692     OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc;
1693   }
1694   // *psource = ";<File>;<Function>;<Line>;<Column>;;";
1695   CGF.EmitStoreOfScalar(OMPDebugLoc, PSource);
1696 
1697   // Our callers always pass this to a runtime function, so for
1698   // convenience, go ahead and return a naked pointer.
1699   return LocValue.getPointer();
1700 }
1701 
1702 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
1703                                           SourceLocation Loc) {
1704   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1705 
1706   llvm::Value *ThreadID = nullptr;
1707   // Check whether we've already cached a load of the thread id in this
1708   // function.
1709   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1710   if (I != OpenMPLocThreadIDMap.end()) {
1711     ThreadID = I->second.ThreadID;
1712     if (ThreadID != nullptr)
1713       return ThreadID;
1714   }
1715   // If exceptions are enabled, do not use parameter to avoid possible crash.
1716   if (auto *OMPRegionInfo =
1717           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
1718     if (OMPRegionInfo->getThreadIDVariable()) {
1719       // Check if this an outlined function with thread id passed as argument.
1720       LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
1721       llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent();
1722       if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions ||
1723           !CGF.getLangOpts().CXXExceptions ||
1724           CGF.Builder.GetInsertBlock() == TopBlock ||
1725           !isa<llvm::Instruction>(LVal.getPointer(CGF)) ||
1726           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1727               TopBlock ||
1728           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1729               CGF.Builder.GetInsertBlock()) {
1730         ThreadID = CGF.EmitLoadOfScalar(LVal, Loc);
1731         // If value loaded in entry block, cache it and use it everywhere in
1732         // function.
1733         if (CGF.Builder.GetInsertBlock() == TopBlock) {
1734           auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1735           Elem.second.ThreadID = ThreadID;
1736         }
1737         return ThreadID;
1738       }
1739     }
1740   }
1741 
1742   // This is not an outlined function region - need to call __kmpc_int32
1743   // kmpc_global_thread_num(ident_t *loc).
1744   // Generate thread id value and cache this value for use across the
1745   // function.
1746   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1747   if (!Elem.second.ServiceInsertPt)
1748     setLocThreadIdInsertPt(CGF);
1749   CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1750   CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1751   llvm::CallInst *Call = CGF.Builder.CreateCall(
1752       createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
1753       emitUpdateLocation(CGF, Loc));
1754   Call->setCallingConv(CGF.getRuntimeCC());
1755   Elem.second.ThreadID = Call;
1756   return Call;
1757 }
1758 
1759 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
1760   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1761   if (OpenMPLocThreadIDMap.count(CGF.CurFn)) {
1762     clearLocThreadIdInsertPt(CGF);
1763     OpenMPLocThreadIDMap.erase(CGF.CurFn);
1764   }
1765   if (FunctionUDRMap.count(CGF.CurFn) > 0) {
1766     for(const auto *D : FunctionUDRMap[CGF.CurFn])
1767       UDRMap.erase(D);
1768     FunctionUDRMap.erase(CGF.CurFn);
1769   }
1770   auto I = FunctionUDMMap.find(CGF.CurFn);
1771   if (I != FunctionUDMMap.end()) {
1772     for(const auto *D : I->second)
1773       UDMMap.erase(D);
1774     FunctionUDMMap.erase(I);
1775   }
1776   LastprivateConditionalToTypes.erase(CGF.CurFn);
1777 }
1778 
1779 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
1780   return IdentTy->getPointerTo();
1781 }
1782 
1783 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() {
1784   if (!Kmpc_MicroTy) {
1785     // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...)
1786     llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty),
1787                                  llvm::PointerType::getUnqual(CGM.Int32Ty)};
1788     Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true);
1789   }
1790   return llvm::PointerType::getUnqual(Kmpc_MicroTy);
1791 }
1792 
1793 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) {
1794   llvm::FunctionCallee RTLFn = nullptr;
1795   switch (static_cast<OpenMPRTLFunction>(Function)) {
1796   case OMPRTL__kmpc_fork_call: {
1797     // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro
1798     // microtask, ...);
1799     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1800                                 getKmpc_MicroPointerTy()};
1801     auto *FnTy =
1802         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
1803     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call");
1804     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
1805       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
1806         llvm::LLVMContext &Ctx = F->getContext();
1807         llvm::MDBuilder MDB(Ctx);
1808         // Annotate the callback behavior of the __kmpc_fork_call:
1809         //  - The callback callee is argument number 2 (microtask).
1810         //  - The first two arguments of the callback callee are unknown (-1).
1811         //  - All variadic arguments to the __kmpc_fork_call are passed to the
1812         //    callback callee.
1813         F->addMetadata(
1814             llvm::LLVMContext::MD_callback,
1815             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
1816                                         2, {-1, -1},
1817                                         /* VarArgsArePassed */ true)}));
1818       }
1819     }
1820     break;
1821   }
1822   case OMPRTL__kmpc_global_thread_num: {
1823     // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc);
1824     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1825     auto *FnTy =
1826         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1827     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num");
1828     break;
1829   }
1830   case OMPRTL__kmpc_threadprivate_cached: {
1831     // Build void *__kmpc_threadprivate_cached(ident_t *loc,
1832     // kmp_int32 global_tid, void *data, size_t size, void ***cache);
1833     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1834                                 CGM.VoidPtrTy, CGM.SizeTy,
1835                                 CGM.VoidPtrTy->getPointerTo()->getPointerTo()};
1836     auto *FnTy =
1837         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false);
1838     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached");
1839     break;
1840   }
1841   case OMPRTL__kmpc_critical: {
1842     // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
1843     // kmp_critical_name *crit);
1844     llvm::Type *TypeParams[] = {
1845         getIdentTyPointerTy(), CGM.Int32Ty,
1846         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1847     auto *FnTy =
1848         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1849     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical");
1850     break;
1851   }
1852   case OMPRTL__kmpc_critical_with_hint: {
1853     // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid,
1854     // kmp_critical_name *crit, uintptr_t hint);
1855     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1856                                 llvm::PointerType::getUnqual(KmpCriticalNameTy),
1857                                 CGM.IntPtrTy};
1858     auto *FnTy =
1859         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1860     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint");
1861     break;
1862   }
1863   case OMPRTL__kmpc_threadprivate_register: {
1864     // Build void __kmpc_threadprivate_register(ident_t *, void *data,
1865     // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
1866     // typedef void *(*kmpc_ctor)(void *);
1867     auto *KmpcCtorTy =
1868         llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
1869                                 /*isVarArg*/ false)->getPointerTo();
1870     // typedef void *(*kmpc_cctor)(void *, void *);
1871     llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
1872     auto *KmpcCopyCtorTy =
1873         llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs,
1874                                 /*isVarArg*/ false)
1875             ->getPointerTo();
1876     // typedef void (*kmpc_dtor)(void *);
1877     auto *KmpcDtorTy =
1878         llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false)
1879             ->getPointerTo();
1880     llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy,
1881                               KmpcCopyCtorTy, KmpcDtorTy};
1882     auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs,
1883                                         /*isVarArg*/ false);
1884     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register");
1885     break;
1886   }
1887   case OMPRTL__kmpc_end_critical: {
1888     // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
1889     // kmp_critical_name *crit);
1890     llvm::Type *TypeParams[] = {
1891         getIdentTyPointerTy(), CGM.Int32Ty,
1892         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1893     auto *FnTy =
1894         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1895     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical");
1896     break;
1897   }
1898   case OMPRTL__kmpc_cancel_barrier: {
1899     // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
1900     // global_tid);
1901     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1902     auto *FnTy =
1903         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1904     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier");
1905     break;
1906   }
1907   case OMPRTL__kmpc_barrier: {
1908     // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
1909     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1910     auto *FnTy =
1911         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1912     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier");
1913     break;
1914   }
1915   case OMPRTL__kmpc_for_static_fini: {
1916     // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
1917     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1918     auto *FnTy =
1919         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1920     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini");
1921     break;
1922   }
1923   case OMPRTL__kmpc_push_num_threads: {
1924     // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
1925     // kmp_int32 num_threads)
1926     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1927                                 CGM.Int32Ty};
1928     auto *FnTy =
1929         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1930     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads");
1931     break;
1932   }
1933   case OMPRTL__kmpc_serialized_parallel: {
1934     // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
1935     // global_tid);
1936     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1937     auto *FnTy =
1938         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1939     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel");
1940     break;
1941   }
1942   case OMPRTL__kmpc_end_serialized_parallel: {
1943     // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
1944     // global_tid);
1945     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1946     auto *FnTy =
1947         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1948     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel");
1949     break;
1950   }
1951   case OMPRTL__kmpc_flush: {
1952     // Build void __kmpc_flush(ident_t *loc);
1953     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1954     auto *FnTy =
1955         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1956     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush");
1957     break;
1958   }
1959   case OMPRTL__kmpc_master: {
1960     // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid);
1961     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1962     auto *FnTy =
1963         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1964     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master");
1965     break;
1966   }
1967   case OMPRTL__kmpc_end_master: {
1968     // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid);
1969     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1970     auto *FnTy =
1971         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
1972     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master");
1973     break;
1974   }
1975   case OMPRTL__kmpc_omp_taskyield: {
1976     // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
1977     // int end_part);
1978     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
1979     auto *FnTy =
1980         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1981     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield");
1982     break;
1983   }
1984   case OMPRTL__kmpc_single: {
1985     // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid);
1986     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1987     auto *FnTy =
1988         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1989     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single");
1990     break;
1991   }
1992   case OMPRTL__kmpc_end_single: {
1993     // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid);
1994     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1995     auto *FnTy =
1996         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
1997     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single");
1998     break;
1999   }
2000   case OMPRTL__kmpc_omp_task_alloc: {
2001     // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
2002     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
2003     // kmp_routine_entry_t *task_entry);
2004     assert(KmpRoutineEntryPtrTy != nullptr &&
2005            "Type kmp_routine_entry_t must be created.");
2006     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2007                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy};
2008     // Return void * and then cast to particular kmp_task_t type.
2009     auto *FnTy =
2010         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2011     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc");
2012     break;
2013   }
2014   case OMPRTL__kmpc_omp_target_task_alloc: {
2015     // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid,
2016     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
2017     // kmp_routine_entry_t *task_entry, kmp_int64 device_id);
2018     assert(KmpRoutineEntryPtrTy != nullptr &&
2019            "Type kmp_routine_entry_t must be created.");
2020     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2021                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy,
2022                                 CGM.Int64Ty};
2023     // Return void * and then cast to particular kmp_task_t type.
2024     auto *FnTy =
2025         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2026     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc");
2027     break;
2028   }
2029   case OMPRTL__kmpc_omp_task: {
2030     // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2031     // *new_task);
2032     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2033                                 CGM.VoidPtrTy};
2034     auto *FnTy =
2035         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2036     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task");
2037     break;
2038   }
2039   case OMPRTL__kmpc_copyprivate: {
2040     // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
2041     // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
2042     // kmp_int32 didit);
2043     llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2044     auto *CpyFnTy =
2045         llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false);
2046     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy,
2047                                 CGM.VoidPtrTy, CpyFnTy->getPointerTo(),
2048                                 CGM.Int32Ty};
2049     auto *FnTy =
2050         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2051     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate");
2052     break;
2053   }
2054   case OMPRTL__kmpc_reduce: {
2055     // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
2056     // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
2057     // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
2058     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2059     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2060                                                /*isVarArg=*/false);
2061     llvm::Type *TypeParams[] = {
2062         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2063         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2064         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2065     auto *FnTy =
2066         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2067     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce");
2068     break;
2069   }
2070   case OMPRTL__kmpc_reduce_nowait: {
2071     // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
2072     // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
2073     // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
2074     // *lck);
2075     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2076     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2077                                                /*isVarArg=*/false);
2078     llvm::Type *TypeParams[] = {
2079         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2080         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2081         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2082     auto *FnTy =
2083         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2084     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait");
2085     break;
2086   }
2087   case OMPRTL__kmpc_end_reduce: {
2088     // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
2089     // kmp_critical_name *lck);
2090     llvm::Type *TypeParams[] = {
2091         getIdentTyPointerTy(), CGM.Int32Ty,
2092         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2093     auto *FnTy =
2094         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2095     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce");
2096     break;
2097   }
2098   case OMPRTL__kmpc_end_reduce_nowait: {
2099     // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
2100     // kmp_critical_name *lck);
2101     llvm::Type *TypeParams[] = {
2102         getIdentTyPointerTy(), CGM.Int32Ty,
2103         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2104     auto *FnTy =
2105         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2106     RTLFn =
2107         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait");
2108     break;
2109   }
2110   case OMPRTL__kmpc_omp_task_begin_if0: {
2111     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2112     // *new_task);
2113     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2114                                 CGM.VoidPtrTy};
2115     auto *FnTy =
2116         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2117     RTLFn =
2118         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0");
2119     break;
2120   }
2121   case OMPRTL__kmpc_omp_task_complete_if0: {
2122     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2123     // *new_task);
2124     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2125                                 CGM.VoidPtrTy};
2126     auto *FnTy =
2127         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2128     RTLFn = CGM.CreateRuntimeFunction(FnTy,
2129                                       /*Name=*/"__kmpc_omp_task_complete_if0");
2130     break;
2131   }
2132   case OMPRTL__kmpc_ordered: {
2133     // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
2134     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2135     auto *FnTy =
2136         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2137     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered");
2138     break;
2139   }
2140   case OMPRTL__kmpc_end_ordered: {
2141     // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
2142     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2143     auto *FnTy =
2144         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2145     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered");
2146     break;
2147   }
2148   case OMPRTL__kmpc_omp_taskwait: {
2149     // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid);
2150     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2151     auto *FnTy =
2152         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2153     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait");
2154     break;
2155   }
2156   case OMPRTL__kmpc_taskgroup: {
2157     // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
2158     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2159     auto *FnTy =
2160         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2161     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup");
2162     break;
2163   }
2164   case OMPRTL__kmpc_end_taskgroup: {
2165     // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
2166     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2167     auto *FnTy =
2168         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2169     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup");
2170     break;
2171   }
2172   case OMPRTL__kmpc_push_proc_bind: {
2173     // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
2174     // int proc_bind)
2175     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2176     auto *FnTy =
2177         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2178     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind");
2179     break;
2180   }
2181   case OMPRTL__kmpc_omp_task_with_deps: {
2182     // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
2183     // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
2184     // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
2185     llvm::Type *TypeParams[] = {
2186         getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty,
2187         CGM.VoidPtrTy,         CGM.Int32Ty, CGM.VoidPtrTy};
2188     auto *FnTy =
2189         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2190     RTLFn =
2191         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps");
2192     break;
2193   }
2194   case OMPRTL__kmpc_omp_wait_deps: {
2195     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
2196     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias,
2197     // kmp_depend_info_t *noalias_dep_list);
2198     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2199                                 CGM.Int32Ty,           CGM.VoidPtrTy,
2200                                 CGM.Int32Ty,           CGM.VoidPtrTy};
2201     auto *FnTy =
2202         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2203     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps");
2204     break;
2205   }
2206   case OMPRTL__kmpc_cancellationpoint: {
2207     // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
2208     // global_tid, kmp_int32 cncl_kind)
2209     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2210     auto *FnTy =
2211         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2212     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint");
2213     break;
2214   }
2215   case OMPRTL__kmpc_cancel: {
2216     // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
2217     // kmp_int32 cncl_kind)
2218     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2219     auto *FnTy =
2220         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2221     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel");
2222     break;
2223   }
2224   case OMPRTL__kmpc_push_num_teams: {
2225     // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid,
2226     // kmp_int32 num_teams, kmp_int32 num_threads)
2227     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2228         CGM.Int32Ty};
2229     auto *FnTy =
2230         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2231     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams");
2232     break;
2233   }
2234   case OMPRTL__kmpc_fork_teams: {
2235     // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
2236     // microtask, ...);
2237     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2238                                 getKmpc_MicroPointerTy()};
2239     auto *FnTy =
2240         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
2241     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams");
2242     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
2243       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
2244         llvm::LLVMContext &Ctx = F->getContext();
2245         llvm::MDBuilder MDB(Ctx);
2246         // Annotate the callback behavior of the __kmpc_fork_teams:
2247         //  - The callback callee is argument number 2 (microtask).
2248         //  - The first two arguments of the callback callee are unknown (-1).
2249         //  - All variadic arguments to the __kmpc_fork_teams are passed to the
2250         //    callback callee.
2251         F->addMetadata(
2252             llvm::LLVMContext::MD_callback,
2253             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
2254                                         2, {-1, -1},
2255                                         /* VarArgsArePassed */ true)}));
2256       }
2257     }
2258     break;
2259   }
2260   case OMPRTL__kmpc_taskloop: {
2261     // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
2262     // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
2263     // sched, kmp_uint64 grainsize, void *task_dup);
2264     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2265                                 CGM.IntTy,
2266                                 CGM.VoidPtrTy,
2267                                 CGM.IntTy,
2268                                 CGM.Int64Ty->getPointerTo(),
2269                                 CGM.Int64Ty->getPointerTo(),
2270                                 CGM.Int64Ty,
2271                                 CGM.IntTy,
2272                                 CGM.IntTy,
2273                                 CGM.Int64Ty,
2274                                 CGM.VoidPtrTy};
2275     auto *FnTy =
2276         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2277     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop");
2278     break;
2279   }
2280   case OMPRTL__kmpc_doacross_init: {
2281     // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
2282     // num_dims, struct kmp_dim *dims);
2283     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2284                                 CGM.Int32Ty,
2285                                 CGM.Int32Ty,
2286                                 CGM.VoidPtrTy};
2287     auto *FnTy =
2288         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2289     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init");
2290     break;
2291   }
2292   case OMPRTL__kmpc_doacross_fini: {
2293     // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
2294     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2295     auto *FnTy =
2296         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2297     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini");
2298     break;
2299   }
2300   case OMPRTL__kmpc_doacross_post: {
2301     // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
2302     // *vec);
2303     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2304                                 CGM.Int64Ty->getPointerTo()};
2305     auto *FnTy =
2306         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2307     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post");
2308     break;
2309   }
2310   case OMPRTL__kmpc_doacross_wait: {
2311     // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
2312     // *vec);
2313     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2314                                 CGM.Int64Ty->getPointerTo()};
2315     auto *FnTy =
2316         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2317     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait");
2318     break;
2319   }
2320   case OMPRTL__kmpc_task_reduction_init: {
2321     // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void
2322     // *data);
2323     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy};
2324     auto *FnTy =
2325         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2326     RTLFn =
2327         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init");
2328     break;
2329   }
2330   case OMPRTL__kmpc_task_reduction_get_th_data: {
2331     // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
2332     // *d);
2333     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2334     auto *FnTy =
2335         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2336     RTLFn = CGM.CreateRuntimeFunction(
2337         FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data");
2338     break;
2339   }
2340   case OMPRTL__kmpc_alloc: {
2341     // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t
2342     // al); omp_allocator_handle_t type is void *.
2343     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy};
2344     auto *FnTy =
2345         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2346     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc");
2347     break;
2348   }
2349   case OMPRTL__kmpc_free: {
2350     // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t
2351     // al); omp_allocator_handle_t type is void *.
2352     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2353     auto *FnTy =
2354         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2355     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free");
2356     break;
2357   }
2358   case OMPRTL__kmpc_push_target_tripcount: {
2359     // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
2360     // size);
2361     llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty};
2362     llvm::FunctionType *FnTy =
2363         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2364     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount");
2365     break;
2366   }
2367   case OMPRTL__tgt_target: {
2368     // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
2369     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2370     // *arg_types);
2371     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2372                                 CGM.VoidPtrTy,
2373                                 CGM.Int32Ty,
2374                                 CGM.VoidPtrPtrTy,
2375                                 CGM.VoidPtrPtrTy,
2376                                 CGM.Int64Ty->getPointerTo(),
2377                                 CGM.Int64Ty->getPointerTo()};
2378     auto *FnTy =
2379         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2380     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target");
2381     break;
2382   }
2383   case OMPRTL__tgt_target_nowait: {
2384     // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
2385     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2386     // int64_t *arg_types);
2387     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2388                                 CGM.VoidPtrTy,
2389                                 CGM.Int32Ty,
2390                                 CGM.VoidPtrPtrTy,
2391                                 CGM.VoidPtrPtrTy,
2392                                 CGM.Int64Ty->getPointerTo(),
2393                                 CGM.Int64Ty->getPointerTo()};
2394     auto *FnTy =
2395         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2396     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait");
2397     break;
2398   }
2399   case OMPRTL__tgt_target_teams: {
2400     // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
2401     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2402     // int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2403     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2404                                 CGM.VoidPtrTy,
2405                                 CGM.Int32Ty,
2406                                 CGM.VoidPtrPtrTy,
2407                                 CGM.VoidPtrPtrTy,
2408                                 CGM.Int64Ty->getPointerTo(),
2409                                 CGM.Int64Ty->getPointerTo(),
2410                                 CGM.Int32Ty,
2411                                 CGM.Int32Ty};
2412     auto *FnTy =
2413         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2414     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams");
2415     break;
2416   }
2417   case OMPRTL__tgt_target_teams_nowait: {
2418     // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void
2419     // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
2420     // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2421     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2422                                 CGM.VoidPtrTy,
2423                                 CGM.Int32Ty,
2424                                 CGM.VoidPtrPtrTy,
2425                                 CGM.VoidPtrPtrTy,
2426                                 CGM.Int64Ty->getPointerTo(),
2427                                 CGM.Int64Ty->getPointerTo(),
2428                                 CGM.Int32Ty,
2429                                 CGM.Int32Ty};
2430     auto *FnTy =
2431         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2432     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait");
2433     break;
2434   }
2435   case OMPRTL__tgt_register_requires: {
2436     // Build void __tgt_register_requires(int64_t flags);
2437     llvm::Type *TypeParams[] = {CGM.Int64Ty};
2438     auto *FnTy =
2439         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2440     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires");
2441     break;
2442   }
2443   case OMPRTL__tgt_target_data_begin: {
2444     // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
2445     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2446     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2447                                 CGM.Int32Ty,
2448                                 CGM.VoidPtrPtrTy,
2449                                 CGM.VoidPtrPtrTy,
2450                                 CGM.Int64Ty->getPointerTo(),
2451                                 CGM.Int64Ty->getPointerTo()};
2452     auto *FnTy =
2453         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2454     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin");
2455     break;
2456   }
2457   case OMPRTL__tgt_target_data_begin_nowait: {
2458     // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
2459     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2460     // *arg_types);
2461     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2462                                 CGM.Int32Ty,
2463                                 CGM.VoidPtrPtrTy,
2464                                 CGM.VoidPtrPtrTy,
2465                                 CGM.Int64Ty->getPointerTo(),
2466                                 CGM.Int64Ty->getPointerTo()};
2467     auto *FnTy =
2468         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2469     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait");
2470     break;
2471   }
2472   case OMPRTL__tgt_target_data_end: {
2473     // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
2474     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2475     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2476                                 CGM.Int32Ty,
2477                                 CGM.VoidPtrPtrTy,
2478                                 CGM.VoidPtrPtrTy,
2479                                 CGM.Int64Ty->getPointerTo(),
2480                                 CGM.Int64Ty->getPointerTo()};
2481     auto *FnTy =
2482         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2483     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end");
2484     break;
2485   }
2486   case OMPRTL__tgt_target_data_end_nowait: {
2487     // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t
2488     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2489     // *arg_types);
2490     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2491                                 CGM.Int32Ty,
2492                                 CGM.VoidPtrPtrTy,
2493                                 CGM.VoidPtrPtrTy,
2494                                 CGM.Int64Ty->getPointerTo(),
2495                                 CGM.Int64Ty->getPointerTo()};
2496     auto *FnTy =
2497         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2498     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait");
2499     break;
2500   }
2501   case OMPRTL__tgt_target_data_update: {
2502     // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
2503     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2504     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2505                                 CGM.Int32Ty,
2506                                 CGM.VoidPtrPtrTy,
2507                                 CGM.VoidPtrPtrTy,
2508                                 CGM.Int64Ty->getPointerTo(),
2509                                 CGM.Int64Ty->getPointerTo()};
2510     auto *FnTy =
2511         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2512     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update");
2513     break;
2514   }
2515   case OMPRTL__tgt_target_data_update_nowait: {
2516     // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t
2517     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2518     // *arg_types);
2519     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2520                                 CGM.Int32Ty,
2521                                 CGM.VoidPtrPtrTy,
2522                                 CGM.VoidPtrPtrTy,
2523                                 CGM.Int64Ty->getPointerTo(),
2524                                 CGM.Int64Ty->getPointerTo()};
2525     auto *FnTy =
2526         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2527     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait");
2528     break;
2529   }
2530   case OMPRTL__tgt_mapper_num_components: {
2531     // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
2532     llvm::Type *TypeParams[] = {CGM.VoidPtrTy};
2533     auto *FnTy =
2534         llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false);
2535     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components");
2536     break;
2537   }
2538   case OMPRTL__tgt_push_mapper_component: {
2539     // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void
2540     // *base, void *begin, int64_t size, int64_t type);
2541     llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy,
2542                                 CGM.Int64Ty, CGM.Int64Ty};
2543     auto *FnTy =
2544         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2545     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component");
2546     break;
2547   }
2548   case OMPRTL__kmpc_task_allow_completion_event: {
2549     // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
2550     // int gtid, kmp_task_t *task);
2551     auto *FnTy = llvm::FunctionType::get(
2552         CGM.VoidPtrTy, {getIdentTyPointerTy(), CGM.IntTy, CGM.VoidPtrTy},
2553         /*isVarArg=*/false);
2554     RTLFn =
2555         CGM.CreateRuntimeFunction(FnTy, "__kmpc_task_allow_completion_event");
2556     break;
2557   }
2558   }
2559   assert(RTLFn && "Unable to find OpenMP runtime function");
2560   return RTLFn;
2561 }
2562 
2563 llvm::FunctionCallee
2564 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) {
2565   assert((IVSize == 32 || IVSize == 64) &&
2566          "IV size is not compatible with the omp runtime");
2567   StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4"
2568                                             : "__kmpc_for_static_init_4u")
2569                                 : (IVSigned ? "__kmpc_for_static_init_8"
2570                                             : "__kmpc_for_static_init_8u");
2571   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2572   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2573   llvm::Type *TypeParams[] = {
2574     getIdentTyPointerTy(),                     // loc
2575     CGM.Int32Ty,                               // tid
2576     CGM.Int32Ty,                               // schedtype
2577     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2578     PtrTy,                                     // p_lower
2579     PtrTy,                                     // p_upper
2580     PtrTy,                                     // p_stride
2581     ITy,                                       // incr
2582     ITy                                        // chunk
2583   };
2584   auto *FnTy =
2585       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2586   return CGM.CreateRuntimeFunction(FnTy, Name);
2587 }
2588 
2589 llvm::FunctionCallee
2590 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) {
2591   assert((IVSize == 32 || IVSize == 64) &&
2592          "IV size is not compatible with the omp runtime");
2593   StringRef Name =
2594       IVSize == 32
2595           ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u")
2596           : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u");
2597   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2598   llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc
2599                                CGM.Int32Ty,           // tid
2600                                CGM.Int32Ty,           // schedtype
2601                                ITy,                   // lower
2602                                ITy,                   // upper
2603                                ITy,                   // stride
2604                                ITy                    // chunk
2605   };
2606   auto *FnTy =
2607       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2608   return CGM.CreateRuntimeFunction(FnTy, Name);
2609 }
2610 
2611 llvm::FunctionCallee
2612 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) {
2613   assert((IVSize == 32 || IVSize == 64) &&
2614          "IV size is not compatible with the omp runtime");
2615   StringRef Name =
2616       IVSize == 32
2617           ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u")
2618           : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u");
2619   llvm::Type *TypeParams[] = {
2620       getIdentTyPointerTy(), // loc
2621       CGM.Int32Ty,           // tid
2622   };
2623   auto *FnTy =
2624       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2625   return CGM.CreateRuntimeFunction(FnTy, Name);
2626 }
2627 
2628 llvm::FunctionCallee
2629 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) {
2630   assert((IVSize == 32 || IVSize == 64) &&
2631          "IV size is not compatible with the omp runtime");
2632   StringRef Name =
2633       IVSize == 32
2634           ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u")
2635           : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u");
2636   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2637   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2638   llvm::Type *TypeParams[] = {
2639     getIdentTyPointerTy(),                     // loc
2640     CGM.Int32Ty,                               // tid
2641     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2642     PtrTy,                                     // p_lower
2643     PtrTy,                                     // p_upper
2644     PtrTy                                      // p_stride
2645   };
2646   auto *FnTy =
2647       llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2648   return CGM.CreateRuntimeFunction(FnTy, Name);
2649 }
2650 
2651 /// Obtain information that uniquely identifies a target entry. This
2652 /// consists of the file and device IDs as well as line number associated with
2653 /// the relevant entry source location.
2654 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc,
2655                                      unsigned &DeviceID, unsigned &FileID,
2656                                      unsigned &LineNum) {
2657   SourceManager &SM = C.getSourceManager();
2658 
2659   // The loc should be always valid and have a file ID (the user cannot use
2660   // #pragma directives in macros)
2661 
2662   assert(Loc.isValid() && "Source location is expected to be always valid.");
2663 
2664   PresumedLoc PLoc = SM.getPresumedLoc(Loc);
2665   assert(PLoc.isValid() && "Source location is expected to be always valid.");
2666 
2667   llvm::sys::fs::UniqueID ID;
2668   if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID))
2669     SM.getDiagnostics().Report(diag::err_cannot_open_file)
2670         << PLoc.getFilename() << EC.message();
2671 
2672   DeviceID = ID.getDevice();
2673   FileID = ID.getFile();
2674   LineNum = PLoc.getLine();
2675 }
2676 
2677 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) {
2678   if (CGM.getLangOpts().OpenMPSimd)
2679     return Address::invalid();
2680   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2681       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2682   if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link ||
2683               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2684                HasRequiresUnifiedSharedMemory))) {
2685     SmallString<64> PtrName;
2686     {
2687       llvm::raw_svector_ostream OS(PtrName);
2688       OS << CGM.getMangledName(GlobalDecl(VD));
2689       if (!VD->isExternallyVisible()) {
2690         unsigned DeviceID, FileID, Line;
2691         getTargetEntryUniqueInfo(CGM.getContext(),
2692                                  VD->getCanonicalDecl()->getBeginLoc(),
2693                                  DeviceID, FileID, Line);
2694         OS << llvm::format("_%x", FileID);
2695       }
2696       OS << "_decl_tgt_ref_ptr";
2697     }
2698     llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName);
2699     if (!Ptr) {
2700       QualType PtrTy = CGM.getContext().getPointerType(VD->getType());
2701       Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy),
2702                                         PtrName);
2703 
2704       auto *GV = cast<llvm::GlobalVariable>(Ptr);
2705       GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
2706 
2707       if (!CGM.getLangOpts().OpenMPIsDevice)
2708         GV->setInitializer(CGM.GetAddrOfGlobal(VD));
2709       registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr));
2710     }
2711     return Address(Ptr, CGM.getContext().getDeclAlign(VD));
2712   }
2713   return Address::invalid();
2714 }
2715 
2716 llvm::Constant *
2717 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
2718   assert(!CGM.getLangOpts().OpenMPUseTLS ||
2719          !CGM.getContext().getTargetInfo().isTLSSupported());
2720   // Lookup the entry, lazily creating it if necessary.
2721   std::string Suffix = getName({"cache", ""});
2722   return getOrCreateInternalVariable(
2723       CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix));
2724 }
2725 
2726 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
2727                                                 const VarDecl *VD,
2728                                                 Address VDAddr,
2729                                                 SourceLocation Loc) {
2730   if (CGM.getLangOpts().OpenMPUseTLS &&
2731       CGM.getContext().getTargetInfo().isTLSSupported())
2732     return VDAddr;
2733 
2734   llvm::Type *VarTy = VDAddr.getElementType();
2735   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2736                          CGF.Builder.CreatePointerCast(VDAddr.getPointer(),
2737                                                        CGM.Int8PtrTy),
2738                          CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
2739                          getOrCreateThreadPrivateCache(VD)};
2740   return Address(CGF.EmitRuntimeCall(
2741       createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
2742                  VDAddr.getAlignment());
2743 }
2744 
2745 void CGOpenMPRuntime::emitThreadPrivateVarInit(
2746     CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor,
2747     llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
2748   // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
2749   // library.
2750   llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc);
2751   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
2752                       OMPLoc);
2753   // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
2754   // to register constructor/destructor for variable.
2755   llvm::Value *Args[] = {
2756       OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy),
2757       Ctor, CopyCtor, Dtor};
2758   CGF.EmitRuntimeCall(
2759       createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args);
2760 }
2761 
2762 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
2763     const VarDecl *VD, Address VDAddr, SourceLocation Loc,
2764     bool PerformInit, CodeGenFunction *CGF) {
2765   if (CGM.getLangOpts().OpenMPUseTLS &&
2766       CGM.getContext().getTargetInfo().isTLSSupported())
2767     return nullptr;
2768 
2769   VD = VD->getDefinition(CGM.getContext());
2770   if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) {
2771     QualType ASTTy = VD->getType();
2772 
2773     llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
2774     const Expr *Init = VD->getAnyInitializer();
2775     if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2776       // Generate function that re-emits the declaration's initializer into the
2777       // threadprivate copy of the variable VD
2778       CodeGenFunction CtorCGF(CGM);
2779       FunctionArgList Args;
2780       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2781                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2782                             ImplicitParamDecl::Other);
2783       Args.push_back(&Dst);
2784 
2785       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2786           CGM.getContext().VoidPtrTy, Args);
2787       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2788       std::string Name = getName({"__kmpc_global_ctor_", ""});
2789       llvm::Function *Fn =
2790           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2791       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
2792                             Args, Loc, Loc);
2793       llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar(
2794           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2795           CGM.getContext().VoidPtrTy, Dst.getLocation());
2796       Address Arg = Address(ArgVal, VDAddr.getAlignment());
2797       Arg = CtorCGF.Builder.CreateElementBitCast(
2798           Arg, CtorCGF.ConvertTypeForMem(ASTTy));
2799       CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
2800                                /*IsInitializer=*/true);
2801       ArgVal = CtorCGF.EmitLoadOfScalar(
2802           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2803           CGM.getContext().VoidPtrTy, Dst.getLocation());
2804       CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
2805       CtorCGF.FinishFunction();
2806       Ctor = Fn;
2807     }
2808     if (VD->getType().isDestructedType() != QualType::DK_none) {
2809       // Generate function that emits destructor call for the threadprivate copy
2810       // of the variable VD
2811       CodeGenFunction DtorCGF(CGM);
2812       FunctionArgList Args;
2813       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2814                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2815                             ImplicitParamDecl::Other);
2816       Args.push_back(&Dst);
2817 
2818       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2819           CGM.getContext().VoidTy, Args);
2820       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2821       std::string Name = getName({"__kmpc_global_dtor_", ""});
2822       llvm::Function *Fn =
2823           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2824       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2825       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
2826                             Loc, Loc);
2827       // Create a scope with an artificial location for the body of this function.
2828       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
2829       llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar(
2830           DtorCGF.GetAddrOfLocalVar(&Dst),
2831           /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation());
2832       DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy,
2833                           DtorCGF.getDestroyer(ASTTy.isDestructedType()),
2834                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
2835       DtorCGF.FinishFunction();
2836       Dtor = Fn;
2837     }
2838     // Do not emit init function if it is not required.
2839     if (!Ctor && !Dtor)
2840       return nullptr;
2841 
2842     llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2843     auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs,
2844                                                /*isVarArg=*/false)
2845                            ->getPointerTo();
2846     // Copying constructor for the threadprivate variable.
2847     // Must be NULL - reserved by runtime, but currently it requires that this
2848     // parameter is always NULL. Otherwise it fires assertion.
2849     CopyCtor = llvm::Constant::getNullValue(CopyCtorTy);
2850     if (Ctor == nullptr) {
2851       auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
2852                                              /*isVarArg=*/false)
2853                          ->getPointerTo();
2854       Ctor = llvm::Constant::getNullValue(CtorTy);
2855     }
2856     if (Dtor == nullptr) {
2857       auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy,
2858                                              /*isVarArg=*/false)
2859                          ->getPointerTo();
2860       Dtor = llvm::Constant::getNullValue(DtorTy);
2861     }
2862     if (!CGF) {
2863       auto *InitFunctionTy =
2864           llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
2865       std::string Name = getName({"__omp_threadprivate_init_", ""});
2866       llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction(
2867           InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction());
2868       CodeGenFunction InitCGF(CGM);
2869       FunctionArgList ArgList;
2870       InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
2871                             CGM.getTypes().arrangeNullaryFunction(), ArgList,
2872                             Loc, Loc);
2873       emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2874       InitCGF.FinishFunction();
2875       return InitFunction;
2876     }
2877     emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2878   }
2879   return nullptr;
2880 }
2881 
2882 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD,
2883                                                      llvm::GlobalVariable *Addr,
2884                                                      bool PerformInit) {
2885   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
2886       !CGM.getLangOpts().OpenMPIsDevice)
2887     return false;
2888   Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2889       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2890   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
2891       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2892        HasRequiresUnifiedSharedMemory))
2893     return CGM.getLangOpts().OpenMPIsDevice;
2894   VD = VD->getDefinition(CGM.getContext());
2895   assert(VD && "Unknown VarDecl");
2896 
2897   if (!DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second)
2898     return CGM.getLangOpts().OpenMPIsDevice;
2899 
2900   QualType ASTTy = VD->getType();
2901   SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc();
2902 
2903   // Produce the unique prefix to identify the new target regions. We use
2904   // the source location of the variable declaration which we know to not
2905   // conflict with any target region.
2906   unsigned DeviceID;
2907   unsigned FileID;
2908   unsigned Line;
2909   getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line);
2910   SmallString<128> Buffer, Out;
2911   {
2912     llvm::raw_svector_ostream OS(Buffer);
2913     OS << "__omp_offloading_" << llvm::format("_%x", DeviceID)
2914        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
2915   }
2916 
2917   const Expr *Init = VD->getAnyInitializer();
2918   if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2919     llvm::Constant *Ctor;
2920     llvm::Constant *ID;
2921     if (CGM.getLangOpts().OpenMPIsDevice) {
2922       // Generate function that re-emits the declaration's initializer into
2923       // the threadprivate copy of the variable VD
2924       CodeGenFunction CtorCGF(CGM);
2925 
2926       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2927       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2928       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2929           FTy, Twine(Buffer, "_ctor"), FI, Loc);
2930       auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF);
2931       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2932                             FunctionArgList(), Loc, Loc);
2933       auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF);
2934       CtorCGF.EmitAnyExprToMem(Init,
2935                                Address(Addr, CGM.getContext().getDeclAlign(VD)),
2936                                Init->getType().getQualifiers(),
2937                                /*IsInitializer=*/true);
2938       CtorCGF.FinishFunction();
2939       Ctor = Fn;
2940       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
2941       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor));
2942     } else {
2943       Ctor = new llvm::GlobalVariable(
2944           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
2945           llvm::GlobalValue::PrivateLinkage,
2946           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor"));
2947       ID = Ctor;
2948     }
2949 
2950     // Register the information for the entry associated with the constructor.
2951     Out.clear();
2952     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2953         DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor,
2954         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor);
2955   }
2956   if (VD->getType().isDestructedType() != QualType::DK_none) {
2957     llvm::Constant *Dtor;
2958     llvm::Constant *ID;
2959     if (CGM.getLangOpts().OpenMPIsDevice) {
2960       // Generate function that emits destructor call for the threadprivate
2961       // copy of the variable VD
2962       CodeGenFunction DtorCGF(CGM);
2963 
2964       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2965       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2966       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2967           FTy, Twine(Buffer, "_dtor"), FI, Loc);
2968       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2969       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2970                             FunctionArgList(), Loc, Loc);
2971       // Create a scope with an artificial location for the body of this
2972       // function.
2973       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
2974       DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)),
2975                           ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()),
2976                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
2977       DtorCGF.FinishFunction();
2978       Dtor = Fn;
2979       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
2980       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor));
2981     } else {
2982       Dtor = new llvm::GlobalVariable(
2983           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
2984           llvm::GlobalValue::PrivateLinkage,
2985           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor"));
2986       ID = Dtor;
2987     }
2988     // Register the information for the entry associated with the destructor.
2989     Out.clear();
2990     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2991         DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor,
2992         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor);
2993   }
2994   return CGM.getLangOpts().OpenMPIsDevice;
2995 }
2996 
2997 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF,
2998                                                           QualType VarType,
2999                                                           StringRef Name) {
3000   std::string Suffix = getName({"artificial", ""});
3001   llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType);
3002   llvm::Value *GAddr =
3003       getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix));
3004   if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS &&
3005       CGM.getTarget().isTLSSupported()) {
3006     cast<llvm::GlobalVariable>(GAddr)->setThreadLocal(/*Val=*/true);
3007     return Address(GAddr, CGM.getContext().getTypeAlignInChars(VarType));
3008   }
3009   std::string CacheSuffix = getName({"cache", ""});
3010   llvm::Value *Args[] = {
3011       emitUpdateLocation(CGF, SourceLocation()),
3012       getThreadID(CGF, SourceLocation()),
3013       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy),
3014       CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy,
3015                                 /*isSigned=*/false),
3016       getOrCreateInternalVariable(
3017           CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))};
3018   return Address(
3019       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3020           CGF.EmitRuntimeCall(
3021               createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
3022           VarLVType->getPointerTo(/*AddrSpace=*/0)),
3023       CGM.getContext().getTypeAlignInChars(VarType));
3024 }
3025 
3026 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond,
3027                                    const RegionCodeGenTy &ThenGen,
3028                                    const RegionCodeGenTy &ElseGen) {
3029   CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
3030 
3031   // If the condition constant folds and can be elided, try to avoid emitting
3032   // the condition and the dead arm of the if/else.
3033   bool CondConstant;
3034   if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
3035     if (CondConstant)
3036       ThenGen(CGF);
3037     else
3038       ElseGen(CGF);
3039     return;
3040   }
3041 
3042   // Otherwise, the condition did not fold, or we couldn't elide it.  Just
3043   // emit the conditional branch.
3044   llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then");
3045   llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else");
3046   llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end");
3047   CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0);
3048 
3049   // Emit the 'then' code.
3050   CGF.EmitBlock(ThenBlock);
3051   ThenGen(CGF);
3052   CGF.EmitBranch(ContBlock);
3053   // Emit the 'else' code if present.
3054   // There is no need to emit line number for unconditional branch.
3055   (void)ApplyDebugLocation::CreateEmpty(CGF);
3056   CGF.EmitBlock(ElseBlock);
3057   ElseGen(CGF);
3058   // There is no need to emit line number for unconditional branch.
3059   (void)ApplyDebugLocation::CreateEmpty(CGF);
3060   CGF.EmitBranch(ContBlock);
3061   // Emit the continuation block for code after the if.
3062   CGF.EmitBlock(ContBlock, /*IsFinished=*/true);
3063 }
3064 
3065 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc,
3066                                        llvm::Function *OutlinedFn,
3067                                        ArrayRef<llvm::Value *> CapturedVars,
3068                                        const Expr *IfCond) {
3069   if (!CGF.HaveInsertPoint())
3070     return;
3071   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
3072   auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF,
3073                                                      PrePostActionTy &) {
3074     // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn);
3075     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3076     llvm::Value *Args[] = {
3077         RTLoc,
3078         CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
3079         CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())};
3080     llvm::SmallVector<llvm::Value *, 16> RealArgs;
3081     RealArgs.append(std::begin(Args), std::end(Args));
3082     RealArgs.append(CapturedVars.begin(), CapturedVars.end());
3083 
3084     llvm::FunctionCallee RTLFn =
3085         RT.createRuntimeFunction(OMPRTL__kmpc_fork_call);
3086     CGF.EmitRuntimeCall(RTLFn, RealArgs);
3087   };
3088   auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF,
3089                                                           PrePostActionTy &) {
3090     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3091     llvm::Value *ThreadID = RT.getThreadID(CGF, Loc);
3092     // Build calls:
3093     // __kmpc_serialized_parallel(&Loc, GTid);
3094     llvm::Value *Args[] = {RTLoc, ThreadID};
3095     CGF.EmitRuntimeCall(
3096         RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args);
3097 
3098     // OutlinedFn(&GTid, &zero_bound, CapturedStruct);
3099     Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc);
3100     Address ZeroAddrBound =
3101         CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty,
3102                                          /*Name=*/".bound.zero.addr");
3103     CGF.InitTempAlloca(ZeroAddrBound, CGF.Builder.getInt32(/*C*/ 0));
3104     llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs;
3105     // ThreadId for serialized parallels is 0.
3106     OutlinedFnArgs.push_back(ThreadIDAddr.getPointer());
3107     OutlinedFnArgs.push_back(ZeroAddrBound.getPointer());
3108     OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end());
3109     RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs);
3110 
3111     // __kmpc_end_serialized_parallel(&Loc, GTid);
3112     llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID};
3113     CGF.EmitRuntimeCall(
3114         RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel),
3115         EndArgs);
3116   };
3117   if (IfCond) {
3118     emitIfClause(CGF, IfCond, ThenGen, ElseGen);
3119   } else {
3120     RegionCodeGenTy ThenRCG(ThenGen);
3121     ThenRCG(CGF);
3122   }
3123 }
3124 
3125 // If we're inside an (outlined) parallel region, use the region info's
3126 // thread-ID variable (it is passed in a first argument of the outlined function
3127 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
3128 // regular serial code region, get thread ID by calling kmp_int32
3129 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
3130 // return the address of that temp.
3131 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
3132                                              SourceLocation Loc) {
3133   if (auto *OMPRegionInfo =
3134           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3135     if (OMPRegionInfo->getThreadIDVariable())
3136       return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF);
3137 
3138   llvm::Value *ThreadID = getThreadID(CGF, Loc);
3139   QualType Int32Ty =
3140       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
3141   Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp.");
3142   CGF.EmitStoreOfScalar(ThreadID,
3143                         CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty));
3144 
3145   return ThreadIDTemp;
3146 }
3147 
3148 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable(
3149     llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) {
3150   SmallString<256> Buffer;
3151   llvm::raw_svector_ostream Out(Buffer);
3152   Out << Name;
3153   StringRef RuntimeName = Out.str();
3154   auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first;
3155   if (Elem.second) {
3156     assert(Elem.second->getType()->getPointerElementType() == Ty &&
3157            "OMP internal variable has different type than requested");
3158     return &*Elem.second;
3159   }
3160 
3161   return Elem.second = new llvm::GlobalVariable(
3162              CGM.getModule(), Ty, /*IsConstant*/ false,
3163              llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty),
3164              Elem.first(), /*InsertBefore=*/nullptr,
3165              llvm::GlobalValue::NotThreadLocal, AddressSpace);
3166 }
3167 
3168 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
3169   std::string Prefix = Twine("gomp_critical_user_", CriticalName).str();
3170   std::string Name = getName({Prefix, "var"});
3171   return getOrCreateInternalVariable(KmpCriticalNameTy, Name);
3172 }
3173 
3174 namespace {
3175 /// Common pre(post)-action for different OpenMP constructs.
3176 class CommonActionTy final : public PrePostActionTy {
3177   llvm::FunctionCallee EnterCallee;
3178   ArrayRef<llvm::Value *> EnterArgs;
3179   llvm::FunctionCallee ExitCallee;
3180   ArrayRef<llvm::Value *> ExitArgs;
3181   bool Conditional;
3182   llvm::BasicBlock *ContBlock = nullptr;
3183 
3184 public:
3185   CommonActionTy(llvm::FunctionCallee EnterCallee,
3186                  ArrayRef<llvm::Value *> EnterArgs,
3187                  llvm::FunctionCallee ExitCallee,
3188                  ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false)
3189       : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee),
3190         ExitArgs(ExitArgs), Conditional(Conditional) {}
3191   void Enter(CodeGenFunction &CGF) override {
3192     llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs);
3193     if (Conditional) {
3194       llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes);
3195       auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
3196       ContBlock = CGF.createBasicBlock("omp_if.end");
3197       // Generate the branch (If-stmt)
3198       CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
3199       CGF.EmitBlock(ThenBlock);
3200     }
3201   }
3202   void Done(CodeGenFunction &CGF) {
3203     // Emit the rest of blocks/branches
3204     CGF.EmitBranch(ContBlock);
3205     CGF.EmitBlock(ContBlock, true);
3206   }
3207   void Exit(CodeGenFunction &CGF) override {
3208     CGF.EmitRuntimeCall(ExitCallee, ExitArgs);
3209   }
3210 };
3211 } // anonymous namespace
3212 
3213 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF,
3214                                          StringRef CriticalName,
3215                                          const RegionCodeGenTy &CriticalOpGen,
3216                                          SourceLocation Loc, const Expr *Hint) {
3217   // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]);
3218   // CriticalOpGen();
3219   // __kmpc_end_critical(ident_t *, gtid, Lock);
3220   // Prepare arguments and build a call to __kmpc_critical
3221   if (!CGF.HaveInsertPoint())
3222     return;
3223   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3224                          getCriticalRegionLock(CriticalName)};
3225   llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args),
3226                                                 std::end(Args));
3227   if (Hint) {
3228     EnterArgs.push_back(CGF.Builder.CreateIntCast(
3229         CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false));
3230   }
3231   CommonActionTy Action(
3232       createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint
3233                                  : OMPRTL__kmpc_critical),
3234       EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args);
3235   CriticalOpGen.setAction(Action);
3236   emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen);
3237 }
3238 
3239 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
3240                                        const RegionCodeGenTy &MasterOpGen,
3241                                        SourceLocation Loc) {
3242   if (!CGF.HaveInsertPoint())
3243     return;
3244   // if(__kmpc_master(ident_t *, gtid)) {
3245   //   MasterOpGen();
3246   //   __kmpc_end_master(ident_t *, gtid);
3247   // }
3248   // Prepare arguments and build a call to __kmpc_master
3249   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3250   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args,
3251                         createRuntimeFunction(OMPRTL__kmpc_end_master), Args,
3252                         /*Conditional=*/true);
3253   MasterOpGen.setAction(Action);
3254   emitInlinedDirective(CGF, OMPD_master, MasterOpGen);
3255   Action.Done(CGF);
3256 }
3257 
3258 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
3259                                         SourceLocation Loc) {
3260   if (!CGF.HaveInsertPoint())
3261     return;
3262   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3263   if (OMPBuilder) {
3264     OMPBuilder->CreateTaskyield(CGF.Builder);
3265   } else {
3266     // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
3267     llvm::Value *Args[] = {
3268         emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3269         llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
3270     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield),
3271                         Args);
3272   }
3273 
3274   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3275     Region->emitUntiedSwitch(CGF);
3276 }
3277 
3278 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF,
3279                                           const RegionCodeGenTy &TaskgroupOpGen,
3280                                           SourceLocation Loc) {
3281   if (!CGF.HaveInsertPoint())
3282     return;
3283   // __kmpc_taskgroup(ident_t *, gtid);
3284   // TaskgroupOpGen();
3285   // __kmpc_end_taskgroup(ident_t *, gtid);
3286   // Prepare arguments and build a call to __kmpc_taskgroup
3287   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3288   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args,
3289                         createRuntimeFunction(OMPRTL__kmpc_end_taskgroup),
3290                         Args);
3291   TaskgroupOpGen.setAction(Action);
3292   emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen);
3293 }
3294 
3295 /// Given an array of pointers to variables, project the address of a
3296 /// given variable.
3297 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array,
3298                                       unsigned Index, const VarDecl *Var) {
3299   // Pull out the pointer to the variable.
3300   Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index);
3301   llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr);
3302 
3303   Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var));
3304   Addr = CGF.Builder.CreateElementBitCast(
3305       Addr, CGF.ConvertTypeForMem(Var->getType()));
3306   return Addr;
3307 }
3308 
3309 static llvm::Value *emitCopyprivateCopyFunction(
3310     CodeGenModule &CGM, llvm::Type *ArgsType,
3311     ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
3312     ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps,
3313     SourceLocation Loc) {
3314   ASTContext &C = CGM.getContext();
3315   // void copy_func(void *LHSArg, void *RHSArg);
3316   FunctionArgList Args;
3317   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3318                            ImplicitParamDecl::Other);
3319   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3320                            ImplicitParamDecl::Other);
3321   Args.push_back(&LHSArg);
3322   Args.push_back(&RHSArg);
3323   const auto &CGFI =
3324       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
3325   std::string Name =
3326       CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"});
3327   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
3328                                     llvm::GlobalValue::InternalLinkage, Name,
3329                                     &CGM.getModule());
3330   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
3331   Fn->setDoesNotRecurse();
3332   CodeGenFunction CGF(CGM);
3333   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
3334   // Dest = (void*[n])(LHSArg);
3335   // Src = (void*[n])(RHSArg);
3336   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3337       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
3338       ArgsType), CGF.getPointerAlign());
3339   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3340       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
3341       ArgsType), CGF.getPointerAlign());
3342   // *(Type0*)Dst[0] = *(Type0*)Src[0];
3343   // *(Type1*)Dst[1] = *(Type1*)Src[1];
3344   // ...
3345   // *(Typen*)Dst[n] = *(Typen*)Src[n];
3346   for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
3347     const auto *DestVar =
3348         cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl());
3349     Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar);
3350 
3351     const auto *SrcVar =
3352         cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl());
3353     Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar);
3354 
3355     const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl();
3356     QualType Type = VD->getType();
3357     CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]);
3358   }
3359   CGF.FinishFunction();
3360   return Fn;
3361 }
3362 
3363 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
3364                                        const RegionCodeGenTy &SingleOpGen,
3365                                        SourceLocation Loc,
3366                                        ArrayRef<const Expr *> CopyprivateVars,
3367                                        ArrayRef<const Expr *> SrcExprs,
3368                                        ArrayRef<const Expr *> DstExprs,
3369                                        ArrayRef<const Expr *> AssignmentOps) {
3370   if (!CGF.HaveInsertPoint())
3371     return;
3372   assert(CopyprivateVars.size() == SrcExprs.size() &&
3373          CopyprivateVars.size() == DstExprs.size() &&
3374          CopyprivateVars.size() == AssignmentOps.size());
3375   ASTContext &C = CGM.getContext();
3376   // int32 did_it = 0;
3377   // if(__kmpc_single(ident_t *, gtid)) {
3378   //   SingleOpGen();
3379   //   __kmpc_end_single(ident_t *, gtid);
3380   //   did_it = 1;
3381   // }
3382   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3383   // <copy_func>, did_it);
3384 
3385   Address DidIt = Address::invalid();
3386   if (!CopyprivateVars.empty()) {
3387     // int32 did_it = 0;
3388     QualType KmpInt32Ty =
3389         C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
3390     DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it");
3391     CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt);
3392   }
3393   // Prepare arguments and build a call to __kmpc_single
3394   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3395   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args,
3396                         createRuntimeFunction(OMPRTL__kmpc_end_single), Args,
3397                         /*Conditional=*/true);
3398   SingleOpGen.setAction(Action);
3399   emitInlinedDirective(CGF, OMPD_single, SingleOpGen);
3400   if (DidIt.isValid()) {
3401     // did_it = 1;
3402     CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt);
3403   }
3404   Action.Done(CGF);
3405   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3406   // <copy_func>, did_it);
3407   if (DidIt.isValid()) {
3408     llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
3409     QualType CopyprivateArrayTy = C.getConstantArrayType(
3410         C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
3411         /*IndexTypeQuals=*/0);
3412     // Create a list of all private variables for copyprivate.
3413     Address CopyprivateList =
3414         CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list");
3415     for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
3416       Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I);
3417       CGF.Builder.CreateStore(
3418           CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3419               CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF),
3420               CGF.VoidPtrTy),
3421           Elem);
3422     }
3423     // Build function that copies private values from single region to all other
3424     // threads in the corresponding parallel region.
3425     llvm::Value *CpyFn = emitCopyprivateCopyFunction(
3426         CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(),
3427         CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc);
3428     llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy);
3429     Address CL =
3430       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList,
3431                                                       CGF.VoidPtrTy);
3432     llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt);
3433     llvm::Value *Args[] = {
3434         emitUpdateLocation(CGF, Loc), // ident_t *<loc>
3435         getThreadID(CGF, Loc),        // i32 <gtid>
3436         BufSize,                      // size_t <buf_size>
3437         CL.getPointer(),              // void *<copyprivate list>
3438         CpyFn,                        // void (*) (void *, void *) <copy_func>
3439         DidItVal                      // i32 did_it
3440     };
3441     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args);
3442   }
3443 }
3444 
3445 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF,
3446                                         const RegionCodeGenTy &OrderedOpGen,
3447                                         SourceLocation Loc, bool IsThreads) {
3448   if (!CGF.HaveInsertPoint())
3449     return;
3450   // __kmpc_ordered(ident_t *, gtid);
3451   // OrderedOpGen();
3452   // __kmpc_end_ordered(ident_t *, gtid);
3453   // Prepare arguments and build a call to __kmpc_ordered
3454   if (IsThreads) {
3455     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3456     CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args,
3457                           createRuntimeFunction(OMPRTL__kmpc_end_ordered),
3458                           Args);
3459     OrderedOpGen.setAction(Action);
3460     emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3461     return;
3462   }
3463   emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3464 }
3465 
3466 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) {
3467   unsigned Flags;
3468   if (Kind == OMPD_for)
3469     Flags = OMP_IDENT_BARRIER_IMPL_FOR;
3470   else if (Kind == OMPD_sections)
3471     Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS;
3472   else if (Kind == OMPD_single)
3473     Flags = OMP_IDENT_BARRIER_IMPL_SINGLE;
3474   else if (Kind == OMPD_barrier)
3475     Flags = OMP_IDENT_BARRIER_EXPL;
3476   else
3477     Flags = OMP_IDENT_BARRIER_IMPL;
3478   return Flags;
3479 }
3480 
3481 void CGOpenMPRuntime::getDefaultScheduleAndChunk(
3482     CodeGenFunction &CGF, const OMPLoopDirective &S,
3483     OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const {
3484   // Check if the loop directive is actually a doacross loop directive. In this
3485   // case choose static, 1 schedule.
3486   if (llvm::any_of(
3487           S.getClausesOfKind<OMPOrderedClause>(),
3488           [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) {
3489     ScheduleKind = OMPC_SCHEDULE_static;
3490     // Chunk size is 1 in this case.
3491     llvm::APInt ChunkSize(32, 1);
3492     ChunkExpr = IntegerLiteral::Create(
3493         CGF.getContext(), ChunkSize,
3494         CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0),
3495         SourceLocation());
3496   }
3497 }
3498 
3499 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
3500                                       OpenMPDirectiveKind Kind, bool EmitChecks,
3501                                       bool ForceSimpleCall) {
3502   // Check if we should use the OMPBuilder
3503   auto *OMPRegionInfo =
3504       dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo);
3505   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3506   if (OMPBuilder) {
3507     CGF.Builder.restoreIP(OMPBuilder->CreateBarrier(
3508         CGF.Builder, Kind, ForceSimpleCall, EmitChecks));
3509     return;
3510   }
3511 
3512   if (!CGF.HaveInsertPoint())
3513     return;
3514   // Build call __kmpc_cancel_barrier(loc, thread_id);
3515   // Build call __kmpc_barrier(loc, thread_id);
3516   unsigned Flags = getDefaultFlagsForBarriers(Kind);
3517   // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc,
3518   // thread_id);
3519   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
3520                          getThreadID(CGF, Loc)};
3521   if (OMPRegionInfo) {
3522     if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) {
3523       llvm::Value *Result = CGF.EmitRuntimeCall(
3524           createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args);
3525       if (EmitChecks) {
3526         // if (__kmpc_cancel_barrier()) {
3527         //   exit from construct;
3528         // }
3529         llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
3530         llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
3531         llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
3532         CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
3533         CGF.EmitBlock(ExitBB);
3534         //   exit from construct;
3535         CodeGenFunction::JumpDest CancelDestination =
3536             CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
3537         CGF.EmitBranchThroughCleanup(CancelDestination);
3538         CGF.EmitBlock(ContBB, /*IsFinished=*/true);
3539       }
3540       return;
3541     }
3542   }
3543   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args);
3544 }
3545 
3546 /// Map the OpenMP loop schedule to the runtime enumeration.
3547 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
3548                                           bool Chunked, bool Ordered) {
3549   switch (ScheduleKind) {
3550   case OMPC_SCHEDULE_static:
3551     return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked)
3552                    : (Ordered ? OMP_ord_static : OMP_sch_static);
3553   case OMPC_SCHEDULE_dynamic:
3554     return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked;
3555   case OMPC_SCHEDULE_guided:
3556     return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked;
3557   case OMPC_SCHEDULE_runtime:
3558     return Ordered ? OMP_ord_runtime : OMP_sch_runtime;
3559   case OMPC_SCHEDULE_auto:
3560     return Ordered ? OMP_ord_auto : OMP_sch_auto;
3561   case OMPC_SCHEDULE_unknown:
3562     assert(!Chunked && "chunk was specified but schedule kind not known");
3563     return Ordered ? OMP_ord_static : OMP_sch_static;
3564   }
3565   llvm_unreachable("Unexpected runtime schedule");
3566 }
3567 
3568 /// Map the OpenMP distribute schedule to the runtime enumeration.
3569 static OpenMPSchedType
3570 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) {
3571   // only static is allowed for dist_schedule
3572   return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static;
3573 }
3574 
3575 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
3576                                          bool Chunked) const {
3577   OpenMPSchedType Schedule =
3578       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3579   return Schedule == OMP_sch_static;
3580 }
3581 
3582 bool CGOpenMPRuntime::isStaticNonchunked(
3583     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3584   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3585   return Schedule == OMP_dist_sch_static;
3586 }
3587 
3588 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind,
3589                                       bool Chunked) const {
3590   OpenMPSchedType Schedule =
3591       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3592   return Schedule == OMP_sch_static_chunked;
3593 }
3594 
3595 bool CGOpenMPRuntime::isStaticChunked(
3596     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3597   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3598   return Schedule == OMP_dist_sch_static_chunked;
3599 }
3600 
3601 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
3602   OpenMPSchedType Schedule =
3603       getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false);
3604   assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
3605   return Schedule != OMP_sch_static;
3606 }
3607 
3608 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule,
3609                                   OpenMPScheduleClauseModifier M1,
3610                                   OpenMPScheduleClauseModifier M2) {
3611   int Modifier = 0;
3612   switch (M1) {
3613   case OMPC_SCHEDULE_MODIFIER_monotonic:
3614     Modifier = OMP_sch_modifier_monotonic;
3615     break;
3616   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3617     Modifier = OMP_sch_modifier_nonmonotonic;
3618     break;
3619   case OMPC_SCHEDULE_MODIFIER_simd:
3620     if (Schedule == OMP_sch_static_chunked)
3621       Schedule = OMP_sch_static_balanced_chunked;
3622     break;
3623   case OMPC_SCHEDULE_MODIFIER_last:
3624   case OMPC_SCHEDULE_MODIFIER_unknown:
3625     break;
3626   }
3627   switch (M2) {
3628   case OMPC_SCHEDULE_MODIFIER_monotonic:
3629     Modifier = OMP_sch_modifier_monotonic;
3630     break;
3631   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3632     Modifier = OMP_sch_modifier_nonmonotonic;
3633     break;
3634   case OMPC_SCHEDULE_MODIFIER_simd:
3635     if (Schedule == OMP_sch_static_chunked)
3636       Schedule = OMP_sch_static_balanced_chunked;
3637     break;
3638   case OMPC_SCHEDULE_MODIFIER_last:
3639   case OMPC_SCHEDULE_MODIFIER_unknown:
3640     break;
3641   }
3642   // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription.
3643   // If the static schedule kind is specified or if the ordered clause is
3644   // specified, and if the nonmonotonic modifier is not specified, the effect is
3645   // as if the monotonic modifier is specified. Otherwise, unless the monotonic
3646   // modifier is specified, the effect is as if the nonmonotonic modifier is
3647   // specified.
3648   if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) {
3649     if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static ||
3650           Schedule == OMP_sch_static_balanced_chunked ||
3651           Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static ||
3652           Schedule == OMP_dist_sch_static_chunked ||
3653           Schedule == OMP_dist_sch_static))
3654       Modifier = OMP_sch_modifier_nonmonotonic;
3655   }
3656   return Schedule | Modifier;
3657 }
3658 
3659 void CGOpenMPRuntime::emitForDispatchInit(
3660     CodeGenFunction &CGF, SourceLocation Loc,
3661     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
3662     bool Ordered, const DispatchRTInput &DispatchValues) {
3663   if (!CGF.HaveInsertPoint())
3664     return;
3665   OpenMPSchedType Schedule = getRuntimeSchedule(
3666       ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered);
3667   assert(Ordered ||
3668          (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked &&
3669           Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked &&
3670           Schedule != OMP_sch_static_balanced_chunked));
3671   // Call __kmpc_dispatch_init(
3672   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
3673   //          kmp_int[32|64] lower, kmp_int[32|64] upper,
3674   //          kmp_int[32|64] stride, kmp_int[32|64] chunk);
3675 
3676   // If the Chunk was not specified in the clause - use default value 1.
3677   llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk
3678                                             : CGF.Builder.getIntN(IVSize, 1);
3679   llvm::Value *Args[] = {
3680       emitUpdateLocation(CGF, Loc),
3681       getThreadID(CGF, Loc),
3682       CGF.Builder.getInt32(addMonoNonMonoModifier(
3683           CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type
3684       DispatchValues.LB,                                     // Lower
3685       DispatchValues.UB,                                     // Upper
3686       CGF.Builder.getIntN(IVSize, 1),                        // Stride
3687       Chunk                                                  // Chunk
3688   };
3689   CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args);
3690 }
3691 
3692 static void emitForStaticInitCall(
3693     CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId,
3694     llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule,
3695     OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2,
3696     const CGOpenMPRuntime::StaticRTInput &Values) {
3697   if (!CGF.HaveInsertPoint())
3698     return;
3699 
3700   assert(!Values.Ordered);
3701   assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked ||
3702          Schedule == OMP_sch_static_balanced_chunked ||
3703          Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked ||
3704          Schedule == OMP_dist_sch_static ||
3705          Schedule == OMP_dist_sch_static_chunked);
3706 
3707   // Call __kmpc_for_static_init(
3708   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
3709   //          kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
3710   //          kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
3711   //          kmp_int[32|64] incr, kmp_int[32|64] chunk);
3712   llvm::Value *Chunk = Values.Chunk;
3713   if (Chunk == nullptr) {
3714     assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static ||
3715             Schedule == OMP_dist_sch_static) &&
3716            "expected static non-chunked schedule");
3717     // If the Chunk was not specified in the clause - use default value 1.
3718     Chunk = CGF.Builder.getIntN(Values.IVSize, 1);
3719   } else {
3720     assert((Schedule == OMP_sch_static_chunked ||
3721             Schedule == OMP_sch_static_balanced_chunked ||
3722             Schedule == OMP_ord_static_chunked ||
3723             Schedule == OMP_dist_sch_static_chunked) &&
3724            "expected static chunked schedule");
3725   }
3726   llvm::Value *Args[] = {
3727       UpdateLocation,
3728       ThreadId,
3729       CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1,
3730                                                   M2)), // Schedule type
3731       Values.IL.getPointer(),                           // &isLastIter
3732       Values.LB.getPointer(),                           // &LB
3733       Values.UB.getPointer(),                           // &UB
3734       Values.ST.getPointer(),                           // &Stride
3735       CGF.Builder.getIntN(Values.IVSize, 1),            // Incr
3736       Chunk                                             // Chunk
3737   };
3738   CGF.EmitRuntimeCall(ForStaticInitFunction, Args);
3739 }
3740 
3741 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF,
3742                                         SourceLocation Loc,
3743                                         OpenMPDirectiveKind DKind,
3744                                         const OpenMPScheduleTy &ScheduleKind,
3745                                         const StaticRTInput &Values) {
3746   OpenMPSchedType ScheduleNum = getRuntimeSchedule(
3747       ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered);
3748   assert(isOpenMPWorksharingDirective(DKind) &&
3749          "Expected loop-based or sections-based directive.");
3750   llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc,
3751                                              isOpenMPLoopDirective(DKind)
3752                                                  ? OMP_IDENT_WORK_LOOP
3753                                                  : OMP_IDENT_WORK_SECTIONS);
3754   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3755   llvm::FunctionCallee StaticInitFunction =
3756       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3757   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
3758   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3759                         ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values);
3760 }
3761 
3762 void CGOpenMPRuntime::emitDistributeStaticInit(
3763     CodeGenFunction &CGF, SourceLocation Loc,
3764     OpenMPDistScheduleClauseKind SchedKind,
3765     const CGOpenMPRuntime::StaticRTInput &Values) {
3766   OpenMPSchedType ScheduleNum =
3767       getRuntimeSchedule(SchedKind, Values.Chunk != nullptr);
3768   llvm::Value *UpdatedLocation =
3769       emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE);
3770   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3771   llvm::FunctionCallee StaticInitFunction =
3772       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3773   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3774                         ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown,
3775                         OMPC_SCHEDULE_MODIFIER_unknown, Values);
3776 }
3777 
3778 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF,
3779                                           SourceLocation Loc,
3780                                           OpenMPDirectiveKind DKind) {
3781   if (!CGF.HaveInsertPoint())
3782     return;
3783   // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
3784   llvm::Value *Args[] = {
3785       emitUpdateLocation(CGF, Loc,
3786                          isOpenMPDistributeDirective(DKind)
3787                              ? OMP_IDENT_WORK_DISTRIBUTE
3788                              : isOpenMPLoopDirective(DKind)
3789                                    ? OMP_IDENT_WORK_LOOP
3790                                    : OMP_IDENT_WORK_SECTIONS),
3791       getThreadID(CGF, Loc)};
3792   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
3793   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini),
3794                       Args);
3795 }
3796 
3797 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
3798                                                  SourceLocation Loc,
3799                                                  unsigned IVSize,
3800                                                  bool IVSigned) {
3801   if (!CGF.HaveInsertPoint())
3802     return;
3803   // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
3804   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3805   CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args);
3806 }
3807 
3808 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
3809                                           SourceLocation Loc, unsigned IVSize,
3810                                           bool IVSigned, Address IL,
3811                                           Address LB, Address UB,
3812                                           Address ST) {
3813   // Call __kmpc_dispatch_next(
3814   //          ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
3815   //          kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
3816   //          kmp_int[32|64] *p_stride);
3817   llvm::Value *Args[] = {
3818       emitUpdateLocation(CGF, Loc),
3819       getThreadID(CGF, Loc),
3820       IL.getPointer(), // &isLastIter
3821       LB.getPointer(), // &Lower
3822       UB.getPointer(), // &Upper
3823       ST.getPointer()  // &Stride
3824   };
3825   llvm::Value *Call =
3826       CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args);
3827   return CGF.EmitScalarConversion(
3828       Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1),
3829       CGF.getContext().BoolTy, Loc);
3830 }
3831 
3832 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
3833                                            llvm::Value *NumThreads,
3834                                            SourceLocation Loc) {
3835   if (!CGF.HaveInsertPoint())
3836     return;
3837   // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
3838   llvm::Value *Args[] = {
3839       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3840       CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)};
3841   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads),
3842                       Args);
3843 }
3844 
3845 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF,
3846                                          ProcBindKind ProcBind,
3847                                          SourceLocation Loc) {
3848   if (!CGF.HaveInsertPoint())
3849     return;
3850   assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value.");
3851   // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind)
3852   llvm::Value *Args[] = {
3853       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3854       llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)};
3855   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args);
3856 }
3857 
3858 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
3859                                 SourceLocation Loc, llvm::AtomicOrdering AO) {
3860   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
3861   if (OMPBuilder) {
3862     OMPBuilder->CreateFlush(CGF.Builder);
3863   } else {
3864     if (!CGF.HaveInsertPoint())
3865       return;
3866     // Build call void __kmpc_flush(ident_t *loc)
3867     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush),
3868                         emitUpdateLocation(CGF, Loc));
3869   }
3870 }
3871 
3872 namespace {
3873 /// Indexes of fields for type kmp_task_t.
3874 enum KmpTaskTFields {
3875   /// List of shared variables.
3876   KmpTaskTShareds,
3877   /// Task routine.
3878   KmpTaskTRoutine,
3879   /// Partition id for the untied tasks.
3880   KmpTaskTPartId,
3881   /// Function with call of destructors for private variables.
3882   Data1,
3883   /// Task priority.
3884   Data2,
3885   /// (Taskloops only) Lower bound.
3886   KmpTaskTLowerBound,
3887   /// (Taskloops only) Upper bound.
3888   KmpTaskTUpperBound,
3889   /// (Taskloops only) Stride.
3890   KmpTaskTStride,
3891   /// (Taskloops only) Is last iteration flag.
3892   KmpTaskTLastIter,
3893   /// (Taskloops only) Reduction data.
3894   KmpTaskTReductions,
3895 };
3896 } // anonymous namespace
3897 
3898 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const {
3899   return OffloadEntriesTargetRegion.empty() &&
3900          OffloadEntriesDeviceGlobalVar.empty();
3901 }
3902 
3903 /// Initialize target region entry.
3904 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3905     initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3906                                     StringRef ParentName, unsigned LineNum,
3907                                     unsigned Order) {
3908   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3909                                              "only required for the device "
3910                                              "code generation.");
3911   OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] =
3912       OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr,
3913                                    OMPTargetRegionEntryTargetRegion);
3914   ++OffloadingEntriesNum;
3915 }
3916 
3917 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3918     registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3919                                   StringRef ParentName, unsigned LineNum,
3920                                   llvm::Constant *Addr, llvm::Constant *ID,
3921                                   OMPTargetRegionEntryKind Flags) {
3922   // If we are emitting code for a target, the entry is already initialized,
3923   // only has to be registered.
3924   if (CGM.getLangOpts().OpenMPIsDevice) {
3925     if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) {
3926       unsigned DiagID = CGM.getDiags().getCustomDiagID(
3927           DiagnosticsEngine::Error,
3928           "Unable to find target region on line '%0' in the device code.");
3929       CGM.getDiags().Report(DiagID) << LineNum;
3930       return;
3931     }
3932     auto &Entry =
3933         OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum];
3934     assert(Entry.isValid() && "Entry not initialized!");
3935     Entry.setAddress(Addr);
3936     Entry.setID(ID);
3937     Entry.setFlags(Flags);
3938   } else {
3939     OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags);
3940     OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry;
3941     ++OffloadingEntriesNum;
3942   }
3943 }
3944 
3945 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo(
3946     unsigned DeviceID, unsigned FileID, StringRef ParentName,
3947     unsigned LineNum) const {
3948   auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID);
3949   if (PerDevice == OffloadEntriesTargetRegion.end())
3950     return false;
3951   auto PerFile = PerDevice->second.find(FileID);
3952   if (PerFile == PerDevice->second.end())
3953     return false;
3954   auto PerParentName = PerFile->second.find(ParentName);
3955   if (PerParentName == PerFile->second.end())
3956     return false;
3957   auto PerLine = PerParentName->second.find(LineNum);
3958   if (PerLine == PerParentName->second.end())
3959     return false;
3960   // Fail if this entry is already registered.
3961   if (PerLine->second.getAddress() || PerLine->second.getID())
3962     return false;
3963   return true;
3964 }
3965 
3966 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo(
3967     const OffloadTargetRegionEntryInfoActTy &Action) {
3968   // Scan all target region entries and perform the provided action.
3969   for (const auto &D : OffloadEntriesTargetRegion)
3970     for (const auto &F : D.second)
3971       for (const auto &P : F.second)
3972         for (const auto &L : P.second)
3973           Action(D.first, F.first, P.first(), L.first, L.second);
3974 }
3975 
3976 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3977     initializeDeviceGlobalVarEntryInfo(StringRef Name,
3978                                        OMPTargetGlobalVarEntryKind Flags,
3979                                        unsigned Order) {
3980   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3981                                              "only required for the device "
3982                                              "code generation.");
3983   OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
3984   ++OffloadingEntriesNum;
3985 }
3986 
3987 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3988     registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr,
3989                                      CharUnits VarSize,
3990                                      OMPTargetGlobalVarEntryKind Flags,
3991                                      llvm::GlobalValue::LinkageTypes Linkage) {
3992   if (CGM.getLangOpts().OpenMPIsDevice) {
3993     auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
3994     assert(Entry.isValid() && Entry.getFlags() == Flags &&
3995            "Entry not initialized!");
3996     assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
3997            "Resetting with the new address.");
3998     if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) {
3999       if (Entry.getVarSize().isZero()) {
4000         Entry.setVarSize(VarSize);
4001         Entry.setLinkage(Linkage);
4002       }
4003       return;
4004     }
4005     Entry.setVarSize(VarSize);
4006     Entry.setLinkage(Linkage);
4007     Entry.setAddress(Addr);
4008   } else {
4009     if (hasDeviceGlobalVarEntryInfo(VarName)) {
4010       auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
4011       assert(Entry.isValid() && Entry.getFlags() == Flags &&
4012              "Entry not initialized!");
4013       assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
4014              "Resetting with the new address.");
4015       if (Entry.getVarSize().isZero()) {
4016         Entry.setVarSize(VarSize);
4017         Entry.setLinkage(Linkage);
4018       }
4019       return;
4020     }
4021     OffloadEntriesDeviceGlobalVar.try_emplace(
4022         VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage);
4023     ++OffloadingEntriesNum;
4024   }
4025 }
4026 
4027 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
4028     actOnDeviceGlobalVarEntriesInfo(
4029         const OffloadDeviceGlobalVarEntryInfoActTy &Action) {
4030   // Scan all target region entries and perform the provided action.
4031   for (const auto &E : OffloadEntriesDeviceGlobalVar)
4032     Action(E.getKey(), E.getValue());
4033 }
4034 
4035 void CGOpenMPRuntime::createOffloadEntry(
4036     llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags,
4037     llvm::GlobalValue::LinkageTypes Linkage) {
4038   StringRef Name = Addr->getName();
4039   llvm::Module &M = CGM.getModule();
4040   llvm::LLVMContext &C = M.getContext();
4041 
4042   // Create constant string with the name.
4043   llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name);
4044 
4045   std::string StringName = getName({"omp_offloading", "entry_name"});
4046   auto *Str = new llvm::GlobalVariable(
4047       M, StrPtrInit->getType(), /*isConstant=*/true,
4048       llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName);
4049   Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
4050 
4051   llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy),
4052                             llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy),
4053                             llvm::ConstantInt::get(CGM.SizeTy, Size),
4054                             llvm::ConstantInt::get(CGM.Int32Ty, Flags),
4055                             llvm::ConstantInt::get(CGM.Int32Ty, 0)};
4056   std::string EntryName = getName({"omp_offloading", "entry", ""});
4057   llvm::GlobalVariable *Entry = createGlobalStruct(
4058       CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data,
4059       Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage);
4060 
4061   // The entry has to be created in the section the linker expects it to be.
4062   Entry->setSection("omp_offloading_entries");
4063 }
4064 
4065 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() {
4066   // Emit the offloading entries and metadata so that the device codegen side
4067   // can easily figure out what to emit. The produced metadata looks like
4068   // this:
4069   //
4070   // !omp_offload.info = !{!1, ...}
4071   //
4072   // Right now we only generate metadata for function that contain target
4073   // regions.
4074 
4075   // If we are in simd mode or there are no entries, we don't need to do
4076   // anything.
4077   if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty())
4078     return;
4079 
4080   llvm::Module &M = CGM.getModule();
4081   llvm::LLVMContext &C = M.getContext();
4082   SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *,
4083                          SourceLocation, StringRef>,
4084               16>
4085       OrderedEntries(OffloadEntriesInfoManager.size());
4086   llvm::SmallVector<StringRef, 16> ParentFunctions(
4087       OffloadEntriesInfoManager.size());
4088 
4089   // Auxiliary methods to create metadata values and strings.
4090   auto &&GetMDInt = [this](unsigned V) {
4091     return llvm::ConstantAsMetadata::get(
4092         llvm::ConstantInt::get(CGM.Int32Ty, V));
4093   };
4094 
4095   auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); };
4096 
4097   // Create the offloading info metadata node.
4098   llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info");
4099 
4100   // Create function that emits metadata for each target region entry;
4101   auto &&TargetRegionMetadataEmitter =
4102       [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt,
4103        &GetMDString](
4104           unsigned DeviceID, unsigned FileID, StringRef ParentName,
4105           unsigned Line,
4106           const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) {
4107         // Generate metadata for target regions. Each entry of this metadata
4108         // contains:
4109         // - Entry 0 -> Kind of this type of metadata (0).
4110         // - Entry 1 -> Device ID of the file where the entry was identified.
4111         // - Entry 2 -> File ID of the file where the entry was identified.
4112         // - Entry 3 -> Mangled name of the function where the entry was
4113         // identified.
4114         // - Entry 4 -> Line in the file where the entry was identified.
4115         // - Entry 5 -> Order the entry was created.
4116         // The first element of the metadata node is the kind.
4117         llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID),
4118                                  GetMDInt(FileID),      GetMDString(ParentName),
4119                                  GetMDInt(Line),        GetMDInt(E.getOrder())};
4120 
4121         SourceLocation Loc;
4122         for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(),
4123                   E = CGM.getContext().getSourceManager().fileinfo_end();
4124              I != E; ++I) {
4125           if (I->getFirst()->getUniqueID().getDevice() == DeviceID &&
4126               I->getFirst()->getUniqueID().getFile() == FileID) {
4127             Loc = CGM.getContext().getSourceManager().translateFileLineCol(
4128                 I->getFirst(), Line, 1);
4129             break;
4130           }
4131         }
4132         // Save this entry in the right position of the ordered entries array.
4133         OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName);
4134         ParentFunctions[E.getOrder()] = ParentName;
4135 
4136         // Add metadata to the named metadata node.
4137         MD->addOperand(llvm::MDNode::get(C, Ops));
4138       };
4139 
4140   OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo(
4141       TargetRegionMetadataEmitter);
4142 
4143   // Create function that emits metadata for each device global variable entry;
4144   auto &&DeviceGlobalVarMetadataEmitter =
4145       [&C, &OrderedEntries, &GetMDInt, &GetMDString,
4146        MD](StringRef MangledName,
4147            const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar
4148                &E) {
4149         // Generate metadata for global variables. Each entry of this metadata
4150         // contains:
4151         // - Entry 0 -> Kind of this type of metadata (1).
4152         // - Entry 1 -> Mangled name of the variable.
4153         // - Entry 2 -> Declare target kind.
4154         // - Entry 3 -> Order the entry was created.
4155         // The first element of the metadata node is the kind.
4156         llvm::Metadata *Ops[] = {
4157             GetMDInt(E.getKind()), GetMDString(MangledName),
4158             GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
4159 
4160         // Save this entry in the right position of the ordered entries array.
4161         OrderedEntries[E.getOrder()] =
4162             std::make_tuple(&E, SourceLocation(), MangledName);
4163 
4164         // Add metadata to the named metadata node.
4165         MD->addOperand(llvm::MDNode::get(C, Ops));
4166       };
4167 
4168   OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo(
4169       DeviceGlobalVarMetadataEmitter);
4170 
4171   for (const auto &E : OrderedEntries) {
4172     assert(std::get<0>(E) && "All ordered entries must exist!");
4173     if (const auto *CE =
4174             dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>(
4175                 std::get<0>(E))) {
4176       if (!CE->getID() || !CE->getAddress()) {
4177         // Do not blame the entry if the parent funtion is not emitted.
4178         StringRef FnName = ParentFunctions[CE->getOrder()];
4179         if (!CGM.GetGlobalValue(FnName))
4180           continue;
4181         unsigned DiagID = CGM.getDiags().getCustomDiagID(
4182             DiagnosticsEngine::Error,
4183             "Offloading entry for target region in %0 is incorrect: either the "
4184             "address or the ID is invalid.");
4185         CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName;
4186         continue;
4187       }
4188       createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0,
4189                          CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage);
4190     } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy::
4191                                              OffloadEntryInfoDeviceGlobalVar>(
4192                    std::get<0>(E))) {
4193       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags =
4194           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4195               CE->getFlags());
4196       switch (Flags) {
4197       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: {
4198         if (CGM.getLangOpts().OpenMPIsDevice &&
4199             CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())
4200           continue;
4201         if (!CE->getAddress()) {
4202           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4203               DiagnosticsEngine::Error, "Offloading entry for declare target "
4204                                         "variable %0 is incorrect: the "
4205                                         "address is invalid.");
4206           CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E);
4207           continue;
4208         }
4209         // The vaiable has no definition - no need to add the entry.
4210         if (CE->getVarSize().isZero())
4211           continue;
4212         break;
4213       }
4214       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink:
4215         assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) ||
4216                 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) &&
4217                "Declaret target link address is set.");
4218         if (CGM.getLangOpts().OpenMPIsDevice)
4219           continue;
4220         if (!CE->getAddress()) {
4221           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4222               DiagnosticsEngine::Error,
4223               "Offloading entry for declare target variable is incorrect: the "
4224               "address is invalid.");
4225           CGM.getDiags().Report(DiagID);
4226           continue;
4227         }
4228         break;
4229       }
4230       createOffloadEntry(CE->getAddress(), CE->getAddress(),
4231                          CE->getVarSize().getQuantity(), Flags,
4232                          CE->getLinkage());
4233     } else {
4234       llvm_unreachable("Unsupported entry kind.");
4235     }
4236   }
4237 }
4238 
4239 /// Loads all the offload entries information from the host IR
4240 /// metadata.
4241 void CGOpenMPRuntime::loadOffloadInfoMetadata() {
4242   // If we are in target mode, load the metadata from the host IR. This code has
4243   // to match the metadaata creation in createOffloadEntriesAndInfoMetadata().
4244 
4245   if (!CGM.getLangOpts().OpenMPIsDevice)
4246     return;
4247 
4248   if (CGM.getLangOpts().OMPHostIRFile.empty())
4249     return;
4250 
4251   auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile);
4252   if (auto EC = Buf.getError()) {
4253     CGM.getDiags().Report(diag::err_cannot_open_file)
4254         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4255     return;
4256   }
4257 
4258   llvm::LLVMContext C;
4259   auto ME = expectedToErrorOrAndEmitErrors(
4260       C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C));
4261 
4262   if (auto EC = ME.getError()) {
4263     unsigned DiagID = CGM.getDiags().getCustomDiagID(
4264         DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'");
4265     CGM.getDiags().Report(DiagID)
4266         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4267     return;
4268   }
4269 
4270   llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info");
4271   if (!MD)
4272     return;
4273 
4274   for (llvm::MDNode *MN : MD->operands()) {
4275     auto &&GetMDInt = [MN](unsigned Idx) {
4276       auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx));
4277       return cast<llvm::ConstantInt>(V->getValue())->getZExtValue();
4278     };
4279 
4280     auto &&GetMDString = [MN](unsigned Idx) {
4281       auto *V = cast<llvm::MDString>(MN->getOperand(Idx));
4282       return V->getString();
4283     };
4284 
4285     switch (GetMDInt(0)) {
4286     default:
4287       llvm_unreachable("Unexpected metadata!");
4288       break;
4289     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4290         OffloadingEntryInfoTargetRegion:
4291       OffloadEntriesInfoManager.initializeTargetRegionEntryInfo(
4292           /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2),
4293           /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4),
4294           /*Order=*/GetMDInt(5));
4295       break;
4296     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4297         OffloadingEntryInfoDeviceGlobalVar:
4298       OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo(
4299           /*MangledName=*/GetMDString(1),
4300           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4301               /*Flags=*/GetMDInt(2)),
4302           /*Order=*/GetMDInt(3));
4303       break;
4304     }
4305   }
4306 }
4307 
4308 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
4309   if (!KmpRoutineEntryPtrTy) {
4310     // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
4311     ASTContext &C = CGM.getContext();
4312     QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
4313     FunctionProtoType::ExtProtoInfo EPI;
4314     KmpRoutineEntryPtrQTy = C.getPointerType(
4315         C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
4316     KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
4317   }
4318 }
4319 
4320 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() {
4321   // Make sure the type of the entry is already created. This is the type we
4322   // have to create:
4323   // struct __tgt_offload_entry{
4324   //   void      *addr;       // Pointer to the offload entry info.
4325   //                          // (function or global)
4326   //   char      *name;       // Name of the function or global.
4327   //   size_t     size;       // Size of the entry info (0 if it a function).
4328   //   int32_t    flags;      // Flags associated with the entry, e.g. 'link'.
4329   //   int32_t    reserved;   // Reserved, to use by the runtime library.
4330   // };
4331   if (TgtOffloadEntryQTy.isNull()) {
4332     ASTContext &C = CGM.getContext();
4333     RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry");
4334     RD->startDefinition();
4335     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4336     addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy));
4337     addFieldToRecordDecl(C, RD, C.getSizeType());
4338     addFieldToRecordDecl(
4339         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4340     addFieldToRecordDecl(
4341         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4342     RD->completeDefinition();
4343     RD->addAttr(PackedAttr::CreateImplicit(C));
4344     TgtOffloadEntryQTy = C.getRecordType(RD);
4345   }
4346   return TgtOffloadEntryQTy;
4347 }
4348 
4349 namespace {
4350 struct PrivateHelpersTy {
4351   PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original,
4352                    const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit)
4353       : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy),
4354         PrivateElemInit(PrivateElemInit) {}
4355   const Expr *OriginalRef = nullptr;
4356   const VarDecl *Original = nullptr;
4357   const VarDecl *PrivateCopy = nullptr;
4358   const VarDecl *PrivateElemInit = nullptr;
4359 };
4360 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
4361 } // anonymous namespace
4362 
4363 static RecordDecl *
4364 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) {
4365   if (!Privates.empty()) {
4366     ASTContext &C = CGM.getContext();
4367     // Build struct .kmp_privates_t. {
4368     //         /*  private vars  */
4369     //       };
4370     RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t");
4371     RD->startDefinition();
4372     for (const auto &Pair : Privates) {
4373       const VarDecl *VD = Pair.second.Original;
4374       QualType Type = VD->getType().getNonReferenceType();
4375       FieldDecl *FD = addFieldToRecordDecl(C, RD, Type);
4376       if (VD->hasAttrs()) {
4377         for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()),
4378              E(VD->getAttrs().end());
4379              I != E; ++I)
4380           FD->addAttr(*I);
4381       }
4382     }
4383     RD->completeDefinition();
4384     return RD;
4385   }
4386   return nullptr;
4387 }
4388 
4389 static RecordDecl *
4390 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind,
4391                          QualType KmpInt32Ty,
4392                          QualType KmpRoutineEntryPointerQTy) {
4393   ASTContext &C = CGM.getContext();
4394   // Build struct kmp_task_t {
4395   //         void *              shareds;
4396   //         kmp_routine_entry_t routine;
4397   //         kmp_int32           part_id;
4398   //         kmp_cmplrdata_t data1;
4399   //         kmp_cmplrdata_t data2;
4400   // For taskloops additional fields:
4401   //         kmp_uint64          lb;
4402   //         kmp_uint64          ub;
4403   //         kmp_int64           st;
4404   //         kmp_int32           liter;
4405   //         void *              reductions;
4406   //       };
4407   RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union);
4408   UD->startDefinition();
4409   addFieldToRecordDecl(C, UD, KmpInt32Ty);
4410   addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy);
4411   UD->completeDefinition();
4412   QualType KmpCmplrdataTy = C.getRecordType(UD);
4413   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t");
4414   RD->startDefinition();
4415   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4416   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
4417   addFieldToRecordDecl(C, RD, KmpInt32Ty);
4418   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4419   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4420   if (isOpenMPTaskLoopDirective(Kind)) {
4421     QualType KmpUInt64Ty =
4422         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
4423     QualType KmpInt64Ty =
4424         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
4425     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4426     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4427     addFieldToRecordDecl(C, RD, KmpInt64Ty);
4428     addFieldToRecordDecl(C, RD, KmpInt32Ty);
4429     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4430   }
4431   RD->completeDefinition();
4432   return RD;
4433 }
4434 
4435 static RecordDecl *
4436 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy,
4437                                      ArrayRef<PrivateDataTy> Privates) {
4438   ASTContext &C = CGM.getContext();
4439   // Build struct kmp_task_t_with_privates {
4440   //         kmp_task_t task_data;
4441   //         .kmp_privates_t. privates;
4442   //       };
4443   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates");
4444   RD->startDefinition();
4445   addFieldToRecordDecl(C, RD, KmpTaskTQTy);
4446   if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates))
4447     addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD));
4448   RD->completeDefinition();
4449   return RD;
4450 }
4451 
4452 /// Emit a proxy function which accepts kmp_task_t as the second
4453 /// argument.
4454 /// \code
4455 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
4456 ///   TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt,
4457 ///   For taskloops:
4458 ///   tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4459 ///   tt->reductions, tt->shareds);
4460 ///   return 0;
4461 /// }
4462 /// \endcode
4463 static llvm::Function *
4464 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
4465                       OpenMPDirectiveKind Kind, QualType KmpInt32Ty,
4466                       QualType KmpTaskTWithPrivatesPtrQTy,
4467                       QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
4468                       QualType SharedsPtrTy, llvm::Function *TaskFunction,
4469                       llvm::Value *TaskPrivatesMap) {
4470   ASTContext &C = CGM.getContext();
4471   FunctionArgList Args;
4472   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4473                             ImplicitParamDecl::Other);
4474   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4475                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4476                                 ImplicitParamDecl::Other);
4477   Args.push_back(&GtidArg);
4478   Args.push_back(&TaskTypeArg);
4479   const auto &TaskEntryFnInfo =
4480       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4481   llvm::FunctionType *TaskEntryTy =
4482       CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
4483   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""});
4484   auto *TaskEntry = llvm::Function::Create(
4485       TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4486   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo);
4487   TaskEntry->setDoesNotRecurse();
4488   CodeGenFunction CGF(CGM);
4489   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args,
4490                     Loc, Loc);
4491 
4492   // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map,
4493   // tt,
4494   // For taskloops:
4495   // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4496   // tt->task_data.shareds);
4497   llvm::Value *GtidParam = CGF.EmitLoadOfScalar(
4498       CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc);
4499   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4500       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4501       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4502   const auto *KmpTaskTWithPrivatesQTyRD =
4503       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4504   LValue Base =
4505       CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4506   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
4507   auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
4508   LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI);
4509   llvm::Value *PartidParam = PartIdLVal.getPointer(CGF);
4510 
4511   auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds);
4512   LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI);
4513   llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4514       CGF.EmitLoadOfScalar(SharedsLVal, Loc),
4515       CGF.ConvertTypeForMem(SharedsPtrTy));
4516 
4517   auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
4518   llvm::Value *PrivatesParam;
4519   if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) {
4520     LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI);
4521     PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4522         PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy);
4523   } else {
4524     PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4525   }
4526 
4527   llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam,
4528                                TaskPrivatesMap,
4529                                CGF.Builder
4530                                    .CreatePointerBitCastOrAddrSpaceCast(
4531                                        TDBase.getAddress(CGF), CGF.VoidPtrTy)
4532                                    .getPointer()};
4533   SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs),
4534                                           std::end(CommonArgs));
4535   if (isOpenMPTaskLoopDirective(Kind)) {
4536     auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound);
4537     LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI);
4538     llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc);
4539     auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound);
4540     LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI);
4541     llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc);
4542     auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride);
4543     LValue StLVal = CGF.EmitLValueForField(Base, *StFI);
4544     llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc);
4545     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4546     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4547     llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc);
4548     auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions);
4549     LValue RLVal = CGF.EmitLValueForField(Base, *RFI);
4550     llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc);
4551     CallArgs.push_back(LBParam);
4552     CallArgs.push_back(UBParam);
4553     CallArgs.push_back(StParam);
4554     CallArgs.push_back(LIParam);
4555     CallArgs.push_back(RParam);
4556   }
4557   CallArgs.push_back(SharedsParam);
4558 
4559   CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction,
4560                                                   CallArgs);
4561   CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)),
4562                              CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty));
4563   CGF.FinishFunction();
4564   return TaskEntry;
4565 }
4566 
4567 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM,
4568                                             SourceLocation Loc,
4569                                             QualType KmpInt32Ty,
4570                                             QualType KmpTaskTWithPrivatesPtrQTy,
4571                                             QualType KmpTaskTWithPrivatesQTy) {
4572   ASTContext &C = CGM.getContext();
4573   FunctionArgList Args;
4574   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4575                             ImplicitParamDecl::Other);
4576   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4577                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4578                                 ImplicitParamDecl::Other);
4579   Args.push_back(&GtidArg);
4580   Args.push_back(&TaskTypeArg);
4581   const auto &DestructorFnInfo =
4582       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4583   llvm::FunctionType *DestructorFnTy =
4584       CGM.getTypes().GetFunctionType(DestructorFnInfo);
4585   std::string Name =
4586       CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""});
4587   auto *DestructorFn =
4588       llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage,
4589                              Name, &CGM.getModule());
4590   CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn,
4591                                     DestructorFnInfo);
4592   DestructorFn->setDoesNotRecurse();
4593   CodeGenFunction CGF(CGM);
4594   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo,
4595                     Args, Loc, Loc);
4596 
4597   LValue Base = CGF.EmitLoadOfPointerLValue(
4598       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4599       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4600   const auto *KmpTaskTWithPrivatesQTyRD =
4601       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4602   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4603   Base = CGF.EmitLValueForField(Base, *FI);
4604   for (const auto *Field :
4605        cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) {
4606     if (QualType::DestructionKind DtorKind =
4607             Field->getType().isDestructedType()) {
4608       LValue FieldLValue = CGF.EmitLValueForField(Base, Field);
4609       CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType());
4610     }
4611   }
4612   CGF.FinishFunction();
4613   return DestructorFn;
4614 }
4615 
4616 /// Emit a privates mapping function for correct handling of private and
4617 /// firstprivate variables.
4618 /// \code
4619 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1>
4620 /// **noalias priv1,...,  <tyn> **noalias privn) {
4621 ///   *priv1 = &.privates.priv1;
4622 ///   ...;
4623 ///   *privn = &.privates.privn;
4624 /// }
4625 /// \endcode
4626 static llvm::Value *
4627 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc,
4628                                ArrayRef<const Expr *> PrivateVars,
4629                                ArrayRef<const Expr *> FirstprivateVars,
4630                                ArrayRef<const Expr *> LastprivateVars,
4631                                QualType PrivatesQTy,
4632                                ArrayRef<PrivateDataTy> Privates) {
4633   ASTContext &C = CGM.getContext();
4634   FunctionArgList Args;
4635   ImplicitParamDecl TaskPrivatesArg(
4636       C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4637       C.getPointerType(PrivatesQTy).withConst().withRestrict(),
4638       ImplicitParamDecl::Other);
4639   Args.push_back(&TaskPrivatesArg);
4640   llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos;
4641   unsigned Counter = 1;
4642   for (const Expr *E : PrivateVars) {
4643     Args.push_back(ImplicitParamDecl::Create(
4644         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4645         C.getPointerType(C.getPointerType(E->getType()))
4646             .withConst()
4647             .withRestrict(),
4648         ImplicitParamDecl::Other));
4649     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4650     PrivateVarsPos[VD] = Counter;
4651     ++Counter;
4652   }
4653   for (const Expr *E : FirstprivateVars) {
4654     Args.push_back(ImplicitParamDecl::Create(
4655         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4656         C.getPointerType(C.getPointerType(E->getType()))
4657             .withConst()
4658             .withRestrict(),
4659         ImplicitParamDecl::Other));
4660     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4661     PrivateVarsPos[VD] = Counter;
4662     ++Counter;
4663   }
4664   for (const Expr *E : LastprivateVars) {
4665     Args.push_back(ImplicitParamDecl::Create(
4666         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4667         C.getPointerType(C.getPointerType(E->getType()))
4668             .withConst()
4669             .withRestrict(),
4670         ImplicitParamDecl::Other));
4671     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4672     PrivateVarsPos[VD] = Counter;
4673     ++Counter;
4674   }
4675   const auto &TaskPrivatesMapFnInfo =
4676       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4677   llvm::FunctionType *TaskPrivatesMapTy =
4678       CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo);
4679   std::string Name =
4680       CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""});
4681   auto *TaskPrivatesMap = llvm::Function::Create(
4682       TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name,
4683       &CGM.getModule());
4684   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap,
4685                                     TaskPrivatesMapFnInfo);
4686   if (CGM.getLangOpts().Optimize) {
4687     TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline);
4688     TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone);
4689     TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline);
4690   }
4691   CodeGenFunction CGF(CGM);
4692   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap,
4693                     TaskPrivatesMapFnInfo, Args, Loc, Loc);
4694 
4695   // *privi = &.privates.privi;
4696   LValue Base = CGF.EmitLoadOfPointerLValue(
4697       CGF.GetAddrOfLocalVar(&TaskPrivatesArg),
4698       TaskPrivatesArg.getType()->castAs<PointerType>());
4699   const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl());
4700   Counter = 0;
4701   for (const FieldDecl *Field : PrivatesQTyRD->fields()) {
4702     LValue FieldLVal = CGF.EmitLValueForField(Base, Field);
4703     const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]];
4704     LValue RefLVal =
4705         CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType());
4706     LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue(
4707         RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>());
4708     CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal);
4709     ++Counter;
4710   }
4711   CGF.FinishFunction();
4712   return TaskPrivatesMap;
4713 }
4714 
4715 /// Emit initialization for private variables in task-based directives.
4716 static void emitPrivatesInit(CodeGenFunction &CGF,
4717                              const OMPExecutableDirective &D,
4718                              Address KmpTaskSharedsPtr, LValue TDBase,
4719                              const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4720                              QualType SharedsTy, QualType SharedsPtrTy,
4721                              const OMPTaskDataTy &Data,
4722                              ArrayRef<PrivateDataTy> Privates, bool ForDup) {
4723   ASTContext &C = CGF.getContext();
4724   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4725   LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI);
4726   OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind())
4727                                  ? OMPD_taskloop
4728                                  : OMPD_task;
4729   const CapturedStmt &CS = *D.getCapturedStmt(Kind);
4730   CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS);
4731   LValue SrcBase;
4732   bool IsTargetTask =
4733       isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) ||
4734       isOpenMPTargetExecutionDirective(D.getDirectiveKind());
4735   // For target-based directives skip 3 firstprivate arrays BasePointersArray,
4736   // PointersArray and SizesArray. The original variables for these arrays are
4737   // not captured and we get their addresses explicitly.
4738   if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) ||
4739       (IsTargetTask && KmpTaskSharedsPtr.isValid())) {
4740     SrcBase = CGF.MakeAddrLValue(
4741         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4742             KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)),
4743         SharedsTy);
4744   }
4745   FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin();
4746   for (const PrivateDataTy &Pair : Privates) {
4747     const VarDecl *VD = Pair.second.PrivateCopy;
4748     const Expr *Init = VD->getAnyInitializer();
4749     if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) &&
4750                              !CGF.isTrivialInitializer(Init)))) {
4751       LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
4752       if (const VarDecl *Elem = Pair.second.PrivateElemInit) {
4753         const VarDecl *OriginalVD = Pair.second.Original;
4754         // Check if the variable is the target-based BasePointersArray,
4755         // PointersArray or SizesArray.
4756         LValue SharedRefLValue;
4757         QualType Type = PrivateLValue.getType();
4758         const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD);
4759         if (IsTargetTask && !SharedField) {
4760           assert(isa<ImplicitParamDecl>(OriginalVD) &&
4761                  isa<CapturedDecl>(OriginalVD->getDeclContext()) &&
4762                  cast<CapturedDecl>(OriginalVD->getDeclContext())
4763                          ->getNumParams() == 0 &&
4764                  isa<TranslationUnitDecl>(
4765                      cast<CapturedDecl>(OriginalVD->getDeclContext())
4766                          ->getDeclContext()) &&
4767                  "Expected artificial target data variable.");
4768           SharedRefLValue =
4769               CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type);
4770         } else if (ForDup) {
4771           SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField);
4772           SharedRefLValue = CGF.MakeAddrLValue(
4773               Address(SharedRefLValue.getPointer(CGF),
4774                       C.getDeclAlign(OriginalVD)),
4775               SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl),
4776               SharedRefLValue.getTBAAInfo());
4777         } else {
4778           InlinedOpenMPRegionRAII Region(
4779               CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown,
4780               /*HasCancel=*/false);
4781           SharedRefLValue =  CGF.EmitLValue(Pair.second.OriginalRef);
4782         }
4783         if (Type->isArrayType()) {
4784           // Initialize firstprivate array.
4785           if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) {
4786             // Perform simple memcpy.
4787             CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type);
4788           } else {
4789             // Initialize firstprivate array using element-by-element
4790             // initialization.
4791             CGF.EmitOMPAggregateAssign(
4792                 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF),
4793                 Type,
4794                 [&CGF, Elem, Init, &CapturesInfo](Address DestElement,
4795                                                   Address SrcElement) {
4796                   // Clean up any temporaries needed by the initialization.
4797                   CodeGenFunction::OMPPrivateScope InitScope(CGF);
4798                   InitScope.addPrivate(
4799                       Elem, [SrcElement]() -> Address { return SrcElement; });
4800                   (void)InitScope.Privatize();
4801                   // Emit initialization for single element.
4802                   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(
4803                       CGF, &CapturesInfo);
4804                   CGF.EmitAnyExprToMem(Init, DestElement,
4805                                        Init->getType().getQualifiers(),
4806                                        /*IsInitializer=*/false);
4807                 });
4808           }
4809         } else {
4810           CodeGenFunction::OMPPrivateScope InitScope(CGF);
4811           InitScope.addPrivate(Elem, [SharedRefLValue, &CGF]() -> Address {
4812             return SharedRefLValue.getAddress(CGF);
4813           });
4814           (void)InitScope.Privatize();
4815           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo);
4816           CGF.EmitExprAsInit(Init, VD, PrivateLValue,
4817                              /*capturedByInit=*/false);
4818         }
4819       } else {
4820         CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false);
4821       }
4822     }
4823     ++FI;
4824   }
4825 }
4826 
4827 /// Check if duplication function is required for taskloops.
4828 static bool checkInitIsRequired(CodeGenFunction &CGF,
4829                                 ArrayRef<PrivateDataTy> Privates) {
4830   bool InitRequired = false;
4831   for (const PrivateDataTy &Pair : Privates) {
4832     const VarDecl *VD = Pair.second.PrivateCopy;
4833     const Expr *Init = VD->getAnyInitializer();
4834     InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) &&
4835                                     !CGF.isTrivialInitializer(Init));
4836     if (InitRequired)
4837       break;
4838   }
4839   return InitRequired;
4840 }
4841 
4842 
4843 /// Emit task_dup function (for initialization of
4844 /// private/firstprivate/lastprivate vars and last_iter flag)
4845 /// \code
4846 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int
4847 /// lastpriv) {
4848 /// // setup lastprivate flag
4849 ///    task_dst->last = lastpriv;
4850 /// // could be constructor calls here...
4851 /// }
4852 /// \endcode
4853 static llvm::Value *
4854 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc,
4855                     const OMPExecutableDirective &D,
4856                     QualType KmpTaskTWithPrivatesPtrQTy,
4857                     const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4858                     const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy,
4859                     QualType SharedsPtrTy, const OMPTaskDataTy &Data,
4860                     ArrayRef<PrivateDataTy> Privates, bool WithLastIter) {
4861   ASTContext &C = CGM.getContext();
4862   FunctionArgList Args;
4863   ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4864                            KmpTaskTWithPrivatesPtrQTy,
4865                            ImplicitParamDecl::Other);
4866   ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4867                            KmpTaskTWithPrivatesPtrQTy,
4868                            ImplicitParamDecl::Other);
4869   ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy,
4870                                 ImplicitParamDecl::Other);
4871   Args.push_back(&DstArg);
4872   Args.push_back(&SrcArg);
4873   Args.push_back(&LastprivArg);
4874   const auto &TaskDupFnInfo =
4875       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4876   llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo);
4877   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""});
4878   auto *TaskDup = llvm::Function::Create(
4879       TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4880   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo);
4881   TaskDup->setDoesNotRecurse();
4882   CodeGenFunction CGF(CGM);
4883   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc,
4884                     Loc);
4885 
4886   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4887       CGF.GetAddrOfLocalVar(&DstArg),
4888       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4889   // task_dst->liter = lastpriv;
4890   if (WithLastIter) {
4891     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4892     LValue Base = CGF.EmitLValueForField(
4893         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4894     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4895     llvm::Value *Lastpriv = CGF.EmitLoadOfScalar(
4896         CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc);
4897     CGF.EmitStoreOfScalar(Lastpriv, LILVal);
4898   }
4899 
4900   // Emit initial values for private copies (if any).
4901   assert(!Privates.empty());
4902   Address KmpTaskSharedsPtr = Address::invalid();
4903   if (!Data.FirstprivateVars.empty()) {
4904     LValue TDBase = CGF.EmitLoadOfPointerLValue(
4905         CGF.GetAddrOfLocalVar(&SrcArg),
4906         KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4907     LValue Base = CGF.EmitLValueForField(
4908         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4909     KmpTaskSharedsPtr = Address(
4910         CGF.EmitLoadOfScalar(CGF.EmitLValueForField(
4911                                  Base, *std::next(KmpTaskTQTyRD->field_begin(),
4912                                                   KmpTaskTShareds)),
4913                              Loc),
4914         CGF.getNaturalTypeAlignment(SharedsTy));
4915   }
4916   emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD,
4917                    SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true);
4918   CGF.FinishFunction();
4919   return TaskDup;
4920 }
4921 
4922 /// Checks if destructor function is required to be generated.
4923 /// \return true if cleanups are required, false otherwise.
4924 static bool
4925 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) {
4926   bool NeedsCleanup = false;
4927   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
4928   const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl());
4929   for (const FieldDecl *FD : PrivateRD->fields()) {
4930     NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType();
4931     if (NeedsCleanup)
4932       break;
4933   }
4934   return NeedsCleanup;
4935 }
4936 
4937 CGOpenMPRuntime::TaskResultTy
4938 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc,
4939                               const OMPExecutableDirective &D,
4940                               llvm::Function *TaskFunction, QualType SharedsTy,
4941                               Address Shareds, const OMPTaskDataTy &Data) {
4942   ASTContext &C = CGM.getContext();
4943   llvm::SmallVector<PrivateDataTy, 4> Privates;
4944   // Aggregate privates and sort them by the alignment.
4945   const auto *I = Data.PrivateCopies.begin();
4946   for (const Expr *E : Data.PrivateVars) {
4947     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4948     Privates.emplace_back(
4949         C.getDeclAlign(VD),
4950         PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4951                          /*PrivateElemInit=*/nullptr));
4952     ++I;
4953   }
4954   I = Data.FirstprivateCopies.begin();
4955   const auto *IElemInitRef = Data.FirstprivateInits.begin();
4956   for (const Expr *E : Data.FirstprivateVars) {
4957     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4958     Privates.emplace_back(
4959         C.getDeclAlign(VD),
4960         PrivateHelpersTy(
4961             E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4962             cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl())));
4963     ++I;
4964     ++IElemInitRef;
4965   }
4966   I = Data.LastprivateCopies.begin();
4967   for (const Expr *E : Data.LastprivateVars) {
4968     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4969     Privates.emplace_back(
4970         C.getDeclAlign(VD),
4971         PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4972                          /*PrivateElemInit=*/nullptr));
4973     ++I;
4974   }
4975   llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) {
4976     return L.first > R.first;
4977   });
4978   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
4979   // Build type kmp_routine_entry_t (if not built yet).
4980   emitKmpRoutineEntryT(KmpInt32Ty);
4981   // Build type kmp_task_t (if not built yet).
4982   if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) {
4983     if (SavedKmpTaskloopTQTy.isNull()) {
4984       SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl(
4985           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
4986     }
4987     KmpTaskTQTy = SavedKmpTaskloopTQTy;
4988   } else {
4989     assert((D.getDirectiveKind() == OMPD_task ||
4990             isOpenMPTargetExecutionDirective(D.getDirectiveKind()) ||
4991             isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) &&
4992            "Expected taskloop, task or target directive");
4993     if (SavedKmpTaskTQTy.isNull()) {
4994       SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl(
4995           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
4996     }
4997     KmpTaskTQTy = SavedKmpTaskTQTy;
4998   }
4999   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
5000   // Build particular struct kmp_task_t for the given task.
5001   const RecordDecl *KmpTaskTWithPrivatesQTyRD =
5002       createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates);
5003   QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD);
5004   QualType KmpTaskTWithPrivatesPtrQTy =
5005       C.getPointerType(KmpTaskTWithPrivatesQTy);
5006   llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy);
5007   llvm::Type *KmpTaskTWithPrivatesPtrTy =
5008       KmpTaskTWithPrivatesTy->getPointerTo();
5009   llvm::Value *KmpTaskTWithPrivatesTySize =
5010       CGF.getTypeSize(KmpTaskTWithPrivatesQTy);
5011   QualType SharedsPtrTy = C.getPointerType(SharedsTy);
5012 
5013   // Emit initial values for private copies (if any).
5014   llvm::Value *TaskPrivatesMap = nullptr;
5015   llvm::Type *TaskPrivatesMapTy =
5016       std::next(TaskFunction->arg_begin(), 3)->getType();
5017   if (!Privates.empty()) {
5018     auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
5019     TaskPrivatesMap = emitTaskPrivateMappingFunction(
5020         CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars,
5021         FI->getType(), Privates);
5022     TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5023         TaskPrivatesMap, TaskPrivatesMapTy);
5024   } else {
5025     TaskPrivatesMap = llvm::ConstantPointerNull::get(
5026         cast<llvm::PointerType>(TaskPrivatesMapTy));
5027   }
5028   // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
5029   // kmp_task_t *tt);
5030   llvm::Function *TaskEntry = emitProxyTaskFunction(
5031       CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5032       KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction,
5033       TaskPrivatesMap);
5034 
5035   // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
5036   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
5037   // kmp_routine_entry_t *task_entry);
5038   // Task flags. Format is taken from
5039   // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h,
5040   // description of kmp_tasking_flags struct.
5041   enum {
5042     TiedFlag = 0x1,
5043     FinalFlag = 0x2,
5044     DestructorsFlag = 0x8,
5045     PriorityFlag = 0x20,
5046     DetachableFlag = 0x40,
5047   };
5048   unsigned Flags = Data.Tied ? TiedFlag : 0;
5049   bool NeedsCleanup = false;
5050   if (!Privates.empty()) {
5051     NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD);
5052     if (NeedsCleanup)
5053       Flags = Flags | DestructorsFlag;
5054   }
5055   if (Data.Priority.getInt())
5056     Flags = Flags | PriorityFlag;
5057   if (D.hasClausesOfKind<OMPDetachClause>())
5058     Flags = Flags | DetachableFlag;
5059   llvm::Value *TaskFlags =
5060       Data.Final.getPointer()
5061           ? CGF.Builder.CreateSelect(Data.Final.getPointer(),
5062                                      CGF.Builder.getInt32(FinalFlag),
5063                                      CGF.Builder.getInt32(/*C=*/0))
5064           : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0);
5065   TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
5066   llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy));
5067   SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc),
5068       getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize,
5069       SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5070           TaskEntry, KmpRoutineEntryPtrTy)};
5071   llvm::Value *NewTask;
5072   if (D.hasClausesOfKind<OMPNowaitClause>()) {
5073     // Check if we have any device clause associated with the directive.
5074     const Expr *Device = nullptr;
5075     if (auto *C = D.getSingleClause<OMPDeviceClause>())
5076       Device = C->getDevice();
5077     // Emit device ID if any otherwise use default value.
5078     llvm::Value *DeviceID;
5079     if (Device)
5080       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
5081                                            CGF.Int64Ty, /*isSigned=*/true);
5082     else
5083       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
5084     AllocArgs.push_back(DeviceID);
5085     NewTask = CGF.EmitRuntimeCall(
5086       createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs);
5087   } else {
5088     NewTask = CGF.EmitRuntimeCall(
5089       createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs);
5090   }
5091   // Emit detach clause initialization.
5092   // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid,
5093   // task_descriptor);
5094   if (const auto *DC = D.getSingleClause<OMPDetachClause>()) {
5095     const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts();
5096     LValue EvtLVal = CGF.EmitLValue(Evt);
5097 
5098     // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
5099     // int gtid, kmp_task_t *task);
5100     llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc());
5101     llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc());
5102     Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false);
5103     llvm::Value *EvtVal = CGF.EmitRuntimeCall(
5104         createRuntimeFunction(OMPRTL__kmpc_task_allow_completion_event),
5105         {Loc, Tid, NewTask});
5106     EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(),
5107                                       Evt->getExprLoc());
5108     CGF.EmitStoreOfScalar(EvtVal, EvtLVal);
5109   }
5110   llvm::Value *NewTaskNewTaskTTy =
5111       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5112           NewTask, KmpTaskTWithPrivatesPtrTy);
5113   LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy,
5114                                                KmpTaskTWithPrivatesQTy);
5115   LValue TDBase =
5116       CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
5117   // Fill the data in the resulting kmp_task_t record.
5118   // Copy shareds if there are any.
5119   Address KmpTaskSharedsPtr = Address::invalid();
5120   if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) {
5121     KmpTaskSharedsPtr =
5122         Address(CGF.EmitLoadOfScalar(
5123                     CGF.EmitLValueForField(
5124                         TDBase, *std::next(KmpTaskTQTyRD->field_begin(),
5125                                            KmpTaskTShareds)),
5126                     Loc),
5127                 CGF.getNaturalTypeAlignment(SharedsTy));
5128     LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy);
5129     LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy);
5130     CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap);
5131   }
5132   // Emit initial values for private copies (if any).
5133   TaskResultTy Result;
5134   if (!Privates.empty()) {
5135     emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD,
5136                      SharedsTy, SharedsPtrTy, Data, Privates,
5137                      /*ForDup=*/false);
5138     if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) &&
5139         (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) {
5140       Result.TaskDupFn = emitTaskDupFunction(
5141           CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD,
5142           KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates,
5143           /*WithLastIter=*/!Data.LastprivateVars.empty());
5144     }
5145   }
5146   // Fields of union "kmp_cmplrdata_t" for destructors and priority.
5147   enum { Priority = 0, Destructors = 1 };
5148   // Provide pointer to function with destructors for privates.
5149   auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1);
5150   const RecordDecl *KmpCmplrdataUD =
5151       (*FI)->getType()->getAsUnionType()->getDecl();
5152   if (NeedsCleanup) {
5153     llvm::Value *DestructorFn = emitDestructorsFunction(
5154         CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5155         KmpTaskTWithPrivatesQTy);
5156     LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI);
5157     LValue DestructorsLV = CGF.EmitLValueForField(
5158         Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors));
5159     CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5160                               DestructorFn, KmpRoutineEntryPtrTy),
5161                           DestructorsLV);
5162   }
5163   // Set priority.
5164   if (Data.Priority.getInt()) {
5165     LValue Data2LV = CGF.EmitLValueForField(
5166         TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2));
5167     LValue PriorityLV = CGF.EmitLValueForField(
5168         Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority));
5169     CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV);
5170   }
5171   Result.NewTask = NewTask;
5172   Result.TaskEntry = TaskEntry;
5173   Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy;
5174   Result.TDBase = TDBase;
5175   Result.KmpTaskTQTyRD = KmpTaskTQTyRD;
5176   return Result;
5177 }
5178 
5179 namespace {
5180 /// Dependence kind for RTL.
5181 enum RTLDependenceKindTy {
5182   DepIn = 0x01,
5183   DepInOut = 0x3,
5184   DepMutexInOutSet = 0x4
5185 };
5186 /// Fields ids in kmp_depend_info record.
5187 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags };
5188 } // namespace
5189 
5190 /// Translates internal dependency kind into the runtime kind.
5191 static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) {
5192   RTLDependenceKindTy DepKind;
5193   switch (K) {
5194   case OMPC_DEPEND_in:
5195     DepKind = DepIn;
5196     break;
5197   // Out and InOut dependencies must use the same code.
5198   case OMPC_DEPEND_out:
5199   case OMPC_DEPEND_inout:
5200     DepKind = DepInOut;
5201     break;
5202   case OMPC_DEPEND_mutexinoutset:
5203     DepKind = DepMutexInOutSet;
5204     break;
5205   case OMPC_DEPEND_source:
5206   case OMPC_DEPEND_sink:
5207   case OMPC_DEPEND_depobj:
5208   case OMPC_DEPEND_unknown:
5209     llvm_unreachable("Unknown task dependence type");
5210   }
5211   return DepKind;
5212 }
5213 
5214 /// Builds kmp_depend_info, if it is not built yet, and builds flags type.
5215 static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy,
5216                            QualType &FlagsTy) {
5217   FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false);
5218   if (KmpDependInfoTy.isNull()) {
5219     RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info");
5220     KmpDependInfoRD->startDefinition();
5221     addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType());
5222     addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType());
5223     addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy);
5224     KmpDependInfoRD->completeDefinition();
5225     KmpDependInfoTy = C.getRecordType(KmpDependInfoRD);
5226   }
5227 }
5228 
5229 std::pair<llvm::Value *, LValue>
5230 CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal,
5231                                    SourceLocation Loc) {
5232   ASTContext &C = CGM.getContext();
5233   QualType FlagsTy;
5234   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5235   RecordDecl *KmpDependInfoRD =
5236       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5237   LValue Base = CGF.EmitLoadOfPointerLValue(
5238       DepobjLVal.getAddress(CGF),
5239       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
5240   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
5241   Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5242           Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy));
5243   Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(),
5244                             Base.getTBAAInfo());
5245   llvm::Value *DepObjAddr = CGF.Builder.CreateGEP(
5246       Addr.getPointer(),
5247       llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
5248   LValue NumDepsBase = CGF.MakeAddrLValue(
5249       Address(DepObjAddr, Addr.getAlignment()), KmpDependInfoTy,
5250       Base.getBaseInfo(), Base.getTBAAInfo());
5251   // NumDeps = deps[i].base_addr;
5252   LValue BaseAddrLVal = CGF.EmitLValueForField(
5253       NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5254   llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc);
5255   return std::make_pair(NumDeps, Base);
5256 }
5257 
5258 std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause(
5259     CodeGenFunction &CGF,
5260     ArrayRef<std::pair<OpenMPDependClauseKind, const Expr *>> Dependencies,
5261     bool ForDepobj, SourceLocation Loc) {
5262   // Process list of dependencies.
5263   ASTContext &C = CGM.getContext();
5264   Address DependenciesArray = Address::invalid();
5265   unsigned NumDependencies = Dependencies.size();
5266   llvm::Value *NumOfElements = nullptr;
5267   if (NumDependencies) {
5268     QualType FlagsTy;
5269     getDependTypes(C, KmpDependInfoTy, FlagsTy);
5270     RecordDecl *KmpDependInfoRD =
5271         cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5272     llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
5273     unsigned NumDepobjDependecies = 0;
5274     SmallVector<std::pair<llvm::Value *, LValue>, 4> Depobjs;
5275     llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0);
5276     // Calculate number of depobj dependecies.
5277     for (const std::pair<OpenMPDependClauseKind, const Expr *> &Pair :
5278          Dependencies) {
5279       if (Pair.first != OMPC_DEPEND_depobj)
5280         continue;
5281       LValue DepobjLVal = CGF.EmitLValue(Pair.second);
5282       llvm::Value *NumDeps;
5283       LValue Base;
5284       std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc);
5285       NumOfDepobjElements =
5286           CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumDeps);
5287       Depobjs.emplace_back(NumDeps, Base);
5288       ++NumDepobjDependecies;
5289     }
5290 
5291     QualType KmpDependInfoArrayTy;
5292     // Define type kmp_depend_info[<Dependencies.size()>];
5293     // For depobj reserve one extra element to store the number of elements.
5294     // It is required to handle depobj(x) update(in) construct.
5295     // kmp_depend_info[<Dependencies.size()>] deps;
5296     if (ForDepobj) {
5297       assert(NumDepobjDependecies == 0 &&
5298              "depobj dependency kind is not expected in depobj directive.");
5299       KmpDependInfoArrayTy = C.getConstantArrayType(
5300           KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1),
5301           nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
5302       // Need to allocate on the dynamic memory.
5303       llvm::Value *ThreadID = getThreadID(CGF, Loc);
5304       // Use default allocator.
5305       llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5306       CharUnits Align = C.getTypeAlignInChars(KmpDependInfoArrayTy);
5307       CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy);
5308       llvm::Value *Size = CGF.CGM.getSize(Sz.alignTo(Align));
5309       llvm::Value *Args[] = {ThreadID, Size, Allocator};
5310 
5311       llvm::Value *Addr = CGF.EmitRuntimeCall(
5312           createRuntimeFunction(OMPRTL__kmpc_alloc), Args, ".dep.arr.addr");
5313       Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5314           Addr, CGF.ConvertTypeForMem(KmpDependInfoArrayTy)->getPointerTo());
5315       DependenciesArray = Address(Addr, Align);
5316       NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies,
5317                                              /*isSigned=*/false);
5318     } else if (NumDepobjDependecies > 0) {
5319       NumOfElements = CGF.Builder.CreateNUWAdd(
5320           NumOfDepobjElements,
5321           llvm::ConstantInt::get(CGM.IntPtrTy,
5322                                  NumDependencies - NumDepobjDependecies,
5323                                  /*isSigned=*/false));
5324       NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty,
5325                                                 /*isSigned=*/false);
5326       OpaqueValueExpr OVE(
5327           Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0),
5328           VK_RValue);
5329       CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE,
5330                                                     RValue::get(NumOfElements));
5331       KmpDependInfoArrayTy =
5332           C.getVariableArrayType(KmpDependInfoTy, &OVE, ArrayType::Normal,
5333                                  /*IndexTypeQuals=*/0, SourceRange(Loc, Loc));
5334       // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy);
5335       // Properly emit variable-sized array.
5336       auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy,
5337                                            ImplicitParamDecl::Other);
5338       CGF.EmitVarDecl(*PD);
5339       DependenciesArray = CGF.GetAddrOfLocalVar(PD);
5340     } else {
5341       KmpDependInfoArrayTy = C.getConstantArrayType(
5342           KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies),
5343           nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
5344       DependenciesArray =
5345           CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr");
5346       NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies,
5347                                              /*isSigned=*/false);
5348     }
5349     if (ForDepobj) {
5350       // Write number of elements in the first element of array for depobj.
5351       llvm::Value *NumVal =
5352           llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies);
5353       LValue Base = CGF.MakeAddrLValue(
5354           CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0),
5355           KmpDependInfoTy);
5356       // deps[i].base_addr = NumDependencies;
5357       LValue BaseAddrLVal = CGF.EmitLValueForField(
5358           Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5359       CGF.EmitStoreOfScalar(NumVal, BaseAddrLVal);
5360     }
5361     unsigned Pos = ForDepobj ? 1 : 0;
5362     for (unsigned I = 0; I < NumDependencies; ++I) {
5363       if (Dependencies[I].first == OMPC_DEPEND_depobj)
5364         continue;
5365       const Expr *E = Dependencies[I].second;
5366       LValue Addr = CGF.EmitLValue(E);
5367       llvm::Value *Size;
5368       QualType Ty = E->getType();
5369       if (const auto *ASE =
5370               dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) {
5371         LValue UpAddrLVal =
5372             CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false);
5373         llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32(
5374             UpAddrLVal.getPointer(CGF), /*Idx0=*/1);
5375         llvm::Value *LowIntPtr =
5376             CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGM.SizeTy);
5377         llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy);
5378         Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr);
5379       } else {
5380         Size = CGF.getTypeSize(Ty);
5381       }
5382       LValue Base;
5383       if (NumDepobjDependecies > 0) {
5384         Base = CGF.MakeAddrLValue(
5385             CGF.Builder.CreateConstGEP(DependenciesArray, Pos),
5386             KmpDependInfoTy);
5387       } else {
5388         Base = CGF.MakeAddrLValue(
5389             CGF.Builder.CreateConstArrayGEP(DependenciesArray, Pos),
5390             KmpDependInfoTy);
5391       }
5392       // deps[i].base_addr = &<Dependencies[i].second>;
5393       LValue BaseAddrLVal = CGF.EmitLValueForField(
5394           Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5395       CGF.EmitStoreOfScalar(
5396           CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGF.IntPtrTy),
5397           BaseAddrLVal);
5398       // deps[i].len = sizeof(<Dependencies[i].second>);
5399       LValue LenLVal = CGF.EmitLValueForField(
5400           Base, *std::next(KmpDependInfoRD->field_begin(), Len));
5401       CGF.EmitStoreOfScalar(Size, LenLVal);
5402       // deps[i].flags = <Dependencies[i].first>;
5403       RTLDependenceKindTy DepKind =
5404           translateDependencyKind(Dependencies[I].first);
5405       LValue FlagsLVal = CGF.EmitLValueForField(
5406           Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
5407       CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
5408                             FlagsLVal);
5409       ++Pos;
5410     }
5411     // Copy final depobj arrays.
5412     if (NumDepobjDependecies > 0) {
5413       llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy);
5414       Address Addr = CGF.Builder.CreateConstGEP(DependenciesArray, Pos);
5415       for (const std::pair<llvm::Value *, LValue> &Pair : Depobjs) {
5416         llvm::Value *Size = CGF.Builder.CreateNUWMul(ElSize, Pair.first);
5417         CGF.Builder.CreateMemCpy(Addr, Pair.second.getAddress(CGF), Size);
5418         Addr =
5419             Address(CGF.Builder.CreateGEP(
5420                         Addr.getElementType(), Addr.getPointer(), Pair.first),
5421                     DependenciesArray.getAlignment().alignmentOfArrayElement(
5422                         C.getTypeSizeInChars(KmpDependInfoTy)));
5423       }
5424       DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5425           DependenciesArray, CGF.VoidPtrTy);
5426     } else {
5427       DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5428           CGF.Builder.CreateConstArrayGEP(DependenciesArray, ForDepobj ? 1 : 0),
5429           CGF.VoidPtrTy);
5430     }
5431   }
5432   return std::make_pair(NumOfElements, DependenciesArray);
5433 }
5434 
5435 void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal,
5436                                         SourceLocation Loc) {
5437   ASTContext &C = CGM.getContext();
5438   QualType FlagsTy;
5439   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5440   LValue Base = CGF.EmitLoadOfPointerLValue(
5441       DepobjLVal.getAddress(CGF),
5442       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
5443   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
5444   Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5445       Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy));
5446   llvm::Value *DepObjAddr = CGF.Builder.CreateGEP(
5447       Addr.getPointer(),
5448       llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
5449   DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr,
5450                                                                CGF.VoidPtrTy);
5451   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5452   // Use default allocator.
5453   llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5454   llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator};
5455 
5456   // _kmpc_free(gtid, addr, nullptr);
5457   (void)CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_free), Args);
5458 }
5459 
5460 void CGOpenMPRuntime::emitUpdateClause(CodeGenFunction &CGF, LValue DepobjLVal,
5461                                        OpenMPDependClauseKind NewDepKind,
5462                                        SourceLocation Loc) {
5463   ASTContext &C = CGM.getContext();
5464   QualType FlagsTy;
5465   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5466   RecordDecl *KmpDependInfoRD =
5467       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5468   llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
5469   llvm::Value *NumDeps;
5470   LValue Base;
5471   std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc);
5472 
5473   Address Begin = Base.getAddress(CGF);
5474   // Cast from pointer to array type to pointer to single element.
5475   llvm::Value *End = CGF.Builder.CreateGEP(Begin.getPointer(), NumDeps);
5476   // The basic structure here is a while-do loop.
5477   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body");
5478   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done");
5479   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5480   CGF.EmitBlock(BodyBB);
5481   llvm::PHINode *ElementPHI =
5482       CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast");
5483   ElementPHI->addIncoming(Begin.getPointer(), EntryBB);
5484   Begin = Address(ElementPHI, Begin.getAlignment());
5485   Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(),
5486                             Base.getTBAAInfo());
5487   // deps[i].flags = NewDepKind;
5488   RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind);
5489   LValue FlagsLVal = CGF.EmitLValueForField(
5490       Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
5491   CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
5492                         FlagsLVal);
5493 
5494   // Shift the address forward by one element.
5495   Address ElementNext =
5496       CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext");
5497   ElementPHI->addIncoming(ElementNext.getPointer(),
5498                           CGF.Builder.GetInsertBlock());
5499   llvm::Value *IsEmpty =
5500       CGF.Builder.CreateICmpEQ(ElementNext.getPointer(), End, "omp.isempty");
5501   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5502   // Done.
5503   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5504 }
5505 
5506 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
5507                                    const OMPExecutableDirective &D,
5508                                    llvm::Function *TaskFunction,
5509                                    QualType SharedsTy, Address Shareds,
5510                                    const Expr *IfCond,
5511                                    const OMPTaskDataTy &Data) {
5512   if (!CGF.HaveInsertPoint())
5513     return;
5514 
5515   TaskResultTy Result =
5516       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5517   llvm::Value *NewTask = Result.NewTask;
5518   llvm::Function *TaskEntry = Result.TaskEntry;
5519   llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy;
5520   LValue TDBase = Result.TDBase;
5521   const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD;
5522   // Process list of dependences.
5523   Address DependenciesArray = Address::invalid();
5524   llvm::Value *NumOfElements;
5525   std::tie(NumOfElements, DependenciesArray) =
5526       emitDependClause(CGF, Data.Dependences, /*ForDepobj=*/false, Loc);
5527 
5528   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5529   // libcall.
5530   // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
5531   // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
5532   // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence
5533   // list is not empty
5534   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5535   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5536   llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask };
5537   llvm::Value *DepTaskArgs[7];
5538   if (!Data.Dependences.empty()) {
5539     DepTaskArgs[0] = UpLoc;
5540     DepTaskArgs[1] = ThreadID;
5541     DepTaskArgs[2] = NewTask;
5542     DepTaskArgs[3] = NumOfElements;
5543     DepTaskArgs[4] = DependenciesArray.getPointer();
5544     DepTaskArgs[5] = CGF.Builder.getInt32(0);
5545     DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5546   }
5547   auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs,
5548                         &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) {
5549     if (!Data.Tied) {
5550       auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
5551       LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI);
5552       CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal);
5553     }
5554     if (!Data.Dependences.empty()) {
5555       CGF.EmitRuntimeCall(
5556           createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs);
5557     } else {
5558       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task),
5559                           TaskArgs);
5560     }
5561     // Check if parent region is untied and build return for untied task;
5562     if (auto *Region =
5563             dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
5564       Region->emitUntiedSwitch(CGF);
5565   };
5566 
5567   llvm::Value *DepWaitTaskArgs[6];
5568   if (!Data.Dependences.empty()) {
5569     DepWaitTaskArgs[0] = UpLoc;
5570     DepWaitTaskArgs[1] = ThreadID;
5571     DepWaitTaskArgs[2] = NumOfElements;
5572     DepWaitTaskArgs[3] = DependenciesArray.getPointer();
5573     DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
5574     DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5575   }
5576   auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry,
5577                         &Data, &DepWaitTaskArgs,
5578                         Loc](CodeGenFunction &CGF, PrePostActionTy &) {
5579     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5580     CodeGenFunction::RunCleanupsScope LocalScope(CGF);
5581     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
5582     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
5583     // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
5584     // is specified.
5585     if (!Data.Dependences.empty())
5586       CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps),
5587                           DepWaitTaskArgs);
5588     // Call proxy_task_entry(gtid, new_task);
5589     auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy,
5590                       Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
5591       Action.Enter(CGF);
5592       llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
5593       CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry,
5594                                                           OutlinedFnArgs);
5595     };
5596 
5597     // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
5598     // kmp_task_t *new_task);
5599     // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
5600     // kmp_task_t *new_task);
5601     RegionCodeGenTy RCG(CodeGen);
5602     CommonActionTy Action(
5603         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs,
5604         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs);
5605     RCG.setAction(Action);
5606     RCG(CGF);
5607   };
5608 
5609   if (IfCond) {
5610     emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen);
5611   } else {
5612     RegionCodeGenTy ThenRCG(ThenCodeGen);
5613     ThenRCG(CGF);
5614   }
5615 }
5616 
5617 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc,
5618                                        const OMPLoopDirective &D,
5619                                        llvm::Function *TaskFunction,
5620                                        QualType SharedsTy, Address Shareds,
5621                                        const Expr *IfCond,
5622                                        const OMPTaskDataTy &Data) {
5623   if (!CGF.HaveInsertPoint())
5624     return;
5625   TaskResultTy Result =
5626       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5627   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5628   // libcall.
5629   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
5630   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
5631   // sched, kmp_uint64 grainsize, void *task_dup);
5632   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5633   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5634   llvm::Value *IfVal;
5635   if (IfCond) {
5636     IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy,
5637                                       /*isSigned=*/true);
5638   } else {
5639     IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1);
5640   }
5641 
5642   LValue LBLVal = CGF.EmitLValueForField(
5643       Result.TDBase,
5644       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound));
5645   const auto *LBVar =
5646       cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl());
5647   CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF),
5648                        LBLVal.getQuals(),
5649                        /*IsInitializer=*/true);
5650   LValue UBLVal = CGF.EmitLValueForField(
5651       Result.TDBase,
5652       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound));
5653   const auto *UBVar =
5654       cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl());
5655   CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF),
5656                        UBLVal.getQuals(),
5657                        /*IsInitializer=*/true);
5658   LValue StLVal = CGF.EmitLValueForField(
5659       Result.TDBase,
5660       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride));
5661   const auto *StVar =
5662       cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl());
5663   CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF),
5664                        StLVal.getQuals(),
5665                        /*IsInitializer=*/true);
5666   // Store reductions address.
5667   LValue RedLVal = CGF.EmitLValueForField(
5668       Result.TDBase,
5669       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions));
5670   if (Data.Reductions) {
5671     CGF.EmitStoreOfScalar(Data.Reductions, RedLVal);
5672   } else {
5673     CGF.EmitNullInitialization(RedLVal.getAddress(CGF),
5674                                CGF.getContext().VoidPtrTy);
5675   }
5676   enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 };
5677   llvm::Value *TaskArgs[] = {
5678       UpLoc,
5679       ThreadID,
5680       Result.NewTask,
5681       IfVal,
5682       LBLVal.getPointer(CGF),
5683       UBLVal.getPointer(CGF),
5684       CGF.EmitLoadOfScalar(StLVal, Loc),
5685       llvm::ConstantInt::getSigned(
5686           CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler
5687       llvm::ConstantInt::getSigned(
5688           CGF.IntTy, Data.Schedule.getPointer()
5689                          ? Data.Schedule.getInt() ? NumTasks : Grainsize
5690                          : NoSchedule),
5691       Data.Schedule.getPointer()
5692           ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty,
5693                                       /*isSigned=*/false)
5694           : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0),
5695       Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5696                              Result.TaskDupFn, CGF.VoidPtrTy)
5697                        : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)};
5698   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs);
5699 }
5700 
5701 /// Emit reduction operation for each element of array (required for
5702 /// array sections) LHS op = RHS.
5703 /// \param Type Type of array.
5704 /// \param LHSVar Variable on the left side of the reduction operation
5705 /// (references element of array in original variable).
5706 /// \param RHSVar Variable on the right side of the reduction operation
5707 /// (references element of array in original variable).
5708 /// \param RedOpGen Generator of reduction operation with use of LHSVar and
5709 /// RHSVar.
5710 static void EmitOMPAggregateReduction(
5711     CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar,
5712     const VarDecl *RHSVar,
5713     const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *,
5714                                   const Expr *, const Expr *)> &RedOpGen,
5715     const Expr *XExpr = nullptr, const Expr *EExpr = nullptr,
5716     const Expr *UpExpr = nullptr) {
5717   // Perform element-by-element initialization.
5718   QualType ElementTy;
5719   Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar);
5720   Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar);
5721 
5722   // Drill down to the base element type on both arrays.
5723   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
5724   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr);
5725 
5726   llvm::Value *RHSBegin = RHSAddr.getPointer();
5727   llvm::Value *LHSBegin = LHSAddr.getPointer();
5728   // Cast from pointer to array type to pointer to single element.
5729   llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements);
5730   // The basic structure here is a while-do loop.
5731   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body");
5732   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done");
5733   llvm::Value *IsEmpty =
5734       CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty");
5735   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5736 
5737   // Enter the loop body, making that address the current address.
5738   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5739   CGF.EmitBlock(BodyBB);
5740 
5741   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
5742 
5743   llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI(
5744       RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast");
5745   RHSElementPHI->addIncoming(RHSBegin, EntryBB);
5746   Address RHSElementCurrent =
5747       Address(RHSElementPHI,
5748               RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5749 
5750   llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI(
5751       LHSBegin->getType(), 2, "omp.arraycpy.destElementPast");
5752   LHSElementPHI->addIncoming(LHSBegin, EntryBB);
5753   Address LHSElementCurrent =
5754       Address(LHSElementPHI,
5755               LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5756 
5757   // Emit copy.
5758   CodeGenFunction::OMPPrivateScope Scope(CGF);
5759   Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; });
5760   Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; });
5761   Scope.Privatize();
5762   RedOpGen(CGF, XExpr, EExpr, UpExpr);
5763   Scope.ForceCleanup();
5764 
5765   // Shift the address forward by one element.
5766   llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32(
5767       LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
5768   llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32(
5769       RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element");
5770   // Check whether we've reached the end.
5771   llvm::Value *Done =
5772       CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done");
5773   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
5774   LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock());
5775   RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock());
5776 
5777   // Done.
5778   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5779 }
5780 
5781 /// Emit reduction combiner. If the combiner is a simple expression emit it as
5782 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of
5783 /// UDR combiner function.
5784 static void emitReductionCombiner(CodeGenFunction &CGF,
5785                                   const Expr *ReductionOp) {
5786   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
5787     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
5788       if (const auto *DRE =
5789               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
5790         if (const auto *DRD =
5791                 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) {
5792           std::pair<llvm::Function *, llvm::Function *> Reduction =
5793               CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
5794           RValue Func = RValue::get(Reduction.first);
5795           CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
5796           CGF.EmitIgnoredExpr(ReductionOp);
5797           return;
5798         }
5799   CGF.EmitIgnoredExpr(ReductionOp);
5800 }
5801 
5802 llvm::Function *CGOpenMPRuntime::emitReductionFunction(
5803     SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates,
5804     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
5805     ArrayRef<const Expr *> ReductionOps) {
5806   ASTContext &C = CGM.getContext();
5807 
5808   // void reduction_func(void *LHSArg, void *RHSArg);
5809   FunctionArgList Args;
5810   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5811                            ImplicitParamDecl::Other);
5812   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5813                            ImplicitParamDecl::Other);
5814   Args.push_back(&LHSArg);
5815   Args.push_back(&RHSArg);
5816   const auto &CGFI =
5817       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5818   std::string Name = getName({"omp", "reduction", "reduction_func"});
5819   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
5820                                     llvm::GlobalValue::InternalLinkage, Name,
5821                                     &CGM.getModule());
5822   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
5823   Fn->setDoesNotRecurse();
5824   CodeGenFunction CGF(CGM);
5825   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
5826 
5827   // Dst = (void*[n])(LHSArg);
5828   // Src = (void*[n])(RHSArg);
5829   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5830       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
5831       ArgsType), CGF.getPointerAlign());
5832   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5833       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
5834       ArgsType), CGF.getPointerAlign());
5835 
5836   //  ...
5837   //  *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
5838   //  ...
5839   CodeGenFunction::OMPPrivateScope Scope(CGF);
5840   auto IPriv = Privates.begin();
5841   unsigned Idx = 0;
5842   for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) {
5843     const auto *RHSVar =
5844         cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl());
5845     Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() {
5846       return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar);
5847     });
5848     const auto *LHSVar =
5849         cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl());
5850     Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() {
5851       return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar);
5852     });
5853     QualType PrivTy = (*IPriv)->getType();
5854     if (PrivTy->isVariablyModifiedType()) {
5855       // Get array size and emit VLA type.
5856       ++Idx;
5857       Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx);
5858       llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem);
5859       const VariableArrayType *VLA =
5860           CGF.getContext().getAsVariableArrayType(PrivTy);
5861       const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr());
5862       CodeGenFunction::OpaqueValueMapping OpaqueMap(
5863           CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy)));
5864       CGF.EmitVariablyModifiedType(PrivTy);
5865     }
5866   }
5867   Scope.Privatize();
5868   IPriv = Privates.begin();
5869   auto ILHS = LHSExprs.begin();
5870   auto IRHS = RHSExprs.begin();
5871   for (const Expr *E : ReductionOps) {
5872     if ((*IPriv)->getType()->isArrayType()) {
5873       // Emit reduction for array section.
5874       const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5875       const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5876       EmitOMPAggregateReduction(
5877           CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5878           [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5879             emitReductionCombiner(CGF, E);
5880           });
5881     } else {
5882       // Emit reduction for array subscript or single variable.
5883       emitReductionCombiner(CGF, E);
5884     }
5885     ++IPriv;
5886     ++ILHS;
5887     ++IRHS;
5888   }
5889   Scope.ForceCleanup();
5890   CGF.FinishFunction();
5891   return Fn;
5892 }
5893 
5894 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF,
5895                                                   const Expr *ReductionOp,
5896                                                   const Expr *PrivateRef,
5897                                                   const DeclRefExpr *LHS,
5898                                                   const DeclRefExpr *RHS) {
5899   if (PrivateRef->getType()->isArrayType()) {
5900     // Emit reduction for array section.
5901     const auto *LHSVar = cast<VarDecl>(LHS->getDecl());
5902     const auto *RHSVar = cast<VarDecl>(RHS->getDecl());
5903     EmitOMPAggregateReduction(
5904         CGF, PrivateRef->getType(), LHSVar, RHSVar,
5905         [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5906           emitReductionCombiner(CGF, ReductionOp);
5907         });
5908   } else {
5909     // Emit reduction for array subscript or single variable.
5910     emitReductionCombiner(CGF, ReductionOp);
5911   }
5912 }
5913 
5914 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc,
5915                                     ArrayRef<const Expr *> Privates,
5916                                     ArrayRef<const Expr *> LHSExprs,
5917                                     ArrayRef<const Expr *> RHSExprs,
5918                                     ArrayRef<const Expr *> ReductionOps,
5919                                     ReductionOptionsTy Options) {
5920   if (!CGF.HaveInsertPoint())
5921     return;
5922 
5923   bool WithNowait = Options.WithNowait;
5924   bool SimpleReduction = Options.SimpleReduction;
5925 
5926   // Next code should be emitted for reduction:
5927   //
5928   // static kmp_critical_name lock = { 0 };
5929   //
5930   // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
5931   //  *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
5932   //  ...
5933   //  *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
5934   //  *(Type<n>-1*)rhs[<n>-1]);
5935   // }
5936   //
5937   // ...
5938   // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
5939   // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5940   // RedList, reduce_func, &<lock>)) {
5941   // case 1:
5942   //  ...
5943   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5944   //  ...
5945   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5946   // break;
5947   // case 2:
5948   //  ...
5949   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5950   //  ...
5951   // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
5952   // break;
5953   // default:;
5954   // }
5955   //
5956   // if SimpleReduction is true, only the next code is generated:
5957   //  ...
5958   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5959   //  ...
5960 
5961   ASTContext &C = CGM.getContext();
5962 
5963   if (SimpleReduction) {
5964     CodeGenFunction::RunCleanupsScope Scope(CGF);
5965     auto IPriv = Privates.begin();
5966     auto ILHS = LHSExprs.begin();
5967     auto IRHS = RHSExprs.begin();
5968     for (const Expr *E : ReductionOps) {
5969       emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5970                                   cast<DeclRefExpr>(*IRHS));
5971       ++IPriv;
5972       ++ILHS;
5973       ++IRHS;
5974     }
5975     return;
5976   }
5977 
5978   // 1. Build a list of reduction variables.
5979   // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
5980   auto Size = RHSExprs.size();
5981   for (const Expr *E : Privates) {
5982     if (E->getType()->isVariablyModifiedType())
5983       // Reserve place for array size.
5984       ++Size;
5985   }
5986   llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size);
5987   QualType ReductionArrayTy =
5988       C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
5989                              /*IndexTypeQuals=*/0);
5990   Address ReductionList =
5991       CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list");
5992   auto IPriv = Privates.begin();
5993   unsigned Idx = 0;
5994   for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) {
5995     Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5996     CGF.Builder.CreateStore(
5997         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5998             CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy),
5999         Elem);
6000     if ((*IPriv)->getType()->isVariablyModifiedType()) {
6001       // Store array size.
6002       ++Idx;
6003       Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
6004       llvm::Value *Size = CGF.Builder.CreateIntCast(
6005           CGF.getVLASize(
6006                  CGF.getContext().getAsVariableArrayType((*IPriv)->getType()))
6007               .NumElts,
6008           CGF.SizeTy, /*isSigned=*/false);
6009       CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy),
6010                               Elem);
6011     }
6012   }
6013 
6014   // 2. Emit reduce_func().
6015   llvm::Function *ReductionFn = emitReductionFunction(
6016       Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates,
6017       LHSExprs, RHSExprs, ReductionOps);
6018 
6019   // 3. Create static kmp_critical_name lock = { 0 };
6020   std::string Name = getName({"reduction"});
6021   llvm::Value *Lock = getCriticalRegionLock(Name);
6022 
6023   // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
6024   // RedList, reduce_func, &<lock>);
6025   llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE);
6026   llvm::Value *ThreadId = getThreadID(CGF, Loc);
6027   llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy);
6028   llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6029       ReductionList.getPointer(), CGF.VoidPtrTy);
6030   llvm::Value *Args[] = {
6031       IdentTLoc,                             // ident_t *<loc>
6032       ThreadId,                              // i32 <gtid>
6033       CGF.Builder.getInt32(RHSExprs.size()), // i32 <n>
6034       ReductionArrayTySize,                  // size_type sizeof(RedList)
6035       RL,                                    // void *RedList
6036       ReductionFn, // void (*) (void *, void *) <reduce_func>
6037       Lock         // kmp_critical_name *&<lock>
6038   };
6039   llvm::Value *Res = CGF.EmitRuntimeCall(
6040       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait
6041                                        : OMPRTL__kmpc_reduce),
6042       Args);
6043 
6044   // 5. Build switch(res)
6045   llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default");
6046   llvm::SwitchInst *SwInst =
6047       CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2);
6048 
6049   // 6. Build case 1:
6050   //  ...
6051   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
6052   //  ...
6053   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
6054   // break;
6055   llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1");
6056   SwInst->addCase(CGF.Builder.getInt32(1), Case1BB);
6057   CGF.EmitBlock(Case1BB);
6058 
6059   // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
6060   llvm::Value *EndArgs[] = {
6061       IdentTLoc, // ident_t *<loc>
6062       ThreadId,  // i32 <gtid>
6063       Lock       // kmp_critical_name *&<lock>
6064   };
6065   auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps](
6066                        CodeGenFunction &CGF, PrePostActionTy &Action) {
6067     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6068     auto IPriv = Privates.begin();
6069     auto ILHS = LHSExprs.begin();
6070     auto IRHS = RHSExprs.begin();
6071     for (const Expr *E : ReductionOps) {
6072       RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
6073                                      cast<DeclRefExpr>(*IRHS));
6074       ++IPriv;
6075       ++ILHS;
6076       ++IRHS;
6077     }
6078   };
6079   RegionCodeGenTy RCG(CodeGen);
6080   CommonActionTy Action(
6081       nullptr, llvm::None,
6082       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait
6083                                        : OMPRTL__kmpc_end_reduce),
6084       EndArgs);
6085   RCG.setAction(Action);
6086   RCG(CGF);
6087 
6088   CGF.EmitBranch(DefaultBB);
6089 
6090   // 7. Build case 2:
6091   //  ...
6092   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
6093   //  ...
6094   // break;
6095   llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2");
6096   SwInst->addCase(CGF.Builder.getInt32(2), Case2BB);
6097   CGF.EmitBlock(Case2BB);
6098 
6099   auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps](
6100                              CodeGenFunction &CGF, PrePostActionTy &Action) {
6101     auto ILHS = LHSExprs.begin();
6102     auto IRHS = RHSExprs.begin();
6103     auto IPriv = Privates.begin();
6104     for (const Expr *E : ReductionOps) {
6105       const Expr *XExpr = nullptr;
6106       const Expr *EExpr = nullptr;
6107       const Expr *UpExpr = nullptr;
6108       BinaryOperatorKind BO = BO_Comma;
6109       if (const auto *BO = dyn_cast<BinaryOperator>(E)) {
6110         if (BO->getOpcode() == BO_Assign) {
6111           XExpr = BO->getLHS();
6112           UpExpr = BO->getRHS();
6113         }
6114       }
6115       // Try to emit update expression as a simple atomic.
6116       const Expr *RHSExpr = UpExpr;
6117       if (RHSExpr) {
6118         // Analyze RHS part of the whole expression.
6119         if (const auto *ACO = dyn_cast<AbstractConditionalOperator>(
6120                 RHSExpr->IgnoreParenImpCasts())) {
6121           // If this is a conditional operator, analyze its condition for
6122           // min/max reduction operator.
6123           RHSExpr = ACO->getCond();
6124         }
6125         if (const auto *BORHS =
6126                 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) {
6127           EExpr = BORHS->getRHS();
6128           BO = BORHS->getOpcode();
6129         }
6130       }
6131       if (XExpr) {
6132         const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
6133         auto &&AtomicRedGen = [BO, VD,
6134                                Loc](CodeGenFunction &CGF, const Expr *XExpr,
6135                                     const Expr *EExpr, const Expr *UpExpr) {
6136           LValue X = CGF.EmitLValue(XExpr);
6137           RValue E;
6138           if (EExpr)
6139             E = CGF.EmitAnyExpr(EExpr);
6140           CGF.EmitOMPAtomicSimpleUpdateExpr(
6141               X, E, BO, /*IsXLHSInRHSPart=*/true,
6142               llvm::AtomicOrdering::Monotonic, Loc,
6143               [&CGF, UpExpr, VD, Loc](RValue XRValue) {
6144                 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
6145                 PrivateScope.addPrivate(
6146                     VD, [&CGF, VD, XRValue, Loc]() {
6147                       Address LHSTemp = CGF.CreateMemTemp(VD->getType());
6148                       CGF.emitOMPSimpleStore(
6149                           CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue,
6150                           VD->getType().getNonReferenceType(), Loc);
6151                       return LHSTemp;
6152                     });
6153                 (void)PrivateScope.Privatize();
6154                 return CGF.EmitAnyExpr(UpExpr);
6155               });
6156         };
6157         if ((*IPriv)->getType()->isArrayType()) {
6158           // Emit atomic reduction for array section.
6159           const auto *RHSVar =
6160               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
6161           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar,
6162                                     AtomicRedGen, XExpr, EExpr, UpExpr);
6163         } else {
6164           // Emit atomic reduction for array subscript or single variable.
6165           AtomicRedGen(CGF, XExpr, EExpr, UpExpr);
6166         }
6167       } else {
6168         // Emit as a critical region.
6169         auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *,
6170                                            const Expr *, const Expr *) {
6171           CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6172           std::string Name = RT.getName({"atomic_reduction"});
6173           RT.emitCriticalRegion(
6174               CGF, Name,
6175               [=](CodeGenFunction &CGF, PrePostActionTy &Action) {
6176                 Action.Enter(CGF);
6177                 emitReductionCombiner(CGF, E);
6178               },
6179               Loc);
6180         };
6181         if ((*IPriv)->getType()->isArrayType()) {
6182           const auto *LHSVar =
6183               cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
6184           const auto *RHSVar =
6185               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
6186           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar,
6187                                     CritRedGen);
6188         } else {
6189           CritRedGen(CGF, nullptr, nullptr, nullptr);
6190         }
6191       }
6192       ++ILHS;
6193       ++IRHS;
6194       ++IPriv;
6195     }
6196   };
6197   RegionCodeGenTy AtomicRCG(AtomicCodeGen);
6198   if (!WithNowait) {
6199     // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
6200     llvm::Value *EndArgs[] = {
6201         IdentTLoc, // ident_t *<loc>
6202         ThreadId,  // i32 <gtid>
6203         Lock       // kmp_critical_name *&<lock>
6204     };
6205     CommonActionTy Action(nullptr, llvm::None,
6206                           createRuntimeFunction(OMPRTL__kmpc_end_reduce),
6207                           EndArgs);
6208     AtomicRCG.setAction(Action);
6209     AtomicRCG(CGF);
6210   } else {
6211     AtomicRCG(CGF);
6212   }
6213 
6214   CGF.EmitBranch(DefaultBB);
6215   CGF.EmitBlock(DefaultBB, /*IsFinished=*/true);
6216 }
6217 
6218 /// Generates unique name for artificial threadprivate variables.
6219 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>"
6220 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix,
6221                                       const Expr *Ref) {
6222   SmallString<256> Buffer;
6223   llvm::raw_svector_ostream Out(Buffer);
6224   const clang::DeclRefExpr *DE;
6225   const VarDecl *D = ::getBaseDecl(Ref, DE);
6226   if (!D)
6227     D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl());
6228   D = D->getCanonicalDecl();
6229   std::string Name = CGM.getOpenMPRuntime().getName(
6230       {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)});
6231   Out << Prefix << Name << "_"
6232       << D->getCanonicalDecl()->getBeginLoc().getRawEncoding();
6233   return std::string(Out.str());
6234 }
6235 
6236 /// Emits reduction initializer function:
6237 /// \code
6238 /// void @.red_init(void* %arg) {
6239 /// %0 = bitcast void* %arg to <type>*
6240 /// store <type> <init>, <type>* %0
6241 /// ret void
6242 /// }
6243 /// \endcode
6244 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM,
6245                                            SourceLocation Loc,
6246                                            ReductionCodeGen &RCG, unsigned N) {
6247   ASTContext &C = CGM.getContext();
6248   FunctionArgList Args;
6249   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6250                           ImplicitParamDecl::Other);
6251   Args.emplace_back(&Param);
6252   const auto &FnInfo =
6253       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6254   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6255   std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""});
6256   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6257                                     Name, &CGM.getModule());
6258   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6259   Fn->setDoesNotRecurse();
6260   CodeGenFunction CGF(CGM);
6261   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6262   Address PrivateAddr = CGF.EmitLoadOfPointer(
6263       CGF.GetAddrOfLocalVar(&Param),
6264       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6265   llvm::Value *Size = nullptr;
6266   // If the size of the reduction item is non-constant, load it from global
6267   // threadprivate variable.
6268   if (RCG.getSizes(N).second) {
6269     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6270         CGF, CGM.getContext().getSizeType(),
6271         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6272     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6273                                 CGM.getContext().getSizeType(), Loc);
6274   }
6275   RCG.emitAggregateType(CGF, N, Size);
6276   LValue SharedLVal;
6277   // If initializer uses initializer from declare reduction construct, emit a
6278   // pointer to the address of the original reduction item (reuired by reduction
6279   // initializer)
6280   if (RCG.usesReductionInitializer(N)) {
6281     Address SharedAddr =
6282         CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6283             CGF, CGM.getContext().VoidPtrTy,
6284             generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6285     SharedAddr = CGF.EmitLoadOfPointer(
6286         SharedAddr,
6287         CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr());
6288     SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy);
6289   } else {
6290     SharedLVal = CGF.MakeNaturalAlignAddrLValue(
6291         llvm::ConstantPointerNull::get(CGM.VoidPtrTy),
6292         CGM.getContext().VoidPtrTy);
6293   }
6294   // Emit the initializer:
6295   // %0 = bitcast void* %arg to <type>*
6296   // store <type> <init>, <type>* %0
6297   RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal,
6298                          [](CodeGenFunction &) { return false; });
6299   CGF.FinishFunction();
6300   return Fn;
6301 }
6302 
6303 /// Emits reduction combiner function:
6304 /// \code
6305 /// void @.red_comb(void* %arg0, void* %arg1) {
6306 /// %lhs = bitcast void* %arg0 to <type>*
6307 /// %rhs = bitcast void* %arg1 to <type>*
6308 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs)
6309 /// store <type> %2, <type>* %lhs
6310 /// ret void
6311 /// }
6312 /// \endcode
6313 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM,
6314                                            SourceLocation Loc,
6315                                            ReductionCodeGen &RCG, unsigned N,
6316                                            const Expr *ReductionOp,
6317                                            const Expr *LHS, const Expr *RHS,
6318                                            const Expr *PrivateRef) {
6319   ASTContext &C = CGM.getContext();
6320   const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl());
6321   const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl());
6322   FunctionArgList Args;
6323   ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
6324                                C.VoidPtrTy, ImplicitParamDecl::Other);
6325   ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6326                             ImplicitParamDecl::Other);
6327   Args.emplace_back(&ParamInOut);
6328   Args.emplace_back(&ParamIn);
6329   const auto &FnInfo =
6330       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6331   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6332   std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""});
6333   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6334                                     Name, &CGM.getModule());
6335   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6336   Fn->setDoesNotRecurse();
6337   CodeGenFunction CGF(CGM);
6338   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6339   llvm::Value *Size = nullptr;
6340   // If the size of the reduction item is non-constant, load it from global
6341   // threadprivate variable.
6342   if (RCG.getSizes(N).second) {
6343     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6344         CGF, CGM.getContext().getSizeType(),
6345         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6346     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6347                                 CGM.getContext().getSizeType(), Loc);
6348   }
6349   RCG.emitAggregateType(CGF, N, Size);
6350   // Remap lhs and rhs variables to the addresses of the function arguments.
6351   // %lhs = bitcast void* %arg0 to <type>*
6352   // %rhs = bitcast void* %arg1 to <type>*
6353   CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
6354   PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() {
6355     // Pull out the pointer to the variable.
6356     Address PtrAddr = CGF.EmitLoadOfPointer(
6357         CGF.GetAddrOfLocalVar(&ParamInOut),
6358         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6359     return CGF.Builder.CreateElementBitCast(
6360         PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType()));
6361   });
6362   PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() {
6363     // Pull out the pointer to the variable.
6364     Address PtrAddr = CGF.EmitLoadOfPointer(
6365         CGF.GetAddrOfLocalVar(&ParamIn),
6366         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6367     return CGF.Builder.CreateElementBitCast(
6368         PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType()));
6369   });
6370   PrivateScope.Privatize();
6371   // Emit the combiner body:
6372   // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs)
6373   // store <type> %2, <type>* %lhs
6374   CGM.getOpenMPRuntime().emitSingleReductionCombiner(
6375       CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS),
6376       cast<DeclRefExpr>(RHS));
6377   CGF.FinishFunction();
6378   return Fn;
6379 }
6380 
6381 /// Emits reduction finalizer function:
6382 /// \code
6383 /// void @.red_fini(void* %arg) {
6384 /// %0 = bitcast void* %arg to <type>*
6385 /// <destroy>(<type>* %0)
6386 /// ret void
6387 /// }
6388 /// \endcode
6389 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM,
6390                                            SourceLocation Loc,
6391                                            ReductionCodeGen &RCG, unsigned N) {
6392   if (!RCG.needCleanups(N))
6393     return nullptr;
6394   ASTContext &C = CGM.getContext();
6395   FunctionArgList Args;
6396   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6397                           ImplicitParamDecl::Other);
6398   Args.emplace_back(&Param);
6399   const auto &FnInfo =
6400       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6401   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6402   std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""});
6403   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6404                                     Name, &CGM.getModule());
6405   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6406   Fn->setDoesNotRecurse();
6407   CodeGenFunction CGF(CGM);
6408   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6409   Address PrivateAddr = CGF.EmitLoadOfPointer(
6410       CGF.GetAddrOfLocalVar(&Param),
6411       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6412   llvm::Value *Size = nullptr;
6413   // If the size of the reduction item is non-constant, load it from global
6414   // threadprivate variable.
6415   if (RCG.getSizes(N).second) {
6416     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6417         CGF, CGM.getContext().getSizeType(),
6418         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6419     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6420                                 CGM.getContext().getSizeType(), Loc);
6421   }
6422   RCG.emitAggregateType(CGF, N, Size);
6423   // Emit the finalizer body:
6424   // <destroy>(<type>* %0)
6425   RCG.emitCleanups(CGF, N, PrivateAddr);
6426   CGF.FinishFunction(Loc);
6427   return Fn;
6428 }
6429 
6430 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit(
6431     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
6432     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
6433   if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty())
6434     return nullptr;
6435 
6436   // Build typedef struct:
6437   // kmp_task_red_input {
6438   //   void *reduce_shar; // shared reduction item
6439   //   size_t reduce_size; // size of data item
6440   //   void *reduce_init; // data initialization routine
6441   //   void *reduce_fini; // data finalization routine
6442   //   void *reduce_comb; // data combiner routine
6443   //   kmp_task_red_flags_t flags; // flags for additional info from compiler
6444   // } kmp_task_red_input_t;
6445   ASTContext &C = CGM.getContext();
6446   RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t");
6447   RD->startDefinition();
6448   const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6449   const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType());
6450   const FieldDecl *InitFD  = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6451   const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6452   const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6453   const FieldDecl *FlagsFD = addFieldToRecordDecl(
6454       C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false));
6455   RD->completeDefinition();
6456   QualType RDType = C.getRecordType(RD);
6457   unsigned Size = Data.ReductionVars.size();
6458   llvm::APInt ArraySize(/*numBits=*/64, Size);
6459   QualType ArrayRDType = C.getConstantArrayType(
6460       RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
6461   // kmp_task_red_input_t .rd_input.[Size];
6462   Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input.");
6463   ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies,
6464                        Data.ReductionOps);
6465   for (unsigned Cnt = 0; Cnt < Size; ++Cnt) {
6466     // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt];
6467     llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0),
6468                            llvm::ConstantInt::get(CGM.SizeTy, Cnt)};
6469     llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP(
6470         TaskRedInput.getPointer(), Idxs,
6471         /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc,
6472         ".rd_input.gep.");
6473     LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType);
6474     // ElemLVal.reduce_shar = &Shareds[Cnt];
6475     LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD);
6476     RCG.emitSharedLValue(CGF, Cnt);
6477     llvm::Value *CastedShared =
6478         CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF));
6479     CGF.EmitStoreOfScalar(CastedShared, SharedLVal);
6480     RCG.emitAggregateType(CGF, Cnt);
6481     llvm::Value *SizeValInChars;
6482     llvm::Value *SizeVal;
6483     std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt);
6484     // We use delayed creation/initialization for VLAs, array sections and
6485     // custom reduction initializations. It is required because runtime does not
6486     // provide the way to pass the sizes of VLAs/array sections to
6487     // initializer/combiner/finalizer functions and does not pass the pointer to
6488     // original reduction item to the initializer. Instead threadprivate global
6489     // variables are used to store these values and use them in the functions.
6490     bool DelayedCreation = !!SizeVal;
6491     SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy,
6492                                                /*isSigned=*/false);
6493     LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD);
6494     CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal);
6495     // ElemLVal.reduce_init = init;
6496     LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD);
6497     llvm::Value *InitAddr =
6498         CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt));
6499     CGF.EmitStoreOfScalar(InitAddr, InitLVal);
6500     DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt);
6501     // ElemLVal.reduce_fini = fini;
6502     LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD);
6503     llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt);
6504     llvm::Value *FiniAddr = Fini
6505                                 ? CGF.EmitCastToVoidPtr(Fini)
6506                                 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy);
6507     CGF.EmitStoreOfScalar(FiniAddr, FiniLVal);
6508     // ElemLVal.reduce_comb = comb;
6509     LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD);
6510     llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction(
6511         CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt],
6512         RHSExprs[Cnt], Data.ReductionCopies[Cnt]));
6513     CGF.EmitStoreOfScalar(CombAddr, CombLVal);
6514     // ElemLVal.flags = 0;
6515     LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD);
6516     if (DelayedCreation) {
6517       CGF.EmitStoreOfScalar(
6518           llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true),
6519           FlagsLVal);
6520     } else
6521       CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF),
6522                                  FlagsLVal.getType());
6523   }
6524   // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void
6525   // *data);
6526   llvm::Value *Args[] = {
6527       CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy,
6528                                 /*isSigned=*/true),
6529       llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6530       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(),
6531                                                       CGM.VoidPtrTy)};
6532   return CGF.EmitRuntimeCall(
6533       createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args);
6534 }
6535 
6536 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
6537                                               SourceLocation Loc,
6538                                               ReductionCodeGen &RCG,
6539                                               unsigned N) {
6540   auto Sizes = RCG.getSizes(N);
6541   // Emit threadprivate global variable if the type is non-constant
6542   // (Sizes.second = nullptr).
6543   if (Sizes.second) {
6544     llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy,
6545                                                      /*isSigned=*/false);
6546     Address SizeAddr = getAddrOfArtificialThreadPrivate(
6547         CGF, CGM.getContext().getSizeType(),
6548         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6549     CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false);
6550   }
6551   // Store address of the original reduction item if custom initializer is used.
6552   if (RCG.usesReductionInitializer(N)) {
6553     Address SharedAddr = getAddrOfArtificialThreadPrivate(
6554         CGF, CGM.getContext().VoidPtrTy,
6555         generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6556     CGF.Builder.CreateStore(
6557         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6558             RCG.getSharedLValue(N).getPointer(CGF), CGM.VoidPtrTy),
6559         SharedAddr, /*IsVolatile=*/false);
6560   }
6561 }
6562 
6563 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF,
6564                                               SourceLocation Loc,
6565                                               llvm::Value *ReductionsPtr,
6566                                               LValue SharedLVal) {
6567   // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
6568   // *d);
6569   llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6570                                                    CGM.IntTy,
6571                                                    /*isSigned=*/true),
6572                          ReductionsPtr,
6573                          CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6574                              SharedLVal.getPointer(CGF), CGM.VoidPtrTy)};
6575   return Address(
6576       CGF.EmitRuntimeCall(
6577           createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args),
6578       SharedLVal.getAlignment());
6579 }
6580 
6581 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
6582                                        SourceLocation Loc) {
6583   if (!CGF.HaveInsertPoint())
6584     return;
6585 
6586   llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder();
6587   if (OMPBuilder) {
6588     OMPBuilder->CreateTaskwait(CGF.Builder);
6589   } else {
6590     // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
6591     // global_tid);
6592     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
6593     // Ignore return result until untied tasks are supported.
6594     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args);
6595   }
6596 
6597   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
6598     Region->emitUntiedSwitch(CGF);
6599 }
6600 
6601 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF,
6602                                            OpenMPDirectiveKind InnerKind,
6603                                            const RegionCodeGenTy &CodeGen,
6604                                            bool HasCancel) {
6605   if (!CGF.HaveInsertPoint())
6606     return;
6607   InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel);
6608   CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
6609 }
6610 
6611 namespace {
6612 enum RTCancelKind {
6613   CancelNoreq = 0,
6614   CancelParallel = 1,
6615   CancelLoop = 2,
6616   CancelSections = 3,
6617   CancelTaskgroup = 4
6618 };
6619 } // anonymous namespace
6620 
6621 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) {
6622   RTCancelKind CancelKind = CancelNoreq;
6623   if (CancelRegion == OMPD_parallel)
6624     CancelKind = CancelParallel;
6625   else if (CancelRegion == OMPD_for)
6626     CancelKind = CancelLoop;
6627   else if (CancelRegion == OMPD_sections)
6628     CancelKind = CancelSections;
6629   else {
6630     assert(CancelRegion == OMPD_taskgroup);
6631     CancelKind = CancelTaskgroup;
6632   }
6633   return CancelKind;
6634 }
6635 
6636 void CGOpenMPRuntime::emitCancellationPointCall(
6637     CodeGenFunction &CGF, SourceLocation Loc,
6638     OpenMPDirectiveKind CancelRegion) {
6639   if (!CGF.HaveInsertPoint())
6640     return;
6641   // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
6642   // global_tid, kmp_int32 cncl_kind);
6643   if (auto *OMPRegionInfo =
6644           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6645     // For 'cancellation point taskgroup', the task region info may not have a
6646     // cancel. This may instead happen in another adjacent task.
6647     if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) {
6648       llvm::Value *Args[] = {
6649           emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
6650           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6651       // Ignore return result until untied tasks are supported.
6652       llvm::Value *Result = CGF.EmitRuntimeCall(
6653           createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args);
6654       // if (__kmpc_cancellationpoint()) {
6655       //   exit from construct;
6656       // }
6657       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6658       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6659       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6660       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6661       CGF.EmitBlock(ExitBB);
6662       // exit from construct;
6663       CodeGenFunction::JumpDest CancelDest =
6664           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6665       CGF.EmitBranchThroughCleanup(CancelDest);
6666       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6667     }
6668   }
6669 }
6670 
6671 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc,
6672                                      const Expr *IfCond,
6673                                      OpenMPDirectiveKind CancelRegion) {
6674   if (!CGF.HaveInsertPoint())
6675     return;
6676   // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
6677   // kmp_int32 cncl_kind);
6678   if (auto *OMPRegionInfo =
6679           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6680     auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF,
6681                                                         PrePostActionTy &) {
6682       CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6683       llvm::Value *Args[] = {
6684           RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc),
6685           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6686       // Ignore return result until untied tasks are supported.
6687       llvm::Value *Result = CGF.EmitRuntimeCall(
6688           RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args);
6689       // if (__kmpc_cancel()) {
6690       //   exit from construct;
6691       // }
6692       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6693       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6694       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6695       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6696       CGF.EmitBlock(ExitBB);
6697       // exit from construct;
6698       CodeGenFunction::JumpDest CancelDest =
6699           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6700       CGF.EmitBranchThroughCleanup(CancelDest);
6701       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6702     };
6703     if (IfCond) {
6704       emitIfClause(CGF, IfCond, ThenGen,
6705                    [](CodeGenFunction &, PrePostActionTy &) {});
6706     } else {
6707       RegionCodeGenTy ThenRCG(ThenGen);
6708       ThenRCG(CGF);
6709     }
6710   }
6711 }
6712 
6713 void CGOpenMPRuntime::emitTargetOutlinedFunction(
6714     const OMPExecutableDirective &D, StringRef ParentName,
6715     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6716     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6717   assert(!ParentName.empty() && "Invalid target region parent name!");
6718   HasEmittedTargetRegion = true;
6719   emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID,
6720                                    IsOffloadEntry, CodeGen);
6721 }
6722 
6723 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper(
6724     const OMPExecutableDirective &D, StringRef ParentName,
6725     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6726     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6727   // Create a unique name for the entry function using the source location
6728   // information of the current target region. The name will be something like:
6729   //
6730   // __omp_offloading_DD_FFFF_PP_lBB
6731   //
6732   // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the
6733   // mangled name of the function that encloses the target region and BB is the
6734   // line number of the target region.
6735 
6736   unsigned DeviceID;
6737   unsigned FileID;
6738   unsigned Line;
6739   getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID,
6740                            Line);
6741   SmallString<64> EntryFnName;
6742   {
6743     llvm::raw_svector_ostream OS(EntryFnName);
6744     OS << "__omp_offloading" << llvm::format("_%x", DeviceID)
6745        << llvm::format("_%x_", FileID) << ParentName << "_l" << Line;
6746   }
6747 
6748   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
6749 
6750   CodeGenFunction CGF(CGM, true);
6751   CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName);
6752   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6753 
6754   OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc());
6755 
6756   // If this target outline function is not an offload entry, we don't need to
6757   // register it.
6758   if (!IsOffloadEntry)
6759     return;
6760 
6761   // The target region ID is used by the runtime library to identify the current
6762   // target region, so it only has to be unique and not necessarily point to
6763   // anything. It could be the pointer to the outlined function that implements
6764   // the target region, but we aren't using that so that the compiler doesn't
6765   // need to keep that, and could therefore inline the host function if proven
6766   // worthwhile during optimization. In the other hand, if emitting code for the
6767   // device, the ID has to be the function address so that it can retrieved from
6768   // the offloading entry and launched by the runtime library. We also mark the
6769   // outlined function to have external linkage in case we are emitting code for
6770   // the device, because these functions will be entry points to the device.
6771 
6772   if (CGM.getLangOpts().OpenMPIsDevice) {
6773     OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy);
6774     OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
6775     OutlinedFn->setDSOLocal(false);
6776   } else {
6777     std::string Name = getName({EntryFnName, "region_id"});
6778     OutlinedFnID = new llvm::GlobalVariable(
6779         CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
6780         llvm::GlobalValue::WeakAnyLinkage,
6781         llvm::Constant::getNullValue(CGM.Int8Ty), Name);
6782   }
6783 
6784   // Register the information for the entry associated with this target region.
6785   OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
6786       DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID,
6787       OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion);
6788 }
6789 
6790 /// Checks if the expression is constant or does not have non-trivial function
6791 /// calls.
6792 static bool isTrivial(ASTContext &Ctx, const Expr * E) {
6793   // We can skip constant expressions.
6794   // We can skip expressions with trivial calls or simple expressions.
6795   return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) ||
6796           !E->hasNonTrivialCall(Ctx)) &&
6797          !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true);
6798 }
6799 
6800 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx,
6801                                                     const Stmt *Body) {
6802   const Stmt *Child = Body->IgnoreContainers();
6803   while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) {
6804     Child = nullptr;
6805     for (const Stmt *S : C->body()) {
6806       if (const auto *E = dyn_cast<Expr>(S)) {
6807         if (isTrivial(Ctx, E))
6808           continue;
6809       }
6810       // Some of the statements can be ignored.
6811       if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) ||
6812           isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S))
6813         continue;
6814       // Analyze declarations.
6815       if (const auto *DS = dyn_cast<DeclStmt>(S)) {
6816         if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) {
6817               if (isa<EmptyDecl>(D) || isa<DeclContext>(D) ||
6818                   isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) ||
6819                   isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) ||
6820                   isa<UsingDirectiveDecl>(D) ||
6821                   isa<OMPDeclareReductionDecl>(D) ||
6822                   isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D))
6823                 return true;
6824               const auto *VD = dyn_cast<VarDecl>(D);
6825               if (!VD)
6826                 return false;
6827               return VD->isConstexpr() ||
6828                      ((VD->getType().isTrivialType(Ctx) ||
6829                        VD->getType()->isReferenceType()) &&
6830                       (!VD->hasInit() || isTrivial(Ctx, VD->getInit())));
6831             }))
6832           continue;
6833       }
6834       // Found multiple children - cannot get the one child only.
6835       if (Child)
6836         return nullptr;
6837       Child = S;
6838     }
6839     if (Child)
6840       Child = Child->IgnoreContainers();
6841   }
6842   return Child;
6843 }
6844 
6845 /// Emit the number of teams for a target directive.  Inspect the num_teams
6846 /// clause associated with a teams construct combined or closely nested
6847 /// with the target directive.
6848 ///
6849 /// Emit a team of size one for directives such as 'target parallel' that
6850 /// have no associated teams construct.
6851 ///
6852 /// Otherwise, return nullptr.
6853 static llvm::Value *
6854 emitNumTeamsForTargetDirective(CodeGenFunction &CGF,
6855                                const OMPExecutableDirective &D) {
6856   assert(!CGF.getLangOpts().OpenMPIsDevice &&
6857          "Clauses associated with the teams directive expected to be emitted "
6858          "only for the host!");
6859   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6860   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6861          "Expected target-based executable directive.");
6862   CGBuilderTy &Bld = CGF.Builder;
6863   switch (DirectiveKind) {
6864   case OMPD_target: {
6865     const auto *CS = D.getInnermostCapturedStmt();
6866     const auto *Body =
6867         CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
6868     const Stmt *ChildStmt =
6869         CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body);
6870     if (const auto *NestedDir =
6871             dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
6872       if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) {
6873         if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) {
6874           CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6875           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6876           const Expr *NumTeams =
6877               NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6878           llvm::Value *NumTeamsVal =
6879               CGF.EmitScalarExpr(NumTeams,
6880                                  /*IgnoreResultAssign*/ true);
6881           return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6882                                    /*isSigned=*/true);
6883         }
6884         return Bld.getInt32(0);
6885       }
6886       if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) ||
6887           isOpenMPSimdDirective(NestedDir->getDirectiveKind()))
6888         return Bld.getInt32(1);
6889       return Bld.getInt32(0);
6890     }
6891     return nullptr;
6892   }
6893   case OMPD_target_teams:
6894   case OMPD_target_teams_distribute:
6895   case OMPD_target_teams_distribute_simd:
6896   case OMPD_target_teams_distribute_parallel_for:
6897   case OMPD_target_teams_distribute_parallel_for_simd: {
6898     if (D.hasClausesOfKind<OMPNumTeamsClause>()) {
6899       CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF);
6900       const Expr *NumTeams =
6901           D.getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6902       llvm::Value *NumTeamsVal =
6903           CGF.EmitScalarExpr(NumTeams,
6904                              /*IgnoreResultAssign*/ true);
6905       return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6906                                /*isSigned=*/true);
6907     }
6908     return Bld.getInt32(0);
6909   }
6910   case OMPD_target_parallel:
6911   case OMPD_target_parallel_for:
6912   case OMPD_target_parallel_for_simd:
6913   case OMPD_target_simd:
6914     return Bld.getInt32(1);
6915   case OMPD_parallel:
6916   case OMPD_for:
6917   case OMPD_parallel_for:
6918   case OMPD_parallel_master:
6919   case OMPD_parallel_sections:
6920   case OMPD_for_simd:
6921   case OMPD_parallel_for_simd:
6922   case OMPD_cancel:
6923   case OMPD_cancellation_point:
6924   case OMPD_ordered:
6925   case OMPD_threadprivate:
6926   case OMPD_allocate:
6927   case OMPD_task:
6928   case OMPD_simd:
6929   case OMPD_sections:
6930   case OMPD_section:
6931   case OMPD_single:
6932   case OMPD_master:
6933   case OMPD_critical:
6934   case OMPD_taskyield:
6935   case OMPD_barrier:
6936   case OMPD_taskwait:
6937   case OMPD_taskgroup:
6938   case OMPD_atomic:
6939   case OMPD_flush:
6940   case OMPD_depobj:
6941   case OMPD_scan:
6942   case OMPD_teams:
6943   case OMPD_target_data:
6944   case OMPD_target_exit_data:
6945   case OMPD_target_enter_data:
6946   case OMPD_distribute:
6947   case OMPD_distribute_simd:
6948   case OMPD_distribute_parallel_for:
6949   case OMPD_distribute_parallel_for_simd:
6950   case OMPD_teams_distribute:
6951   case OMPD_teams_distribute_simd:
6952   case OMPD_teams_distribute_parallel_for:
6953   case OMPD_teams_distribute_parallel_for_simd:
6954   case OMPD_target_update:
6955   case OMPD_declare_simd:
6956   case OMPD_declare_variant:
6957   case OMPD_begin_declare_variant:
6958   case OMPD_end_declare_variant:
6959   case OMPD_declare_target:
6960   case OMPD_end_declare_target:
6961   case OMPD_declare_reduction:
6962   case OMPD_declare_mapper:
6963   case OMPD_taskloop:
6964   case OMPD_taskloop_simd:
6965   case OMPD_master_taskloop:
6966   case OMPD_master_taskloop_simd:
6967   case OMPD_parallel_master_taskloop:
6968   case OMPD_parallel_master_taskloop_simd:
6969   case OMPD_requires:
6970   case OMPD_unknown:
6971     break;
6972   }
6973   llvm_unreachable("Unexpected directive kind.");
6974 }
6975 
6976 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS,
6977                                   llvm::Value *DefaultThreadLimitVal) {
6978   const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6979       CGF.getContext(), CS->getCapturedStmt());
6980   if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6981     if (isOpenMPParallelDirective(Dir->getDirectiveKind())) {
6982       llvm::Value *NumThreads = nullptr;
6983       llvm::Value *CondVal = nullptr;
6984       // Handle if clause. If if clause present, the number of threads is
6985       // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
6986       if (Dir->hasClausesOfKind<OMPIfClause>()) {
6987         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6988         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6989         const OMPIfClause *IfClause = nullptr;
6990         for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) {
6991           if (C->getNameModifier() == OMPD_unknown ||
6992               C->getNameModifier() == OMPD_parallel) {
6993             IfClause = C;
6994             break;
6995           }
6996         }
6997         if (IfClause) {
6998           const Expr *Cond = IfClause->getCondition();
6999           bool Result;
7000           if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
7001             if (!Result)
7002               return CGF.Builder.getInt32(1);
7003           } else {
7004             CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange());
7005             if (const auto *PreInit =
7006                     cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) {
7007               for (const auto *I : PreInit->decls()) {
7008                 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7009                   CGF.EmitVarDecl(cast<VarDecl>(*I));
7010                 } else {
7011                   CodeGenFunction::AutoVarEmission Emission =
7012                       CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
7013                   CGF.EmitAutoVarCleanups(Emission);
7014                 }
7015               }
7016             }
7017             CondVal = CGF.EvaluateExprAsBool(Cond);
7018           }
7019         }
7020       }
7021       // Check the value of num_threads clause iff if clause was not specified
7022       // or is not evaluated to false.
7023       if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) {
7024         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
7025         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
7026         const auto *NumThreadsClause =
7027             Dir->getSingleClause<OMPNumThreadsClause>();
7028         CodeGenFunction::LexicalScope Scope(
7029             CGF, NumThreadsClause->getNumThreads()->getSourceRange());
7030         if (const auto *PreInit =
7031                 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) {
7032           for (const auto *I : PreInit->decls()) {
7033             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7034               CGF.EmitVarDecl(cast<VarDecl>(*I));
7035             } else {
7036               CodeGenFunction::AutoVarEmission Emission =
7037                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
7038               CGF.EmitAutoVarCleanups(Emission);
7039             }
7040           }
7041         }
7042         NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads());
7043         NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty,
7044                                                /*isSigned=*/false);
7045         if (DefaultThreadLimitVal)
7046           NumThreads = CGF.Builder.CreateSelect(
7047               CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads),
7048               DefaultThreadLimitVal, NumThreads);
7049       } else {
7050         NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal
7051                                            : CGF.Builder.getInt32(0);
7052       }
7053       // Process condition of the if clause.
7054       if (CondVal) {
7055         NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads,
7056                                               CGF.Builder.getInt32(1));
7057       }
7058       return NumThreads;
7059     }
7060     if (isOpenMPSimdDirective(Dir->getDirectiveKind()))
7061       return CGF.Builder.getInt32(1);
7062     return DefaultThreadLimitVal;
7063   }
7064   return DefaultThreadLimitVal ? DefaultThreadLimitVal
7065                                : CGF.Builder.getInt32(0);
7066 }
7067 
7068 /// Emit the number of threads for a target directive.  Inspect the
7069 /// thread_limit clause associated with a teams construct combined or closely
7070 /// nested with the target directive.
7071 ///
7072 /// Emit the num_threads clause for directives such as 'target parallel' that
7073 /// have no associated teams construct.
7074 ///
7075 /// Otherwise, return nullptr.
7076 static llvm::Value *
7077 emitNumThreadsForTargetDirective(CodeGenFunction &CGF,
7078                                  const OMPExecutableDirective &D) {
7079   assert(!CGF.getLangOpts().OpenMPIsDevice &&
7080          "Clauses associated with the teams directive expected to be emitted "
7081          "only for the host!");
7082   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
7083   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
7084          "Expected target-based executable directive.");
7085   CGBuilderTy &Bld = CGF.Builder;
7086   llvm::Value *ThreadLimitVal = nullptr;
7087   llvm::Value *NumThreadsVal = nullptr;
7088   switch (DirectiveKind) {
7089   case OMPD_target: {
7090     const CapturedStmt *CS = D.getInnermostCapturedStmt();
7091     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7092       return NumThreads;
7093     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7094         CGF.getContext(), CS->getCapturedStmt());
7095     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7096       if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) {
7097         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
7098         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
7099         const auto *ThreadLimitClause =
7100             Dir->getSingleClause<OMPThreadLimitClause>();
7101         CodeGenFunction::LexicalScope Scope(
7102             CGF, ThreadLimitClause->getThreadLimit()->getSourceRange());
7103         if (const auto *PreInit =
7104                 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) {
7105           for (const auto *I : PreInit->decls()) {
7106             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7107               CGF.EmitVarDecl(cast<VarDecl>(*I));
7108             } else {
7109               CodeGenFunction::AutoVarEmission Emission =
7110                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
7111               CGF.EmitAutoVarCleanups(Emission);
7112             }
7113           }
7114         }
7115         llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7116             ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7117         ThreadLimitVal =
7118             Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7119       }
7120       if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) &&
7121           !isOpenMPDistributeDirective(Dir->getDirectiveKind())) {
7122         CS = Dir->getInnermostCapturedStmt();
7123         const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7124             CGF.getContext(), CS->getCapturedStmt());
7125         Dir = dyn_cast_or_null<OMPExecutableDirective>(Child);
7126       }
7127       if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) &&
7128           !isOpenMPSimdDirective(Dir->getDirectiveKind())) {
7129         CS = Dir->getInnermostCapturedStmt();
7130         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7131           return NumThreads;
7132       }
7133       if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind()))
7134         return Bld.getInt32(1);
7135     }
7136     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7137   }
7138   case OMPD_target_teams: {
7139     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7140       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7141       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7142       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7143           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7144       ThreadLimitVal =
7145           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7146     }
7147     const CapturedStmt *CS = D.getInnermostCapturedStmt();
7148     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7149       return NumThreads;
7150     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7151         CGF.getContext(), CS->getCapturedStmt());
7152     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7153       if (Dir->getDirectiveKind() == OMPD_distribute) {
7154         CS = Dir->getInnermostCapturedStmt();
7155         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7156           return NumThreads;
7157       }
7158     }
7159     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7160   }
7161   case OMPD_target_teams_distribute:
7162     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7163       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7164       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7165       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7166           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7167       ThreadLimitVal =
7168           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7169     }
7170     return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal);
7171   case OMPD_target_parallel:
7172   case OMPD_target_parallel_for:
7173   case OMPD_target_parallel_for_simd:
7174   case OMPD_target_teams_distribute_parallel_for:
7175   case OMPD_target_teams_distribute_parallel_for_simd: {
7176     llvm::Value *CondVal = nullptr;
7177     // Handle if clause. If if clause present, the number of threads is
7178     // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
7179     if (D.hasClausesOfKind<OMPIfClause>()) {
7180       const OMPIfClause *IfClause = nullptr;
7181       for (const auto *C : D.getClausesOfKind<OMPIfClause>()) {
7182         if (C->getNameModifier() == OMPD_unknown ||
7183             C->getNameModifier() == OMPD_parallel) {
7184           IfClause = C;
7185           break;
7186         }
7187       }
7188       if (IfClause) {
7189         const Expr *Cond = IfClause->getCondition();
7190         bool Result;
7191         if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
7192           if (!Result)
7193             return Bld.getInt32(1);
7194         } else {
7195           CodeGenFunction::RunCleanupsScope Scope(CGF);
7196           CondVal = CGF.EvaluateExprAsBool(Cond);
7197         }
7198       }
7199     }
7200     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7201       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7202       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7203       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7204           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7205       ThreadLimitVal =
7206           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7207     }
7208     if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
7209       CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
7210       const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
7211       llvm::Value *NumThreads = CGF.EmitScalarExpr(
7212           NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true);
7213       NumThreadsVal =
7214           Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false);
7215       ThreadLimitVal = ThreadLimitVal
7216                            ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal,
7217                                                                 ThreadLimitVal),
7218                                               NumThreadsVal, ThreadLimitVal)
7219                            : NumThreadsVal;
7220     }
7221     if (!ThreadLimitVal)
7222       ThreadLimitVal = Bld.getInt32(0);
7223     if (CondVal)
7224       return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1));
7225     return ThreadLimitVal;
7226   }
7227   case OMPD_target_teams_distribute_simd:
7228   case OMPD_target_simd:
7229     return Bld.getInt32(1);
7230   case OMPD_parallel:
7231   case OMPD_for:
7232   case OMPD_parallel_for:
7233   case OMPD_parallel_master:
7234   case OMPD_parallel_sections:
7235   case OMPD_for_simd:
7236   case OMPD_parallel_for_simd:
7237   case OMPD_cancel:
7238   case OMPD_cancellation_point:
7239   case OMPD_ordered:
7240   case OMPD_threadprivate:
7241   case OMPD_allocate:
7242   case OMPD_task:
7243   case OMPD_simd:
7244   case OMPD_sections:
7245   case OMPD_section:
7246   case OMPD_single:
7247   case OMPD_master:
7248   case OMPD_critical:
7249   case OMPD_taskyield:
7250   case OMPD_barrier:
7251   case OMPD_taskwait:
7252   case OMPD_taskgroup:
7253   case OMPD_atomic:
7254   case OMPD_flush:
7255   case OMPD_depobj:
7256   case OMPD_scan:
7257   case OMPD_teams:
7258   case OMPD_target_data:
7259   case OMPD_target_exit_data:
7260   case OMPD_target_enter_data:
7261   case OMPD_distribute:
7262   case OMPD_distribute_simd:
7263   case OMPD_distribute_parallel_for:
7264   case OMPD_distribute_parallel_for_simd:
7265   case OMPD_teams_distribute:
7266   case OMPD_teams_distribute_simd:
7267   case OMPD_teams_distribute_parallel_for:
7268   case OMPD_teams_distribute_parallel_for_simd:
7269   case OMPD_target_update:
7270   case OMPD_declare_simd:
7271   case OMPD_declare_variant:
7272   case OMPD_begin_declare_variant:
7273   case OMPD_end_declare_variant:
7274   case OMPD_declare_target:
7275   case OMPD_end_declare_target:
7276   case OMPD_declare_reduction:
7277   case OMPD_declare_mapper:
7278   case OMPD_taskloop:
7279   case OMPD_taskloop_simd:
7280   case OMPD_master_taskloop:
7281   case OMPD_master_taskloop_simd:
7282   case OMPD_parallel_master_taskloop:
7283   case OMPD_parallel_master_taskloop_simd:
7284   case OMPD_requires:
7285   case OMPD_unknown:
7286     break;
7287   }
7288   llvm_unreachable("Unsupported directive kind.");
7289 }
7290 
7291 namespace {
7292 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
7293 
7294 // Utility to handle information from clauses associated with a given
7295 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause).
7296 // It provides a convenient interface to obtain the information and generate
7297 // code for that information.
7298 class MappableExprsHandler {
7299 public:
7300   /// Values for bit flags used to specify the mapping type for
7301   /// offloading.
7302   enum OpenMPOffloadMappingFlags : uint64_t {
7303     /// No flags
7304     OMP_MAP_NONE = 0x0,
7305     /// Allocate memory on the device and move data from host to device.
7306     OMP_MAP_TO = 0x01,
7307     /// Allocate memory on the device and move data from device to host.
7308     OMP_MAP_FROM = 0x02,
7309     /// Always perform the requested mapping action on the element, even
7310     /// if it was already mapped before.
7311     OMP_MAP_ALWAYS = 0x04,
7312     /// Delete the element from the device environment, ignoring the
7313     /// current reference count associated with the element.
7314     OMP_MAP_DELETE = 0x08,
7315     /// The element being mapped is a pointer-pointee pair; both the
7316     /// pointer and the pointee should be mapped.
7317     OMP_MAP_PTR_AND_OBJ = 0x10,
7318     /// This flags signals that the base address of an entry should be
7319     /// passed to the target kernel as an argument.
7320     OMP_MAP_TARGET_PARAM = 0x20,
7321     /// Signal that the runtime library has to return the device pointer
7322     /// in the current position for the data being mapped. Used when we have the
7323     /// use_device_ptr clause.
7324     OMP_MAP_RETURN_PARAM = 0x40,
7325     /// This flag signals that the reference being passed is a pointer to
7326     /// private data.
7327     OMP_MAP_PRIVATE = 0x80,
7328     /// Pass the element to the device by value.
7329     OMP_MAP_LITERAL = 0x100,
7330     /// Implicit map
7331     OMP_MAP_IMPLICIT = 0x200,
7332     /// Close is a hint to the runtime to allocate memory close to
7333     /// the target device.
7334     OMP_MAP_CLOSE = 0x400,
7335     /// The 16 MSBs of the flags indicate whether the entry is member of some
7336     /// struct/class.
7337     OMP_MAP_MEMBER_OF = 0xffff000000000000,
7338     LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF),
7339   };
7340 
7341   /// Get the offset of the OMP_MAP_MEMBER_OF field.
7342   static unsigned getFlagMemberOffset() {
7343     unsigned Offset = 0;
7344     for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1);
7345          Remain = Remain >> 1)
7346       Offset++;
7347     return Offset;
7348   }
7349 
7350   /// Class that associates information with a base pointer to be passed to the
7351   /// runtime library.
7352   class BasePointerInfo {
7353     /// The base pointer.
7354     llvm::Value *Ptr = nullptr;
7355     /// The base declaration that refers to this device pointer, or null if
7356     /// there is none.
7357     const ValueDecl *DevPtrDecl = nullptr;
7358 
7359   public:
7360     BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr)
7361         : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {}
7362     llvm::Value *operator*() const { return Ptr; }
7363     const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; }
7364     void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; }
7365   };
7366 
7367   using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>;
7368   using MapValuesArrayTy = SmallVector<llvm::Value *, 4>;
7369   using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>;
7370 
7371   /// Map between a struct and the its lowest & highest elements which have been
7372   /// mapped.
7373   /// [ValueDecl *] --> {LE(FieldIndex, Pointer),
7374   ///                    HE(FieldIndex, Pointer)}
7375   struct StructRangeInfoTy {
7376     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = {
7377         0, Address::invalid()};
7378     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = {
7379         0, Address::invalid()};
7380     Address Base = Address::invalid();
7381   };
7382 
7383 private:
7384   /// Kind that defines how a device pointer has to be returned.
7385   struct MapInfo {
7386     OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
7387     OpenMPMapClauseKind MapType = OMPC_MAP_unknown;
7388     ArrayRef<OpenMPMapModifierKind> MapModifiers;
7389     bool ReturnDevicePointer = false;
7390     bool IsImplicit = false;
7391 
7392     MapInfo() = default;
7393     MapInfo(
7394         OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7395         OpenMPMapClauseKind MapType,
7396         ArrayRef<OpenMPMapModifierKind> MapModifiers,
7397         bool ReturnDevicePointer, bool IsImplicit)
7398         : Components(Components), MapType(MapType), MapModifiers(MapModifiers),
7399           ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {}
7400   };
7401 
7402   /// If use_device_ptr is used on a pointer which is a struct member and there
7403   /// is no map information about it, then emission of that entry is deferred
7404   /// until the whole struct has been processed.
7405   struct DeferredDevicePtrEntryTy {
7406     const Expr *IE = nullptr;
7407     const ValueDecl *VD = nullptr;
7408 
7409     DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD)
7410         : IE(IE), VD(VD) {}
7411   };
7412 
7413   /// The target directive from where the mappable clauses were extracted. It
7414   /// is either a executable directive or a user-defined mapper directive.
7415   llvm::PointerUnion<const OMPExecutableDirective *,
7416                      const OMPDeclareMapperDecl *>
7417       CurDir;
7418 
7419   /// Function the directive is being generated for.
7420   CodeGenFunction &CGF;
7421 
7422   /// Set of all first private variables in the current directive.
7423   /// bool data is set to true if the variable is implicitly marked as
7424   /// firstprivate, false otherwise.
7425   llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls;
7426 
7427   /// Map between device pointer declarations and their expression components.
7428   /// The key value for declarations in 'this' is null.
7429   llvm::DenseMap<
7430       const ValueDecl *,
7431       SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7432       DevPointersMap;
7433 
7434   llvm::Value *getExprTypeSize(const Expr *E) const {
7435     QualType ExprTy = E->getType().getCanonicalType();
7436 
7437     // Reference types are ignored for mapping purposes.
7438     if (const auto *RefTy = ExprTy->getAs<ReferenceType>())
7439       ExprTy = RefTy->getPointeeType().getCanonicalType();
7440 
7441     // Given that an array section is considered a built-in type, we need to
7442     // do the calculation based on the length of the section instead of relying
7443     // on CGF.getTypeSize(E->getType()).
7444     if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) {
7445       QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType(
7446                             OAE->getBase()->IgnoreParenImpCasts())
7447                             .getCanonicalType();
7448 
7449       // If there is no length associated with the expression and lower bound is
7450       // not specified too, that means we are using the whole length of the
7451       // base.
7452       if (!OAE->getLength() && OAE->getColonLoc().isValid() &&
7453           !OAE->getLowerBound())
7454         return CGF.getTypeSize(BaseTy);
7455 
7456       llvm::Value *ElemSize;
7457       if (const auto *PTy = BaseTy->getAs<PointerType>()) {
7458         ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType());
7459       } else {
7460         const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr());
7461         assert(ATy && "Expecting array type if not a pointer type.");
7462         ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType());
7463       }
7464 
7465       // If we don't have a length at this point, that is because we have an
7466       // array section with a single element.
7467       if (!OAE->getLength() && OAE->getColonLoc().isInvalid())
7468         return ElemSize;
7469 
7470       if (const Expr *LenExpr = OAE->getLength()) {
7471         llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr);
7472         LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(),
7473                                              CGF.getContext().getSizeType(),
7474                                              LenExpr->getExprLoc());
7475         return CGF.Builder.CreateNUWMul(LengthVal, ElemSize);
7476       }
7477       assert(!OAE->getLength() && OAE->getColonLoc().isValid() &&
7478              OAE->getLowerBound() && "expected array_section[lb:].");
7479       // Size = sizetype - lb * elemtype;
7480       llvm::Value *LengthVal = CGF.getTypeSize(BaseTy);
7481       llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound());
7482       LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(),
7483                                        CGF.getContext().getSizeType(),
7484                                        OAE->getLowerBound()->getExprLoc());
7485       LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize);
7486       llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal);
7487       llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal);
7488       LengthVal = CGF.Builder.CreateSelect(
7489           Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0));
7490       return LengthVal;
7491     }
7492     return CGF.getTypeSize(ExprTy);
7493   }
7494 
7495   /// Return the corresponding bits for a given map clause modifier. Add
7496   /// a flag marking the map as a pointer if requested. Add a flag marking the
7497   /// map as the first one of a series of maps that relate to the same map
7498   /// expression.
7499   OpenMPOffloadMappingFlags getMapTypeBits(
7500       OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7501       bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const {
7502     OpenMPOffloadMappingFlags Bits =
7503         IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE;
7504     switch (MapType) {
7505     case OMPC_MAP_alloc:
7506     case OMPC_MAP_release:
7507       // alloc and release is the default behavior in the runtime library,  i.e.
7508       // if we don't pass any bits alloc/release that is what the runtime is
7509       // going to do. Therefore, we don't need to signal anything for these two
7510       // type modifiers.
7511       break;
7512     case OMPC_MAP_to:
7513       Bits |= OMP_MAP_TO;
7514       break;
7515     case OMPC_MAP_from:
7516       Bits |= OMP_MAP_FROM;
7517       break;
7518     case OMPC_MAP_tofrom:
7519       Bits |= OMP_MAP_TO | OMP_MAP_FROM;
7520       break;
7521     case OMPC_MAP_delete:
7522       Bits |= OMP_MAP_DELETE;
7523       break;
7524     case OMPC_MAP_unknown:
7525       llvm_unreachable("Unexpected map type!");
7526     }
7527     if (AddPtrFlag)
7528       Bits |= OMP_MAP_PTR_AND_OBJ;
7529     if (AddIsTargetParamFlag)
7530       Bits |= OMP_MAP_TARGET_PARAM;
7531     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always)
7532         != MapModifiers.end())
7533       Bits |= OMP_MAP_ALWAYS;
7534     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close)
7535         != MapModifiers.end())
7536       Bits |= OMP_MAP_CLOSE;
7537     return Bits;
7538   }
7539 
7540   /// Return true if the provided expression is a final array section. A
7541   /// final array section, is one whose length can't be proved to be one.
7542   bool isFinalArraySectionExpression(const Expr *E) const {
7543     const auto *OASE = dyn_cast<OMPArraySectionExpr>(E);
7544 
7545     // It is not an array section and therefore not a unity-size one.
7546     if (!OASE)
7547       return false;
7548 
7549     // An array section with no colon always refer to a single element.
7550     if (OASE->getColonLoc().isInvalid())
7551       return false;
7552 
7553     const Expr *Length = OASE->getLength();
7554 
7555     // If we don't have a length we have to check if the array has size 1
7556     // for this dimension. Also, we should always expect a length if the
7557     // base type is pointer.
7558     if (!Length) {
7559       QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType(
7560                              OASE->getBase()->IgnoreParenImpCasts())
7561                              .getCanonicalType();
7562       if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr()))
7563         return ATy->getSize().getSExtValue() != 1;
7564       // If we don't have a constant dimension length, we have to consider
7565       // the current section as having any size, so it is not necessarily
7566       // unitary. If it happen to be unity size, that's user fault.
7567       return true;
7568     }
7569 
7570     // Check if the length evaluates to 1.
7571     Expr::EvalResult Result;
7572     if (!Length->EvaluateAsInt(Result, CGF.getContext()))
7573       return true; // Can have more that size 1.
7574 
7575     llvm::APSInt ConstLength = Result.Val.getInt();
7576     return ConstLength.getSExtValue() != 1;
7577   }
7578 
7579   /// Generate the base pointers, section pointers, sizes and map type
7580   /// bits for the provided map type, map modifier, and expression components.
7581   /// \a IsFirstComponent should be set to true if the provided set of
7582   /// components is the first associated with a capture.
7583   void generateInfoForComponentList(
7584       OpenMPMapClauseKind MapType,
7585       ArrayRef<OpenMPMapModifierKind> MapModifiers,
7586       OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7587       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
7588       MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
7589       StructRangeInfoTy &PartialStruct, bool IsFirstComponentList,
7590       bool IsImplicit,
7591       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
7592           OverlappedElements = llvm::None) const {
7593     // The following summarizes what has to be generated for each map and the
7594     // types below. The generated information is expressed in this order:
7595     // base pointer, section pointer, size, flags
7596     // (to add to the ones that come from the map type and modifier).
7597     //
7598     // double d;
7599     // int i[100];
7600     // float *p;
7601     //
7602     // struct S1 {
7603     //   int i;
7604     //   float f[50];
7605     // }
7606     // struct S2 {
7607     //   int i;
7608     //   float f[50];
7609     //   S1 s;
7610     //   double *p;
7611     //   struct S2 *ps;
7612     // }
7613     // S2 s;
7614     // S2 *ps;
7615     //
7616     // map(d)
7617     // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM
7618     //
7619     // map(i)
7620     // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM
7621     //
7622     // map(i[1:23])
7623     // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM
7624     //
7625     // map(p)
7626     // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
7627     //
7628     // map(p[1:24])
7629     // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM
7630     //
7631     // map(s)
7632     // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM
7633     //
7634     // map(s.i)
7635     // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM
7636     //
7637     // map(s.s.f)
7638     // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7639     //
7640     // map(s.p)
7641     // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM
7642     //
7643     // map(to: s.p[:22])
7644     // &s, &(s.p), sizeof(double*), TARGET_PARAM (*)
7645     // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**)
7646     // &(s.p), &(s.p[0]), 22*sizeof(double),
7647     //   MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
7648     // (*) alloc space for struct members, only this is a target parameter
7649     // (**) map the pointer (nothing to be mapped in this example) (the compiler
7650     //      optimizes this entry out, same in the examples below)
7651     // (***) map the pointee (map: to)
7652     //
7653     // map(s.ps)
7654     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7655     //
7656     // map(from: s.ps->s.i)
7657     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7658     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7659     // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ  | FROM
7660     //
7661     // map(to: s.ps->ps)
7662     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7663     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7664     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ  | TO
7665     //
7666     // map(s.ps->ps->ps)
7667     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7668     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7669     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7670     // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7671     //
7672     // map(to: s.ps->ps->s.f[:22])
7673     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7674     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7675     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7676     // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7677     //
7678     // map(ps)
7679     // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM
7680     //
7681     // map(ps->i)
7682     // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM
7683     //
7684     // map(ps->s.f)
7685     // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7686     //
7687     // map(from: ps->p)
7688     // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM
7689     //
7690     // map(to: ps->p[:22])
7691     // ps, &(ps->p), sizeof(double*), TARGET_PARAM
7692     // ps, &(ps->p), sizeof(double*), MEMBER_OF(1)
7693     // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO
7694     //
7695     // map(ps->ps)
7696     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7697     //
7698     // map(from: ps->ps->s.i)
7699     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7700     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7701     // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7702     //
7703     // map(from: ps->ps->ps)
7704     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7705     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7706     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7707     //
7708     // map(ps->ps->ps->ps)
7709     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7710     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7711     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7712     // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7713     //
7714     // map(to: ps->ps->ps->s.f[:22])
7715     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7716     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7717     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7718     // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7719     //
7720     // map(to: s.f[:22]) map(from: s.p[:33])
7721     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) +
7722     //     sizeof(double*) (**), TARGET_PARAM
7723     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO
7724     // &s, &(s.p), sizeof(double*), MEMBER_OF(1)
7725     // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7726     // (*) allocate contiguous space needed to fit all mapped members even if
7727     //     we allocate space for members not mapped (in this example,
7728     //     s.f[22..49] and s.s are not mapped, yet we must allocate space for
7729     //     them as well because they fall between &s.f[0] and &s.p)
7730     //
7731     // map(from: s.f[:22]) map(to: ps->p[:33])
7732     // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM
7733     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7734     // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*)
7735     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO
7736     // (*) the struct this entry pertains to is the 2nd element in the list of
7737     //     arguments, hence MEMBER_OF(2)
7738     //
7739     // map(from: s.f[:22], s.s) map(to: ps->p[:33])
7740     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM
7741     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM
7742     // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM
7743     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7744     // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*)
7745     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO
7746     // (*) the struct this entry pertains to is the 4th element in the list
7747     //     of arguments, hence MEMBER_OF(4)
7748 
7749     // Track if the map information being generated is the first for a capture.
7750     bool IsCaptureFirstInfo = IsFirstComponentList;
7751     // When the variable is on a declare target link or in a to clause with
7752     // unified memory, a reference is needed to hold the host/device address
7753     // of the variable.
7754     bool RequiresReference = false;
7755 
7756     // Scan the components from the base to the complete expression.
7757     auto CI = Components.rbegin();
7758     auto CE = Components.rend();
7759     auto I = CI;
7760 
7761     // Track if the map information being generated is the first for a list of
7762     // components.
7763     bool IsExpressionFirstInfo = true;
7764     Address BP = Address::invalid();
7765     const Expr *AssocExpr = I->getAssociatedExpression();
7766     const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr);
7767     const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr);
7768 
7769     if (isa<MemberExpr>(AssocExpr)) {
7770       // The base is the 'this' pointer. The content of the pointer is going
7771       // to be the base of the field being mapped.
7772       BP = CGF.LoadCXXThisAddress();
7773     } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) ||
7774                (OASE &&
7775                 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) {
7776       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7777     } else {
7778       // The base is the reference to the variable.
7779       // BP = &Var.
7780       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7781       if (const auto *VD =
7782               dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) {
7783         if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
7784                 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) {
7785           if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
7786               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
7787                CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) {
7788             RequiresReference = true;
7789             BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
7790           }
7791         }
7792       }
7793 
7794       // If the variable is a pointer and is being dereferenced (i.e. is not
7795       // the last component), the base has to be the pointer itself, not its
7796       // reference. References are ignored for mapping purposes.
7797       QualType Ty =
7798           I->getAssociatedDeclaration()->getType().getNonReferenceType();
7799       if (Ty->isAnyPointerType() && std::next(I) != CE) {
7800         BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
7801 
7802         // We do not need to generate individual map information for the
7803         // pointer, it can be associated with the combined storage.
7804         ++I;
7805       }
7806     }
7807 
7808     // Track whether a component of the list should be marked as MEMBER_OF some
7809     // combined entry (for partial structs). Only the first PTR_AND_OBJ entry
7810     // in a component list should be marked as MEMBER_OF, all subsequent entries
7811     // do not belong to the base struct. E.g.
7812     // struct S2 s;
7813     // s.ps->ps->ps->f[:]
7814     //   (1) (2) (3) (4)
7815     // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a
7816     // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3)
7817     // is the pointee of ps(2) which is not member of struct s, so it should not
7818     // be marked as such (it is still PTR_AND_OBJ).
7819     // The variable is initialized to false so that PTR_AND_OBJ entries which
7820     // are not struct members are not considered (e.g. array of pointers to
7821     // data).
7822     bool ShouldBeMemberOf = false;
7823 
7824     // Variable keeping track of whether or not we have encountered a component
7825     // in the component list which is a member expression. Useful when we have a
7826     // pointer or a final array section, in which case it is the previous
7827     // component in the list which tells us whether we have a member expression.
7828     // E.g. X.f[:]
7829     // While processing the final array section "[:]" it is "f" which tells us
7830     // whether we are dealing with a member of a declared struct.
7831     const MemberExpr *EncounteredME = nullptr;
7832 
7833     for (; I != CE; ++I) {
7834       // If the current component is member of a struct (parent struct) mark it.
7835       if (!EncounteredME) {
7836         EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression());
7837         // If we encounter a PTR_AND_OBJ entry from now on it should be marked
7838         // as MEMBER_OF the parent struct.
7839         if (EncounteredME)
7840           ShouldBeMemberOf = true;
7841       }
7842 
7843       auto Next = std::next(I);
7844 
7845       // We need to generate the addresses and sizes if this is the last
7846       // component, if the component is a pointer or if it is an array section
7847       // whose length can't be proved to be one. If this is a pointer, it
7848       // becomes the base address for the following components.
7849 
7850       // A final array section, is one whose length can't be proved to be one.
7851       bool IsFinalArraySection =
7852           isFinalArraySectionExpression(I->getAssociatedExpression());
7853 
7854       // Get information on whether the element is a pointer. Have to do a
7855       // special treatment for array sections given that they are built-in
7856       // types.
7857       const auto *OASE =
7858           dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression());
7859       const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression());
7860       const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression());
7861       bool IsPointer =
7862           (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE)
7863                        .getCanonicalType()
7864                        ->isAnyPointerType()) ||
7865           I->getAssociatedExpression()->getType()->isAnyPointerType();
7866       bool IsNonDerefPointer = IsPointer && !UO && !BO;
7867 
7868       if (Next == CE || IsNonDerefPointer || IsFinalArraySection) {
7869         // If this is not the last component, we expect the pointer to be
7870         // associated with an array expression or member expression.
7871         assert((Next == CE ||
7872                 isa<MemberExpr>(Next->getAssociatedExpression()) ||
7873                 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) ||
7874                 isa<OMPArraySectionExpr>(Next->getAssociatedExpression()) ||
7875                 isa<UnaryOperator>(Next->getAssociatedExpression()) ||
7876                 isa<BinaryOperator>(Next->getAssociatedExpression())) &&
7877                "Unexpected expression");
7878 
7879         Address LB = CGF.EmitOMPSharedLValue(I->getAssociatedExpression())
7880                          .getAddress(CGF);
7881 
7882         // If this component is a pointer inside the base struct then we don't
7883         // need to create any entry for it - it will be combined with the object
7884         // it is pointing to into a single PTR_AND_OBJ entry.
7885         bool IsMemberPointer =
7886             IsPointer && EncounteredME &&
7887             (dyn_cast<MemberExpr>(I->getAssociatedExpression()) ==
7888              EncounteredME);
7889         if (!OverlappedElements.empty()) {
7890           // Handle base element with the info for overlapped elements.
7891           assert(!PartialStruct.Base.isValid() && "The base element is set.");
7892           assert(Next == CE &&
7893                  "Expected last element for the overlapped elements.");
7894           assert(!IsPointer &&
7895                  "Unexpected base element with the pointer type.");
7896           // Mark the whole struct as the struct that requires allocation on the
7897           // device.
7898           PartialStruct.LowestElem = {0, LB};
7899           CharUnits TypeSize = CGF.getContext().getTypeSizeInChars(
7900               I->getAssociatedExpression()->getType());
7901           Address HB = CGF.Builder.CreateConstGEP(
7902               CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB,
7903                                                               CGF.VoidPtrTy),
7904               TypeSize.getQuantity() - 1);
7905           PartialStruct.HighestElem = {
7906               std::numeric_limits<decltype(
7907                   PartialStruct.HighestElem.first)>::max(),
7908               HB};
7909           PartialStruct.Base = BP;
7910           // Emit data for non-overlapped data.
7911           OpenMPOffloadMappingFlags Flags =
7912               OMP_MAP_MEMBER_OF |
7913               getMapTypeBits(MapType, MapModifiers, IsImplicit,
7914                              /*AddPtrFlag=*/false,
7915                              /*AddIsTargetParamFlag=*/false);
7916           LB = BP;
7917           llvm::Value *Size = nullptr;
7918           // Do bitcopy of all non-overlapped structure elements.
7919           for (OMPClauseMappableExprCommon::MappableExprComponentListRef
7920                    Component : OverlappedElements) {
7921             Address ComponentLB = Address::invalid();
7922             for (const OMPClauseMappableExprCommon::MappableComponent &MC :
7923                  Component) {
7924               if (MC.getAssociatedDeclaration()) {
7925                 ComponentLB =
7926                     CGF.EmitOMPSharedLValue(MC.getAssociatedExpression())
7927                         .getAddress(CGF);
7928                 Size = CGF.Builder.CreatePtrDiff(
7929                     CGF.EmitCastToVoidPtr(ComponentLB.getPointer()),
7930                     CGF.EmitCastToVoidPtr(LB.getPointer()));
7931                 break;
7932               }
7933             }
7934             BasePointers.push_back(BP.getPointer());
7935             Pointers.push_back(LB.getPointer());
7936             Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty,
7937                                                       /*isSigned=*/true));
7938             Types.push_back(Flags);
7939             LB = CGF.Builder.CreateConstGEP(ComponentLB, 1);
7940           }
7941           BasePointers.push_back(BP.getPointer());
7942           Pointers.push_back(LB.getPointer());
7943           Size = CGF.Builder.CreatePtrDiff(
7944               CGF.EmitCastToVoidPtr(
7945                   CGF.Builder.CreateConstGEP(HB, 1).getPointer()),
7946               CGF.EmitCastToVoidPtr(LB.getPointer()));
7947           Sizes.push_back(
7948               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7949           Types.push_back(Flags);
7950           break;
7951         }
7952         llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression());
7953         if (!IsMemberPointer) {
7954           BasePointers.push_back(BP.getPointer());
7955           Pointers.push_back(LB.getPointer());
7956           Sizes.push_back(
7957               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7958 
7959           // We need to add a pointer flag for each map that comes from the
7960           // same expression except for the first one. We also need to signal
7961           // this map is the first one that relates with the current capture
7962           // (there is a set of entries for each capture).
7963           OpenMPOffloadMappingFlags Flags = getMapTypeBits(
7964               MapType, MapModifiers, IsImplicit,
7965               !IsExpressionFirstInfo || RequiresReference,
7966               IsCaptureFirstInfo && !RequiresReference);
7967 
7968           if (!IsExpressionFirstInfo) {
7969             // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well,
7970             // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags.
7971             if (IsPointer)
7972               Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS |
7973                          OMP_MAP_DELETE | OMP_MAP_CLOSE);
7974 
7975             if (ShouldBeMemberOf) {
7976               // Set placeholder value MEMBER_OF=FFFF to indicate that the flag
7977               // should be later updated with the correct value of MEMBER_OF.
7978               Flags |= OMP_MAP_MEMBER_OF;
7979               // From now on, all subsequent PTR_AND_OBJ entries should not be
7980               // marked as MEMBER_OF.
7981               ShouldBeMemberOf = false;
7982             }
7983           }
7984 
7985           Types.push_back(Flags);
7986         }
7987 
7988         // If we have encountered a member expression so far, keep track of the
7989         // mapped member. If the parent is "*this", then the value declaration
7990         // is nullptr.
7991         if (EncounteredME) {
7992           const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl());
7993           unsigned FieldIndex = FD->getFieldIndex();
7994 
7995           // Update info about the lowest and highest elements for this struct
7996           if (!PartialStruct.Base.isValid()) {
7997             PartialStruct.LowestElem = {FieldIndex, LB};
7998             PartialStruct.HighestElem = {FieldIndex, LB};
7999             PartialStruct.Base = BP;
8000           } else if (FieldIndex < PartialStruct.LowestElem.first) {
8001             PartialStruct.LowestElem = {FieldIndex, LB};
8002           } else if (FieldIndex > PartialStruct.HighestElem.first) {
8003             PartialStruct.HighestElem = {FieldIndex, LB};
8004           }
8005         }
8006 
8007         // If we have a final array section, we are done with this expression.
8008         if (IsFinalArraySection)
8009           break;
8010 
8011         // The pointer becomes the base for the next element.
8012         if (Next != CE)
8013           BP = LB;
8014 
8015         IsExpressionFirstInfo = false;
8016         IsCaptureFirstInfo = false;
8017       }
8018     }
8019   }
8020 
8021   /// Return the adjusted map modifiers if the declaration a capture refers to
8022   /// appears in a first-private clause. This is expected to be used only with
8023   /// directives that start with 'target'.
8024   MappableExprsHandler::OpenMPOffloadMappingFlags
8025   getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const {
8026     assert(Cap.capturesVariable() && "Expected capture by reference only!");
8027 
8028     // A first private variable captured by reference will use only the
8029     // 'private ptr' and 'map to' flag. Return the right flags if the captured
8030     // declaration is known as first-private in this handler.
8031     if (FirstPrivateDecls.count(Cap.getCapturedVar())) {
8032       if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) &&
8033           Cap.getCaptureKind() == CapturedStmt::VCK_ByRef)
8034         return MappableExprsHandler::OMP_MAP_ALWAYS |
8035                MappableExprsHandler::OMP_MAP_TO;
8036       if (Cap.getCapturedVar()->getType()->isAnyPointerType())
8037         return MappableExprsHandler::OMP_MAP_TO |
8038                MappableExprsHandler::OMP_MAP_PTR_AND_OBJ;
8039       return MappableExprsHandler::OMP_MAP_PRIVATE |
8040              MappableExprsHandler::OMP_MAP_TO;
8041     }
8042     return MappableExprsHandler::OMP_MAP_TO |
8043            MappableExprsHandler::OMP_MAP_FROM;
8044   }
8045 
8046   static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) {
8047     // Rotate by getFlagMemberOffset() bits.
8048     return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1)
8049                                                   << getFlagMemberOffset());
8050   }
8051 
8052   static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags,
8053                                      OpenMPOffloadMappingFlags MemberOfFlag) {
8054     // If the entry is PTR_AND_OBJ but has not been marked with the special
8055     // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be
8056     // marked as MEMBER_OF.
8057     if ((Flags & OMP_MAP_PTR_AND_OBJ) &&
8058         ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF))
8059       return;
8060 
8061     // Reset the placeholder value to prepare the flag for the assignment of the
8062     // proper MEMBER_OF value.
8063     Flags &= ~OMP_MAP_MEMBER_OF;
8064     Flags |= MemberOfFlag;
8065   }
8066 
8067   void getPlainLayout(const CXXRecordDecl *RD,
8068                       llvm::SmallVectorImpl<const FieldDecl *> &Layout,
8069                       bool AsBase) const {
8070     const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD);
8071 
8072     llvm::StructType *St =
8073         AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType();
8074 
8075     unsigned NumElements = St->getNumElements();
8076     llvm::SmallVector<
8077         llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4>
8078         RecordLayout(NumElements);
8079 
8080     // Fill bases.
8081     for (const auto &I : RD->bases()) {
8082       if (I.isVirtual())
8083         continue;
8084       const auto *Base = I.getType()->getAsCXXRecordDecl();
8085       // Ignore empty bases.
8086       if (Base->isEmpty() || CGF.getContext()
8087                                  .getASTRecordLayout(Base)
8088                                  .getNonVirtualSize()
8089                                  .isZero())
8090         continue;
8091 
8092       unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base);
8093       RecordLayout[FieldIndex] = Base;
8094     }
8095     // Fill in virtual bases.
8096     for (const auto &I : RD->vbases()) {
8097       const auto *Base = I.getType()->getAsCXXRecordDecl();
8098       // Ignore empty bases.
8099       if (Base->isEmpty())
8100         continue;
8101       unsigned FieldIndex = RL.getVirtualBaseIndex(Base);
8102       if (RecordLayout[FieldIndex])
8103         continue;
8104       RecordLayout[FieldIndex] = Base;
8105     }
8106     // Fill in all the fields.
8107     assert(!RD->isUnion() && "Unexpected union.");
8108     for (const auto *Field : RD->fields()) {
8109       // Fill in non-bitfields. (Bitfields always use a zero pattern, which we
8110       // will fill in later.)
8111       if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) {
8112         unsigned FieldIndex = RL.getLLVMFieldNo(Field);
8113         RecordLayout[FieldIndex] = Field;
8114       }
8115     }
8116     for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>
8117              &Data : RecordLayout) {
8118       if (Data.isNull())
8119         continue;
8120       if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>())
8121         getPlainLayout(Base, Layout, /*AsBase=*/true);
8122       else
8123         Layout.push_back(Data.get<const FieldDecl *>());
8124     }
8125   }
8126 
8127 public:
8128   MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF)
8129       : CurDir(&Dir), CGF(CGF) {
8130     // Extract firstprivate clause information.
8131     for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>())
8132       for (const auto *D : C->varlists())
8133         FirstPrivateDecls.try_emplace(
8134             cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit());
8135     // Extract device pointer clause information.
8136     for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
8137       for (auto L : C->component_lists())
8138         DevPointersMap[L.first].push_back(L.second);
8139   }
8140 
8141   /// Constructor for the declare mapper directive.
8142   MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF)
8143       : CurDir(&Dir), CGF(CGF) {}
8144 
8145   /// Generate code for the combined entry if we have a partially mapped struct
8146   /// and take care of the mapping flags of the arguments corresponding to
8147   /// individual struct members.
8148   void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers,
8149                          MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8150                          MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes,
8151                          const StructRangeInfoTy &PartialStruct) const {
8152     // Base is the base of the struct
8153     BasePointers.push_back(PartialStruct.Base.getPointer());
8154     // Pointer is the address of the lowest element
8155     llvm::Value *LB = PartialStruct.LowestElem.second.getPointer();
8156     Pointers.push_back(LB);
8157     // Size is (addr of {highest+1} element) - (addr of lowest element)
8158     llvm::Value *HB = PartialStruct.HighestElem.second.getPointer();
8159     llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1);
8160     llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy);
8161     llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy);
8162     llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr);
8163     llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty,
8164                                                   /*isSigned=*/false);
8165     Sizes.push_back(Size);
8166     // Map type is always TARGET_PARAM
8167     Types.push_back(OMP_MAP_TARGET_PARAM);
8168     // Remove TARGET_PARAM flag from the first element
8169     (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM;
8170 
8171     // All other current entries will be MEMBER_OF the combined entry
8172     // (except for PTR_AND_OBJ entries which do not have a placeholder value
8173     // 0xFFFF in the MEMBER_OF field).
8174     OpenMPOffloadMappingFlags MemberOfFlag =
8175         getMemberOfFlag(BasePointers.size() - 1);
8176     for (auto &M : CurTypes)
8177       setCorrectMemberOfFlag(M, MemberOfFlag);
8178   }
8179 
8180   /// Generate all the base pointers, section pointers, sizes and map
8181   /// types for the extracted mappable expressions. Also, for each item that
8182   /// relates with a device pointer, a pair of the relevant declaration and
8183   /// index where it occurs is appended to the device pointers info array.
8184   void generateAllInfo(MapBaseValuesArrayTy &BasePointers,
8185                        MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8186                        MapFlagsArrayTy &Types) const {
8187     // We have to process the component lists that relate with the same
8188     // declaration in a single chunk so that we can generate the map flags
8189     // correctly. Therefore, we organize all lists in a map.
8190     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
8191 
8192     // Helper function to fill the information map for the different supported
8193     // clauses.
8194     auto &&InfoGen = [&Info](
8195         const ValueDecl *D,
8196         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8197         OpenMPMapClauseKind MapType,
8198         ArrayRef<OpenMPMapModifierKind> MapModifiers,
8199         bool ReturnDevicePointer, bool IsImplicit) {
8200       const ValueDecl *VD =
8201           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
8202       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
8203                             IsImplicit);
8204     };
8205 
8206     assert(CurDir.is<const OMPExecutableDirective *>() &&
8207            "Expect a executable directive");
8208     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8209     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>())
8210       for (const auto L : C->component_lists()) {
8211         InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(),
8212             /*ReturnDevicePointer=*/false, C->isImplicit());
8213       }
8214     for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>())
8215       for (const auto L : C->component_lists()) {
8216         InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None,
8217             /*ReturnDevicePointer=*/false, C->isImplicit());
8218       }
8219     for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>())
8220       for (const auto L : C->component_lists()) {
8221         InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None,
8222             /*ReturnDevicePointer=*/false, C->isImplicit());
8223       }
8224 
8225     // Look at the use_device_ptr clause information and mark the existing map
8226     // entries as such. If there is no map information for an entry in the
8227     // use_device_ptr list, we create one with map type 'alloc' and zero size
8228     // section. It is the user fault if that was not mapped before. If there is
8229     // no map information and the pointer is a struct member, then we defer the
8230     // emission of that entry until the whole struct has been processed.
8231     llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>>
8232         DeferredInfo;
8233 
8234     for (const auto *C :
8235          CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) {
8236       for (const auto L : C->component_lists()) {
8237         assert(!L.second.empty() && "Not expecting empty list of components!");
8238         const ValueDecl *VD = L.second.back().getAssociatedDeclaration();
8239         VD = cast<ValueDecl>(VD->getCanonicalDecl());
8240         const Expr *IE = L.second.back().getAssociatedExpression();
8241         // If the first component is a member expression, we have to look into
8242         // 'this', which maps to null in the map of map information. Otherwise
8243         // look directly for the information.
8244         auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD);
8245 
8246         // We potentially have map information for this declaration already.
8247         // Look for the first set of components that refer to it.
8248         if (It != Info.end()) {
8249           auto CI = std::find_if(
8250               It->second.begin(), It->second.end(), [VD](const MapInfo &MI) {
8251                 return MI.Components.back().getAssociatedDeclaration() == VD;
8252               });
8253           // If we found a map entry, signal that the pointer has to be returned
8254           // and move on to the next declaration.
8255           if (CI != It->second.end()) {
8256             CI->ReturnDevicePointer = true;
8257             continue;
8258           }
8259         }
8260 
8261         // We didn't find any match in our map information - generate a zero
8262         // size array section - if the pointer is a struct member we defer this
8263         // action until the whole struct has been processed.
8264         if (isa<MemberExpr>(IE)) {
8265           // Insert the pointer into Info to be processed by
8266           // generateInfoForComponentList. Because it is a member pointer
8267           // without a pointee, no entry will be generated for it, therefore
8268           // we need to generate one after the whole struct has been processed.
8269           // Nonetheless, generateInfoForComponentList must be called to take
8270           // the pointer into account for the calculation of the range of the
8271           // partial struct.
8272           InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None,
8273                   /*ReturnDevicePointer=*/false, C->isImplicit());
8274           DeferredInfo[nullptr].emplace_back(IE, VD);
8275         } else {
8276           llvm::Value *Ptr =
8277               CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc());
8278           BasePointers.emplace_back(Ptr, VD);
8279           Pointers.push_back(Ptr);
8280           Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8281           Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM);
8282         }
8283       }
8284     }
8285 
8286     for (const auto &M : Info) {
8287       // We need to know when we generate information for the first component
8288       // associated with a capture, because the mapping flags depend on it.
8289       bool IsFirstComponentList = true;
8290 
8291       // Temporary versions of arrays
8292       MapBaseValuesArrayTy CurBasePointers;
8293       MapValuesArrayTy CurPointers;
8294       MapValuesArrayTy CurSizes;
8295       MapFlagsArrayTy CurTypes;
8296       StructRangeInfoTy PartialStruct;
8297 
8298       for (const MapInfo &L : M.second) {
8299         assert(!L.Components.empty() &&
8300                "Not expecting declaration with no component lists.");
8301 
8302         // Remember the current base pointer index.
8303         unsigned CurrentBasePointersIdx = CurBasePointers.size();
8304         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8305                                      CurBasePointers, CurPointers, CurSizes,
8306                                      CurTypes, PartialStruct,
8307                                      IsFirstComponentList, L.IsImplicit);
8308 
8309         // If this entry relates with a device pointer, set the relevant
8310         // declaration and add the 'return pointer' flag.
8311         if (L.ReturnDevicePointer) {
8312           assert(CurBasePointers.size() > CurrentBasePointersIdx &&
8313                  "Unexpected number of mapped base pointers.");
8314 
8315           const ValueDecl *RelevantVD =
8316               L.Components.back().getAssociatedDeclaration();
8317           assert(RelevantVD &&
8318                  "No relevant declaration related with device pointer??");
8319 
8320           CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD);
8321           CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM;
8322         }
8323         IsFirstComponentList = false;
8324       }
8325 
8326       // Append any pending zero-length pointers which are struct members and
8327       // used with use_device_ptr.
8328       auto CI = DeferredInfo.find(M.first);
8329       if (CI != DeferredInfo.end()) {
8330         for (const DeferredDevicePtrEntryTy &L : CI->second) {
8331           llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF);
8332           llvm::Value *Ptr = this->CGF.EmitLoadOfScalar(
8333               this->CGF.EmitLValue(L.IE), L.IE->getExprLoc());
8334           CurBasePointers.emplace_back(BasePtr, L.VD);
8335           CurPointers.push_back(Ptr);
8336           CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty));
8337           // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder
8338           // value MEMBER_OF=FFFF so that the entry is later updated with the
8339           // correct value of MEMBER_OF.
8340           CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM |
8341                              OMP_MAP_MEMBER_OF);
8342         }
8343       }
8344 
8345       // If there is an entry in PartialStruct it means we have a struct with
8346       // individual members mapped. Emit an extra combined entry.
8347       if (PartialStruct.Base.isValid())
8348         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8349                           PartialStruct);
8350 
8351       // We need to append the results of this capture to what we already have.
8352       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8353       Pointers.append(CurPointers.begin(), CurPointers.end());
8354       Sizes.append(CurSizes.begin(), CurSizes.end());
8355       Types.append(CurTypes.begin(), CurTypes.end());
8356     }
8357   }
8358 
8359   /// Generate all the base pointers, section pointers, sizes and map types for
8360   /// the extracted map clauses of user-defined mapper.
8361   void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers,
8362                                 MapValuesArrayTy &Pointers,
8363                                 MapValuesArrayTy &Sizes,
8364                                 MapFlagsArrayTy &Types) const {
8365     assert(CurDir.is<const OMPDeclareMapperDecl *>() &&
8366            "Expect a declare mapper directive");
8367     const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>();
8368     // We have to process the component lists that relate with the same
8369     // declaration in a single chunk so that we can generate the map flags
8370     // correctly. Therefore, we organize all lists in a map.
8371     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
8372 
8373     // Helper function to fill the information map for the different supported
8374     // clauses.
8375     auto &&InfoGen = [&Info](
8376         const ValueDecl *D,
8377         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8378         OpenMPMapClauseKind MapType,
8379         ArrayRef<OpenMPMapModifierKind> MapModifiers,
8380         bool ReturnDevicePointer, bool IsImplicit) {
8381       const ValueDecl *VD =
8382           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
8383       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
8384                             IsImplicit);
8385     };
8386 
8387     for (const auto *C : CurMapperDir->clauselists()) {
8388       const auto *MC = cast<OMPMapClause>(C);
8389       for (const auto L : MC->component_lists()) {
8390         InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(),
8391                 /*ReturnDevicePointer=*/false, MC->isImplicit());
8392       }
8393     }
8394 
8395     for (const auto &M : Info) {
8396       // We need to know when we generate information for the first component
8397       // associated with a capture, because the mapping flags depend on it.
8398       bool IsFirstComponentList = true;
8399 
8400       // Temporary versions of arrays
8401       MapBaseValuesArrayTy CurBasePointers;
8402       MapValuesArrayTy CurPointers;
8403       MapValuesArrayTy CurSizes;
8404       MapFlagsArrayTy CurTypes;
8405       StructRangeInfoTy PartialStruct;
8406 
8407       for (const MapInfo &L : M.second) {
8408         assert(!L.Components.empty() &&
8409                "Not expecting declaration with no component lists.");
8410         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8411                                      CurBasePointers, CurPointers, CurSizes,
8412                                      CurTypes, PartialStruct,
8413                                      IsFirstComponentList, L.IsImplicit);
8414         IsFirstComponentList = false;
8415       }
8416 
8417       // If there is an entry in PartialStruct it means we have a struct with
8418       // individual members mapped. Emit an extra combined entry.
8419       if (PartialStruct.Base.isValid())
8420         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8421                           PartialStruct);
8422 
8423       // We need to append the results of this capture to what we already have.
8424       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8425       Pointers.append(CurPointers.begin(), CurPointers.end());
8426       Sizes.append(CurSizes.begin(), CurSizes.end());
8427       Types.append(CurTypes.begin(), CurTypes.end());
8428     }
8429   }
8430 
8431   /// Emit capture info for lambdas for variables captured by reference.
8432   void generateInfoForLambdaCaptures(
8433       const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers,
8434       MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8435       MapFlagsArrayTy &Types,
8436       llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const {
8437     const auto *RD = VD->getType()
8438                          .getCanonicalType()
8439                          .getNonReferenceType()
8440                          ->getAsCXXRecordDecl();
8441     if (!RD || !RD->isLambda())
8442       return;
8443     Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD));
8444     LValue VDLVal = CGF.MakeAddrLValue(
8445         VDAddr, VD->getType().getCanonicalType().getNonReferenceType());
8446     llvm::DenseMap<const VarDecl *, FieldDecl *> Captures;
8447     FieldDecl *ThisCapture = nullptr;
8448     RD->getCaptureFields(Captures, ThisCapture);
8449     if (ThisCapture) {
8450       LValue ThisLVal =
8451           CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture);
8452       LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture);
8453       LambdaPointers.try_emplace(ThisLVal.getPointer(CGF),
8454                                  VDLVal.getPointer(CGF));
8455       BasePointers.push_back(ThisLVal.getPointer(CGF));
8456       Pointers.push_back(ThisLValVal.getPointer(CGF));
8457       Sizes.push_back(
8458           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8459                                     CGF.Int64Ty, /*isSigned=*/true));
8460       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8461                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8462     }
8463     for (const LambdaCapture &LC : RD->captures()) {
8464       if (!LC.capturesVariable())
8465         continue;
8466       const VarDecl *VD = LC.getCapturedVar();
8467       if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType())
8468         continue;
8469       auto It = Captures.find(VD);
8470       assert(It != Captures.end() && "Found lambda capture without field.");
8471       LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second);
8472       if (LC.getCaptureKind() == LCK_ByRef) {
8473         LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second);
8474         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
8475                                    VDLVal.getPointer(CGF));
8476         BasePointers.push_back(VarLVal.getPointer(CGF));
8477         Pointers.push_back(VarLValVal.getPointer(CGF));
8478         Sizes.push_back(CGF.Builder.CreateIntCast(
8479             CGF.getTypeSize(
8480                 VD->getType().getCanonicalType().getNonReferenceType()),
8481             CGF.Int64Ty, /*isSigned=*/true));
8482       } else {
8483         RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation());
8484         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
8485                                    VDLVal.getPointer(CGF));
8486         BasePointers.push_back(VarLVal.getPointer(CGF));
8487         Pointers.push_back(VarRVal.getScalarVal());
8488         Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0));
8489       }
8490       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8491                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8492     }
8493   }
8494 
8495   /// Set correct indices for lambdas captures.
8496   void adjustMemberOfForLambdaCaptures(
8497       const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers,
8498       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
8499       MapFlagsArrayTy &Types) const {
8500     for (unsigned I = 0, E = Types.size(); I < E; ++I) {
8501       // Set correct member_of idx for all implicit lambda captures.
8502       if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8503                        OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT))
8504         continue;
8505       llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]);
8506       assert(BasePtr && "Unable to find base lambda address.");
8507       int TgtIdx = -1;
8508       for (unsigned J = I; J > 0; --J) {
8509         unsigned Idx = J - 1;
8510         if (Pointers[Idx] != BasePtr)
8511           continue;
8512         TgtIdx = Idx;
8513         break;
8514       }
8515       assert(TgtIdx != -1 && "Unable to find parent lambda.");
8516       // All other current entries will be MEMBER_OF the combined entry
8517       // (except for PTR_AND_OBJ entries which do not have a placeholder value
8518       // 0xFFFF in the MEMBER_OF field).
8519       OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx);
8520       setCorrectMemberOfFlag(Types[I], MemberOfFlag);
8521     }
8522   }
8523 
8524   /// Generate the base pointers, section pointers, sizes and map types
8525   /// associated to a given capture.
8526   void generateInfoForCapture(const CapturedStmt::Capture *Cap,
8527                               llvm::Value *Arg,
8528                               MapBaseValuesArrayTy &BasePointers,
8529                               MapValuesArrayTy &Pointers,
8530                               MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
8531                               StructRangeInfoTy &PartialStruct) const {
8532     assert(!Cap->capturesVariableArrayType() &&
8533            "Not expecting to generate map info for a variable array type!");
8534 
8535     // We need to know when we generating information for the first component
8536     const ValueDecl *VD = Cap->capturesThis()
8537                               ? nullptr
8538                               : Cap->getCapturedVar()->getCanonicalDecl();
8539 
8540     // If this declaration appears in a is_device_ptr clause we just have to
8541     // pass the pointer by value. If it is a reference to a declaration, we just
8542     // pass its value.
8543     if (DevPointersMap.count(VD)) {
8544       BasePointers.emplace_back(Arg, VD);
8545       Pointers.push_back(Arg);
8546       Sizes.push_back(
8547           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8548                                     CGF.Int64Ty, /*isSigned=*/true));
8549       Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM);
8550       return;
8551     }
8552 
8553     using MapData =
8554         std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef,
8555                    OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>;
8556     SmallVector<MapData, 4> DeclComponentLists;
8557     assert(CurDir.is<const OMPExecutableDirective *>() &&
8558            "Expect a executable directive");
8559     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8560     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8561       for (const auto L : C->decl_component_lists(VD)) {
8562         assert(L.first == VD &&
8563                "We got information for the wrong declaration??");
8564         assert(!L.second.empty() &&
8565                "Not expecting declaration with no component lists.");
8566         DeclComponentLists.emplace_back(L.second, C->getMapType(),
8567                                         C->getMapTypeModifiers(),
8568                                         C->isImplicit());
8569       }
8570     }
8571 
8572     // Find overlapping elements (including the offset from the base element).
8573     llvm::SmallDenseMap<
8574         const MapData *,
8575         llvm::SmallVector<
8576             OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>,
8577         4>
8578         OverlappedData;
8579     size_t Count = 0;
8580     for (const MapData &L : DeclComponentLists) {
8581       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8582       OpenMPMapClauseKind MapType;
8583       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8584       bool IsImplicit;
8585       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8586       ++Count;
8587       for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) {
8588         OMPClauseMappableExprCommon::MappableExprComponentListRef Components1;
8589         std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1;
8590         auto CI = Components.rbegin();
8591         auto CE = Components.rend();
8592         auto SI = Components1.rbegin();
8593         auto SE = Components1.rend();
8594         for (; CI != CE && SI != SE; ++CI, ++SI) {
8595           if (CI->getAssociatedExpression()->getStmtClass() !=
8596               SI->getAssociatedExpression()->getStmtClass())
8597             break;
8598           // Are we dealing with different variables/fields?
8599           if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration())
8600             break;
8601         }
8602         // Found overlapping if, at least for one component, reached the head of
8603         // the components list.
8604         if (CI == CE || SI == SE) {
8605           assert((CI != CE || SI != SE) &&
8606                  "Unexpected full match of the mapping components.");
8607           const MapData &BaseData = CI == CE ? L : L1;
8608           OMPClauseMappableExprCommon::MappableExprComponentListRef SubData =
8609               SI == SE ? Components : Components1;
8610           auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData);
8611           OverlappedElements.getSecond().push_back(SubData);
8612         }
8613       }
8614     }
8615     // Sort the overlapped elements for each item.
8616     llvm::SmallVector<const FieldDecl *, 4> Layout;
8617     if (!OverlappedData.empty()) {
8618       if (const auto *CRD =
8619               VD->getType().getCanonicalType()->getAsCXXRecordDecl())
8620         getPlainLayout(CRD, Layout, /*AsBase=*/false);
8621       else {
8622         const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl();
8623         Layout.append(RD->field_begin(), RD->field_end());
8624       }
8625     }
8626     for (auto &Pair : OverlappedData) {
8627       llvm::sort(
8628           Pair.getSecond(),
8629           [&Layout](
8630               OMPClauseMappableExprCommon::MappableExprComponentListRef First,
8631               OMPClauseMappableExprCommon::MappableExprComponentListRef
8632                   Second) {
8633             auto CI = First.rbegin();
8634             auto CE = First.rend();
8635             auto SI = Second.rbegin();
8636             auto SE = Second.rend();
8637             for (; CI != CE && SI != SE; ++CI, ++SI) {
8638               if (CI->getAssociatedExpression()->getStmtClass() !=
8639                   SI->getAssociatedExpression()->getStmtClass())
8640                 break;
8641               // Are we dealing with different variables/fields?
8642               if (CI->getAssociatedDeclaration() !=
8643                   SI->getAssociatedDeclaration())
8644                 break;
8645             }
8646 
8647             // Lists contain the same elements.
8648             if (CI == CE && SI == SE)
8649               return false;
8650 
8651             // List with less elements is less than list with more elements.
8652             if (CI == CE || SI == SE)
8653               return CI == CE;
8654 
8655             const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration());
8656             const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration());
8657             if (FD1->getParent() == FD2->getParent())
8658               return FD1->getFieldIndex() < FD2->getFieldIndex();
8659             const auto It =
8660                 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) {
8661                   return FD == FD1 || FD == FD2;
8662                 });
8663             return *It == FD1;
8664           });
8665     }
8666 
8667     // Associated with a capture, because the mapping flags depend on it.
8668     // Go through all of the elements with the overlapped elements.
8669     for (const auto &Pair : OverlappedData) {
8670       const MapData &L = *Pair.getFirst();
8671       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8672       OpenMPMapClauseKind MapType;
8673       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8674       bool IsImplicit;
8675       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8676       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
8677           OverlappedComponents = Pair.getSecond();
8678       bool IsFirstComponentList = true;
8679       generateInfoForComponentList(MapType, MapModifiers, Components,
8680                                    BasePointers, Pointers, Sizes, Types,
8681                                    PartialStruct, IsFirstComponentList,
8682                                    IsImplicit, OverlappedComponents);
8683     }
8684     // Go through other elements without overlapped elements.
8685     bool IsFirstComponentList = OverlappedData.empty();
8686     for (const MapData &L : DeclComponentLists) {
8687       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8688       OpenMPMapClauseKind MapType;
8689       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8690       bool IsImplicit;
8691       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8692       auto It = OverlappedData.find(&L);
8693       if (It == OverlappedData.end())
8694         generateInfoForComponentList(MapType, MapModifiers, Components,
8695                                      BasePointers, Pointers, Sizes, Types,
8696                                      PartialStruct, IsFirstComponentList,
8697                                      IsImplicit);
8698       IsFirstComponentList = false;
8699     }
8700   }
8701 
8702   /// Generate the base pointers, section pointers, sizes and map types
8703   /// associated with the declare target link variables.
8704   void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers,
8705                                         MapValuesArrayTy &Pointers,
8706                                         MapValuesArrayTy &Sizes,
8707                                         MapFlagsArrayTy &Types) const {
8708     assert(CurDir.is<const OMPExecutableDirective *>() &&
8709            "Expect a executable directive");
8710     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8711     // Map other list items in the map clause which are not captured variables
8712     // but "declare target link" global variables.
8713     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8714       for (const auto L : C->component_lists()) {
8715         if (!L.first)
8716           continue;
8717         const auto *VD = dyn_cast<VarDecl>(L.first);
8718         if (!VD)
8719           continue;
8720         llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
8721             OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
8722         if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
8723             !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link)
8724           continue;
8725         StructRangeInfoTy PartialStruct;
8726         generateInfoForComponentList(
8727             C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers,
8728             Pointers, Sizes, Types, PartialStruct,
8729             /*IsFirstComponentList=*/true, C->isImplicit());
8730         assert(!PartialStruct.Base.isValid() &&
8731                "No partial structs for declare target link expected.");
8732       }
8733     }
8734   }
8735 
8736   /// Generate the default map information for a given capture \a CI,
8737   /// record field declaration \a RI and captured value \a CV.
8738   void generateDefaultMapInfo(const CapturedStmt::Capture &CI,
8739                               const FieldDecl &RI, llvm::Value *CV,
8740                               MapBaseValuesArrayTy &CurBasePointers,
8741                               MapValuesArrayTy &CurPointers,
8742                               MapValuesArrayTy &CurSizes,
8743                               MapFlagsArrayTy &CurMapTypes) const {
8744     bool IsImplicit = true;
8745     // Do the default mapping.
8746     if (CI.capturesThis()) {
8747       CurBasePointers.push_back(CV);
8748       CurPointers.push_back(CV);
8749       const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr());
8750       CurSizes.push_back(
8751           CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()),
8752                                     CGF.Int64Ty, /*isSigned=*/true));
8753       // Default map type.
8754       CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM);
8755     } else if (CI.capturesVariableByCopy()) {
8756       CurBasePointers.push_back(CV);
8757       CurPointers.push_back(CV);
8758       if (!RI.getType()->isAnyPointerType()) {
8759         // We have to signal to the runtime captures passed by value that are
8760         // not pointers.
8761         CurMapTypes.push_back(OMP_MAP_LITERAL);
8762         CurSizes.push_back(CGF.Builder.CreateIntCast(
8763             CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true));
8764       } else {
8765         // Pointers are implicitly mapped with a zero size and no flags
8766         // (other than first map that is added for all implicit maps).
8767         CurMapTypes.push_back(OMP_MAP_NONE);
8768         CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8769       }
8770       const VarDecl *VD = CI.getCapturedVar();
8771       auto I = FirstPrivateDecls.find(VD);
8772       if (I != FirstPrivateDecls.end())
8773         IsImplicit = I->getSecond();
8774     } else {
8775       assert(CI.capturesVariable() && "Expected captured reference.");
8776       const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr());
8777       QualType ElementType = PtrTy->getPointeeType();
8778       CurSizes.push_back(CGF.Builder.CreateIntCast(
8779           CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true));
8780       // The default map type for a scalar/complex type is 'to' because by
8781       // default the value doesn't have to be retrieved. For an aggregate
8782       // type, the default is 'tofrom'.
8783       CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI));
8784       const VarDecl *VD = CI.getCapturedVar();
8785       auto I = FirstPrivateDecls.find(VD);
8786       if (I != FirstPrivateDecls.end() &&
8787           VD->getType().isConstant(CGF.getContext())) {
8788         llvm::Constant *Addr =
8789             CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD);
8790         // Copy the value of the original variable to the new global copy.
8791         CGF.Builder.CreateMemCpy(
8792             CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(CGF),
8793             Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)),
8794             CurSizes.back(), /*IsVolatile=*/false);
8795         // Use new global variable as the base pointers.
8796         CurBasePointers.push_back(Addr);
8797         CurPointers.push_back(Addr);
8798       } else {
8799         CurBasePointers.push_back(CV);
8800         if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) {
8801           Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue(
8802               CV, ElementType, CGF.getContext().getDeclAlign(VD),
8803               AlignmentSource::Decl));
8804           CurPointers.push_back(PtrAddr.getPointer());
8805         } else {
8806           CurPointers.push_back(CV);
8807         }
8808       }
8809       if (I != FirstPrivateDecls.end())
8810         IsImplicit = I->getSecond();
8811     }
8812     // Every default map produces a single argument which is a target parameter.
8813     CurMapTypes.back() |= OMP_MAP_TARGET_PARAM;
8814 
8815     // Add flag stating this is an implicit map.
8816     if (IsImplicit)
8817       CurMapTypes.back() |= OMP_MAP_IMPLICIT;
8818   }
8819 };
8820 } // anonymous namespace
8821 
8822 /// Emit the arrays used to pass the captures and map information to the
8823 /// offloading runtime library. If there is no map or capture information,
8824 /// return nullptr by reference.
8825 static void
8826 emitOffloadingArrays(CodeGenFunction &CGF,
8827                      MappableExprsHandler::MapBaseValuesArrayTy &BasePointers,
8828                      MappableExprsHandler::MapValuesArrayTy &Pointers,
8829                      MappableExprsHandler::MapValuesArrayTy &Sizes,
8830                      MappableExprsHandler::MapFlagsArrayTy &MapTypes,
8831                      CGOpenMPRuntime::TargetDataInfo &Info) {
8832   CodeGenModule &CGM = CGF.CGM;
8833   ASTContext &Ctx = CGF.getContext();
8834 
8835   // Reset the array information.
8836   Info.clearArrayInfo();
8837   Info.NumberOfPtrs = BasePointers.size();
8838 
8839   if (Info.NumberOfPtrs) {
8840     // Detect if we have any capture size requiring runtime evaluation of the
8841     // size so that a constant array could be eventually used.
8842     bool hasRuntimeEvaluationCaptureSize = false;
8843     for (llvm::Value *S : Sizes)
8844       if (!isa<llvm::Constant>(S)) {
8845         hasRuntimeEvaluationCaptureSize = true;
8846         break;
8847       }
8848 
8849     llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true);
8850     QualType PointerArrayType = Ctx.getConstantArrayType(
8851         Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal,
8852         /*IndexTypeQuals=*/0);
8853 
8854     Info.BasePointersArray =
8855         CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer();
8856     Info.PointersArray =
8857         CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer();
8858 
8859     // If we don't have any VLA types or other types that require runtime
8860     // evaluation, we can use a constant array for the map sizes, otherwise we
8861     // need to fill up the arrays as we do for the pointers.
8862     QualType Int64Ty =
8863         Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
8864     if (hasRuntimeEvaluationCaptureSize) {
8865       QualType SizeArrayType = Ctx.getConstantArrayType(
8866           Int64Ty, PointerNumAP, nullptr, ArrayType::Normal,
8867           /*IndexTypeQuals=*/0);
8868       Info.SizesArray =
8869           CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer();
8870     } else {
8871       // We expect all the sizes to be constant, so we collect them to create
8872       // a constant array.
8873       SmallVector<llvm::Constant *, 16> ConstSizes;
8874       for (llvm::Value *S : Sizes)
8875         ConstSizes.push_back(cast<llvm::Constant>(S));
8876 
8877       auto *SizesArrayInit = llvm::ConstantArray::get(
8878           llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes);
8879       std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"});
8880       auto *SizesArrayGbl = new llvm::GlobalVariable(
8881           CGM.getModule(), SizesArrayInit->getType(),
8882           /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8883           SizesArrayInit, Name);
8884       SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8885       Info.SizesArray = SizesArrayGbl;
8886     }
8887 
8888     // The map types are always constant so we don't need to generate code to
8889     // fill arrays. Instead, we create an array constant.
8890     SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0);
8891     llvm::copy(MapTypes, Mapping.begin());
8892     llvm::Constant *MapTypesArrayInit =
8893         llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping);
8894     std::string MaptypesName =
8895         CGM.getOpenMPRuntime().getName({"offload_maptypes"});
8896     auto *MapTypesArrayGbl = new llvm::GlobalVariable(
8897         CGM.getModule(), MapTypesArrayInit->getType(),
8898         /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8899         MapTypesArrayInit, MaptypesName);
8900     MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8901     Info.MapTypesArray = MapTypesArrayGbl;
8902 
8903     for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) {
8904       llvm::Value *BPVal = *BasePointers[I];
8905       llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32(
8906           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8907           Info.BasePointersArray, 0, I);
8908       BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8909           BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0));
8910       Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8911       CGF.Builder.CreateStore(BPVal, BPAddr);
8912 
8913       if (Info.requiresDevicePointerInfo())
8914         if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl())
8915           Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr);
8916 
8917       llvm::Value *PVal = Pointers[I];
8918       llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32(
8919           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8920           Info.PointersArray, 0, I);
8921       P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8922           P, PVal->getType()->getPointerTo(/*AddrSpace=*/0));
8923       Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8924       CGF.Builder.CreateStore(PVal, PAddr);
8925 
8926       if (hasRuntimeEvaluationCaptureSize) {
8927         llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32(
8928             llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8929             Info.SizesArray,
8930             /*Idx0=*/0,
8931             /*Idx1=*/I);
8932         Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty));
8933         CGF.Builder.CreateStore(
8934             CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true),
8935             SAddr);
8936       }
8937     }
8938   }
8939 }
8940 
8941 /// Emit the arguments to be passed to the runtime library based on the
8942 /// arrays of pointers, sizes and map types.
8943 static void emitOffloadingArraysArgument(
8944     CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg,
8945     llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg,
8946     llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) {
8947   CodeGenModule &CGM = CGF.CGM;
8948   if (Info.NumberOfPtrs) {
8949     BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8950         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8951         Info.BasePointersArray,
8952         /*Idx0=*/0, /*Idx1=*/0);
8953     PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8954         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8955         Info.PointersArray,
8956         /*Idx0=*/0,
8957         /*Idx1=*/0);
8958     SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8959         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray,
8960         /*Idx0=*/0, /*Idx1=*/0);
8961     MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8962         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8963         Info.MapTypesArray,
8964         /*Idx0=*/0,
8965         /*Idx1=*/0);
8966   } else {
8967     BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8968     PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8969     SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8970     MapTypesArrayArg =
8971         llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8972   }
8973 }
8974 
8975 /// Check for inner distribute directive.
8976 static const OMPExecutableDirective *
8977 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) {
8978   const auto *CS = D.getInnermostCapturedStmt();
8979   const auto *Body =
8980       CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
8981   const Stmt *ChildStmt =
8982       CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
8983 
8984   if (const auto *NestedDir =
8985           dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
8986     OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind();
8987     switch (D.getDirectiveKind()) {
8988     case OMPD_target:
8989       if (isOpenMPDistributeDirective(DKind))
8990         return NestedDir;
8991       if (DKind == OMPD_teams) {
8992         Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers(
8993             /*IgnoreCaptured=*/true);
8994         if (!Body)
8995           return nullptr;
8996         ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
8997         if (const auto *NND =
8998                 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
8999           DKind = NND->getDirectiveKind();
9000           if (isOpenMPDistributeDirective(DKind))
9001             return NND;
9002         }
9003       }
9004       return nullptr;
9005     case OMPD_target_teams:
9006       if (isOpenMPDistributeDirective(DKind))
9007         return NestedDir;
9008       return nullptr;
9009     case OMPD_target_parallel:
9010     case OMPD_target_simd:
9011     case OMPD_target_parallel_for:
9012     case OMPD_target_parallel_for_simd:
9013       return nullptr;
9014     case OMPD_target_teams_distribute:
9015     case OMPD_target_teams_distribute_simd:
9016     case OMPD_target_teams_distribute_parallel_for:
9017     case OMPD_target_teams_distribute_parallel_for_simd:
9018     case OMPD_parallel:
9019     case OMPD_for:
9020     case OMPD_parallel_for:
9021     case OMPD_parallel_master:
9022     case OMPD_parallel_sections:
9023     case OMPD_for_simd:
9024     case OMPD_parallel_for_simd:
9025     case OMPD_cancel:
9026     case OMPD_cancellation_point:
9027     case OMPD_ordered:
9028     case OMPD_threadprivate:
9029     case OMPD_allocate:
9030     case OMPD_task:
9031     case OMPD_simd:
9032     case OMPD_sections:
9033     case OMPD_section:
9034     case OMPD_single:
9035     case OMPD_master:
9036     case OMPD_critical:
9037     case OMPD_taskyield:
9038     case OMPD_barrier:
9039     case OMPD_taskwait:
9040     case OMPD_taskgroup:
9041     case OMPD_atomic:
9042     case OMPD_flush:
9043     case OMPD_depobj:
9044     case OMPD_scan:
9045     case OMPD_teams:
9046     case OMPD_target_data:
9047     case OMPD_target_exit_data:
9048     case OMPD_target_enter_data:
9049     case OMPD_distribute:
9050     case OMPD_distribute_simd:
9051     case OMPD_distribute_parallel_for:
9052     case OMPD_distribute_parallel_for_simd:
9053     case OMPD_teams_distribute:
9054     case OMPD_teams_distribute_simd:
9055     case OMPD_teams_distribute_parallel_for:
9056     case OMPD_teams_distribute_parallel_for_simd:
9057     case OMPD_target_update:
9058     case OMPD_declare_simd:
9059     case OMPD_declare_variant:
9060     case OMPD_begin_declare_variant:
9061     case OMPD_end_declare_variant:
9062     case OMPD_declare_target:
9063     case OMPD_end_declare_target:
9064     case OMPD_declare_reduction:
9065     case OMPD_declare_mapper:
9066     case OMPD_taskloop:
9067     case OMPD_taskloop_simd:
9068     case OMPD_master_taskloop:
9069     case OMPD_master_taskloop_simd:
9070     case OMPD_parallel_master_taskloop:
9071     case OMPD_parallel_master_taskloop_simd:
9072     case OMPD_requires:
9073     case OMPD_unknown:
9074       llvm_unreachable("Unexpected directive.");
9075     }
9076   }
9077 
9078   return nullptr;
9079 }
9080 
9081 /// Emit the user-defined mapper function. The code generation follows the
9082 /// pattern in the example below.
9083 /// \code
9084 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle,
9085 ///                                           void *base, void *begin,
9086 ///                                           int64_t size, int64_t type) {
9087 ///   // Allocate space for an array section first.
9088 ///   if (size > 1 && !maptype.IsDelete)
9089 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
9090 ///                                 size*sizeof(Ty), clearToFrom(type));
9091 ///   // Map members.
9092 ///   for (unsigned i = 0; i < size; i++) {
9093 ///     // For each component specified by this mapper:
9094 ///     for (auto c : all_components) {
9095 ///       if (c.hasMapper())
9096 ///         (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size,
9097 ///                       c.arg_type);
9098 ///       else
9099 ///         __tgt_push_mapper_component(rt_mapper_handle, c.arg_base,
9100 ///                                     c.arg_begin, c.arg_size, c.arg_type);
9101 ///     }
9102 ///   }
9103 ///   // Delete the array section.
9104 ///   if (size > 1 && maptype.IsDelete)
9105 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
9106 ///                                 size*sizeof(Ty), clearToFrom(type));
9107 /// }
9108 /// \endcode
9109 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D,
9110                                             CodeGenFunction *CGF) {
9111   if (UDMMap.count(D) > 0)
9112     return;
9113   ASTContext &C = CGM.getContext();
9114   QualType Ty = D->getType();
9115   QualType PtrTy = C.getPointerType(Ty).withRestrict();
9116   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
9117   auto *MapperVarDecl =
9118       cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl());
9119   SourceLocation Loc = D->getLocation();
9120   CharUnits ElementSize = C.getTypeSizeInChars(Ty);
9121 
9122   // Prepare mapper function arguments and attributes.
9123   ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
9124                               C.VoidPtrTy, ImplicitParamDecl::Other);
9125   ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
9126                             ImplicitParamDecl::Other);
9127   ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
9128                              C.VoidPtrTy, ImplicitParamDecl::Other);
9129   ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
9130                             ImplicitParamDecl::Other);
9131   ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
9132                             ImplicitParamDecl::Other);
9133   FunctionArgList Args;
9134   Args.push_back(&HandleArg);
9135   Args.push_back(&BaseArg);
9136   Args.push_back(&BeginArg);
9137   Args.push_back(&SizeArg);
9138   Args.push_back(&TypeArg);
9139   const CGFunctionInfo &FnInfo =
9140       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
9141   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
9142   SmallString<64> TyStr;
9143   llvm::raw_svector_ostream Out(TyStr);
9144   CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out);
9145   std::string Name = getName({"omp_mapper", TyStr, D->getName()});
9146   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
9147                                     Name, &CGM.getModule());
9148   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
9149   Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
9150   // Start the mapper function code generation.
9151   CodeGenFunction MapperCGF(CGM);
9152   MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
9153   // Compute the starting and end addreses of array elements.
9154   llvm::Value *Size = MapperCGF.EmitLoadOfScalar(
9155       MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false,
9156       C.getPointerType(Int64Ty), Loc);
9157   llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast(
9158       MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(),
9159       CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy)));
9160   llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size);
9161   llvm::Value *MapType = MapperCGF.EmitLoadOfScalar(
9162       MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false,
9163       C.getPointerType(Int64Ty), Loc);
9164   // Prepare common arguments for array initiation and deletion.
9165   llvm::Value *Handle = MapperCGF.EmitLoadOfScalar(
9166       MapperCGF.GetAddrOfLocalVar(&HandleArg),
9167       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9168   llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar(
9169       MapperCGF.GetAddrOfLocalVar(&BaseArg),
9170       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9171   llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar(
9172       MapperCGF.GetAddrOfLocalVar(&BeginArg),
9173       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9174 
9175   // Emit array initiation if this is an array section and \p MapType indicates
9176   // that memory allocation is required.
9177   llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head");
9178   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
9179                              ElementSize, HeadBB, /*IsInit=*/true);
9180 
9181   // Emit a for loop to iterate through SizeArg of elements and map all of them.
9182 
9183   // Emit the loop header block.
9184   MapperCGF.EmitBlock(HeadBB);
9185   llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body");
9186   llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done");
9187   // Evaluate whether the initial condition is satisfied.
9188   llvm::Value *IsEmpty =
9189       MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty");
9190   MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
9191   llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock();
9192 
9193   // Emit the loop body block.
9194   MapperCGF.EmitBlock(BodyBB);
9195   llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI(
9196       PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent");
9197   PtrPHI->addIncoming(PtrBegin, EntryBB);
9198   Address PtrCurrent =
9199       Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg)
9200                           .getAlignment()
9201                           .alignmentOfArrayElement(ElementSize));
9202   // Privatize the declared variable of mapper to be the current array element.
9203   CodeGenFunction::OMPPrivateScope Scope(MapperCGF);
9204   Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() {
9205     return MapperCGF
9206         .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>())
9207         .getAddress(MapperCGF);
9208   });
9209   (void)Scope.Privatize();
9210 
9211   // Get map clause information. Fill up the arrays with all mapped variables.
9212   MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9213   MappableExprsHandler::MapValuesArrayTy Pointers;
9214   MappableExprsHandler::MapValuesArrayTy Sizes;
9215   MappableExprsHandler::MapFlagsArrayTy MapTypes;
9216   MappableExprsHandler MEHandler(*D, MapperCGF);
9217   MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes);
9218 
9219   // Call the runtime API __tgt_mapper_num_components to get the number of
9220   // pre-existing components.
9221   llvm::Value *OffloadingArgs[] = {Handle};
9222   llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall(
9223       createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs);
9224   llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl(
9225       PreviousSize,
9226       MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset()));
9227 
9228   // Fill up the runtime mapper handle for all components.
9229   for (unsigned I = 0; I < BasePointers.size(); ++I) {
9230     llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast(
9231         *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
9232     llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast(
9233         Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
9234     llvm::Value *CurSizeArg = Sizes[I];
9235 
9236     // Extract the MEMBER_OF field from the map type.
9237     llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member");
9238     MapperCGF.EmitBlock(MemberBB);
9239     llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]);
9240     llvm::Value *Member = MapperCGF.Builder.CreateAnd(
9241         OriMapType,
9242         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF));
9243     llvm::BasicBlock *MemberCombineBB =
9244         MapperCGF.createBasicBlock("omp.member.combine");
9245     llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type");
9246     llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member);
9247     MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB);
9248     // Add the number of pre-existing components to the MEMBER_OF field if it
9249     // is valid.
9250     MapperCGF.EmitBlock(MemberCombineBB);
9251     llvm::Value *CombinedMember =
9252         MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
9253     // Do nothing if it is not a member of previous components.
9254     MapperCGF.EmitBlock(TypeBB);
9255     llvm::PHINode *MemberMapType =
9256         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype");
9257     MemberMapType->addIncoming(OriMapType, MemberBB);
9258     MemberMapType->addIncoming(CombinedMember, MemberCombineBB);
9259 
9260     // Combine the map type inherited from user-defined mapper with that
9261     // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM
9262     // bits of the \a MapType, which is the input argument of the mapper
9263     // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM
9264     // bits of MemberMapType.
9265     // [OpenMP 5.0], 1.2.6. map-type decay.
9266     //        | alloc |  to   | from  | tofrom | release | delete
9267     // ----------------------------------------------------------
9268     // alloc  | alloc | alloc | alloc | alloc  | release | delete
9269     // to     | alloc |  to   | alloc |   to   | release | delete
9270     // from   | alloc | alloc | from  |  from  | release | delete
9271     // tofrom | alloc |  to   | from  | tofrom | release | delete
9272     llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd(
9273         MapType,
9274         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO |
9275                                    MappableExprsHandler::OMP_MAP_FROM));
9276     llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc");
9277     llvm::BasicBlock *AllocElseBB =
9278         MapperCGF.createBasicBlock("omp.type.alloc.else");
9279     llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to");
9280     llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else");
9281     llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from");
9282     llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end");
9283     llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom);
9284     MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
9285     // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM.
9286     MapperCGF.EmitBlock(AllocBB);
9287     llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd(
9288         MemberMapType,
9289         MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9290                                      MappableExprsHandler::OMP_MAP_FROM)));
9291     MapperCGF.Builder.CreateBr(EndBB);
9292     MapperCGF.EmitBlock(AllocElseBB);
9293     llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ(
9294         LeftToFrom,
9295         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO));
9296     MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
9297     // In case of to, clear OMP_MAP_FROM.
9298     MapperCGF.EmitBlock(ToBB);
9299     llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd(
9300         MemberMapType,
9301         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM));
9302     MapperCGF.Builder.CreateBr(EndBB);
9303     MapperCGF.EmitBlock(ToElseBB);
9304     llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ(
9305         LeftToFrom,
9306         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM));
9307     MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB);
9308     // In case of from, clear OMP_MAP_TO.
9309     MapperCGF.EmitBlock(FromBB);
9310     llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd(
9311         MemberMapType,
9312         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO));
9313     // In case of tofrom, do nothing.
9314     MapperCGF.EmitBlock(EndBB);
9315     llvm::PHINode *CurMapType =
9316         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype");
9317     CurMapType->addIncoming(AllocMapType, AllocBB);
9318     CurMapType->addIncoming(ToMapType, ToBB);
9319     CurMapType->addIncoming(FromMapType, FromBB);
9320     CurMapType->addIncoming(MemberMapType, ToElseBB);
9321 
9322     // TODO: call the corresponding mapper function if a user-defined mapper is
9323     // associated with this map clause.
9324     // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9325     // data structure.
9326     llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg,
9327                                      CurSizeArg, CurMapType};
9328     MapperCGF.EmitRuntimeCall(
9329         createRuntimeFunction(OMPRTL__tgt_push_mapper_component),
9330         OffloadingArgs);
9331   }
9332 
9333   // Update the pointer to point to the next element that needs to be mapped,
9334   // and check whether we have mapped all elements.
9335   llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32(
9336       PtrPHI, /*Idx0=*/1, "omp.arraymap.next");
9337   PtrPHI->addIncoming(PtrNext, BodyBB);
9338   llvm::Value *IsDone =
9339       MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone");
9340   llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit");
9341   MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
9342 
9343   MapperCGF.EmitBlock(ExitBB);
9344   // Emit array deletion if this is an array section and \p MapType indicates
9345   // that deletion is required.
9346   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
9347                              ElementSize, DoneBB, /*IsInit=*/false);
9348 
9349   // Emit the function exit block.
9350   MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true);
9351   MapperCGF.FinishFunction();
9352   UDMMap.try_emplace(D, Fn);
9353   if (CGF) {
9354     auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn);
9355     Decls.second.push_back(D);
9356   }
9357 }
9358 
9359 /// Emit the array initialization or deletion portion for user-defined mapper
9360 /// code generation. First, it evaluates whether an array section is mapped and
9361 /// whether the \a MapType instructs to delete this section. If \a IsInit is
9362 /// true, and \a MapType indicates to not delete this array, array
9363 /// initialization code is generated. If \a IsInit is false, and \a MapType
9364 /// indicates to not this array, array deletion code is generated.
9365 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel(
9366     CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base,
9367     llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType,
9368     CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) {
9369   StringRef Prefix = IsInit ? ".init" : ".del";
9370 
9371   // Evaluate if this is an array section.
9372   llvm::BasicBlock *IsDeleteBB =
9373       MapperCGF.createBasicBlock(getName({"omp.array", Prefix, ".evaldelete"}));
9374   llvm::BasicBlock *BodyBB =
9375       MapperCGF.createBasicBlock(getName({"omp.array", Prefix}));
9376   llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE(
9377       Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray");
9378   MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB);
9379 
9380   // Evaluate if we are going to delete this section.
9381   MapperCGF.EmitBlock(IsDeleteBB);
9382   llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd(
9383       MapType,
9384       MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE));
9385   llvm::Value *DeleteCond;
9386   if (IsInit) {
9387     DeleteCond = MapperCGF.Builder.CreateIsNull(
9388         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
9389   } else {
9390     DeleteCond = MapperCGF.Builder.CreateIsNotNull(
9391         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
9392   }
9393   MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB);
9394 
9395   MapperCGF.EmitBlock(BodyBB);
9396   // Get the array size by multiplying element size and element number (i.e., \p
9397   // Size).
9398   llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul(
9399       Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity()));
9400   // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves
9401   // memory allocation/deletion purpose only.
9402   llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd(
9403       MapType,
9404       MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9405                                    MappableExprsHandler::OMP_MAP_FROM)));
9406   // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9407   // data structure.
9408   llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg};
9409   MapperCGF.EmitRuntimeCall(
9410       createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs);
9411 }
9412 
9413 void CGOpenMPRuntime::emitTargetNumIterationsCall(
9414     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9415     llvm::Value *DeviceID,
9416     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9417                                      const OMPLoopDirective &D)>
9418         SizeEmitter) {
9419   OpenMPDirectiveKind Kind = D.getDirectiveKind();
9420   const OMPExecutableDirective *TD = &D;
9421   // Get nested teams distribute kind directive, if any.
9422   if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind))
9423     TD = getNestedDistributeDirective(CGM.getContext(), D);
9424   if (!TD)
9425     return;
9426   const auto *LD = cast<OMPLoopDirective>(TD);
9427   auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF,
9428                                                      PrePostActionTy &) {
9429     if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) {
9430       llvm::Value *Args[] = {DeviceID, NumIterations};
9431       CGF.EmitRuntimeCall(
9432           createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args);
9433     }
9434   };
9435   emitInlinedDirective(CGF, OMPD_unknown, CodeGen);
9436 }
9437 
9438 void CGOpenMPRuntime::emitTargetCall(
9439     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9440     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
9441     llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
9442     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9443                                      const OMPLoopDirective &D)>
9444         SizeEmitter) {
9445   if (!CGF.HaveInsertPoint())
9446     return;
9447 
9448   assert(OutlinedFn && "Invalid outlined function!");
9449 
9450   const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>();
9451   llvm::SmallVector<llvm::Value *, 16> CapturedVars;
9452   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
9453   auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF,
9454                                             PrePostActionTy &) {
9455     CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9456   };
9457   emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen);
9458 
9459   CodeGenFunction::OMPTargetDataInfo InputInfo;
9460   llvm::Value *MapTypesArray = nullptr;
9461   // Fill up the pointer arrays and transfer execution to the device.
9462   auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo,
9463                     &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars,
9464                     SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) {
9465     if (Device.getInt() == OMPC_DEVICE_ancestor) {
9466       // Reverse offloading is not supported, so just execute on the host.
9467       if (RequiresOuterTask) {
9468         CapturedVars.clear();
9469         CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9470       }
9471       emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9472       return;
9473     }
9474 
9475     // On top of the arrays that were filled up, the target offloading call
9476     // takes as arguments the device id as well as the host pointer. The host
9477     // pointer is used by the runtime library to identify the current target
9478     // region, so it only has to be unique and not necessarily point to
9479     // anything. It could be the pointer to the outlined function that
9480     // implements the target region, but we aren't using that so that the
9481     // compiler doesn't need to keep that, and could therefore inline the host
9482     // function if proven worthwhile during optimization.
9483 
9484     // From this point on, we need to have an ID of the target region defined.
9485     assert(OutlinedFnID && "Invalid outlined function ID!");
9486 
9487     // Emit device ID if any.
9488     llvm::Value *DeviceID;
9489     if (Device.getPointer()) {
9490       assert((Device.getInt() == OMPC_DEVICE_unknown ||
9491               Device.getInt() == OMPC_DEVICE_device_num) &&
9492              "Expected device_num modifier.");
9493       llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer());
9494       DeviceID =
9495           CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true);
9496     } else {
9497       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
9498     }
9499 
9500     // Emit the number of elements in the offloading arrays.
9501     llvm::Value *PointerNum =
9502         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
9503 
9504     // Return value of the runtime offloading call.
9505     llvm::Value *Return;
9506 
9507     llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D);
9508     llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D);
9509 
9510     // Emit tripcount for the target loop-based directive.
9511     emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter);
9512 
9513     bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
9514     // The target region is an outlined function launched by the runtime
9515     // via calls __tgt_target() or __tgt_target_teams().
9516     //
9517     // __tgt_target() launches a target region with one team and one thread,
9518     // executing a serial region.  This master thread may in turn launch
9519     // more threads within its team upon encountering a parallel region,
9520     // however, no additional teams can be launched on the device.
9521     //
9522     // __tgt_target_teams() launches a target region with one or more teams,
9523     // each with one or more threads.  This call is required for target
9524     // constructs such as:
9525     //  'target teams'
9526     //  'target' / 'teams'
9527     //  'target teams distribute parallel for'
9528     //  'target parallel'
9529     // and so on.
9530     //
9531     // Note that on the host and CPU targets, the runtime implementation of
9532     // these calls simply call the outlined function without forking threads.
9533     // The outlined functions themselves have runtime calls to
9534     // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by
9535     // the compiler in emitTeamsCall() and emitParallelCall().
9536     //
9537     // In contrast, on the NVPTX target, the implementation of
9538     // __tgt_target_teams() launches a GPU kernel with the requested number
9539     // of teams and threads so no additional calls to the runtime are required.
9540     if (NumTeams) {
9541       // If we have NumTeams defined this means that we have an enclosed teams
9542       // region. Therefore we also expect to have NumThreads defined. These two
9543       // values should be defined in the presence of a teams directive,
9544       // regardless of having any clauses associated. If the user is using teams
9545       // but no clauses, these two values will be the default that should be
9546       // passed to the runtime library - a 32-bit integer with the value zero.
9547       assert(NumThreads && "Thread limit expression should be available along "
9548                            "with number of teams.");
9549       llvm::Value *OffloadingArgs[] = {DeviceID,
9550                                        OutlinedFnID,
9551                                        PointerNum,
9552                                        InputInfo.BasePointersArray.getPointer(),
9553                                        InputInfo.PointersArray.getPointer(),
9554                                        InputInfo.SizesArray.getPointer(),
9555                                        MapTypesArray,
9556                                        NumTeams,
9557                                        NumThreads};
9558       Return = CGF.EmitRuntimeCall(
9559           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait
9560                                           : OMPRTL__tgt_target_teams),
9561           OffloadingArgs);
9562     } else {
9563       llvm::Value *OffloadingArgs[] = {DeviceID,
9564                                        OutlinedFnID,
9565                                        PointerNum,
9566                                        InputInfo.BasePointersArray.getPointer(),
9567                                        InputInfo.PointersArray.getPointer(),
9568                                        InputInfo.SizesArray.getPointer(),
9569                                        MapTypesArray};
9570       Return = CGF.EmitRuntimeCall(
9571           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait
9572                                           : OMPRTL__tgt_target),
9573           OffloadingArgs);
9574     }
9575 
9576     // Check the error code and execute the host version if required.
9577     llvm::BasicBlock *OffloadFailedBlock =
9578         CGF.createBasicBlock("omp_offload.failed");
9579     llvm::BasicBlock *OffloadContBlock =
9580         CGF.createBasicBlock("omp_offload.cont");
9581     llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return);
9582     CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock);
9583 
9584     CGF.EmitBlock(OffloadFailedBlock);
9585     if (RequiresOuterTask) {
9586       CapturedVars.clear();
9587       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9588     }
9589     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9590     CGF.EmitBranch(OffloadContBlock);
9591 
9592     CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true);
9593   };
9594 
9595   // Notify that the host version must be executed.
9596   auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars,
9597                     RequiresOuterTask](CodeGenFunction &CGF,
9598                                        PrePostActionTy &) {
9599     if (RequiresOuterTask) {
9600       CapturedVars.clear();
9601       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9602     }
9603     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9604   };
9605 
9606   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
9607                           &CapturedVars, RequiresOuterTask,
9608                           &CS](CodeGenFunction &CGF, PrePostActionTy &) {
9609     // Fill up the arrays with all the captured variables.
9610     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9611     MappableExprsHandler::MapValuesArrayTy Pointers;
9612     MappableExprsHandler::MapValuesArrayTy Sizes;
9613     MappableExprsHandler::MapFlagsArrayTy MapTypes;
9614 
9615     // Get mappable expression information.
9616     MappableExprsHandler MEHandler(D, CGF);
9617     llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers;
9618 
9619     auto RI = CS.getCapturedRecordDecl()->field_begin();
9620     auto CV = CapturedVars.begin();
9621     for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(),
9622                                               CE = CS.capture_end();
9623          CI != CE; ++CI, ++RI, ++CV) {
9624       MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers;
9625       MappableExprsHandler::MapValuesArrayTy CurPointers;
9626       MappableExprsHandler::MapValuesArrayTy CurSizes;
9627       MappableExprsHandler::MapFlagsArrayTy CurMapTypes;
9628       MappableExprsHandler::StructRangeInfoTy PartialStruct;
9629 
9630       // VLA sizes are passed to the outlined region by copy and do not have map
9631       // information associated.
9632       if (CI->capturesVariableArrayType()) {
9633         CurBasePointers.push_back(*CV);
9634         CurPointers.push_back(*CV);
9635         CurSizes.push_back(CGF.Builder.CreateIntCast(
9636             CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true));
9637         // Copy to the device as an argument. No need to retrieve it.
9638         CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL |
9639                               MappableExprsHandler::OMP_MAP_TARGET_PARAM |
9640                               MappableExprsHandler::OMP_MAP_IMPLICIT);
9641       } else {
9642         // If we have any information in the map clause, we use it, otherwise we
9643         // just do a default mapping.
9644         MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers,
9645                                          CurSizes, CurMapTypes, PartialStruct);
9646         if (CurBasePointers.empty())
9647           MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers,
9648                                            CurPointers, CurSizes, CurMapTypes);
9649         // Generate correct mapping for variables captured by reference in
9650         // lambdas.
9651         if (CI->capturesVariable())
9652           MEHandler.generateInfoForLambdaCaptures(
9653               CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes,
9654               CurMapTypes, LambdaPointers);
9655       }
9656       // We expect to have at least an element of information for this capture.
9657       assert(!CurBasePointers.empty() &&
9658              "Non-existing map pointer for capture!");
9659       assert(CurBasePointers.size() == CurPointers.size() &&
9660              CurBasePointers.size() == CurSizes.size() &&
9661              CurBasePointers.size() == CurMapTypes.size() &&
9662              "Inconsistent map information sizes!");
9663 
9664       // If there is an entry in PartialStruct it means we have a struct with
9665       // individual members mapped. Emit an extra combined entry.
9666       if (PartialStruct.Base.isValid())
9667         MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes,
9668                                     CurMapTypes, PartialStruct);
9669 
9670       // We need to append the results of this capture to what we already have.
9671       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
9672       Pointers.append(CurPointers.begin(), CurPointers.end());
9673       Sizes.append(CurSizes.begin(), CurSizes.end());
9674       MapTypes.append(CurMapTypes.begin(), CurMapTypes.end());
9675     }
9676     // Adjust MEMBER_OF flags for the lambdas captures.
9677     MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers,
9678                                               Pointers, MapTypes);
9679     // Map other list items in the map clause which are not captured variables
9680     // but "declare target link" global variables.
9681     MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes,
9682                                                MapTypes);
9683 
9684     TargetDataInfo Info;
9685     // Fill up the arrays and create the arguments.
9686     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
9687     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
9688                                  Info.PointersArray, Info.SizesArray,
9689                                  Info.MapTypesArray, Info);
9690     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
9691     InputInfo.BasePointersArray =
9692         Address(Info.BasePointersArray, CGM.getPointerAlign());
9693     InputInfo.PointersArray =
9694         Address(Info.PointersArray, CGM.getPointerAlign());
9695     InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign());
9696     MapTypesArray = Info.MapTypesArray;
9697     if (RequiresOuterTask)
9698       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
9699     else
9700       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
9701   };
9702 
9703   auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask](
9704                              CodeGenFunction &CGF, PrePostActionTy &) {
9705     if (RequiresOuterTask) {
9706       CodeGenFunction::OMPTargetDataInfo InputInfo;
9707       CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo);
9708     } else {
9709       emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen);
9710     }
9711   };
9712 
9713   // If we have a target function ID it means that we need to support
9714   // offloading, otherwise, just execute on the host. We need to execute on host
9715   // regardless of the conditional in the if clause if, e.g., the user do not
9716   // specify target triples.
9717   if (OutlinedFnID) {
9718     if (IfCond) {
9719       emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen);
9720     } else {
9721       RegionCodeGenTy ThenRCG(TargetThenGen);
9722       ThenRCG(CGF);
9723     }
9724   } else {
9725     RegionCodeGenTy ElseRCG(TargetElseGen);
9726     ElseRCG(CGF);
9727   }
9728 }
9729 
9730 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S,
9731                                                     StringRef ParentName) {
9732   if (!S)
9733     return;
9734 
9735   // Codegen OMP target directives that offload compute to the device.
9736   bool RequiresDeviceCodegen =
9737       isa<OMPExecutableDirective>(S) &&
9738       isOpenMPTargetExecutionDirective(
9739           cast<OMPExecutableDirective>(S)->getDirectiveKind());
9740 
9741   if (RequiresDeviceCodegen) {
9742     const auto &E = *cast<OMPExecutableDirective>(S);
9743     unsigned DeviceID;
9744     unsigned FileID;
9745     unsigned Line;
9746     getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID,
9747                              FileID, Line);
9748 
9749     // Is this a target region that should not be emitted as an entry point? If
9750     // so just signal we are done with this target region.
9751     if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID,
9752                                                             ParentName, Line))
9753       return;
9754 
9755     switch (E.getDirectiveKind()) {
9756     case OMPD_target:
9757       CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName,
9758                                                    cast<OMPTargetDirective>(E));
9759       break;
9760     case OMPD_target_parallel:
9761       CodeGenFunction::EmitOMPTargetParallelDeviceFunction(
9762           CGM, ParentName, cast<OMPTargetParallelDirective>(E));
9763       break;
9764     case OMPD_target_teams:
9765       CodeGenFunction::EmitOMPTargetTeamsDeviceFunction(
9766           CGM, ParentName, cast<OMPTargetTeamsDirective>(E));
9767       break;
9768     case OMPD_target_teams_distribute:
9769       CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction(
9770           CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E));
9771       break;
9772     case OMPD_target_teams_distribute_simd:
9773       CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction(
9774           CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E));
9775       break;
9776     case OMPD_target_parallel_for:
9777       CodeGenFunction::EmitOMPTargetParallelForDeviceFunction(
9778           CGM, ParentName, cast<OMPTargetParallelForDirective>(E));
9779       break;
9780     case OMPD_target_parallel_for_simd:
9781       CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction(
9782           CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E));
9783       break;
9784     case OMPD_target_simd:
9785       CodeGenFunction::EmitOMPTargetSimdDeviceFunction(
9786           CGM, ParentName, cast<OMPTargetSimdDirective>(E));
9787       break;
9788     case OMPD_target_teams_distribute_parallel_for:
9789       CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction(
9790           CGM, ParentName,
9791           cast<OMPTargetTeamsDistributeParallelForDirective>(E));
9792       break;
9793     case OMPD_target_teams_distribute_parallel_for_simd:
9794       CodeGenFunction::
9795           EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(
9796               CGM, ParentName,
9797               cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E));
9798       break;
9799     case OMPD_parallel:
9800     case OMPD_for:
9801     case OMPD_parallel_for:
9802     case OMPD_parallel_master:
9803     case OMPD_parallel_sections:
9804     case OMPD_for_simd:
9805     case OMPD_parallel_for_simd:
9806     case OMPD_cancel:
9807     case OMPD_cancellation_point:
9808     case OMPD_ordered:
9809     case OMPD_threadprivate:
9810     case OMPD_allocate:
9811     case OMPD_task:
9812     case OMPD_simd:
9813     case OMPD_sections:
9814     case OMPD_section:
9815     case OMPD_single:
9816     case OMPD_master:
9817     case OMPD_critical:
9818     case OMPD_taskyield:
9819     case OMPD_barrier:
9820     case OMPD_taskwait:
9821     case OMPD_taskgroup:
9822     case OMPD_atomic:
9823     case OMPD_flush:
9824     case OMPD_depobj:
9825     case OMPD_scan:
9826     case OMPD_teams:
9827     case OMPD_target_data:
9828     case OMPD_target_exit_data:
9829     case OMPD_target_enter_data:
9830     case OMPD_distribute:
9831     case OMPD_distribute_simd:
9832     case OMPD_distribute_parallel_for:
9833     case OMPD_distribute_parallel_for_simd:
9834     case OMPD_teams_distribute:
9835     case OMPD_teams_distribute_simd:
9836     case OMPD_teams_distribute_parallel_for:
9837     case OMPD_teams_distribute_parallel_for_simd:
9838     case OMPD_target_update:
9839     case OMPD_declare_simd:
9840     case OMPD_declare_variant:
9841     case OMPD_begin_declare_variant:
9842     case OMPD_end_declare_variant:
9843     case OMPD_declare_target:
9844     case OMPD_end_declare_target:
9845     case OMPD_declare_reduction:
9846     case OMPD_declare_mapper:
9847     case OMPD_taskloop:
9848     case OMPD_taskloop_simd:
9849     case OMPD_master_taskloop:
9850     case OMPD_master_taskloop_simd:
9851     case OMPD_parallel_master_taskloop:
9852     case OMPD_parallel_master_taskloop_simd:
9853     case OMPD_requires:
9854     case OMPD_unknown:
9855       llvm_unreachable("Unknown target directive for OpenMP device codegen.");
9856     }
9857     return;
9858   }
9859 
9860   if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) {
9861     if (!E->hasAssociatedStmt() || !E->getAssociatedStmt())
9862       return;
9863 
9864     scanForTargetRegionsFunctions(
9865         E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName);
9866     return;
9867   }
9868 
9869   // If this is a lambda function, look into its body.
9870   if (const auto *L = dyn_cast<LambdaExpr>(S))
9871     S = L->getBody();
9872 
9873   // Keep looking for target regions recursively.
9874   for (const Stmt *II : S->children())
9875     scanForTargetRegionsFunctions(II, ParentName);
9876 }
9877 
9878 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) {
9879   // If emitting code for the host, we do not process FD here. Instead we do
9880   // the normal code generation.
9881   if (!CGM.getLangOpts().OpenMPIsDevice) {
9882     if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) {
9883       Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9884           OMPDeclareTargetDeclAttr::getDeviceType(FD);
9885       // Do not emit device_type(nohost) functions for the host.
9886       if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost)
9887         return true;
9888     }
9889     return false;
9890   }
9891 
9892   const ValueDecl *VD = cast<ValueDecl>(GD.getDecl());
9893   // Try to detect target regions in the function.
9894   if (const auto *FD = dyn_cast<FunctionDecl>(VD)) {
9895     StringRef Name = CGM.getMangledName(GD);
9896     scanForTargetRegionsFunctions(FD->getBody(), Name);
9897     Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9898         OMPDeclareTargetDeclAttr::getDeviceType(FD);
9899     // Do not emit device_type(nohost) functions for the host.
9900     if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host)
9901       return true;
9902   }
9903 
9904   // Do not to emit function if it is not marked as declare target.
9905   return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) &&
9906          AlreadyEmittedTargetDecls.count(VD) == 0;
9907 }
9908 
9909 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
9910   if (!CGM.getLangOpts().OpenMPIsDevice)
9911     return false;
9912 
9913   // Check if there are Ctors/Dtors in this declaration and look for target
9914   // regions in it. We use the complete variant to produce the kernel name
9915   // mangling.
9916   QualType RDTy = cast<VarDecl>(GD.getDecl())->getType();
9917   if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) {
9918     for (const CXXConstructorDecl *Ctor : RD->ctors()) {
9919       StringRef ParentName =
9920           CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete));
9921       scanForTargetRegionsFunctions(Ctor->getBody(), ParentName);
9922     }
9923     if (const CXXDestructorDecl *Dtor = RD->getDestructor()) {
9924       StringRef ParentName =
9925           CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete));
9926       scanForTargetRegionsFunctions(Dtor->getBody(), ParentName);
9927     }
9928   }
9929 
9930   // Do not to emit variable if it is not marked as declare target.
9931   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9932       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(
9933           cast<VarDecl>(GD.getDecl()));
9934   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
9935       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9936        HasRequiresUnifiedSharedMemory)) {
9937     DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl()));
9938     return true;
9939   }
9940   return false;
9941 }
9942 
9943 llvm::Constant *
9944 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF,
9945                                                 const VarDecl *VD) {
9946   assert(VD->getType().isConstant(CGM.getContext()) &&
9947          "Expected constant variable.");
9948   StringRef VarName;
9949   llvm::Constant *Addr;
9950   llvm::GlobalValue::LinkageTypes Linkage;
9951   QualType Ty = VD->getType();
9952   SmallString<128> Buffer;
9953   {
9954     unsigned DeviceID;
9955     unsigned FileID;
9956     unsigned Line;
9957     getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID,
9958                              FileID, Line);
9959     llvm::raw_svector_ostream OS(Buffer);
9960     OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID)
9961        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
9962     VarName = OS.str();
9963   }
9964   Linkage = llvm::GlobalValue::InternalLinkage;
9965   Addr =
9966       getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName,
9967                                   getDefaultFirstprivateAddressSpace());
9968   cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage);
9969   CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty);
9970   CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr));
9971   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
9972       VarName, Addr, VarSize,
9973       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage);
9974   return Addr;
9975 }
9976 
9977 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD,
9978                                                    llvm::Constant *Addr) {
9979   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
9980       !CGM.getLangOpts().OpenMPIsDevice)
9981     return;
9982   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9983       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
9984   if (!Res) {
9985     if (CGM.getLangOpts().OpenMPIsDevice) {
9986       // Register non-target variables being emitted in device code (debug info
9987       // may cause this).
9988       StringRef VarName = CGM.getMangledName(VD);
9989       EmittedNonTargetVariables.try_emplace(VarName, Addr);
9990     }
9991     return;
9992   }
9993   // Register declare target variables.
9994   OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags;
9995   StringRef VarName;
9996   CharUnits VarSize;
9997   llvm::GlobalValue::LinkageTypes Linkage;
9998 
9999   if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10000       !HasRequiresUnifiedSharedMemory) {
10001     Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
10002     VarName = CGM.getMangledName(VD);
10003     if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) {
10004       VarSize = CGM.getContext().getTypeSizeInChars(VD->getType());
10005       assert(!VarSize.isZero() && "Expected non-zero size of the variable");
10006     } else {
10007       VarSize = CharUnits::Zero();
10008     }
10009     Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false);
10010     // Temp solution to prevent optimizations of the internal variables.
10011     if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) {
10012       std::string RefName = getName({VarName, "ref"});
10013       if (!CGM.GetGlobalValue(RefName)) {
10014         llvm::Constant *AddrRef =
10015             getOrCreateInternalVariable(Addr->getType(), RefName);
10016         auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef);
10017         GVAddrRef->setConstant(/*Val=*/true);
10018         GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage);
10019         GVAddrRef->setInitializer(Addr);
10020         CGM.addCompilerUsedGlobal(GVAddrRef);
10021       }
10022     }
10023   } else {
10024     assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
10025             (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10026              HasRequiresUnifiedSharedMemory)) &&
10027            "Declare target attribute must link or to with unified memory.");
10028     if (*Res == OMPDeclareTargetDeclAttr::MT_Link)
10029       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink;
10030     else
10031       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
10032 
10033     if (CGM.getLangOpts().OpenMPIsDevice) {
10034       VarName = Addr->getName();
10035       Addr = nullptr;
10036     } else {
10037       VarName = getAddrOfDeclareTargetVar(VD).getName();
10038       Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer());
10039     }
10040     VarSize = CGM.getPointerSize();
10041     Linkage = llvm::GlobalValue::WeakAnyLinkage;
10042   }
10043 
10044   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
10045       VarName, Addr, VarSize, Flags, Linkage);
10046 }
10047 
10048 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) {
10049   if (isa<FunctionDecl>(GD.getDecl()) ||
10050       isa<OMPDeclareReductionDecl>(GD.getDecl()))
10051     return emitTargetFunctions(GD);
10052 
10053   return emitTargetGlobalVariable(GD);
10054 }
10055 
10056 void CGOpenMPRuntime::emitDeferredTargetDecls() const {
10057   for (const VarDecl *VD : DeferredGlobalVariables) {
10058     llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
10059         OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
10060     if (!Res)
10061       continue;
10062     if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10063         !HasRequiresUnifiedSharedMemory) {
10064       CGM.EmitGlobal(VD);
10065     } else {
10066       assert((*Res == OMPDeclareTargetDeclAttr::MT_Link ||
10067               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10068                HasRequiresUnifiedSharedMemory)) &&
10069              "Expected link clause or to clause with unified memory.");
10070       (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
10071     }
10072   }
10073 }
10074 
10075 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas(
10076     CodeGenFunction &CGF, const OMPExecutableDirective &D) const {
10077   assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) &&
10078          " Expected target-based directive.");
10079 }
10080 
10081 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) {
10082   for (const OMPClause *Clause : D->clauselists()) {
10083     if (Clause->getClauseKind() == OMPC_unified_shared_memory) {
10084       HasRequiresUnifiedSharedMemory = true;
10085     } else if (const auto *AC =
10086                    dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) {
10087       switch (AC->getAtomicDefaultMemOrderKind()) {
10088       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel:
10089         RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease;
10090         break;
10091       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst:
10092         RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent;
10093         break;
10094       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed:
10095         RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic;
10096         break;
10097       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown:
10098         break;
10099       }
10100     }
10101   }
10102 }
10103 
10104 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const {
10105   return RequiresAtomicOrdering;
10106 }
10107 
10108 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD,
10109                                                        LangAS &AS) {
10110   if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())
10111     return false;
10112   const auto *A = VD->getAttr<OMPAllocateDeclAttr>();
10113   switch(A->getAllocatorType()) {
10114   case OMPAllocateDeclAttr::OMPDefaultMemAlloc:
10115   // Not supported, fallback to the default mem space.
10116   case OMPAllocateDeclAttr::OMPLargeCapMemAlloc:
10117   case OMPAllocateDeclAttr::OMPCGroupMemAlloc:
10118   case OMPAllocateDeclAttr::OMPHighBWMemAlloc:
10119   case OMPAllocateDeclAttr::OMPLowLatMemAlloc:
10120   case OMPAllocateDeclAttr::OMPThreadMemAlloc:
10121   case OMPAllocateDeclAttr::OMPConstMemAlloc:
10122   case OMPAllocateDeclAttr::OMPPTeamMemAlloc:
10123     AS = LangAS::Default;
10124     return true;
10125   case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc:
10126     llvm_unreachable("Expected predefined allocator for the variables with the "
10127                      "static storage.");
10128   }
10129   return false;
10130 }
10131 
10132 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const {
10133   return HasRequiresUnifiedSharedMemory;
10134 }
10135 
10136 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII(
10137     CodeGenModule &CGM)
10138     : CGM(CGM) {
10139   if (CGM.getLangOpts().OpenMPIsDevice) {
10140     SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal;
10141     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false;
10142   }
10143 }
10144 
10145 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() {
10146   if (CGM.getLangOpts().OpenMPIsDevice)
10147     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal;
10148 }
10149 
10150 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) {
10151   if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal)
10152     return true;
10153 
10154   const auto *D = cast<FunctionDecl>(GD.getDecl());
10155   // Do not to emit function if it is marked as declare target as it was already
10156   // emitted.
10157   if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) {
10158     if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) {
10159       if (auto *F = dyn_cast_or_null<llvm::Function>(
10160               CGM.GetGlobalValue(CGM.getMangledName(GD))))
10161         return !F->isDeclaration();
10162       return false;
10163     }
10164     return true;
10165   }
10166 
10167   return !AlreadyEmittedTargetDecls.insert(D).second;
10168 }
10169 
10170 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() {
10171   // If we don't have entries or if we are emitting code for the device, we
10172   // don't need to do anything.
10173   if (CGM.getLangOpts().OMPTargetTriples.empty() ||
10174       CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice ||
10175       (OffloadEntriesInfoManager.empty() &&
10176        !HasEmittedDeclareTargetRegion &&
10177        !HasEmittedTargetRegion))
10178     return nullptr;
10179 
10180   // Create and register the function that handles the requires directives.
10181   ASTContext &C = CGM.getContext();
10182 
10183   llvm::Function *RequiresRegFn;
10184   {
10185     CodeGenFunction CGF(CGM);
10186     const auto &FI = CGM.getTypes().arrangeNullaryFunction();
10187     llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
10188     std::string ReqName = getName({"omp_offloading", "requires_reg"});
10189     RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI);
10190     CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {});
10191     OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE;
10192     // TODO: check for other requires clauses.
10193     // The requires directive takes effect only when a target region is
10194     // present in the compilation unit. Otherwise it is ignored and not
10195     // passed to the runtime. This avoids the runtime from throwing an error
10196     // for mismatching requires clauses across compilation units that don't
10197     // contain at least 1 target region.
10198     assert((HasEmittedTargetRegion ||
10199             HasEmittedDeclareTargetRegion ||
10200             !OffloadEntriesInfoManager.empty()) &&
10201            "Target or declare target region expected.");
10202     if (HasRequiresUnifiedSharedMemory)
10203       Flags = OMP_REQ_UNIFIED_SHARED_MEMORY;
10204     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires),
10205         llvm::ConstantInt::get(CGM.Int64Ty, Flags));
10206     CGF.FinishFunction();
10207   }
10208   return RequiresRegFn;
10209 }
10210 
10211 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF,
10212                                     const OMPExecutableDirective &D,
10213                                     SourceLocation Loc,
10214                                     llvm::Function *OutlinedFn,
10215                                     ArrayRef<llvm::Value *> CapturedVars) {
10216   if (!CGF.HaveInsertPoint())
10217     return;
10218 
10219   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
10220   CodeGenFunction::RunCleanupsScope Scope(CGF);
10221 
10222   // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn);
10223   llvm::Value *Args[] = {
10224       RTLoc,
10225       CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
10226       CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())};
10227   llvm::SmallVector<llvm::Value *, 16> RealArgs;
10228   RealArgs.append(std::begin(Args), std::end(Args));
10229   RealArgs.append(CapturedVars.begin(), CapturedVars.end());
10230 
10231   llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams);
10232   CGF.EmitRuntimeCall(RTLFn, RealArgs);
10233 }
10234 
10235 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
10236                                          const Expr *NumTeams,
10237                                          const Expr *ThreadLimit,
10238                                          SourceLocation Loc) {
10239   if (!CGF.HaveInsertPoint())
10240     return;
10241 
10242   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
10243 
10244   llvm::Value *NumTeamsVal =
10245       NumTeams
10246           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams),
10247                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
10248           : CGF.Builder.getInt32(0);
10249 
10250   llvm::Value *ThreadLimitVal =
10251       ThreadLimit
10252           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit),
10253                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
10254           : CGF.Builder.getInt32(0);
10255 
10256   // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit)
10257   llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal,
10258                                      ThreadLimitVal};
10259   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams),
10260                       PushNumTeamsArgs);
10261 }
10262 
10263 void CGOpenMPRuntime::emitTargetDataCalls(
10264     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
10265     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
10266   if (!CGF.HaveInsertPoint())
10267     return;
10268 
10269   // Action used to replace the default codegen action and turn privatization
10270   // off.
10271   PrePostActionTy NoPrivAction;
10272 
10273   // Generate the code for the opening of the data environment. Capture all the
10274   // arguments of the runtime call by reference because they are used in the
10275   // closing of the region.
10276   auto &&BeginThenGen = [this, &D, Device, &Info,
10277                          &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) {
10278     // Fill up the arrays with all the mapped variables.
10279     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
10280     MappableExprsHandler::MapValuesArrayTy Pointers;
10281     MappableExprsHandler::MapValuesArrayTy Sizes;
10282     MappableExprsHandler::MapFlagsArrayTy MapTypes;
10283 
10284     // Get map clause information.
10285     MappableExprsHandler MCHandler(D, CGF);
10286     MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
10287 
10288     // Fill up the arrays and create the arguments.
10289     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
10290 
10291     llvm::Value *BasePointersArrayArg = nullptr;
10292     llvm::Value *PointersArrayArg = nullptr;
10293     llvm::Value *SizesArrayArg = nullptr;
10294     llvm::Value *MapTypesArrayArg = nullptr;
10295     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10296                                  SizesArrayArg, MapTypesArrayArg, Info);
10297 
10298     // Emit device ID if any.
10299     llvm::Value *DeviceID = nullptr;
10300     if (Device) {
10301       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10302                                            CGF.Int64Ty, /*isSigned=*/true);
10303     } else {
10304       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10305     }
10306 
10307     // Emit the number of elements in the offloading arrays.
10308     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10309 
10310     llvm::Value *OffloadingArgs[] = {
10311         DeviceID,         PointerNum,    BasePointersArrayArg,
10312         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10313     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin),
10314                         OffloadingArgs);
10315 
10316     // If device pointer privatization is required, emit the body of the region
10317     // here. It will have to be duplicated: with and without privatization.
10318     if (!Info.CaptureDeviceAddrMap.empty())
10319       CodeGen(CGF);
10320   };
10321 
10322   // Generate code for the closing of the data region.
10323   auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF,
10324                                             PrePostActionTy &) {
10325     assert(Info.isValid() && "Invalid data environment closing arguments.");
10326 
10327     llvm::Value *BasePointersArrayArg = nullptr;
10328     llvm::Value *PointersArrayArg = nullptr;
10329     llvm::Value *SizesArrayArg = nullptr;
10330     llvm::Value *MapTypesArrayArg = nullptr;
10331     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10332                                  SizesArrayArg, MapTypesArrayArg, Info);
10333 
10334     // Emit device ID if any.
10335     llvm::Value *DeviceID = nullptr;
10336     if (Device) {
10337       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10338                                            CGF.Int64Ty, /*isSigned=*/true);
10339     } else {
10340       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10341     }
10342 
10343     // Emit the number of elements in the offloading arrays.
10344     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10345 
10346     llvm::Value *OffloadingArgs[] = {
10347         DeviceID,         PointerNum,    BasePointersArrayArg,
10348         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10349     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end),
10350                         OffloadingArgs);
10351   };
10352 
10353   // If we need device pointer privatization, we need to emit the body of the
10354   // region with no privatization in the 'else' branch of the conditional.
10355   // Otherwise, we don't have to do anything.
10356   auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF,
10357                                                          PrePostActionTy &) {
10358     if (!Info.CaptureDeviceAddrMap.empty()) {
10359       CodeGen.setAction(NoPrivAction);
10360       CodeGen(CGF);
10361     }
10362   };
10363 
10364   // We don't have to do anything to close the region if the if clause evaluates
10365   // to false.
10366   auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {};
10367 
10368   if (IfCond) {
10369     emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen);
10370   } else {
10371     RegionCodeGenTy RCG(BeginThenGen);
10372     RCG(CGF);
10373   }
10374 
10375   // If we don't require privatization of device pointers, we emit the body in
10376   // between the runtime calls. This avoids duplicating the body code.
10377   if (Info.CaptureDeviceAddrMap.empty()) {
10378     CodeGen.setAction(NoPrivAction);
10379     CodeGen(CGF);
10380   }
10381 
10382   if (IfCond) {
10383     emitIfClause(CGF, IfCond, EndThenGen, EndElseGen);
10384   } else {
10385     RegionCodeGenTy RCG(EndThenGen);
10386     RCG(CGF);
10387   }
10388 }
10389 
10390 void CGOpenMPRuntime::emitTargetDataStandAloneCall(
10391     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
10392     const Expr *Device) {
10393   if (!CGF.HaveInsertPoint())
10394     return;
10395 
10396   assert((isa<OMPTargetEnterDataDirective>(D) ||
10397           isa<OMPTargetExitDataDirective>(D) ||
10398           isa<OMPTargetUpdateDirective>(D)) &&
10399          "Expecting either target enter, exit data, or update directives.");
10400 
10401   CodeGenFunction::OMPTargetDataInfo InputInfo;
10402   llvm::Value *MapTypesArray = nullptr;
10403   // Generate the code for the opening of the data environment.
10404   auto &&ThenGen = [this, &D, Device, &InputInfo,
10405                     &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) {
10406     // Emit device ID if any.
10407     llvm::Value *DeviceID = nullptr;
10408     if (Device) {
10409       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10410                                            CGF.Int64Ty, /*isSigned=*/true);
10411     } else {
10412       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10413     }
10414 
10415     // Emit the number of elements in the offloading arrays.
10416     llvm::Constant *PointerNum =
10417         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
10418 
10419     llvm::Value *OffloadingArgs[] = {DeviceID,
10420                                      PointerNum,
10421                                      InputInfo.BasePointersArray.getPointer(),
10422                                      InputInfo.PointersArray.getPointer(),
10423                                      InputInfo.SizesArray.getPointer(),
10424                                      MapTypesArray};
10425 
10426     // Select the right runtime function call for each expected standalone
10427     // directive.
10428     const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
10429     OpenMPRTLFunction RTLFn;
10430     switch (D.getDirectiveKind()) {
10431     case OMPD_target_enter_data:
10432       RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait
10433                         : OMPRTL__tgt_target_data_begin;
10434       break;
10435     case OMPD_target_exit_data:
10436       RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait
10437                         : OMPRTL__tgt_target_data_end;
10438       break;
10439     case OMPD_target_update:
10440       RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait
10441                         : OMPRTL__tgt_target_data_update;
10442       break;
10443     case OMPD_parallel:
10444     case OMPD_for:
10445     case OMPD_parallel_for:
10446     case OMPD_parallel_master:
10447     case OMPD_parallel_sections:
10448     case OMPD_for_simd:
10449     case OMPD_parallel_for_simd:
10450     case OMPD_cancel:
10451     case OMPD_cancellation_point:
10452     case OMPD_ordered:
10453     case OMPD_threadprivate:
10454     case OMPD_allocate:
10455     case OMPD_task:
10456     case OMPD_simd:
10457     case OMPD_sections:
10458     case OMPD_section:
10459     case OMPD_single:
10460     case OMPD_master:
10461     case OMPD_critical:
10462     case OMPD_taskyield:
10463     case OMPD_barrier:
10464     case OMPD_taskwait:
10465     case OMPD_taskgroup:
10466     case OMPD_atomic:
10467     case OMPD_flush:
10468     case OMPD_depobj:
10469     case OMPD_scan:
10470     case OMPD_teams:
10471     case OMPD_target_data:
10472     case OMPD_distribute:
10473     case OMPD_distribute_simd:
10474     case OMPD_distribute_parallel_for:
10475     case OMPD_distribute_parallel_for_simd:
10476     case OMPD_teams_distribute:
10477     case OMPD_teams_distribute_simd:
10478     case OMPD_teams_distribute_parallel_for:
10479     case OMPD_teams_distribute_parallel_for_simd:
10480     case OMPD_declare_simd:
10481     case OMPD_declare_variant:
10482     case OMPD_begin_declare_variant:
10483     case OMPD_end_declare_variant:
10484     case OMPD_declare_target:
10485     case OMPD_end_declare_target:
10486     case OMPD_declare_reduction:
10487     case OMPD_declare_mapper:
10488     case OMPD_taskloop:
10489     case OMPD_taskloop_simd:
10490     case OMPD_master_taskloop:
10491     case OMPD_master_taskloop_simd:
10492     case OMPD_parallel_master_taskloop:
10493     case OMPD_parallel_master_taskloop_simd:
10494     case OMPD_target:
10495     case OMPD_target_simd:
10496     case OMPD_target_teams_distribute:
10497     case OMPD_target_teams_distribute_simd:
10498     case OMPD_target_teams_distribute_parallel_for:
10499     case OMPD_target_teams_distribute_parallel_for_simd:
10500     case OMPD_target_teams:
10501     case OMPD_target_parallel:
10502     case OMPD_target_parallel_for:
10503     case OMPD_target_parallel_for_simd:
10504     case OMPD_requires:
10505     case OMPD_unknown:
10506       llvm_unreachable("Unexpected standalone target data directive.");
10507       break;
10508     }
10509     CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs);
10510   };
10511 
10512   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray](
10513                              CodeGenFunction &CGF, PrePostActionTy &) {
10514     // Fill up the arrays with all the mapped variables.
10515     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
10516     MappableExprsHandler::MapValuesArrayTy Pointers;
10517     MappableExprsHandler::MapValuesArrayTy Sizes;
10518     MappableExprsHandler::MapFlagsArrayTy MapTypes;
10519 
10520     // Get map clause information.
10521     MappableExprsHandler MEHandler(D, CGF);
10522     MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
10523 
10524     TargetDataInfo Info;
10525     // Fill up the arrays and create the arguments.
10526     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
10527     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
10528                                  Info.PointersArray, Info.SizesArray,
10529                                  Info.MapTypesArray, Info);
10530     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
10531     InputInfo.BasePointersArray =
10532         Address(Info.BasePointersArray, CGM.getPointerAlign());
10533     InputInfo.PointersArray =
10534         Address(Info.PointersArray, CGM.getPointerAlign());
10535     InputInfo.SizesArray =
10536         Address(Info.SizesArray, CGM.getPointerAlign());
10537     MapTypesArray = Info.MapTypesArray;
10538     if (D.hasClausesOfKind<OMPDependClause>())
10539       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
10540     else
10541       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
10542   };
10543 
10544   if (IfCond) {
10545     emitIfClause(CGF, IfCond, TargetThenGen,
10546                  [](CodeGenFunction &CGF, PrePostActionTy &) {});
10547   } else {
10548     RegionCodeGenTy ThenRCG(TargetThenGen);
10549     ThenRCG(CGF);
10550   }
10551 }
10552 
10553 namespace {
10554   /// Kind of parameter in a function with 'declare simd' directive.
10555   enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector };
10556   /// Attribute set of the parameter.
10557   struct ParamAttrTy {
10558     ParamKindTy Kind = Vector;
10559     llvm::APSInt StrideOrArg;
10560     llvm::APSInt Alignment;
10561   };
10562 } // namespace
10563 
10564 static unsigned evaluateCDTSize(const FunctionDecl *FD,
10565                                 ArrayRef<ParamAttrTy> ParamAttrs) {
10566   // Every vector variant of a SIMD-enabled function has a vector length (VLEN).
10567   // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument
10568   // of that clause. The VLEN value must be power of 2.
10569   // In other case the notion of the function`s "characteristic data type" (CDT)
10570   // is used to compute the vector length.
10571   // CDT is defined in the following order:
10572   //   a) For non-void function, the CDT is the return type.
10573   //   b) If the function has any non-uniform, non-linear parameters, then the
10574   //   CDT is the type of the first such parameter.
10575   //   c) If the CDT determined by a) or b) above is struct, union, or class
10576   //   type which is pass-by-value (except for the type that maps to the
10577   //   built-in complex data type), the characteristic data type is int.
10578   //   d) If none of the above three cases is applicable, the CDT is int.
10579   // The VLEN is then determined based on the CDT and the size of vector
10580   // register of that ISA for which current vector version is generated. The
10581   // VLEN is computed using the formula below:
10582   //   VLEN  = sizeof(vector_register) / sizeof(CDT),
10583   // where vector register size specified in section 3.2.1 Registers and the
10584   // Stack Frame of original AMD64 ABI document.
10585   QualType RetType = FD->getReturnType();
10586   if (RetType.isNull())
10587     return 0;
10588   ASTContext &C = FD->getASTContext();
10589   QualType CDT;
10590   if (!RetType.isNull() && !RetType->isVoidType()) {
10591     CDT = RetType;
10592   } else {
10593     unsigned Offset = 0;
10594     if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) {
10595       if (ParamAttrs[Offset].Kind == Vector)
10596         CDT = C.getPointerType(C.getRecordType(MD->getParent()));
10597       ++Offset;
10598     }
10599     if (CDT.isNull()) {
10600       for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10601         if (ParamAttrs[I + Offset].Kind == Vector) {
10602           CDT = FD->getParamDecl(I)->getType();
10603           break;
10604         }
10605       }
10606     }
10607   }
10608   if (CDT.isNull())
10609     CDT = C.IntTy;
10610   CDT = CDT->getCanonicalTypeUnqualified();
10611   if (CDT->isRecordType() || CDT->isUnionType())
10612     CDT = C.IntTy;
10613   return C.getTypeSize(CDT);
10614 }
10615 
10616 static void
10617 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn,
10618                            const llvm::APSInt &VLENVal,
10619                            ArrayRef<ParamAttrTy> ParamAttrs,
10620                            OMPDeclareSimdDeclAttr::BranchStateTy State) {
10621   struct ISADataTy {
10622     char ISA;
10623     unsigned VecRegSize;
10624   };
10625   ISADataTy ISAData[] = {
10626       {
10627           'b', 128
10628       }, // SSE
10629       {
10630           'c', 256
10631       }, // AVX
10632       {
10633           'd', 256
10634       }, // AVX2
10635       {
10636           'e', 512
10637       }, // AVX512
10638   };
10639   llvm::SmallVector<char, 2> Masked;
10640   switch (State) {
10641   case OMPDeclareSimdDeclAttr::BS_Undefined:
10642     Masked.push_back('N');
10643     Masked.push_back('M');
10644     break;
10645   case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10646     Masked.push_back('N');
10647     break;
10648   case OMPDeclareSimdDeclAttr::BS_Inbranch:
10649     Masked.push_back('M');
10650     break;
10651   }
10652   for (char Mask : Masked) {
10653     for (const ISADataTy &Data : ISAData) {
10654       SmallString<256> Buffer;
10655       llvm::raw_svector_ostream Out(Buffer);
10656       Out << "_ZGV" << Data.ISA << Mask;
10657       if (!VLENVal) {
10658         unsigned NumElts = evaluateCDTSize(FD, ParamAttrs);
10659         assert(NumElts && "Non-zero simdlen/cdtsize expected");
10660         Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts);
10661       } else {
10662         Out << VLENVal;
10663       }
10664       for (const ParamAttrTy &ParamAttr : ParamAttrs) {
10665         switch (ParamAttr.Kind){
10666         case LinearWithVarStride:
10667           Out << 's' << ParamAttr.StrideOrArg;
10668           break;
10669         case Linear:
10670           Out << 'l';
10671           if (!!ParamAttr.StrideOrArg)
10672             Out << ParamAttr.StrideOrArg;
10673           break;
10674         case Uniform:
10675           Out << 'u';
10676           break;
10677         case Vector:
10678           Out << 'v';
10679           break;
10680         }
10681         if (!!ParamAttr.Alignment)
10682           Out << 'a' << ParamAttr.Alignment;
10683       }
10684       Out << '_' << Fn->getName();
10685       Fn->addFnAttr(Out.str());
10686     }
10687   }
10688 }
10689 
10690 // This are the Functions that are needed to mangle the name of the
10691 // vector functions generated by the compiler, according to the rules
10692 // defined in the "Vector Function ABI specifications for AArch64",
10693 // available at
10694 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi.
10695 
10696 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI.
10697 ///
10698 /// TODO: Need to implement the behavior for reference marked with a
10699 /// var or no linear modifiers (1.b in the section). For this, we
10700 /// need to extend ParamKindTy to support the linear modifiers.
10701 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) {
10702   QT = QT.getCanonicalType();
10703 
10704   if (QT->isVoidType())
10705     return false;
10706 
10707   if (Kind == ParamKindTy::Uniform)
10708     return false;
10709 
10710   if (Kind == ParamKindTy::Linear)
10711     return false;
10712 
10713   // TODO: Handle linear references with modifiers
10714 
10715   if (Kind == ParamKindTy::LinearWithVarStride)
10716     return false;
10717 
10718   return true;
10719 }
10720 
10721 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
10722 static bool getAArch64PBV(QualType QT, ASTContext &C) {
10723   QT = QT.getCanonicalType();
10724   unsigned Size = C.getTypeSize(QT);
10725 
10726   // Only scalars and complex within 16 bytes wide set PVB to true.
10727   if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128)
10728     return false;
10729 
10730   if (QT->isFloatingType())
10731     return true;
10732 
10733   if (QT->isIntegerType())
10734     return true;
10735 
10736   if (QT->isPointerType())
10737     return true;
10738 
10739   // TODO: Add support for complex types (section 3.1.2, item 2).
10740 
10741   return false;
10742 }
10743 
10744 /// Computes the lane size (LS) of a return type or of an input parameter,
10745 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI.
10746 /// TODO: Add support for references, section 3.2.1, item 1.
10747 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) {
10748   if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) {
10749     QualType PTy = QT.getCanonicalType()->getPointeeType();
10750     if (getAArch64PBV(PTy, C))
10751       return C.getTypeSize(PTy);
10752   }
10753   if (getAArch64PBV(QT, C))
10754     return C.getTypeSize(QT);
10755 
10756   return C.getTypeSize(C.getUIntPtrType());
10757 }
10758 
10759 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the
10760 // signature of the scalar function, as defined in 3.2.2 of the
10761 // AAVFABI.
10762 static std::tuple<unsigned, unsigned, bool>
10763 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) {
10764   QualType RetType = FD->getReturnType().getCanonicalType();
10765 
10766   ASTContext &C = FD->getASTContext();
10767 
10768   bool OutputBecomesInput = false;
10769 
10770   llvm::SmallVector<unsigned, 8> Sizes;
10771   if (!RetType->isVoidType()) {
10772     Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C));
10773     if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {}))
10774       OutputBecomesInput = true;
10775   }
10776   for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10777     QualType QT = FD->getParamDecl(I)->getType().getCanonicalType();
10778     Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C));
10779   }
10780 
10781   assert(!Sizes.empty() && "Unable to determine NDS and WDS.");
10782   // The LS of a function parameter / return value can only be a power
10783   // of 2, starting from 8 bits, up to 128.
10784   assert(std::all_of(Sizes.begin(), Sizes.end(),
10785                      [](unsigned Size) {
10786                        return Size == 8 || Size == 16 || Size == 32 ||
10787                               Size == 64 || Size == 128;
10788                      }) &&
10789          "Invalid size");
10790 
10791   return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)),
10792                          *std::max_element(std::begin(Sizes), std::end(Sizes)),
10793                          OutputBecomesInput);
10794 }
10795 
10796 /// Mangle the parameter part of the vector function name according to
10797 /// their OpenMP classification. The mangling function is defined in
10798 /// section 3.5 of the AAVFABI.
10799 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) {
10800   SmallString<256> Buffer;
10801   llvm::raw_svector_ostream Out(Buffer);
10802   for (const auto &ParamAttr : ParamAttrs) {
10803     switch (ParamAttr.Kind) {
10804     case LinearWithVarStride:
10805       Out << "ls" << ParamAttr.StrideOrArg;
10806       break;
10807     case Linear:
10808       Out << 'l';
10809       // Don't print the step value if it is not present or if it is
10810       // equal to 1.
10811       if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1)
10812         Out << ParamAttr.StrideOrArg;
10813       break;
10814     case Uniform:
10815       Out << 'u';
10816       break;
10817     case Vector:
10818       Out << 'v';
10819       break;
10820     }
10821 
10822     if (!!ParamAttr.Alignment)
10823       Out << 'a' << ParamAttr.Alignment;
10824   }
10825 
10826   return std::string(Out.str());
10827 }
10828 
10829 // Function used to add the attribute. The parameter `VLEN` is
10830 // templated to allow the use of "x" when targeting scalable functions
10831 // for SVE.
10832 template <typename T>
10833 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix,
10834                                  char ISA, StringRef ParSeq,
10835                                  StringRef MangledName, bool OutputBecomesInput,
10836                                  llvm::Function *Fn) {
10837   SmallString<256> Buffer;
10838   llvm::raw_svector_ostream Out(Buffer);
10839   Out << Prefix << ISA << LMask << VLEN;
10840   if (OutputBecomesInput)
10841     Out << "v";
10842   Out << ParSeq << "_" << MangledName;
10843   Fn->addFnAttr(Out.str());
10844 }
10845 
10846 // Helper function to generate the Advanced SIMD names depending on
10847 // the value of the NDS when simdlen is not present.
10848 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask,
10849                                       StringRef Prefix, char ISA,
10850                                       StringRef ParSeq, StringRef MangledName,
10851                                       bool OutputBecomesInput,
10852                                       llvm::Function *Fn) {
10853   switch (NDS) {
10854   case 8:
10855     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10856                          OutputBecomesInput, Fn);
10857     addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName,
10858                          OutputBecomesInput, Fn);
10859     break;
10860   case 16:
10861     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10862                          OutputBecomesInput, Fn);
10863     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10864                          OutputBecomesInput, Fn);
10865     break;
10866   case 32:
10867     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10868                          OutputBecomesInput, Fn);
10869     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10870                          OutputBecomesInput, Fn);
10871     break;
10872   case 64:
10873   case 128:
10874     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10875                          OutputBecomesInput, Fn);
10876     break;
10877   default:
10878     llvm_unreachable("Scalar type is too wide.");
10879   }
10880 }
10881 
10882 /// Emit vector function attributes for AArch64, as defined in the AAVFABI.
10883 static void emitAArch64DeclareSimdFunction(
10884     CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN,
10885     ArrayRef<ParamAttrTy> ParamAttrs,
10886     OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName,
10887     char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) {
10888 
10889   // Get basic data for building the vector signature.
10890   const auto Data = getNDSWDS(FD, ParamAttrs);
10891   const unsigned NDS = std::get<0>(Data);
10892   const unsigned WDS = std::get<1>(Data);
10893   const bool OutputBecomesInput = std::get<2>(Data);
10894 
10895   // Check the values provided via `simdlen` by the user.
10896   // 1. A `simdlen(1)` doesn't produce vector signatures,
10897   if (UserVLEN == 1) {
10898     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10899         DiagnosticsEngine::Warning,
10900         "The clause simdlen(1) has no effect when targeting aarch64.");
10901     CGM.getDiags().Report(SLoc, DiagID);
10902     return;
10903   }
10904 
10905   // 2. Section 3.3.1, item 1: user input must be a power of 2 for
10906   // Advanced SIMD output.
10907   if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) {
10908     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10909         DiagnosticsEngine::Warning, "The value specified in simdlen must be a "
10910                                     "power of 2 when targeting Advanced SIMD.");
10911     CGM.getDiags().Report(SLoc, DiagID);
10912     return;
10913   }
10914 
10915   // 3. Section 3.4.1. SVE fixed lengh must obey the architectural
10916   // limits.
10917   if (ISA == 's' && UserVLEN != 0) {
10918     if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) {
10919       unsigned DiagID = CGM.getDiags().getCustomDiagID(
10920           DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit "
10921                                       "lanes in the architectural constraints "
10922                                       "for SVE (min is 128-bit, max is "
10923                                       "2048-bit, by steps of 128-bit)");
10924       CGM.getDiags().Report(SLoc, DiagID) << WDS;
10925       return;
10926     }
10927   }
10928 
10929   // Sort out parameter sequence.
10930   const std::string ParSeq = mangleVectorParameters(ParamAttrs);
10931   StringRef Prefix = "_ZGV";
10932   // Generate simdlen from user input (if any).
10933   if (UserVLEN) {
10934     if (ISA == 's') {
10935       // SVE generates only a masked function.
10936       addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10937                            OutputBecomesInput, Fn);
10938     } else {
10939       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10940       // Advanced SIMD generates one or two functions, depending on
10941       // the `[not]inbranch` clause.
10942       switch (State) {
10943       case OMPDeclareSimdDeclAttr::BS_Undefined:
10944         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10945                              OutputBecomesInput, Fn);
10946         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10947                              OutputBecomesInput, Fn);
10948         break;
10949       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10950         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10951                              OutputBecomesInput, Fn);
10952         break;
10953       case OMPDeclareSimdDeclAttr::BS_Inbranch:
10954         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10955                              OutputBecomesInput, Fn);
10956         break;
10957       }
10958     }
10959   } else {
10960     // If no user simdlen is provided, follow the AAVFABI rules for
10961     // generating the vector length.
10962     if (ISA == 's') {
10963       // SVE, section 3.4.1, item 1.
10964       addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName,
10965                            OutputBecomesInput, Fn);
10966     } else {
10967       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10968       // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or
10969       // two vector names depending on the use of the clause
10970       // `[not]inbranch`.
10971       switch (State) {
10972       case OMPDeclareSimdDeclAttr::BS_Undefined:
10973         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10974                                   OutputBecomesInput, Fn);
10975         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
10976                                   OutputBecomesInput, Fn);
10977         break;
10978       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10979         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10980                                   OutputBecomesInput, Fn);
10981         break;
10982       case OMPDeclareSimdDeclAttr::BS_Inbranch:
10983         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
10984                                   OutputBecomesInput, Fn);
10985         break;
10986       }
10987     }
10988   }
10989 }
10990 
10991 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD,
10992                                               llvm::Function *Fn) {
10993   ASTContext &C = CGM.getContext();
10994   FD = FD->getMostRecentDecl();
10995   // Map params to their positions in function decl.
10996   llvm::DenseMap<const Decl *, unsigned> ParamPositions;
10997   if (isa<CXXMethodDecl>(FD))
10998     ParamPositions.try_emplace(FD, 0);
10999   unsigned ParamPos = ParamPositions.size();
11000   for (const ParmVarDecl *P : FD->parameters()) {
11001     ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos);
11002     ++ParamPos;
11003   }
11004   while (FD) {
11005     for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) {
11006       llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size());
11007       // Mark uniform parameters.
11008       for (const Expr *E : Attr->uniforms()) {
11009         E = E->IgnoreParenImpCasts();
11010         unsigned Pos;
11011         if (isa<CXXThisExpr>(E)) {
11012           Pos = ParamPositions[FD];
11013         } else {
11014           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
11015                                 ->getCanonicalDecl();
11016           Pos = ParamPositions[PVD];
11017         }
11018         ParamAttrs[Pos].Kind = Uniform;
11019       }
11020       // Get alignment info.
11021       auto NI = Attr->alignments_begin();
11022       for (const Expr *E : Attr->aligneds()) {
11023         E = E->IgnoreParenImpCasts();
11024         unsigned Pos;
11025         QualType ParmTy;
11026         if (isa<CXXThisExpr>(E)) {
11027           Pos = ParamPositions[FD];
11028           ParmTy = E->getType();
11029         } else {
11030           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
11031                                 ->getCanonicalDecl();
11032           Pos = ParamPositions[PVD];
11033           ParmTy = PVD->getType();
11034         }
11035         ParamAttrs[Pos].Alignment =
11036             (*NI)
11037                 ? (*NI)->EvaluateKnownConstInt(C)
11038                 : llvm::APSInt::getUnsigned(
11039                       C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy))
11040                           .getQuantity());
11041         ++NI;
11042       }
11043       // Mark linear parameters.
11044       auto SI = Attr->steps_begin();
11045       auto MI = Attr->modifiers_begin();
11046       for (const Expr *E : Attr->linears()) {
11047         E = E->IgnoreParenImpCasts();
11048         unsigned Pos;
11049         if (isa<CXXThisExpr>(E)) {
11050           Pos = ParamPositions[FD];
11051         } else {
11052           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
11053                                 ->getCanonicalDecl();
11054           Pos = ParamPositions[PVD];
11055         }
11056         ParamAttrTy &ParamAttr = ParamAttrs[Pos];
11057         ParamAttr.Kind = Linear;
11058         if (*SI) {
11059           Expr::EvalResult Result;
11060           if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) {
11061             if (const auto *DRE =
11062                     cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) {
11063               if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) {
11064                 ParamAttr.Kind = LinearWithVarStride;
11065                 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(
11066                     ParamPositions[StridePVD->getCanonicalDecl()]);
11067               }
11068             }
11069           } else {
11070             ParamAttr.StrideOrArg = Result.Val.getInt();
11071           }
11072         }
11073         ++SI;
11074         ++MI;
11075       }
11076       llvm::APSInt VLENVal;
11077       SourceLocation ExprLoc;
11078       const Expr *VLENExpr = Attr->getSimdlen();
11079       if (VLENExpr) {
11080         VLENVal = VLENExpr->EvaluateKnownConstInt(C);
11081         ExprLoc = VLENExpr->getExprLoc();
11082       }
11083       OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState();
11084       if (CGM.getTriple().isX86()) {
11085         emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State);
11086       } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) {
11087         unsigned VLEN = VLENVal.getExtValue();
11088         StringRef MangledName = Fn->getName();
11089         if (CGM.getTarget().hasFeature("sve"))
11090           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
11091                                          MangledName, 's', 128, Fn, ExprLoc);
11092         if (CGM.getTarget().hasFeature("neon"))
11093           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
11094                                          MangledName, 'n', 128, Fn, ExprLoc);
11095       }
11096     }
11097     FD = FD->getPreviousDecl();
11098   }
11099 }
11100 
11101 namespace {
11102 /// Cleanup action for doacross support.
11103 class DoacrossCleanupTy final : public EHScopeStack::Cleanup {
11104 public:
11105   static const int DoacrossFinArgs = 2;
11106 
11107 private:
11108   llvm::FunctionCallee RTLFn;
11109   llvm::Value *Args[DoacrossFinArgs];
11110 
11111 public:
11112   DoacrossCleanupTy(llvm::FunctionCallee RTLFn,
11113                     ArrayRef<llvm::Value *> CallArgs)
11114       : RTLFn(RTLFn) {
11115     assert(CallArgs.size() == DoacrossFinArgs);
11116     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
11117   }
11118   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
11119     if (!CGF.HaveInsertPoint())
11120       return;
11121     CGF.EmitRuntimeCall(RTLFn, Args);
11122   }
11123 };
11124 } // namespace
11125 
11126 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF,
11127                                        const OMPLoopDirective &D,
11128                                        ArrayRef<Expr *> NumIterations) {
11129   if (!CGF.HaveInsertPoint())
11130     return;
11131 
11132   ASTContext &C = CGM.getContext();
11133   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
11134   RecordDecl *RD;
11135   if (KmpDimTy.isNull()) {
11136     // Build struct kmp_dim {  // loop bounds info casted to kmp_int64
11137     //  kmp_int64 lo; // lower
11138     //  kmp_int64 up; // upper
11139     //  kmp_int64 st; // stride
11140     // };
11141     RD = C.buildImplicitRecord("kmp_dim");
11142     RD->startDefinition();
11143     addFieldToRecordDecl(C, RD, Int64Ty);
11144     addFieldToRecordDecl(C, RD, Int64Ty);
11145     addFieldToRecordDecl(C, RD, Int64Ty);
11146     RD->completeDefinition();
11147     KmpDimTy = C.getRecordType(RD);
11148   } else {
11149     RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl());
11150   }
11151   llvm::APInt Size(/*numBits=*/32, NumIterations.size());
11152   QualType ArrayTy =
11153       C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0);
11154 
11155   Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims");
11156   CGF.EmitNullInitialization(DimsAddr, ArrayTy);
11157   enum { LowerFD = 0, UpperFD, StrideFD };
11158   // Fill dims with data.
11159   for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) {
11160     LValue DimsLVal = CGF.MakeAddrLValue(
11161         CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy);
11162     // dims.upper = num_iterations;
11163     LValue UpperLVal = CGF.EmitLValueForField(
11164         DimsLVal, *std::next(RD->field_begin(), UpperFD));
11165     llvm::Value *NumIterVal =
11166         CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]),
11167                                  D.getNumIterations()->getType(), Int64Ty,
11168                                  D.getNumIterations()->getExprLoc());
11169     CGF.EmitStoreOfScalar(NumIterVal, UpperLVal);
11170     // dims.stride = 1;
11171     LValue StrideLVal = CGF.EmitLValueForField(
11172         DimsLVal, *std::next(RD->field_begin(), StrideFD));
11173     CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1),
11174                           StrideLVal);
11175   }
11176 
11177   // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid,
11178   // kmp_int32 num_dims, struct kmp_dim * dims);
11179   llvm::Value *Args[] = {
11180       emitUpdateLocation(CGF, D.getBeginLoc()),
11181       getThreadID(CGF, D.getBeginLoc()),
11182       llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()),
11183       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11184           CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(),
11185           CGM.VoidPtrTy)};
11186 
11187   llvm::FunctionCallee RTLFn =
11188       createRuntimeFunction(OMPRTL__kmpc_doacross_init);
11189   CGF.EmitRuntimeCall(RTLFn, Args);
11190   llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = {
11191       emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())};
11192   llvm::FunctionCallee FiniRTLFn =
11193       createRuntimeFunction(OMPRTL__kmpc_doacross_fini);
11194   CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
11195                                              llvm::makeArrayRef(FiniArgs));
11196 }
11197 
11198 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
11199                                           const OMPDependClause *C) {
11200   QualType Int64Ty =
11201       CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
11202   llvm::APInt Size(/*numBits=*/32, C->getNumLoops());
11203   QualType ArrayTy = CGM.getContext().getConstantArrayType(
11204       Int64Ty, Size, nullptr, ArrayType::Normal, 0);
11205   Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr");
11206   for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) {
11207     const Expr *CounterVal = C->getLoopData(I);
11208     assert(CounterVal);
11209     llvm::Value *CntVal = CGF.EmitScalarConversion(
11210         CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty,
11211         CounterVal->getExprLoc());
11212     CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I),
11213                           /*Volatile=*/false, Int64Ty);
11214   }
11215   llvm::Value *Args[] = {
11216       emitUpdateLocation(CGF, C->getBeginLoc()),
11217       getThreadID(CGF, C->getBeginLoc()),
11218       CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()};
11219   llvm::FunctionCallee RTLFn;
11220   if (C->getDependencyKind() == OMPC_DEPEND_source) {
11221     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post);
11222   } else {
11223     assert(C->getDependencyKind() == OMPC_DEPEND_sink);
11224     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait);
11225   }
11226   CGF.EmitRuntimeCall(RTLFn, Args);
11227 }
11228 
11229 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc,
11230                                llvm::FunctionCallee Callee,
11231                                ArrayRef<llvm::Value *> Args) const {
11232   assert(Loc.isValid() && "Outlined function call location must be valid.");
11233   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
11234 
11235   if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) {
11236     if (Fn->doesNotThrow()) {
11237       CGF.EmitNounwindRuntimeCall(Fn, Args);
11238       return;
11239     }
11240   }
11241   CGF.EmitRuntimeCall(Callee, Args);
11242 }
11243 
11244 void CGOpenMPRuntime::emitOutlinedFunctionCall(
11245     CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn,
11246     ArrayRef<llvm::Value *> Args) const {
11247   emitCall(CGF, Loc, OutlinedFn, Args);
11248 }
11249 
11250 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) {
11251   if (const auto *FD = dyn_cast<FunctionDecl>(D))
11252     if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD))
11253       HasEmittedDeclareTargetRegion = true;
11254 }
11255 
11256 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF,
11257                                              const VarDecl *NativeParam,
11258                                              const VarDecl *TargetParam) const {
11259   return CGF.GetAddrOfLocalVar(NativeParam);
11260 }
11261 
11262 namespace {
11263 /// Cleanup action for allocate support.
11264 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup {
11265 public:
11266   static const int CleanupArgs = 3;
11267 
11268 private:
11269   llvm::FunctionCallee RTLFn;
11270   llvm::Value *Args[CleanupArgs];
11271 
11272 public:
11273   OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn,
11274                        ArrayRef<llvm::Value *> CallArgs)
11275       : RTLFn(RTLFn) {
11276     assert(CallArgs.size() == CleanupArgs &&
11277            "Size of arguments does not match.");
11278     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
11279   }
11280   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
11281     if (!CGF.HaveInsertPoint())
11282       return;
11283     CGF.EmitRuntimeCall(RTLFn, Args);
11284   }
11285 };
11286 } // namespace
11287 
11288 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF,
11289                                                    const VarDecl *VD) {
11290   if (!VD)
11291     return Address::invalid();
11292   const VarDecl *CVD = VD->getCanonicalDecl();
11293   if (!CVD->hasAttr<OMPAllocateDeclAttr>())
11294     return Address::invalid();
11295   const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
11296   // Use the default allocation.
11297   if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc &&
11298       !AA->getAllocator())
11299     return Address::invalid();
11300   llvm::Value *Size;
11301   CharUnits Align = CGM.getContext().getDeclAlign(CVD);
11302   if (CVD->getType()->isVariablyModifiedType()) {
11303     Size = CGF.getTypeSize(CVD->getType());
11304     // Align the size: ((size + align - 1) / align) * align
11305     Size = CGF.Builder.CreateNUWAdd(
11306         Size, CGM.getSize(Align - CharUnits::fromQuantity(1)));
11307     Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align));
11308     Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align));
11309   } else {
11310     CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType());
11311     Size = CGM.getSize(Sz.alignTo(Align));
11312   }
11313   llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc());
11314   assert(AA->getAllocator() &&
11315          "Expected allocator expression for non-default allocator.");
11316   llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator());
11317   // According to the standard, the original allocator type is a enum (integer).
11318   // Convert to pointer type, if required.
11319   if (Allocator->getType()->isIntegerTy())
11320     Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy);
11321   else if (Allocator->getType()->isPointerTy())
11322     Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator,
11323                                                                 CGM.VoidPtrTy);
11324   llvm::Value *Args[] = {ThreadID, Size, Allocator};
11325 
11326   llvm::Value *Addr =
11327       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args,
11328                           getName({CVD->getName(), ".void.addr"}));
11329   llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr,
11330                                                               Allocator};
11331   llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free);
11332 
11333   CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
11334                                                 llvm::makeArrayRef(FiniArgs));
11335   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11336       Addr,
11337       CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())),
11338       getName({CVD->getName(), ".addr"}));
11339   return Address(Addr, Align);
11340 }
11341 
11342 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII(
11343     CodeGenModule &CGM, const OMPLoopDirective &S)
11344     : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) {
11345   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11346   if (!NeedToPush)
11347     return;
11348   NontemporalDeclsSet &DS =
11349       CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back();
11350   for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) {
11351     for (const Stmt *Ref : C->private_refs()) {
11352       const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts();
11353       const ValueDecl *VD;
11354       if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) {
11355         VD = DRE->getDecl();
11356       } else {
11357         const auto *ME = cast<MemberExpr>(SimpleRefExpr);
11358         assert((ME->isImplicitCXXThis() ||
11359                 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) &&
11360                "Expected member of current class.");
11361         VD = ME->getMemberDecl();
11362       }
11363       DS.insert(VD);
11364     }
11365   }
11366 }
11367 
11368 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() {
11369   if (!NeedToPush)
11370     return;
11371   CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back();
11372 }
11373 
11374 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const {
11375   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11376 
11377   return llvm::any_of(
11378       CGM.getOpenMPRuntime().NontemporalDeclsStack,
11379       [VD](const NontemporalDeclsSet &Set) { return Set.count(VD) > 0; });
11380 }
11381 
11382 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis(
11383     const OMPExecutableDirective &S,
11384     llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled)
11385     const {
11386   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs;
11387   // Vars in target/task regions must be excluded completely.
11388   if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) ||
11389       isOpenMPTaskingDirective(S.getDirectiveKind())) {
11390     SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
11391     getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind());
11392     const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front());
11393     for (const CapturedStmt::Capture &Cap : CS->captures()) {
11394       if (Cap.capturesVariable() || Cap.capturesVariableByCopy())
11395         NeedToCheckForLPCs.insert(Cap.getCapturedVar());
11396     }
11397   }
11398   // Exclude vars in private clauses.
11399   for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) {
11400     for (const Expr *Ref : C->varlists()) {
11401       if (!Ref->getType()->isScalarType())
11402         continue;
11403       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11404       if (!DRE)
11405         continue;
11406       NeedToCheckForLPCs.insert(DRE->getDecl());
11407     }
11408   }
11409   for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
11410     for (const Expr *Ref : C->varlists()) {
11411       if (!Ref->getType()->isScalarType())
11412         continue;
11413       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11414       if (!DRE)
11415         continue;
11416       NeedToCheckForLPCs.insert(DRE->getDecl());
11417     }
11418   }
11419   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
11420     for (const Expr *Ref : C->varlists()) {
11421       if (!Ref->getType()->isScalarType())
11422         continue;
11423       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11424       if (!DRE)
11425         continue;
11426       NeedToCheckForLPCs.insert(DRE->getDecl());
11427     }
11428   }
11429   for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
11430     for (const Expr *Ref : C->varlists()) {
11431       if (!Ref->getType()->isScalarType())
11432         continue;
11433       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11434       if (!DRE)
11435         continue;
11436       NeedToCheckForLPCs.insert(DRE->getDecl());
11437     }
11438   }
11439   for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) {
11440     for (const Expr *Ref : C->varlists()) {
11441       if (!Ref->getType()->isScalarType())
11442         continue;
11443       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
11444       if (!DRE)
11445         continue;
11446       NeedToCheckForLPCs.insert(DRE->getDecl());
11447     }
11448   }
11449   for (const Decl *VD : NeedToCheckForLPCs) {
11450     for (const LastprivateConditionalData &Data :
11451          llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) {
11452       if (Data.DeclToUniqueName.count(VD) > 0) {
11453         if (!Data.Disabled)
11454           NeedToAddForLPCsAsDisabled.insert(VD);
11455         break;
11456       }
11457     }
11458   }
11459 }
11460 
11461 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
11462     CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal)
11463     : CGM(CGF.CGM),
11464       Action((CGM.getLangOpts().OpenMP >= 50 &&
11465               llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(),
11466                            [](const OMPLastprivateClause *C) {
11467                              return C->getKind() ==
11468                                     OMPC_LASTPRIVATE_conditional;
11469                            }))
11470                  ? ActionToDo::PushAsLastprivateConditional
11471                  : ActionToDo::DoNotPush) {
11472   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11473   if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush)
11474     return;
11475   assert(Action == ActionToDo::PushAsLastprivateConditional &&
11476          "Expected a push action.");
11477   LastprivateConditionalData &Data =
11478       CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
11479   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
11480     if (C->getKind() != OMPC_LASTPRIVATE_conditional)
11481       continue;
11482 
11483     for (const Expr *Ref : C->varlists()) {
11484       Data.DeclToUniqueName.insert(std::make_pair(
11485           cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(),
11486           SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref))));
11487     }
11488   }
11489   Data.IVLVal = IVLVal;
11490   Data.Fn = CGF.CurFn;
11491 }
11492 
11493 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
11494     CodeGenFunction &CGF, const OMPExecutableDirective &S)
11495     : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) {
11496   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
11497   if (CGM.getLangOpts().OpenMP < 50)
11498     return;
11499   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled;
11500   tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled);
11501   if (!NeedToAddForLPCsAsDisabled.empty()) {
11502     Action = ActionToDo::DisableLastprivateConditional;
11503     LastprivateConditionalData &Data =
11504         CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
11505     for (const Decl *VD : NeedToAddForLPCsAsDisabled)
11506       Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>()));
11507     Data.Fn = CGF.CurFn;
11508     Data.Disabled = true;
11509   }
11510 }
11511 
11512 CGOpenMPRuntime::LastprivateConditionalRAII
11513 CGOpenMPRuntime::LastprivateConditionalRAII::disable(
11514     CodeGenFunction &CGF, const OMPExecutableDirective &S) {
11515   return LastprivateConditionalRAII(CGF, S);
11516 }
11517 
11518 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() {
11519   if (CGM.getLangOpts().OpenMP < 50)
11520     return;
11521   if (Action == ActionToDo::DisableLastprivateConditional) {
11522     assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
11523            "Expected list of disabled private vars.");
11524     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
11525   }
11526   if (Action == ActionToDo::PushAsLastprivateConditional) {
11527     assert(
11528         !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
11529         "Expected list of lastprivate conditional vars.");
11530     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
11531   }
11532 }
11533 
11534 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF,
11535                                                         const VarDecl *VD) {
11536   ASTContext &C = CGM.getContext();
11537   auto I = LastprivateConditionalToTypes.find(CGF.CurFn);
11538   if (I == LastprivateConditionalToTypes.end())
11539     I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first;
11540   QualType NewType;
11541   const FieldDecl *VDField;
11542   const FieldDecl *FiredField;
11543   LValue BaseLVal;
11544   auto VI = I->getSecond().find(VD);
11545   if (VI == I->getSecond().end()) {
11546     RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional");
11547     RD->startDefinition();
11548     VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType());
11549     FiredField = addFieldToRecordDecl(C, RD, C.CharTy);
11550     RD->completeDefinition();
11551     NewType = C.getRecordType(RD);
11552     Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName());
11553     BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl);
11554     I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal);
11555   } else {
11556     NewType = std::get<0>(VI->getSecond());
11557     VDField = std::get<1>(VI->getSecond());
11558     FiredField = std::get<2>(VI->getSecond());
11559     BaseLVal = std::get<3>(VI->getSecond());
11560   }
11561   LValue FiredLVal =
11562       CGF.EmitLValueForField(BaseLVal, FiredField);
11563   CGF.EmitStoreOfScalar(
11564       llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)),
11565       FiredLVal);
11566   return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF);
11567 }
11568 
11569 namespace {
11570 /// Checks if the lastprivate conditional variable is referenced in LHS.
11571 class LastprivateConditionalRefChecker final
11572     : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> {
11573   ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM;
11574   const Expr *FoundE = nullptr;
11575   const Decl *FoundD = nullptr;
11576   StringRef UniqueDeclName;
11577   LValue IVLVal;
11578   llvm::Function *FoundFn = nullptr;
11579   SourceLocation Loc;
11580 
11581 public:
11582   bool VisitDeclRefExpr(const DeclRefExpr *E) {
11583     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
11584          llvm::reverse(LPM)) {
11585       auto It = D.DeclToUniqueName.find(E->getDecl());
11586       if (It == D.DeclToUniqueName.end())
11587         continue;
11588       if (D.Disabled)
11589         return false;
11590       FoundE = E;
11591       FoundD = E->getDecl()->getCanonicalDecl();
11592       UniqueDeclName = It->second;
11593       IVLVal = D.IVLVal;
11594       FoundFn = D.Fn;
11595       break;
11596     }
11597     return FoundE == E;
11598   }
11599   bool VisitMemberExpr(const MemberExpr *E) {
11600     if (!CodeGenFunction::IsWrappedCXXThis(E->getBase()))
11601       return false;
11602     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
11603          llvm::reverse(LPM)) {
11604       auto It = D.DeclToUniqueName.find(E->getMemberDecl());
11605       if (It == D.DeclToUniqueName.end())
11606         continue;
11607       if (D.Disabled)
11608         return false;
11609       FoundE = E;
11610       FoundD = E->getMemberDecl()->getCanonicalDecl();
11611       UniqueDeclName = It->second;
11612       IVLVal = D.IVLVal;
11613       FoundFn = D.Fn;
11614       break;
11615     }
11616     return FoundE == E;
11617   }
11618   bool VisitStmt(const Stmt *S) {
11619     for (const Stmt *Child : S->children()) {
11620       if (!Child)
11621         continue;
11622       if (const auto *E = dyn_cast<Expr>(Child))
11623         if (!E->isGLValue())
11624           continue;
11625       if (Visit(Child))
11626         return true;
11627     }
11628     return false;
11629   }
11630   explicit LastprivateConditionalRefChecker(
11631       ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM)
11632       : LPM(LPM) {}
11633   std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *>
11634   getFoundData() const {
11635     return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn);
11636   }
11637 };
11638 } // namespace
11639 
11640 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF,
11641                                                        LValue IVLVal,
11642                                                        StringRef UniqueDeclName,
11643                                                        LValue LVal,
11644                                                        SourceLocation Loc) {
11645   // Last updated loop counter for the lastprivate conditional var.
11646   // int<xx> last_iv = 0;
11647   llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType());
11648   llvm::Constant *LastIV =
11649       getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"}));
11650   cast<llvm::GlobalVariable>(LastIV)->setAlignment(
11651       IVLVal.getAlignment().getAsAlign());
11652   LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType());
11653 
11654   // Last value of the lastprivate conditional.
11655   // decltype(priv_a) last_a;
11656   llvm::Constant *Last = getOrCreateInternalVariable(
11657       CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName);
11658   cast<llvm::GlobalVariable>(Last)->setAlignment(
11659       LVal.getAlignment().getAsAlign());
11660   LValue LastLVal =
11661       CGF.MakeAddrLValue(Last, LVal.getType(), LVal.getAlignment());
11662 
11663   // Global loop counter. Required to handle inner parallel-for regions.
11664   // iv
11665   llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc);
11666 
11667   // #pragma omp critical(a)
11668   // if (last_iv <= iv) {
11669   //   last_iv = iv;
11670   //   last_a = priv_a;
11671   // }
11672   auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal,
11673                     Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
11674     Action.Enter(CGF);
11675     llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc);
11676     // (last_iv <= iv) ? Check if the variable is updated and store new
11677     // value in global var.
11678     llvm::Value *CmpRes;
11679     if (IVLVal.getType()->isSignedIntegerType()) {
11680       CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal);
11681     } else {
11682       assert(IVLVal.getType()->isUnsignedIntegerType() &&
11683              "Loop iteration variable must be integer.");
11684       CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal);
11685     }
11686     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then");
11687     llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit");
11688     CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB);
11689     // {
11690     CGF.EmitBlock(ThenBB);
11691 
11692     //   last_iv = iv;
11693     CGF.EmitStoreOfScalar(IVVal, LastIVLVal);
11694 
11695     //   last_a = priv_a;
11696     switch (CGF.getEvaluationKind(LVal.getType())) {
11697     case TEK_Scalar: {
11698       llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc);
11699       CGF.EmitStoreOfScalar(PrivVal, LastLVal);
11700       break;
11701     }
11702     case TEK_Complex: {
11703       CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc);
11704       CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false);
11705       break;
11706     }
11707     case TEK_Aggregate:
11708       llvm_unreachable(
11709           "Aggregates are not supported in lastprivate conditional.");
11710     }
11711     // }
11712     CGF.EmitBranch(ExitBB);
11713     // There is no need to emit line number for unconditional branch.
11714     (void)ApplyDebugLocation::CreateEmpty(CGF);
11715     CGF.EmitBlock(ExitBB, /*IsFinished=*/true);
11716   };
11717 
11718   if (CGM.getLangOpts().OpenMPSimd) {
11719     // Do not emit as a critical region as no parallel region could be emitted.
11720     RegionCodeGenTy ThenRCG(CodeGen);
11721     ThenRCG(CGF);
11722   } else {
11723     emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc);
11724   }
11725 }
11726 
11727 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF,
11728                                                          const Expr *LHS) {
11729   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
11730     return;
11731   LastprivateConditionalRefChecker Checker(LastprivateConditionalStack);
11732   if (!Checker.Visit(LHS))
11733     return;
11734   const Expr *FoundE;
11735   const Decl *FoundD;
11736   StringRef UniqueDeclName;
11737   LValue IVLVal;
11738   llvm::Function *FoundFn;
11739   std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) =
11740       Checker.getFoundData();
11741   if (FoundFn != CGF.CurFn) {
11742     // Special codegen for inner parallel regions.
11743     // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1;
11744     auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD);
11745     assert(It != LastprivateConditionalToTypes[FoundFn].end() &&
11746            "Lastprivate conditional is not found in outer region.");
11747     QualType StructTy = std::get<0>(It->getSecond());
11748     const FieldDecl* FiredDecl = std::get<2>(It->getSecond());
11749     LValue PrivLVal = CGF.EmitLValue(FoundE);
11750     Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11751         PrivLVal.getAddress(CGF),
11752         CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy)));
11753     LValue BaseLVal =
11754         CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl);
11755     LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl);
11756     CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get(
11757                             CGF.ConvertTypeForMem(FiredDecl->getType()), 1)),
11758                         FiredLVal, llvm::AtomicOrdering::Unordered,
11759                         /*IsVolatile=*/true, /*isInit=*/false);
11760     return;
11761   }
11762 
11763   // Private address of the lastprivate conditional in the current context.
11764   // priv_a
11765   LValue LVal = CGF.EmitLValue(FoundE);
11766   emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal,
11767                                    FoundE->getExprLoc());
11768 }
11769 
11770 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional(
11771     CodeGenFunction &CGF, const OMPExecutableDirective &D,
11772     const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) {
11773   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
11774     return;
11775   auto Range = llvm::reverse(LastprivateConditionalStack);
11776   auto It = llvm::find_if(
11777       Range, [](const LastprivateConditionalData &D) { return !D.Disabled; });
11778   if (It == Range.end() || It->Fn != CGF.CurFn)
11779     return;
11780   auto LPCI = LastprivateConditionalToTypes.find(It->Fn);
11781   assert(LPCI != LastprivateConditionalToTypes.end() &&
11782          "Lastprivates must be registered already.");
11783   SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
11784   getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind());
11785   const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back());
11786   for (const auto &Pair : It->DeclToUniqueName) {
11787     const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl());
11788     if (!CS->capturesVariable(VD) || IgnoredDecls.count(VD) > 0)
11789       continue;
11790     auto I = LPCI->getSecond().find(Pair.first);
11791     assert(I != LPCI->getSecond().end() &&
11792            "Lastprivate must be rehistered already.");
11793     // bool Cmp = priv_a.Fired != 0;
11794     LValue BaseLVal = std::get<3>(I->getSecond());
11795     LValue FiredLVal =
11796         CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond()));
11797     llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc());
11798     llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res);
11799     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then");
11800     llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done");
11801     // if (Cmp) {
11802     CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB);
11803     CGF.EmitBlock(ThenBB);
11804     Address Addr = CGF.GetAddrOfLocalVar(VD);
11805     LValue LVal;
11806     if (VD->getType()->isReferenceType())
11807       LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(),
11808                                            AlignmentSource::Decl);
11809     else
11810       LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(),
11811                                 AlignmentSource::Decl);
11812     emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal,
11813                                      D.getBeginLoc());
11814     auto AL = ApplyDebugLocation::CreateArtificial(CGF);
11815     CGF.EmitBlock(DoneBB, /*IsFinal=*/true);
11816     // }
11817   }
11818 }
11819 
11820 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate(
11821     CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD,
11822     SourceLocation Loc) {
11823   if (CGF.getLangOpts().OpenMP < 50)
11824     return;
11825   auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD);
11826   assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() &&
11827          "Unknown lastprivate conditional variable.");
11828   StringRef UniqueName = It->second;
11829   llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName);
11830   // The variable was not updated in the region - exit.
11831   if (!GV)
11832     return;
11833   LValue LPLVal = CGF.MakeAddrLValue(
11834       GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment());
11835   llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc);
11836   CGF.EmitStoreOfScalar(Res, PrivLVal);
11837 }
11838 
11839 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction(
11840     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11841     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11842   llvm_unreachable("Not supported in SIMD-only mode");
11843 }
11844 
11845 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction(
11846     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11847     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11848   llvm_unreachable("Not supported in SIMD-only mode");
11849 }
11850 
11851 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction(
11852     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11853     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
11854     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
11855     bool Tied, unsigned &NumberOfParts) {
11856   llvm_unreachable("Not supported in SIMD-only mode");
11857 }
11858 
11859 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF,
11860                                            SourceLocation Loc,
11861                                            llvm::Function *OutlinedFn,
11862                                            ArrayRef<llvm::Value *> CapturedVars,
11863                                            const Expr *IfCond) {
11864   llvm_unreachable("Not supported in SIMD-only mode");
11865 }
11866 
11867 void CGOpenMPSIMDRuntime::emitCriticalRegion(
11868     CodeGenFunction &CGF, StringRef CriticalName,
11869     const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc,
11870     const Expr *Hint) {
11871   llvm_unreachable("Not supported in SIMD-only mode");
11872 }
11873 
11874 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF,
11875                                            const RegionCodeGenTy &MasterOpGen,
11876                                            SourceLocation Loc) {
11877   llvm_unreachable("Not supported in SIMD-only mode");
11878 }
11879 
11880 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
11881                                             SourceLocation Loc) {
11882   llvm_unreachable("Not supported in SIMD-only mode");
11883 }
11884 
11885 void CGOpenMPSIMDRuntime::emitTaskgroupRegion(
11886     CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen,
11887     SourceLocation Loc) {
11888   llvm_unreachable("Not supported in SIMD-only mode");
11889 }
11890 
11891 void CGOpenMPSIMDRuntime::emitSingleRegion(
11892     CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen,
11893     SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars,
11894     ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs,
11895     ArrayRef<const Expr *> AssignmentOps) {
11896   llvm_unreachable("Not supported in SIMD-only mode");
11897 }
11898 
11899 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF,
11900                                             const RegionCodeGenTy &OrderedOpGen,
11901                                             SourceLocation Loc,
11902                                             bool IsThreads) {
11903   llvm_unreachable("Not supported in SIMD-only mode");
11904 }
11905 
11906 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF,
11907                                           SourceLocation Loc,
11908                                           OpenMPDirectiveKind Kind,
11909                                           bool EmitChecks,
11910                                           bool ForceSimpleCall) {
11911   llvm_unreachable("Not supported in SIMD-only mode");
11912 }
11913 
11914 void CGOpenMPSIMDRuntime::emitForDispatchInit(
11915     CodeGenFunction &CGF, SourceLocation Loc,
11916     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
11917     bool Ordered, const DispatchRTInput &DispatchValues) {
11918   llvm_unreachable("Not supported in SIMD-only mode");
11919 }
11920 
11921 void CGOpenMPSIMDRuntime::emitForStaticInit(
11922     CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind,
11923     const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) {
11924   llvm_unreachable("Not supported in SIMD-only mode");
11925 }
11926 
11927 void CGOpenMPSIMDRuntime::emitDistributeStaticInit(
11928     CodeGenFunction &CGF, SourceLocation Loc,
11929     OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) {
11930   llvm_unreachable("Not supported in SIMD-only mode");
11931 }
11932 
11933 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
11934                                                      SourceLocation Loc,
11935                                                      unsigned IVSize,
11936                                                      bool IVSigned) {
11937   llvm_unreachable("Not supported in SIMD-only mode");
11938 }
11939 
11940 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF,
11941                                               SourceLocation Loc,
11942                                               OpenMPDirectiveKind DKind) {
11943   llvm_unreachable("Not supported in SIMD-only mode");
11944 }
11945 
11946 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF,
11947                                               SourceLocation Loc,
11948                                               unsigned IVSize, bool IVSigned,
11949                                               Address IL, Address LB,
11950                                               Address UB, Address ST) {
11951   llvm_unreachable("Not supported in SIMD-only mode");
11952 }
11953 
11954 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
11955                                                llvm::Value *NumThreads,
11956                                                SourceLocation Loc) {
11957   llvm_unreachable("Not supported in SIMD-only mode");
11958 }
11959 
11960 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF,
11961                                              ProcBindKind ProcBind,
11962                                              SourceLocation Loc) {
11963   llvm_unreachable("Not supported in SIMD-only mode");
11964 }
11965 
11966 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
11967                                                     const VarDecl *VD,
11968                                                     Address VDAddr,
11969                                                     SourceLocation Loc) {
11970   llvm_unreachable("Not supported in SIMD-only mode");
11971 }
11972 
11973 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition(
11974     const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit,
11975     CodeGenFunction *CGF) {
11976   llvm_unreachable("Not supported in SIMD-only mode");
11977 }
11978 
11979 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate(
11980     CodeGenFunction &CGF, QualType VarType, StringRef Name) {
11981   llvm_unreachable("Not supported in SIMD-only mode");
11982 }
11983 
11984 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF,
11985                                     ArrayRef<const Expr *> Vars,
11986                                     SourceLocation Loc,
11987                                     llvm::AtomicOrdering AO) {
11988   llvm_unreachable("Not supported in SIMD-only mode");
11989 }
11990 
11991 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
11992                                        const OMPExecutableDirective &D,
11993                                        llvm::Function *TaskFunction,
11994                                        QualType SharedsTy, Address Shareds,
11995                                        const Expr *IfCond,
11996                                        const OMPTaskDataTy &Data) {
11997   llvm_unreachable("Not supported in SIMD-only mode");
11998 }
11999 
12000 void CGOpenMPSIMDRuntime::emitTaskLoopCall(
12001     CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D,
12002     llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds,
12003     const Expr *IfCond, const OMPTaskDataTy &Data) {
12004   llvm_unreachable("Not supported in SIMD-only mode");
12005 }
12006 
12007 void CGOpenMPSIMDRuntime::emitReduction(
12008     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates,
12009     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
12010     ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) {
12011   assert(Options.SimpleReduction && "Only simple reduction is expected.");
12012   CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs,
12013                                  ReductionOps, Options);
12014 }
12015 
12016 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit(
12017     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
12018     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
12019   llvm_unreachable("Not supported in SIMD-only mode");
12020 }
12021 
12022 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
12023                                                   SourceLocation Loc,
12024                                                   ReductionCodeGen &RCG,
12025                                                   unsigned N) {
12026   llvm_unreachable("Not supported in SIMD-only mode");
12027 }
12028 
12029 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF,
12030                                                   SourceLocation Loc,
12031                                                   llvm::Value *ReductionsPtr,
12032                                                   LValue SharedLVal) {
12033   llvm_unreachable("Not supported in SIMD-only mode");
12034 }
12035 
12036 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
12037                                            SourceLocation Loc) {
12038   llvm_unreachable("Not supported in SIMD-only mode");
12039 }
12040 
12041 void CGOpenMPSIMDRuntime::emitCancellationPointCall(
12042     CodeGenFunction &CGF, SourceLocation Loc,
12043     OpenMPDirectiveKind CancelRegion) {
12044   llvm_unreachable("Not supported in SIMD-only mode");
12045 }
12046 
12047 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF,
12048                                          SourceLocation Loc, const Expr *IfCond,
12049                                          OpenMPDirectiveKind CancelRegion) {
12050   llvm_unreachable("Not supported in SIMD-only mode");
12051 }
12052 
12053 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction(
12054     const OMPExecutableDirective &D, StringRef ParentName,
12055     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
12056     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
12057   llvm_unreachable("Not supported in SIMD-only mode");
12058 }
12059 
12060 void CGOpenMPSIMDRuntime::emitTargetCall(
12061     CodeGenFunction &CGF, const OMPExecutableDirective &D,
12062     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
12063     llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
12064     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
12065                                      const OMPLoopDirective &D)>
12066         SizeEmitter) {
12067   llvm_unreachable("Not supported in SIMD-only mode");
12068 }
12069 
12070 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) {
12071   llvm_unreachable("Not supported in SIMD-only mode");
12072 }
12073 
12074 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
12075   llvm_unreachable("Not supported in SIMD-only mode");
12076 }
12077 
12078 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) {
12079   return false;
12080 }
12081 
12082 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF,
12083                                         const OMPExecutableDirective &D,
12084                                         SourceLocation Loc,
12085                                         llvm::Function *OutlinedFn,
12086                                         ArrayRef<llvm::Value *> CapturedVars) {
12087   llvm_unreachable("Not supported in SIMD-only mode");
12088 }
12089 
12090 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
12091                                              const Expr *NumTeams,
12092                                              const Expr *ThreadLimit,
12093                                              SourceLocation Loc) {
12094   llvm_unreachable("Not supported in SIMD-only mode");
12095 }
12096 
12097 void CGOpenMPSIMDRuntime::emitTargetDataCalls(
12098     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12099     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
12100   llvm_unreachable("Not supported in SIMD-only mode");
12101 }
12102 
12103 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall(
12104     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
12105     const Expr *Device) {
12106   llvm_unreachable("Not supported in SIMD-only mode");
12107 }
12108 
12109 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF,
12110                                            const OMPLoopDirective &D,
12111                                            ArrayRef<Expr *> NumIterations) {
12112   llvm_unreachable("Not supported in SIMD-only mode");
12113 }
12114 
12115 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
12116                                               const OMPDependClause *C) {
12117   llvm_unreachable("Not supported in SIMD-only mode");
12118 }
12119 
12120 const VarDecl *
12121 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD,
12122                                         const VarDecl *NativeParam) const {
12123   llvm_unreachable("Not supported in SIMD-only mode");
12124 }
12125 
12126 Address
12127 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF,
12128                                          const VarDecl *NativeParam,
12129                                          const VarDecl *TargetParam) const {
12130   llvm_unreachable("Not supported in SIMD-only mode");
12131 }
12132