1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This provides a class for OpenMP runtime code generation.
10 //
11 //===----------------------------------------------------------------------===//
12 
13 #include "CGCXXABI.h"
14 #include "CGCleanup.h"
15 #include "CGOpenMPRuntime.h"
16 #include "CGRecordLayout.h"
17 #include "CodeGenFunction.h"
18 #include "clang/CodeGen/ConstantInitBuilder.h"
19 #include "clang/AST/Decl.h"
20 #include "clang/AST/StmtOpenMP.h"
21 #include "clang/Basic/BitmaskEnum.h"
22 #include "llvm/ADT/ArrayRef.h"
23 #include "llvm/Bitcode/BitcodeReader.h"
24 #include "llvm/IR/DerivedTypes.h"
25 #include "llvm/IR/GlobalValue.h"
26 #include "llvm/IR/Value.h"
27 #include "llvm/Support/Format.h"
28 #include "llvm/Support/raw_ostream.h"
29 #include <cassert>
30 
31 using namespace clang;
32 using namespace CodeGen;
33 
34 namespace {
35 /// Base class for handling code generation inside OpenMP regions.
36 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
37 public:
38   /// Kinds of OpenMP regions used in codegen.
39   enum CGOpenMPRegionKind {
40     /// Region with outlined function for standalone 'parallel'
41     /// directive.
42     ParallelOutlinedRegion,
43     /// Region with outlined function for standalone 'task' directive.
44     TaskOutlinedRegion,
45     /// Region for constructs that do not require function outlining,
46     /// like 'for', 'sections', 'atomic' etc. directives.
47     InlinedRegion,
48     /// Region with outlined function for standalone 'target' directive.
49     TargetRegion,
50   };
51 
52   CGOpenMPRegionInfo(const CapturedStmt &CS,
53                      const CGOpenMPRegionKind RegionKind,
54                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
55                      bool HasCancel)
56       : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
57         CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {}
58 
59   CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
60                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
61                      bool HasCancel)
62       : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen),
63         Kind(Kind), HasCancel(HasCancel) {}
64 
65   /// Get a variable or parameter for storing global thread id
66   /// inside OpenMP construct.
67   virtual const VarDecl *getThreadIDVariable() const = 0;
68 
69   /// Emit the captured statement body.
70   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
71 
72   /// Get an LValue for the current ThreadID variable.
73   /// \return LValue for thread id variable. This LValue always has type int32*.
74   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
75 
76   virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {}
77 
78   CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
79 
80   OpenMPDirectiveKind getDirectiveKind() const { return Kind; }
81 
82   bool hasCancel() const { return HasCancel; }
83 
84   static bool classof(const CGCapturedStmtInfo *Info) {
85     return Info->getKind() == CR_OpenMP;
86   }
87 
88   ~CGOpenMPRegionInfo() override = default;
89 
90 protected:
91   CGOpenMPRegionKind RegionKind;
92   RegionCodeGenTy CodeGen;
93   OpenMPDirectiveKind Kind;
94   bool HasCancel;
95 };
96 
97 /// API for captured statement code generation in OpenMP constructs.
98 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo {
99 public:
100   CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
101                              const RegionCodeGenTy &CodeGen,
102                              OpenMPDirectiveKind Kind, bool HasCancel,
103                              StringRef HelperName)
104       : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind,
105                            HasCancel),
106         ThreadIDVar(ThreadIDVar), HelperName(HelperName) {
107     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
108   }
109 
110   /// Get a variable or parameter for storing global thread id
111   /// inside OpenMP construct.
112   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
113 
114   /// Get the name of the capture helper.
115   StringRef getHelperName() const override { return HelperName; }
116 
117   static bool classof(const CGCapturedStmtInfo *Info) {
118     return CGOpenMPRegionInfo::classof(Info) &&
119            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
120                ParallelOutlinedRegion;
121   }
122 
123 private:
124   /// A variable or parameter storing global thread id for OpenMP
125   /// constructs.
126   const VarDecl *ThreadIDVar;
127   StringRef HelperName;
128 };
129 
130 /// API for captured statement code generation in OpenMP constructs.
131 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo {
132 public:
133   class UntiedTaskActionTy final : public PrePostActionTy {
134     bool Untied;
135     const VarDecl *PartIDVar;
136     const RegionCodeGenTy UntiedCodeGen;
137     llvm::SwitchInst *UntiedSwitch = nullptr;
138 
139   public:
140     UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar,
141                        const RegionCodeGenTy &UntiedCodeGen)
142         : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {}
143     void Enter(CodeGenFunction &CGF) override {
144       if (Untied) {
145         // Emit task switching point.
146         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
147             CGF.GetAddrOfLocalVar(PartIDVar),
148             PartIDVar->getType()->castAs<PointerType>());
149         llvm::Value *Res =
150             CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation());
151         llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done.");
152         UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB);
153         CGF.EmitBlock(DoneBB);
154         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
155         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
156         UntiedSwitch->addCase(CGF.Builder.getInt32(0),
157                               CGF.Builder.GetInsertBlock());
158         emitUntiedSwitch(CGF);
159       }
160     }
161     void emitUntiedSwitch(CodeGenFunction &CGF) const {
162       if (Untied) {
163         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
164             CGF.GetAddrOfLocalVar(PartIDVar),
165             PartIDVar->getType()->castAs<PointerType>());
166         CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
167                               PartIdLVal);
168         UntiedCodeGen(CGF);
169         CodeGenFunction::JumpDest CurPoint =
170             CGF.getJumpDestInCurrentScope(".untied.next.");
171         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
172         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
173         UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
174                               CGF.Builder.GetInsertBlock());
175         CGF.EmitBranchThroughCleanup(CurPoint);
176         CGF.EmitBlock(CurPoint.getBlock());
177       }
178     }
179     unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); }
180   };
181   CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
182                                  const VarDecl *ThreadIDVar,
183                                  const RegionCodeGenTy &CodeGen,
184                                  OpenMPDirectiveKind Kind, bool HasCancel,
185                                  const UntiedTaskActionTy &Action)
186       : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel),
187         ThreadIDVar(ThreadIDVar), Action(Action) {
188     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
189   }
190 
191   /// Get a variable or parameter for storing global thread id
192   /// inside OpenMP construct.
193   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
194 
195   /// Get an LValue for the current ThreadID variable.
196   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
197 
198   /// Get the name of the capture helper.
199   StringRef getHelperName() const override { return ".omp_outlined."; }
200 
201   void emitUntiedSwitch(CodeGenFunction &CGF) override {
202     Action.emitUntiedSwitch(CGF);
203   }
204 
205   static bool classof(const CGCapturedStmtInfo *Info) {
206     return CGOpenMPRegionInfo::classof(Info) &&
207            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
208                TaskOutlinedRegion;
209   }
210 
211 private:
212   /// A variable or parameter storing global thread id for OpenMP
213   /// constructs.
214   const VarDecl *ThreadIDVar;
215   /// Action for emitting code for untied tasks.
216   const UntiedTaskActionTy &Action;
217 };
218 
219 /// API for inlined captured statement code generation in OpenMP
220 /// constructs.
221 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
222 public:
223   CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
224                             const RegionCodeGenTy &CodeGen,
225                             OpenMPDirectiveKind Kind, bool HasCancel)
226       : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel),
227         OldCSI(OldCSI),
228         OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
229 
230   // Retrieve the value of the context parameter.
231   llvm::Value *getContextValue() const override {
232     if (OuterRegionInfo)
233       return OuterRegionInfo->getContextValue();
234     llvm_unreachable("No context value for inlined OpenMP region");
235   }
236 
237   void setContextValue(llvm::Value *V) override {
238     if (OuterRegionInfo) {
239       OuterRegionInfo->setContextValue(V);
240       return;
241     }
242     llvm_unreachable("No context value for inlined OpenMP region");
243   }
244 
245   /// Lookup the captured field decl for a variable.
246   const FieldDecl *lookup(const VarDecl *VD) const override {
247     if (OuterRegionInfo)
248       return OuterRegionInfo->lookup(VD);
249     // If there is no outer outlined region,no need to lookup in a list of
250     // captured variables, we can use the original one.
251     return nullptr;
252   }
253 
254   FieldDecl *getThisFieldDecl() const override {
255     if (OuterRegionInfo)
256       return OuterRegionInfo->getThisFieldDecl();
257     return nullptr;
258   }
259 
260   /// Get a variable or parameter for storing global thread id
261   /// inside OpenMP construct.
262   const VarDecl *getThreadIDVariable() const override {
263     if (OuterRegionInfo)
264       return OuterRegionInfo->getThreadIDVariable();
265     return nullptr;
266   }
267 
268   /// Get an LValue for the current ThreadID variable.
269   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override {
270     if (OuterRegionInfo)
271       return OuterRegionInfo->getThreadIDVariableLValue(CGF);
272     llvm_unreachable("No LValue for inlined OpenMP construct");
273   }
274 
275   /// Get the name of the capture helper.
276   StringRef getHelperName() const override {
277     if (auto *OuterRegionInfo = getOldCSI())
278       return OuterRegionInfo->getHelperName();
279     llvm_unreachable("No helper name for inlined OpenMP construct");
280   }
281 
282   void emitUntiedSwitch(CodeGenFunction &CGF) override {
283     if (OuterRegionInfo)
284       OuterRegionInfo->emitUntiedSwitch(CGF);
285   }
286 
287   CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
288 
289   static bool classof(const CGCapturedStmtInfo *Info) {
290     return CGOpenMPRegionInfo::classof(Info) &&
291            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion;
292   }
293 
294   ~CGOpenMPInlinedRegionInfo() override = default;
295 
296 private:
297   /// CodeGen info about outer OpenMP region.
298   CodeGenFunction::CGCapturedStmtInfo *OldCSI;
299   CGOpenMPRegionInfo *OuterRegionInfo;
300 };
301 
302 /// API for captured statement code generation in OpenMP target
303 /// constructs. For this captures, implicit parameters are used instead of the
304 /// captured fields. The name of the target region has to be unique in a given
305 /// application so it is provided by the client, because only the client has
306 /// the information to generate that.
307 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo {
308 public:
309   CGOpenMPTargetRegionInfo(const CapturedStmt &CS,
310                            const RegionCodeGenTy &CodeGen, StringRef HelperName)
311       : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target,
312                            /*HasCancel=*/false),
313         HelperName(HelperName) {}
314 
315   /// This is unused for target regions because each starts executing
316   /// with a single thread.
317   const VarDecl *getThreadIDVariable() const override { return nullptr; }
318 
319   /// Get the name of the capture helper.
320   StringRef getHelperName() const override { return HelperName; }
321 
322   static bool classof(const CGCapturedStmtInfo *Info) {
323     return CGOpenMPRegionInfo::classof(Info) &&
324            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion;
325   }
326 
327 private:
328   StringRef HelperName;
329 };
330 
331 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) {
332   llvm_unreachable("No codegen for expressions");
333 }
334 /// API for generation of expressions captured in a innermost OpenMP
335 /// region.
336 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo {
337 public:
338   CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS)
339       : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen,
340                                   OMPD_unknown,
341                                   /*HasCancel=*/false),
342         PrivScope(CGF) {
343     // Make sure the globals captured in the provided statement are local by
344     // using the privatization logic. We assume the same variable is not
345     // captured more than once.
346     for (const auto &C : CS.captures()) {
347       if (!C.capturesVariable() && !C.capturesVariableByCopy())
348         continue;
349 
350       const VarDecl *VD = C.getCapturedVar();
351       if (VD->isLocalVarDeclOrParm())
352         continue;
353 
354       DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
355                       /*RefersToEnclosingVariableOrCapture=*/false,
356                       VD->getType().getNonReferenceType(), VK_LValue,
357                       C.getLocation());
358       PrivScope.addPrivate(
359           VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(); });
360     }
361     (void)PrivScope.Privatize();
362   }
363 
364   /// Lookup the captured field decl for a variable.
365   const FieldDecl *lookup(const VarDecl *VD) const override {
366     if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD))
367       return FD;
368     return nullptr;
369   }
370 
371   /// Emit the captured statement body.
372   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override {
373     llvm_unreachable("No body for expressions");
374   }
375 
376   /// Get a variable or parameter for storing global thread id
377   /// inside OpenMP construct.
378   const VarDecl *getThreadIDVariable() const override {
379     llvm_unreachable("No thread id for expressions");
380   }
381 
382   /// Get the name of the capture helper.
383   StringRef getHelperName() const override {
384     llvm_unreachable("No helper name for expressions");
385   }
386 
387   static bool classof(const CGCapturedStmtInfo *Info) { return false; }
388 
389 private:
390   /// Private scope to capture global variables.
391   CodeGenFunction::OMPPrivateScope PrivScope;
392 };
393 
394 /// RAII for emitting code of OpenMP constructs.
395 class InlinedOpenMPRegionRAII {
396   CodeGenFunction &CGF;
397   llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields;
398   FieldDecl *LambdaThisCaptureField = nullptr;
399   const CodeGen::CGBlockInfo *BlockInfo = nullptr;
400 
401 public:
402   /// Constructs region for combined constructs.
403   /// \param CodeGen Code generation sequence for combined directives. Includes
404   /// a list of functions used for code generation of implicitly inlined
405   /// regions.
406   InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen,
407                           OpenMPDirectiveKind Kind, bool HasCancel)
408       : CGF(CGF) {
409     // Start emission for the construct.
410     CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(
411         CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel);
412     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
413     LambdaThisCaptureField = CGF.LambdaThisCaptureField;
414     CGF.LambdaThisCaptureField = nullptr;
415     BlockInfo = CGF.BlockInfo;
416     CGF.BlockInfo = nullptr;
417   }
418 
419   ~InlinedOpenMPRegionRAII() {
420     // Restore original CapturedStmtInfo only if we're done with code emission.
421     auto *OldCSI =
422         cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
423     delete CGF.CapturedStmtInfo;
424     CGF.CapturedStmtInfo = OldCSI;
425     std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
426     CGF.LambdaThisCaptureField = LambdaThisCaptureField;
427     CGF.BlockInfo = BlockInfo;
428   }
429 };
430 
431 /// Values for bit flags used in the ident_t to describe the fields.
432 /// All enumeric elements are named and described in accordance with the code
433 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
434 enum OpenMPLocationFlags : unsigned {
435   /// Use trampoline for internal microtask.
436   OMP_IDENT_IMD = 0x01,
437   /// Use c-style ident structure.
438   OMP_IDENT_KMPC = 0x02,
439   /// Atomic reduction option for kmpc_reduce.
440   OMP_ATOMIC_REDUCE = 0x10,
441   /// Explicit 'barrier' directive.
442   OMP_IDENT_BARRIER_EXPL = 0x20,
443   /// Implicit barrier in code.
444   OMP_IDENT_BARRIER_IMPL = 0x40,
445   /// Implicit barrier in 'for' directive.
446   OMP_IDENT_BARRIER_IMPL_FOR = 0x40,
447   /// Implicit barrier in 'sections' directive.
448   OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0,
449   /// Implicit barrier in 'single' directive.
450   OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140,
451   /// Call of __kmp_for_static_init for static loop.
452   OMP_IDENT_WORK_LOOP = 0x200,
453   /// Call of __kmp_for_static_init for sections.
454   OMP_IDENT_WORK_SECTIONS = 0x400,
455   /// Call of __kmp_for_static_init for distribute.
456   OMP_IDENT_WORK_DISTRIBUTE = 0x800,
457   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE)
458 };
459 
460 namespace {
461 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
462 /// Values for bit flags for marking which requires clauses have been used.
463 enum OpenMPOffloadingRequiresDirFlags : int64_t {
464   /// flag undefined.
465   OMP_REQ_UNDEFINED               = 0x000,
466   /// no requires clause present.
467   OMP_REQ_NONE                    = 0x001,
468   /// reverse_offload clause.
469   OMP_REQ_REVERSE_OFFLOAD         = 0x002,
470   /// unified_address clause.
471   OMP_REQ_UNIFIED_ADDRESS         = 0x004,
472   /// unified_shared_memory clause.
473   OMP_REQ_UNIFIED_SHARED_MEMORY   = 0x008,
474   /// dynamic_allocators clause.
475   OMP_REQ_DYNAMIC_ALLOCATORS      = 0x010,
476   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS)
477 };
478 
479 enum OpenMPOffloadingReservedDeviceIDs {
480   /// Device ID if the device was not defined, runtime should get it
481   /// from environment variables in the spec.
482   OMP_DEVICEID_UNDEF = -1,
483 };
484 } // anonymous namespace
485 
486 /// Describes ident structure that describes a source location.
487 /// All descriptions are taken from
488 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h
489 /// Original structure:
490 /// typedef struct ident {
491 ///    kmp_int32 reserved_1;   /**<  might be used in Fortran;
492 ///                                  see above  */
493 ///    kmp_int32 flags;        /**<  also f.flags; KMP_IDENT_xxx flags;
494 ///                                  KMP_IDENT_KMPC identifies this union
495 ///                                  member  */
496 ///    kmp_int32 reserved_2;   /**<  not really used in Fortran any more;
497 ///                                  see above */
498 ///#if USE_ITT_BUILD
499 ///                            /*  but currently used for storing
500 ///                                region-specific ITT */
501 ///                            /*  contextual information. */
502 ///#endif /* USE_ITT_BUILD */
503 ///    kmp_int32 reserved_3;   /**< source[4] in Fortran, do not use for
504 ///                                 C++  */
505 ///    char const *psource;    /**< String describing the source location.
506 ///                            The string is composed of semi-colon separated
507 //                             fields which describe the source file,
508 ///                            the function and a pair of line numbers that
509 ///                            delimit the construct.
510 ///                             */
511 /// } ident_t;
512 enum IdentFieldIndex {
513   /// might be used in Fortran
514   IdentField_Reserved_1,
515   /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member.
516   IdentField_Flags,
517   /// Not really used in Fortran any more
518   IdentField_Reserved_2,
519   /// Source[4] in Fortran, do not use for C++
520   IdentField_Reserved_3,
521   /// String describing the source location. The string is composed of
522   /// semi-colon separated fields which describe the source file, the function
523   /// and a pair of line numbers that delimit the construct.
524   IdentField_PSource
525 };
526 
527 /// Schedule types for 'omp for' loops (these enumerators are taken from
528 /// the enum sched_type in kmp.h).
529 enum OpenMPSchedType {
530   /// Lower bound for default (unordered) versions.
531   OMP_sch_lower = 32,
532   OMP_sch_static_chunked = 33,
533   OMP_sch_static = 34,
534   OMP_sch_dynamic_chunked = 35,
535   OMP_sch_guided_chunked = 36,
536   OMP_sch_runtime = 37,
537   OMP_sch_auto = 38,
538   /// static with chunk adjustment (e.g., simd)
539   OMP_sch_static_balanced_chunked = 45,
540   /// Lower bound for 'ordered' versions.
541   OMP_ord_lower = 64,
542   OMP_ord_static_chunked = 65,
543   OMP_ord_static = 66,
544   OMP_ord_dynamic_chunked = 67,
545   OMP_ord_guided_chunked = 68,
546   OMP_ord_runtime = 69,
547   OMP_ord_auto = 70,
548   OMP_sch_default = OMP_sch_static,
549   /// dist_schedule types
550   OMP_dist_sch_static_chunked = 91,
551   OMP_dist_sch_static = 92,
552   /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers.
553   /// Set if the monotonic schedule modifier was present.
554   OMP_sch_modifier_monotonic = (1 << 29),
555   /// Set if the nonmonotonic schedule modifier was present.
556   OMP_sch_modifier_nonmonotonic = (1 << 30),
557 };
558 
559 enum OpenMPRTLFunction {
560   /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc,
561   /// kmpc_micro microtask, ...);
562   OMPRTL__kmpc_fork_call,
563   /// Call to void *__kmpc_threadprivate_cached(ident_t *loc,
564   /// kmp_int32 global_tid, void *data, size_t size, void ***cache);
565   OMPRTL__kmpc_threadprivate_cached,
566   /// Call to void __kmpc_threadprivate_register( ident_t *,
567   /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
568   OMPRTL__kmpc_threadprivate_register,
569   // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc);
570   OMPRTL__kmpc_global_thread_num,
571   // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
572   // kmp_critical_name *crit);
573   OMPRTL__kmpc_critical,
574   // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32
575   // global_tid, kmp_critical_name *crit, uintptr_t hint);
576   OMPRTL__kmpc_critical_with_hint,
577   // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
578   // kmp_critical_name *crit);
579   OMPRTL__kmpc_end_critical,
580   // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
581   // global_tid);
582   OMPRTL__kmpc_cancel_barrier,
583   // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
584   OMPRTL__kmpc_barrier,
585   // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
586   OMPRTL__kmpc_for_static_fini,
587   // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
588   // global_tid);
589   OMPRTL__kmpc_serialized_parallel,
590   // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
591   // global_tid);
592   OMPRTL__kmpc_end_serialized_parallel,
593   // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
594   // kmp_int32 num_threads);
595   OMPRTL__kmpc_push_num_threads,
596   // Call to void __kmpc_flush(ident_t *loc);
597   OMPRTL__kmpc_flush,
598   // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid);
599   OMPRTL__kmpc_master,
600   // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid);
601   OMPRTL__kmpc_end_master,
602   // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
603   // int end_part);
604   OMPRTL__kmpc_omp_taskyield,
605   // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid);
606   OMPRTL__kmpc_single,
607   // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid);
608   OMPRTL__kmpc_end_single,
609   // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
610   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
611   // kmp_routine_entry_t *task_entry);
612   OMPRTL__kmpc_omp_task_alloc,
613   // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *,
614   // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t,
615   // size_t sizeof_shareds, kmp_routine_entry_t *task_entry,
616   // kmp_int64 device_id);
617   OMPRTL__kmpc_omp_target_task_alloc,
618   // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t *
619   // new_task);
620   OMPRTL__kmpc_omp_task,
621   // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
622   // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
623   // kmp_int32 didit);
624   OMPRTL__kmpc_copyprivate,
625   // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
626   // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
627   // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
628   OMPRTL__kmpc_reduce,
629   // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
630   // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
631   // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
632   // *lck);
633   OMPRTL__kmpc_reduce_nowait,
634   // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
635   // kmp_critical_name *lck);
636   OMPRTL__kmpc_end_reduce,
637   // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
638   // kmp_critical_name *lck);
639   OMPRTL__kmpc_end_reduce_nowait,
640   // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
641   // kmp_task_t * new_task);
642   OMPRTL__kmpc_omp_task_begin_if0,
643   // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
644   // kmp_task_t * new_task);
645   OMPRTL__kmpc_omp_task_complete_if0,
646   // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
647   OMPRTL__kmpc_ordered,
648   // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
649   OMPRTL__kmpc_end_ordered,
650   // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
651   // global_tid);
652   OMPRTL__kmpc_omp_taskwait,
653   // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
654   OMPRTL__kmpc_taskgroup,
655   // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
656   OMPRTL__kmpc_end_taskgroup,
657   // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
658   // int proc_bind);
659   OMPRTL__kmpc_push_proc_bind,
660   // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32
661   // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t
662   // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
663   OMPRTL__kmpc_omp_task_with_deps,
664   // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32
665   // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
666   // ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
667   OMPRTL__kmpc_omp_wait_deps,
668   // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
669   // global_tid, kmp_int32 cncl_kind);
670   OMPRTL__kmpc_cancellationpoint,
671   // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
672   // kmp_int32 cncl_kind);
673   OMPRTL__kmpc_cancel,
674   // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid,
675   // kmp_int32 num_teams, kmp_int32 thread_limit);
676   OMPRTL__kmpc_push_num_teams,
677   // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
678   // microtask, ...);
679   OMPRTL__kmpc_fork_teams,
680   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
681   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
682   // sched, kmp_uint64 grainsize, void *task_dup);
683   OMPRTL__kmpc_taskloop,
684   // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
685   // num_dims, struct kmp_dim *dims);
686   OMPRTL__kmpc_doacross_init,
687   // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
688   OMPRTL__kmpc_doacross_fini,
689   // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
690   // *vec);
691   OMPRTL__kmpc_doacross_post,
692   // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
693   // *vec);
694   OMPRTL__kmpc_doacross_wait,
695   // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void
696   // *data);
697   OMPRTL__kmpc_task_reduction_init,
698   // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
699   // *d);
700   OMPRTL__kmpc_task_reduction_get_th_data,
701   // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al);
702   OMPRTL__kmpc_alloc,
703   // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al);
704   OMPRTL__kmpc_free,
705 
706   //
707   // Offloading related calls
708   //
709   // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
710   // size);
711   OMPRTL__kmpc_push_target_tripcount,
712   // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
713   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
714   // *arg_types);
715   OMPRTL__tgt_target,
716   // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
717   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
718   // *arg_types);
719   OMPRTL__tgt_target_nowait,
720   // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
721   // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
722   // *arg_types, int32_t num_teams, int32_t thread_limit);
723   OMPRTL__tgt_target_teams,
724   // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void
725   // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
726   // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
727   OMPRTL__tgt_target_teams_nowait,
728   // Call to void __tgt_register_requires(int64_t flags);
729   OMPRTL__tgt_register_requires,
730   // Call to void __tgt_register_lib(__tgt_bin_desc *desc);
731   OMPRTL__tgt_register_lib,
732   // Call to void __tgt_unregister_lib(__tgt_bin_desc *desc);
733   OMPRTL__tgt_unregister_lib,
734   // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
735   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
736   OMPRTL__tgt_target_data_begin,
737   // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
738   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
739   // *arg_types);
740   OMPRTL__tgt_target_data_begin_nowait,
741   // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
742   // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types);
743   OMPRTL__tgt_target_data_end,
744   // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t
745   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
746   // *arg_types);
747   OMPRTL__tgt_target_data_end_nowait,
748   // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
749   // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
750   OMPRTL__tgt_target_data_update,
751   // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t
752   // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
753   // *arg_types);
754   OMPRTL__tgt_target_data_update_nowait,
755   // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
756   OMPRTL__tgt_mapper_num_components,
757   // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void
758   // *base, void *begin, int64_t size, int64_t type);
759   OMPRTL__tgt_push_mapper_component,
760 };
761 
762 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP
763 /// region.
764 class CleanupTy final : public EHScopeStack::Cleanup {
765   PrePostActionTy *Action;
766 
767 public:
768   explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {}
769   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
770     if (!CGF.HaveInsertPoint())
771       return;
772     Action->Exit(CGF);
773   }
774 };
775 
776 } // anonymous namespace
777 
778 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const {
779   CodeGenFunction::RunCleanupsScope Scope(CGF);
780   if (PrePostAction) {
781     CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction);
782     Callback(CodeGen, CGF, *PrePostAction);
783   } else {
784     PrePostActionTy Action;
785     Callback(CodeGen, CGF, Action);
786   }
787 }
788 
789 /// Check if the combiner is a call to UDR combiner and if it is so return the
790 /// UDR decl used for reduction.
791 static const OMPDeclareReductionDecl *
792 getReductionInit(const Expr *ReductionOp) {
793   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
794     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
795       if (const auto *DRE =
796               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
797         if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl()))
798           return DRD;
799   return nullptr;
800 }
801 
802 static void emitInitWithReductionInitializer(CodeGenFunction &CGF,
803                                              const OMPDeclareReductionDecl *DRD,
804                                              const Expr *InitOp,
805                                              Address Private, Address Original,
806                                              QualType Ty) {
807   if (DRD->getInitializer()) {
808     std::pair<llvm::Function *, llvm::Function *> Reduction =
809         CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
810     const auto *CE = cast<CallExpr>(InitOp);
811     const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee());
812     const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts();
813     const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts();
814     const auto *LHSDRE =
815         cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr());
816     const auto *RHSDRE =
817         cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr());
818     CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
819     PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()),
820                             [=]() { return Private; });
821     PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()),
822                             [=]() { return Original; });
823     (void)PrivateScope.Privatize();
824     RValue Func = RValue::get(Reduction.second);
825     CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
826     CGF.EmitIgnoredExpr(InitOp);
827   } else {
828     llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty);
829     std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"});
830     auto *GV = new llvm::GlobalVariable(
831         CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true,
832         llvm::GlobalValue::PrivateLinkage, Init, Name);
833     LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty);
834     RValue InitRVal;
835     switch (CGF.getEvaluationKind(Ty)) {
836     case TEK_Scalar:
837       InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation());
838       break;
839     case TEK_Complex:
840       InitRVal =
841           RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation()));
842       break;
843     case TEK_Aggregate:
844       InitRVal = RValue::getAggregate(LV.getAddress());
845       break;
846     }
847     OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue);
848     CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal);
849     CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
850                          /*IsInitializer=*/false);
851   }
852 }
853 
854 /// Emit initialization of arrays of complex types.
855 /// \param DestAddr Address of the array.
856 /// \param Type Type of array.
857 /// \param Init Initial expression of array.
858 /// \param SrcAddr Address of the original array.
859 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr,
860                                  QualType Type, bool EmitDeclareReductionInit,
861                                  const Expr *Init,
862                                  const OMPDeclareReductionDecl *DRD,
863                                  Address SrcAddr = Address::invalid()) {
864   // Perform element-by-element initialization.
865   QualType ElementTy;
866 
867   // Drill down to the base element type on both arrays.
868   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
869   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr);
870   DestAddr =
871       CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType());
872   if (DRD)
873     SrcAddr =
874         CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType());
875 
876   llvm::Value *SrcBegin = nullptr;
877   if (DRD)
878     SrcBegin = SrcAddr.getPointer();
879   llvm::Value *DestBegin = DestAddr.getPointer();
880   // Cast from pointer to array type to pointer to single element.
881   llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements);
882   // The basic structure here is a while-do loop.
883   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body");
884   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done");
885   llvm::Value *IsEmpty =
886       CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty");
887   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
888 
889   // Enter the loop body, making that address the current address.
890   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
891   CGF.EmitBlock(BodyBB);
892 
893   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
894 
895   llvm::PHINode *SrcElementPHI = nullptr;
896   Address SrcElementCurrent = Address::invalid();
897   if (DRD) {
898     SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2,
899                                           "omp.arraycpy.srcElementPast");
900     SrcElementPHI->addIncoming(SrcBegin, EntryBB);
901     SrcElementCurrent =
902         Address(SrcElementPHI,
903                 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize));
904   }
905   llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI(
906       DestBegin->getType(), 2, "omp.arraycpy.destElementPast");
907   DestElementPHI->addIncoming(DestBegin, EntryBB);
908   Address DestElementCurrent =
909       Address(DestElementPHI,
910               DestAddr.getAlignment().alignmentOfArrayElement(ElementSize));
911 
912   // Emit copy.
913   {
914     CodeGenFunction::RunCleanupsScope InitScope(CGF);
915     if (EmitDeclareReductionInit) {
916       emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent,
917                                        SrcElementCurrent, ElementTy);
918     } else
919       CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(),
920                            /*IsInitializer=*/false);
921   }
922 
923   if (DRD) {
924     // Shift the address forward by one element.
925     llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32(
926         SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
927     SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock());
928   }
929 
930   // Shift the address forward by one element.
931   llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32(
932       DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
933   // Check whether we've reached the end.
934   llvm::Value *Done =
935       CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done");
936   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
937   DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock());
938 
939   // Done.
940   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
941 }
942 
943 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) {
944   return CGF.EmitOMPSharedLValue(E);
945 }
946 
947 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF,
948                                             const Expr *E) {
949   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E))
950     return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false);
951   return LValue();
952 }
953 
954 void ReductionCodeGen::emitAggregateInitialization(
955     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
956     const OMPDeclareReductionDecl *DRD) {
957   // Emit VarDecl with copy init for arrays.
958   // Get the address of the original variable captured in current
959   // captured region.
960   const auto *PrivateVD =
961       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
962   bool EmitDeclareReductionInit =
963       DRD && (DRD->getInitializer() || !PrivateVD->hasInit());
964   EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(),
965                        EmitDeclareReductionInit,
966                        EmitDeclareReductionInit ? ClausesData[N].ReductionOp
967                                                 : PrivateVD->getInit(),
968                        DRD, SharedLVal.getAddress());
969 }
970 
971 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds,
972                                    ArrayRef<const Expr *> Privates,
973                                    ArrayRef<const Expr *> ReductionOps) {
974   ClausesData.reserve(Shareds.size());
975   SharedAddresses.reserve(Shareds.size());
976   Sizes.reserve(Shareds.size());
977   BaseDecls.reserve(Shareds.size());
978   auto IPriv = Privates.begin();
979   auto IRed = ReductionOps.begin();
980   for (const Expr *Ref : Shareds) {
981     ClausesData.emplace_back(Ref, *IPriv, *IRed);
982     std::advance(IPriv, 1);
983     std::advance(IRed, 1);
984   }
985 }
986 
987 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) {
988   assert(SharedAddresses.size() == N &&
989          "Number of generated lvalues must be exactly N.");
990   LValue First = emitSharedLValue(CGF, ClausesData[N].Ref);
991   LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref);
992   SharedAddresses.emplace_back(First, Second);
993 }
994 
995 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) {
996   const auto *PrivateVD =
997       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
998   QualType PrivateType = PrivateVD->getType();
999   bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref);
1000   if (!PrivateType->isVariablyModifiedType()) {
1001     Sizes.emplace_back(
1002         CGF.getTypeSize(
1003             SharedAddresses[N].first.getType().getNonReferenceType()),
1004         nullptr);
1005     return;
1006   }
1007   llvm::Value *Size;
1008   llvm::Value *SizeInChars;
1009   auto *ElemType =
1010       cast<llvm::PointerType>(SharedAddresses[N].first.getPointer()->getType())
1011           ->getElementType();
1012   auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType);
1013   if (AsArraySection) {
1014     Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(),
1015                                      SharedAddresses[N].first.getPointer());
1016     Size = CGF.Builder.CreateNUWAdd(
1017         Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1));
1018     SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf);
1019   } else {
1020     SizeInChars = CGF.getTypeSize(
1021         SharedAddresses[N].first.getType().getNonReferenceType());
1022     Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf);
1023   }
1024   Sizes.emplace_back(SizeInChars, Size);
1025   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1026       CGF,
1027       cast<OpaqueValueExpr>(
1028           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1029       RValue::get(Size));
1030   CGF.EmitVariablyModifiedType(PrivateType);
1031 }
1032 
1033 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N,
1034                                          llvm::Value *Size) {
1035   const auto *PrivateVD =
1036       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1037   QualType PrivateType = PrivateVD->getType();
1038   if (!PrivateType->isVariablyModifiedType()) {
1039     assert(!Size && !Sizes[N].second &&
1040            "Size should be nullptr for non-variably modified reduction "
1041            "items.");
1042     return;
1043   }
1044   CodeGenFunction::OpaqueValueMapping OpaqueMap(
1045       CGF,
1046       cast<OpaqueValueExpr>(
1047           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
1048       RValue::get(Size));
1049   CGF.EmitVariablyModifiedType(PrivateType);
1050 }
1051 
1052 void ReductionCodeGen::emitInitialization(
1053     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal,
1054     llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) {
1055   assert(SharedAddresses.size() > N && "No variable was generated");
1056   const auto *PrivateVD =
1057       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1058   const OMPDeclareReductionDecl *DRD =
1059       getReductionInit(ClausesData[N].ReductionOp);
1060   QualType PrivateType = PrivateVD->getType();
1061   PrivateAddr = CGF.Builder.CreateElementBitCast(
1062       PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1063   QualType SharedType = SharedAddresses[N].first.getType();
1064   SharedLVal = CGF.MakeAddrLValue(
1065       CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(),
1066                                        CGF.ConvertTypeForMem(SharedType)),
1067       SharedType, SharedAddresses[N].first.getBaseInfo(),
1068       CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType));
1069   if (CGF.getContext().getAsArrayType(PrivateVD->getType())) {
1070     emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD);
1071   } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) {
1072     emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp,
1073                                      PrivateAddr, SharedLVal.getAddress(),
1074                                      SharedLVal.getType());
1075   } else if (!DefaultInit(CGF) && PrivateVD->hasInit() &&
1076              !CGF.isTrivialInitializer(PrivateVD->getInit())) {
1077     CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr,
1078                          PrivateVD->getType().getQualifiers(),
1079                          /*IsInitializer=*/false);
1080   }
1081 }
1082 
1083 bool ReductionCodeGen::needCleanups(unsigned N) {
1084   const auto *PrivateVD =
1085       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1086   QualType PrivateType = PrivateVD->getType();
1087   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1088   return DTorKind != QualType::DK_none;
1089 }
1090 
1091 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N,
1092                                     Address PrivateAddr) {
1093   const auto *PrivateVD =
1094       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
1095   QualType PrivateType = PrivateVD->getType();
1096   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
1097   if (needCleanups(N)) {
1098     PrivateAddr = CGF.Builder.CreateElementBitCast(
1099         PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
1100     CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType);
1101   }
1102 }
1103 
1104 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1105                           LValue BaseLV) {
1106   BaseTy = BaseTy.getNonReferenceType();
1107   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1108          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1109     if (const auto *PtrTy = BaseTy->getAs<PointerType>()) {
1110       BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(), PtrTy);
1111     } else {
1112       LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(), BaseTy);
1113       BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal);
1114     }
1115     BaseTy = BaseTy->getPointeeType();
1116   }
1117   return CGF.MakeAddrLValue(
1118       CGF.Builder.CreateElementBitCast(BaseLV.getAddress(),
1119                                        CGF.ConvertTypeForMem(ElTy)),
1120       BaseLV.getType(), BaseLV.getBaseInfo(),
1121       CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType()));
1122 }
1123 
1124 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
1125                           llvm::Type *BaseLVType, CharUnits BaseLVAlignment,
1126                           llvm::Value *Addr) {
1127   Address Tmp = Address::invalid();
1128   Address TopTmp = Address::invalid();
1129   Address MostTopTmp = Address::invalid();
1130   BaseTy = BaseTy.getNonReferenceType();
1131   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
1132          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
1133     Tmp = CGF.CreateMemTemp(BaseTy);
1134     if (TopTmp.isValid())
1135       CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp);
1136     else
1137       MostTopTmp = Tmp;
1138     TopTmp = Tmp;
1139     BaseTy = BaseTy->getPointeeType();
1140   }
1141   llvm::Type *Ty = BaseLVType;
1142   if (Tmp.isValid())
1143     Ty = Tmp.getElementType();
1144   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty);
1145   if (Tmp.isValid()) {
1146     CGF.Builder.CreateStore(Addr, Tmp);
1147     return MostTopTmp;
1148   }
1149   return Address(Addr, BaseLVAlignment);
1150 }
1151 
1152 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) {
1153   const VarDecl *OrigVD = nullptr;
1154   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) {
1155     const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
1156     while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base))
1157       Base = TempOASE->getBase()->IgnoreParenImpCasts();
1158     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1159       Base = TempASE->getBase()->IgnoreParenImpCasts();
1160     DE = cast<DeclRefExpr>(Base);
1161     OrigVD = cast<VarDecl>(DE->getDecl());
1162   } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) {
1163     const Expr *Base = ASE->getBase()->IgnoreParenImpCasts();
1164     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
1165       Base = TempASE->getBase()->IgnoreParenImpCasts();
1166     DE = cast<DeclRefExpr>(Base);
1167     OrigVD = cast<VarDecl>(DE->getDecl());
1168   }
1169   return OrigVD;
1170 }
1171 
1172 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N,
1173                                                Address PrivateAddr) {
1174   const DeclRefExpr *DE;
1175   if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) {
1176     BaseDecls.emplace_back(OrigVD);
1177     LValue OriginalBaseLValue = CGF.EmitLValue(DE);
1178     LValue BaseLValue =
1179         loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(),
1180                     OriginalBaseLValue);
1181     llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff(
1182         BaseLValue.getPointer(), SharedAddresses[N].first.getPointer());
1183     llvm::Value *PrivatePointer =
1184         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1185             PrivateAddr.getPointer(),
1186             SharedAddresses[N].first.getAddress().getType());
1187     llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment);
1188     return castToBase(CGF, OrigVD->getType(),
1189                       SharedAddresses[N].first.getType(),
1190                       OriginalBaseLValue.getAddress().getType(),
1191                       OriginalBaseLValue.getAlignment(), Ptr);
1192   }
1193   BaseDecls.emplace_back(
1194       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl()));
1195   return PrivateAddr;
1196 }
1197 
1198 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const {
1199   const OMPDeclareReductionDecl *DRD =
1200       getReductionInit(ClausesData[N].ReductionOp);
1201   return DRD && DRD->getInitializer();
1202 }
1203 
1204 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
1205   return CGF.EmitLoadOfPointerLValue(
1206       CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1207       getThreadIDVariable()->getType()->castAs<PointerType>());
1208 }
1209 
1210 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) {
1211   if (!CGF.HaveInsertPoint())
1212     return;
1213   // 1.2.2 OpenMP Language Terminology
1214   // Structured block - An executable statement with a single entry at the
1215   // top and a single exit at the bottom.
1216   // The point of exit cannot be a branch out of the structured block.
1217   // longjmp() and throw() must not violate the entry/exit criteria.
1218   CGF.EHStack.pushTerminate();
1219   CodeGen(CGF);
1220   CGF.EHStack.popTerminate();
1221 }
1222 
1223 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
1224     CodeGenFunction &CGF) {
1225   return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1226                             getThreadIDVariable()->getType(),
1227                             AlignmentSource::Decl);
1228 }
1229 
1230 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1231                                        QualType FieldTy) {
1232   auto *Field = FieldDecl::Create(
1233       C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1234       C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1235       /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1236   Field->setAccess(AS_public);
1237   DC->addDecl(Field);
1238   return Field;
1239 }
1240 
1241 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator,
1242                                  StringRef Separator)
1243     : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator),
1244       OffloadEntriesInfoManager(CGM) {
1245   ASTContext &C = CGM.getContext();
1246   RecordDecl *RD = C.buildImplicitRecord("ident_t");
1247   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
1248   RD->startDefinition();
1249   // reserved_1
1250   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1251   // flags
1252   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1253   // reserved_2
1254   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1255   // reserved_3
1256   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1257   // psource
1258   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
1259   RD->completeDefinition();
1260   IdentQTy = C.getRecordType(RD);
1261   IdentTy = CGM.getTypes().ConvertRecordDeclType(RD);
1262   KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
1263 
1264   loadOffloadInfoMetadata();
1265 }
1266 
1267 bool CGOpenMPRuntime::tryEmitDeclareVariant(const GlobalDecl &NewGD,
1268                                             const GlobalDecl &OldGD,
1269                                             llvm::GlobalValue *OrigAddr,
1270                                             bool IsForDefinition) {
1271   // Emit at least a definition for the aliasee if the the address of the
1272   // original function is requested.
1273   if (IsForDefinition || OrigAddr)
1274     (void)CGM.GetAddrOfGlobal(NewGD);
1275   StringRef NewMangledName = CGM.getMangledName(NewGD);
1276   llvm::GlobalValue *Addr = CGM.GetGlobalValue(NewMangledName);
1277   if (Addr && !Addr->isDeclaration()) {
1278     const auto *D = cast<FunctionDecl>(OldGD.getDecl());
1279     const CGFunctionInfo &FI = CGM.getTypes().arrangeGlobalDeclaration(OldGD);
1280     llvm::Type *DeclTy = CGM.getTypes().GetFunctionType(FI);
1281 
1282     // Create a reference to the named value.  This ensures that it is emitted
1283     // if a deferred decl.
1284     llvm::GlobalValue::LinkageTypes LT = CGM.getFunctionLinkage(OldGD);
1285 
1286     // Create the new alias itself, but don't set a name yet.
1287     auto *GA =
1288         llvm::GlobalAlias::create(DeclTy, 0, LT, "", Addr, &CGM.getModule());
1289 
1290     if (OrigAddr) {
1291       assert(OrigAddr->isDeclaration() && "Expected declaration");
1292 
1293       GA->takeName(OrigAddr);
1294       OrigAddr->replaceAllUsesWith(
1295           llvm::ConstantExpr::getBitCast(GA, OrigAddr->getType()));
1296       OrigAddr->eraseFromParent();
1297     } else {
1298       GA->setName(CGM.getMangledName(OldGD));
1299     }
1300 
1301     // Set attributes which are particular to an alias; this is a
1302     // specialization of the attributes which may be set on a global function.
1303     if (D->hasAttr<WeakAttr>() || D->hasAttr<WeakRefAttr>() ||
1304         D->isWeakImported())
1305       GA->setLinkage(llvm::Function::WeakAnyLinkage);
1306 
1307     CGM.SetCommonAttributes(OldGD, GA);
1308     return true;
1309   }
1310   return false;
1311 }
1312 
1313 void CGOpenMPRuntime::clear() {
1314   InternalVars.clear();
1315   // Clean non-target variable declarations possibly used only in debug info.
1316   for (const auto &Data : EmittedNonTargetVariables) {
1317     if (!Data.getValue().pointsToAliveValue())
1318       continue;
1319     auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue());
1320     if (!GV)
1321       continue;
1322     if (!GV->isDeclaration() || GV->getNumUses() > 0)
1323       continue;
1324     GV->eraseFromParent();
1325   }
1326   // Emit aliases for the deferred aliasees.
1327   for (const auto &Pair : DeferredVariantFunction) {
1328     StringRef MangledName = CGM.getMangledName(Pair.second.second);
1329     llvm::GlobalValue *Addr = CGM.GetGlobalValue(MangledName);
1330     // If not able to emit alias, just emit original declaration.
1331     (void)tryEmitDeclareVariant(Pair.second.first, Pair.second.second, Addr,
1332                                 /*IsForDefinition=*/false);
1333   }
1334 }
1335 
1336 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const {
1337   SmallString<128> Buffer;
1338   llvm::raw_svector_ostream OS(Buffer);
1339   StringRef Sep = FirstSeparator;
1340   for (StringRef Part : Parts) {
1341     OS << Sep << Part;
1342     Sep = Separator;
1343   }
1344   return OS.str();
1345 }
1346 
1347 static llvm::Function *
1348 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty,
1349                           const Expr *CombinerInitializer, const VarDecl *In,
1350                           const VarDecl *Out, bool IsCombiner) {
1351   // void .omp_combiner.(Ty *in, Ty *out);
1352   ASTContext &C = CGM.getContext();
1353   QualType PtrTy = C.getPointerType(Ty).withRestrict();
1354   FunctionArgList Args;
1355   ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(),
1356                                /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1357   ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(),
1358                               /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1359   Args.push_back(&OmpOutParm);
1360   Args.push_back(&OmpInParm);
1361   const CGFunctionInfo &FnInfo =
1362       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
1363   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
1364   std::string Name = CGM.getOpenMPRuntime().getName(
1365       {IsCombiner ? "omp_combiner" : "omp_initializer", ""});
1366   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
1367                                     Name, &CGM.getModule());
1368   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
1369   if (CGM.getLangOpts().Optimize) {
1370     Fn->removeFnAttr(llvm::Attribute::NoInline);
1371     Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
1372     Fn->addFnAttr(llvm::Attribute::AlwaysInline);
1373   }
1374   CodeGenFunction CGF(CGM);
1375   // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions.
1376   // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions.
1377   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(),
1378                     Out->getLocation());
1379   CodeGenFunction::OMPPrivateScope Scope(CGF);
1380   Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm);
1381   Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() {
1382     return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>())
1383         .getAddress();
1384   });
1385   Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm);
1386   Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() {
1387     return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>())
1388         .getAddress();
1389   });
1390   (void)Scope.Privatize();
1391   if (!IsCombiner && Out->hasInit() &&
1392       !CGF.isTrivialInitializer(Out->getInit())) {
1393     CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out),
1394                          Out->getType().getQualifiers(),
1395                          /*IsInitializer=*/true);
1396   }
1397   if (CombinerInitializer)
1398     CGF.EmitIgnoredExpr(CombinerInitializer);
1399   Scope.ForceCleanup();
1400   CGF.FinishFunction();
1401   return Fn;
1402 }
1403 
1404 void CGOpenMPRuntime::emitUserDefinedReduction(
1405     CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) {
1406   if (UDRMap.count(D) > 0)
1407     return;
1408   llvm::Function *Combiner = emitCombinerOrInitializer(
1409       CGM, D->getType(), D->getCombiner(),
1410       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()),
1411       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()),
1412       /*IsCombiner=*/true);
1413   llvm::Function *Initializer = nullptr;
1414   if (const Expr *Init = D->getInitializer()) {
1415     Initializer = emitCombinerOrInitializer(
1416         CGM, D->getType(),
1417         D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init
1418                                                                      : nullptr,
1419         cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()),
1420         cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()),
1421         /*IsCombiner=*/false);
1422   }
1423   UDRMap.try_emplace(D, Combiner, Initializer);
1424   if (CGF) {
1425     auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn);
1426     Decls.second.push_back(D);
1427   }
1428 }
1429 
1430 std::pair<llvm::Function *, llvm::Function *>
1431 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) {
1432   auto I = UDRMap.find(D);
1433   if (I != UDRMap.end())
1434     return I->second;
1435   emitUserDefinedReduction(/*CGF=*/nullptr, D);
1436   return UDRMap.lookup(D);
1437 }
1438 
1439 static llvm::Function *emitParallelOrTeamsOutlinedFunction(
1440     CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS,
1441     const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1442     const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) {
1443   assert(ThreadIDVar->getType()->isPointerType() &&
1444          "thread id variable must be of type kmp_int32 *");
1445   CodeGenFunction CGF(CGM, true);
1446   bool HasCancel = false;
1447   if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D))
1448     HasCancel = OPD->hasCancel();
1449   else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D))
1450     HasCancel = OPSD->hasCancel();
1451   else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D))
1452     HasCancel = OPFD->hasCancel();
1453   else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D))
1454     HasCancel = OPFD->hasCancel();
1455   else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D))
1456     HasCancel = OPFD->hasCancel();
1457   else if (const auto *OPFD =
1458                dyn_cast<OMPTeamsDistributeParallelForDirective>(&D))
1459     HasCancel = OPFD->hasCancel();
1460   else if (const auto *OPFD =
1461                dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D))
1462     HasCancel = OPFD->hasCancel();
1463   CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind,
1464                                     HasCancel, OutlinedHelperName);
1465   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1466   return CGF.GenerateOpenMPCapturedStmtFunction(*CS);
1467 }
1468 
1469 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction(
1470     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1471     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1472   const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel);
1473   return emitParallelOrTeamsOutlinedFunction(
1474       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1475 }
1476 
1477 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction(
1478     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1479     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1480   const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams);
1481   return emitParallelOrTeamsOutlinedFunction(
1482       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1483 }
1484 
1485 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction(
1486     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1487     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
1488     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
1489     bool Tied, unsigned &NumberOfParts) {
1490   auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF,
1491                                               PrePostActionTy &) {
1492     llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc());
1493     llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc());
1494     llvm::Value *TaskArgs[] = {
1495         UpLoc, ThreadID,
1496         CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar),
1497                                     TaskTVar->getType()->castAs<PointerType>())
1498             .getPointer()};
1499     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs);
1500   };
1501   CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar,
1502                                                             UntiedCodeGen);
1503   CodeGen.setAction(Action);
1504   assert(!ThreadIDVar->getType()->isPointerType() &&
1505          "thread id variable must be of type kmp_int32 for tasks");
1506   const OpenMPDirectiveKind Region =
1507       isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop
1508                                                       : OMPD_task;
1509   const CapturedStmt *CS = D.getCapturedStmt(Region);
1510   const auto *TD = dyn_cast<OMPTaskDirective>(&D);
1511   CodeGenFunction CGF(CGM, true);
1512   CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen,
1513                                         InnermostKind,
1514                                         TD ? TD->hasCancel() : false, Action);
1515   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1516   llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS);
1517   if (!Tied)
1518     NumberOfParts = Action.getNumberOfParts();
1519   return Res;
1520 }
1521 
1522 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM,
1523                              const RecordDecl *RD, const CGRecordLayout &RL,
1524                              ArrayRef<llvm::Constant *> Data) {
1525   llvm::StructType *StructTy = RL.getLLVMType();
1526   unsigned PrevIdx = 0;
1527   ConstantInitBuilder CIBuilder(CGM);
1528   auto DI = Data.begin();
1529   for (const FieldDecl *FD : RD->fields()) {
1530     unsigned Idx = RL.getLLVMFieldNo(FD);
1531     // Fill the alignment.
1532     for (unsigned I = PrevIdx; I < Idx; ++I)
1533       Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I)));
1534     PrevIdx = Idx + 1;
1535     Fields.add(*DI);
1536     ++DI;
1537   }
1538 }
1539 
1540 template <class... As>
1541 static llvm::GlobalVariable *
1542 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant,
1543                    ArrayRef<llvm::Constant *> Data, const Twine &Name,
1544                    As &&... Args) {
1545   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1546   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1547   ConstantInitBuilder CIBuilder(CGM);
1548   ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType());
1549   buildStructValue(Fields, CGM, RD, RL, Data);
1550   return Fields.finishAndCreateGlobal(
1551       Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant,
1552       std::forward<As>(Args)...);
1553 }
1554 
1555 template <typename T>
1556 static void
1557 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty,
1558                                          ArrayRef<llvm::Constant *> Data,
1559                                          T &Parent) {
1560   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1561   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1562   ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType());
1563   buildStructValue(Fields, CGM, RD, RL, Data);
1564   Fields.finishAndAddTo(Parent);
1565 }
1566 
1567 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) {
1568   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1569   unsigned Reserved2Flags = getDefaultLocationReserved2Flags();
1570   FlagsTy FlagsKey(Flags, Reserved2Flags);
1571   llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey);
1572   if (!Entry) {
1573     if (!DefaultOpenMPPSource) {
1574       // Initialize default location for psource field of ident_t structure of
1575       // all ident_t objects. Format is ";file;function;line;column;;".
1576       // Taken from
1577       // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp
1578       DefaultOpenMPPSource =
1579           CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer();
1580       DefaultOpenMPPSource =
1581           llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy);
1582     }
1583 
1584     llvm::Constant *Data[] = {
1585         llvm::ConstantInt::getNullValue(CGM.Int32Ty),
1586         llvm::ConstantInt::get(CGM.Int32Ty, Flags),
1587         llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags),
1588         llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource};
1589     llvm::GlobalValue *DefaultOpenMPLocation =
1590         createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "",
1591                            llvm::GlobalValue::PrivateLinkage);
1592     DefaultOpenMPLocation->setUnnamedAddr(
1593         llvm::GlobalValue::UnnamedAddr::Global);
1594 
1595     OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation;
1596   }
1597   return Address(Entry, Align);
1598 }
1599 
1600 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF,
1601                                              bool AtCurrentPoint) {
1602   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1603   assert(!Elem.second.ServiceInsertPt && "Insert point is set already.");
1604 
1605   llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty);
1606   if (AtCurrentPoint) {
1607     Elem.second.ServiceInsertPt = new llvm::BitCastInst(
1608         Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock());
1609   } else {
1610     Elem.second.ServiceInsertPt =
1611         new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt");
1612     Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt);
1613   }
1614 }
1615 
1616 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) {
1617   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1618   if (Elem.second.ServiceInsertPt) {
1619     llvm::Instruction *Ptr = Elem.second.ServiceInsertPt;
1620     Elem.second.ServiceInsertPt = nullptr;
1621     Ptr->eraseFromParent();
1622   }
1623 }
1624 
1625 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
1626                                                  SourceLocation Loc,
1627                                                  unsigned Flags) {
1628   Flags |= OMP_IDENT_KMPC;
1629   // If no debug info is generated - return global default location.
1630   if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo ||
1631       Loc.isInvalid())
1632     return getOrCreateDefaultLocation(Flags).getPointer();
1633 
1634   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1635 
1636   CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy);
1637   Address LocValue = Address::invalid();
1638   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1639   if (I != OpenMPLocThreadIDMap.end())
1640     LocValue = Address(I->second.DebugLoc, Align);
1641 
1642   // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if
1643   // GetOpenMPThreadID was called before this routine.
1644   if (!LocValue.isValid()) {
1645     // Generate "ident_t .kmpc_loc.addr;"
1646     Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr");
1647     auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1648     Elem.second.DebugLoc = AI.getPointer();
1649     LocValue = AI;
1650 
1651     if (!Elem.second.ServiceInsertPt)
1652       setLocThreadIdInsertPt(CGF);
1653     CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1654     CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1655     CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags),
1656                              CGF.getTypeSize(IdentQTy));
1657   }
1658 
1659   // char **psource = &.kmpc_loc_<flags>.addr.psource;
1660   LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy);
1661   auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin();
1662   LValue PSource =
1663       CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource));
1664 
1665   llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding());
1666   if (OMPDebugLoc == nullptr) {
1667     SmallString<128> Buffer2;
1668     llvm::raw_svector_ostream OS2(Buffer2);
1669     // Build debug location
1670     PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1671     OS2 << ";" << PLoc.getFilename() << ";";
1672     if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1673       OS2 << FD->getQualifiedNameAsString();
1674     OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
1675     OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str());
1676     OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc;
1677   }
1678   // *psource = ";<File>;<Function>;<Line>;<Column>;;";
1679   CGF.EmitStoreOfScalar(OMPDebugLoc, PSource);
1680 
1681   // Our callers always pass this to a runtime function, so for
1682   // convenience, go ahead and return a naked pointer.
1683   return LocValue.getPointer();
1684 }
1685 
1686 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
1687                                           SourceLocation Loc) {
1688   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1689 
1690   llvm::Value *ThreadID = nullptr;
1691   // Check whether we've already cached a load of the thread id in this
1692   // function.
1693   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1694   if (I != OpenMPLocThreadIDMap.end()) {
1695     ThreadID = I->second.ThreadID;
1696     if (ThreadID != nullptr)
1697       return ThreadID;
1698   }
1699   // If exceptions are enabled, do not use parameter to avoid possible crash.
1700   if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions ||
1701       !CGF.getLangOpts().CXXExceptions ||
1702       CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) {
1703     if (auto *OMPRegionInfo =
1704             dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
1705       if (OMPRegionInfo->getThreadIDVariable()) {
1706         // Check if this an outlined function with thread id passed as argument.
1707         LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
1708         ThreadID = CGF.EmitLoadOfScalar(LVal, Loc);
1709         // If value loaded in entry block, cache it and use it everywhere in
1710         // function.
1711         if (CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) {
1712           auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1713           Elem.second.ThreadID = ThreadID;
1714         }
1715         return ThreadID;
1716       }
1717     }
1718   }
1719 
1720   // This is not an outlined function region - need to call __kmpc_int32
1721   // kmpc_global_thread_num(ident_t *loc).
1722   // Generate thread id value and cache this value for use across the
1723   // function.
1724   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1725   if (!Elem.second.ServiceInsertPt)
1726     setLocThreadIdInsertPt(CGF);
1727   CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1728   CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1729   llvm::CallInst *Call = CGF.Builder.CreateCall(
1730       createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
1731       emitUpdateLocation(CGF, Loc));
1732   Call->setCallingConv(CGF.getRuntimeCC());
1733   Elem.second.ThreadID = Call;
1734   return Call;
1735 }
1736 
1737 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
1738   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1739   if (OpenMPLocThreadIDMap.count(CGF.CurFn)) {
1740     clearLocThreadIdInsertPt(CGF);
1741     OpenMPLocThreadIDMap.erase(CGF.CurFn);
1742   }
1743   if (FunctionUDRMap.count(CGF.CurFn) > 0) {
1744     for(auto *D : FunctionUDRMap[CGF.CurFn])
1745       UDRMap.erase(D);
1746     FunctionUDRMap.erase(CGF.CurFn);
1747   }
1748   auto I = FunctionUDMMap.find(CGF.CurFn);
1749   if (I != FunctionUDMMap.end()) {
1750     for(auto *D : I->second)
1751       UDMMap.erase(D);
1752     FunctionUDMMap.erase(I);
1753   }
1754 }
1755 
1756 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
1757   return IdentTy->getPointerTo();
1758 }
1759 
1760 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() {
1761   if (!Kmpc_MicroTy) {
1762     // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...)
1763     llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty),
1764                                  llvm::PointerType::getUnqual(CGM.Int32Ty)};
1765     Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true);
1766   }
1767   return llvm::PointerType::getUnqual(Kmpc_MicroTy);
1768 }
1769 
1770 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) {
1771   llvm::FunctionCallee RTLFn = nullptr;
1772   switch (static_cast<OpenMPRTLFunction>(Function)) {
1773   case OMPRTL__kmpc_fork_call: {
1774     // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro
1775     // microtask, ...);
1776     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1777                                 getKmpc_MicroPointerTy()};
1778     auto *FnTy =
1779         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
1780     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call");
1781     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
1782       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
1783         llvm::LLVMContext &Ctx = F->getContext();
1784         llvm::MDBuilder MDB(Ctx);
1785         // Annotate the callback behavior of the __kmpc_fork_call:
1786         //  - The callback callee is argument number 2 (microtask).
1787         //  - The first two arguments of the callback callee are unknown (-1).
1788         //  - All variadic arguments to the __kmpc_fork_call are passed to the
1789         //    callback callee.
1790         F->addMetadata(
1791             llvm::LLVMContext::MD_callback,
1792             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
1793                                         2, {-1, -1},
1794                                         /* VarArgsArePassed */ true)}));
1795       }
1796     }
1797     break;
1798   }
1799   case OMPRTL__kmpc_global_thread_num: {
1800     // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc);
1801     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1802     auto *FnTy =
1803         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1804     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num");
1805     break;
1806   }
1807   case OMPRTL__kmpc_threadprivate_cached: {
1808     // Build void *__kmpc_threadprivate_cached(ident_t *loc,
1809     // kmp_int32 global_tid, void *data, size_t size, void ***cache);
1810     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1811                                 CGM.VoidPtrTy, CGM.SizeTy,
1812                                 CGM.VoidPtrTy->getPointerTo()->getPointerTo()};
1813     auto *FnTy =
1814         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false);
1815     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached");
1816     break;
1817   }
1818   case OMPRTL__kmpc_critical: {
1819     // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
1820     // kmp_critical_name *crit);
1821     llvm::Type *TypeParams[] = {
1822         getIdentTyPointerTy(), CGM.Int32Ty,
1823         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1824     auto *FnTy =
1825         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1826     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical");
1827     break;
1828   }
1829   case OMPRTL__kmpc_critical_with_hint: {
1830     // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid,
1831     // kmp_critical_name *crit, uintptr_t hint);
1832     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1833                                 llvm::PointerType::getUnqual(KmpCriticalNameTy),
1834                                 CGM.IntPtrTy};
1835     auto *FnTy =
1836         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1837     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint");
1838     break;
1839   }
1840   case OMPRTL__kmpc_threadprivate_register: {
1841     // Build void __kmpc_threadprivate_register(ident_t *, void *data,
1842     // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
1843     // typedef void *(*kmpc_ctor)(void *);
1844     auto *KmpcCtorTy =
1845         llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
1846                                 /*isVarArg*/ false)->getPointerTo();
1847     // typedef void *(*kmpc_cctor)(void *, void *);
1848     llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
1849     auto *KmpcCopyCtorTy =
1850         llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs,
1851                                 /*isVarArg*/ false)
1852             ->getPointerTo();
1853     // typedef void (*kmpc_dtor)(void *);
1854     auto *KmpcDtorTy =
1855         llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false)
1856             ->getPointerTo();
1857     llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy,
1858                               KmpcCopyCtorTy, KmpcDtorTy};
1859     auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs,
1860                                         /*isVarArg*/ false);
1861     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register");
1862     break;
1863   }
1864   case OMPRTL__kmpc_end_critical: {
1865     // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
1866     // kmp_critical_name *crit);
1867     llvm::Type *TypeParams[] = {
1868         getIdentTyPointerTy(), CGM.Int32Ty,
1869         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
1870     auto *FnTy =
1871         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1872     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical");
1873     break;
1874   }
1875   case OMPRTL__kmpc_cancel_barrier: {
1876     // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
1877     // global_tid);
1878     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1879     auto *FnTy =
1880         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1881     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier");
1882     break;
1883   }
1884   case OMPRTL__kmpc_barrier: {
1885     // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid);
1886     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1887     auto *FnTy =
1888         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1889     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier");
1890     break;
1891   }
1892   case OMPRTL__kmpc_for_static_fini: {
1893     // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
1894     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1895     auto *FnTy =
1896         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1897     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini");
1898     break;
1899   }
1900   case OMPRTL__kmpc_push_num_threads: {
1901     // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
1902     // kmp_int32 num_threads)
1903     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
1904                                 CGM.Int32Ty};
1905     auto *FnTy =
1906         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1907     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads");
1908     break;
1909   }
1910   case OMPRTL__kmpc_serialized_parallel: {
1911     // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
1912     // global_tid);
1913     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1914     auto *FnTy =
1915         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1916     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel");
1917     break;
1918   }
1919   case OMPRTL__kmpc_end_serialized_parallel: {
1920     // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
1921     // global_tid);
1922     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1923     auto *FnTy =
1924         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1925     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel");
1926     break;
1927   }
1928   case OMPRTL__kmpc_flush: {
1929     // Build void __kmpc_flush(ident_t *loc);
1930     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
1931     auto *FnTy =
1932         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1933     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush");
1934     break;
1935   }
1936   case OMPRTL__kmpc_master: {
1937     // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid);
1938     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1939     auto *FnTy =
1940         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1941     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master");
1942     break;
1943   }
1944   case OMPRTL__kmpc_end_master: {
1945     // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid);
1946     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1947     auto *FnTy =
1948         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
1949     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master");
1950     break;
1951   }
1952   case OMPRTL__kmpc_omp_taskyield: {
1953     // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
1954     // int end_part);
1955     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
1956     auto *FnTy =
1957         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1958     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield");
1959     break;
1960   }
1961   case OMPRTL__kmpc_single: {
1962     // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid);
1963     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1964     auto *FnTy =
1965         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
1966     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single");
1967     break;
1968   }
1969   case OMPRTL__kmpc_end_single: {
1970     // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid);
1971     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
1972     auto *FnTy =
1973         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
1974     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single");
1975     break;
1976   }
1977   case OMPRTL__kmpc_omp_task_alloc: {
1978     // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
1979     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
1980     // kmp_routine_entry_t *task_entry);
1981     assert(KmpRoutineEntryPtrTy != nullptr &&
1982            "Type kmp_routine_entry_t must be created.");
1983     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
1984                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy};
1985     // Return void * and then cast to particular kmp_task_t type.
1986     auto *FnTy =
1987         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
1988     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc");
1989     break;
1990   }
1991   case OMPRTL__kmpc_omp_target_task_alloc: {
1992     // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid,
1993     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
1994     // kmp_routine_entry_t *task_entry, kmp_int64 device_id);
1995     assert(KmpRoutineEntryPtrTy != nullptr &&
1996            "Type kmp_routine_entry_t must be created.");
1997     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
1998                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy,
1999                                 CGM.Int64Ty};
2000     // Return void * and then cast to particular kmp_task_t type.
2001     auto *FnTy =
2002         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2003     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc");
2004     break;
2005   }
2006   case OMPRTL__kmpc_omp_task: {
2007     // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2008     // *new_task);
2009     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2010                                 CGM.VoidPtrTy};
2011     auto *FnTy =
2012         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2013     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task");
2014     break;
2015   }
2016   case OMPRTL__kmpc_copyprivate: {
2017     // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
2018     // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
2019     // kmp_int32 didit);
2020     llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2021     auto *CpyFnTy =
2022         llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false);
2023     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy,
2024                                 CGM.VoidPtrTy, CpyFnTy->getPointerTo(),
2025                                 CGM.Int32Ty};
2026     auto *FnTy =
2027         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2028     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate");
2029     break;
2030   }
2031   case OMPRTL__kmpc_reduce: {
2032     // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
2033     // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
2034     // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
2035     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2036     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2037                                                /*isVarArg=*/false);
2038     llvm::Type *TypeParams[] = {
2039         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2040         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2041         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2042     auto *FnTy =
2043         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2044     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce");
2045     break;
2046   }
2047   case OMPRTL__kmpc_reduce_nowait: {
2048     // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
2049     // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
2050     // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
2051     // *lck);
2052     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2053     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
2054                                                /*isVarArg=*/false);
2055     llvm::Type *TypeParams[] = {
2056         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
2057         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
2058         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2059     auto *FnTy =
2060         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2061     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait");
2062     break;
2063   }
2064   case OMPRTL__kmpc_end_reduce: {
2065     // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
2066     // kmp_critical_name *lck);
2067     llvm::Type *TypeParams[] = {
2068         getIdentTyPointerTy(), CGM.Int32Ty,
2069         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2070     auto *FnTy =
2071         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2072     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce");
2073     break;
2074   }
2075   case OMPRTL__kmpc_end_reduce_nowait: {
2076     // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
2077     // kmp_critical_name *lck);
2078     llvm::Type *TypeParams[] = {
2079         getIdentTyPointerTy(), CGM.Int32Ty,
2080         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
2081     auto *FnTy =
2082         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2083     RTLFn =
2084         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait");
2085     break;
2086   }
2087   case OMPRTL__kmpc_omp_task_begin_if0: {
2088     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2089     // *new_task);
2090     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2091                                 CGM.VoidPtrTy};
2092     auto *FnTy =
2093         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2094     RTLFn =
2095         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0");
2096     break;
2097   }
2098   case OMPRTL__kmpc_omp_task_complete_if0: {
2099     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2100     // *new_task);
2101     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2102                                 CGM.VoidPtrTy};
2103     auto *FnTy =
2104         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2105     RTLFn = CGM.CreateRuntimeFunction(FnTy,
2106                                       /*Name=*/"__kmpc_omp_task_complete_if0");
2107     break;
2108   }
2109   case OMPRTL__kmpc_ordered: {
2110     // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
2111     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2112     auto *FnTy =
2113         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2114     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered");
2115     break;
2116   }
2117   case OMPRTL__kmpc_end_ordered: {
2118     // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid);
2119     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2120     auto *FnTy =
2121         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2122     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered");
2123     break;
2124   }
2125   case OMPRTL__kmpc_omp_taskwait: {
2126     // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid);
2127     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2128     auto *FnTy =
2129         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2130     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait");
2131     break;
2132   }
2133   case OMPRTL__kmpc_taskgroup: {
2134     // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid);
2135     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2136     auto *FnTy =
2137         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2138     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup");
2139     break;
2140   }
2141   case OMPRTL__kmpc_end_taskgroup: {
2142     // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid);
2143     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2144     auto *FnTy =
2145         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2146     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup");
2147     break;
2148   }
2149   case OMPRTL__kmpc_push_proc_bind: {
2150     // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid,
2151     // int proc_bind)
2152     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2153     auto *FnTy =
2154         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2155     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind");
2156     break;
2157   }
2158   case OMPRTL__kmpc_omp_task_with_deps: {
2159     // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
2160     // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
2161     // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list);
2162     llvm::Type *TypeParams[] = {
2163         getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty,
2164         CGM.VoidPtrTy,         CGM.Int32Ty, CGM.VoidPtrTy};
2165     auto *FnTy =
2166         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
2167     RTLFn =
2168         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps");
2169     break;
2170   }
2171   case OMPRTL__kmpc_omp_wait_deps: {
2172     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
2173     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias,
2174     // kmp_depend_info_t *noalias_dep_list);
2175     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2176                                 CGM.Int32Ty,           CGM.VoidPtrTy,
2177                                 CGM.Int32Ty,           CGM.VoidPtrTy};
2178     auto *FnTy =
2179         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2180     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps");
2181     break;
2182   }
2183   case OMPRTL__kmpc_cancellationpoint: {
2184     // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
2185     // global_tid, kmp_int32 cncl_kind)
2186     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2187     auto *FnTy =
2188         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2189     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint");
2190     break;
2191   }
2192   case OMPRTL__kmpc_cancel: {
2193     // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
2194     // kmp_int32 cncl_kind)
2195     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
2196     auto *FnTy =
2197         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2198     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel");
2199     break;
2200   }
2201   case OMPRTL__kmpc_push_num_teams: {
2202     // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid,
2203     // kmp_int32 num_teams, kmp_int32 num_threads)
2204     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
2205         CGM.Int32Ty};
2206     auto *FnTy =
2207         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2208     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams");
2209     break;
2210   }
2211   case OMPRTL__kmpc_fork_teams: {
2212     // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro
2213     // microtask, ...);
2214     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2215                                 getKmpc_MicroPointerTy()};
2216     auto *FnTy =
2217         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
2218     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams");
2219     if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) {
2220       if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) {
2221         llvm::LLVMContext &Ctx = F->getContext();
2222         llvm::MDBuilder MDB(Ctx);
2223         // Annotate the callback behavior of the __kmpc_fork_teams:
2224         //  - The callback callee is argument number 2 (microtask).
2225         //  - The first two arguments of the callback callee are unknown (-1).
2226         //  - All variadic arguments to the __kmpc_fork_teams are passed to the
2227         //    callback callee.
2228         F->addMetadata(
2229             llvm::LLVMContext::MD_callback,
2230             *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding(
2231                                         2, {-1, -1},
2232                                         /* VarArgsArePassed */ true)}));
2233       }
2234     }
2235     break;
2236   }
2237   case OMPRTL__kmpc_taskloop: {
2238     // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
2239     // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
2240     // sched, kmp_uint64 grainsize, void *task_dup);
2241     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2242                                 CGM.IntTy,
2243                                 CGM.VoidPtrTy,
2244                                 CGM.IntTy,
2245                                 CGM.Int64Ty->getPointerTo(),
2246                                 CGM.Int64Ty->getPointerTo(),
2247                                 CGM.Int64Ty,
2248                                 CGM.IntTy,
2249                                 CGM.IntTy,
2250                                 CGM.Int64Ty,
2251                                 CGM.VoidPtrTy};
2252     auto *FnTy =
2253         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2254     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop");
2255     break;
2256   }
2257   case OMPRTL__kmpc_doacross_init: {
2258     // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32
2259     // num_dims, struct kmp_dim *dims);
2260     llvm::Type *TypeParams[] = {getIdentTyPointerTy(),
2261                                 CGM.Int32Ty,
2262                                 CGM.Int32Ty,
2263                                 CGM.VoidPtrTy};
2264     auto *FnTy =
2265         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2266     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init");
2267     break;
2268   }
2269   case OMPRTL__kmpc_doacross_fini: {
2270     // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid);
2271     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
2272     auto *FnTy =
2273         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2274     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini");
2275     break;
2276   }
2277   case OMPRTL__kmpc_doacross_post: {
2278     // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64
2279     // *vec);
2280     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2281                                 CGM.Int64Ty->getPointerTo()};
2282     auto *FnTy =
2283         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2284     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post");
2285     break;
2286   }
2287   case OMPRTL__kmpc_doacross_wait: {
2288     // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64
2289     // *vec);
2290     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
2291                                 CGM.Int64Ty->getPointerTo()};
2292     auto *FnTy =
2293         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2294     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait");
2295     break;
2296   }
2297   case OMPRTL__kmpc_task_reduction_init: {
2298     // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void
2299     // *data);
2300     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy};
2301     auto *FnTy =
2302         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2303     RTLFn =
2304         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init");
2305     break;
2306   }
2307   case OMPRTL__kmpc_task_reduction_get_th_data: {
2308     // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
2309     // *d);
2310     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2311     auto *FnTy =
2312         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2313     RTLFn = CGM.CreateRuntimeFunction(
2314         FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data");
2315     break;
2316   }
2317   case OMPRTL__kmpc_alloc: {
2318     // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t
2319     // al); omp_allocator_handle_t type is void *.
2320     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy};
2321     auto *FnTy =
2322         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
2323     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc");
2324     break;
2325   }
2326   case OMPRTL__kmpc_free: {
2327     // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t
2328     // al); omp_allocator_handle_t type is void *.
2329     llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy};
2330     auto *FnTy =
2331         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2332     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free");
2333     break;
2334   }
2335   case OMPRTL__kmpc_push_target_tripcount: {
2336     // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64
2337     // size);
2338     llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty};
2339     llvm::FunctionType *FnTy =
2340         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2341     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount");
2342     break;
2343   }
2344   case OMPRTL__tgt_target: {
2345     // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t
2346     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2347     // *arg_types);
2348     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2349                                 CGM.VoidPtrTy,
2350                                 CGM.Int32Ty,
2351                                 CGM.VoidPtrPtrTy,
2352                                 CGM.VoidPtrPtrTy,
2353                                 CGM.Int64Ty->getPointerTo(),
2354                                 CGM.Int64Ty->getPointerTo()};
2355     auto *FnTy =
2356         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2357     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target");
2358     break;
2359   }
2360   case OMPRTL__tgt_target_nowait: {
2361     // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr,
2362     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2363     // int64_t *arg_types);
2364     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2365                                 CGM.VoidPtrTy,
2366                                 CGM.Int32Ty,
2367                                 CGM.VoidPtrPtrTy,
2368                                 CGM.VoidPtrPtrTy,
2369                                 CGM.Int64Ty->getPointerTo(),
2370                                 CGM.Int64Ty->getPointerTo()};
2371     auto *FnTy =
2372         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2373     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait");
2374     break;
2375   }
2376   case OMPRTL__tgt_target_teams: {
2377     // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr,
2378     // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes,
2379     // int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2380     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2381                                 CGM.VoidPtrTy,
2382                                 CGM.Int32Ty,
2383                                 CGM.VoidPtrPtrTy,
2384                                 CGM.VoidPtrPtrTy,
2385                                 CGM.Int64Ty->getPointerTo(),
2386                                 CGM.Int64Ty->getPointerTo(),
2387                                 CGM.Int32Ty,
2388                                 CGM.Int32Ty};
2389     auto *FnTy =
2390         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2391     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams");
2392     break;
2393   }
2394   case OMPRTL__tgt_target_teams_nowait: {
2395     // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void
2396     // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t
2397     // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit);
2398     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2399                                 CGM.VoidPtrTy,
2400                                 CGM.Int32Ty,
2401                                 CGM.VoidPtrPtrTy,
2402                                 CGM.VoidPtrPtrTy,
2403                                 CGM.Int64Ty->getPointerTo(),
2404                                 CGM.Int64Ty->getPointerTo(),
2405                                 CGM.Int32Ty,
2406                                 CGM.Int32Ty};
2407     auto *FnTy =
2408         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2409     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait");
2410     break;
2411   }
2412   case OMPRTL__tgt_register_requires: {
2413     // Build void __tgt_register_requires(int64_t flags);
2414     llvm::Type *TypeParams[] = {CGM.Int64Ty};
2415     auto *FnTy =
2416         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2417     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires");
2418     break;
2419   }
2420   case OMPRTL__tgt_register_lib: {
2421     // Build void __tgt_register_lib(__tgt_bin_desc *desc);
2422     QualType ParamTy =
2423         CGM.getContext().getPointerType(getTgtBinaryDescriptorQTy());
2424     llvm::Type *TypeParams[] = {CGM.getTypes().ConvertTypeForMem(ParamTy)};
2425     auto *FnTy =
2426         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2427     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_lib");
2428     break;
2429   }
2430   case OMPRTL__tgt_unregister_lib: {
2431     // Build void __tgt_unregister_lib(__tgt_bin_desc *desc);
2432     QualType ParamTy =
2433         CGM.getContext().getPointerType(getTgtBinaryDescriptorQTy());
2434     llvm::Type *TypeParams[] = {CGM.getTypes().ConvertTypeForMem(ParamTy)};
2435     auto *FnTy =
2436         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2437     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_unregister_lib");
2438     break;
2439   }
2440   case OMPRTL__tgt_target_data_begin: {
2441     // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num,
2442     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2443     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2444                                 CGM.Int32Ty,
2445                                 CGM.VoidPtrPtrTy,
2446                                 CGM.VoidPtrPtrTy,
2447                                 CGM.Int64Ty->getPointerTo(),
2448                                 CGM.Int64Ty->getPointerTo()};
2449     auto *FnTy =
2450         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2451     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin");
2452     break;
2453   }
2454   case OMPRTL__tgt_target_data_begin_nowait: {
2455     // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t
2456     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2457     // *arg_types);
2458     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2459                                 CGM.Int32Ty,
2460                                 CGM.VoidPtrPtrTy,
2461                                 CGM.VoidPtrPtrTy,
2462                                 CGM.Int64Ty->getPointerTo(),
2463                                 CGM.Int64Ty->getPointerTo()};
2464     auto *FnTy =
2465         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2466     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait");
2467     break;
2468   }
2469   case OMPRTL__tgt_target_data_end: {
2470     // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num,
2471     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2472     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2473                                 CGM.Int32Ty,
2474                                 CGM.VoidPtrPtrTy,
2475                                 CGM.VoidPtrPtrTy,
2476                                 CGM.Int64Ty->getPointerTo(),
2477                                 CGM.Int64Ty->getPointerTo()};
2478     auto *FnTy =
2479         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2480     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end");
2481     break;
2482   }
2483   case OMPRTL__tgt_target_data_end_nowait: {
2484     // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t
2485     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2486     // *arg_types);
2487     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2488                                 CGM.Int32Ty,
2489                                 CGM.VoidPtrPtrTy,
2490                                 CGM.VoidPtrPtrTy,
2491                                 CGM.Int64Ty->getPointerTo(),
2492                                 CGM.Int64Ty->getPointerTo()};
2493     auto *FnTy =
2494         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2495     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait");
2496     break;
2497   }
2498   case OMPRTL__tgt_target_data_update: {
2499     // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num,
2500     // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types);
2501     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2502                                 CGM.Int32Ty,
2503                                 CGM.VoidPtrPtrTy,
2504                                 CGM.VoidPtrPtrTy,
2505                                 CGM.Int64Ty->getPointerTo(),
2506                                 CGM.Int64Ty->getPointerTo()};
2507     auto *FnTy =
2508         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2509     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update");
2510     break;
2511   }
2512   case OMPRTL__tgt_target_data_update_nowait: {
2513     // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t
2514     // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t
2515     // *arg_types);
2516     llvm::Type *TypeParams[] = {CGM.Int64Ty,
2517                                 CGM.Int32Ty,
2518                                 CGM.VoidPtrPtrTy,
2519                                 CGM.VoidPtrPtrTy,
2520                                 CGM.Int64Ty->getPointerTo(),
2521                                 CGM.Int64Ty->getPointerTo()};
2522     auto *FnTy =
2523         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2524     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait");
2525     break;
2526   }
2527   case OMPRTL__tgt_mapper_num_components: {
2528     // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle);
2529     llvm::Type *TypeParams[] = {CGM.VoidPtrTy};
2530     auto *FnTy =
2531         llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false);
2532     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components");
2533     break;
2534   }
2535   case OMPRTL__tgt_push_mapper_component: {
2536     // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void
2537     // *base, void *begin, int64_t size, int64_t type);
2538     llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy,
2539                                 CGM.Int64Ty, CGM.Int64Ty};
2540     auto *FnTy =
2541         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2542     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component");
2543     break;
2544   }
2545   }
2546   assert(RTLFn && "Unable to find OpenMP runtime function");
2547   return RTLFn;
2548 }
2549 
2550 llvm::FunctionCallee
2551 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) {
2552   assert((IVSize == 32 || IVSize == 64) &&
2553          "IV size is not compatible with the omp runtime");
2554   StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4"
2555                                             : "__kmpc_for_static_init_4u")
2556                                 : (IVSigned ? "__kmpc_for_static_init_8"
2557                                             : "__kmpc_for_static_init_8u");
2558   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2559   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2560   llvm::Type *TypeParams[] = {
2561     getIdentTyPointerTy(),                     // loc
2562     CGM.Int32Ty,                               // tid
2563     CGM.Int32Ty,                               // schedtype
2564     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2565     PtrTy,                                     // p_lower
2566     PtrTy,                                     // p_upper
2567     PtrTy,                                     // p_stride
2568     ITy,                                       // incr
2569     ITy                                        // chunk
2570   };
2571   auto *FnTy =
2572       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2573   return CGM.CreateRuntimeFunction(FnTy, Name);
2574 }
2575 
2576 llvm::FunctionCallee
2577 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) {
2578   assert((IVSize == 32 || IVSize == 64) &&
2579          "IV size is not compatible with the omp runtime");
2580   StringRef Name =
2581       IVSize == 32
2582           ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u")
2583           : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u");
2584   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2585   llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc
2586                                CGM.Int32Ty,           // tid
2587                                CGM.Int32Ty,           // schedtype
2588                                ITy,                   // lower
2589                                ITy,                   // upper
2590                                ITy,                   // stride
2591                                ITy                    // chunk
2592   };
2593   auto *FnTy =
2594       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
2595   return CGM.CreateRuntimeFunction(FnTy, Name);
2596 }
2597 
2598 llvm::FunctionCallee
2599 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) {
2600   assert((IVSize == 32 || IVSize == 64) &&
2601          "IV size is not compatible with the omp runtime");
2602   StringRef Name =
2603       IVSize == 32
2604           ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u")
2605           : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u");
2606   llvm::Type *TypeParams[] = {
2607       getIdentTyPointerTy(), // loc
2608       CGM.Int32Ty,           // tid
2609   };
2610   auto *FnTy =
2611       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
2612   return CGM.CreateRuntimeFunction(FnTy, Name);
2613 }
2614 
2615 llvm::FunctionCallee
2616 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) {
2617   assert((IVSize == 32 || IVSize == 64) &&
2618          "IV size is not compatible with the omp runtime");
2619   StringRef Name =
2620       IVSize == 32
2621           ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u")
2622           : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u");
2623   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
2624   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
2625   llvm::Type *TypeParams[] = {
2626     getIdentTyPointerTy(),                     // loc
2627     CGM.Int32Ty,                               // tid
2628     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
2629     PtrTy,                                     // p_lower
2630     PtrTy,                                     // p_upper
2631     PtrTy                                      // p_stride
2632   };
2633   auto *FnTy =
2634       llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
2635   return CGM.CreateRuntimeFunction(FnTy, Name);
2636 }
2637 
2638 /// Obtain information that uniquely identifies a target entry. This
2639 /// consists of the file and device IDs as well as line number associated with
2640 /// the relevant entry source location.
2641 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc,
2642                                      unsigned &DeviceID, unsigned &FileID,
2643                                      unsigned &LineNum) {
2644   SourceManager &SM = C.getSourceManager();
2645 
2646   // The loc should be always valid and have a file ID (the user cannot use
2647   // #pragma directives in macros)
2648 
2649   assert(Loc.isValid() && "Source location is expected to be always valid.");
2650 
2651   PresumedLoc PLoc = SM.getPresumedLoc(Loc);
2652   assert(PLoc.isValid() && "Source location is expected to be always valid.");
2653 
2654   llvm::sys::fs::UniqueID ID;
2655   if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID))
2656     SM.getDiagnostics().Report(diag::err_cannot_open_file)
2657         << PLoc.getFilename() << EC.message();
2658 
2659   DeviceID = ID.getDevice();
2660   FileID = ID.getFile();
2661   LineNum = PLoc.getLine();
2662 }
2663 
2664 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) {
2665   if (CGM.getLangOpts().OpenMPSimd)
2666     return Address::invalid();
2667   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2668       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2669   if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link ||
2670               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2671                HasRequiresUnifiedSharedMemory))) {
2672     SmallString<64> PtrName;
2673     {
2674       llvm::raw_svector_ostream OS(PtrName);
2675       OS << CGM.getMangledName(GlobalDecl(VD));
2676       if (!VD->isExternallyVisible()) {
2677         unsigned DeviceID, FileID, Line;
2678         getTargetEntryUniqueInfo(CGM.getContext(),
2679                                  VD->getCanonicalDecl()->getBeginLoc(),
2680                                  DeviceID, FileID, Line);
2681         OS << llvm::format("_%x", FileID);
2682       }
2683       OS << "_decl_tgt_ref_ptr";
2684     }
2685     llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName);
2686     if (!Ptr) {
2687       QualType PtrTy = CGM.getContext().getPointerType(VD->getType());
2688       Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy),
2689                                         PtrName);
2690 
2691       auto *GV = cast<llvm::GlobalVariable>(Ptr);
2692       GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
2693 
2694       if (!CGM.getLangOpts().OpenMPIsDevice)
2695         GV->setInitializer(CGM.GetAddrOfGlobal(VD));
2696       registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr));
2697     }
2698     return Address(Ptr, CGM.getContext().getDeclAlign(VD));
2699   }
2700   return Address::invalid();
2701 }
2702 
2703 llvm::Constant *
2704 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
2705   assert(!CGM.getLangOpts().OpenMPUseTLS ||
2706          !CGM.getContext().getTargetInfo().isTLSSupported());
2707   // Lookup the entry, lazily creating it if necessary.
2708   std::string Suffix = getName({"cache", ""});
2709   return getOrCreateInternalVariable(
2710       CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix));
2711 }
2712 
2713 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
2714                                                 const VarDecl *VD,
2715                                                 Address VDAddr,
2716                                                 SourceLocation Loc) {
2717   if (CGM.getLangOpts().OpenMPUseTLS &&
2718       CGM.getContext().getTargetInfo().isTLSSupported())
2719     return VDAddr;
2720 
2721   llvm::Type *VarTy = VDAddr.getElementType();
2722   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2723                          CGF.Builder.CreatePointerCast(VDAddr.getPointer(),
2724                                                        CGM.Int8PtrTy),
2725                          CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
2726                          getOrCreateThreadPrivateCache(VD)};
2727   return Address(CGF.EmitRuntimeCall(
2728       createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
2729                  VDAddr.getAlignment());
2730 }
2731 
2732 void CGOpenMPRuntime::emitThreadPrivateVarInit(
2733     CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor,
2734     llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
2735   // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
2736   // library.
2737   llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc);
2738   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
2739                       OMPLoc);
2740   // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
2741   // to register constructor/destructor for variable.
2742   llvm::Value *Args[] = {
2743       OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy),
2744       Ctor, CopyCtor, Dtor};
2745   CGF.EmitRuntimeCall(
2746       createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args);
2747 }
2748 
2749 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
2750     const VarDecl *VD, Address VDAddr, SourceLocation Loc,
2751     bool PerformInit, CodeGenFunction *CGF) {
2752   if (CGM.getLangOpts().OpenMPUseTLS &&
2753       CGM.getContext().getTargetInfo().isTLSSupported())
2754     return nullptr;
2755 
2756   VD = VD->getDefinition(CGM.getContext());
2757   if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) {
2758     QualType ASTTy = VD->getType();
2759 
2760     llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
2761     const Expr *Init = VD->getAnyInitializer();
2762     if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2763       // Generate function that re-emits the declaration's initializer into the
2764       // threadprivate copy of the variable VD
2765       CodeGenFunction CtorCGF(CGM);
2766       FunctionArgList Args;
2767       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2768                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2769                             ImplicitParamDecl::Other);
2770       Args.push_back(&Dst);
2771 
2772       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2773           CGM.getContext().VoidPtrTy, Args);
2774       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2775       std::string Name = getName({"__kmpc_global_ctor_", ""});
2776       llvm::Function *Fn =
2777           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2778       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
2779                             Args, Loc, Loc);
2780       llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar(
2781           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2782           CGM.getContext().VoidPtrTy, Dst.getLocation());
2783       Address Arg = Address(ArgVal, VDAddr.getAlignment());
2784       Arg = CtorCGF.Builder.CreateElementBitCast(
2785           Arg, CtorCGF.ConvertTypeForMem(ASTTy));
2786       CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
2787                                /*IsInitializer=*/true);
2788       ArgVal = CtorCGF.EmitLoadOfScalar(
2789           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
2790           CGM.getContext().VoidPtrTy, Dst.getLocation());
2791       CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
2792       CtorCGF.FinishFunction();
2793       Ctor = Fn;
2794     }
2795     if (VD->getType().isDestructedType() != QualType::DK_none) {
2796       // Generate function that emits destructor call for the threadprivate copy
2797       // of the variable VD
2798       CodeGenFunction DtorCGF(CGM);
2799       FunctionArgList Args;
2800       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
2801                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
2802                             ImplicitParamDecl::Other);
2803       Args.push_back(&Dst);
2804 
2805       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
2806           CGM.getContext().VoidTy, Args);
2807       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2808       std::string Name = getName({"__kmpc_global_dtor_", ""});
2809       llvm::Function *Fn =
2810           CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc);
2811       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2812       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
2813                             Loc, Loc);
2814       // Create a scope with an artificial location for the body of this function.
2815       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
2816       llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar(
2817           DtorCGF.GetAddrOfLocalVar(&Dst),
2818           /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation());
2819       DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy,
2820                           DtorCGF.getDestroyer(ASTTy.isDestructedType()),
2821                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
2822       DtorCGF.FinishFunction();
2823       Dtor = Fn;
2824     }
2825     // Do not emit init function if it is not required.
2826     if (!Ctor && !Dtor)
2827       return nullptr;
2828 
2829     llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
2830     auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs,
2831                                                /*isVarArg=*/false)
2832                            ->getPointerTo();
2833     // Copying constructor for the threadprivate variable.
2834     // Must be NULL - reserved by runtime, but currently it requires that this
2835     // parameter is always NULL. Otherwise it fires assertion.
2836     CopyCtor = llvm::Constant::getNullValue(CopyCtorTy);
2837     if (Ctor == nullptr) {
2838       auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
2839                                              /*isVarArg=*/false)
2840                          ->getPointerTo();
2841       Ctor = llvm::Constant::getNullValue(CtorTy);
2842     }
2843     if (Dtor == nullptr) {
2844       auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy,
2845                                              /*isVarArg=*/false)
2846                          ->getPointerTo();
2847       Dtor = llvm::Constant::getNullValue(DtorTy);
2848     }
2849     if (!CGF) {
2850       auto *InitFunctionTy =
2851           llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
2852       std::string Name = getName({"__omp_threadprivate_init_", ""});
2853       llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction(
2854           InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction());
2855       CodeGenFunction InitCGF(CGM);
2856       FunctionArgList ArgList;
2857       InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
2858                             CGM.getTypes().arrangeNullaryFunction(), ArgList,
2859                             Loc, Loc);
2860       emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2861       InitCGF.FinishFunction();
2862       return InitFunction;
2863     }
2864     emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
2865   }
2866   return nullptr;
2867 }
2868 
2869 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD,
2870                                                      llvm::GlobalVariable *Addr,
2871                                                      bool PerformInit) {
2872   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
2873       !CGM.getLangOpts().OpenMPIsDevice)
2874     return false;
2875   Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
2876       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
2877   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
2878       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
2879        HasRequiresUnifiedSharedMemory))
2880     return CGM.getLangOpts().OpenMPIsDevice;
2881   VD = VD->getDefinition(CGM.getContext());
2882   if (VD && !DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second)
2883     return CGM.getLangOpts().OpenMPIsDevice;
2884 
2885   QualType ASTTy = VD->getType();
2886 
2887   SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc();
2888   // Produce the unique prefix to identify the new target regions. We use
2889   // the source location of the variable declaration which we know to not
2890   // conflict with any target region.
2891   unsigned DeviceID;
2892   unsigned FileID;
2893   unsigned Line;
2894   getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line);
2895   SmallString<128> Buffer, Out;
2896   {
2897     llvm::raw_svector_ostream OS(Buffer);
2898     OS << "__omp_offloading_" << llvm::format("_%x", DeviceID)
2899        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
2900   }
2901 
2902   const Expr *Init = VD->getAnyInitializer();
2903   if (CGM.getLangOpts().CPlusPlus && PerformInit) {
2904     llvm::Constant *Ctor;
2905     llvm::Constant *ID;
2906     if (CGM.getLangOpts().OpenMPIsDevice) {
2907       // Generate function that re-emits the declaration's initializer into
2908       // the threadprivate copy of the variable VD
2909       CodeGenFunction CtorCGF(CGM);
2910 
2911       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2912       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2913       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2914           FTy, Twine(Buffer, "_ctor"), FI, Loc);
2915       auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF);
2916       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2917                             FunctionArgList(), Loc, Loc);
2918       auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF);
2919       CtorCGF.EmitAnyExprToMem(Init,
2920                                Address(Addr, CGM.getContext().getDeclAlign(VD)),
2921                                Init->getType().getQualifiers(),
2922                                /*IsInitializer=*/true);
2923       CtorCGF.FinishFunction();
2924       Ctor = Fn;
2925       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
2926       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor));
2927     } else {
2928       Ctor = new llvm::GlobalVariable(
2929           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
2930           llvm::GlobalValue::PrivateLinkage,
2931           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor"));
2932       ID = Ctor;
2933     }
2934 
2935     // Register the information for the entry associated with the constructor.
2936     Out.clear();
2937     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2938         DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor,
2939         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor);
2940   }
2941   if (VD->getType().isDestructedType() != QualType::DK_none) {
2942     llvm::Constant *Dtor;
2943     llvm::Constant *ID;
2944     if (CGM.getLangOpts().OpenMPIsDevice) {
2945       // Generate function that emits destructor call for the threadprivate
2946       // copy of the variable VD
2947       CodeGenFunction DtorCGF(CGM);
2948 
2949       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
2950       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
2951       llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction(
2952           FTy, Twine(Buffer, "_dtor"), FI, Loc);
2953       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
2954       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
2955                             FunctionArgList(), Loc, Loc);
2956       // Create a scope with an artificial location for the body of this
2957       // function.
2958       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
2959       DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)),
2960                           ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()),
2961                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
2962       DtorCGF.FinishFunction();
2963       Dtor = Fn;
2964       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
2965       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor));
2966     } else {
2967       Dtor = new llvm::GlobalVariable(
2968           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
2969           llvm::GlobalValue::PrivateLinkage,
2970           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor"));
2971       ID = Dtor;
2972     }
2973     // Register the information for the entry associated with the destructor.
2974     Out.clear();
2975     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2976         DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor,
2977         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor);
2978   }
2979   return CGM.getLangOpts().OpenMPIsDevice;
2980 }
2981 
2982 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF,
2983                                                           QualType VarType,
2984                                                           StringRef Name) {
2985   std::string Suffix = getName({"artificial", ""});
2986   std::string CacheSuffix = getName({"cache", ""});
2987   llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType);
2988   llvm::Value *GAddr =
2989       getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix));
2990   llvm::Value *Args[] = {
2991       emitUpdateLocation(CGF, SourceLocation()),
2992       getThreadID(CGF, SourceLocation()),
2993       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy),
2994       CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy,
2995                                 /*isSigned=*/false),
2996       getOrCreateInternalVariable(
2997           CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))};
2998   return Address(
2999       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3000           CGF.EmitRuntimeCall(
3001               createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args),
3002           VarLVType->getPointerTo(/*AddrSpace=*/0)),
3003       CGM.getPointerAlign());
3004 }
3005 
3006 void CGOpenMPRuntime::emitOMPIfClause(CodeGenFunction &CGF, const Expr *Cond,
3007                                       const RegionCodeGenTy &ThenGen,
3008                                       const RegionCodeGenTy &ElseGen) {
3009   CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
3010 
3011   // If the condition constant folds and can be elided, try to avoid emitting
3012   // the condition and the dead arm of the if/else.
3013   bool CondConstant;
3014   if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
3015     if (CondConstant)
3016       ThenGen(CGF);
3017     else
3018       ElseGen(CGF);
3019     return;
3020   }
3021 
3022   // Otherwise, the condition did not fold, or we couldn't elide it.  Just
3023   // emit the conditional branch.
3024   llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then");
3025   llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else");
3026   llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end");
3027   CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0);
3028 
3029   // Emit the 'then' code.
3030   CGF.EmitBlock(ThenBlock);
3031   ThenGen(CGF);
3032   CGF.EmitBranch(ContBlock);
3033   // Emit the 'else' code if present.
3034   // There is no need to emit line number for unconditional branch.
3035   (void)ApplyDebugLocation::CreateEmpty(CGF);
3036   CGF.EmitBlock(ElseBlock);
3037   ElseGen(CGF);
3038   // There is no need to emit line number for unconditional branch.
3039   (void)ApplyDebugLocation::CreateEmpty(CGF);
3040   CGF.EmitBranch(ContBlock);
3041   // Emit the continuation block for code after the if.
3042   CGF.EmitBlock(ContBlock, /*IsFinished=*/true);
3043 }
3044 
3045 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc,
3046                                        llvm::Function *OutlinedFn,
3047                                        ArrayRef<llvm::Value *> CapturedVars,
3048                                        const Expr *IfCond) {
3049   if (!CGF.HaveInsertPoint())
3050     return;
3051   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
3052   auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF,
3053                                                      PrePostActionTy &) {
3054     // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn);
3055     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3056     llvm::Value *Args[] = {
3057         RTLoc,
3058         CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
3059         CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())};
3060     llvm::SmallVector<llvm::Value *, 16> RealArgs;
3061     RealArgs.append(std::begin(Args), std::end(Args));
3062     RealArgs.append(CapturedVars.begin(), CapturedVars.end());
3063 
3064     llvm::FunctionCallee RTLFn =
3065         RT.createRuntimeFunction(OMPRTL__kmpc_fork_call);
3066     CGF.EmitRuntimeCall(RTLFn, RealArgs);
3067   };
3068   auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF,
3069                                                           PrePostActionTy &) {
3070     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
3071     llvm::Value *ThreadID = RT.getThreadID(CGF, Loc);
3072     // Build calls:
3073     // __kmpc_serialized_parallel(&Loc, GTid);
3074     llvm::Value *Args[] = {RTLoc, ThreadID};
3075     CGF.EmitRuntimeCall(
3076         RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args);
3077 
3078     // OutlinedFn(&GTid, &zero, CapturedStruct);
3079     Address ZeroAddr = CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty,
3080                                                         /*Name*/ ".zero.addr");
3081     CGF.InitTempAlloca(ZeroAddr, CGF.Builder.getInt32(/*C*/ 0));
3082     llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs;
3083     // ThreadId for serialized parallels is 0.
3084     OutlinedFnArgs.push_back(ZeroAddr.getPointer());
3085     OutlinedFnArgs.push_back(ZeroAddr.getPointer());
3086     OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end());
3087     RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs);
3088 
3089     // __kmpc_end_serialized_parallel(&Loc, GTid);
3090     llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID};
3091     CGF.EmitRuntimeCall(
3092         RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel),
3093         EndArgs);
3094   };
3095   if (IfCond) {
3096     emitOMPIfClause(CGF, IfCond, ThenGen, ElseGen);
3097   } else {
3098     RegionCodeGenTy ThenRCG(ThenGen);
3099     ThenRCG(CGF);
3100   }
3101 }
3102 
3103 // If we're inside an (outlined) parallel region, use the region info's
3104 // thread-ID variable (it is passed in a first argument of the outlined function
3105 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
3106 // regular serial code region, get thread ID by calling kmp_int32
3107 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
3108 // return the address of that temp.
3109 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
3110                                              SourceLocation Loc) {
3111   if (auto *OMPRegionInfo =
3112           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3113     if (OMPRegionInfo->getThreadIDVariable())
3114       return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress();
3115 
3116   llvm::Value *ThreadID = getThreadID(CGF, Loc);
3117   QualType Int32Ty =
3118       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
3119   Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp.");
3120   CGF.EmitStoreOfScalar(ThreadID,
3121                         CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty));
3122 
3123   return ThreadIDTemp;
3124 }
3125 
3126 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable(
3127     llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) {
3128   SmallString<256> Buffer;
3129   llvm::raw_svector_ostream Out(Buffer);
3130   Out << Name;
3131   StringRef RuntimeName = Out.str();
3132   auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first;
3133   if (Elem.second) {
3134     assert(Elem.second->getType()->getPointerElementType() == Ty &&
3135            "OMP internal variable has different type than requested");
3136     return &*Elem.second;
3137   }
3138 
3139   return Elem.second = new llvm::GlobalVariable(
3140              CGM.getModule(), Ty, /*IsConstant*/ false,
3141              llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty),
3142              Elem.first(), /*InsertBefore=*/nullptr,
3143              llvm::GlobalValue::NotThreadLocal, AddressSpace);
3144 }
3145 
3146 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
3147   std::string Prefix = Twine("gomp_critical_user_", CriticalName).str();
3148   std::string Name = getName({Prefix, "var"});
3149   return getOrCreateInternalVariable(KmpCriticalNameTy, Name);
3150 }
3151 
3152 namespace {
3153 /// Common pre(post)-action for different OpenMP constructs.
3154 class CommonActionTy final : public PrePostActionTy {
3155   llvm::FunctionCallee EnterCallee;
3156   ArrayRef<llvm::Value *> EnterArgs;
3157   llvm::FunctionCallee ExitCallee;
3158   ArrayRef<llvm::Value *> ExitArgs;
3159   bool Conditional;
3160   llvm::BasicBlock *ContBlock = nullptr;
3161 
3162 public:
3163   CommonActionTy(llvm::FunctionCallee EnterCallee,
3164                  ArrayRef<llvm::Value *> EnterArgs,
3165                  llvm::FunctionCallee ExitCallee,
3166                  ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false)
3167       : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee),
3168         ExitArgs(ExitArgs), Conditional(Conditional) {}
3169   void Enter(CodeGenFunction &CGF) override {
3170     llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs);
3171     if (Conditional) {
3172       llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes);
3173       auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
3174       ContBlock = CGF.createBasicBlock("omp_if.end");
3175       // Generate the branch (If-stmt)
3176       CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
3177       CGF.EmitBlock(ThenBlock);
3178     }
3179   }
3180   void Done(CodeGenFunction &CGF) {
3181     // Emit the rest of blocks/branches
3182     CGF.EmitBranch(ContBlock);
3183     CGF.EmitBlock(ContBlock, true);
3184   }
3185   void Exit(CodeGenFunction &CGF) override {
3186     CGF.EmitRuntimeCall(ExitCallee, ExitArgs);
3187   }
3188 };
3189 } // anonymous namespace
3190 
3191 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF,
3192                                          StringRef CriticalName,
3193                                          const RegionCodeGenTy &CriticalOpGen,
3194                                          SourceLocation Loc, const Expr *Hint) {
3195   // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]);
3196   // CriticalOpGen();
3197   // __kmpc_end_critical(ident_t *, gtid, Lock);
3198   // Prepare arguments and build a call to __kmpc_critical
3199   if (!CGF.HaveInsertPoint())
3200     return;
3201   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3202                          getCriticalRegionLock(CriticalName)};
3203   llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args),
3204                                                 std::end(Args));
3205   if (Hint) {
3206     EnterArgs.push_back(CGF.Builder.CreateIntCast(
3207         CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false));
3208   }
3209   CommonActionTy Action(
3210       createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint
3211                                  : OMPRTL__kmpc_critical),
3212       EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args);
3213   CriticalOpGen.setAction(Action);
3214   emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen);
3215 }
3216 
3217 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
3218                                        const RegionCodeGenTy &MasterOpGen,
3219                                        SourceLocation Loc) {
3220   if (!CGF.HaveInsertPoint())
3221     return;
3222   // if(__kmpc_master(ident_t *, gtid)) {
3223   //   MasterOpGen();
3224   //   __kmpc_end_master(ident_t *, gtid);
3225   // }
3226   // Prepare arguments and build a call to __kmpc_master
3227   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3228   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args,
3229                         createRuntimeFunction(OMPRTL__kmpc_end_master), Args,
3230                         /*Conditional=*/true);
3231   MasterOpGen.setAction(Action);
3232   emitInlinedDirective(CGF, OMPD_master, MasterOpGen);
3233   Action.Done(CGF);
3234 }
3235 
3236 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
3237                                         SourceLocation Loc) {
3238   if (!CGF.HaveInsertPoint())
3239     return;
3240   // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
3241   llvm::Value *Args[] = {
3242       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3243       llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
3244   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), Args);
3245   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
3246     Region->emitUntiedSwitch(CGF);
3247 }
3248 
3249 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF,
3250                                           const RegionCodeGenTy &TaskgroupOpGen,
3251                                           SourceLocation Loc) {
3252   if (!CGF.HaveInsertPoint())
3253     return;
3254   // __kmpc_taskgroup(ident_t *, gtid);
3255   // TaskgroupOpGen();
3256   // __kmpc_end_taskgroup(ident_t *, gtid);
3257   // Prepare arguments and build a call to __kmpc_taskgroup
3258   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3259   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args,
3260                         createRuntimeFunction(OMPRTL__kmpc_end_taskgroup),
3261                         Args);
3262   TaskgroupOpGen.setAction(Action);
3263   emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen);
3264 }
3265 
3266 /// Given an array of pointers to variables, project the address of a
3267 /// given variable.
3268 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array,
3269                                       unsigned Index, const VarDecl *Var) {
3270   // Pull out the pointer to the variable.
3271   Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index);
3272   llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr);
3273 
3274   Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var));
3275   Addr = CGF.Builder.CreateElementBitCast(
3276       Addr, CGF.ConvertTypeForMem(Var->getType()));
3277   return Addr;
3278 }
3279 
3280 static llvm::Value *emitCopyprivateCopyFunction(
3281     CodeGenModule &CGM, llvm::Type *ArgsType,
3282     ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
3283     ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps,
3284     SourceLocation Loc) {
3285   ASTContext &C = CGM.getContext();
3286   // void copy_func(void *LHSArg, void *RHSArg);
3287   FunctionArgList Args;
3288   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3289                            ImplicitParamDecl::Other);
3290   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
3291                            ImplicitParamDecl::Other);
3292   Args.push_back(&LHSArg);
3293   Args.push_back(&RHSArg);
3294   const auto &CGFI =
3295       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
3296   std::string Name =
3297       CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"});
3298   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
3299                                     llvm::GlobalValue::InternalLinkage, Name,
3300                                     &CGM.getModule());
3301   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
3302   Fn->setDoesNotRecurse();
3303   CodeGenFunction CGF(CGM);
3304   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
3305   // Dest = (void*[n])(LHSArg);
3306   // Src = (void*[n])(RHSArg);
3307   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3308       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
3309       ArgsType), CGF.getPointerAlign());
3310   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3311       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
3312       ArgsType), CGF.getPointerAlign());
3313   // *(Type0*)Dst[0] = *(Type0*)Src[0];
3314   // *(Type1*)Dst[1] = *(Type1*)Src[1];
3315   // ...
3316   // *(Typen*)Dst[n] = *(Typen*)Src[n];
3317   for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
3318     const auto *DestVar =
3319         cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl());
3320     Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar);
3321 
3322     const auto *SrcVar =
3323         cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl());
3324     Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar);
3325 
3326     const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl();
3327     QualType Type = VD->getType();
3328     CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]);
3329   }
3330   CGF.FinishFunction();
3331   return Fn;
3332 }
3333 
3334 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
3335                                        const RegionCodeGenTy &SingleOpGen,
3336                                        SourceLocation Loc,
3337                                        ArrayRef<const Expr *> CopyprivateVars,
3338                                        ArrayRef<const Expr *> SrcExprs,
3339                                        ArrayRef<const Expr *> DstExprs,
3340                                        ArrayRef<const Expr *> AssignmentOps) {
3341   if (!CGF.HaveInsertPoint())
3342     return;
3343   assert(CopyprivateVars.size() == SrcExprs.size() &&
3344          CopyprivateVars.size() == DstExprs.size() &&
3345          CopyprivateVars.size() == AssignmentOps.size());
3346   ASTContext &C = CGM.getContext();
3347   // int32 did_it = 0;
3348   // if(__kmpc_single(ident_t *, gtid)) {
3349   //   SingleOpGen();
3350   //   __kmpc_end_single(ident_t *, gtid);
3351   //   did_it = 1;
3352   // }
3353   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3354   // <copy_func>, did_it);
3355 
3356   Address DidIt = Address::invalid();
3357   if (!CopyprivateVars.empty()) {
3358     // int32 did_it = 0;
3359     QualType KmpInt32Ty =
3360         C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
3361     DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it");
3362     CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt);
3363   }
3364   // Prepare arguments and build a call to __kmpc_single
3365   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3366   CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args,
3367                         createRuntimeFunction(OMPRTL__kmpc_end_single), Args,
3368                         /*Conditional=*/true);
3369   SingleOpGen.setAction(Action);
3370   emitInlinedDirective(CGF, OMPD_single, SingleOpGen);
3371   if (DidIt.isValid()) {
3372     // did_it = 1;
3373     CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt);
3374   }
3375   Action.Done(CGF);
3376   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
3377   // <copy_func>, did_it);
3378   if (DidIt.isValid()) {
3379     llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
3380     QualType CopyprivateArrayTy = C.getConstantArrayType(
3381         C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
3382         /*IndexTypeQuals=*/0);
3383     // Create a list of all private variables for copyprivate.
3384     Address CopyprivateList =
3385         CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list");
3386     for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
3387       Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I);
3388       CGF.Builder.CreateStore(
3389           CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3390               CGF.EmitLValue(CopyprivateVars[I]).getPointer(), CGF.VoidPtrTy),
3391           Elem);
3392     }
3393     // Build function that copies private values from single region to all other
3394     // threads in the corresponding parallel region.
3395     llvm::Value *CpyFn = emitCopyprivateCopyFunction(
3396         CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(),
3397         CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc);
3398     llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy);
3399     Address CL =
3400       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList,
3401                                                       CGF.VoidPtrTy);
3402     llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt);
3403     llvm::Value *Args[] = {
3404         emitUpdateLocation(CGF, Loc), // ident_t *<loc>
3405         getThreadID(CGF, Loc),        // i32 <gtid>
3406         BufSize,                      // size_t <buf_size>
3407         CL.getPointer(),              // void *<copyprivate list>
3408         CpyFn,                        // void (*) (void *, void *) <copy_func>
3409         DidItVal                      // i32 did_it
3410     };
3411     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args);
3412   }
3413 }
3414 
3415 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF,
3416                                         const RegionCodeGenTy &OrderedOpGen,
3417                                         SourceLocation Loc, bool IsThreads) {
3418   if (!CGF.HaveInsertPoint())
3419     return;
3420   // __kmpc_ordered(ident_t *, gtid);
3421   // OrderedOpGen();
3422   // __kmpc_end_ordered(ident_t *, gtid);
3423   // Prepare arguments and build a call to __kmpc_ordered
3424   if (IsThreads) {
3425     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3426     CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args,
3427                           createRuntimeFunction(OMPRTL__kmpc_end_ordered),
3428                           Args);
3429     OrderedOpGen.setAction(Action);
3430     emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3431     return;
3432   }
3433   emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
3434 }
3435 
3436 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) {
3437   unsigned Flags;
3438   if (Kind == OMPD_for)
3439     Flags = OMP_IDENT_BARRIER_IMPL_FOR;
3440   else if (Kind == OMPD_sections)
3441     Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS;
3442   else if (Kind == OMPD_single)
3443     Flags = OMP_IDENT_BARRIER_IMPL_SINGLE;
3444   else if (Kind == OMPD_barrier)
3445     Flags = OMP_IDENT_BARRIER_EXPL;
3446   else
3447     Flags = OMP_IDENT_BARRIER_IMPL;
3448   return Flags;
3449 }
3450 
3451 void CGOpenMPRuntime::getDefaultScheduleAndChunk(
3452     CodeGenFunction &CGF, const OMPLoopDirective &S,
3453     OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const {
3454   // Check if the loop directive is actually a doacross loop directive. In this
3455   // case choose static, 1 schedule.
3456   if (llvm::any_of(
3457           S.getClausesOfKind<OMPOrderedClause>(),
3458           [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) {
3459     ScheduleKind = OMPC_SCHEDULE_static;
3460     // Chunk size is 1 in this case.
3461     llvm::APInt ChunkSize(32, 1);
3462     ChunkExpr = IntegerLiteral::Create(
3463         CGF.getContext(), ChunkSize,
3464         CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0),
3465         SourceLocation());
3466   }
3467 }
3468 
3469 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
3470                                       OpenMPDirectiveKind Kind, bool EmitChecks,
3471                                       bool ForceSimpleCall) {
3472   if (!CGF.HaveInsertPoint())
3473     return;
3474   // Build call __kmpc_cancel_barrier(loc, thread_id);
3475   // Build call __kmpc_barrier(loc, thread_id);
3476   unsigned Flags = getDefaultFlagsForBarriers(Kind);
3477   // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc,
3478   // thread_id);
3479   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
3480                          getThreadID(CGF, Loc)};
3481   if (auto *OMPRegionInfo =
3482           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
3483     if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) {
3484       llvm::Value *Result = CGF.EmitRuntimeCall(
3485           createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args);
3486       if (EmitChecks) {
3487         // if (__kmpc_cancel_barrier()) {
3488         //   exit from construct;
3489         // }
3490         llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
3491         llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
3492         llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
3493         CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
3494         CGF.EmitBlock(ExitBB);
3495         //   exit from construct;
3496         CodeGenFunction::JumpDest CancelDestination =
3497             CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
3498         CGF.EmitBranchThroughCleanup(CancelDestination);
3499         CGF.EmitBlock(ContBB, /*IsFinished=*/true);
3500       }
3501       return;
3502     }
3503   }
3504   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args);
3505 }
3506 
3507 /// Map the OpenMP loop schedule to the runtime enumeration.
3508 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
3509                                           bool Chunked, bool Ordered) {
3510   switch (ScheduleKind) {
3511   case OMPC_SCHEDULE_static:
3512     return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked)
3513                    : (Ordered ? OMP_ord_static : OMP_sch_static);
3514   case OMPC_SCHEDULE_dynamic:
3515     return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked;
3516   case OMPC_SCHEDULE_guided:
3517     return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked;
3518   case OMPC_SCHEDULE_runtime:
3519     return Ordered ? OMP_ord_runtime : OMP_sch_runtime;
3520   case OMPC_SCHEDULE_auto:
3521     return Ordered ? OMP_ord_auto : OMP_sch_auto;
3522   case OMPC_SCHEDULE_unknown:
3523     assert(!Chunked && "chunk was specified but schedule kind not known");
3524     return Ordered ? OMP_ord_static : OMP_sch_static;
3525   }
3526   llvm_unreachable("Unexpected runtime schedule");
3527 }
3528 
3529 /// Map the OpenMP distribute schedule to the runtime enumeration.
3530 static OpenMPSchedType
3531 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) {
3532   // only static is allowed for dist_schedule
3533   return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static;
3534 }
3535 
3536 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
3537                                          bool Chunked) const {
3538   OpenMPSchedType Schedule =
3539       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3540   return Schedule == OMP_sch_static;
3541 }
3542 
3543 bool CGOpenMPRuntime::isStaticNonchunked(
3544     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3545   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3546   return Schedule == OMP_dist_sch_static;
3547 }
3548 
3549 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind,
3550                                       bool Chunked) const {
3551   OpenMPSchedType Schedule =
3552       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
3553   return Schedule == OMP_sch_static_chunked;
3554 }
3555 
3556 bool CGOpenMPRuntime::isStaticChunked(
3557     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
3558   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
3559   return Schedule == OMP_dist_sch_static_chunked;
3560 }
3561 
3562 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
3563   OpenMPSchedType Schedule =
3564       getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false);
3565   assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
3566   return Schedule != OMP_sch_static;
3567 }
3568 
3569 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule,
3570                                   OpenMPScheduleClauseModifier M1,
3571                                   OpenMPScheduleClauseModifier M2) {
3572   int Modifier = 0;
3573   switch (M1) {
3574   case OMPC_SCHEDULE_MODIFIER_monotonic:
3575     Modifier = OMP_sch_modifier_monotonic;
3576     break;
3577   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3578     Modifier = OMP_sch_modifier_nonmonotonic;
3579     break;
3580   case OMPC_SCHEDULE_MODIFIER_simd:
3581     if (Schedule == OMP_sch_static_chunked)
3582       Schedule = OMP_sch_static_balanced_chunked;
3583     break;
3584   case OMPC_SCHEDULE_MODIFIER_last:
3585   case OMPC_SCHEDULE_MODIFIER_unknown:
3586     break;
3587   }
3588   switch (M2) {
3589   case OMPC_SCHEDULE_MODIFIER_monotonic:
3590     Modifier = OMP_sch_modifier_monotonic;
3591     break;
3592   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
3593     Modifier = OMP_sch_modifier_nonmonotonic;
3594     break;
3595   case OMPC_SCHEDULE_MODIFIER_simd:
3596     if (Schedule == OMP_sch_static_chunked)
3597       Schedule = OMP_sch_static_balanced_chunked;
3598     break;
3599   case OMPC_SCHEDULE_MODIFIER_last:
3600   case OMPC_SCHEDULE_MODIFIER_unknown:
3601     break;
3602   }
3603   // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription.
3604   // If the static schedule kind is specified or if the ordered clause is
3605   // specified, and if the nonmonotonic modifier is not specified, the effect is
3606   // as if the monotonic modifier is specified. Otherwise, unless the monotonic
3607   // modifier is specified, the effect is as if the nonmonotonic modifier is
3608   // specified.
3609   if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) {
3610     if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static ||
3611           Schedule == OMP_sch_static_balanced_chunked ||
3612           Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static))
3613       Modifier = OMP_sch_modifier_nonmonotonic;
3614   }
3615   return Schedule | Modifier;
3616 }
3617 
3618 void CGOpenMPRuntime::emitForDispatchInit(
3619     CodeGenFunction &CGF, SourceLocation Loc,
3620     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
3621     bool Ordered, const DispatchRTInput &DispatchValues) {
3622   if (!CGF.HaveInsertPoint())
3623     return;
3624   OpenMPSchedType Schedule = getRuntimeSchedule(
3625       ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered);
3626   assert(Ordered ||
3627          (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked &&
3628           Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked &&
3629           Schedule != OMP_sch_static_balanced_chunked));
3630   // Call __kmpc_dispatch_init(
3631   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
3632   //          kmp_int[32|64] lower, kmp_int[32|64] upper,
3633   //          kmp_int[32|64] stride, kmp_int[32|64] chunk);
3634 
3635   // If the Chunk was not specified in the clause - use default value 1.
3636   llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk
3637                                             : CGF.Builder.getIntN(IVSize, 1);
3638   llvm::Value *Args[] = {
3639       emitUpdateLocation(CGF, Loc),
3640       getThreadID(CGF, Loc),
3641       CGF.Builder.getInt32(addMonoNonMonoModifier(
3642           CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type
3643       DispatchValues.LB,                                     // Lower
3644       DispatchValues.UB,                                     // Upper
3645       CGF.Builder.getIntN(IVSize, 1),                        // Stride
3646       Chunk                                                  // Chunk
3647   };
3648   CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args);
3649 }
3650 
3651 static void emitForStaticInitCall(
3652     CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId,
3653     llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule,
3654     OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2,
3655     const CGOpenMPRuntime::StaticRTInput &Values) {
3656   if (!CGF.HaveInsertPoint())
3657     return;
3658 
3659   assert(!Values.Ordered);
3660   assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked ||
3661          Schedule == OMP_sch_static_balanced_chunked ||
3662          Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked ||
3663          Schedule == OMP_dist_sch_static ||
3664          Schedule == OMP_dist_sch_static_chunked);
3665 
3666   // Call __kmpc_for_static_init(
3667   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
3668   //          kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
3669   //          kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
3670   //          kmp_int[32|64] incr, kmp_int[32|64] chunk);
3671   llvm::Value *Chunk = Values.Chunk;
3672   if (Chunk == nullptr) {
3673     assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static ||
3674             Schedule == OMP_dist_sch_static) &&
3675            "expected static non-chunked schedule");
3676     // If the Chunk was not specified in the clause - use default value 1.
3677     Chunk = CGF.Builder.getIntN(Values.IVSize, 1);
3678   } else {
3679     assert((Schedule == OMP_sch_static_chunked ||
3680             Schedule == OMP_sch_static_balanced_chunked ||
3681             Schedule == OMP_ord_static_chunked ||
3682             Schedule == OMP_dist_sch_static_chunked) &&
3683            "expected static chunked schedule");
3684   }
3685   llvm::Value *Args[] = {
3686       UpdateLocation,
3687       ThreadId,
3688       CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1,
3689                                                   M2)), // Schedule type
3690       Values.IL.getPointer(),                           // &isLastIter
3691       Values.LB.getPointer(),                           // &LB
3692       Values.UB.getPointer(),                           // &UB
3693       Values.ST.getPointer(),                           // &Stride
3694       CGF.Builder.getIntN(Values.IVSize, 1),            // Incr
3695       Chunk                                             // Chunk
3696   };
3697   CGF.EmitRuntimeCall(ForStaticInitFunction, Args);
3698 }
3699 
3700 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF,
3701                                         SourceLocation Loc,
3702                                         OpenMPDirectiveKind DKind,
3703                                         const OpenMPScheduleTy &ScheduleKind,
3704                                         const StaticRTInput &Values) {
3705   OpenMPSchedType ScheduleNum = getRuntimeSchedule(
3706       ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered);
3707   assert(isOpenMPWorksharingDirective(DKind) &&
3708          "Expected loop-based or sections-based directive.");
3709   llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc,
3710                                              isOpenMPLoopDirective(DKind)
3711                                                  ? OMP_IDENT_WORK_LOOP
3712                                                  : OMP_IDENT_WORK_SECTIONS);
3713   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3714   llvm::FunctionCallee StaticInitFunction =
3715       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3716   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3717                         ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values);
3718 }
3719 
3720 void CGOpenMPRuntime::emitDistributeStaticInit(
3721     CodeGenFunction &CGF, SourceLocation Loc,
3722     OpenMPDistScheduleClauseKind SchedKind,
3723     const CGOpenMPRuntime::StaticRTInput &Values) {
3724   OpenMPSchedType ScheduleNum =
3725       getRuntimeSchedule(SchedKind, Values.Chunk != nullptr);
3726   llvm::Value *UpdatedLocation =
3727       emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE);
3728   llvm::Value *ThreadId = getThreadID(CGF, Loc);
3729   llvm::FunctionCallee StaticInitFunction =
3730       createForStaticInitFunction(Values.IVSize, Values.IVSigned);
3731   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
3732                         ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown,
3733                         OMPC_SCHEDULE_MODIFIER_unknown, Values);
3734 }
3735 
3736 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF,
3737                                           SourceLocation Loc,
3738                                           OpenMPDirectiveKind DKind) {
3739   if (!CGF.HaveInsertPoint())
3740     return;
3741   // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
3742   llvm::Value *Args[] = {
3743       emitUpdateLocation(CGF, Loc,
3744                          isOpenMPDistributeDirective(DKind)
3745                              ? OMP_IDENT_WORK_DISTRIBUTE
3746                              : isOpenMPLoopDirective(DKind)
3747                                    ? OMP_IDENT_WORK_LOOP
3748                                    : OMP_IDENT_WORK_SECTIONS),
3749       getThreadID(CGF, Loc)};
3750   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini),
3751                       Args);
3752 }
3753 
3754 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
3755                                                  SourceLocation Loc,
3756                                                  unsigned IVSize,
3757                                                  bool IVSigned) {
3758   if (!CGF.HaveInsertPoint())
3759     return;
3760   // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
3761   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
3762   CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args);
3763 }
3764 
3765 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
3766                                           SourceLocation Loc, unsigned IVSize,
3767                                           bool IVSigned, Address IL,
3768                                           Address LB, Address UB,
3769                                           Address ST) {
3770   // Call __kmpc_dispatch_next(
3771   //          ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
3772   //          kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
3773   //          kmp_int[32|64] *p_stride);
3774   llvm::Value *Args[] = {
3775       emitUpdateLocation(CGF, Loc),
3776       getThreadID(CGF, Loc),
3777       IL.getPointer(), // &isLastIter
3778       LB.getPointer(), // &Lower
3779       UB.getPointer(), // &Upper
3780       ST.getPointer()  // &Stride
3781   };
3782   llvm::Value *Call =
3783       CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args);
3784   return CGF.EmitScalarConversion(
3785       Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1),
3786       CGF.getContext().BoolTy, Loc);
3787 }
3788 
3789 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
3790                                            llvm::Value *NumThreads,
3791                                            SourceLocation Loc) {
3792   if (!CGF.HaveInsertPoint())
3793     return;
3794   // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
3795   llvm::Value *Args[] = {
3796       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3797       CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)};
3798   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads),
3799                       Args);
3800 }
3801 
3802 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF,
3803                                          OpenMPProcBindClauseKind ProcBind,
3804                                          SourceLocation Loc) {
3805   if (!CGF.HaveInsertPoint())
3806     return;
3807   // Constants for proc bind value accepted by the runtime.
3808   enum ProcBindTy {
3809     ProcBindFalse = 0,
3810     ProcBindTrue,
3811     ProcBindMaster,
3812     ProcBindClose,
3813     ProcBindSpread,
3814     ProcBindIntel,
3815     ProcBindDefault
3816   } RuntimeProcBind;
3817   switch (ProcBind) {
3818   case OMPC_PROC_BIND_master:
3819     RuntimeProcBind = ProcBindMaster;
3820     break;
3821   case OMPC_PROC_BIND_close:
3822     RuntimeProcBind = ProcBindClose;
3823     break;
3824   case OMPC_PROC_BIND_spread:
3825     RuntimeProcBind = ProcBindSpread;
3826     break;
3827   case OMPC_PROC_BIND_unknown:
3828     llvm_unreachable("Unsupported proc_bind value.");
3829   }
3830   // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind)
3831   llvm::Value *Args[] = {
3832       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
3833       llvm::ConstantInt::get(CGM.IntTy, RuntimeProcBind, /*isSigned=*/true)};
3834   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args);
3835 }
3836 
3837 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
3838                                 SourceLocation Loc) {
3839   if (!CGF.HaveInsertPoint())
3840     return;
3841   // Build call void __kmpc_flush(ident_t *loc)
3842   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush),
3843                       emitUpdateLocation(CGF, Loc));
3844 }
3845 
3846 namespace {
3847 /// Indexes of fields for type kmp_task_t.
3848 enum KmpTaskTFields {
3849   /// List of shared variables.
3850   KmpTaskTShareds,
3851   /// Task routine.
3852   KmpTaskTRoutine,
3853   /// Partition id for the untied tasks.
3854   KmpTaskTPartId,
3855   /// Function with call of destructors for private variables.
3856   Data1,
3857   /// Task priority.
3858   Data2,
3859   /// (Taskloops only) Lower bound.
3860   KmpTaskTLowerBound,
3861   /// (Taskloops only) Upper bound.
3862   KmpTaskTUpperBound,
3863   /// (Taskloops only) Stride.
3864   KmpTaskTStride,
3865   /// (Taskloops only) Is last iteration flag.
3866   KmpTaskTLastIter,
3867   /// (Taskloops only) Reduction data.
3868   KmpTaskTReductions,
3869 };
3870 } // anonymous namespace
3871 
3872 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const {
3873   return OffloadEntriesTargetRegion.empty() &&
3874          OffloadEntriesDeviceGlobalVar.empty();
3875 }
3876 
3877 /// Initialize target region entry.
3878 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3879     initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3880                                     StringRef ParentName, unsigned LineNum,
3881                                     unsigned Order) {
3882   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3883                                              "only required for the device "
3884                                              "code generation.");
3885   OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] =
3886       OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr,
3887                                    OMPTargetRegionEntryTargetRegion);
3888   ++OffloadingEntriesNum;
3889 }
3890 
3891 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3892     registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3893                                   StringRef ParentName, unsigned LineNum,
3894                                   llvm::Constant *Addr, llvm::Constant *ID,
3895                                   OMPTargetRegionEntryKind Flags) {
3896   // If we are emitting code for a target, the entry is already initialized,
3897   // only has to be registered.
3898   if (CGM.getLangOpts().OpenMPIsDevice) {
3899     if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) {
3900       unsigned DiagID = CGM.getDiags().getCustomDiagID(
3901           DiagnosticsEngine::Error,
3902           "Unable to find target region on line '%0' in the device code.");
3903       CGM.getDiags().Report(DiagID) << LineNum;
3904       return;
3905     }
3906     auto &Entry =
3907         OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum];
3908     assert(Entry.isValid() && "Entry not initialized!");
3909     Entry.setAddress(Addr);
3910     Entry.setID(ID);
3911     Entry.setFlags(Flags);
3912   } else {
3913     OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags);
3914     OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry;
3915     ++OffloadingEntriesNum;
3916   }
3917 }
3918 
3919 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo(
3920     unsigned DeviceID, unsigned FileID, StringRef ParentName,
3921     unsigned LineNum) const {
3922   auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID);
3923   if (PerDevice == OffloadEntriesTargetRegion.end())
3924     return false;
3925   auto PerFile = PerDevice->second.find(FileID);
3926   if (PerFile == PerDevice->second.end())
3927     return false;
3928   auto PerParentName = PerFile->second.find(ParentName);
3929   if (PerParentName == PerFile->second.end())
3930     return false;
3931   auto PerLine = PerParentName->second.find(LineNum);
3932   if (PerLine == PerParentName->second.end())
3933     return false;
3934   // Fail if this entry is already registered.
3935   if (PerLine->second.getAddress() || PerLine->second.getID())
3936     return false;
3937   return true;
3938 }
3939 
3940 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo(
3941     const OffloadTargetRegionEntryInfoActTy &Action) {
3942   // Scan all target region entries and perform the provided action.
3943   for (const auto &D : OffloadEntriesTargetRegion)
3944     for (const auto &F : D.second)
3945       for (const auto &P : F.second)
3946         for (const auto &L : P.second)
3947           Action(D.first, F.first, P.first(), L.first, L.second);
3948 }
3949 
3950 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3951     initializeDeviceGlobalVarEntryInfo(StringRef Name,
3952                                        OMPTargetGlobalVarEntryKind Flags,
3953                                        unsigned Order) {
3954   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3955                                              "only required for the device "
3956                                              "code generation.");
3957   OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
3958   ++OffloadingEntriesNum;
3959 }
3960 
3961 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3962     registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr,
3963                                      CharUnits VarSize,
3964                                      OMPTargetGlobalVarEntryKind Flags,
3965                                      llvm::GlobalValue::LinkageTypes Linkage) {
3966   if (CGM.getLangOpts().OpenMPIsDevice) {
3967     auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
3968     assert(Entry.isValid() && Entry.getFlags() == Flags &&
3969            "Entry not initialized!");
3970     assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
3971            "Resetting with the new address.");
3972     if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) {
3973       if (Entry.getVarSize().isZero()) {
3974         Entry.setVarSize(VarSize);
3975         Entry.setLinkage(Linkage);
3976       }
3977       return;
3978     }
3979     Entry.setVarSize(VarSize);
3980     Entry.setLinkage(Linkage);
3981     Entry.setAddress(Addr);
3982   } else {
3983     if (hasDeviceGlobalVarEntryInfo(VarName)) {
3984       auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
3985       assert(Entry.isValid() && Entry.getFlags() == Flags &&
3986              "Entry not initialized!");
3987       assert((!Entry.getAddress() || Entry.getAddress() == Addr) &&
3988              "Resetting with the new address.");
3989       if (Entry.getVarSize().isZero()) {
3990         Entry.setVarSize(VarSize);
3991         Entry.setLinkage(Linkage);
3992       }
3993       return;
3994     }
3995     OffloadEntriesDeviceGlobalVar.try_emplace(
3996         VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage);
3997     ++OffloadingEntriesNum;
3998   }
3999 }
4000 
4001 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
4002     actOnDeviceGlobalVarEntriesInfo(
4003         const OffloadDeviceGlobalVarEntryInfoActTy &Action) {
4004   // Scan all target region entries and perform the provided action.
4005   for (const auto &E : OffloadEntriesDeviceGlobalVar)
4006     Action(E.getKey(), E.getValue());
4007 }
4008 
4009 llvm::Function *
4010 CGOpenMPRuntime::createOffloadingBinaryDescriptorRegistration() {
4011   // If we don't have entries or if we are emitting code for the device, we
4012   // don't need to do anything.
4013   if (CGM.getLangOpts().OpenMPIsDevice || OffloadEntriesInfoManager.empty())
4014     return nullptr;
4015 
4016   llvm::Module &M = CGM.getModule();
4017   ASTContext &C = CGM.getContext();
4018 
4019   // Get list of devices we care about
4020   const std::vector<llvm::Triple> &Devices = CGM.getLangOpts().OMPTargetTriples;
4021 
4022   // We should be creating an offloading descriptor only if there are devices
4023   // specified.
4024   assert(!Devices.empty() && "No OpenMP offloading devices??");
4025 
4026   // Create the external variables that will point to the begin and end of the
4027   // host entries section. These will be defined by the linker.
4028   llvm::Type *OffloadEntryTy =
4029       CGM.getTypes().ConvertTypeForMem(getTgtOffloadEntryQTy());
4030   auto *HostEntriesBegin = new llvm::GlobalVariable(
4031       M, OffloadEntryTy, /*isConstant=*/true,
4032       llvm::GlobalValue::ExternalLinkage, /*Initializer=*/nullptr,
4033       "__start_omp_offloading_entries");
4034   HostEntriesBegin->setVisibility(llvm::GlobalValue::HiddenVisibility);
4035   auto *HostEntriesEnd = new llvm::GlobalVariable(
4036       M, OffloadEntryTy, /*isConstant=*/true,
4037       llvm::GlobalValue::ExternalLinkage,
4038       /*Initializer=*/nullptr, "__stop_omp_offloading_entries");
4039   HostEntriesEnd->setVisibility(llvm::GlobalValue::HiddenVisibility);
4040 
4041   // Create all device images
4042   auto *DeviceImageTy = cast<llvm::StructType>(
4043       CGM.getTypes().ConvertTypeForMem(getTgtDeviceImageQTy()));
4044   ConstantInitBuilder DeviceImagesBuilder(CGM);
4045   ConstantArrayBuilder DeviceImagesEntries =
4046       DeviceImagesBuilder.beginArray(DeviceImageTy);
4047 
4048   for (const llvm::Triple &Device : Devices) {
4049     StringRef T = Device.getTriple();
4050     std::string BeginName = getName({"omp_offloading", "img_start", ""});
4051     auto *ImgBegin = new llvm::GlobalVariable(
4052         M, CGM.Int8Ty, /*isConstant=*/true,
4053         llvm::GlobalValue::ExternalWeakLinkage,
4054         /*Initializer=*/nullptr, Twine(BeginName).concat(T));
4055     std::string EndName = getName({"omp_offloading", "img_end", ""});
4056     auto *ImgEnd = new llvm::GlobalVariable(
4057         M, CGM.Int8Ty, /*isConstant=*/true,
4058         llvm::GlobalValue::ExternalWeakLinkage,
4059         /*Initializer=*/nullptr, Twine(EndName).concat(T));
4060 
4061     llvm::Constant *Data[] = {ImgBegin, ImgEnd, HostEntriesBegin,
4062                               HostEntriesEnd};
4063     createConstantGlobalStructAndAddToParent(CGM, getTgtDeviceImageQTy(), Data,
4064                                              DeviceImagesEntries);
4065   }
4066 
4067   // Create device images global array.
4068   std::string ImagesName = getName({"omp_offloading", "device_images"});
4069   llvm::GlobalVariable *DeviceImages =
4070       DeviceImagesEntries.finishAndCreateGlobal(ImagesName,
4071                                                 CGM.getPointerAlign(),
4072                                                 /*isConstant=*/true);
4073   DeviceImages->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
4074 
4075   // This is a Zero array to be used in the creation of the constant expressions
4076   llvm::Constant *Index[] = {llvm::Constant::getNullValue(CGM.Int32Ty),
4077                              llvm::Constant::getNullValue(CGM.Int32Ty)};
4078 
4079   // Create the target region descriptor.
4080   llvm::Constant *Data[] = {
4081       llvm::ConstantInt::get(CGM.Int32Ty, Devices.size()),
4082       llvm::ConstantExpr::getGetElementPtr(DeviceImages->getValueType(),
4083                                            DeviceImages, Index),
4084       HostEntriesBegin, HostEntriesEnd};
4085   std::string Descriptor = getName({"omp_offloading", "descriptor"});
4086   llvm::GlobalVariable *Desc = createGlobalStruct(
4087       CGM, getTgtBinaryDescriptorQTy(), /*IsConstant=*/true, Data, Descriptor);
4088 
4089   // Emit code to register or unregister the descriptor at execution
4090   // startup or closing, respectively.
4091 
4092   llvm::Function *UnRegFn;
4093   {
4094     FunctionArgList Args;
4095     ImplicitParamDecl DummyPtr(C, C.VoidPtrTy, ImplicitParamDecl::Other);
4096     Args.push_back(&DummyPtr);
4097 
4098     CodeGenFunction CGF(CGM);
4099     // Disable debug info for global (de-)initializer because they are not part
4100     // of some particular construct.
4101     CGF.disableDebugInfo();
4102     const auto &FI =
4103         CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4104     llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
4105     std::string UnregName = getName({"omp_offloading", "descriptor_unreg"});
4106     UnRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, UnregName, FI);
4107     CGF.StartFunction(GlobalDecl(), C.VoidTy, UnRegFn, FI, Args);
4108     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_unregister_lib),
4109                         Desc);
4110     CGF.FinishFunction();
4111   }
4112   llvm::Function *RegFn;
4113   {
4114     CodeGenFunction CGF(CGM);
4115     // Disable debug info for global (de-)initializer because they are not part
4116     // of some particular construct.
4117     CGF.disableDebugInfo();
4118     const auto &FI = CGM.getTypes().arrangeNullaryFunction();
4119     llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
4120 
4121     // Encode offload target triples into the registration function name. It
4122     // will serve as a comdat key for the registration/unregistration code for
4123     // this particular combination of offloading targets.
4124     SmallVector<StringRef, 4U> RegFnNameParts(Devices.size() + 2U);
4125     RegFnNameParts[0] = "omp_offloading";
4126     RegFnNameParts[1] = "descriptor_reg";
4127     llvm::transform(Devices, std::next(RegFnNameParts.begin(), 2),
4128                     [](const llvm::Triple &T) -> const std::string& {
4129                       return T.getTriple();
4130                     });
4131     llvm::sort(std::next(RegFnNameParts.begin(), 2), RegFnNameParts.end());
4132     std::string Descriptor = getName(RegFnNameParts);
4133     RegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, Descriptor, FI);
4134     CGF.StartFunction(GlobalDecl(), C.VoidTy, RegFn, FI, FunctionArgList());
4135     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_lib), Desc);
4136     // Create a variable to drive the registration and unregistration of the
4137     // descriptor, so we can reuse the logic that emits Ctors and Dtors.
4138     ImplicitParamDecl RegUnregVar(C, C.getTranslationUnitDecl(),
4139                                   SourceLocation(), nullptr, C.CharTy,
4140                                   ImplicitParamDecl::Other);
4141     CGM.getCXXABI().registerGlobalDtor(CGF, RegUnregVar, UnRegFn, Desc);
4142     CGF.FinishFunction();
4143   }
4144   if (CGM.supportsCOMDAT()) {
4145     // It is sufficient to call registration function only once, so create a
4146     // COMDAT group for registration/unregistration functions and associated
4147     // data. That would reduce startup time and code size. Registration
4148     // function serves as a COMDAT group key.
4149     llvm::Comdat *ComdatKey = M.getOrInsertComdat(RegFn->getName());
4150     RegFn->setLinkage(llvm::GlobalValue::LinkOnceAnyLinkage);
4151     RegFn->setVisibility(llvm::GlobalValue::HiddenVisibility);
4152     RegFn->setComdat(ComdatKey);
4153     UnRegFn->setComdat(ComdatKey);
4154     DeviceImages->setComdat(ComdatKey);
4155     Desc->setComdat(ComdatKey);
4156   }
4157   return RegFn;
4158 }
4159 
4160 void CGOpenMPRuntime::createOffloadEntry(
4161     llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags,
4162     llvm::GlobalValue::LinkageTypes Linkage) {
4163   StringRef Name = Addr->getName();
4164   llvm::Module &M = CGM.getModule();
4165   llvm::LLVMContext &C = M.getContext();
4166 
4167   // Create constant string with the name.
4168   llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name);
4169 
4170   std::string StringName = getName({"omp_offloading", "entry_name"});
4171   auto *Str = new llvm::GlobalVariable(
4172       M, StrPtrInit->getType(), /*isConstant=*/true,
4173       llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName);
4174   Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
4175 
4176   llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy),
4177                             llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy),
4178                             llvm::ConstantInt::get(CGM.SizeTy, Size),
4179                             llvm::ConstantInt::get(CGM.Int32Ty, Flags),
4180                             llvm::ConstantInt::get(CGM.Int32Ty, 0)};
4181   std::string EntryName = getName({"omp_offloading", "entry", ""});
4182   llvm::GlobalVariable *Entry = createGlobalStruct(
4183       CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data,
4184       Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage);
4185 
4186   // The entry has to be created in the section the linker expects it to be.
4187   Entry->setSection("omp_offloading_entries");
4188 }
4189 
4190 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() {
4191   // Emit the offloading entries and metadata so that the device codegen side
4192   // can easily figure out what to emit. The produced metadata looks like
4193   // this:
4194   //
4195   // !omp_offload.info = !{!1, ...}
4196   //
4197   // Right now we only generate metadata for function that contain target
4198   // regions.
4199 
4200   // If we do not have entries, we don't need to do anything.
4201   if (OffloadEntriesInfoManager.empty())
4202     return;
4203 
4204   llvm::Module &M = CGM.getModule();
4205   llvm::LLVMContext &C = M.getContext();
4206   SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *,
4207                          SourceLocation, StringRef>,
4208               16>
4209       OrderedEntries(OffloadEntriesInfoManager.size());
4210   llvm::SmallVector<StringRef, 16> ParentFunctions(
4211       OffloadEntriesInfoManager.size());
4212 
4213   // Auxiliary methods to create metadata values and strings.
4214   auto &&GetMDInt = [this](unsigned V) {
4215     return llvm::ConstantAsMetadata::get(
4216         llvm::ConstantInt::get(CGM.Int32Ty, V));
4217   };
4218 
4219   auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); };
4220 
4221   // Create the offloading info metadata node.
4222   llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info");
4223 
4224   // Create function that emits metadata for each target region entry;
4225   auto &&TargetRegionMetadataEmitter =
4226       [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt,
4227        &GetMDString](
4228           unsigned DeviceID, unsigned FileID, StringRef ParentName,
4229           unsigned Line,
4230           const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) {
4231         // Generate metadata for target regions. Each entry of this metadata
4232         // contains:
4233         // - Entry 0 -> Kind of this type of metadata (0).
4234         // - Entry 1 -> Device ID of the file where the entry was identified.
4235         // - Entry 2 -> File ID of the file where the entry was identified.
4236         // - Entry 3 -> Mangled name of the function where the entry was
4237         // identified.
4238         // - Entry 4 -> Line in the file where the entry was identified.
4239         // - Entry 5 -> Order the entry was created.
4240         // The first element of the metadata node is the kind.
4241         llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID),
4242                                  GetMDInt(FileID),      GetMDString(ParentName),
4243                                  GetMDInt(Line),        GetMDInt(E.getOrder())};
4244 
4245         SourceLocation Loc;
4246         for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(),
4247                   E = CGM.getContext().getSourceManager().fileinfo_end();
4248              I != E; ++I) {
4249           if (I->getFirst()->getUniqueID().getDevice() == DeviceID &&
4250               I->getFirst()->getUniqueID().getFile() == FileID) {
4251             Loc = CGM.getContext().getSourceManager().translateFileLineCol(
4252                 I->getFirst(), Line, 1);
4253             break;
4254           }
4255         }
4256         // Save this entry in the right position of the ordered entries array.
4257         OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName);
4258         ParentFunctions[E.getOrder()] = ParentName;
4259 
4260         // Add metadata to the named metadata node.
4261         MD->addOperand(llvm::MDNode::get(C, Ops));
4262       };
4263 
4264   OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo(
4265       TargetRegionMetadataEmitter);
4266 
4267   // Create function that emits metadata for each device global variable entry;
4268   auto &&DeviceGlobalVarMetadataEmitter =
4269       [&C, &OrderedEntries, &GetMDInt, &GetMDString,
4270        MD](StringRef MangledName,
4271            const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar
4272                &E) {
4273         // Generate metadata for global variables. Each entry of this metadata
4274         // contains:
4275         // - Entry 0 -> Kind of this type of metadata (1).
4276         // - Entry 1 -> Mangled name of the variable.
4277         // - Entry 2 -> Declare target kind.
4278         // - Entry 3 -> Order the entry was created.
4279         // The first element of the metadata node is the kind.
4280         llvm::Metadata *Ops[] = {
4281             GetMDInt(E.getKind()), GetMDString(MangledName),
4282             GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
4283 
4284         // Save this entry in the right position of the ordered entries array.
4285         OrderedEntries[E.getOrder()] =
4286             std::make_tuple(&E, SourceLocation(), MangledName);
4287 
4288         // Add metadata to the named metadata node.
4289         MD->addOperand(llvm::MDNode::get(C, Ops));
4290       };
4291 
4292   OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo(
4293       DeviceGlobalVarMetadataEmitter);
4294 
4295   for (const auto &E : OrderedEntries) {
4296     assert(std::get<0>(E) && "All ordered entries must exist!");
4297     if (const auto *CE =
4298             dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>(
4299                 std::get<0>(E))) {
4300       if (!CE->getID() || !CE->getAddress()) {
4301         // Do not blame the entry if the parent funtion is not emitted.
4302         StringRef FnName = ParentFunctions[CE->getOrder()];
4303         if (!CGM.GetGlobalValue(FnName))
4304           continue;
4305         unsigned DiagID = CGM.getDiags().getCustomDiagID(
4306             DiagnosticsEngine::Error,
4307             "Offloading entry for target region in %0 is incorrect: either the "
4308             "address or the ID is invalid.");
4309         CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName;
4310         continue;
4311       }
4312       createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0,
4313                          CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage);
4314     } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy::
4315                                              OffloadEntryInfoDeviceGlobalVar>(
4316                    std::get<0>(E))) {
4317       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags =
4318           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4319               CE->getFlags());
4320       switch (Flags) {
4321       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: {
4322         if (CGM.getLangOpts().OpenMPIsDevice &&
4323             CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())
4324           continue;
4325         if (!CE->getAddress()) {
4326           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4327               DiagnosticsEngine::Error, "Offloading entry for declare target "
4328                                         "variable %0 is incorrect: the "
4329                                         "address is invalid.");
4330           CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E);
4331           continue;
4332         }
4333         // The vaiable has no definition - no need to add the entry.
4334         if (CE->getVarSize().isZero())
4335           continue;
4336         break;
4337       }
4338       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink:
4339         assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) ||
4340                 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) &&
4341                "Declaret target link address is set.");
4342         if (CGM.getLangOpts().OpenMPIsDevice)
4343           continue;
4344         if (!CE->getAddress()) {
4345           unsigned DiagID = CGM.getDiags().getCustomDiagID(
4346               DiagnosticsEngine::Error,
4347               "Offloading entry for declare target variable is incorrect: the "
4348               "address is invalid.");
4349           CGM.getDiags().Report(DiagID);
4350           continue;
4351         }
4352         break;
4353       }
4354       createOffloadEntry(CE->getAddress(), CE->getAddress(),
4355                          CE->getVarSize().getQuantity(), Flags,
4356                          CE->getLinkage());
4357     } else {
4358       llvm_unreachable("Unsupported entry kind.");
4359     }
4360   }
4361 }
4362 
4363 /// Loads all the offload entries information from the host IR
4364 /// metadata.
4365 void CGOpenMPRuntime::loadOffloadInfoMetadata() {
4366   // If we are in target mode, load the metadata from the host IR. This code has
4367   // to match the metadaata creation in createOffloadEntriesAndInfoMetadata().
4368 
4369   if (!CGM.getLangOpts().OpenMPIsDevice)
4370     return;
4371 
4372   if (CGM.getLangOpts().OMPHostIRFile.empty())
4373     return;
4374 
4375   auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile);
4376   if (auto EC = Buf.getError()) {
4377     CGM.getDiags().Report(diag::err_cannot_open_file)
4378         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4379     return;
4380   }
4381 
4382   llvm::LLVMContext C;
4383   auto ME = expectedToErrorOrAndEmitErrors(
4384       C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C));
4385 
4386   if (auto EC = ME.getError()) {
4387     unsigned DiagID = CGM.getDiags().getCustomDiagID(
4388         DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'");
4389     CGM.getDiags().Report(DiagID)
4390         << CGM.getLangOpts().OMPHostIRFile << EC.message();
4391     return;
4392   }
4393 
4394   llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info");
4395   if (!MD)
4396     return;
4397 
4398   for (llvm::MDNode *MN : MD->operands()) {
4399     auto &&GetMDInt = [MN](unsigned Idx) {
4400       auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx));
4401       return cast<llvm::ConstantInt>(V->getValue())->getZExtValue();
4402     };
4403 
4404     auto &&GetMDString = [MN](unsigned Idx) {
4405       auto *V = cast<llvm::MDString>(MN->getOperand(Idx));
4406       return V->getString();
4407     };
4408 
4409     switch (GetMDInt(0)) {
4410     default:
4411       llvm_unreachable("Unexpected metadata!");
4412       break;
4413     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4414         OffloadingEntryInfoTargetRegion:
4415       OffloadEntriesInfoManager.initializeTargetRegionEntryInfo(
4416           /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2),
4417           /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4),
4418           /*Order=*/GetMDInt(5));
4419       break;
4420     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
4421         OffloadingEntryInfoDeviceGlobalVar:
4422       OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo(
4423           /*MangledName=*/GetMDString(1),
4424           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
4425               /*Flags=*/GetMDInt(2)),
4426           /*Order=*/GetMDInt(3));
4427       break;
4428     }
4429   }
4430 }
4431 
4432 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
4433   if (!KmpRoutineEntryPtrTy) {
4434     // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
4435     ASTContext &C = CGM.getContext();
4436     QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
4437     FunctionProtoType::ExtProtoInfo EPI;
4438     KmpRoutineEntryPtrQTy = C.getPointerType(
4439         C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
4440     KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
4441   }
4442 }
4443 
4444 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() {
4445   // Make sure the type of the entry is already created. This is the type we
4446   // have to create:
4447   // struct __tgt_offload_entry{
4448   //   void      *addr;       // Pointer to the offload entry info.
4449   //                          // (function or global)
4450   //   char      *name;       // Name of the function or global.
4451   //   size_t     size;       // Size of the entry info (0 if it a function).
4452   //   int32_t    flags;      // Flags associated with the entry, e.g. 'link'.
4453   //   int32_t    reserved;   // Reserved, to use by the runtime library.
4454   // };
4455   if (TgtOffloadEntryQTy.isNull()) {
4456     ASTContext &C = CGM.getContext();
4457     RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry");
4458     RD->startDefinition();
4459     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4460     addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy));
4461     addFieldToRecordDecl(C, RD, C.getSizeType());
4462     addFieldToRecordDecl(
4463         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4464     addFieldToRecordDecl(
4465         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4466     RD->completeDefinition();
4467     RD->addAttr(PackedAttr::CreateImplicit(C));
4468     TgtOffloadEntryQTy = C.getRecordType(RD);
4469   }
4470   return TgtOffloadEntryQTy;
4471 }
4472 
4473 QualType CGOpenMPRuntime::getTgtDeviceImageQTy() {
4474   // These are the types we need to build:
4475   // struct __tgt_device_image{
4476   // void   *ImageStart;       // Pointer to the target code start.
4477   // void   *ImageEnd;         // Pointer to the target code end.
4478   // // We also add the host entries to the device image, as it may be useful
4479   // // for the target runtime to have access to that information.
4480   // __tgt_offload_entry  *EntriesBegin;   // Begin of the table with all
4481   //                                       // the entries.
4482   // __tgt_offload_entry  *EntriesEnd;     // End of the table with all the
4483   //                                       // entries (non inclusive).
4484   // };
4485   if (TgtDeviceImageQTy.isNull()) {
4486     ASTContext &C = CGM.getContext();
4487     RecordDecl *RD = C.buildImplicitRecord("__tgt_device_image");
4488     RD->startDefinition();
4489     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4490     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4491     addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy()));
4492     addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy()));
4493     RD->completeDefinition();
4494     TgtDeviceImageQTy = C.getRecordType(RD);
4495   }
4496   return TgtDeviceImageQTy;
4497 }
4498 
4499 QualType CGOpenMPRuntime::getTgtBinaryDescriptorQTy() {
4500   // struct __tgt_bin_desc{
4501   //   int32_t              NumDevices;      // Number of devices supported.
4502   //   __tgt_device_image   *DeviceImages;   // Arrays of device images
4503   //                                         // (one per device).
4504   //   __tgt_offload_entry  *EntriesBegin;   // Begin of the table with all the
4505   //                                         // entries.
4506   //   __tgt_offload_entry  *EntriesEnd;     // End of the table with all the
4507   //                                         // entries (non inclusive).
4508   // };
4509   if (TgtBinaryDescriptorQTy.isNull()) {
4510     ASTContext &C = CGM.getContext();
4511     RecordDecl *RD = C.buildImplicitRecord("__tgt_bin_desc");
4512     RD->startDefinition();
4513     addFieldToRecordDecl(
4514         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
4515     addFieldToRecordDecl(C, RD, C.getPointerType(getTgtDeviceImageQTy()));
4516     addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy()));
4517     addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy()));
4518     RD->completeDefinition();
4519     TgtBinaryDescriptorQTy = C.getRecordType(RD);
4520   }
4521   return TgtBinaryDescriptorQTy;
4522 }
4523 
4524 namespace {
4525 struct PrivateHelpersTy {
4526   PrivateHelpersTy(const VarDecl *Original, const VarDecl *PrivateCopy,
4527                    const VarDecl *PrivateElemInit)
4528       : Original(Original), PrivateCopy(PrivateCopy),
4529         PrivateElemInit(PrivateElemInit) {}
4530   const VarDecl *Original;
4531   const VarDecl *PrivateCopy;
4532   const VarDecl *PrivateElemInit;
4533 };
4534 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
4535 } // anonymous namespace
4536 
4537 static RecordDecl *
4538 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) {
4539   if (!Privates.empty()) {
4540     ASTContext &C = CGM.getContext();
4541     // Build struct .kmp_privates_t. {
4542     //         /*  private vars  */
4543     //       };
4544     RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t");
4545     RD->startDefinition();
4546     for (const auto &Pair : Privates) {
4547       const VarDecl *VD = Pair.second.Original;
4548       QualType Type = VD->getType().getNonReferenceType();
4549       FieldDecl *FD = addFieldToRecordDecl(C, RD, Type);
4550       if (VD->hasAttrs()) {
4551         for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()),
4552              E(VD->getAttrs().end());
4553              I != E; ++I)
4554           FD->addAttr(*I);
4555       }
4556     }
4557     RD->completeDefinition();
4558     return RD;
4559   }
4560   return nullptr;
4561 }
4562 
4563 static RecordDecl *
4564 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind,
4565                          QualType KmpInt32Ty,
4566                          QualType KmpRoutineEntryPointerQTy) {
4567   ASTContext &C = CGM.getContext();
4568   // Build struct kmp_task_t {
4569   //         void *              shareds;
4570   //         kmp_routine_entry_t routine;
4571   //         kmp_int32           part_id;
4572   //         kmp_cmplrdata_t data1;
4573   //         kmp_cmplrdata_t data2;
4574   // For taskloops additional fields:
4575   //         kmp_uint64          lb;
4576   //         kmp_uint64          ub;
4577   //         kmp_int64           st;
4578   //         kmp_int32           liter;
4579   //         void *              reductions;
4580   //       };
4581   RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union);
4582   UD->startDefinition();
4583   addFieldToRecordDecl(C, UD, KmpInt32Ty);
4584   addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy);
4585   UD->completeDefinition();
4586   QualType KmpCmplrdataTy = C.getRecordType(UD);
4587   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t");
4588   RD->startDefinition();
4589   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4590   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
4591   addFieldToRecordDecl(C, RD, KmpInt32Ty);
4592   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4593   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
4594   if (isOpenMPTaskLoopDirective(Kind)) {
4595     QualType KmpUInt64Ty =
4596         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
4597     QualType KmpInt64Ty =
4598         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
4599     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4600     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
4601     addFieldToRecordDecl(C, RD, KmpInt64Ty);
4602     addFieldToRecordDecl(C, RD, KmpInt32Ty);
4603     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
4604   }
4605   RD->completeDefinition();
4606   return RD;
4607 }
4608 
4609 static RecordDecl *
4610 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy,
4611                                      ArrayRef<PrivateDataTy> Privates) {
4612   ASTContext &C = CGM.getContext();
4613   // Build struct kmp_task_t_with_privates {
4614   //         kmp_task_t task_data;
4615   //         .kmp_privates_t. privates;
4616   //       };
4617   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates");
4618   RD->startDefinition();
4619   addFieldToRecordDecl(C, RD, KmpTaskTQTy);
4620   if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates))
4621     addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD));
4622   RD->completeDefinition();
4623   return RD;
4624 }
4625 
4626 /// Emit a proxy function which accepts kmp_task_t as the second
4627 /// argument.
4628 /// \code
4629 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
4630 ///   TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt,
4631 ///   For taskloops:
4632 ///   tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4633 ///   tt->reductions, tt->shareds);
4634 ///   return 0;
4635 /// }
4636 /// \endcode
4637 static llvm::Function *
4638 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
4639                       OpenMPDirectiveKind Kind, QualType KmpInt32Ty,
4640                       QualType KmpTaskTWithPrivatesPtrQTy,
4641                       QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
4642                       QualType SharedsPtrTy, llvm::Function *TaskFunction,
4643                       llvm::Value *TaskPrivatesMap) {
4644   ASTContext &C = CGM.getContext();
4645   FunctionArgList Args;
4646   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4647                             ImplicitParamDecl::Other);
4648   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4649                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4650                                 ImplicitParamDecl::Other);
4651   Args.push_back(&GtidArg);
4652   Args.push_back(&TaskTypeArg);
4653   const auto &TaskEntryFnInfo =
4654       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4655   llvm::FunctionType *TaskEntryTy =
4656       CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
4657   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""});
4658   auto *TaskEntry = llvm::Function::Create(
4659       TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4660   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo);
4661   TaskEntry->setDoesNotRecurse();
4662   CodeGenFunction CGF(CGM);
4663   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args,
4664                     Loc, Loc);
4665 
4666   // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map,
4667   // tt,
4668   // For taskloops:
4669   // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
4670   // tt->task_data.shareds);
4671   llvm::Value *GtidParam = CGF.EmitLoadOfScalar(
4672       CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc);
4673   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4674       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4675       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4676   const auto *KmpTaskTWithPrivatesQTyRD =
4677       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4678   LValue Base =
4679       CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4680   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
4681   auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
4682   LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI);
4683   llvm::Value *PartidParam = PartIdLVal.getPointer();
4684 
4685   auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds);
4686   LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI);
4687   llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4688       CGF.EmitLoadOfScalar(SharedsLVal, Loc),
4689       CGF.ConvertTypeForMem(SharedsPtrTy));
4690 
4691   auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
4692   llvm::Value *PrivatesParam;
4693   if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) {
4694     LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI);
4695     PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4696         PrivatesLVal.getPointer(), CGF.VoidPtrTy);
4697   } else {
4698     PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
4699   }
4700 
4701   llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam,
4702                                TaskPrivatesMap,
4703                                CGF.Builder
4704                                    .CreatePointerBitCastOrAddrSpaceCast(
4705                                        TDBase.getAddress(), CGF.VoidPtrTy)
4706                                    .getPointer()};
4707   SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs),
4708                                           std::end(CommonArgs));
4709   if (isOpenMPTaskLoopDirective(Kind)) {
4710     auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound);
4711     LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI);
4712     llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc);
4713     auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound);
4714     LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI);
4715     llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc);
4716     auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride);
4717     LValue StLVal = CGF.EmitLValueForField(Base, *StFI);
4718     llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc);
4719     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4720     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4721     llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc);
4722     auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions);
4723     LValue RLVal = CGF.EmitLValueForField(Base, *RFI);
4724     llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc);
4725     CallArgs.push_back(LBParam);
4726     CallArgs.push_back(UBParam);
4727     CallArgs.push_back(StParam);
4728     CallArgs.push_back(LIParam);
4729     CallArgs.push_back(RParam);
4730   }
4731   CallArgs.push_back(SharedsParam);
4732 
4733   CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction,
4734                                                   CallArgs);
4735   CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)),
4736                              CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty));
4737   CGF.FinishFunction();
4738   return TaskEntry;
4739 }
4740 
4741 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM,
4742                                             SourceLocation Loc,
4743                                             QualType KmpInt32Ty,
4744                                             QualType KmpTaskTWithPrivatesPtrQTy,
4745                                             QualType KmpTaskTWithPrivatesQTy) {
4746   ASTContext &C = CGM.getContext();
4747   FunctionArgList Args;
4748   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
4749                             ImplicitParamDecl::Other);
4750   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4751                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
4752                                 ImplicitParamDecl::Other);
4753   Args.push_back(&GtidArg);
4754   Args.push_back(&TaskTypeArg);
4755   const auto &DestructorFnInfo =
4756       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
4757   llvm::FunctionType *DestructorFnTy =
4758       CGM.getTypes().GetFunctionType(DestructorFnInfo);
4759   std::string Name =
4760       CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""});
4761   auto *DestructorFn =
4762       llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage,
4763                              Name, &CGM.getModule());
4764   CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn,
4765                                     DestructorFnInfo);
4766   DestructorFn->setDoesNotRecurse();
4767   CodeGenFunction CGF(CGM);
4768   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo,
4769                     Args, Loc, Loc);
4770 
4771   LValue Base = CGF.EmitLoadOfPointerLValue(
4772       CGF.GetAddrOfLocalVar(&TaskTypeArg),
4773       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4774   const auto *KmpTaskTWithPrivatesQTyRD =
4775       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
4776   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4777   Base = CGF.EmitLValueForField(Base, *FI);
4778   for (const auto *Field :
4779        cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) {
4780     if (QualType::DestructionKind DtorKind =
4781             Field->getType().isDestructedType()) {
4782       LValue FieldLValue = CGF.EmitLValueForField(Base, Field);
4783       CGF.pushDestroy(DtorKind, FieldLValue.getAddress(), Field->getType());
4784     }
4785   }
4786   CGF.FinishFunction();
4787   return DestructorFn;
4788 }
4789 
4790 /// Emit a privates mapping function for correct handling of private and
4791 /// firstprivate variables.
4792 /// \code
4793 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1>
4794 /// **noalias priv1,...,  <tyn> **noalias privn) {
4795 ///   *priv1 = &.privates.priv1;
4796 ///   ...;
4797 ///   *privn = &.privates.privn;
4798 /// }
4799 /// \endcode
4800 static llvm::Value *
4801 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc,
4802                                ArrayRef<const Expr *> PrivateVars,
4803                                ArrayRef<const Expr *> FirstprivateVars,
4804                                ArrayRef<const Expr *> LastprivateVars,
4805                                QualType PrivatesQTy,
4806                                ArrayRef<PrivateDataTy> Privates) {
4807   ASTContext &C = CGM.getContext();
4808   FunctionArgList Args;
4809   ImplicitParamDecl TaskPrivatesArg(
4810       C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4811       C.getPointerType(PrivatesQTy).withConst().withRestrict(),
4812       ImplicitParamDecl::Other);
4813   Args.push_back(&TaskPrivatesArg);
4814   llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos;
4815   unsigned Counter = 1;
4816   for (const Expr *E : PrivateVars) {
4817     Args.push_back(ImplicitParamDecl::Create(
4818         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4819         C.getPointerType(C.getPointerType(E->getType()))
4820             .withConst()
4821             .withRestrict(),
4822         ImplicitParamDecl::Other));
4823     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4824     PrivateVarsPos[VD] = Counter;
4825     ++Counter;
4826   }
4827   for (const Expr *E : FirstprivateVars) {
4828     Args.push_back(ImplicitParamDecl::Create(
4829         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4830         C.getPointerType(C.getPointerType(E->getType()))
4831             .withConst()
4832             .withRestrict(),
4833         ImplicitParamDecl::Other));
4834     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4835     PrivateVarsPos[VD] = Counter;
4836     ++Counter;
4837   }
4838   for (const Expr *E : LastprivateVars) {
4839     Args.push_back(ImplicitParamDecl::Create(
4840         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4841         C.getPointerType(C.getPointerType(E->getType()))
4842             .withConst()
4843             .withRestrict(),
4844         ImplicitParamDecl::Other));
4845     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4846     PrivateVarsPos[VD] = Counter;
4847     ++Counter;
4848   }
4849   const auto &TaskPrivatesMapFnInfo =
4850       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4851   llvm::FunctionType *TaskPrivatesMapTy =
4852       CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo);
4853   std::string Name =
4854       CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""});
4855   auto *TaskPrivatesMap = llvm::Function::Create(
4856       TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name,
4857       &CGM.getModule());
4858   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap,
4859                                     TaskPrivatesMapFnInfo);
4860   if (CGM.getLangOpts().Optimize) {
4861     TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline);
4862     TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone);
4863     TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline);
4864   }
4865   CodeGenFunction CGF(CGM);
4866   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap,
4867                     TaskPrivatesMapFnInfo, Args, Loc, Loc);
4868 
4869   // *privi = &.privates.privi;
4870   LValue Base = CGF.EmitLoadOfPointerLValue(
4871       CGF.GetAddrOfLocalVar(&TaskPrivatesArg),
4872       TaskPrivatesArg.getType()->castAs<PointerType>());
4873   const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl());
4874   Counter = 0;
4875   for (const FieldDecl *Field : PrivatesQTyRD->fields()) {
4876     LValue FieldLVal = CGF.EmitLValueForField(Base, Field);
4877     const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]];
4878     LValue RefLVal =
4879         CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType());
4880     LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue(
4881         RefLVal.getAddress(), RefLVal.getType()->castAs<PointerType>());
4882     CGF.EmitStoreOfScalar(FieldLVal.getPointer(), RefLoadLVal);
4883     ++Counter;
4884   }
4885   CGF.FinishFunction();
4886   return TaskPrivatesMap;
4887 }
4888 
4889 /// Emit initialization for private variables in task-based directives.
4890 static void emitPrivatesInit(CodeGenFunction &CGF,
4891                              const OMPExecutableDirective &D,
4892                              Address KmpTaskSharedsPtr, LValue TDBase,
4893                              const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4894                              QualType SharedsTy, QualType SharedsPtrTy,
4895                              const OMPTaskDataTy &Data,
4896                              ArrayRef<PrivateDataTy> Privates, bool ForDup) {
4897   ASTContext &C = CGF.getContext();
4898   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4899   LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI);
4900   OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind())
4901                                  ? OMPD_taskloop
4902                                  : OMPD_task;
4903   const CapturedStmt &CS = *D.getCapturedStmt(Kind);
4904   CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS);
4905   LValue SrcBase;
4906   bool IsTargetTask =
4907       isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) ||
4908       isOpenMPTargetExecutionDirective(D.getDirectiveKind());
4909   // For target-based directives skip 3 firstprivate arrays BasePointersArray,
4910   // PointersArray and SizesArray. The original variables for these arrays are
4911   // not captured and we get their addresses explicitly.
4912   if ((!IsTargetTask && !Data.FirstprivateVars.empty()) ||
4913       (IsTargetTask && KmpTaskSharedsPtr.isValid())) {
4914     SrcBase = CGF.MakeAddrLValue(
4915         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4916             KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)),
4917         SharedsTy);
4918   }
4919   FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin();
4920   for (const PrivateDataTy &Pair : Privates) {
4921     const VarDecl *VD = Pair.second.PrivateCopy;
4922     const Expr *Init = VD->getAnyInitializer();
4923     if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) &&
4924                              !CGF.isTrivialInitializer(Init)))) {
4925       LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
4926       if (const VarDecl *Elem = Pair.second.PrivateElemInit) {
4927         const VarDecl *OriginalVD = Pair.second.Original;
4928         // Check if the variable is the target-based BasePointersArray,
4929         // PointersArray or SizesArray.
4930         LValue SharedRefLValue;
4931         QualType Type = PrivateLValue.getType();
4932         const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD);
4933         if (IsTargetTask && !SharedField) {
4934           assert(isa<ImplicitParamDecl>(OriginalVD) &&
4935                  isa<CapturedDecl>(OriginalVD->getDeclContext()) &&
4936                  cast<CapturedDecl>(OriginalVD->getDeclContext())
4937                          ->getNumParams() == 0 &&
4938                  isa<TranslationUnitDecl>(
4939                      cast<CapturedDecl>(OriginalVD->getDeclContext())
4940                          ->getDeclContext()) &&
4941                  "Expected artificial target data variable.");
4942           SharedRefLValue =
4943               CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type);
4944         } else {
4945           SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField);
4946           SharedRefLValue = CGF.MakeAddrLValue(
4947               Address(SharedRefLValue.getPointer(), C.getDeclAlign(OriginalVD)),
4948               SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl),
4949               SharedRefLValue.getTBAAInfo());
4950         }
4951         if (Type->isArrayType()) {
4952           // Initialize firstprivate array.
4953           if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) {
4954             // Perform simple memcpy.
4955             CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type);
4956           } else {
4957             // Initialize firstprivate array using element-by-element
4958             // initialization.
4959             CGF.EmitOMPAggregateAssign(
4960                 PrivateLValue.getAddress(), SharedRefLValue.getAddress(), Type,
4961                 [&CGF, Elem, Init, &CapturesInfo](Address DestElement,
4962                                                   Address SrcElement) {
4963                   // Clean up any temporaries needed by the initialization.
4964                   CodeGenFunction::OMPPrivateScope InitScope(CGF);
4965                   InitScope.addPrivate(
4966                       Elem, [SrcElement]() -> Address { return SrcElement; });
4967                   (void)InitScope.Privatize();
4968                   // Emit initialization for single element.
4969                   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(
4970                       CGF, &CapturesInfo);
4971                   CGF.EmitAnyExprToMem(Init, DestElement,
4972                                        Init->getType().getQualifiers(),
4973                                        /*IsInitializer=*/false);
4974                 });
4975           }
4976         } else {
4977           CodeGenFunction::OMPPrivateScope InitScope(CGF);
4978           InitScope.addPrivate(Elem, [SharedRefLValue]() -> Address {
4979             return SharedRefLValue.getAddress();
4980           });
4981           (void)InitScope.Privatize();
4982           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo);
4983           CGF.EmitExprAsInit(Init, VD, PrivateLValue,
4984                              /*capturedByInit=*/false);
4985         }
4986       } else {
4987         CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false);
4988       }
4989     }
4990     ++FI;
4991   }
4992 }
4993 
4994 /// Check if duplication function is required for taskloops.
4995 static bool checkInitIsRequired(CodeGenFunction &CGF,
4996                                 ArrayRef<PrivateDataTy> Privates) {
4997   bool InitRequired = false;
4998   for (const PrivateDataTy &Pair : Privates) {
4999     const VarDecl *VD = Pair.second.PrivateCopy;
5000     const Expr *Init = VD->getAnyInitializer();
5001     InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) &&
5002                                     !CGF.isTrivialInitializer(Init));
5003     if (InitRequired)
5004       break;
5005   }
5006   return InitRequired;
5007 }
5008 
5009 
5010 /// Emit task_dup function (for initialization of
5011 /// private/firstprivate/lastprivate vars and last_iter flag)
5012 /// \code
5013 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int
5014 /// lastpriv) {
5015 /// // setup lastprivate flag
5016 ///    task_dst->last = lastpriv;
5017 /// // could be constructor calls here...
5018 /// }
5019 /// \endcode
5020 static llvm::Value *
5021 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc,
5022                     const OMPExecutableDirective &D,
5023                     QualType KmpTaskTWithPrivatesPtrQTy,
5024                     const RecordDecl *KmpTaskTWithPrivatesQTyRD,
5025                     const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy,
5026                     QualType SharedsPtrTy, const OMPTaskDataTy &Data,
5027                     ArrayRef<PrivateDataTy> Privates, bool WithLastIter) {
5028   ASTContext &C = CGM.getContext();
5029   FunctionArgList Args;
5030   ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5031                            KmpTaskTWithPrivatesPtrQTy,
5032                            ImplicitParamDecl::Other);
5033   ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5034                            KmpTaskTWithPrivatesPtrQTy,
5035                            ImplicitParamDecl::Other);
5036   ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy,
5037                                 ImplicitParamDecl::Other);
5038   Args.push_back(&DstArg);
5039   Args.push_back(&SrcArg);
5040   Args.push_back(&LastprivArg);
5041   const auto &TaskDupFnInfo =
5042       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5043   llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo);
5044   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""});
5045   auto *TaskDup = llvm::Function::Create(
5046       TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
5047   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo);
5048   TaskDup->setDoesNotRecurse();
5049   CodeGenFunction CGF(CGM);
5050   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc,
5051                     Loc);
5052 
5053   LValue TDBase = CGF.EmitLoadOfPointerLValue(
5054       CGF.GetAddrOfLocalVar(&DstArg),
5055       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
5056   // task_dst->liter = lastpriv;
5057   if (WithLastIter) {
5058     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
5059     LValue Base = CGF.EmitLValueForField(
5060         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
5061     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
5062     llvm::Value *Lastpriv = CGF.EmitLoadOfScalar(
5063         CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc);
5064     CGF.EmitStoreOfScalar(Lastpriv, LILVal);
5065   }
5066 
5067   // Emit initial values for private copies (if any).
5068   assert(!Privates.empty());
5069   Address KmpTaskSharedsPtr = Address::invalid();
5070   if (!Data.FirstprivateVars.empty()) {
5071     LValue TDBase = CGF.EmitLoadOfPointerLValue(
5072         CGF.GetAddrOfLocalVar(&SrcArg),
5073         KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
5074     LValue Base = CGF.EmitLValueForField(
5075         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
5076     KmpTaskSharedsPtr = Address(
5077         CGF.EmitLoadOfScalar(CGF.EmitLValueForField(
5078                                  Base, *std::next(KmpTaskTQTyRD->field_begin(),
5079                                                   KmpTaskTShareds)),
5080                              Loc),
5081         CGF.getNaturalTypeAlignment(SharedsTy));
5082   }
5083   emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD,
5084                    SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true);
5085   CGF.FinishFunction();
5086   return TaskDup;
5087 }
5088 
5089 /// Checks if destructor function is required to be generated.
5090 /// \return true if cleanups are required, false otherwise.
5091 static bool
5092 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) {
5093   bool NeedsCleanup = false;
5094   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
5095   const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl());
5096   for (const FieldDecl *FD : PrivateRD->fields()) {
5097     NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType();
5098     if (NeedsCleanup)
5099       break;
5100   }
5101   return NeedsCleanup;
5102 }
5103 
5104 CGOpenMPRuntime::TaskResultTy
5105 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc,
5106                               const OMPExecutableDirective &D,
5107                               llvm::Function *TaskFunction, QualType SharedsTy,
5108                               Address Shareds, const OMPTaskDataTy &Data) {
5109   ASTContext &C = CGM.getContext();
5110   llvm::SmallVector<PrivateDataTy, 4> Privates;
5111   // Aggregate privates and sort them by the alignment.
5112   auto I = Data.PrivateCopies.begin();
5113   for (const Expr *E : Data.PrivateVars) {
5114     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
5115     Privates.emplace_back(
5116         C.getDeclAlign(VD),
5117         PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
5118                          /*PrivateElemInit=*/nullptr));
5119     ++I;
5120   }
5121   I = Data.FirstprivateCopies.begin();
5122   auto IElemInitRef = Data.FirstprivateInits.begin();
5123   for (const Expr *E : Data.FirstprivateVars) {
5124     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
5125     Privates.emplace_back(
5126         C.getDeclAlign(VD),
5127         PrivateHelpersTy(
5128             VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
5129             cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl())));
5130     ++I;
5131     ++IElemInitRef;
5132   }
5133   I = Data.LastprivateCopies.begin();
5134   for (const Expr *E : Data.LastprivateVars) {
5135     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
5136     Privates.emplace_back(
5137         C.getDeclAlign(VD),
5138         PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
5139                          /*PrivateElemInit=*/nullptr));
5140     ++I;
5141   }
5142   llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) {
5143     return L.first > R.first;
5144   });
5145   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
5146   // Build type kmp_routine_entry_t (if not built yet).
5147   emitKmpRoutineEntryT(KmpInt32Ty);
5148   // Build type kmp_task_t (if not built yet).
5149   if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) {
5150     if (SavedKmpTaskloopTQTy.isNull()) {
5151       SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl(
5152           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
5153     }
5154     KmpTaskTQTy = SavedKmpTaskloopTQTy;
5155   } else {
5156     assert((D.getDirectiveKind() == OMPD_task ||
5157             isOpenMPTargetExecutionDirective(D.getDirectiveKind()) ||
5158             isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) &&
5159            "Expected taskloop, task or target directive");
5160     if (SavedKmpTaskTQTy.isNull()) {
5161       SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl(
5162           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
5163     }
5164     KmpTaskTQTy = SavedKmpTaskTQTy;
5165   }
5166   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
5167   // Build particular struct kmp_task_t for the given task.
5168   const RecordDecl *KmpTaskTWithPrivatesQTyRD =
5169       createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates);
5170   QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD);
5171   QualType KmpTaskTWithPrivatesPtrQTy =
5172       C.getPointerType(KmpTaskTWithPrivatesQTy);
5173   llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy);
5174   llvm::Type *KmpTaskTWithPrivatesPtrTy =
5175       KmpTaskTWithPrivatesTy->getPointerTo();
5176   llvm::Value *KmpTaskTWithPrivatesTySize =
5177       CGF.getTypeSize(KmpTaskTWithPrivatesQTy);
5178   QualType SharedsPtrTy = C.getPointerType(SharedsTy);
5179 
5180   // Emit initial values for private copies (if any).
5181   llvm::Value *TaskPrivatesMap = nullptr;
5182   llvm::Type *TaskPrivatesMapTy =
5183       std::next(TaskFunction->arg_begin(), 3)->getType();
5184   if (!Privates.empty()) {
5185     auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
5186     TaskPrivatesMap = emitTaskPrivateMappingFunction(
5187         CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars,
5188         FI->getType(), Privates);
5189     TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5190         TaskPrivatesMap, TaskPrivatesMapTy);
5191   } else {
5192     TaskPrivatesMap = llvm::ConstantPointerNull::get(
5193         cast<llvm::PointerType>(TaskPrivatesMapTy));
5194   }
5195   // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
5196   // kmp_task_t *tt);
5197   llvm::Function *TaskEntry = emitProxyTaskFunction(
5198       CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5199       KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction,
5200       TaskPrivatesMap);
5201 
5202   // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
5203   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
5204   // kmp_routine_entry_t *task_entry);
5205   // Task flags. Format is taken from
5206   // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h,
5207   // description of kmp_tasking_flags struct.
5208   enum {
5209     TiedFlag = 0x1,
5210     FinalFlag = 0x2,
5211     DestructorsFlag = 0x8,
5212     PriorityFlag = 0x20
5213   };
5214   unsigned Flags = Data.Tied ? TiedFlag : 0;
5215   bool NeedsCleanup = false;
5216   if (!Privates.empty()) {
5217     NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD);
5218     if (NeedsCleanup)
5219       Flags = Flags | DestructorsFlag;
5220   }
5221   if (Data.Priority.getInt())
5222     Flags = Flags | PriorityFlag;
5223   llvm::Value *TaskFlags =
5224       Data.Final.getPointer()
5225           ? CGF.Builder.CreateSelect(Data.Final.getPointer(),
5226                                      CGF.Builder.getInt32(FinalFlag),
5227                                      CGF.Builder.getInt32(/*C=*/0))
5228           : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0);
5229   TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
5230   llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy));
5231   SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc),
5232       getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize,
5233       SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5234           TaskEntry, KmpRoutineEntryPtrTy)};
5235   llvm::Value *NewTask;
5236   if (D.hasClausesOfKind<OMPNowaitClause>()) {
5237     // Check if we have any device clause associated with the directive.
5238     const Expr *Device = nullptr;
5239     if (auto *C = D.getSingleClause<OMPDeviceClause>())
5240       Device = C->getDevice();
5241     // Emit device ID if any otherwise use default value.
5242     llvm::Value *DeviceID;
5243     if (Device)
5244       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
5245                                            CGF.Int64Ty, /*isSigned=*/true);
5246     else
5247       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
5248     AllocArgs.push_back(DeviceID);
5249     NewTask = CGF.EmitRuntimeCall(
5250       createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs);
5251   } else {
5252     NewTask = CGF.EmitRuntimeCall(
5253       createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs);
5254   }
5255   llvm::Value *NewTaskNewTaskTTy =
5256       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5257           NewTask, KmpTaskTWithPrivatesPtrTy);
5258   LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy,
5259                                                KmpTaskTWithPrivatesQTy);
5260   LValue TDBase =
5261       CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
5262   // Fill the data in the resulting kmp_task_t record.
5263   // Copy shareds if there are any.
5264   Address KmpTaskSharedsPtr = Address::invalid();
5265   if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) {
5266     KmpTaskSharedsPtr =
5267         Address(CGF.EmitLoadOfScalar(
5268                     CGF.EmitLValueForField(
5269                         TDBase, *std::next(KmpTaskTQTyRD->field_begin(),
5270                                            KmpTaskTShareds)),
5271                     Loc),
5272                 CGF.getNaturalTypeAlignment(SharedsTy));
5273     LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy);
5274     LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy);
5275     CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap);
5276   }
5277   // Emit initial values for private copies (if any).
5278   TaskResultTy Result;
5279   if (!Privates.empty()) {
5280     emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD,
5281                      SharedsTy, SharedsPtrTy, Data, Privates,
5282                      /*ForDup=*/false);
5283     if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) &&
5284         (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) {
5285       Result.TaskDupFn = emitTaskDupFunction(
5286           CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD,
5287           KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates,
5288           /*WithLastIter=*/!Data.LastprivateVars.empty());
5289     }
5290   }
5291   // Fields of union "kmp_cmplrdata_t" for destructors and priority.
5292   enum { Priority = 0, Destructors = 1 };
5293   // Provide pointer to function with destructors for privates.
5294   auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1);
5295   const RecordDecl *KmpCmplrdataUD =
5296       (*FI)->getType()->getAsUnionType()->getDecl();
5297   if (NeedsCleanup) {
5298     llvm::Value *DestructorFn = emitDestructorsFunction(
5299         CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
5300         KmpTaskTWithPrivatesQTy);
5301     LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI);
5302     LValue DestructorsLV = CGF.EmitLValueForField(
5303         Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors));
5304     CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5305                               DestructorFn, KmpRoutineEntryPtrTy),
5306                           DestructorsLV);
5307   }
5308   // Set priority.
5309   if (Data.Priority.getInt()) {
5310     LValue Data2LV = CGF.EmitLValueForField(
5311         TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2));
5312     LValue PriorityLV = CGF.EmitLValueForField(
5313         Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority));
5314     CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV);
5315   }
5316   Result.NewTask = NewTask;
5317   Result.TaskEntry = TaskEntry;
5318   Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy;
5319   Result.TDBase = TDBase;
5320   Result.KmpTaskTQTyRD = KmpTaskTQTyRD;
5321   return Result;
5322 }
5323 
5324 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
5325                                    const OMPExecutableDirective &D,
5326                                    llvm::Function *TaskFunction,
5327                                    QualType SharedsTy, Address Shareds,
5328                                    const Expr *IfCond,
5329                                    const OMPTaskDataTy &Data) {
5330   if (!CGF.HaveInsertPoint())
5331     return;
5332 
5333   TaskResultTy Result =
5334       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5335   llvm::Value *NewTask = Result.NewTask;
5336   llvm::Function *TaskEntry = Result.TaskEntry;
5337   llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy;
5338   LValue TDBase = Result.TDBase;
5339   const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD;
5340   ASTContext &C = CGM.getContext();
5341   // Process list of dependences.
5342   Address DependenciesArray = Address::invalid();
5343   unsigned NumDependencies = Data.Dependences.size();
5344   if (NumDependencies) {
5345     // Dependence kind for RTL.
5346     enum RTLDependenceKindTy { DepIn = 0x01, DepInOut = 0x3, DepMutexInOutSet = 0x4 };
5347     enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags };
5348     RecordDecl *KmpDependInfoRD;
5349     QualType FlagsTy =
5350         C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false);
5351     llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
5352     if (KmpDependInfoTy.isNull()) {
5353       KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info");
5354       KmpDependInfoRD->startDefinition();
5355       addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType());
5356       addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType());
5357       addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy);
5358       KmpDependInfoRD->completeDefinition();
5359       KmpDependInfoTy = C.getRecordType(KmpDependInfoRD);
5360     } else {
5361       KmpDependInfoRD = cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5362     }
5363     // Define type kmp_depend_info[<Dependences.size()>];
5364     QualType KmpDependInfoArrayTy = C.getConstantArrayType(
5365         KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies),
5366         nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
5367     // kmp_depend_info[<Dependences.size()>] deps;
5368     DependenciesArray =
5369         CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr");
5370     for (unsigned I = 0; I < NumDependencies; ++I) {
5371       const Expr *E = Data.Dependences[I].second;
5372       LValue Addr = CGF.EmitLValue(E);
5373       llvm::Value *Size;
5374       QualType Ty = E->getType();
5375       if (const auto *ASE =
5376               dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) {
5377         LValue UpAddrLVal =
5378             CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false);
5379         llvm::Value *UpAddr =
5380             CGF.Builder.CreateConstGEP1_32(UpAddrLVal.getPointer(), /*Idx0=*/1);
5381         llvm::Value *LowIntPtr =
5382             CGF.Builder.CreatePtrToInt(Addr.getPointer(), CGM.SizeTy);
5383         llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy);
5384         Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr);
5385       } else {
5386         Size = CGF.getTypeSize(Ty);
5387       }
5388       LValue Base = CGF.MakeAddrLValue(
5389           CGF.Builder.CreateConstArrayGEP(DependenciesArray, I),
5390           KmpDependInfoTy);
5391       // deps[i].base_addr = &<Dependences[i].second>;
5392       LValue BaseAddrLVal = CGF.EmitLValueForField(
5393           Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5394       CGF.EmitStoreOfScalar(
5395           CGF.Builder.CreatePtrToInt(Addr.getPointer(), CGF.IntPtrTy),
5396           BaseAddrLVal);
5397       // deps[i].len = sizeof(<Dependences[i].second>);
5398       LValue LenLVal = CGF.EmitLValueForField(
5399           Base, *std::next(KmpDependInfoRD->field_begin(), Len));
5400       CGF.EmitStoreOfScalar(Size, LenLVal);
5401       // deps[i].flags = <Dependences[i].first>;
5402       RTLDependenceKindTy DepKind;
5403       switch (Data.Dependences[I].first) {
5404       case OMPC_DEPEND_in:
5405         DepKind = DepIn;
5406         break;
5407       // Out and InOut dependencies must use the same code.
5408       case OMPC_DEPEND_out:
5409       case OMPC_DEPEND_inout:
5410         DepKind = DepInOut;
5411         break;
5412       case OMPC_DEPEND_mutexinoutset:
5413         DepKind = DepMutexInOutSet;
5414         break;
5415       case OMPC_DEPEND_source:
5416       case OMPC_DEPEND_sink:
5417       case OMPC_DEPEND_unknown:
5418         llvm_unreachable("Unknown task dependence type");
5419       }
5420       LValue FlagsLVal = CGF.EmitLValueForField(
5421           Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
5422       CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
5423                             FlagsLVal);
5424     }
5425     DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5426         CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0), CGF.VoidPtrTy);
5427   }
5428 
5429   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5430   // libcall.
5431   // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
5432   // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
5433   // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence
5434   // list is not empty
5435   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5436   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5437   llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask };
5438   llvm::Value *DepTaskArgs[7];
5439   if (NumDependencies) {
5440     DepTaskArgs[0] = UpLoc;
5441     DepTaskArgs[1] = ThreadID;
5442     DepTaskArgs[2] = NewTask;
5443     DepTaskArgs[3] = CGF.Builder.getInt32(NumDependencies);
5444     DepTaskArgs[4] = DependenciesArray.getPointer();
5445     DepTaskArgs[5] = CGF.Builder.getInt32(0);
5446     DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5447   }
5448   auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, NumDependencies,
5449                         &TaskArgs,
5450                         &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) {
5451     if (!Data.Tied) {
5452       auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
5453       LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI);
5454       CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal);
5455     }
5456     if (NumDependencies) {
5457       CGF.EmitRuntimeCall(
5458           createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs);
5459     } else {
5460       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task),
5461                           TaskArgs);
5462     }
5463     // Check if parent region is untied and build return for untied task;
5464     if (auto *Region =
5465             dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
5466       Region->emitUntiedSwitch(CGF);
5467   };
5468 
5469   llvm::Value *DepWaitTaskArgs[6];
5470   if (NumDependencies) {
5471     DepWaitTaskArgs[0] = UpLoc;
5472     DepWaitTaskArgs[1] = ThreadID;
5473     DepWaitTaskArgs[2] = CGF.Builder.getInt32(NumDependencies);
5474     DepWaitTaskArgs[3] = DependenciesArray.getPointer();
5475     DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
5476     DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5477   }
5478   auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry,
5479                         NumDependencies, &DepWaitTaskArgs,
5480                         Loc](CodeGenFunction &CGF, PrePostActionTy &) {
5481     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5482     CodeGenFunction::RunCleanupsScope LocalScope(CGF);
5483     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
5484     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
5485     // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
5486     // is specified.
5487     if (NumDependencies)
5488       CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps),
5489                           DepWaitTaskArgs);
5490     // Call proxy_task_entry(gtid, new_task);
5491     auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy,
5492                       Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
5493       Action.Enter(CGF);
5494       llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
5495       CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry,
5496                                                           OutlinedFnArgs);
5497     };
5498 
5499     // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
5500     // kmp_task_t *new_task);
5501     // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
5502     // kmp_task_t *new_task);
5503     RegionCodeGenTy RCG(CodeGen);
5504     CommonActionTy Action(
5505         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs,
5506         RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs);
5507     RCG.setAction(Action);
5508     RCG(CGF);
5509   };
5510 
5511   if (IfCond) {
5512     emitOMPIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen);
5513   } else {
5514     RegionCodeGenTy ThenRCG(ThenCodeGen);
5515     ThenRCG(CGF);
5516   }
5517 }
5518 
5519 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc,
5520                                        const OMPLoopDirective &D,
5521                                        llvm::Function *TaskFunction,
5522                                        QualType SharedsTy, Address Shareds,
5523                                        const Expr *IfCond,
5524                                        const OMPTaskDataTy &Data) {
5525   if (!CGF.HaveInsertPoint())
5526     return;
5527   TaskResultTy Result =
5528       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5529   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5530   // libcall.
5531   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
5532   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
5533   // sched, kmp_uint64 grainsize, void *task_dup);
5534   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5535   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5536   llvm::Value *IfVal;
5537   if (IfCond) {
5538     IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy,
5539                                       /*isSigned=*/true);
5540   } else {
5541     IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1);
5542   }
5543 
5544   LValue LBLVal = CGF.EmitLValueForField(
5545       Result.TDBase,
5546       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound));
5547   const auto *LBVar =
5548       cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl());
5549   CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(), LBLVal.getQuals(),
5550                        /*IsInitializer=*/true);
5551   LValue UBLVal = CGF.EmitLValueForField(
5552       Result.TDBase,
5553       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound));
5554   const auto *UBVar =
5555       cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl());
5556   CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(), UBLVal.getQuals(),
5557                        /*IsInitializer=*/true);
5558   LValue StLVal = CGF.EmitLValueForField(
5559       Result.TDBase,
5560       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride));
5561   const auto *StVar =
5562       cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl());
5563   CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(), StLVal.getQuals(),
5564                        /*IsInitializer=*/true);
5565   // Store reductions address.
5566   LValue RedLVal = CGF.EmitLValueForField(
5567       Result.TDBase,
5568       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions));
5569   if (Data.Reductions) {
5570     CGF.EmitStoreOfScalar(Data.Reductions, RedLVal);
5571   } else {
5572     CGF.EmitNullInitialization(RedLVal.getAddress(),
5573                                CGF.getContext().VoidPtrTy);
5574   }
5575   enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 };
5576   llvm::Value *TaskArgs[] = {
5577       UpLoc,
5578       ThreadID,
5579       Result.NewTask,
5580       IfVal,
5581       LBLVal.getPointer(),
5582       UBLVal.getPointer(),
5583       CGF.EmitLoadOfScalar(StLVal, Loc),
5584       llvm::ConstantInt::getSigned(
5585               CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler
5586       llvm::ConstantInt::getSigned(
5587           CGF.IntTy, Data.Schedule.getPointer()
5588                          ? Data.Schedule.getInt() ? NumTasks : Grainsize
5589                          : NoSchedule),
5590       Data.Schedule.getPointer()
5591           ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty,
5592                                       /*isSigned=*/false)
5593           : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0),
5594       Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5595                              Result.TaskDupFn, CGF.VoidPtrTy)
5596                        : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)};
5597   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs);
5598 }
5599 
5600 /// Emit reduction operation for each element of array (required for
5601 /// array sections) LHS op = RHS.
5602 /// \param Type Type of array.
5603 /// \param LHSVar Variable on the left side of the reduction operation
5604 /// (references element of array in original variable).
5605 /// \param RHSVar Variable on the right side of the reduction operation
5606 /// (references element of array in original variable).
5607 /// \param RedOpGen Generator of reduction operation with use of LHSVar and
5608 /// RHSVar.
5609 static void EmitOMPAggregateReduction(
5610     CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar,
5611     const VarDecl *RHSVar,
5612     const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *,
5613                                   const Expr *, const Expr *)> &RedOpGen,
5614     const Expr *XExpr = nullptr, const Expr *EExpr = nullptr,
5615     const Expr *UpExpr = nullptr) {
5616   // Perform element-by-element initialization.
5617   QualType ElementTy;
5618   Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar);
5619   Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar);
5620 
5621   // Drill down to the base element type on both arrays.
5622   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
5623   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr);
5624 
5625   llvm::Value *RHSBegin = RHSAddr.getPointer();
5626   llvm::Value *LHSBegin = LHSAddr.getPointer();
5627   // Cast from pointer to array type to pointer to single element.
5628   llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements);
5629   // The basic structure here is a while-do loop.
5630   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body");
5631   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done");
5632   llvm::Value *IsEmpty =
5633       CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty");
5634   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5635 
5636   // Enter the loop body, making that address the current address.
5637   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5638   CGF.EmitBlock(BodyBB);
5639 
5640   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
5641 
5642   llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI(
5643       RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast");
5644   RHSElementPHI->addIncoming(RHSBegin, EntryBB);
5645   Address RHSElementCurrent =
5646       Address(RHSElementPHI,
5647               RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5648 
5649   llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI(
5650       LHSBegin->getType(), 2, "omp.arraycpy.destElementPast");
5651   LHSElementPHI->addIncoming(LHSBegin, EntryBB);
5652   Address LHSElementCurrent =
5653       Address(LHSElementPHI,
5654               LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5655 
5656   // Emit copy.
5657   CodeGenFunction::OMPPrivateScope Scope(CGF);
5658   Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; });
5659   Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; });
5660   Scope.Privatize();
5661   RedOpGen(CGF, XExpr, EExpr, UpExpr);
5662   Scope.ForceCleanup();
5663 
5664   // Shift the address forward by one element.
5665   llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32(
5666       LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element");
5667   llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32(
5668       RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element");
5669   // Check whether we've reached the end.
5670   llvm::Value *Done =
5671       CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done");
5672   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
5673   LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock());
5674   RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock());
5675 
5676   // Done.
5677   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5678 }
5679 
5680 /// Emit reduction combiner. If the combiner is a simple expression emit it as
5681 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of
5682 /// UDR combiner function.
5683 static void emitReductionCombiner(CodeGenFunction &CGF,
5684                                   const Expr *ReductionOp) {
5685   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
5686     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
5687       if (const auto *DRE =
5688               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
5689         if (const auto *DRD =
5690                 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) {
5691           std::pair<llvm::Function *, llvm::Function *> Reduction =
5692               CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
5693           RValue Func = RValue::get(Reduction.first);
5694           CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
5695           CGF.EmitIgnoredExpr(ReductionOp);
5696           return;
5697         }
5698   CGF.EmitIgnoredExpr(ReductionOp);
5699 }
5700 
5701 llvm::Function *CGOpenMPRuntime::emitReductionFunction(
5702     SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates,
5703     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
5704     ArrayRef<const Expr *> ReductionOps) {
5705   ASTContext &C = CGM.getContext();
5706 
5707   // void reduction_func(void *LHSArg, void *RHSArg);
5708   FunctionArgList Args;
5709   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5710                            ImplicitParamDecl::Other);
5711   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5712                            ImplicitParamDecl::Other);
5713   Args.push_back(&LHSArg);
5714   Args.push_back(&RHSArg);
5715   const auto &CGFI =
5716       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5717   std::string Name = getName({"omp", "reduction", "reduction_func"});
5718   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
5719                                     llvm::GlobalValue::InternalLinkage, Name,
5720                                     &CGM.getModule());
5721   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
5722   Fn->setDoesNotRecurse();
5723   CodeGenFunction CGF(CGM);
5724   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
5725 
5726   // Dst = (void*[n])(LHSArg);
5727   // Src = (void*[n])(RHSArg);
5728   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5729       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
5730       ArgsType), CGF.getPointerAlign());
5731   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5732       CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
5733       ArgsType), CGF.getPointerAlign());
5734 
5735   //  ...
5736   //  *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
5737   //  ...
5738   CodeGenFunction::OMPPrivateScope Scope(CGF);
5739   auto IPriv = Privates.begin();
5740   unsigned Idx = 0;
5741   for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) {
5742     const auto *RHSVar =
5743         cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl());
5744     Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() {
5745       return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar);
5746     });
5747     const auto *LHSVar =
5748         cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl());
5749     Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() {
5750       return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar);
5751     });
5752     QualType PrivTy = (*IPriv)->getType();
5753     if (PrivTy->isVariablyModifiedType()) {
5754       // Get array size and emit VLA type.
5755       ++Idx;
5756       Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx);
5757       llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem);
5758       const VariableArrayType *VLA =
5759           CGF.getContext().getAsVariableArrayType(PrivTy);
5760       const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr());
5761       CodeGenFunction::OpaqueValueMapping OpaqueMap(
5762           CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy)));
5763       CGF.EmitVariablyModifiedType(PrivTy);
5764     }
5765   }
5766   Scope.Privatize();
5767   IPriv = Privates.begin();
5768   auto ILHS = LHSExprs.begin();
5769   auto IRHS = RHSExprs.begin();
5770   for (const Expr *E : ReductionOps) {
5771     if ((*IPriv)->getType()->isArrayType()) {
5772       // Emit reduction for array section.
5773       const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5774       const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5775       EmitOMPAggregateReduction(
5776           CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5777           [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5778             emitReductionCombiner(CGF, E);
5779           });
5780     } else {
5781       // Emit reduction for array subscript or single variable.
5782       emitReductionCombiner(CGF, E);
5783     }
5784     ++IPriv;
5785     ++ILHS;
5786     ++IRHS;
5787   }
5788   Scope.ForceCleanup();
5789   CGF.FinishFunction();
5790   return Fn;
5791 }
5792 
5793 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF,
5794                                                   const Expr *ReductionOp,
5795                                                   const Expr *PrivateRef,
5796                                                   const DeclRefExpr *LHS,
5797                                                   const DeclRefExpr *RHS) {
5798   if (PrivateRef->getType()->isArrayType()) {
5799     // Emit reduction for array section.
5800     const auto *LHSVar = cast<VarDecl>(LHS->getDecl());
5801     const auto *RHSVar = cast<VarDecl>(RHS->getDecl());
5802     EmitOMPAggregateReduction(
5803         CGF, PrivateRef->getType(), LHSVar, RHSVar,
5804         [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5805           emitReductionCombiner(CGF, ReductionOp);
5806         });
5807   } else {
5808     // Emit reduction for array subscript or single variable.
5809     emitReductionCombiner(CGF, ReductionOp);
5810   }
5811 }
5812 
5813 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc,
5814                                     ArrayRef<const Expr *> Privates,
5815                                     ArrayRef<const Expr *> LHSExprs,
5816                                     ArrayRef<const Expr *> RHSExprs,
5817                                     ArrayRef<const Expr *> ReductionOps,
5818                                     ReductionOptionsTy Options) {
5819   if (!CGF.HaveInsertPoint())
5820     return;
5821 
5822   bool WithNowait = Options.WithNowait;
5823   bool SimpleReduction = Options.SimpleReduction;
5824 
5825   // Next code should be emitted for reduction:
5826   //
5827   // static kmp_critical_name lock = { 0 };
5828   //
5829   // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
5830   //  *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
5831   //  ...
5832   //  *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
5833   //  *(Type<n>-1*)rhs[<n>-1]);
5834   // }
5835   //
5836   // ...
5837   // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
5838   // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5839   // RedList, reduce_func, &<lock>)) {
5840   // case 1:
5841   //  ...
5842   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5843   //  ...
5844   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5845   // break;
5846   // case 2:
5847   //  ...
5848   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5849   //  ...
5850   // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
5851   // break;
5852   // default:;
5853   // }
5854   //
5855   // if SimpleReduction is true, only the next code is generated:
5856   //  ...
5857   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5858   //  ...
5859 
5860   ASTContext &C = CGM.getContext();
5861 
5862   if (SimpleReduction) {
5863     CodeGenFunction::RunCleanupsScope Scope(CGF);
5864     auto IPriv = Privates.begin();
5865     auto ILHS = LHSExprs.begin();
5866     auto IRHS = RHSExprs.begin();
5867     for (const Expr *E : ReductionOps) {
5868       emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5869                                   cast<DeclRefExpr>(*IRHS));
5870       ++IPriv;
5871       ++ILHS;
5872       ++IRHS;
5873     }
5874     return;
5875   }
5876 
5877   // 1. Build a list of reduction variables.
5878   // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
5879   auto Size = RHSExprs.size();
5880   for (const Expr *E : Privates) {
5881     if (E->getType()->isVariablyModifiedType())
5882       // Reserve place for array size.
5883       ++Size;
5884   }
5885   llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size);
5886   QualType ReductionArrayTy =
5887       C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
5888                              /*IndexTypeQuals=*/0);
5889   Address ReductionList =
5890       CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list");
5891   auto IPriv = Privates.begin();
5892   unsigned Idx = 0;
5893   for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) {
5894     Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5895     CGF.Builder.CreateStore(
5896         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5897             CGF.EmitLValue(RHSExprs[I]).getPointer(), CGF.VoidPtrTy),
5898         Elem);
5899     if ((*IPriv)->getType()->isVariablyModifiedType()) {
5900       // Store array size.
5901       ++Idx;
5902       Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5903       llvm::Value *Size = CGF.Builder.CreateIntCast(
5904           CGF.getVLASize(
5905                  CGF.getContext().getAsVariableArrayType((*IPriv)->getType()))
5906               .NumElts,
5907           CGF.SizeTy, /*isSigned=*/false);
5908       CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy),
5909                               Elem);
5910     }
5911   }
5912 
5913   // 2. Emit reduce_func().
5914   llvm::Function *ReductionFn = emitReductionFunction(
5915       Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates,
5916       LHSExprs, RHSExprs, ReductionOps);
5917 
5918   // 3. Create static kmp_critical_name lock = { 0 };
5919   std::string Name = getName({"reduction"});
5920   llvm::Value *Lock = getCriticalRegionLock(Name);
5921 
5922   // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5923   // RedList, reduce_func, &<lock>);
5924   llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE);
5925   llvm::Value *ThreadId = getThreadID(CGF, Loc);
5926   llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy);
5927   llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5928       ReductionList.getPointer(), CGF.VoidPtrTy);
5929   llvm::Value *Args[] = {
5930       IdentTLoc,                             // ident_t *<loc>
5931       ThreadId,                              // i32 <gtid>
5932       CGF.Builder.getInt32(RHSExprs.size()), // i32 <n>
5933       ReductionArrayTySize,                  // size_type sizeof(RedList)
5934       RL,                                    // void *RedList
5935       ReductionFn, // void (*) (void *, void *) <reduce_func>
5936       Lock         // kmp_critical_name *&<lock>
5937   };
5938   llvm::Value *Res = CGF.EmitRuntimeCall(
5939       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait
5940                                        : OMPRTL__kmpc_reduce),
5941       Args);
5942 
5943   // 5. Build switch(res)
5944   llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default");
5945   llvm::SwitchInst *SwInst =
5946       CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2);
5947 
5948   // 6. Build case 1:
5949   //  ...
5950   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5951   //  ...
5952   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5953   // break;
5954   llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1");
5955   SwInst->addCase(CGF.Builder.getInt32(1), Case1BB);
5956   CGF.EmitBlock(Case1BB);
5957 
5958   // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5959   llvm::Value *EndArgs[] = {
5960       IdentTLoc, // ident_t *<loc>
5961       ThreadId,  // i32 <gtid>
5962       Lock       // kmp_critical_name *&<lock>
5963   };
5964   auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps](
5965                        CodeGenFunction &CGF, PrePostActionTy &Action) {
5966     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5967     auto IPriv = Privates.begin();
5968     auto ILHS = LHSExprs.begin();
5969     auto IRHS = RHSExprs.begin();
5970     for (const Expr *E : ReductionOps) {
5971       RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5972                                      cast<DeclRefExpr>(*IRHS));
5973       ++IPriv;
5974       ++ILHS;
5975       ++IRHS;
5976     }
5977   };
5978   RegionCodeGenTy RCG(CodeGen);
5979   CommonActionTy Action(
5980       nullptr, llvm::None,
5981       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait
5982                                        : OMPRTL__kmpc_end_reduce),
5983       EndArgs);
5984   RCG.setAction(Action);
5985   RCG(CGF);
5986 
5987   CGF.EmitBranch(DefaultBB);
5988 
5989   // 7. Build case 2:
5990   //  ...
5991   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5992   //  ...
5993   // break;
5994   llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2");
5995   SwInst->addCase(CGF.Builder.getInt32(2), Case2BB);
5996   CGF.EmitBlock(Case2BB);
5997 
5998   auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps](
5999                              CodeGenFunction &CGF, PrePostActionTy &Action) {
6000     auto ILHS = LHSExprs.begin();
6001     auto IRHS = RHSExprs.begin();
6002     auto IPriv = Privates.begin();
6003     for (const Expr *E : ReductionOps) {
6004       const Expr *XExpr = nullptr;
6005       const Expr *EExpr = nullptr;
6006       const Expr *UpExpr = nullptr;
6007       BinaryOperatorKind BO = BO_Comma;
6008       if (const auto *BO = dyn_cast<BinaryOperator>(E)) {
6009         if (BO->getOpcode() == BO_Assign) {
6010           XExpr = BO->getLHS();
6011           UpExpr = BO->getRHS();
6012         }
6013       }
6014       // Try to emit update expression as a simple atomic.
6015       const Expr *RHSExpr = UpExpr;
6016       if (RHSExpr) {
6017         // Analyze RHS part of the whole expression.
6018         if (const auto *ACO = dyn_cast<AbstractConditionalOperator>(
6019                 RHSExpr->IgnoreParenImpCasts())) {
6020           // If this is a conditional operator, analyze its condition for
6021           // min/max reduction operator.
6022           RHSExpr = ACO->getCond();
6023         }
6024         if (const auto *BORHS =
6025                 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) {
6026           EExpr = BORHS->getRHS();
6027           BO = BORHS->getOpcode();
6028         }
6029       }
6030       if (XExpr) {
6031         const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
6032         auto &&AtomicRedGen = [BO, VD,
6033                                Loc](CodeGenFunction &CGF, const Expr *XExpr,
6034                                     const Expr *EExpr, const Expr *UpExpr) {
6035           LValue X = CGF.EmitLValue(XExpr);
6036           RValue E;
6037           if (EExpr)
6038             E = CGF.EmitAnyExpr(EExpr);
6039           CGF.EmitOMPAtomicSimpleUpdateExpr(
6040               X, E, BO, /*IsXLHSInRHSPart=*/true,
6041               llvm::AtomicOrdering::Monotonic, Loc,
6042               [&CGF, UpExpr, VD, Loc](RValue XRValue) {
6043                 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
6044                 PrivateScope.addPrivate(
6045                     VD, [&CGF, VD, XRValue, Loc]() {
6046                       Address LHSTemp = CGF.CreateMemTemp(VD->getType());
6047                       CGF.emitOMPSimpleStore(
6048                           CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue,
6049                           VD->getType().getNonReferenceType(), Loc);
6050                       return LHSTemp;
6051                     });
6052                 (void)PrivateScope.Privatize();
6053                 return CGF.EmitAnyExpr(UpExpr);
6054               });
6055         };
6056         if ((*IPriv)->getType()->isArrayType()) {
6057           // Emit atomic reduction for array section.
6058           const auto *RHSVar =
6059               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
6060           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar,
6061                                     AtomicRedGen, XExpr, EExpr, UpExpr);
6062         } else {
6063           // Emit atomic reduction for array subscript or single variable.
6064           AtomicRedGen(CGF, XExpr, EExpr, UpExpr);
6065         }
6066       } else {
6067         // Emit as a critical region.
6068         auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *,
6069                                            const Expr *, const Expr *) {
6070           CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6071           std::string Name = RT.getName({"atomic_reduction"});
6072           RT.emitCriticalRegion(
6073               CGF, Name,
6074               [=](CodeGenFunction &CGF, PrePostActionTy &Action) {
6075                 Action.Enter(CGF);
6076                 emitReductionCombiner(CGF, E);
6077               },
6078               Loc);
6079         };
6080         if ((*IPriv)->getType()->isArrayType()) {
6081           const auto *LHSVar =
6082               cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
6083           const auto *RHSVar =
6084               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
6085           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar,
6086                                     CritRedGen);
6087         } else {
6088           CritRedGen(CGF, nullptr, nullptr, nullptr);
6089         }
6090       }
6091       ++ILHS;
6092       ++IRHS;
6093       ++IPriv;
6094     }
6095   };
6096   RegionCodeGenTy AtomicRCG(AtomicCodeGen);
6097   if (!WithNowait) {
6098     // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
6099     llvm::Value *EndArgs[] = {
6100         IdentTLoc, // ident_t *<loc>
6101         ThreadId,  // i32 <gtid>
6102         Lock       // kmp_critical_name *&<lock>
6103     };
6104     CommonActionTy Action(nullptr, llvm::None,
6105                           createRuntimeFunction(OMPRTL__kmpc_end_reduce),
6106                           EndArgs);
6107     AtomicRCG.setAction(Action);
6108     AtomicRCG(CGF);
6109   } else {
6110     AtomicRCG(CGF);
6111   }
6112 
6113   CGF.EmitBranch(DefaultBB);
6114   CGF.EmitBlock(DefaultBB, /*IsFinished=*/true);
6115 }
6116 
6117 /// Generates unique name for artificial threadprivate variables.
6118 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>"
6119 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix,
6120                                       const Expr *Ref) {
6121   SmallString<256> Buffer;
6122   llvm::raw_svector_ostream Out(Buffer);
6123   const clang::DeclRefExpr *DE;
6124   const VarDecl *D = ::getBaseDecl(Ref, DE);
6125   if (!D)
6126     D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl());
6127   D = D->getCanonicalDecl();
6128   std::string Name = CGM.getOpenMPRuntime().getName(
6129       {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)});
6130   Out << Prefix << Name << "_"
6131       << D->getCanonicalDecl()->getBeginLoc().getRawEncoding();
6132   return Out.str();
6133 }
6134 
6135 /// Emits reduction initializer function:
6136 /// \code
6137 /// void @.red_init(void* %arg) {
6138 /// %0 = bitcast void* %arg to <type>*
6139 /// store <type> <init>, <type>* %0
6140 /// ret void
6141 /// }
6142 /// \endcode
6143 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM,
6144                                            SourceLocation Loc,
6145                                            ReductionCodeGen &RCG, unsigned N) {
6146   ASTContext &C = CGM.getContext();
6147   FunctionArgList Args;
6148   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6149                           ImplicitParamDecl::Other);
6150   Args.emplace_back(&Param);
6151   const auto &FnInfo =
6152       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6153   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6154   std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""});
6155   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6156                                     Name, &CGM.getModule());
6157   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6158   Fn->setDoesNotRecurse();
6159   CodeGenFunction CGF(CGM);
6160   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6161   Address PrivateAddr = CGF.EmitLoadOfPointer(
6162       CGF.GetAddrOfLocalVar(&Param),
6163       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6164   llvm::Value *Size = nullptr;
6165   // If the size of the reduction item is non-constant, load it from global
6166   // threadprivate variable.
6167   if (RCG.getSizes(N).second) {
6168     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6169         CGF, CGM.getContext().getSizeType(),
6170         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6171     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6172                                 CGM.getContext().getSizeType(), Loc);
6173   }
6174   RCG.emitAggregateType(CGF, N, Size);
6175   LValue SharedLVal;
6176   // If initializer uses initializer from declare reduction construct, emit a
6177   // pointer to the address of the original reduction item (reuired by reduction
6178   // initializer)
6179   if (RCG.usesReductionInitializer(N)) {
6180     Address SharedAddr =
6181         CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6182             CGF, CGM.getContext().VoidPtrTy,
6183             generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6184     SharedAddr = CGF.EmitLoadOfPointer(
6185         SharedAddr,
6186         CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr());
6187     SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy);
6188   } else {
6189     SharedLVal = CGF.MakeNaturalAlignAddrLValue(
6190         llvm::ConstantPointerNull::get(CGM.VoidPtrTy),
6191         CGM.getContext().VoidPtrTy);
6192   }
6193   // Emit the initializer:
6194   // %0 = bitcast void* %arg to <type>*
6195   // store <type> <init>, <type>* %0
6196   RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal,
6197                          [](CodeGenFunction &) { return false; });
6198   CGF.FinishFunction();
6199   return Fn;
6200 }
6201 
6202 /// Emits reduction combiner function:
6203 /// \code
6204 /// void @.red_comb(void* %arg0, void* %arg1) {
6205 /// %lhs = bitcast void* %arg0 to <type>*
6206 /// %rhs = bitcast void* %arg1 to <type>*
6207 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs)
6208 /// store <type> %2, <type>* %lhs
6209 /// ret void
6210 /// }
6211 /// \endcode
6212 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM,
6213                                            SourceLocation Loc,
6214                                            ReductionCodeGen &RCG, unsigned N,
6215                                            const Expr *ReductionOp,
6216                                            const Expr *LHS, const Expr *RHS,
6217                                            const Expr *PrivateRef) {
6218   ASTContext &C = CGM.getContext();
6219   const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl());
6220   const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl());
6221   FunctionArgList Args;
6222   ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
6223                                C.VoidPtrTy, ImplicitParamDecl::Other);
6224   ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6225                             ImplicitParamDecl::Other);
6226   Args.emplace_back(&ParamInOut);
6227   Args.emplace_back(&ParamIn);
6228   const auto &FnInfo =
6229       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6230   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6231   std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""});
6232   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6233                                     Name, &CGM.getModule());
6234   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6235   Fn->setDoesNotRecurse();
6236   CodeGenFunction CGF(CGM);
6237   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6238   llvm::Value *Size = nullptr;
6239   // If the size of the reduction item is non-constant, load it from global
6240   // threadprivate variable.
6241   if (RCG.getSizes(N).second) {
6242     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6243         CGF, CGM.getContext().getSizeType(),
6244         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6245     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6246                                 CGM.getContext().getSizeType(), Loc);
6247   }
6248   RCG.emitAggregateType(CGF, N, Size);
6249   // Remap lhs and rhs variables to the addresses of the function arguments.
6250   // %lhs = bitcast void* %arg0 to <type>*
6251   // %rhs = bitcast void* %arg1 to <type>*
6252   CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
6253   PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() {
6254     // Pull out the pointer to the variable.
6255     Address PtrAddr = CGF.EmitLoadOfPointer(
6256         CGF.GetAddrOfLocalVar(&ParamInOut),
6257         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6258     return CGF.Builder.CreateElementBitCast(
6259         PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType()));
6260   });
6261   PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() {
6262     // Pull out the pointer to the variable.
6263     Address PtrAddr = CGF.EmitLoadOfPointer(
6264         CGF.GetAddrOfLocalVar(&ParamIn),
6265         C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6266     return CGF.Builder.CreateElementBitCast(
6267         PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType()));
6268   });
6269   PrivateScope.Privatize();
6270   // Emit the combiner body:
6271   // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs)
6272   // store <type> %2, <type>* %lhs
6273   CGM.getOpenMPRuntime().emitSingleReductionCombiner(
6274       CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS),
6275       cast<DeclRefExpr>(RHS));
6276   CGF.FinishFunction();
6277   return Fn;
6278 }
6279 
6280 /// Emits reduction finalizer function:
6281 /// \code
6282 /// void @.red_fini(void* %arg) {
6283 /// %0 = bitcast void* %arg to <type>*
6284 /// <destroy>(<type>* %0)
6285 /// ret void
6286 /// }
6287 /// \endcode
6288 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM,
6289                                            SourceLocation Loc,
6290                                            ReductionCodeGen &RCG, unsigned N) {
6291   if (!RCG.needCleanups(N))
6292     return nullptr;
6293   ASTContext &C = CGM.getContext();
6294   FunctionArgList Args;
6295   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6296                           ImplicitParamDecl::Other);
6297   Args.emplace_back(&Param);
6298   const auto &FnInfo =
6299       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6300   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6301   std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""});
6302   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6303                                     Name, &CGM.getModule());
6304   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6305   Fn->setDoesNotRecurse();
6306   CodeGenFunction CGF(CGM);
6307   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6308   Address PrivateAddr = CGF.EmitLoadOfPointer(
6309       CGF.GetAddrOfLocalVar(&Param),
6310       C.getPointerType(C.VoidPtrTy).castAs<PointerType>());
6311   llvm::Value *Size = nullptr;
6312   // If the size of the reduction item is non-constant, load it from global
6313   // threadprivate variable.
6314   if (RCG.getSizes(N).second) {
6315     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6316         CGF, CGM.getContext().getSizeType(),
6317         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6318     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6319                                 CGM.getContext().getSizeType(), Loc);
6320   }
6321   RCG.emitAggregateType(CGF, N, Size);
6322   // Emit the finalizer body:
6323   // <destroy>(<type>* %0)
6324   RCG.emitCleanups(CGF, N, PrivateAddr);
6325   CGF.FinishFunction();
6326   return Fn;
6327 }
6328 
6329 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit(
6330     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
6331     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
6332   if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty())
6333     return nullptr;
6334 
6335   // Build typedef struct:
6336   // kmp_task_red_input {
6337   //   void *reduce_shar; // shared reduction item
6338   //   size_t reduce_size; // size of data item
6339   //   void *reduce_init; // data initialization routine
6340   //   void *reduce_fini; // data finalization routine
6341   //   void *reduce_comb; // data combiner routine
6342   //   kmp_task_red_flags_t flags; // flags for additional info from compiler
6343   // } kmp_task_red_input_t;
6344   ASTContext &C = CGM.getContext();
6345   RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t");
6346   RD->startDefinition();
6347   const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6348   const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType());
6349   const FieldDecl *InitFD  = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6350   const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6351   const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6352   const FieldDecl *FlagsFD = addFieldToRecordDecl(
6353       C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false));
6354   RD->completeDefinition();
6355   QualType RDType = C.getRecordType(RD);
6356   unsigned Size = Data.ReductionVars.size();
6357   llvm::APInt ArraySize(/*numBits=*/64, Size);
6358   QualType ArrayRDType = C.getConstantArrayType(
6359       RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
6360   // kmp_task_red_input_t .rd_input.[Size];
6361   Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input.");
6362   ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies,
6363                        Data.ReductionOps);
6364   for (unsigned Cnt = 0; Cnt < Size; ++Cnt) {
6365     // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt];
6366     llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0),
6367                            llvm::ConstantInt::get(CGM.SizeTy, Cnt)};
6368     llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP(
6369         TaskRedInput.getPointer(), Idxs,
6370         /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc,
6371         ".rd_input.gep.");
6372     LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType);
6373     // ElemLVal.reduce_shar = &Shareds[Cnt];
6374     LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD);
6375     RCG.emitSharedLValue(CGF, Cnt);
6376     llvm::Value *CastedShared =
6377         CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer());
6378     CGF.EmitStoreOfScalar(CastedShared, SharedLVal);
6379     RCG.emitAggregateType(CGF, Cnt);
6380     llvm::Value *SizeValInChars;
6381     llvm::Value *SizeVal;
6382     std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt);
6383     // We use delayed creation/initialization for VLAs, array sections and
6384     // custom reduction initializations. It is required because runtime does not
6385     // provide the way to pass the sizes of VLAs/array sections to
6386     // initializer/combiner/finalizer functions and does not pass the pointer to
6387     // original reduction item to the initializer. Instead threadprivate global
6388     // variables are used to store these values and use them in the functions.
6389     bool DelayedCreation = !!SizeVal;
6390     SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy,
6391                                                /*isSigned=*/false);
6392     LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD);
6393     CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal);
6394     // ElemLVal.reduce_init = init;
6395     LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD);
6396     llvm::Value *InitAddr =
6397         CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt));
6398     CGF.EmitStoreOfScalar(InitAddr, InitLVal);
6399     DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt);
6400     // ElemLVal.reduce_fini = fini;
6401     LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD);
6402     llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt);
6403     llvm::Value *FiniAddr = Fini
6404                                 ? CGF.EmitCastToVoidPtr(Fini)
6405                                 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy);
6406     CGF.EmitStoreOfScalar(FiniAddr, FiniLVal);
6407     // ElemLVal.reduce_comb = comb;
6408     LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD);
6409     llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction(
6410         CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt],
6411         RHSExprs[Cnt], Data.ReductionCopies[Cnt]));
6412     CGF.EmitStoreOfScalar(CombAddr, CombLVal);
6413     // ElemLVal.flags = 0;
6414     LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD);
6415     if (DelayedCreation) {
6416       CGF.EmitStoreOfScalar(
6417           llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true),
6418           FlagsLVal);
6419     } else
6420       CGF.EmitNullInitialization(FlagsLVal.getAddress(), FlagsLVal.getType());
6421   }
6422   // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void
6423   // *data);
6424   llvm::Value *Args[] = {
6425       CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy,
6426                                 /*isSigned=*/true),
6427       llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6428       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(),
6429                                                       CGM.VoidPtrTy)};
6430   return CGF.EmitRuntimeCall(
6431       createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args);
6432 }
6433 
6434 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
6435                                               SourceLocation Loc,
6436                                               ReductionCodeGen &RCG,
6437                                               unsigned N) {
6438   auto Sizes = RCG.getSizes(N);
6439   // Emit threadprivate global variable if the type is non-constant
6440   // (Sizes.second = nullptr).
6441   if (Sizes.second) {
6442     llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy,
6443                                                      /*isSigned=*/false);
6444     Address SizeAddr = getAddrOfArtificialThreadPrivate(
6445         CGF, CGM.getContext().getSizeType(),
6446         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6447     CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false);
6448   }
6449   // Store address of the original reduction item if custom initializer is used.
6450   if (RCG.usesReductionInitializer(N)) {
6451     Address SharedAddr = getAddrOfArtificialThreadPrivate(
6452         CGF, CGM.getContext().VoidPtrTy,
6453         generateUniqueName(CGM, "reduction", RCG.getRefExpr(N)));
6454     CGF.Builder.CreateStore(
6455         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6456             RCG.getSharedLValue(N).getPointer(), CGM.VoidPtrTy),
6457         SharedAddr, /*IsVolatile=*/false);
6458   }
6459 }
6460 
6461 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF,
6462                                               SourceLocation Loc,
6463                                               llvm::Value *ReductionsPtr,
6464                                               LValue SharedLVal) {
6465   // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
6466   // *d);
6467   llvm::Value *Args[] = {
6468       CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy,
6469                                 /*isSigned=*/true),
6470       ReductionsPtr,
6471       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(SharedLVal.getPointer(),
6472                                                       CGM.VoidPtrTy)};
6473   return Address(
6474       CGF.EmitRuntimeCall(
6475           createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args),
6476       SharedLVal.getAlignment());
6477 }
6478 
6479 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
6480                                        SourceLocation Loc) {
6481   if (!CGF.HaveInsertPoint())
6482     return;
6483   // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
6484   // global_tid);
6485   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
6486   // Ignore return result until untied tasks are supported.
6487   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args);
6488   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
6489     Region->emitUntiedSwitch(CGF);
6490 }
6491 
6492 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF,
6493                                            OpenMPDirectiveKind InnerKind,
6494                                            const RegionCodeGenTy &CodeGen,
6495                                            bool HasCancel) {
6496   if (!CGF.HaveInsertPoint())
6497     return;
6498   InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel);
6499   CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
6500 }
6501 
6502 namespace {
6503 enum RTCancelKind {
6504   CancelNoreq = 0,
6505   CancelParallel = 1,
6506   CancelLoop = 2,
6507   CancelSections = 3,
6508   CancelTaskgroup = 4
6509 };
6510 } // anonymous namespace
6511 
6512 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) {
6513   RTCancelKind CancelKind = CancelNoreq;
6514   if (CancelRegion == OMPD_parallel)
6515     CancelKind = CancelParallel;
6516   else if (CancelRegion == OMPD_for)
6517     CancelKind = CancelLoop;
6518   else if (CancelRegion == OMPD_sections)
6519     CancelKind = CancelSections;
6520   else {
6521     assert(CancelRegion == OMPD_taskgroup);
6522     CancelKind = CancelTaskgroup;
6523   }
6524   return CancelKind;
6525 }
6526 
6527 void CGOpenMPRuntime::emitCancellationPointCall(
6528     CodeGenFunction &CGF, SourceLocation Loc,
6529     OpenMPDirectiveKind CancelRegion) {
6530   if (!CGF.HaveInsertPoint())
6531     return;
6532   // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
6533   // global_tid, kmp_int32 cncl_kind);
6534   if (auto *OMPRegionInfo =
6535           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6536     // For 'cancellation point taskgroup', the task region info may not have a
6537     // cancel. This may instead happen in another adjacent task.
6538     if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) {
6539       llvm::Value *Args[] = {
6540           emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
6541           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6542       // Ignore return result until untied tasks are supported.
6543       llvm::Value *Result = CGF.EmitRuntimeCall(
6544           createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args);
6545       // if (__kmpc_cancellationpoint()) {
6546       //   exit from construct;
6547       // }
6548       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6549       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6550       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6551       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6552       CGF.EmitBlock(ExitBB);
6553       // exit from construct;
6554       CodeGenFunction::JumpDest CancelDest =
6555           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6556       CGF.EmitBranchThroughCleanup(CancelDest);
6557       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6558     }
6559   }
6560 }
6561 
6562 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc,
6563                                      const Expr *IfCond,
6564                                      OpenMPDirectiveKind CancelRegion) {
6565   if (!CGF.HaveInsertPoint())
6566     return;
6567   // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
6568   // kmp_int32 cncl_kind);
6569   if (auto *OMPRegionInfo =
6570           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6571     auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF,
6572                                                         PrePostActionTy &) {
6573       CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6574       llvm::Value *Args[] = {
6575           RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc),
6576           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6577       // Ignore return result until untied tasks are supported.
6578       llvm::Value *Result = CGF.EmitRuntimeCall(
6579           RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args);
6580       // if (__kmpc_cancel()) {
6581       //   exit from construct;
6582       // }
6583       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6584       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6585       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6586       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6587       CGF.EmitBlock(ExitBB);
6588       // exit from construct;
6589       CodeGenFunction::JumpDest CancelDest =
6590           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6591       CGF.EmitBranchThroughCleanup(CancelDest);
6592       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6593     };
6594     if (IfCond) {
6595       emitOMPIfClause(CGF, IfCond, ThenGen,
6596                       [](CodeGenFunction &, PrePostActionTy &) {});
6597     } else {
6598       RegionCodeGenTy ThenRCG(ThenGen);
6599       ThenRCG(CGF);
6600     }
6601   }
6602 }
6603 
6604 void CGOpenMPRuntime::emitTargetOutlinedFunction(
6605     const OMPExecutableDirective &D, StringRef ParentName,
6606     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6607     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6608   assert(!ParentName.empty() && "Invalid target region parent name!");
6609   HasEmittedTargetRegion = true;
6610   emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID,
6611                                    IsOffloadEntry, CodeGen);
6612 }
6613 
6614 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper(
6615     const OMPExecutableDirective &D, StringRef ParentName,
6616     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6617     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6618   // Create a unique name for the entry function using the source location
6619   // information of the current target region. The name will be something like:
6620   //
6621   // __omp_offloading_DD_FFFF_PP_lBB
6622   //
6623   // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the
6624   // mangled name of the function that encloses the target region and BB is the
6625   // line number of the target region.
6626 
6627   unsigned DeviceID;
6628   unsigned FileID;
6629   unsigned Line;
6630   getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID,
6631                            Line);
6632   SmallString<64> EntryFnName;
6633   {
6634     llvm::raw_svector_ostream OS(EntryFnName);
6635     OS << "__omp_offloading" << llvm::format("_%x", DeviceID)
6636        << llvm::format("_%x_", FileID) << ParentName << "_l" << Line;
6637   }
6638 
6639   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
6640 
6641   CodeGenFunction CGF(CGM, true);
6642   CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName);
6643   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6644 
6645   OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS);
6646 
6647   // If this target outline function is not an offload entry, we don't need to
6648   // register it.
6649   if (!IsOffloadEntry)
6650     return;
6651 
6652   // The target region ID is used by the runtime library to identify the current
6653   // target region, so it only has to be unique and not necessarily point to
6654   // anything. It could be the pointer to the outlined function that implements
6655   // the target region, but we aren't using that so that the compiler doesn't
6656   // need to keep that, and could therefore inline the host function if proven
6657   // worthwhile during optimization. In the other hand, if emitting code for the
6658   // device, the ID has to be the function address so that it can retrieved from
6659   // the offloading entry and launched by the runtime library. We also mark the
6660   // outlined function to have external linkage in case we are emitting code for
6661   // the device, because these functions will be entry points to the device.
6662 
6663   if (CGM.getLangOpts().OpenMPIsDevice) {
6664     OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy);
6665     OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
6666     OutlinedFn->setDSOLocal(false);
6667   } else {
6668     std::string Name = getName({EntryFnName, "region_id"});
6669     OutlinedFnID = new llvm::GlobalVariable(
6670         CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
6671         llvm::GlobalValue::WeakAnyLinkage,
6672         llvm::Constant::getNullValue(CGM.Int8Ty), Name);
6673   }
6674 
6675   // Register the information for the entry associated with this target region.
6676   OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
6677       DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID,
6678       OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion);
6679 }
6680 
6681 /// Checks if the expression is constant or does not have non-trivial function
6682 /// calls.
6683 static bool isTrivial(ASTContext &Ctx, const Expr * E) {
6684   // We can skip constant expressions.
6685   // We can skip expressions with trivial calls or simple expressions.
6686   return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) ||
6687           !E->hasNonTrivialCall(Ctx)) &&
6688          !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true);
6689 }
6690 
6691 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx,
6692                                                     const Stmt *Body) {
6693   const Stmt *Child = Body->IgnoreContainers();
6694   while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) {
6695     Child = nullptr;
6696     for (const Stmt *S : C->body()) {
6697       if (const auto *E = dyn_cast<Expr>(S)) {
6698         if (isTrivial(Ctx, E))
6699           continue;
6700       }
6701       // Some of the statements can be ignored.
6702       if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) ||
6703           isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S))
6704         continue;
6705       // Analyze declarations.
6706       if (const auto *DS = dyn_cast<DeclStmt>(S)) {
6707         if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) {
6708               if (isa<EmptyDecl>(D) || isa<DeclContext>(D) ||
6709                   isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) ||
6710                   isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) ||
6711                   isa<UsingDirectiveDecl>(D) ||
6712                   isa<OMPDeclareReductionDecl>(D) ||
6713                   isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D))
6714                 return true;
6715               const auto *VD = dyn_cast<VarDecl>(D);
6716               if (!VD)
6717                 return false;
6718               return VD->isConstexpr() ||
6719                      ((VD->getType().isTrivialType(Ctx) ||
6720                        VD->getType()->isReferenceType()) &&
6721                       (!VD->hasInit() || isTrivial(Ctx, VD->getInit())));
6722             }))
6723           continue;
6724       }
6725       // Found multiple children - cannot get the one child only.
6726       if (Child)
6727         return nullptr;
6728       Child = S;
6729     }
6730     if (Child)
6731       Child = Child->IgnoreContainers();
6732   }
6733   return Child;
6734 }
6735 
6736 /// Emit the number of teams for a target directive.  Inspect the num_teams
6737 /// clause associated with a teams construct combined or closely nested
6738 /// with the target directive.
6739 ///
6740 /// Emit a team of size one for directives such as 'target parallel' that
6741 /// have no associated teams construct.
6742 ///
6743 /// Otherwise, return nullptr.
6744 static llvm::Value *
6745 emitNumTeamsForTargetDirective(CodeGenFunction &CGF,
6746                                const OMPExecutableDirective &D) {
6747   assert(!CGF.getLangOpts().OpenMPIsDevice &&
6748          "Clauses associated with the teams directive expected to be emitted "
6749          "only for the host!");
6750   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6751   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6752          "Expected target-based executable directive.");
6753   CGBuilderTy &Bld = CGF.Builder;
6754   switch (DirectiveKind) {
6755   case OMPD_target: {
6756     const auto *CS = D.getInnermostCapturedStmt();
6757     const auto *Body =
6758         CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
6759     const Stmt *ChildStmt =
6760         CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body);
6761     if (const auto *NestedDir =
6762             dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
6763       if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) {
6764         if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) {
6765           CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6766           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6767           const Expr *NumTeams =
6768               NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6769           llvm::Value *NumTeamsVal =
6770               CGF.EmitScalarExpr(NumTeams,
6771                                  /*IgnoreResultAssign*/ true);
6772           return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6773                                    /*isSigned=*/true);
6774         }
6775         return Bld.getInt32(0);
6776       }
6777       if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) ||
6778           isOpenMPSimdDirective(NestedDir->getDirectiveKind()))
6779         return Bld.getInt32(1);
6780       return Bld.getInt32(0);
6781     }
6782     return nullptr;
6783   }
6784   case OMPD_target_teams:
6785   case OMPD_target_teams_distribute:
6786   case OMPD_target_teams_distribute_simd:
6787   case OMPD_target_teams_distribute_parallel_for:
6788   case OMPD_target_teams_distribute_parallel_for_simd: {
6789     if (D.hasClausesOfKind<OMPNumTeamsClause>()) {
6790       CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF);
6791       const Expr *NumTeams =
6792           D.getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6793       llvm::Value *NumTeamsVal =
6794           CGF.EmitScalarExpr(NumTeams,
6795                              /*IgnoreResultAssign*/ true);
6796       return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6797                                /*isSigned=*/true);
6798     }
6799     return Bld.getInt32(0);
6800   }
6801   case OMPD_target_parallel:
6802   case OMPD_target_parallel_for:
6803   case OMPD_target_parallel_for_simd:
6804   case OMPD_target_simd:
6805     return Bld.getInt32(1);
6806   case OMPD_parallel:
6807   case OMPD_for:
6808   case OMPD_parallel_for:
6809   case OMPD_parallel_sections:
6810   case OMPD_for_simd:
6811   case OMPD_parallel_for_simd:
6812   case OMPD_cancel:
6813   case OMPD_cancellation_point:
6814   case OMPD_ordered:
6815   case OMPD_threadprivate:
6816   case OMPD_allocate:
6817   case OMPD_task:
6818   case OMPD_simd:
6819   case OMPD_sections:
6820   case OMPD_section:
6821   case OMPD_single:
6822   case OMPD_master:
6823   case OMPD_critical:
6824   case OMPD_taskyield:
6825   case OMPD_barrier:
6826   case OMPD_taskwait:
6827   case OMPD_taskgroup:
6828   case OMPD_atomic:
6829   case OMPD_flush:
6830   case OMPD_teams:
6831   case OMPD_target_data:
6832   case OMPD_target_exit_data:
6833   case OMPD_target_enter_data:
6834   case OMPD_distribute:
6835   case OMPD_distribute_simd:
6836   case OMPD_distribute_parallel_for:
6837   case OMPD_distribute_parallel_for_simd:
6838   case OMPD_teams_distribute:
6839   case OMPD_teams_distribute_simd:
6840   case OMPD_teams_distribute_parallel_for:
6841   case OMPD_teams_distribute_parallel_for_simd:
6842   case OMPD_target_update:
6843   case OMPD_declare_simd:
6844   case OMPD_declare_variant:
6845   case OMPD_declare_target:
6846   case OMPD_end_declare_target:
6847   case OMPD_declare_reduction:
6848   case OMPD_declare_mapper:
6849   case OMPD_taskloop:
6850   case OMPD_taskloop_simd:
6851   case OMPD_master_taskloop:
6852   case OMPD_requires:
6853   case OMPD_unknown:
6854     break;
6855   }
6856   llvm_unreachable("Unexpected directive kind.");
6857 }
6858 
6859 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS,
6860                                   llvm::Value *DefaultThreadLimitVal) {
6861   const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6862       CGF.getContext(), CS->getCapturedStmt());
6863   if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6864     if (isOpenMPParallelDirective(Dir->getDirectiveKind())) {
6865       llvm::Value *NumThreads = nullptr;
6866       llvm::Value *CondVal = nullptr;
6867       // Handle if clause. If if clause present, the number of threads is
6868       // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
6869       if (Dir->hasClausesOfKind<OMPIfClause>()) {
6870         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6871         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6872         const OMPIfClause *IfClause = nullptr;
6873         for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) {
6874           if (C->getNameModifier() == OMPD_unknown ||
6875               C->getNameModifier() == OMPD_parallel) {
6876             IfClause = C;
6877             break;
6878           }
6879         }
6880         if (IfClause) {
6881           const Expr *Cond = IfClause->getCondition();
6882           bool Result;
6883           if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
6884             if (!Result)
6885               return CGF.Builder.getInt32(1);
6886           } else {
6887             CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange());
6888             if (const auto *PreInit =
6889                     cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) {
6890               for (const auto *I : PreInit->decls()) {
6891                 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6892                   CGF.EmitVarDecl(cast<VarDecl>(*I));
6893                 } else {
6894                   CodeGenFunction::AutoVarEmission Emission =
6895                       CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6896                   CGF.EmitAutoVarCleanups(Emission);
6897                 }
6898               }
6899             }
6900             CondVal = CGF.EvaluateExprAsBool(Cond);
6901           }
6902         }
6903       }
6904       // Check the value of num_threads clause iff if clause was not specified
6905       // or is not evaluated to false.
6906       if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) {
6907         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6908         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6909         const auto *NumThreadsClause =
6910             Dir->getSingleClause<OMPNumThreadsClause>();
6911         CodeGenFunction::LexicalScope Scope(
6912             CGF, NumThreadsClause->getNumThreads()->getSourceRange());
6913         if (const auto *PreInit =
6914                 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) {
6915           for (const auto *I : PreInit->decls()) {
6916             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6917               CGF.EmitVarDecl(cast<VarDecl>(*I));
6918             } else {
6919               CodeGenFunction::AutoVarEmission Emission =
6920                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6921               CGF.EmitAutoVarCleanups(Emission);
6922             }
6923           }
6924         }
6925         NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads());
6926         NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty,
6927                                                /*isSigned=*/false);
6928         if (DefaultThreadLimitVal)
6929           NumThreads = CGF.Builder.CreateSelect(
6930               CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads),
6931               DefaultThreadLimitVal, NumThreads);
6932       } else {
6933         NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal
6934                                            : CGF.Builder.getInt32(0);
6935       }
6936       // Process condition of the if clause.
6937       if (CondVal) {
6938         NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads,
6939                                               CGF.Builder.getInt32(1));
6940       }
6941       return NumThreads;
6942     }
6943     if (isOpenMPSimdDirective(Dir->getDirectiveKind()))
6944       return CGF.Builder.getInt32(1);
6945     return DefaultThreadLimitVal;
6946   }
6947   return DefaultThreadLimitVal ? DefaultThreadLimitVal
6948                                : CGF.Builder.getInt32(0);
6949 }
6950 
6951 /// Emit the number of threads for a target directive.  Inspect the
6952 /// thread_limit clause associated with a teams construct combined or closely
6953 /// nested with the target directive.
6954 ///
6955 /// Emit the num_threads clause for directives such as 'target parallel' that
6956 /// have no associated teams construct.
6957 ///
6958 /// Otherwise, return nullptr.
6959 static llvm::Value *
6960 emitNumThreadsForTargetDirective(CodeGenFunction &CGF,
6961                                  const OMPExecutableDirective &D) {
6962   assert(!CGF.getLangOpts().OpenMPIsDevice &&
6963          "Clauses associated with the teams directive expected to be emitted "
6964          "only for the host!");
6965   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6966   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6967          "Expected target-based executable directive.");
6968   CGBuilderTy &Bld = CGF.Builder;
6969   llvm::Value *ThreadLimitVal = nullptr;
6970   llvm::Value *NumThreadsVal = nullptr;
6971   switch (DirectiveKind) {
6972   case OMPD_target: {
6973     const CapturedStmt *CS = D.getInnermostCapturedStmt();
6974     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
6975       return NumThreads;
6976     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6977         CGF.getContext(), CS->getCapturedStmt());
6978     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6979       if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) {
6980         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6981         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6982         const auto *ThreadLimitClause =
6983             Dir->getSingleClause<OMPThreadLimitClause>();
6984         CodeGenFunction::LexicalScope Scope(
6985             CGF, ThreadLimitClause->getThreadLimit()->getSourceRange());
6986         if (const auto *PreInit =
6987                 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) {
6988           for (const auto *I : PreInit->decls()) {
6989             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6990               CGF.EmitVarDecl(cast<VarDecl>(*I));
6991             } else {
6992               CodeGenFunction::AutoVarEmission Emission =
6993                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6994               CGF.EmitAutoVarCleanups(Emission);
6995             }
6996           }
6997         }
6998         llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
6999             ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7000         ThreadLimitVal =
7001             Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7002       }
7003       if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) &&
7004           !isOpenMPDistributeDirective(Dir->getDirectiveKind())) {
7005         CS = Dir->getInnermostCapturedStmt();
7006         const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7007             CGF.getContext(), CS->getCapturedStmt());
7008         Dir = dyn_cast_or_null<OMPExecutableDirective>(Child);
7009       }
7010       if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) &&
7011           !isOpenMPSimdDirective(Dir->getDirectiveKind())) {
7012         CS = Dir->getInnermostCapturedStmt();
7013         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7014           return NumThreads;
7015       }
7016       if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind()))
7017         return Bld.getInt32(1);
7018     }
7019     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7020   }
7021   case OMPD_target_teams: {
7022     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7023       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7024       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7025       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7026           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7027       ThreadLimitVal =
7028           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7029     }
7030     const CapturedStmt *CS = D.getInnermostCapturedStmt();
7031     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7032       return NumThreads;
7033     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7034         CGF.getContext(), CS->getCapturedStmt());
7035     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7036       if (Dir->getDirectiveKind() == OMPD_distribute) {
7037         CS = Dir->getInnermostCapturedStmt();
7038         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7039           return NumThreads;
7040       }
7041     }
7042     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7043   }
7044   case OMPD_target_teams_distribute:
7045     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7046       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7047       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7048       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7049           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7050       ThreadLimitVal =
7051           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7052     }
7053     return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal);
7054   case OMPD_target_parallel:
7055   case OMPD_target_parallel_for:
7056   case OMPD_target_parallel_for_simd:
7057   case OMPD_target_teams_distribute_parallel_for:
7058   case OMPD_target_teams_distribute_parallel_for_simd: {
7059     llvm::Value *CondVal = nullptr;
7060     // Handle if clause. If if clause present, the number of threads is
7061     // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
7062     if (D.hasClausesOfKind<OMPIfClause>()) {
7063       const OMPIfClause *IfClause = nullptr;
7064       for (const auto *C : D.getClausesOfKind<OMPIfClause>()) {
7065         if (C->getNameModifier() == OMPD_unknown ||
7066             C->getNameModifier() == OMPD_parallel) {
7067           IfClause = C;
7068           break;
7069         }
7070       }
7071       if (IfClause) {
7072         const Expr *Cond = IfClause->getCondition();
7073         bool Result;
7074         if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
7075           if (!Result)
7076             return Bld.getInt32(1);
7077         } else {
7078           CodeGenFunction::RunCleanupsScope Scope(CGF);
7079           CondVal = CGF.EvaluateExprAsBool(Cond);
7080         }
7081       }
7082     }
7083     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7084       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7085       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7086       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7087           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7088       ThreadLimitVal =
7089           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7090     }
7091     if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
7092       CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
7093       const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
7094       llvm::Value *NumThreads = CGF.EmitScalarExpr(
7095           NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true);
7096       NumThreadsVal =
7097           Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false);
7098       ThreadLimitVal = ThreadLimitVal
7099                            ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal,
7100                                                                 ThreadLimitVal),
7101                                               NumThreadsVal, ThreadLimitVal)
7102                            : NumThreadsVal;
7103     }
7104     if (!ThreadLimitVal)
7105       ThreadLimitVal = Bld.getInt32(0);
7106     if (CondVal)
7107       return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1));
7108     return ThreadLimitVal;
7109   }
7110   case OMPD_target_teams_distribute_simd:
7111   case OMPD_target_simd:
7112     return Bld.getInt32(1);
7113   case OMPD_parallel:
7114   case OMPD_for:
7115   case OMPD_parallel_for:
7116   case OMPD_parallel_sections:
7117   case OMPD_for_simd:
7118   case OMPD_parallel_for_simd:
7119   case OMPD_cancel:
7120   case OMPD_cancellation_point:
7121   case OMPD_ordered:
7122   case OMPD_threadprivate:
7123   case OMPD_allocate:
7124   case OMPD_task:
7125   case OMPD_simd:
7126   case OMPD_sections:
7127   case OMPD_section:
7128   case OMPD_single:
7129   case OMPD_master:
7130   case OMPD_critical:
7131   case OMPD_taskyield:
7132   case OMPD_barrier:
7133   case OMPD_taskwait:
7134   case OMPD_taskgroup:
7135   case OMPD_atomic:
7136   case OMPD_flush:
7137   case OMPD_teams:
7138   case OMPD_target_data:
7139   case OMPD_target_exit_data:
7140   case OMPD_target_enter_data:
7141   case OMPD_distribute:
7142   case OMPD_distribute_simd:
7143   case OMPD_distribute_parallel_for:
7144   case OMPD_distribute_parallel_for_simd:
7145   case OMPD_teams_distribute:
7146   case OMPD_teams_distribute_simd:
7147   case OMPD_teams_distribute_parallel_for:
7148   case OMPD_teams_distribute_parallel_for_simd:
7149   case OMPD_target_update:
7150   case OMPD_declare_simd:
7151   case OMPD_declare_variant:
7152   case OMPD_declare_target:
7153   case OMPD_end_declare_target:
7154   case OMPD_declare_reduction:
7155   case OMPD_declare_mapper:
7156   case OMPD_taskloop:
7157   case OMPD_taskloop_simd:
7158   case OMPD_master_taskloop:
7159   case OMPD_requires:
7160   case OMPD_unknown:
7161     break;
7162   }
7163   llvm_unreachable("Unsupported directive kind.");
7164 }
7165 
7166 namespace {
7167 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
7168 
7169 // Utility to handle information from clauses associated with a given
7170 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause).
7171 // It provides a convenient interface to obtain the information and generate
7172 // code for that information.
7173 class MappableExprsHandler {
7174 public:
7175   /// Values for bit flags used to specify the mapping type for
7176   /// offloading.
7177   enum OpenMPOffloadMappingFlags : uint64_t {
7178     /// No flags
7179     OMP_MAP_NONE = 0x0,
7180     /// Allocate memory on the device and move data from host to device.
7181     OMP_MAP_TO = 0x01,
7182     /// Allocate memory on the device and move data from device to host.
7183     OMP_MAP_FROM = 0x02,
7184     /// Always perform the requested mapping action on the element, even
7185     /// if it was already mapped before.
7186     OMP_MAP_ALWAYS = 0x04,
7187     /// Delete the element from the device environment, ignoring the
7188     /// current reference count associated with the element.
7189     OMP_MAP_DELETE = 0x08,
7190     /// The element being mapped is a pointer-pointee pair; both the
7191     /// pointer and the pointee should be mapped.
7192     OMP_MAP_PTR_AND_OBJ = 0x10,
7193     /// This flags signals that the base address of an entry should be
7194     /// passed to the target kernel as an argument.
7195     OMP_MAP_TARGET_PARAM = 0x20,
7196     /// Signal that the runtime library has to return the device pointer
7197     /// in the current position for the data being mapped. Used when we have the
7198     /// use_device_ptr clause.
7199     OMP_MAP_RETURN_PARAM = 0x40,
7200     /// This flag signals that the reference being passed is a pointer to
7201     /// private data.
7202     OMP_MAP_PRIVATE = 0x80,
7203     /// Pass the element to the device by value.
7204     OMP_MAP_LITERAL = 0x100,
7205     /// Implicit map
7206     OMP_MAP_IMPLICIT = 0x200,
7207     /// Close is a hint to the runtime to allocate memory close to
7208     /// the target device.
7209     OMP_MAP_CLOSE = 0x400,
7210     /// The 16 MSBs of the flags indicate whether the entry is member of some
7211     /// struct/class.
7212     OMP_MAP_MEMBER_OF = 0xffff000000000000,
7213     LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF),
7214   };
7215 
7216   /// Get the offset of the OMP_MAP_MEMBER_OF field.
7217   static unsigned getFlagMemberOffset() {
7218     unsigned Offset = 0;
7219     for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1);
7220          Remain = Remain >> 1)
7221       Offset++;
7222     return Offset;
7223   }
7224 
7225   /// Class that associates information with a base pointer to be passed to the
7226   /// runtime library.
7227   class BasePointerInfo {
7228     /// The base pointer.
7229     llvm::Value *Ptr = nullptr;
7230     /// The base declaration that refers to this device pointer, or null if
7231     /// there is none.
7232     const ValueDecl *DevPtrDecl = nullptr;
7233 
7234   public:
7235     BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr)
7236         : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {}
7237     llvm::Value *operator*() const { return Ptr; }
7238     const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; }
7239     void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; }
7240   };
7241 
7242   using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>;
7243   using MapValuesArrayTy = SmallVector<llvm::Value *, 4>;
7244   using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>;
7245 
7246   /// Map between a struct and the its lowest & highest elements which have been
7247   /// mapped.
7248   /// [ValueDecl *] --> {LE(FieldIndex, Pointer),
7249   ///                    HE(FieldIndex, Pointer)}
7250   struct StructRangeInfoTy {
7251     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = {
7252         0, Address::invalid()};
7253     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = {
7254         0, Address::invalid()};
7255     Address Base = Address::invalid();
7256   };
7257 
7258 private:
7259   /// Kind that defines how a device pointer has to be returned.
7260   struct MapInfo {
7261     OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
7262     OpenMPMapClauseKind MapType = OMPC_MAP_unknown;
7263     ArrayRef<OpenMPMapModifierKind> MapModifiers;
7264     bool ReturnDevicePointer = false;
7265     bool IsImplicit = false;
7266 
7267     MapInfo() = default;
7268     MapInfo(
7269         OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7270         OpenMPMapClauseKind MapType,
7271         ArrayRef<OpenMPMapModifierKind> MapModifiers,
7272         bool ReturnDevicePointer, bool IsImplicit)
7273         : Components(Components), MapType(MapType), MapModifiers(MapModifiers),
7274           ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {}
7275   };
7276 
7277   /// If use_device_ptr is used on a pointer which is a struct member and there
7278   /// is no map information about it, then emission of that entry is deferred
7279   /// until the whole struct has been processed.
7280   struct DeferredDevicePtrEntryTy {
7281     const Expr *IE = nullptr;
7282     const ValueDecl *VD = nullptr;
7283 
7284     DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD)
7285         : IE(IE), VD(VD) {}
7286   };
7287 
7288   /// The target directive from where the mappable clauses were extracted. It
7289   /// is either a executable directive or a user-defined mapper directive.
7290   llvm::PointerUnion<const OMPExecutableDirective *,
7291                      const OMPDeclareMapperDecl *>
7292       CurDir;
7293 
7294   /// Function the directive is being generated for.
7295   CodeGenFunction &CGF;
7296 
7297   /// Set of all first private variables in the current directive.
7298   /// bool data is set to true if the variable is implicitly marked as
7299   /// firstprivate, false otherwise.
7300   llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls;
7301 
7302   /// Map between device pointer declarations and their expression components.
7303   /// The key value for declarations in 'this' is null.
7304   llvm::DenseMap<
7305       const ValueDecl *,
7306       SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7307       DevPointersMap;
7308 
7309   llvm::Value *getExprTypeSize(const Expr *E) const {
7310     QualType ExprTy = E->getType().getCanonicalType();
7311 
7312     // Reference types are ignored for mapping purposes.
7313     if (const auto *RefTy = ExprTy->getAs<ReferenceType>())
7314       ExprTy = RefTy->getPointeeType().getCanonicalType();
7315 
7316     // Given that an array section is considered a built-in type, we need to
7317     // do the calculation based on the length of the section instead of relying
7318     // on CGF.getTypeSize(E->getType()).
7319     if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) {
7320       QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType(
7321                             OAE->getBase()->IgnoreParenImpCasts())
7322                             .getCanonicalType();
7323 
7324       // If there is no length associated with the expression and lower bound is
7325       // not specified too, that means we are using the whole length of the
7326       // base.
7327       if (!OAE->getLength() && OAE->getColonLoc().isValid() &&
7328           !OAE->getLowerBound())
7329         return CGF.getTypeSize(BaseTy);
7330 
7331       llvm::Value *ElemSize;
7332       if (const auto *PTy = BaseTy->getAs<PointerType>()) {
7333         ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType());
7334       } else {
7335         const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr());
7336         assert(ATy && "Expecting array type if not a pointer type.");
7337         ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType());
7338       }
7339 
7340       // If we don't have a length at this point, that is because we have an
7341       // array section with a single element.
7342       if (!OAE->getLength() && OAE->getColonLoc().isInvalid())
7343         return ElemSize;
7344 
7345       if (const Expr *LenExpr = OAE->getLength()) {
7346         llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr);
7347         LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(),
7348                                              CGF.getContext().getSizeType(),
7349                                              LenExpr->getExprLoc());
7350         return CGF.Builder.CreateNUWMul(LengthVal, ElemSize);
7351       }
7352       assert(!OAE->getLength() && OAE->getColonLoc().isValid() &&
7353              OAE->getLowerBound() && "expected array_section[lb:].");
7354       // Size = sizetype - lb * elemtype;
7355       llvm::Value *LengthVal = CGF.getTypeSize(BaseTy);
7356       llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound());
7357       LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(),
7358                                        CGF.getContext().getSizeType(),
7359                                        OAE->getLowerBound()->getExprLoc());
7360       LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize);
7361       llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal);
7362       llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal);
7363       LengthVal = CGF.Builder.CreateSelect(
7364           Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0));
7365       return LengthVal;
7366     }
7367     return CGF.getTypeSize(ExprTy);
7368   }
7369 
7370   /// Return the corresponding bits for a given map clause modifier. Add
7371   /// a flag marking the map as a pointer if requested. Add a flag marking the
7372   /// map as the first one of a series of maps that relate to the same map
7373   /// expression.
7374   OpenMPOffloadMappingFlags getMapTypeBits(
7375       OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7376       bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const {
7377     OpenMPOffloadMappingFlags Bits =
7378         IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE;
7379     switch (MapType) {
7380     case OMPC_MAP_alloc:
7381     case OMPC_MAP_release:
7382       // alloc and release is the default behavior in the runtime library,  i.e.
7383       // if we don't pass any bits alloc/release that is what the runtime is
7384       // going to do. Therefore, we don't need to signal anything for these two
7385       // type modifiers.
7386       break;
7387     case OMPC_MAP_to:
7388       Bits |= OMP_MAP_TO;
7389       break;
7390     case OMPC_MAP_from:
7391       Bits |= OMP_MAP_FROM;
7392       break;
7393     case OMPC_MAP_tofrom:
7394       Bits |= OMP_MAP_TO | OMP_MAP_FROM;
7395       break;
7396     case OMPC_MAP_delete:
7397       Bits |= OMP_MAP_DELETE;
7398       break;
7399     case OMPC_MAP_unknown:
7400       llvm_unreachable("Unexpected map type!");
7401     }
7402     if (AddPtrFlag)
7403       Bits |= OMP_MAP_PTR_AND_OBJ;
7404     if (AddIsTargetParamFlag)
7405       Bits |= OMP_MAP_TARGET_PARAM;
7406     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always)
7407         != MapModifiers.end())
7408       Bits |= OMP_MAP_ALWAYS;
7409     if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close)
7410         != MapModifiers.end())
7411       Bits |= OMP_MAP_CLOSE;
7412     return Bits;
7413   }
7414 
7415   /// Return true if the provided expression is a final array section. A
7416   /// final array section, is one whose length can't be proved to be one.
7417   bool isFinalArraySectionExpression(const Expr *E) const {
7418     const auto *OASE = dyn_cast<OMPArraySectionExpr>(E);
7419 
7420     // It is not an array section and therefore not a unity-size one.
7421     if (!OASE)
7422       return false;
7423 
7424     // An array section with no colon always refer to a single element.
7425     if (OASE->getColonLoc().isInvalid())
7426       return false;
7427 
7428     const Expr *Length = OASE->getLength();
7429 
7430     // If we don't have a length we have to check if the array has size 1
7431     // for this dimension. Also, we should always expect a length if the
7432     // base type is pointer.
7433     if (!Length) {
7434       QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType(
7435                              OASE->getBase()->IgnoreParenImpCasts())
7436                              .getCanonicalType();
7437       if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr()))
7438         return ATy->getSize().getSExtValue() != 1;
7439       // If we don't have a constant dimension length, we have to consider
7440       // the current section as having any size, so it is not necessarily
7441       // unitary. If it happen to be unity size, that's user fault.
7442       return true;
7443     }
7444 
7445     // Check if the length evaluates to 1.
7446     Expr::EvalResult Result;
7447     if (!Length->EvaluateAsInt(Result, CGF.getContext()))
7448       return true; // Can have more that size 1.
7449 
7450     llvm::APSInt ConstLength = Result.Val.getInt();
7451     return ConstLength.getSExtValue() != 1;
7452   }
7453 
7454   /// Generate the base pointers, section pointers, sizes and map type
7455   /// bits for the provided map type, map modifier, and expression components.
7456   /// \a IsFirstComponent should be set to true if the provided set of
7457   /// components is the first associated with a capture.
7458   void generateInfoForComponentList(
7459       OpenMPMapClauseKind MapType,
7460       ArrayRef<OpenMPMapModifierKind> MapModifiers,
7461       OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7462       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
7463       MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
7464       StructRangeInfoTy &PartialStruct, bool IsFirstComponentList,
7465       bool IsImplicit,
7466       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
7467           OverlappedElements = llvm::None) const {
7468     // The following summarizes what has to be generated for each map and the
7469     // types below. The generated information is expressed in this order:
7470     // base pointer, section pointer, size, flags
7471     // (to add to the ones that come from the map type and modifier).
7472     //
7473     // double d;
7474     // int i[100];
7475     // float *p;
7476     //
7477     // struct S1 {
7478     //   int i;
7479     //   float f[50];
7480     // }
7481     // struct S2 {
7482     //   int i;
7483     //   float f[50];
7484     //   S1 s;
7485     //   double *p;
7486     //   struct S2 *ps;
7487     // }
7488     // S2 s;
7489     // S2 *ps;
7490     //
7491     // map(d)
7492     // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM
7493     //
7494     // map(i)
7495     // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM
7496     //
7497     // map(i[1:23])
7498     // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM
7499     //
7500     // map(p)
7501     // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
7502     //
7503     // map(p[1:24])
7504     // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM
7505     //
7506     // map(s)
7507     // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM
7508     //
7509     // map(s.i)
7510     // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM
7511     //
7512     // map(s.s.f)
7513     // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7514     //
7515     // map(s.p)
7516     // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM
7517     //
7518     // map(to: s.p[:22])
7519     // &s, &(s.p), sizeof(double*), TARGET_PARAM (*)
7520     // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**)
7521     // &(s.p), &(s.p[0]), 22*sizeof(double),
7522     //   MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
7523     // (*) alloc space for struct members, only this is a target parameter
7524     // (**) map the pointer (nothing to be mapped in this example) (the compiler
7525     //      optimizes this entry out, same in the examples below)
7526     // (***) map the pointee (map: to)
7527     //
7528     // map(s.ps)
7529     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7530     //
7531     // map(from: s.ps->s.i)
7532     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7533     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7534     // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ  | FROM
7535     //
7536     // map(to: s.ps->ps)
7537     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7538     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7539     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ  | TO
7540     //
7541     // map(s.ps->ps->ps)
7542     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7543     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7544     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7545     // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7546     //
7547     // map(to: s.ps->ps->s.f[:22])
7548     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7549     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7550     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7551     // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7552     //
7553     // map(ps)
7554     // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM
7555     //
7556     // map(ps->i)
7557     // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM
7558     //
7559     // map(ps->s.f)
7560     // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7561     //
7562     // map(from: ps->p)
7563     // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM
7564     //
7565     // map(to: ps->p[:22])
7566     // ps, &(ps->p), sizeof(double*), TARGET_PARAM
7567     // ps, &(ps->p), sizeof(double*), MEMBER_OF(1)
7568     // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO
7569     //
7570     // map(ps->ps)
7571     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7572     //
7573     // map(from: ps->ps->s.i)
7574     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7575     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7576     // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7577     //
7578     // map(from: ps->ps->ps)
7579     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7580     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7581     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7582     //
7583     // map(ps->ps->ps->ps)
7584     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7585     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7586     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7587     // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7588     //
7589     // map(to: ps->ps->ps->s.f[:22])
7590     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7591     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7592     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7593     // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7594     //
7595     // map(to: s.f[:22]) map(from: s.p[:33])
7596     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) +
7597     //     sizeof(double*) (**), TARGET_PARAM
7598     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO
7599     // &s, &(s.p), sizeof(double*), MEMBER_OF(1)
7600     // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7601     // (*) allocate contiguous space needed to fit all mapped members even if
7602     //     we allocate space for members not mapped (in this example,
7603     //     s.f[22..49] and s.s are not mapped, yet we must allocate space for
7604     //     them as well because they fall between &s.f[0] and &s.p)
7605     //
7606     // map(from: s.f[:22]) map(to: ps->p[:33])
7607     // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM
7608     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7609     // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*)
7610     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO
7611     // (*) the struct this entry pertains to is the 2nd element in the list of
7612     //     arguments, hence MEMBER_OF(2)
7613     //
7614     // map(from: s.f[:22], s.s) map(to: ps->p[:33])
7615     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM
7616     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM
7617     // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM
7618     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7619     // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*)
7620     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO
7621     // (*) the struct this entry pertains to is the 4th element in the list
7622     //     of arguments, hence MEMBER_OF(4)
7623 
7624     // Track if the map information being generated is the first for a capture.
7625     bool IsCaptureFirstInfo = IsFirstComponentList;
7626     // When the variable is on a declare target link or in a to clause with
7627     // unified memory, a reference is needed to hold the host/device address
7628     // of the variable.
7629     bool RequiresReference = false;
7630 
7631     // Scan the components from the base to the complete expression.
7632     auto CI = Components.rbegin();
7633     auto CE = Components.rend();
7634     auto I = CI;
7635 
7636     // Track if the map information being generated is the first for a list of
7637     // components.
7638     bool IsExpressionFirstInfo = true;
7639     Address BP = Address::invalid();
7640     const Expr *AssocExpr = I->getAssociatedExpression();
7641     const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr);
7642     const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr);
7643 
7644     if (isa<MemberExpr>(AssocExpr)) {
7645       // The base is the 'this' pointer. The content of the pointer is going
7646       // to be the base of the field being mapped.
7647       BP = CGF.LoadCXXThisAddress();
7648     } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) ||
7649                (OASE &&
7650                 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) {
7651       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress();
7652     } else {
7653       // The base is the reference to the variable.
7654       // BP = &Var.
7655       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress();
7656       if (const auto *VD =
7657               dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) {
7658         if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
7659                 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) {
7660           if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
7661               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
7662                CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) {
7663             RequiresReference = true;
7664             BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
7665           }
7666         }
7667       }
7668 
7669       // If the variable is a pointer and is being dereferenced (i.e. is not
7670       // the last component), the base has to be the pointer itself, not its
7671       // reference. References are ignored for mapping purposes.
7672       QualType Ty =
7673           I->getAssociatedDeclaration()->getType().getNonReferenceType();
7674       if (Ty->isAnyPointerType() && std::next(I) != CE) {
7675         BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
7676 
7677         // We do not need to generate individual map information for the
7678         // pointer, it can be associated with the combined storage.
7679         ++I;
7680       }
7681     }
7682 
7683     // Track whether a component of the list should be marked as MEMBER_OF some
7684     // combined entry (for partial structs). Only the first PTR_AND_OBJ entry
7685     // in a component list should be marked as MEMBER_OF, all subsequent entries
7686     // do not belong to the base struct. E.g.
7687     // struct S2 s;
7688     // s.ps->ps->ps->f[:]
7689     //   (1) (2) (3) (4)
7690     // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a
7691     // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3)
7692     // is the pointee of ps(2) which is not member of struct s, so it should not
7693     // be marked as such (it is still PTR_AND_OBJ).
7694     // The variable is initialized to false so that PTR_AND_OBJ entries which
7695     // are not struct members are not considered (e.g. array of pointers to
7696     // data).
7697     bool ShouldBeMemberOf = false;
7698 
7699     // Variable keeping track of whether or not we have encountered a component
7700     // in the component list which is a member expression. Useful when we have a
7701     // pointer or a final array section, in which case it is the previous
7702     // component in the list which tells us whether we have a member expression.
7703     // E.g. X.f[:]
7704     // While processing the final array section "[:]" it is "f" which tells us
7705     // whether we are dealing with a member of a declared struct.
7706     const MemberExpr *EncounteredME = nullptr;
7707 
7708     for (; I != CE; ++I) {
7709       // If the current component is member of a struct (parent struct) mark it.
7710       if (!EncounteredME) {
7711         EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression());
7712         // If we encounter a PTR_AND_OBJ entry from now on it should be marked
7713         // as MEMBER_OF the parent struct.
7714         if (EncounteredME)
7715           ShouldBeMemberOf = true;
7716       }
7717 
7718       auto Next = std::next(I);
7719 
7720       // We need to generate the addresses and sizes if this is the last
7721       // component, if the component is a pointer or if it is an array section
7722       // whose length can't be proved to be one. If this is a pointer, it
7723       // becomes the base address for the following components.
7724 
7725       // A final array section, is one whose length can't be proved to be one.
7726       bool IsFinalArraySection =
7727           isFinalArraySectionExpression(I->getAssociatedExpression());
7728 
7729       // Get information on whether the element is a pointer. Have to do a
7730       // special treatment for array sections given that they are built-in
7731       // types.
7732       const auto *OASE =
7733           dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression());
7734       bool IsPointer =
7735           (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE)
7736                        .getCanonicalType()
7737                        ->isAnyPointerType()) ||
7738           I->getAssociatedExpression()->getType()->isAnyPointerType();
7739 
7740       if (Next == CE || IsPointer || IsFinalArraySection) {
7741         // If this is not the last component, we expect the pointer to be
7742         // associated with an array expression or member expression.
7743         assert((Next == CE ||
7744                 isa<MemberExpr>(Next->getAssociatedExpression()) ||
7745                 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) ||
7746                 isa<OMPArraySectionExpr>(Next->getAssociatedExpression())) &&
7747                "Unexpected expression");
7748 
7749         Address LB =
7750             CGF.EmitOMPSharedLValue(I->getAssociatedExpression()).getAddress();
7751 
7752         // If this component is a pointer inside the base struct then we don't
7753         // need to create any entry for it - it will be combined with the object
7754         // it is pointing to into a single PTR_AND_OBJ entry.
7755         bool IsMemberPointer =
7756             IsPointer && EncounteredME &&
7757             (dyn_cast<MemberExpr>(I->getAssociatedExpression()) ==
7758              EncounteredME);
7759         if (!OverlappedElements.empty()) {
7760           // Handle base element with the info for overlapped elements.
7761           assert(!PartialStruct.Base.isValid() && "The base element is set.");
7762           assert(Next == CE &&
7763                  "Expected last element for the overlapped elements.");
7764           assert(!IsPointer &&
7765                  "Unexpected base element with the pointer type.");
7766           // Mark the whole struct as the struct that requires allocation on the
7767           // device.
7768           PartialStruct.LowestElem = {0, LB};
7769           CharUnits TypeSize = CGF.getContext().getTypeSizeInChars(
7770               I->getAssociatedExpression()->getType());
7771           Address HB = CGF.Builder.CreateConstGEP(
7772               CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB,
7773                                                               CGF.VoidPtrTy),
7774               TypeSize.getQuantity() - 1);
7775           PartialStruct.HighestElem = {
7776               std::numeric_limits<decltype(
7777                   PartialStruct.HighestElem.first)>::max(),
7778               HB};
7779           PartialStruct.Base = BP;
7780           // Emit data for non-overlapped data.
7781           OpenMPOffloadMappingFlags Flags =
7782               OMP_MAP_MEMBER_OF |
7783               getMapTypeBits(MapType, MapModifiers, IsImplicit,
7784                              /*AddPtrFlag=*/false,
7785                              /*AddIsTargetParamFlag=*/false);
7786           LB = BP;
7787           llvm::Value *Size = nullptr;
7788           // Do bitcopy of all non-overlapped structure elements.
7789           for (OMPClauseMappableExprCommon::MappableExprComponentListRef
7790                    Component : OverlappedElements) {
7791             Address ComponentLB = Address::invalid();
7792             for (const OMPClauseMappableExprCommon::MappableComponent &MC :
7793                  Component) {
7794               if (MC.getAssociatedDeclaration()) {
7795                 ComponentLB =
7796                     CGF.EmitOMPSharedLValue(MC.getAssociatedExpression())
7797                         .getAddress();
7798                 Size = CGF.Builder.CreatePtrDiff(
7799                     CGF.EmitCastToVoidPtr(ComponentLB.getPointer()),
7800                     CGF.EmitCastToVoidPtr(LB.getPointer()));
7801                 break;
7802               }
7803             }
7804             BasePointers.push_back(BP.getPointer());
7805             Pointers.push_back(LB.getPointer());
7806             Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty,
7807                                                       /*isSigned=*/true));
7808             Types.push_back(Flags);
7809             LB = CGF.Builder.CreateConstGEP(ComponentLB, 1);
7810           }
7811           BasePointers.push_back(BP.getPointer());
7812           Pointers.push_back(LB.getPointer());
7813           Size = CGF.Builder.CreatePtrDiff(
7814               CGF.EmitCastToVoidPtr(
7815                   CGF.Builder.CreateConstGEP(HB, 1).getPointer()),
7816               CGF.EmitCastToVoidPtr(LB.getPointer()));
7817           Sizes.push_back(
7818               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7819           Types.push_back(Flags);
7820           break;
7821         }
7822         llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression());
7823         if (!IsMemberPointer) {
7824           BasePointers.push_back(BP.getPointer());
7825           Pointers.push_back(LB.getPointer());
7826           Sizes.push_back(
7827               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
7828 
7829           // We need to add a pointer flag for each map that comes from the
7830           // same expression except for the first one. We also need to signal
7831           // this map is the first one that relates with the current capture
7832           // (there is a set of entries for each capture).
7833           OpenMPOffloadMappingFlags Flags = getMapTypeBits(
7834               MapType, MapModifiers, IsImplicit,
7835               !IsExpressionFirstInfo || RequiresReference,
7836               IsCaptureFirstInfo && !RequiresReference);
7837 
7838           if (!IsExpressionFirstInfo) {
7839             // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well,
7840             // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags.
7841             if (IsPointer)
7842               Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS |
7843                          OMP_MAP_DELETE | OMP_MAP_CLOSE);
7844 
7845             if (ShouldBeMemberOf) {
7846               // Set placeholder value MEMBER_OF=FFFF to indicate that the flag
7847               // should be later updated with the correct value of MEMBER_OF.
7848               Flags |= OMP_MAP_MEMBER_OF;
7849               // From now on, all subsequent PTR_AND_OBJ entries should not be
7850               // marked as MEMBER_OF.
7851               ShouldBeMemberOf = false;
7852             }
7853           }
7854 
7855           Types.push_back(Flags);
7856         }
7857 
7858         // If we have encountered a member expression so far, keep track of the
7859         // mapped member. If the parent is "*this", then the value declaration
7860         // is nullptr.
7861         if (EncounteredME) {
7862           const auto *FD = dyn_cast<FieldDecl>(EncounteredME->getMemberDecl());
7863           unsigned FieldIndex = FD->getFieldIndex();
7864 
7865           // Update info about the lowest and highest elements for this struct
7866           if (!PartialStruct.Base.isValid()) {
7867             PartialStruct.LowestElem = {FieldIndex, LB};
7868             PartialStruct.HighestElem = {FieldIndex, LB};
7869             PartialStruct.Base = BP;
7870           } else if (FieldIndex < PartialStruct.LowestElem.first) {
7871             PartialStruct.LowestElem = {FieldIndex, LB};
7872           } else if (FieldIndex > PartialStruct.HighestElem.first) {
7873             PartialStruct.HighestElem = {FieldIndex, LB};
7874           }
7875         }
7876 
7877         // If we have a final array section, we are done with this expression.
7878         if (IsFinalArraySection)
7879           break;
7880 
7881         // The pointer becomes the base for the next element.
7882         if (Next != CE)
7883           BP = LB;
7884 
7885         IsExpressionFirstInfo = false;
7886         IsCaptureFirstInfo = false;
7887       }
7888     }
7889   }
7890 
7891   /// Return the adjusted map modifiers if the declaration a capture refers to
7892   /// appears in a first-private clause. This is expected to be used only with
7893   /// directives that start with 'target'.
7894   MappableExprsHandler::OpenMPOffloadMappingFlags
7895   getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const {
7896     assert(Cap.capturesVariable() && "Expected capture by reference only!");
7897 
7898     // A first private variable captured by reference will use only the
7899     // 'private ptr' and 'map to' flag. Return the right flags if the captured
7900     // declaration is known as first-private in this handler.
7901     if (FirstPrivateDecls.count(Cap.getCapturedVar())) {
7902       if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) &&
7903           Cap.getCaptureKind() == CapturedStmt::VCK_ByRef)
7904         return MappableExprsHandler::OMP_MAP_ALWAYS |
7905                MappableExprsHandler::OMP_MAP_TO;
7906       if (Cap.getCapturedVar()->getType()->isAnyPointerType())
7907         return MappableExprsHandler::OMP_MAP_TO |
7908                MappableExprsHandler::OMP_MAP_PTR_AND_OBJ;
7909       return MappableExprsHandler::OMP_MAP_PRIVATE |
7910              MappableExprsHandler::OMP_MAP_TO;
7911     }
7912     return MappableExprsHandler::OMP_MAP_TO |
7913            MappableExprsHandler::OMP_MAP_FROM;
7914   }
7915 
7916   static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) {
7917     // Rotate by getFlagMemberOffset() bits.
7918     return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1)
7919                                                   << getFlagMemberOffset());
7920   }
7921 
7922   static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags,
7923                                      OpenMPOffloadMappingFlags MemberOfFlag) {
7924     // If the entry is PTR_AND_OBJ but has not been marked with the special
7925     // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be
7926     // marked as MEMBER_OF.
7927     if ((Flags & OMP_MAP_PTR_AND_OBJ) &&
7928         ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF))
7929       return;
7930 
7931     // Reset the placeholder value to prepare the flag for the assignment of the
7932     // proper MEMBER_OF value.
7933     Flags &= ~OMP_MAP_MEMBER_OF;
7934     Flags |= MemberOfFlag;
7935   }
7936 
7937   void getPlainLayout(const CXXRecordDecl *RD,
7938                       llvm::SmallVectorImpl<const FieldDecl *> &Layout,
7939                       bool AsBase) const {
7940     const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD);
7941 
7942     llvm::StructType *St =
7943         AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType();
7944 
7945     unsigned NumElements = St->getNumElements();
7946     llvm::SmallVector<
7947         llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4>
7948         RecordLayout(NumElements);
7949 
7950     // Fill bases.
7951     for (const auto &I : RD->bases()) {
7952       if (I.isVirtual())
7953         continue;
7954       const auto *Base = I.getType()->getAsCXXRecordDecl();
7955       // Ignore empty bases.
7956       if (Base->isEmpty() || CGF.getContext()
7957                                  .getASTRecordLayout(Base)
7958                                  .getNonVirtualSize()
7959                                  .isZero())
7960         continue;
7961 
7962       unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base);
7963       RecordLayout[FieldIndex] = Base;
7964     }
7965     // Fill in virtual bases.
7966     for (const auto &I : RD->vbases()) {
7967       const auto *Base = I.getType()->getAsCXXRecordDecl();
7968       // Ignore empty bases.
7969       if (Base->isEmpty())
7970         continue;
7971       unsigned FieldIndex = RL.getVirtualBaseIndex(Base);
7972       if (RecordLayout[FieldIndex])
7973         continue;
7974       RecordLayout[FieldIndex] = Base;
7975     }
7976     // Fill in all the fields.
7977     assert(!RD->isUnion() && "Unexpected union.");
7978     for (const auto *Field : RD->fields()) {
7979       // Fill in non-bitfields. (Bitfields always use a zero pattern, which we
7980       // will fill in later.)
7981       if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) {
7982         unsigned FieldIndex = RL.getLLVMFieldNo(Field);
7983         RecordLayout[FieldIndex] = Field;
7984       }
7985     }
7986     for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>
7987              &Data : RecordLayout) {
7988       if (Data.isNull())
7989         continue;
7990       if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>())
7991         getPlainLayout(Base, Layout, /*AsBase=*/true);
7992       else
7993         Layout.push_back(Data.get<const FieldDecl *>());
7994     }
7995   }
7996 
7997 public:
7998   MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF)
7999       : CurDir(&Dir), CGF(CGF) {
8000     // Extract firstprivate clause information.
8001     for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>())
8002       for (const auto *D : C->varlists())
8003         FirstPrivateDecls.try_emplace(
8004             cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit());
8005     // Extract device pointer clause information.
8006     for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
8007       for (auto L : C->component_lists())
8008         DevPointersMap[L.first].push_back(L.second);
8009   }
8010 
8011   /// Constructor for the declare mapper directive.
8012   MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF)
8013       : CurDir(&Dir), CGF(CGF) {}
8014 
8015   /// Generate code for the combined entry if we have a partially mapped struct
8016   /// and take care of the mapping flags of the arguments corresponding to
8017   /// individual struct members.
8018   void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers,
8019                          MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8020                          MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes,
8021                          const StructRangeInfoTy &PartialStruct) const {
8022     // Base is the base of the struct
8023     BasePointers.push_back(PartialStruct.Base.getPointer());
8024     // Pointer is the address of the lowest element
8025     llvm::Value *LB = PartialStruct.LowestElem.second.getPointer();
8026     Pointers.push_back(LB);
8027     // Size is (addr of {highest+1} element) - (addr of lowest element)
8028     llvm::Value *HB = PartialStruct.HighestElem.second.getPointer();
8029     llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1);
8030     llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy);
8031     llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy);
8032     llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr);
8033     llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty,
8034                                                   /*isSigned=*/false);
8035     Sizes.push_back(Size);
8036     // Map type is always TARGET_PARAM
8037     Types.push_back(OMP_MAP_TARGET_PARAM);
8038     // Remove TARGET_PARAM flag from the first element
8039     (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM;
8040 
8041     // All other current entries will be MEMBER_OF the combined entry
8042     // (except for PTR_AND_OBJ entries which do not have a placeholder value
8043     // 0xFFFF in the MEMBER_OF field).
8044     OpenMPOffloadMappingFlags MemberOfFlag =
8045         getMemberOfFlag(BasePointers.size() - 1);
8046     for (auto &M : CurTypes)
8047       setCorrectMemberOfFlag(M, MemberOfFlag);
8048   }
8049 
8050   /// Generate all the base pointers, section pointers, sizes and map
8051   /// types for the extracted mappable expressions. Also, for each item that
8052   /// relates with a device pointer, a pair of the relevant declaration and
8053   /// index where it occurs is appended to the device pointers info array.
8054   void generateAllInfo(MapBaseValuesArrayTy &BasePointers,
8055                        MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8056                        MapFlagsArrayTy &Types) const {
8057     // We have to process the component lists that relate with the same
8058     // declaration in a single chunk so that we can generate the map flags
8059     // correctly. Therefore, we organize all lists in a map.
8060     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
8061 
8062     // Helper function to fill the information map for the different supported
8063     // clauses.
8064     auto &&InfoGen = [&Info](
8065         const ValueDecl *D,
8066         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8067         OpenMPMapClauseKind MapType,
8068         ArrayRef<OpenMPMapModifierKind> MapModifiers,
8069         bool ReturnDevicePointer, bool IsImplicit) {
8070       const ValueDecl *VD =
8071           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
8072       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
8073                             IsImplicit);
8074     };
8075 
8076     assert(CurDir.is<const OMPExecutableDirective *>() &&
8077            "Expect a executable directive");
8078     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8079     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>())
8080       for (const auto &L : C->component_lists()) {
8081         InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(),
8082             /*ReturnDevicePointer=*/false, C->isImplicit());
8083       }
8084     for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>())
8085       for (const auto &L : C->component_lists()) {
8086         InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None,
8087             /*ReturnDevicePointer=*/false, C->isImplicit());
8088       }
8089     for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>())
8090       for (const auto &L : C->component_lists()) {
8091         InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None,
8092             /*ReturnDevicePointer=*/false, C->isImplicit());
8093       }
8094 
8095     // Look at the use_device_ptr clause information and mark the existing map
8096     // entries as such. If there is no map information for an entry in the
8097     // use_device_ptr list, we create one with map type 'alloc' and zero size
8098     // section. It is the user fault if that was not mapped before. If there is
8099     // no map information and the pointer is a struct member, then we defer the
8100     // emission of that entry until the whole struct has been processed.
8101     llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>>
8102         DeferredInfo;
8103 
8104     for (const auto *C :
8105          CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) {
8106       for (const auto &L : C->component_lists()) {
8107         assert(!L.second.empty() && "Not expecting empty list of components!");
8108         const ValueDecl *VD = L.second.back().getAssociatedDeclaration();
8109         VD = cast<ValueDecl>(VD->getCanonicalDecl());
8110         const Expr *IE = L.second.back().getAssociatedExpression();
8111         // If the first component is a member expression, we have to look into
8112         // 'this', which maps to null in the map of map information. Otherwise
8113         // look directly for the information.
8114         auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD);
8115 
8116         // We potentially have map information for this declaration already.
8117         // Look for the first set of components that refer to it.
8118         if (It != Info.end()) {
8119           auto CI = std::find_if(
8120               It->second.begin(), It->second.end(), [VD](const MapInfo &MI) {
8121                 return MI.Components.back().getAssociatedDeclaration() == VD;
8122               });
8123           // If we found a map entry, signal that the pointer has to be returned
8124           // and move on to the next declaration.
8125           if (CI != It->second.end()) {
8126             CI->ReturnDevicePointer = true;
8127             continue;
8128           }
8129         }
8130 
8131         // We didn't find any match in our map information - generate a zero
8132         // size array section - if the pointer is a struct member we defer this
8133         // action until the whole struct has been processed.
8134         if (isa<MemberExpr>(IE)) {
8135           // Insert the pointer into Info to be processed by
8136           // generateInfoForComponentList. Because it is a member pointer
8137           // without a pointee, no entry will be generated for it, therefore
8138           // we need to generate one after the whole struct has been processed.
8139           // Nonetheless, generateInfoForComponentList must be called to take
8140           // the pointer into account for the calculation of the range of the
8141           // partial struct.
8142           InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None,
8143                   /*ReturnDevicePointer=*/false, C->isImplicit());
8144           DeferredInfo[nullptr].emplace_back(IE, VD);
8145         } else {
8146           llvm::Value *Ptr =
8147               CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc());
8148           BasePointers.emplace_back(Ptr, VD);
8149           Pointers.push_back(Ptr);
8150           Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8151           Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM);
8152         }
8153       }
8154     }
8155 
8156     for (const auto &M : Info) {
8157       // We need to know when we generate information for the first component
8158       // associated with a capture, because the mapping flags depend on it.
8159       bool IsFirstComponentList = true;
8160 
8161       // Temporary versions of arrays
8162       MapBaseValuesArrayTy CurBasePointers;
8163       MapValuesArrayTy CurPointers;
8164       MapValuesArrayTy CurSizes;
8165       MapFlagsArrayTy CurTypes;
8166       StructRangeInfoTy PartialStruct;
8167 
8168       for (const MapInfo &L : M.second) {
8169         assert(!L.Components.empty() &&
8170                "Not expecting declaration with no component lists.");
8171 
8172         // Remember the current base pointer index.
8173         unsigned CurrentBasePointersIdx = CurBasePointers.size();
8174         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8175                                      CurBasePointers, CurPointers, CurSizes,
8176                                      CurTypes, PartialStruct,
8177                                      IsFirstComponentList, L.IsImplicit);
8178 
8179         // If this entry relates with a device pointer, set the relevant
8180         // declaration and add the 'return pointer' flag.
8181         if (L.ReturnDevicePointer) {
8182           assert(CurBasePointers.size() > CurrentBasePointersIdx &&
8183                  "Unexpected number of mapped base pointers.");
8184 
8185           const ValueDecl *RelevantVD =
8186               L.Components.back().getAssociatedDeclaration();
8187           assert(RelevantVD &&
8188                  "No relevant declaration related with device pointer??");
8189 
8190           CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD);
8191           CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM;
8192         }
8193         IsFirstComponentList = false;
8194       }
8195 
8196       // Append any pending zero-length pointers which are struct members and
8197       // used with use_device_ptr.
8198       auto CI = DeferredInfo.find(M.first);
8199       if (CI != DeferredInfo.end()) {
8200         for (const DeferredDevicePtrEntryTy &L : CI->second) {
8201           llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer();
8202           llvm::Value *Ptr = this->CGF.EmitLoadOfScalar(
8203               this->CGF.EmitLValue(L.IE), L.IE->getExprLoc());
8204           CurBasePointers.emplace_back(BasePtr, L.VD);
8205           CurPointers.push_back(Ptr);
8206           CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty));
8207           // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder
8208           // value MEMBER_OF=FFFF so that the entry is later updated with the
8209           // correct value of MEMBER_OF.
8210           CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM |
8211                              OMP_MAP_MEMBER_OF);
8212         }
8213       }
8214 
8215       // If there is an entry in PartialStruct it means we have a struct with
8216       // individual members mapped. Emit an extra combined entry.
8217       if (PartialStruct.Base.isValid())
8218         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8219                           PartialStruct);
8220 
8221       // We need to append the results of this capture to what we already have.
8222       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8223       Pointers.append(CurPointers.begin(), CurPointers.end());
8224       Sizes.append(CurSizes.begin(), CurSizes.end());
8225       Types.append(CurTypes.begin(), CurTypes.end());
8226     }
8227   }
8228 
8229   /// Generate all the base pointers, section pointers, sizes and map types for
8230   /// the extracted map clauses of user-defined mapper.
8231   void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers,
8232                                 MapValuesArrayTy &Pointers,
8233                                 MapValuesArrayTy &Sizes,
8234                                 MapFlagsArrayTy &Types) const {
8235     assert(CurDir.is<const OMPDeclareMapperDecl *>() &&
8236            "Expect a declare mapper directive");
8237     const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>();
8238     // We have to process the component lists that relate with the same
8239     // declaration in a single chunk so that we can generate the map flags
8240     // correctly. Therefore, we organize all lists in a map.
8241     llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info;
8242 
8243     // Helper function to fill the information map for the different supported
8244     // clauses.
8245     auto &&InfoGen = [&Info](
8246         const ValueDecl *D,
8247         OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8248         OpenMPMapClauseKind MapType,
8249         ArrayRef<OpenMPMapModifierKind> MapModifiers,
8250         bool ReturnDevicePointer, bool IsImplicit) {
8251       const ValueDecl *VD =
8252           D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr;
8253       Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer,
8254                             IsImplicit);
8255     };
8256 
8257     for (const auto *C : CurMapperDir->clauselists()) {
8258       const auto *MC = cast<OMPMapClause>(C);
8259       for (const auto &L : MC->component_lists()) {
8260         InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(),
8261                 /*ReturnDevicePointer=*/false, MC->isImplicit());
8262       }
8263     }
8264 
8265     for (const auto &M : Info) {
8266       // We need to know when we generate information for the first component
8267       // associated with a capture, because the mapping flags depend on it.
8268       bool IsFirstComponentList = true;
8269 
8270       // Temporary versions of arrays
8271       MapBaseValuesArrayTy CurBasePointers;
8272       MapValuesArrayTy CurPointers;
8273       MapValuesArrayTy CurSizes;
8274       MapFlagsArrayTy CurTypes;
8275       StructRangeInfoTy PartialStruct;
8276 
8277       for (const MapInfo &L : M.second) {
8278         assert(!L.Components.empty() &&
8279                "Not expecting declaration with no component lists.");
8280         generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components,
8281                                      CurBasePointers, CurPointers, CurSizes,
8282                                      CurTypes, PartialStruct,
8283                                      IsFirstComponentList, L.IsImplicit);
8284         IsFirstComponentList = false;
8285       }
8286 
8287       // If there is an entry in PartialStruct it means we have a struct with
8288       // individual members mapped. Emit an extra combined entry.
8289       if (PartialStruct.Base.isValid())
8290         emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes,
8291                           PartialStruct);
8292 
8293       // We need to append the results of this capture to what we already have.
8294       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
8295       Pointers.append(CurPointers.begin(), CurPointers.end());
8296       Sizes.append(CurSizes.begin(), CurSizes.end());
8297       Types.append(CurTypes.begin(), CurTypes.end());
8298     }
8299   }
8300 
8301   /// Emit capture info for lambdas for variables captured by reference.
8302   void generateInfoForLambdaCaptures(
8303       const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers,
8304       MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes,
8305       MapFlagsArrayTy &Types,
8306       llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const {
8307     const auto *RD = VD->getType()
8308                          .getCanonicalType()
8309                          .getNonReferenceType()
8310                          ->getAsCXXRecordDecl();
8311     if (!RD || !RD->isLambda())
8312       return;
8313     Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD));
8314     LValue VDLVal = CGF.MakeAddrLValue(
8315         VDAddr, VD->getType().getCanonicalType().getNonReferenceType());
8316     llvm::DenseMap<const VarDecl *, FieldDecl *> Captures;
8317     FieldDecl *ThisCapture = nullptr;
8318     RD->getCaptureFields(Captures, ThisCapture);
8319     if (ThisCapture) {
8320       LValue ThisLVal =
8321           CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture);
8322       LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture);
8323       LambdaPointers.try_emplace(ThisLVal.getPointer(), VDLVal.getPointer());
8324       BasePointers.push_back(ThisLVal.getPointer());
8325       Pointers.push_back(ThisLValVal.getPointer());
8326       Sizes.push_back(
8327           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8328                                     CGF.Int64Ty, /*isSigned=*/true));
8329       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8330                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8331     }
8332     for (const LambdaCapture &LC : RD->captures()) {
8333       if (!LC.capturesVariable())
8334         continue;
8335       const VarDecl *VD = LC.getCapturedVar();
8336       if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType())
8337         continue;
8338       auto It = Captures.find(VD);
8339       assert(It != Captures.end() && "Found lambda capture without field.");
8340       LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second);
8341       if (LC.getCaptureKind() == LCK_ByRef) {
8342         LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second);
8343         LambdaPointers.try_emplace(VarLVal.getPointer(), VDLVal.getPointer());
8344         BasePointers.push_back(VarLVal.getPointer());
8345         Pointers.push_back(VarLValVal.getPointer());
8346         Sizes.push_back(CGF.Builder.CreateIntCast(
8347             CGF.getTypeSize(
8348                 VD->getType().getCanonicalType().getNonReferenceType()),
8349             CGF.Int64Ty, /*isSigned=*/true));
8350       } else {
8351         RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation());
8352         LambdaPointers.try_emplace(VarLVal.getPointer(), VDLVal.getPointer());
8353         BasePointers.push_back(VarLVal.getPointer());
8354         Pointers.push_back(VarRVal.getScalarVal());
8355         Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0));
8356       }
8357       Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8358                       OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
8359     }
8360   }
8361 
8362   /// Set correct indices for lambdas captures.
8363   void adjustMemberOfForLambdaCaptures(
8364       const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers,
8365       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
8366       MapFlagsArrayTy &Types) const {
8367     for (unsigned I = 0, E = Types.size(); I < E; ++I) {
8368       // Set correct member_of idx for all implicit lambda captures.
8369       if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
8370                        OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT))
8371         continue;
8372       llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]);
8373       assert(BasePtr && "Unable to find base lambda address.");
8374       int TgtIdx = -1;
8375       for (unsigned J = I; J > 0; --J) {
8376         unsigned Idx = J - 1;
8377         if (Pointers[Idx] != BasePtr)
8378           continue;
8379         TgtIdx = Idx;
8380         break;
8381       }
8382       assert(TgtIdx != -1 && "Unable to find parent lambda.");
8383       // All other current entries will be MEMBER_OF the combined entry
8384       // (except for PTR_AND_OBJ entries which do not have a placeholder value
8385       // 0xFFFF in the MEMBER_OF field).
8386       OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx);
8387       setCorrectMemberOfFlag(Types[I], MemberOfFlag);
8388     }
8389   }
8390 
8391   /// Generate the base pointers, section pointers, sizes and map types
8392   /// associated to a given capture.
8393   void generateInfoForCapture(const CapturedStmt::Capture *Cap,
8394                               llvm::Value *Arg,
8395                               MapBaseValuesArrayTy &BasePointers,
8396                               MapValuesArrayTy &Pointers,
8397                               MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types,
8398                               StructRangeInfoTy &PartialStruct) const {
8399     assert(!Cap->capturesVariableArrayType() &&
8400            "Not expecting to generate map info for a variable array type!");
8401 
8402     // We need to know when we generating information for the first component
8403     const ValueDecl *VD = Cap->capturesThis()
8404                               ? nullptr
8405                               : Cap->getCapturedVar()->getCanonicalDecl();
8406 
8407     // If this declaration appears in a is_device_ptr clause we just have to
8408     // pass the pointer by value. If it is a reference to a declaration, we just
8409     // pass its value.
8410     if (DevPointersMap.count(VD)) {
8411       BasePointers.emplace_back(Arg, VD);
8412       Pointers.push_back(Arg);
8413       Sizes.push_back(
8414           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
8415                                     CGF.Int64Ty, /*isSigned=*/true));
8416       Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM);
8417       return;
8418     }
8419 
8420     using MapData =
8421         std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef,
8422                    OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>;
8423     SmallVector<MapData, 4> DeclComponentLists;
8424     assert(CurDir.is<const OMPExecutableDirective *>() &&
8425            "Expect a executable directive");
8426     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8427     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8428       for (const auto &L : C->decl_component_lists(VD)) {
8429         assert(L.first == VD &&
8430                "We got information for the wrong declaration??");
8431         assert(!L.second.empty() &&
8432                "Not expecting declaration with no component lists.");
8433         DeclComponentLists.emplace_back(L.second, C->getMapType(),
8434                                         C->getMapTypeModifiers(),
8435                                         C->isImplicit());
8436       }
8437     }
8438 
8439     // Find overlapping elements (including the offset from the base element).
8440     llvm::SmallDenseMap<
8441         const MapData *,
8442         llvm::SmallVector<
8443             OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>,
8444         4>
8445         OverlappedData;
8446     size_t Count = 0;
8447     for (const MapData &L : DeclComponentLists) {
8448       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8449       OpenMPMapClauseKind MapType;
8450       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8451       bool IsImplicit;
8452       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8453       ++Count;
8454       for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) {
8455         OMPClauseMappableExprCommon::MappableExprComponentListRef Components1;
8456         std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1;
8457         auto CI = Components.rbegin();
8458         auto CE = Components.rend();
8459         auto SI = Components1.rbegin();
8460         auto SE = Components1.rend();
8461         for (; CI != CE && SI != SE; ++CI, ++SI) {
8462           if (CI->getAssociatedExpression()->getStmtClass() !=
8463               SI->getAssociatedExpression()->getStmtClass())
8464             break;
8465           // Are we dealing with different variables/fields?
8466           if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration())
8467             break;
8468         }
8469         // Found overlapping if, at least for one component, reached the head of
8470         // the components list.
8471         if (CI == CE || SI == SE) {
8472           assert((CI != CE || SI != SE) &&
8473                  "Unexpected full match of the mapping components.");
8474           const MapData &BaseData = CI == CE ? L : L1;
8475           OMPClauseMappableExprCommon::MappableExprComponentListRef SubData =
8476               SI == SE ? Components : Components1;
8477           auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData);
8478           OverlappedElements.getSecond().push_back(SubData);
8479         }
8480       }
8481     }
8482     // Sort the overlapped elements for each item.
8483     llvm::SmallVector<const FieldDecl *, 4> Layout;
8484     if (!OverlappedData.empty()) {
8485       if (const auto *CRD =
8486               VD->getType().getCanonicalType()->getAsCXXRecordDecl())
8487         getPlainLayout(CRD, Layout, /*AsBase=*/false);
8488       else {
8489         const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl();
8490         Layout.append(RD->field_begin(), RD->field_end());
8491       }
8492     }
8493     for (auto &Pair : OverlappedData) {
8494       llvm::sort(
8495           Pair.getSecond(),
8496           [&Layout](
8497               OMPClauseMappableExprCommon::MappableExprComponentListRef First,
8498               OMPClauseMappableExprCommon::MappableExprComponentListRef
8499                   Second) {
8500             auto CI = First.rbegin();
8501             auto CE = First.rend();
8502             auto SI = Second.rbegin();
8503             auto SE = Second.rend();
8504             for (; CI != CE && SI != SE; ++CI, ++SI) {
8505               if (CI->getAssociatedExpression()->getStmtClass() !=
8506                   SI->getAssociatedExpression()->getStmtClass())
8507                 break;
8508               // Are we dealing with different variables/fields?
8509               if (CI->getAssociatedDeclaration() !=
8510                   SI->getAssociatedDeclaration())
8511                 break;
8512             }
8513 
8514             // Lists contain the same elements.
8515             if (CI == CE && SI == SE)
8516               return false;
8517 
8518             // List with less elements is less than list with more elements.
8519             if (CI == CE || SI == SE)
8520               return CI == CE;
8521 
8522             const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration());
8523             const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration());
8524             if (FD1->getParent() == FD2->getParent())
8525               return FD1->getFieldIndex() < FD2->getFieldIndex();
8526             const auto It =
8527                 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) {
8528                   return FD == FD1 || FD == FD2;
8529                 });
8530             return *It == FD1;
8531           });
8532     }
8533 
8534     // Associated with a capture, because the mapping flags depend on it.
8535     // Go through all of the elements with the overlapped elements.
8536     for (const auto &Pair : OverlappedData) {
8537       const MapData &L = *Pair.getFirst();
8538       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8539       OpenMPMapClauseKind MapType;
8540       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8541       bool IsImplicit;
8542       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8543       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
8544           OverlappedComponents = Pair.getSecond();
8545       bool IsFirstComponentList = true;
8546       generateInfoForComponentList(MapType, MapModifiers, Components,
8547                                    BasePointers, Pointers, Sizes, Types,
8548                                    PartialStruct, IsFirstComponentList,
8549                                    IsImplicit, OverlappedComponents);
8550     }
8551     // Go through other elements without overlapped elements.
8552     bool IsFirstComponentList = OverlappedData.empty();
8553     for (const MapData &L : DeclComponentLists) {
8554       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
8555       OpenMPMapClauseKind MapType;
8556       ArrayRef<OpenMPMapModifierKind> MapModifiers;
8557       bool IsImplicit;
8558       std::tie(Components, MapType, MapModifiers, IsImplicit) = L;
8559       auto It = OverlappedData.find(&L);
8560       if (It == OverlappedData.end())
8561         generateInfoForComponentList(MapType, MapModifiers, Components,
8562                                      BasePointers, Pointers, Sizes, Types,
8563                                      PartialStruct, IsFirstComponentList,
8564                                      IsImplicit);
8565       IsFirstComponentList = false;
8566     }
8567   }
8568 
8569   /// Generate the base pointers, section pointers, sizes and map types
8570   /// associated with the declare target link variables.
8571   void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers,
8572                                         MapValuesArrayTy &Pointers,
8573                                         MapValuesArrayTy &Sizes,
8574                                         MapFlagsArrayTy &Types) const {
8575     assert(CurDir.is<const OMPExecutableDirective *>() &&
8576            "Expect a executable directive");
8577     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
8578     // Map other list items in the map clause which are not captured variables
8579     // but "declare target link" global variables.
8580     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
8581       for (const auto &L : C->component_lists()) {
8582         if (!L.first)
8583           continue;
8584         const auto *VD = dyn_cast<VarDecl>(L.first);
8585         if (!VD)
8586           continue;
8587         llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
8588             OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
8589         if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
8590             !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link)
8591           continue;
8592         StructRangeInfoTy PartialStruct;
8593         generateInfoForComponentList(
8594             C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers,
8595             Pointers, Sizes, Types, PartialStruct,
8596             /*IsFirstComponentList=*/true, C->isImplicit());
8597         assert(!PartialStruct.Base.isValid() &&
8598                "No partial structs for declare target link expected.");
8599       }
8600     }
8601   }
8602 
8603   /// Generate the default map information for a given capture \a CI,
8604   /// record field declaration \a RI and captured value \a CV.
8605   void generateDefaultMapInfo(const CapturedStmt::Capture &CI,
8606                               const FieldDecl &RI, llvm::Value *CV,
8607                               MapBaseValuesArrayTy &CurBasePointers,
8608                               MapValuesArrayTy &CurPointers,
8609                               MapValuesArrayTy &CurSizes,
8610                               MapFlagsArrayTy &CurMapTypes) const {
8611     bool IsImplicit = true;
8612     // Do the default mapping.
8613     if (CI.capturesThis()) {
8614       CurBasePointers.push_back(CV);
8615       CurPointers.push_back(CV);
8616       const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr());
8617       CurSizes.push_back(
8618           CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()),
8619                                     CGF.Int64Ty, /*isSigned=*/true));
8620       // Default map type.
8621       CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM);
8622     } else if (CI.capturesVariableByCopy()) {
8623       CurBasePointers.push_back(CV);
8624       CurPointers.push_back(CV);
8625       if (!RI.getType()->isAnyPointerType()) {
8626         // We have to signal to the runtime captures passed by value that are
8627         // not pointers.
8628         CurMapTypes.push_back(OMP_MAP_LITERAL);
8629         CurSizes.push_back(CGF.Builder.CreateIntCast(
8630             CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true));
8631       } else {
8632         // Pointers are implicitly mapped with a zero size and no flags
8633         // (other than first map that is added for all implicit maps).
8634         CurMapTypes.push_back(OMP_MAP_NONE);
8635         CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
8636       }
8637       const VarDecl *VD = CI.getCapturedVar();
8638       auto I = FirstPrivateDecls.find(VD);
8639       if (I != FirstPrivateDecls.end())
8640         IsImplicit = I->getSecond();
8641     } else {
8642       assert(CI.capturesVariable() && "Expected captured reference.");
8643       const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr());
8644       QualType ElementType = PtrTy->getPointeeType();
8645       CurSizes.push_back(CGF.Builder.CreateIntCast(
8646           CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true));
8647       // The default map type for a scalar/complex type is 'to' because by
8648       // default the value doesn't have to be retrieved. For an aggregate
8649       // type, the default is 'tofrom'.
8650       CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI));
8651       const VarDecl *VD = CI.getCapturedVar();
8652       auto I = FirstPrivateDecls.find(VD);
8653       if (I != FirstPrivateDecls.end() &&
8654           VD->getType().isConstant(CGF.getContext())) {
8655         llvm::Constant *Addr =
8656             CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD);
8657         // Copy the value of the original variable to the new global copy.
8658         CGF.Builder.CreateMemCpy(
8659             CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(),
8660             Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)),
8661             CurSizes.back(), /*IsVolatile=*/false);
8662         // Use new global variable as the base pointers.
8663         CurBasePointers.push_back(Addr);
8664         CurPointers.push_back(Addr);
8665       } else {
8666         CurBasePointers.push_back(CV);
8667         if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) {
8668           Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue(
8669               CV, ElementType, CGF.getContext().getDeclAlign(VD),
8670               AlignmentSource::Decl));
8671           CurPointers.push_back(PtrAddr.getPointer());
8672         } else {
8673           CurPointers.push_back(CV);
8674         }
8675       }
8676       if (I != FirstPrivateDecls.end())
8677         IsImplicit = I->getSecond();
8678     }
8679     // Every default map produces a single argument which is a target parameter.
8680     CurMapTypes.back() |= OMP_MAP_TARGET_PARAM;
8681 
8682     // Add flag stating this is an implicit map.
8683     if (IsImplicit)
8684       CurMapTypes.back() |= OMP_MAP_IMPLICIT;
8685   }
8686 };
8687 } // anonymous namespace
8688 
8689 /// Emit the arrays used to pass the captures and map information to the
8690 /// offloading runtime library. If there is no map or capture information,
8691 /// return nullptr by reference.
8692 static void
8693 emitOffloadingArrays(CodeGenFunction &CGF,
8694                      MappableExprsHandler::MapBaseValuesArrayTy &BasePointers,
8695                      MappableExprsHandler::MapValuesArrayTy &Pointers,
8696                      MappableExprsHandler::MapValuesArrayTy &Sizes,
8697                      MappableExprsHandler::MapFlagsArrayTy &MapTypes,
8698                      CGOpenMPRuntime::TargetDataInfo &Info) {
8699   CodeGenModule &CGM = CGF.CGM;
8700   ASTContext &Ctx = CGF.getContext();
8701 
8702   // Reset the array information.
8703   Info.clearArrayInfo();
8704   Info.NumberOfPtrs = BasePointers.size();
8705 
8706   if (Info.NumberOfPtrs) {
8707     // Detect if we have any capture size requiring runtime evaluation of the
8708     // size so that a constant array could be eventually used.
8709     bool hasRuntimeEvaluationCaptureSize = false;
8710     for (llvm::Value *S : Sizes)
8711       if (!isa<llvm::Constant>(S)) {
8712         hasRuntimeEvaluationCaptureSize = true;
8713         break;
8714       }
8715 
8716     llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true);
8717     QualType PointerArrayType = Ctx.getConstantArrayType(
8718         Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal,
8719         /*IndexTypeQuals=*/0);
8720 
8721     Info.BasePointersArray =
8722         CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer();
8723     Info.PointersArray =
8724         CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer();
8725 
8726     // If we don't have any VLA types or other types that require runtime
8727     // evaluation, we can use a constant array for the map sizes, otherwise we
8728     // need to fill up the arrays as we do for the pointers.
8729     QualType Int64Ty =
8730         Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
8731     if (hasRuntimeEvaluationCaptureSize) {
8732       QualType SizeArrayType = Ctx.getConstantArrayType(
8733           Int64Ty, PointerNumAP, nullptr, ArrayType::Normal,
8734           /*IndexTypeQuals=*/0);
8735       Info.SizesArray =
8736           CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer();
8737     } else {
8738       // We expect all the sizes to be constant, so we collect them to create
8739       // a constant array.
8740       SmallVector<llvm::Constant *, 16> ConstSizes;
8741       for (llvm::Value *S : Sizes)
8742         ConstSizes.push_back(cast<llvm::Constant>(S));
8743 
8744       auto *SizesArrayInit = llvm::ConstantArray::get(
8745           llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes);
8746       std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"});
8747       auto *SizesArrayGbl = new llvm::GlobalVariable(
8748           CGM.getModule(), SizesArrayInit->getType(),
8749           /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8750           SizesArrayInit, Name);
8751       SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8752       Info.SizesArray = SizesArrayGbl;
8753     }
8754 
8755     // The map types are always constant so we don't need to generate code to
8756     // fill arrays. Instead, we create an array constant.
8757     SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0);
8758     llvm::copy(MapTypes, Mapping.begin());
8759     llvm::Constant *MapTypesArrayInit =
8760         llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping);
8761     std::string MaptypesName =
8762         CGM.getOpenMPRuntime().getName({"offload_maptypes"});
8763     auto *MapTypesArrayGbl = new llvm::GlobalVariable(
8764         CGM.getModule(), MapTypesArrayInit->getType(),
8765         /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage,
8766         MapTypesArrayInit, MaptypesName);
8767     MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
8768     Info.MapTypesArray = MapTypesArrayGbl;
8769 
8770     for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) {
8771       llvm::Value *BPVal = *BasePointers[I];
8772       llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32(
8773           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8774           Info.BasePointersArray, 0, I);
8775       BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8776           BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0));
8777       Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8778       CGF.Builder.CreateStore(BPVal, BPAddr);
8779 
8780       if (Info.requiresDevicePointerInfo())
8781         if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl())
8782           Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr);
8783 
8784       llvm::Value *PVal = Pointers[I];
8785       llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32(
8786           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8787           Info.PointersArray, 0, I);
8788       P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8789           P, PVal->getType()->getPointerTo(/*AddrSpace=*/0));
8790       Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
8791       CGF.Builder.CreateStore(PVal, PAddr);
8792 
8793       if (hasRuntimeEvaluationCaptureSize) {
8794         llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32(
8795             llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8796             Info.SizesArray,
8797             /*Idx0=*/0,
8798             /*Idx1=*/I);
8799         Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty));
8800         CGF.Builder.CreateStore(
8801             CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true),
8802             SAddr);
8803       }
8804     }
8805   }
8806 }
8807 
8808 /// Emit the arguments to be passed to the runtime library based on the
8809 /// arrays of pointers, sizes and map types.
8810 static void emitOffloadingArraysArgument(
8811     CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg,
8812     llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg,
8813     llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) {
8814   CodeGenModule &CGM = CGF.CGM;
8815   if (Info.NumberOfPtrs) {
8816     BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8817         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8818         Info.BasePointersArray,
8819         /*Idx0=*/0, /*Idx1=*/0);
8820     PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8821         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
8822         Info.PointersArray,
8823         /*Idx0=*/0,
8824         /*Idx1=*/0);
8825     SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8826         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray,
8827         /*Idx0=*/0, /*Idx1=*/0);
8828     MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
8829         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
8830         Info.MapTypesArray,
8831         /*Idx0=*/0,
8832         /*Idx1=*/0);
8833   } else {
8834     BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8835     PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
8836     SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8837     MapTypesArrayArg =
8838         llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
8839   }
8840 }
8841 
8842 /// Check for inner distribute directive.
8843 static const OMPExecutableDirective *
8844 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) {
8845   const auto *CS = D.getInnermostCapturedStmt();
8846   const auto *Body =
8847       CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
8848   const Stmt *ChildStmt =
8849       CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
8850 
8851   if (const auto *NestedDir =
8852           dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
8853     OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind();
8854     switch (D.getDirectiveKind()) {
8855     case OMPD_target:
8856       if (isOpenMPDistributeDirective(DKind))
8857         return NestedDir;
8858       if (DKind == OMPD_teams) {
8859         Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers(
8860             /*IgnoreCaptured=*/true);
8861         if (!Body)
8862           return nullptr;
8863         ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
8864         if (const auto *NND =
8865                 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
8866           DKind = NND->getDirectiveKind();
8867           if (isOpenMPDistributeDirective(DKind))
8868             return NND;
8869         }
8870       }
8871       return nullptr;
8872     case OMPD_target_teams:
8873       if (isOpenMPDistributeDirective(DKind))
8874         return NestedDir;
8875       return nullptr;
8876     case OMPD_target_parallel:
8877     case OMPD_target_simd:
8878     case OMPD_target_parallel_for:
8879     case OMPD_target_parallel_for_simd:
8880       return nullptr;
8881     case OMPD_target_teams_distribute:
8882     case OMPD_target_teams_distribute_simd:
8883     case OMPD_target_teams_distribute_parallel_for:
8884     case OMPD_target_teams_distribute_parallel_for_simd:
8885     case OMPD_parallel:
8886     case OMPD_for:
8887     case OMPD_parallel_for:
8888     case OMPD_parallel_sections:
8889     case OMPD_for_simd:
8890     case OMPD_parallel_for_simd:
8891     case OMPD_cancel:
8892     case OMPD_cancellation_point:
8893     case OMPD_ordered:
8894     case OMPD_threadprivate:
8895     case OMPD_allocate:
8896     case OMPD_task:
8897     case OMPD_simd:
8898     case OMPD_sections:
8899     case OMPD_section:
8900     case OMPD_single:
8901     case OMPD_master:
8902     case OMPD_critical:
8903     case OMPD_taskyield:
8904     case OMPD_barrier:
8905     case OMPD_taskwait:
8906     case OMPD_taskgroup:
8907     case OMPD_atomic:
8908     case OMPD_flush:
8909     case OMPD_teams:
8910     case OMPD_target_data:
8911     case OMPD_target_exit_data:
8912     case OMPD_target_enter_data:
8913     case OMPD_distribute:
8914     case OMPD_distribute_simd:
8915     case OMPD_distribute_parallel_for:
8916     case OMPD_distribute_parallel_for_simd:
8917     case OMPD_teams_distribute:
8918     case OMPD_teams_distribute_simd:
8919     case OMPD_teams_distribute_parallel_for:
8920     case OMPD_teams_distribute_parallel_for_simd:
8921     case OMPD_target_update:
8922     case OMPD_declare_simd:
8923     case OMPD_declare_variant:
8924     case OMPD_declare_target:
8925     case OMPD_end_declare_target:
8926     case OMPD_declare_reduction:
8927     case OMPD_declare_mapper:
8928     case OMPD_taskloop:
8929     case OMPD_taskloop_simd:
8930     case OMPD_master_taskloop:
8931     case OMPD_requires:
8932     case OMPD_unknown:
8933       llvm_unreachable("Unexpected directive.");
8934     }
8935   }
8936 
8937   return nullptr;
8938 }
8939 
8940 /// Emit the user-defined mapper function. The code generation follows the
8941 /// pattern in the example below.
8942 /// \code
8943 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle,
8944 ///                                           void *base, void *begin,
8945 ///                                           int64_t size, int64_t type) {
8946 ///   // Allocate space for an array section first.
8947 ///   if (size > 1 && !maptype.IsDelete)
8948 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
8949 ///                                 size*sizeof(Ty), clearToFrom(type));
8950 ///   // Map members.
8951 ///   for (unsigned i = 0; i < size; i++) {
8952 ///     // For each component specified by this mapper:
8953 ///     for (auto c : all_components) {
8954 ///       if (c.hasMapper())
8955 ///         (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size,
8956 ///                       c.arg_type);
8957 ///       else
8958 ///         __tgt_push_mapper_component(rt_mapper_handle, c.arg_base,
8959 ///                                     c.arg_begin, c.arg_size, c.arg_type);
8960 ///     }
8961 ///   }
8962 ///   // Delete the array section.
8963 ///   if (size > 1 && maptype.IsDelete)
8964 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
8965 ///                                 size*sizeof(Ty), clearToFrom(type));
8966 /// }
8967 /// \endcode
8968 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D,
8969                                             CodeGenFunction *CGF) {
8970   if (UDMMap.count(D) > 0)
8971     return;
8972   ASTContext &C = CGM.getContext();
8973   QualType Ty = D->getType();
8974   QualType PtrTy = C.getPointerType(Ty).withRestrict();
8975   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
8976   auto *MapperVarDecl =
8977       cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl());
8978   SourceLocation Loc = D->getLocation();
8979   CharUnits ElementSize = C.getTypeSizeInChars(Ty);
8980 
8981   // Prepare mapper function arguments and attributes.
8982   ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
8983                               C.VoidPtrTy, ImplicitParamDecl::Other);
8984   ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
8985                             ImplicitParamDecl::Other);
8986   ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
8987                              C.VoidPtrTy, ImplicitParamDecl::Other);
8988   ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
8989                             ImplicitParamDecl::Other);
8990   ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
8991                             ImplicitParamDecl::Other);
8992   FunctionArgList Args;
8993   Args.push_back(&HandleArg);
8994   Args.push_back(&BaseArg);
8995   Args.push_back(&BeginArg);
8996   Args.push_back(&SizeArg);
8997   Args.push_back(&TypeArg);
8998   const CGFunctionInfo &FnInfo =
8999       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
9000   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
9001   SmallString<64> TyStr;
9002   llvm::raw_svector_ostream Out(TyStr);
9003   CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out);
9004   std::string Name = getName({"omp_mapper", TyStr, D->getName()});
9005   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
9006                                     Name, &CGM.getModule());
9007   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
9008   Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
9009   // Start the mapper function code generation.
9010   CodeGenFunction MapperCGF(CGM);
9011   MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
9012   // Compute the starting and end addreses of array elements.
9013   llvm::Value *Size = MapperCGF.EmitLoadOfScalar(
9014       MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false,
9015       C.getPointerType(Int64Ty), Loc);
9016   llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast(
9017       MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(),
9018       CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy)));
9019   llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size);
9020   llvm::Value *MapType = MapperCGF.EmitLoadOfScalar(
9021       MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false,
9022       C.getPointerType(Int64Ty), Loc);
9023   // Prepare common arguments for array initiation and deletion.
9024   llvm::Value *Handle = MapperCGF.EmitLoadOfScalar(
9025       MapperCGF.GetAddrOfLocalVar(&HandleArg),
9026       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9027   llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar(
9028       MapperCGF.GetAddrOfLocalVar(&BaseArg),
9029       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9030   llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar(
9031       MapperCGF.GetAddrOfLocalVar(&BeginArg),
9032       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
9033 
9034   // Emit array initiation if this is an array section and \p MapType indicates
9035   // that memory allocation is required.
9036   llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head");
9037   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
9038                              ElementSize, HeadBB, /*IsInit=*/true);
9039 
9040   // Emit a for loop to iterate through SizeArg of elements and map all of them.
9041 
9042   // Emit the loop header block.
9043   MapperCGF.EmitBlock(HeadBB);
9044   llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body");
9045   llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done");
9046   // Evaluate whether the initial condition is satisfied.
9047   llvm::Value *IsEmpty =
9048       MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty");
9049   MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
9050   llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock();
9051 
9052   // Emit the loop body block.
9053   MapperCGF.EmitBlock(BodyBB);
9054   llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI(
9055       PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent");
9056   PtrPHI->addIncoming(PtrBegin, EntryBB);
9057   Address PtrCurrent =
9058       Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg)
9059                           .getAlignment()
9060                           .alignmentOfArrayElement(ElementSize));
9061   // Privatize the declared variable of mapper to be the current array element.
9062   CodeGenFunction::OMPPrivateScope Scope(MapperCGF);
9063   Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() {
9064     return MapperCGF
9065         .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>())
9066         .getAddress();
9067   });
9068   (void)Scope.Privatize();
9069 
9070   // Get map clause information. Fill up the arrays with all mapped variables.
9071   MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9072   MappableExprsHandler::MapValuesArrayTy Pointers;
9073   MappableExprsHandler::MapValuesArrayTy Sizes;
9074   MappableExprsHandler::MapFlagsArrayTy MapTypes;
9075   MappableExprsHandler MEHandler(*D, MapperCGF);
9076   MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes);
9077 
9078   // Call the runtime API __tgt_mapper_num_components to get the number of
9079   // pre-existing components.
9080   llvm::Value *OffloadingArgs[] = {Handle};
9081   llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall(
9082       createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs);
9083   llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl(
9084       PreviousSize,
9085       MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset()));
9086 
9087   // Fill up the runtime mapper handle for all components.
9088   for (unsigned I = 0; I < BasePointers.size(); ++I) {
9089     llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast(
9090         *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
9091     llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast(
9092         Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
9093     llvm::Value *CurSizeArg = Sizes[I];
9094 
9095     // Extract the MEMBER_OF field from the map type.
9096     llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member");
9097     MapperCGF.EmitBlock(MemberBB);
9098     llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]);
9099     llvm::Value *Member = MapperCGF.Builder.CreateAnd(
9100         OriMapType,
9101         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF));
9102     llvm::BasicBlock *MemberCombineBB =
9103         MapperCGF.createBasicBlock("omp.member.combine");
9104     llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type");
9105     llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member);
9106     MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB);
9107     // Add the number of pre-existing components to the MEMBER_OF field if it
9108     // is valid.
9109     MapperCGF.EmitBlock(MemberCombineBB);
9110     llvm::Value *CombinedMember =
9111         MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
9112     // Do nothing if it is not a member of previous components.
9113     MapperCGF.EmitBlock(TypeBB);
9114     llvm::PHINode *MemberMapType =
9115         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype");
9116     MemberMapType->addIncoming(OriMapType, MemberBB);
9117     MemberMapType->addIncoming(CombinedMember, MemberCombineBB);
9118 
9119     // Combine the map type inherited from user-defined mapper with that
9120     // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM
9121     // bits of the \a MapType, which is the input argument of the mapper
9122     // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM
9123     // bits of MemberMapType.
9124     // [OpenMP 5.0], 1.2.6. map-type decay.
9125     //        | alloc |  to   | from  | tofrom | release | delete
9126     // ----------------------------------------------------------
9127     // alloc  | alloc | alloc | alloc | alloc  | release | delete
9128     // to     | alloc |  to   | alloc |   to   | release | delete
9129     // from   | alloc | alloc | from  |  from  | release | delete
9130     // tofrom | alloc |  to   | from  | tofrom | release | delete
9131     llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd(
9132         MapType,
9133         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO |
9134                                    MappableExprsHandler::OMP_MAP_FROM));
9135     llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc");
9136     llvm::BasicBlock *AllocElseBB =
9137         MapperCGF.createBasicBlock("omp.type.alloc.else");
9138     llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to");
9139     llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else");
9140     llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from");
9141     llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end");
9142     llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom);
9143     MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
9144     // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM.
9145     MapperCGF.EmitBlock(AllocBB);
9146     llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd(
9147         MemberMapType,
9148         MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9149                                      MappableExprsHandler::OMP_MAP_FROM)));
9150     MapperCGF.Builder.CreateBr(EndBB);
9151     MapperCGF.EmitBlock(AllocElseBB);
9152     llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ(
9153         LeftToFrom,
9154         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO));
9155     MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
9156     // In case of to, clear OMP_MAP_FROM.
9157     MapperCGF.EmitBlock(ToBB);
9158     llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd(
9159         MemberMapType,
9160         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM));
9161     MapperCGF.Builder.CreateBr(EndBB);
9162     MapperCGF.EmitBlock(ToElseBB);
9163     llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ(
9164         LeftToFrom,
9165         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM));
9166     MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB);
9167     // In case of from, clear OMP_MAP_TO.
9168     MapperCGF.EmitBlock(FromBB);
9169     llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd(
9170         MemberMapType,
9171         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO));
9172     // In case of tofrom, do nothing.
9173     MapperCGF.EmitBlock(EndBB);
9174     llvm::PHINode *CurMapType =
9175         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype");
9176     CurMapType->addIncoming(AllocMapType, AllocBB);
9177     CurMapType->addIncoming(ToMapType, ToBB);
9178     CurMapType->addIncoming(FromMapType, FromBB);
9179     CurMapType->addIncoming(MemberMapType, ToElseBB);
9180 
9181     // TODO: call the corresponding mapper function if a user-defined mapper is
9182     // associated with this map clause.
9183     // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9184     // data structure.
9185     llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg,
9186                                      CurSizeArg, CurMapType};
9187     MapperCGF.EmitRuntimeCall(
9188         createRuntimeFunction(OMPRTL__tgt_push_mapper_component),
9189         OffloadingArgs);
9190   }
9191 
9192   // Update the pointer to point to the next element that needs to be mapped,
9193   // and check whether we have mapped all elements.
9194   llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32(
9195       PtrPHI, /*Idx0=*/1, "omp.arraymap.next");
9196   PtrPHI->addIncoming(PtrNext, BodyBB);
9197   llvm::Value *IsDone =
9198       MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone");
9199   llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit");
9200   MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
9201 
9202   MapperCGF.EmitBlock(ExitBB);
9203   // Emit array deletion if this is an array section and \p MapType indicates
9204   // that deletion is required.
9205   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
9206                              ElementSize, DoneBB, /*IsInit=*/false);
9207 
9208   // Emit the function exit block.
9209   MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true);
9210   MapperCGF.FinishFunction();
9211   UDMMap.try_emplace(D, Fn);
9212   if (CGF) {
9213     auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn);
9214     Decls.second.push_back(D);
9215   }
9216 }
9217 
9218 /// Emit the array initialization or deletion portion for user-defined mapper
9219 /// code generation. First, it evaluates whether an array section is mapped and
9220 /// whether the \a MapType instructs to delete this section. If \a IsInit is
9221 /// true, and \a MapType indicates to not delete this array, array
9222 /// initialization code is generated. If \a IsInit is false, and \a MapType
9223 /// indicates to not this array, array deletion code is generated.
9224 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel(
9225     CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base,
9226     llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType,
9227     CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) {
9228   StringRef Prefix = IsInit ? ".init" : ".del";
9229 
9230   // Evaluate if this is an array section.
9231   llvm::BasicBlock *IsDeleteBB =
9232       MapperCGF.createBasicBlock("omp.array" + Prefix + ".evaldelete");
9233   llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.array" + Prefix);
9234   llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE(
9235       Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray");
9236   MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB);
9237 
9238   // Evaluate if we are going to delete this section.
9239   MapperCGF.EmitBlock(IsDeleteBB);
9240   llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd(
9241       MapType,
9242       MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE));
9243   llvm::Value *DeleteCond;
9244   if (IsInit) {
9245     DeleteCond = MapperCGF.Builder.CreateIsNull(
9246         DeleteBit, "omp.array" + Prefix + ".delete");
9247   } else {
9248     DeleteCond = MapperCGF.Builder.CreateIsNotNull(
9249         DeleteBit, "omp.array" + Prefix + ".delete");
9250   }
9251   MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB);
9252 
9253   MapperCGF.EmitBlock(BodyBB);
9254   // Get the array size by multiplying element size and element number (i.e., \p
9255   // Size).
9256   llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul(
9257       Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity()));
9258   // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves
9259   // memory allocation/deletion purpose only.
9260   llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd(
9261       MapType,
9262       MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
9263                                    MappableExprsHandler::OMP_MAP_FROM)));
9264   // Call the runtime API __tgt_push_mapper_component to fill up the runtime
9265   // data structure.
9266   llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg};
9267   MapperCGF.EmitRuntimeCall(
9268       createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs);
9269 }
9270 
9271 void CGOpenMPRuntime::emitTargetNumIterationsCall(
9272     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9273     llvm::Value *DeviceID,
9274     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9275                                      const OMPLoopDirective &D)>
9276         SizeEmitter) {
9277   OpenMPDirectiveKind Kind = D.getDirectiveKind();
9278   const OMPExecutableDirective *TD = &D;
9279   // Get nested teams distribute kind directive, if any.
9280   if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind))
9281     TD = getNestedDistributeDirective(CGM.getContext(), D);
9282   if (!TD)
9283     return;
9284   const auto *LD = cast<OMPLoopDirective>(TD);
9285   auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF,
9286                                                      PrePostActionTy &) {
9287     if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) {
9288       llvm::Value *Args[] = {DeviceID, NumIterations};
9289       CGF.EmitRuntimeCall(
9290           createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args);
9291     }
9292   };
9293   emitInlinedDirective(CGF, OMPD_unknown, CodeGen);
9294 }
9295 
9296 void CGOpenMPRuntime::emitTargetCall(
9297     CodeGenFunction &CGF, const OMPExecutableDirective &D,
9298     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
9299     const Expr *Device,
9300     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
9301                                      const OMPLoopDirective &D)>
9302         SizeEmitter) {
9303   if (!CGF.HaveInsertPoint())
9304     return;
9305 
9306   assert(OutlinedFn && "Invalid outlined function!");
9307 
9308   const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>();
9309   llvm::SmallVector<llvm::Value *, 16> CapturedVars;
9310   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
9311   auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF,
9312                                             PrePostActionTy &) {
9313     CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9314   };
9315   emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen);
9316 
9317   CodeGenFunction::OMPTargetDataInfo InputInfo;
9318   llvm::Value *MapTypesArray = nullptr;
9319   // Fill up the pointer arrays and transfer execution to the device.
9320   auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo,
9321                     &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars,
9322                     SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) {
9323     // On top of the arrays that were filled up, the target offloading call
9324     // takes as arguments the device id as well as the host pointer. The host
9325     // pointer is used by the runtime library to identify the current target
9326     // region, so it only has to be unique and not necessarily point to
9327     // anything. It could be the pointer to the outlined function that
9328     // implements the target region, but we aren't using that so that the
9329     // compiler doesn't need to keep that, and could therefore inline the host
9330     // function if proven worthwhile during optimization.
9331 
9332     // From this point on, we need to have an ID of the target region defined.
9333     assert(OutlinedFnID && "Invalid outlined function ID!");
9334 
9335     // Emit device ID if any.
9336     llvm::Value *DeviceID;
9337     if (Device) {
9338       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
9339                                            CGF.Int64Ty, /*isSigned=*/true);
9340     } else {
9341       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
9342     }
9343 
9344     // Emit the number of elements in the offloading arrays.
9345     llvm::Value *PointerNum =
9346         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
9347 
9348     // Return value of the runtime offloading call.
9349     llvm::Value *Return;
9350 
9351     llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D);
9352     llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D);
9353 
9354     // Emit tripcount for the target loop-based directive.
9355     emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter);
9356 
9357     bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
9358     // The target region is an outlined function launched by the runtime
9359     // via calls __tgt_target() or __tgt_target_teams().
9360     //
9361     // __tgt_target() launches a target region with one team and one thread,
9362     // executing a serial region.  This master thread may in turn launch
9363     // more threads within its team upon encountering a parallel region,
9364     // however, no additional teams can be launched on the device.
9365     //
9366     // __tgt_target_teams() launches a target region with one or more teams,
9367     // each with one or more threads.  This call is required for target
9368     // constructs such as:
9369     //  'target teams'
9370     //  'target' / 'teams'
9371     //  'target teams distribute parallel for'
9372     //  'target parallel'
9373     // and so on.
9374     //
9375     // Note that on the host and CPU targets, the runtime implementation of
9376     // these calls simply call the outlined function without forking threads.
9377     // The outlined functions themselves have runtime calls to
9378     // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by
9379     // the compiler in emitTeamsCall() and emitParallelCall().
9380     //
9381     // In contrast, on the NVPTX target, the implementation of
9382     // __tgt_target_teams() launches a GPU kernel with the requested number
9383     // of teams and threads so no additional calls to the runtime are required.
9384     if (NumTeams) {
9385       // If we have NumTeams defined this means that we have an enclosed teams
9386       // region. Therefore we also expect to have NumThreads defined. These two
9387       // values should be defined in the presence of a teams directive,
9388       // regardless of having any clauses associated. If the user is using teams
9389       // but no clauses, these two values will be the default that should be
9390       // passed to the runtime library - a 32-bit integer with the value zero.
9391       assert(NumThreads && "Thread limit expression should be available along "
9392                            "with number of teams.");
9393       llvm::Value *OffloadingArgs[] = {DeviceID,
9394                                        OutlinedFnID,
9395                                        PointerNum,
9396                                        InputInfo.BasePointersArray.getPointer(),
9397                                        InputInfo.PointersArray.getPointer(),
9398                                        InputInfo.SizesArray.getPointer(),
9399                                        MapTypesArray,
9400                                        NumTeams,
9401                                        NumThreads};
9402       Return = CGF.EmitRuntimeCall(
9403           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait
9404                                           : OMPRTL__tgt_target_teams),
9405           OffloadingArgs);
9406     } else {
9407       llvm::Value *OffloadingArgs[] = {DeviceID,
9408                                        OutlinedFnID,
9409                                        PointerNum,
9410                                        InputInfo.BasePointersArray.getPointer(),
9411                                        InputInfo.PointersArray.getPointer(),
9412                                        InputInfo.SizesArray.getPointer(),
9413                                        MapTypesArray};
9414       Return = CGF.EmitRuntimeCall(
9415           createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait
9416                                           : OMPRTL__tgt_target),
9417           OffloadingArgs);
9418     }
9419 
9420     // Check the error code and execute the host version if required.
9421     llvm::BasicBlock *OffloadFailedBlock =
9422         CGF.createBasicBlock("omp_offload.failed");
9423     llvm::BasicBlock *OffloadContBlock =
9424         CGF.createBasicBlock("omp_offload.cont");
9425     llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return);
9426     CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock);
9427 
9428     CGF.EmitBlock(OffloadFailedBlock);
9429     if (RequiresOuterTask) {
9430       CapturedVars.clear();
9431       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9432     }
9433     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9434     CGF.EmitBranch(OffloadContBlock);
9435 
9436     CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true);
9437   };
9438 
9439   // Notify that the host version must be executed.
9440   auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars,
9441                     RequiresOuterTask](CodeGenFunction &CGF,
9442                                        PrePostActionTy &) {
9443     if (RequiresOuterTask) {
9444       CapturedVars.clear();
9445       CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
9446     }
9447     emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
9448   };
9449 
9450   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
9451                           &CapturedVars, RequiresOuterTask,
9452                           &CS](CodeGenFunction &CGF, PrePostActionTy &) {
9453     // Fill up the arrays with all the captured variables.
9454     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
9455     MappableExprsHandler::MapValuesArrayTy Pointers;
9456     MappableExprsHandler::MapValuesArrayTy Sizes;
9457     MappableExprsHandler::MapFlagsArrayTy MapTypes;
9458 
9459     // Get mappable expression information.
9460     MappableExprsHandler MEHandler(D, CGF);
9461     llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers;
9462 
9463     auto RI = CS.getCapturedRecordDecl()->field_begin();
9464     auto CV = CapturedVars.begin();
9465     for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(),
9466                                               CE = CS.capture_end();
9467          CI != CE; ++CI, ++RI, ++CV) {
9468       MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers;
9469       MappableExprsHandler::MapValuesArrayTy CurPointers;
9470       MappableExprsHandler::MapValuesArrayTy CurSizes;
9471       MappableExprsHandler::MapFlagsArrayTy CurMapTypes;
9472       MappableExprsHandler::StructRangeInfoTy PartialStruct;
9473 
9474       // VLA sizes are passed to the outlined region by copy and do not have map
9475       // information associated.
9476       if (CI->capturesVariableArrayType()) {
9477         CurBasePointers.push_back(*CV);
9478         CurPointers.push_back(*CV);
9479         CurSizes.push_back(CGF.Builder.CreateIntCast(
9480             CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true));
9481         // Copy to the device as an argument. No need to retrieve it.
9482         CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL |
9483                               MappableExprsHandler::OMP_MAP_TARGET_PARAM |
9484                               MappableExprsHandler::OMP_MAP_IMPLICIT);
9485       } else {
9486         // If we have any information in the map clause, we use it, otherwise we
9487         // just do a default mapping.
9488         MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers,
9489                                          CurSizes, CurMapTypes, PartialStruct);
9490         if (CurBasePointers.empty())
9491           MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers,
9492                                            CurPointers, CurSizes, CurMapTypes);
9493         // Generate correct mapping for variables captured by reference in
9494         // lambdas.
9495         if (CI->capturesVariable())
9496           MEHandler.generateInfoForLambdaCaptures(
9497               CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes,
9498               CurMapTypes, LambdaPointers);
9499       }
9500       // We expect to have at least an element of information for this capture.
9501       assert(!CurBasePointers.empty() &&
9502              "Non-existing map pointer for capture!");
9503       assert(CurBasePointers.size() == CurPointers.size() &&
9504              CurBasePointers.size() == CurSizes.size() &&
9505              CurBasePointers.size() == CurMapTypes.size() &&
9506              "Inconsistent map information sizes!");
9507 
9508       // If there is an entry in PartialStruct it means we have a struct with
9509       // individual members mapped. Emit an extra combined entry.
9510       if (PartialStruct.Base.isValid())
9511         MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes,
9512                                     CurMapTypes, PartialStruct);
9513 
9514       // We need to append the results of this capture to what we already have.
9515       BasePointers.append(CurBasePointers.begin(), CurBasePointers.end());
9516       Pointers.append(CurPointers.begin(), CurPointers.end());
9517       Sizes.append(CurSizes.begin(), CurSizes.end());
9518       MapTypes.append(CurMapTypes.begin(), CurMapTypes.end());
9519     }
9520     // Adjust MEMBER_OF flags for the lambdas captures.
9521     MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers,
9522                                               Pointers, MapTypes);
9523     // Map other list items in the map clause which are not captured variables
9524     // but "declare target link" global variables.
9525     MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes,
9526                                                MapTypes);
9527 
9528     TargetDataInfo Info;
9529     // Fill up the arrays and create the arguments.
9530     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
9531     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
9532                                  Info.PointersArray, Info.SizesArray,
9533                                  Info.MapTypesArray, Info);
9534     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
9535     InputInfo.BasePointersArray =
9536         Address(Info.BasePointersArray, CGM.getPointerAlign());
9537     InputInfo.PointersArray =
9538         Address(Info.PointersArray, CGM.getPointerAlign());
9539     InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign());
9540     MapTypesArray = Info.MapTypesArray;
9541     if (RequiresOuterTask)
9542       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
9543     else
9544       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
9545   };
9546 
9547   auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask](
9548                              CodeGenFunction &CGF, PrePostActionTy &) {
9549     if (RequiresOuterTask) {
9550       CodeGenFunction::OMPTargetDataInfo InputInfo;
9551       CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo);
9552     } else {
9553       emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen);
9554     }
9555   };
9556 
9557   // If we have a target function ID it means that we need to support
9558   // offloading, otherwise, just execute on the host. We need to execute on host
9559   // regardless of the conditional in the if clause if, e.g., the user do not
9560   // specify target triples.
9561   if (OutlinedFnID) {
9562     if (IfCond) {
9563       emitOMPIfClause(CGF, IfCond, TargetThenGen, TargetElseGen);
9564     } else {
9565       RegionCodeGenTy ThenRCG(TargetThenGen);
9566       ThenRCG(CGF);
9567     }
9568   } else {
9569     RegionCodeGenTy ElseRCG(TargetElseGen);
9570     ElseRCG(CGF);
9571   }
9572 }
9573 
9574 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S,
9575                                                     StringRef ParentName) {
9576   if (!S)
9577     return;
9578 
9579   // Codegen OMP target directives that offload compute to the device.
9580   bool RequiresDeviceCodegen =
9581       isa<OMPExecutableDirective>(S) &&
9582       isOpenMPTargetExecutionDirective(
9583           cast<OMPExecutableDirective>(S)->getDirectiveKind());
9584 
9585   if (RequiresDeviceCodegen) {
9586     const auto &E = *cast<OMPExecutableDirective>(S);
9587     unsigned DeviceID;
9588     unsigned FileID;
9589     unsigned Line;
9590     getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID,
9591                              FileID, Line);
9592 
9593     // Is this a target region that should not be emitted as an entry point? If
9594     // so just signal we are done with this target region.
9595     if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID,
9596                                                             ParentName, Line))
9597       return;
9598 
9599     switch (E.getDirectiveKind()) {
9600     case OMPD_target:
9601       CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName,
9602                                                    cast<OMPTargetDirective>(E));
9603       break;
9604     case OMPD_target_parallel:
9605       CodeGenFunction::EmitOMPTargetParallelDeviceFunction(
9606           CGM, ParentName, cast<OMPTargetParallelDirective>(E));
9607       break;
9608     case OMPD_target_teams:
9609       CodeGenFunction::EmitOMPTargetTeamsDeviceFunction(
9610           CGM, ParentName, cast<OMPTargetTeamsDirective>(E));
9611       break;
9612     case OMPD_target_teams_distribute:
9613       CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction(
9614           CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E));
9615       break;
9616     case OMPD_target_teams_distribute_simd:
9617       CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction(
9618           CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E));
9619       break;
9620     case OMPD_target_parallel_for:
9621       CodeGenFunction::EmitOMPTargetParallelForDeviceFunction(
9622           CGM, ParentName, cast<OMPTargetParallelForDirective>(E));
9623       break;
9624     case OMPD_target_parallel_for_simd:
9625       CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction(
9626           CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E));
9627       break;
9628     case OMPD_target_simd:
9629       CodeGenFunction::EmitOMPTargetSimdDeviceFunction(
9630           CGM, ParentName, cast<OMPTargetSimdDirective>(E));
9631       break;
9632     case OMPD_target_teams_distribute_parallel_for:
9633       CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction(
9634           CGM, ParentName,
9635           cast<OMPTargetTeamsDistributeParallelForDirective>(E));
9636       break;
9637     case OMPD_target_teams_distribute_parallel_for_simd:
9638       CodeGenFunction::
9639           EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(
9640               CGM, ParentName,
9641               cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E));
9642       break;
9643     case OMPD_parallel:
9644     case OMPD_for:
9645     case OMPD_parallel_for:
9646     case OMPD_parallel_sections:
9647     case OMPD_for_simd:
9648     case OMPD_parallel_for_simd:
9649     case OMPD_cancel:
9650     case OMPD_cancellation_point:
9651     case OMPD_ordered:
9652     case OMPD_threadprivate:
9653     case OMPD_allocate:
9654     case OMPD_task:
9655     case OMPD_simd:
9656     case OMPD_sections:
9657     case OMPD_section:
9658     case OMPD_single:
9659     case OMPD_master:
9660     case OMPD_critical:
9661     case OMPD_taskyield:
9662     case OMPD_barrier:
9663     case OMPD_taskwait:
9664     case OMPD_taskgroup:
9665     case OMPD_atomic:
9666     case OMPD_flush:
9667     case OMPD_teams:
9668     case OMPD_target_data:
9669     case OMPD_target_exit_data:
9670     case OMPD_target_enter_data:
9671     case OMPD_distribute:
9672     case OMPD_distribute_simd:
9673     case OMPD_distribute_parallel_for:
9674     case OMPD_distribute_parallel_for_simd:
9675     case OMPD_teams_distribute:
9676     case OMPD_teams_distribute_simd:
9677     case OMPD_teams_distribute_parallel_for:
9678     case OMPD_teams_distribute_parallel_for_simd:
9679     case OMPD_target_update:
9680     case OMPD_declare_simd:
9681     case OMPD_declare_variant:
9682     case OMPD_declare_target:
9683     case OMPD_end_declare_target:
9684     case OMPD_declare_reduction:
9685     case OMPD_declare_mapper:
9686     case OMPD_taskloop:
9687     case OMPD_taskloop_simd:
9688     case OMPD_master_taskloop:
9689     case OMPD_requires:
9690     case OMPD_unknown:
9691       llvm_unreachable("Unknown target directive for OpenMP device codegen.");
9692     }
9693     return;
9694   }
9695 
9696   if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) {
9697     if (!E->hasAssociatedStmt() || !E->getAssociatedStmt())
9698       return;
9699 
9700     scanForTargetRegionsFunctions(
9701         E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName);
9702     return;
9703   }
9704 
9705   // If this is a lambda function, look into its body.
9706   if (const auto *L = dyn_cast<LambdaExpr>(S))
9707     S = L->getBody();
9708 
9709   // Keep looking for target regions recursively.
9710   for (const Stmt *II : S->children())
9711     scanForTargetRegionsFunctions(II, ParentName);
9712 }
9713 
9714 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) {
9715   // If emitting code for the host, we do not process FD here. Instead we do
9716   // the normal code generation.
9717   if (!CGM.getLangOpts().OpenMPIsDevice) {
9718     if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) {
9719       Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9720           OMPDeclareTargetDeclAttr::getDeviceType(FD);
9721       // Do not emit device_type(nohost) functions for the host.
9722       if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost)
9723         return true;
9724     }
9725     return false;
9726   }
9727 
9728   const ValueDecl *VD = cast<ValueDecl>(GD.getDecl());
9729   StringRef Name = CGM.getMangledName(GD);
9730   // Try to detect target regions in the function.
9731   if (const auto *FD = dyn_cast<FunctionDecl>(VD)) {
9732     scanForTargetRegionsFunctions(FD->getBody(), Name);
9733     Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
9734         OMPDeclareTargetDeclAttr::getDeviceType(FD);
9735     // Do not emit device_type(nohost) functions for the host.
9736     if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host)
9737       return true;
9738   }
9739 
9740   // Do not to emit function if it is not marked as declare target.
9741   return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) &&
9742          AlreadyEmittedTargetFunctions.count(Name) == 0;
9743 }
9744 
9745 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
9746   if (!CGM.getLangOpts().OpenMPIsDevice)
9747     return false;
9748 
9749   // Check if there are Ctors/Dtors in this declaration and look for target
9750   // regions in it. We use the complete variant to produce the kernel name
9751   // mangling.
9752   QualType RDTy = cast<VarDecl>(GD.getDecl())->getType();
9753   if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) {
9754     for (const CXXConstructorDecl *Ctor : RD->ctors()) {
9755       StringRef ParentName =
9756           CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete));
9757       scanForTargetRegionsFunctions(Ctor->getBody(), ParentName);
9758     }
9759     if (const CXXDestructorDecl *Dtor = RD->getDestructor()) {
9760       StringRef ParentName =
9761           CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete));
9762       scanForTargetRegionsFunctions(Dtor->getBody(), ParentName);
9763     }
9764   }
9765 
9766   // Do not to emit variable if it is not marked as declare target.
9767   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9768       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(
9769           cast<VarDecl>(GD.getDecl()));
9770   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
9771       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9772        HasRequiresUnifiedSharedMemory)) {
9773     DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl()));
9774     return true;
9775   }
9776   return false;
9777 }
9778 
9779 llvm::Constant *
9780 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF,
9781                                                 const VarDecl *VD) {
9782   assert(VD->getType().isConstant(CGM.getContext()) &&
9783          "Expected constant variable.");
9784   StringRef VarName;
9785   llvm::Constant *Addr;
9786   llvm::GlobalValue::LinkageTypes Linkage;
9787   QualType Ty = VD->getType();
9788   SmallString<128> Buffer;
9789   {
9790     unsigned DeviceID;
9791     unsigned FileID;
9792     unsigned Line;
9793     getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID,
9794                              FileID, Line);
9795     llvm::raw_svector_ostream OS(Buffer);
9796     OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID)
9797        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
9798     VarName = OS.str();
9799   }
9800   Linkage = llvm::GlobalValue::InternalLinkage;
9801   Addr =
9802       getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName,
9803                                   getDefaultFirstprivateAddressSpace());
9804   cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage);
9805   CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty);
9806   CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr));
9807   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
9808       VarName, Addr, VarSize,
9809       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage);
9810   return Addr;
9811 }
9812 
9813 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD,
9814                                                    llvm::Constant *Addr) {
9815   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
9816       !CGM.getLangOpts().OpenMPIsDevice)
9817     return;
9818   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9819       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
9820   if (!Res) {
9821     if (CGM.getLangOpts().OpenMPIsDevice) {
9822       // Register non-target variables being emitted in device code (debug info
9823       // may cause this).
9824       StringRef VarName = CGM.getMangledName(VD);
9825       EmittedNonTargetVariables.try_emplace(VarName, Addr);
9826     }
9827     return;
9828   }
9829   // Register declare target variables.
9830   OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags;
9831   StringRef VarName;
9832   CharUnits VarSize;
9833   llvm::GlobalValue::LinkageTypes Linkage;
9834 
9835   if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9836       !HasRequiresUnifiedSharedMemory) {
9837     Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
9838     VarName = CGM.getMangledName(VD);
9839     if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) {
9840       VarSize = CGM.getContext().getTypeSizeInChars(VD->getType());
9841       assert(!VarSize.isZero() && "Expected non-zero size of the variable");
9842     } else {
9843       VarSize = CharUnits::Zero();
9844     }
9845     Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false);
9846     // Temp solution to prevent optimizations of the internal variables.
9847     if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) {
9848       std::string RefName = getName({VarName, "ref"});
9849       if (!CGM.GetGlobalValue(RefName)) {
9850         llvm::Constant *AddrRef =
9851             getOrCreateInternalVariable(Addr->getType(), RefName);
9852         auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef);
9853         GVAddrRef->setConstant(/*Val=*/true);
9854         GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage);
9855         GVAddrRef->setInitializer(Addr);
9856         CGM.addCompilerUsedGlobal(GVAddrRef);
9857       }
9858     }
9859   } else {
9860     assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
9861             (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9862              HasRequiresUnifiedSharedMemory)) &&
9863            "Declare target attribute must link or to with unified memory.");
9864     if (*Res == OMPDeclareTargetDeclAttr::MT_Link)
9865       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink;
9866     else
9867       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
9868 
9869     if (CGM.getLangOpts().OpenMPIsDevice) {
9870       VarName = Addr->getName();
9871       Addr = nullptr;
9872     } else {
9873       VarName = getAddrOfDeclareTargetVar(VD).getName();
9874       Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer());
9875     }
9876     VarSize = CGM.getPointerSize();
9877     Linkage = llvm::GlobalValue::WeakAnyLinkage;
9878   }
9879 
9880   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
9881       VarName, Addr, VarSize, Flags, Linkage);
9882 }
9883 
9884 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) {
9885   if (isa<FunctionDecl>(GD.getDecl()) ||
9886       isa<OMPDeclareReductionDecl>(GD.getDecl()))
9887     return emitTargetFunctions(GD);
9888 
9889   return emitTargetGlobalVariable(GD);
9890 }
9891 
9892 void CGOpenMPRuntime::emitDeferredTargetDecls() const {
9893   for (const VarDecl *VD : DeferredGlobalVariables) {
9894     llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
9895         OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
9896     if (!Res)
9897       continue;
9898     if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9899         !HasRequiresUnifiedSharedMemory) {
9900       CGM.EmitGlobal(VD);
9901     } else {
9902       assert((*Res == OMPDeclareTargetDeclAttr::MT_Link ||
9903               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
9904                HasRequiresUnifiedSharedMemory)) &&
9905              "Expected link clause or to clause with unified memory.");
9906       (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
9907     }
9908   }
9909 }
9910 
9911 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas(
9912     CodeGenFunction &CGF, const OMPExecutableDirective &D) const {
9913   assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) &&
9914          " Expected target-based directive.");
9915 }
9916 
9917 void CGOpenMPRuntime::checkArchForUnifiedAddressing(
9918     const OMPRequiresDecl *D) {
9919   for (const OMPClause *Clause : D->clauselists()) {
9920     if (Clause->getClauseKind() == OMPC_unified_shared_memory) {
9921       HasRequiresUnifiedSharedMemory = true;
9922       break;
9923     }
9924   }
9925 }
9926 
9927 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD,
9928                                                        LangAS &AS) {
9929   if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())
9930     return false;
9931   const auto *A = VD->getAttr<OMPAllocateDeclAttr>();
9932   switch(A->getAllocatorType()) {
9933   case OMPAllocateDeclAttr::OMPDefaultMemAlloc:
9934   // Not supported, fallback to the default mem space.
9935   case OMPAllocateDeclAttr::OMPLargeCapMemAlloc:
9936   case OMPAllocateDeclAttr::OMPCGroupMemAlloc:
9937   case OMPAllocateDeclAttr::OMPHighBWMemAlloc:
9938   case OMPAllocateDeclAttr::OMPLowLatMemAlloc:
9939   case OMPAllocateDeclAttr::OMPThreadMemAlloc:
9940   case OMPAllocateDeclAttr::OMPConstMemAlloc:
9941   case OMPAllocateDeclAttr::OMPPTeamMemAlloc:
9942     AS = LangAS::Default;
9943     return true;
9944   case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc:
9945     llvm_unreachable("Expected predefined allocator for the variables with the "
9946                      "static storage.");
9947   }
9948   return false;
9949 }
9950 
9951 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const {
9952   return HasRequiresUnifiedSharedMemory;
9953 }
9954 
9955 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII(
9956     CodeGenModule &CGM)
9957     : CGM(CGM) {
9958   if (CGM.getLangOpts().OpenMPIsDevice) {
9959     SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal;
9960     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false;
9961   }
9962 }
9963 
9964 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() {
9965   if (CGM.getLangOpts().OpenMPIsDevice)
9966     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal;
9967 }
9968 
9969 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) {
9970   if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal)
9971     return true;
9972 
9973   StringRef Name = CGM.getMangledName(GD);
9974   const auto *D = cast<FunctionDecl>(GD.getDecl());
9975   // Do not to emit function if it is marked as declare target as it was already
9976   // emitted.
9977   if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) {
9978     if (D->hasBody() && AlreadyEmittedTargetFunctions.count(Name) == 0) {
9979       if (auto *F = dyn_cast_or_null<llvm::Function>(CGM.GetGlobalValue(Name)))
9980         return !F->isDeclaration();
9981       return false;
9982     }
9983     return true;
9984   }
9985 
9986   return !AlreadyEmittedTargetFunctions.insert(Name).second;
9987 }
9988 
9989 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() {
9990   // If we don't have entries or if we are emitting code for the device, we
9991   // don't need to do anything.
9992   if (CGM.getLangOpts().OMPTargetTriples.empty() ||
9993       CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice ||
9994       (OffloadEntriesInfoManager.empty() &&
9995        !HasEmittedDeclareTargetRegion &&
9996        !HasEmittedTargetRegion))
9997     return nullptr;
9998 
9999   // Create and register the function that handles the requires directives.
10000   ASTContext &C = CGM.getContext();
10001 
10002   llvm::Function *RequiresRegFn;
10003   {
10004     CodeGenFunction CGF(CGM);
10005     const auto &FI = CGM.getTypes().arrangeNullaryFunction();
10006     llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
10007     std::string ReqName = getName({"omp_offloading", "requires_reg"});
10008     RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI);
10009     CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {});
10010     OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE;
10011     // TODO: check for other requires clauses.
10012     // The requires directive takes effect only when a target region is
10013     // present in the compilation unit. Otherwise it is ignored and not
10014     // passed to the runtime. This avoids the runtime from throwing an error
10015     // for mismatching requires clauses across compilation units that don't
10016     // contain at least 1 target region.
10017     assert((HasEmittedTargetRegion ||
10018             HasEmittedDeclareTargetRegion ||
10019             !OffloadEntriesInfoManager.empty()) &&
10020            "Target or declare target region expected.");
10021     if (HasRequiresUnifiedSharedMemory)
10022       Flags = OMP_REQ_UNIFIED_SHARED_MEMORY;
10023     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires),
10024         llvm::ConstantInt::get(CGM.Int64Ty, Flags));
10025     CGF.FinishFunction();
10026   }
10027   return RequiresRegFn;
10028 }
10029 
10030 llvm::Function *CGOpenMPRuntime::emitRegistrationFunction() {
10031   // If we have offloading in the current module, we need to emit the entries
10032   // now and register the offloading descriptor.
10033   createOffloadEntriesAndInfoMetadata();
10034 
10035   // Create and register the offloading binary descriptors. This is the main
10036   // entity that captures all the information about offloading in the current
10037   // compilation unit.
10038   return createOffloadingBinaryDescriptorRegistration();
10039 }
10040 
10041 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF,
10042                                     const OMPExecutableDirective &D,
10043                                     SourceLocation Loc,
10044                                     llvm::Function *OutlinedFn,
10045                                     ArrayRef<llvm::Value *> CapturedVars) {
10046   if (!CGF.HaveInsertPoint())
10047     return;
10048 
10049   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
10050   CodeGenFunction::RunCleanupsScope Scope(CGF);
10051 
10052   // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn);
10053   llvm::Value *Args[] = {
10054       RTLoc,
10055       CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
10056       CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())};
10057   llvm::SmallVector<llvm::Value *, 16> RealArgs;
10058   RealArgs.append(std::begin(Args), std::end(Args));
10059   RealArgs.append(CapturedVars.begin(), CapturedVars.end());
10060 
10061   llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams);
10062   CGF.EmitRuntimeCall(RTLFn, RealArgs);
10063 }
10064 
10065 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
10066                                          const Expr *NumTeams,
10067                                          const Expr *ThreadLimit,
10068                                          SourceLocation Loc) {
10069   if (!CGF.HaveInsertPoint())
10070     return;
10071 
10072   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
10073 
10074   llvm::Value *NumTeamsVal =
10075       NumTeams
10076           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams),
10077                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
10078           : CGF.Builder.getInt32(0);
10079 
10080   llvm::Value *ThreadLimitVal =
10081       ThreadLimit
10082           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit),
10083                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
10084           : CGF.Builder.getInt32(0);
10085 
10086   // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit)
10087   llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal,
10088                                      ThreadLimitVal};
10089   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams),
10090                       PushNumTeamsArgs);
10091 }
10092 
10093 void CGOpenMPRuntime::emitTargetDataCalls(
10094     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
10095     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
10096   if (!CGF.HaveInsertPoint())
10097     return;
10098 
10099   // Action used to replace the default codegen action and turn privatization
10100   // off.
10101   PrePostActionTy NoPrivAction;
10102 
10103   // Generate the code for the opening of the data environment. Capture all the
10104   // arguments of the runtime call by reference because they are used in the
10105   // closing of the region.
10106   auto &&BeginThenGen = [this, &D, Device, &Info,
10107                          &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) {
10108     // Fill up the arrays with all the mapped variables.
10109     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
10110     MappableExprsHandler::MapValuesArrayTy Pointers;
10111     MappableExprsHandler::MapValuesArrayTy Sizes;
10112     MappableExprsHandler::MapFlagsArrayTy MapTypes;
10113 
10114     // Get map clause information.
10115     MappableExprsHandler MCHandler(D, CGF);
10116     MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
10117 
10118     // Fill up the arrays and create the arguments.
10119     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
10120 
10121     llvm::Value *BasePointersArrayArg = nullptr;
10122     llvm::Value *PointersArrayArg = nullptr;
10123     llvm::Value *SizesArrayArg = nullptr;
10124     llvm::Value *MapTypesArrayArg = nullptr;
10125     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10126                                  SizesArrayArg, MapTypesArrayArg, Info);
10127 
10128     // Emit device ID if any.
10129     llvm::Value *DeviceID = nullptr;
10130     if (Device) {
10131       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10132                                            CGF.Int64Ty, /*isSigned=*/true);
10133     } else {
10134       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10135     }
10136 
10137     // Emit the number of elements in the offloading arrays.
10138     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10139 
10140     llvm::Value *OffloadingArgs[] = {
10141         DeviceID,         PointerNum,    BasePointersArrayArg,
10142         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10143     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin),
10144                         OffloadingArgs);
10145 
10146     // If device pointer privatization is required, emit the body of the region
10147     // here. It will have to be duplicated: with and without privatization.
10148     if (!Info.CaptureDeviceAddrMap.empty())
10149       CodeGen(CGF);
10150   };
10151 
10152   // Generate code for the closing of the data region.
10153   auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF,
10154                                             PrePostActionTy &) {
10155     assert(Info.isValid() && "Invalid data environment closing arguments.");
10156 
10157     llvm::Value *BasePointersArrayArg = nullptr;
10158     llvm::Value *PointersArrayArg = nullptr;
10159     llvm::Value *SizesArrayArg = nullptr;
10160     llvm::Value *MapTypesArrayArg = nullptr;
10161     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
10162                                  SizesArrayArg, MapTypesArrayArg, Info);
10163 
10164     // Emit device ID if any.
10165     llvm::Value *DeviceID = nullptr;
10166     if (Device) {
10167       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10168                                            CGF.Int64Ty, /*isSigned=*/true);
10169     } else {
10170       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10171     }
10172 
10173     // Emit the number of elements in the offloading arrays.
10174     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
10175 
10176     llvm::Value *OffloadingArgs[] = {
10177         DeviceID,         PointerNum,    BasePointersArrayArg,
10178         PointersArrayArg, SizesArrayArg, MapTypesArrayArg};
10179     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end),
10180                         OffloadingArgs);
10181   };
10182 
10183   // If we need device pointer privatization, we need to emit the body of the
10184   // region with no privatization in the 'else' branch of the conditional.
10185   // Otherwise, we don't have to do anything.
10186   auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF,
10187                                                          PrePostActionTy &) {
10188     if (!Info.CaptureDeviceAddrMap.empty()) {
10189       CodeGen.setAction(NoPrivAction);
10190       CodeGen(CGF);
10191     }
10192   };
10193 
10194   // We don't have to do anything to close the region if the if clause evaluates
10195   // to false.
10196   auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {};
10197 
10198   if (IfCond) {
10199     emitOMPIfClause(CGF, IfCond, BeginThenGen, BeginElseGen);
10200   } else {
10201     RegionCodeGenTy RCG(BeginThenGen);
10202     RCG(CGF);
10203   }
10204 
10205   // If we don't require privatization of device pointers, we emit the body in
10206   // between the runtime calls. This avoids duplicating the body code.
10207   if (Info.CaptureDeviceAddrMap.empty()) {
10208     CodeGen.setAction(NoPrivAction);
10209     CodeGen(CGF);
10210   }
10211 
10212   if (IfCond) {
10213     emitOMPIfClause(CGF, IfCond, EndThenGen, EndElseGen);
10214   } else {
10215     RegionCodeGenTy RCG(EndThenGen);
10216     RCG(CGF);
10217   }
10218 }
10219 
10220 void CGOpenMPRuntime::emitTargetDataStandAloneCall(
10221     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
10222     const Expr *Device) {
10223   if (!CGF.HaveInsertPoint())
10224     return;
10225 
10226   assert((isa<OMPTargetEnterDataDirective>(D) ||
10227           isa<OMPTargetExitDataDirective>(D) ||
10228           isa<OMPTargetUpdateDirective>(D)) &&
10229          "Expecting either target enter, exit data, or update directives.");
10230 
10231   CodeGenFunction::OMPTargetDataInfo InputInfo;
10232   llvm::Value *MapTypesArray = nullptr;
10233   // Generate the code for the opening of the data environment.
10234   auto &&ThenGen = [this, &D, Device, &InputInfo,
10235                     &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) {
10236     // Emit device ID if any.
10237     llvm::Value *DeviceID = nullptr;
10238     if (Device) {
10239       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
10240                                            CGF.Int64Ty, /*isSigned=*/true);
10241     } else {
10242       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10243     }
10244 
10245     // Emit the number of elements in the offloading arrays.
10246     llvm::Constant *PointerNum =
10247         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
10248 
10249     llvm::Value *OffloadingArgs[] = {DeviceID,
10250                                      PointerNum,
10251                                      InputInfo.BasePointersArray.getPointer(),
10252                                      InputInfo.PointersArray.getPointer(),
10253                                      InputInfo.SizesArray.getPointer(),
10254                                      MapTypesArray};
10255 
10256     // Select the right runtime function call for each expected standalone
10257     // directive.
10258     const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
10259     OpenMPRTLFunction RTLFn;
10260     switch (D.getDirectiveKind()) {
10261     case OMPD_target_enter_data:
10262       RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait
10263                         : OMPRTL__tgt_target_data_begin;
10264       break;
10265     case OMPD_target_exit_data:
10266       RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait
10267                         : OMPRTL__tgt_target_data_end;
10268       break;
10269     case OMPD_target_update:
10270       RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait
10271                         : OMPRTL__tgt_target_data_update;
10272       break;
10273     case OMPD_parallel:
10274     case OMPD_for:
10275     case OMPD_parallel_for:
10276     case OMPD_parallel_sections:
10277     case OMPD_for_simd:
10278     case OMPD_parallel_for_simd:
10279     case OMPD_cancel:
10280     case OMPD_cancellation_point:
10281     case OMPD_ordered:
10282     case OMPD_threadprivate:
10283     case OMPD_allocate:
10284     case OMPD_task:
10285     case OMPD_simd:
10286     case OMPD_sections:
10287     case OMPD_section:
10288     case OMPD_single:
10289     case OMPD_master:
10290     case OMPD_critical:
10291     case OMPD_taskyield:
10292     case OMPD_barrier:
10293     case OMPD_taskwait:
10294     case OMPD_taskgroup:
10295     case OMPD_atomic:
10296     case OMPD_flush:
10297     case OMPD_teams:
10298     case OMPD_target_data:
10299     case OMPD_distribute:
10300     case OMPD_distribute_simd:
10301     case OMPD_distribute_parallel_for:
10302     case OMPD_distribute_parallel_for_simd:
10303     case OMPD_teams_distribute:
10304     case OMPD_teams_distribute_simd:
10305     case OMPD_teams_distribute_parallel_for:
10306     case OMPD_teams_distribute_parallel_for_simd:
10307     case OMPD_declare_simd:
10308     case OMPD_declare_variant:
10309     case OMPD_declare_target:
10310     case OMPD_end_declare_target:
10311     case OMPD_declare_reduction:
10312     case OMPD_declare_mapper:
10313     case OMPD_taskloop:
10314     case OMPD_taskloop_simd:
10315     case OMPD_master_taskloop:
10316     case OMPD_target:
10317     case OMPD_target_simd:
10318     case OMPD_target_teams_distribute:
10319     case OMPD_target_teams_distribute_simd:
10320     case OMPD_target_teams_distribute_parallel_for:
10321     case OMPD_target_teams_distribute_parallel_for_simd:
10322     case OMPD_target_teams:
10323     case OMPD_target_parallel:
10324     case OMPD_target_parallel_for:
10325     case OMPD_target_parallel_for_simd:
10326     case OMPD_requires:
10327     case OMPD_unknown:
10328       llvm_unreachable("Unexpected standalone target data directive.");
10329       break;
10330     }
10331     CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs);
10332   };
10333 
10334   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray](
10335                              CodeGenFunction &CGF, PrePostActionTy &) {
10336     // Fill up the arrays with all the mapped variables.
10337     MappableExprsHandler::MapBaseValuesArrayTy BasePointers;
10338     MappableExprsHandler::MapValuesArrayTy Pointers;
10339     MappableExprsHandler::MapValuesArrayTy Sizes;
10340     MappableExprsHandler::MapFlagsArrayTy MapTypes;
10341 
10342     // Get map clause information.
10343     MappableExprsHandler MEHandler(D, CGF);
10344     MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes);
10345 
10346     TargetDataInfo Info;
10347     // Fill up the arrays and create the arguments.
10348     emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info);
10349     emitOffloadingArraysArgument(CGF, Info.BasePointersArray,
10350                                  Info.PointersArray, Info.SizesArray,
10351                                  Info.MapTypesArray, Info);
10352     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
10353     InputInfo.BasePointersArray =
10354         Address(Info.BasePointersArray, CGM.getPointerAlign());
10355     InputInfo.PointersArray =
10356         Address(Info.PointersArray, CGM.getPointerAlign());
10357     InputInfo.SizesArray =
10358         Address(Info.SizesArray, CGM.getPointerAlign());
10359     MapTypesArray = Info.MapTypesArray;
10360     if (D.hasClausesOfKind<OMPDependClause>())
10361       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
10362     else
10363       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
10364   };
10365 
10366   if (IfCond) {
10367     emitOMPIfClause(CGF, IfCond, TargetThenGen,
10368                     [](CodeGenFunction &CGF, PrePostActionTy &) {});
10369   } else {
10370     RegionCodeGenTy ThenRCG(TargetThenGen);
10371     ThenRCG(CGF);
10372   }
10373 }
10374 
10375 namespace {
10376   /// Kind of parameter in a function with 'declare simd' directive.
10377   enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector };
10378   /// Attribute set of the parameter.
10379   struct ParamAttrTy {
10380     ParamKindTy Kind = Vector;
10381     llvm::APSInt StrideOrArg;
10382     llvm::APSInt Alignment;
10383   };
10384 } // namespace
10385 
10386 static unsigned evaluateCDTSize(const FunctionDecl *FD,
10387                                 ArrayRef<ParamAttrTy> ParamAttrs) {
10388   // Every vector variant of a SIMD-enabled function has a vector length (VLEN).
10389   // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument
10390   // of that clause. The VLEN value must be power of 2.
10391   // In other case the notion of the function`s "characteristic data type" (CDT)
10392   // is used to compute the vector length.
10393   // CDT is defined in the following order:
10394   //   a) For non-void function, the CDT is the return type.
10395   //   b) If the function has any non-uniform, non-linear parameters, then the
10396   //   CDT is the type of the first such parameter.
10397   //   c) If the CDT determined by a) or b) above is struct, union, or class
10398   //   type which is pass-by-value (except for the type that maps to the
10399   //   built-in complex data type), the characteristic data type is int.
10400   //   d) If none of the above three cases is applicable, the CDT is int.
10401   // The VLEN is then determined based on the CDT and the size of vector
10402   // register of that ISA for which current vector version is generated. The
10403   // VLEN is computed using the formula below:
10404   //   VLEN  = sizeof(vector_register) / sizeof(CDT),
10405   // where vector register size specified in section 3.2.1 Registers and the
10406   // Stack Frame of original AMD64 ABI document.
10407   QualType RetType = FD->getReturnType();
10408   if (RetType.isNull())
10409     return 0;
10410   ASTContext &C = FD->getASTContext();
10411   QualType CDT;
10412   if (!RetType.isNull() && !RetType->isVoidType()) {
10413     CDT = RetType;
10414   } else {
10415     unsigned Offset = 0;
10416     if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) {
10417       if (ParamAttrs[Offset].Kind == Vector)
10418         CDT = C.getPointerType(C.getRecordType(MD->getParent()));
10419       ++Offset;
10420     }
10421     if (CDT.isNull()) {
10422       for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10423         if (ParamAttrs[I + Offset].Kind == Vector) {
10424           CDT = FD->getParamDecl(I)->getType();
10425           break;
10426         }
10427       }
10428     }
10429   }
10430   if (CDT.isNull())
10431     CDT = C.IntTy;
10432   CDT = CDT->getCanonicalTypeUnqualified();
10433   if (CDT->isRecordType() || CDT->isUnionType())
10434     CDT = C.IntTy;
10435   return C.getTypeSize(CDT);
10436 }
10437 
10438 static void
10439 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn,
10440                            const llvm::APSInt &VLENVal,
10441                            ArrayRef<ParamAttrTy> ParamAttrs,
10442                            OMPDeclareSimdDeclAttr::BranchStateTy State) {
10443   struct ISADataTy {
10444     char ISA;
10445     unsigned VecRegSize;
10446   };
10447   ISADataTy ISAData[] = {
10448       {
10449           'b', 128
10450       }, // SSE
10451       {
10452           'c', 256
10453       }, // AVX
10454       {
10455           'd', 256
10456       }, // AVX2
10457       {
10458           'e', 512
10459       }, // AVX512
10460   };
10461   llvm::SmallVector<char, 2> Masked;
10462   switch (State) {
10463   case OMPDeclareSimdDeclAttr::BS_Undefined:
10464     Masked.push_back('N');
10465     Masked.push_back('M');
10466     break;
10467   case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10468     Masked.push_back('N');
10469     break;
10470   case OMPDeclareSimdDeclAttr::BS_Inbranch:
10471     Masked.push_back('M');
10472     break;
10473   }
10474   for (char Mask : Masked) {
10475     for (const ISADataTy &Data : ISAData) {
10476       SmallString<256> Buffer;
10477       llvm::raw_svector_ostream Out(Buffer);
10478       Out << "_ZGV" << Data.ISA << Mask;
10479       if (!VLENVal) {
10480         unsigned NumElts = evaluateCDTSize(FD, ParamAttrs);
10481         assert(NumElts && "Non-zero simdlen/cdtsize expected");
10482         Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts);
10483       } else {
10484         Out << VLENVal;
10485       }
10486       for (const ParamAttrTy &ParamAttr : ParamAttrs) {
10487         switch (ParamAttr.Kind){
10488         case LinearWithVarStride:
10489           Out << 's' << ParamAttr.StrideOrArg;
10490           break;
10491         case Linear:
10492           Out << 'l';
10493           if (!!ParamAttr.StrideOrArg)
10494             Out << ParamAttr.StrideOrArg;
10495           break;
10496         case Uniform:
10497           Out << 'u';
10498           break;
10499         case Vector:
10500           Out << 'v';
10501           break;
10502         }
10503         if (!!ParamAttr.Alignment)
10504           Out << 'a' << ParamAttr.Alignment;
10505       }
10506       Out << '_' << Fn->getName();
10507       Fn->addFnAttr(Out.str());
10508     }
10509   }
10510 }
10511 
10512 // This are the Functions that are needed to mangle the name of the
10513 // vector functions generated by the compiler, according to the rules
10514 // defined in the "Vector Function ABI specifications for AArch64",
10515 // available at
10516 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi.
10517 
10518 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI.
10519 ///
10520 /// TODO: Need to implement the behavior for reference marked with a
10521 /// var or no linear modifiers (1.b in the section). For this, we
10522 /// need to extend ParamKindTy to support the linear modifiers.
10523 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) {
10524   QT = QT.getCanonicalType();
10525 
10526   if (QT->isVoidType())
10527     return false;
10528 
10529   if (Kind == ParamKindTy::Uniform)
10530     return false;
10531 
10532   if (Kind == ParamKindTy::Linear)
10533     return false;
10534 
10535   // TODO: Handle linear references with modifiers
10536 
10537   if (Kind == ParamKindTy::LinearWithVarStride)
10538     return false;
10539 
10540   return true;
10541 }
10542 
10543 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
10544 static bool getAArch64PBV(QualType QT, ASTContext &C) {
10545   QT = QT.getCanonicalType();
10546   unsigned Size = C.getTypeSize(QT);
10547 
10548   // Only scalars and complex within 16 bytes wide set PVB to true.
10549   if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128)
10550     return false;
10551 
10552   if (QT->isFloatingType())
10553     return true;
10554 
10555   if (QT->isIntegerType())
10556     return true;
10557 
10558   if (QT->isPointerType())
10559     return true;
10560 
10561   // TODO: Add support for complex types (section 3.1.2, item 2).
10562 
10563   return false;
10564 }
10565 
10566 /// Computes the lane size (LS) of a return type or of an input parameter,
10567 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI.
10568 /// TODO: Add support for references, section 3.2.1, item 1.
10569 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) {
10570   if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) {
10571     QualType PTy = QT.getCanonicalType()->getPointeeType();
10572     if (getAArch64PBV(PTy, C))
10573       return C.getTypeSize(PTy);
10574   }
10575   if (getAArch64PBV(QT, C))
10576     return C.getTypeSize(QT);
10577 
10578   return C.getTypeSize(C.getUIntPtrType());
10579 }
10580 
10581 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the
10582 // signature of the scalar function, as defined in 3.2.2 of the
10583 // AAVFABI.
10584 static std::tuple<unsigned, unsigned, bool>
10585 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) {
10586   QualType RetType = FD->getReturnType().getCanonicalType();
10587 
10588   ASTContext &C = FD->getASTContext();
10589 
10590   bool OutputBecomesInput = false;
10591 
10592   llvm::SmallVector<unsigned, 8> Sizes;
10593   if (!RetType->isVoidType()) {
10594     Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C));
10595     if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {}))
10596       OutputBecomesInput = true;
10597   }
10598   for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
10599     QualType QT = FD->getParamDecl(I)->getType().getCanonicalType();
10600     Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C));
10601   }
10602 
10603   assert(!Sizes.empty() && "Unable to determine NDS and WDS.");
10604   // The LS of a function parameter / return value can only be a power
10605   // of 2, starting from 8 bits, up to 128.
10606   assert(std::all_of(Sizes.begin(), Sizes.end(),
10607                      [](unsigned Size) {
10608                        return Size == 8 || Size == 16 || Size == 32 ||
10609                               Size == 64 || Size == 128;
10610                      }) &&
10611          "Invalid size");
10612 
10613   return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)),
10614                          *std::max_element(std::begin(Sizes), std::end(Sizes)),
10615                          OutputBecomesInput);
10616 }
10617 
10618 /// Mangle the parameter part of the vector function name according to
10619 /// their OpenMP classification. The mangling function is defined in
10620 /// section 3.5 of the AAVFABI.
10621 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) {
10622   SmallString<256> Buffer;
10623   llvm::raw_svector_ostream Out(Buffer);
10624   for (const auto &ParamAttr : ParamAttrs) {
10625     switch (ParamAttr.Kind) {
10626     case LinearWithVarStride:
10627       Out << "ls" << ParamAttr.StrideOrArg;
10628       break;
10629     case Linear:
10630       Out << 'l';
10631       // Don't print the step value if it is not present or if it is
10632       // equal to 1.
10633       if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1)
10634         Out << ParamAttr.StrideOrArg;
10635       break;
10636     case Uniform:
10637       Out << 'u';
10638       break;
10639     case Vector:
10640       Out << 'v';
10641       break;
10642     }
10643 
10644     if (!!ParamAttr.Alignment)
10645       Out << 'a' << ParamAttr.Alignment;
10646   }
10647 
10648   return Out.str();
10649 }
10650 
10651 // Function used to add the attribute. The parameter `VLEN` is
10652 // templated to allow the use of "x" when targeting scalable functions
10653 // for SVE.
10654 template <typename T>
10655 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix,
10656                                  char ISA, StringRef ParSeq,
10657                                  StringRef MangledName, bool OutputBecomesInput,
10658                                  llvm::Function *Fn) {
10659   SmallString<256> Buffer;
10660   llvm::raw_svector_ostream Out(Buffer);
10661   Out << Prefix << ISA << LMask << VLEN;
10662   if (OutputBecomesInput)
10663     Out << "v";
10664   Out << ParSeq << "_" << MangledName;
10665   Fn->addFnAttr(Out.str());
10666 }
10667 
10668 // Helper function to generate the Advanced SIMD names depending on
10669 // the value of the NDS when simdlen is not present.
10670 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask,
10671                                       StringRef Prefix, char ISA,
10672                                       StringRef ParSeq, StringRef MangledName,
10673                                       bool OutputBecomesInput,
10674                                       llvm::Function *Fn) {
10675   switch (NDS) {
10676   case 8:
10677     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10678                          OutputBecomesInput, Fn);
10679     addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName,
10680                          OutputBecomesInput, Fn);
10681     break;
10682   case 16:
10683     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10684                          OutputBecomesInput, Fn);
10685     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
10686                          OutputBecomesInput, Fn);
10687     break;
10688   case 32:
10689     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10690                          OutputBecomesInput, Fn);
10691     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
10692                          OutputBecomesInput, Fn);
10693     break;
10694   case 64:
10695   case 128:
10696     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
10697                          OutputBecomesInput, Fn);
10698     break;
10699   default:
10700     llvm_unreachable("Scalar type is too wide.");
10701   }
10702 }
10703 
10704 /// Emit vector function attributes for AArch64, as defined in the AAVFABI.
10705 static void emitAArch64DeclareSimdFunction(
10706     CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN,
10707     ArrayRef<ParamAttrTy> ParamAttrs,
10708     OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName,
10709     char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) {
10710 
10711   // Get basic data for building the vector signature.
10712   const auto Data = getNDSWDS(FD, ParamAttrs);
10713   const unsigned NDS = std::get<0>(Data);
10714   const unsigned WDS = std::get<1>(Data);
10715   const bool OutputBecomesInput = std::get<2>(Data);
10716 
10717   // Check the values provided via `simdlen` by the user.
10718   // 1. A `simdlen(1)` doesn't produce vector signatures,
10719   if (UserVLEN == 1) {
10720     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10721         DiagnosticsEngine::Warning,
10722         "The clause simdlen(1) has no effect when targeting aarch64.");
10723     CGM.getDiags().Report(SLoc, DiagID);
10724     return;
10725   }
10726 
10727   // 2. Section 3.3.1, item 1: user input must be a power of 2 for
10728   // Advanced SIMD output.
10729   if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) {
10730     unsigned DiagID = CGM.getDiags().getCustomDiagID(
10731         DiagnosticsEngine::Warning, "The value specified in simdlen must be a "
10732                                     "power of 2 when targeting Advanced SIMD.");
10733     CGM.getDiags().Report(SLoc, DiagID);
10734     return;
10735   }
10736 
10737   // 3. Section 3.4.1. SVE fixed lengh must obey the architectural
10738   // limits.
10739   if (ISA == 's' && UserVLEN != 0) {
10740     if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) {
10741       unsigned DiagID = CGM.getDiags().getCustomDiagID(
10742           DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit "
10743                                       "lanes in the architectural constraints "
10744                                       "for SVE (min is 128-bit, max is "
10745                                       "2048-bit, by steps of 128-bit)");
10746       CGM.getDiags().Report(SLoc, DiagID) << WDS;
10747       return;
10748     }
10749   }
10750 
10751   // Sort out parameter sequence.
10752   const std::string ParSeq = mangleVectorParameters(ParamAttrs);
10753   StringRef Prefix = "_ZGV";
10754   // Generate simdlen from user input (if any).
10755   if (UserVLEN) {
10756     if (ISA == 's') {
10757       // SVE generates only a masked function.
10758       addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10759                            OutputBecomesInput, Fn);
10760     } else {
10761       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10762       // Advanced SIMD generates one or two functions, depending on
10763       // the `[not]inbranch` clause.
10764       switch (State) {
10765       case OMPDeclareSimdDeclAttr::BS_Undefined:
10766         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10767                              OutputBecomesInput, Fn);
10768         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10769                              OutputBecomesInput, Fn);
10770         break;
10771       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10772         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
10773                              OutputBecomesInput, Fn);
10774         break;
10775       case OMPDeclareSimdDeclAttr::BS_Inbranch:
10776         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
10777                              OutputBecomesInput, Fn);
10778         break;
10779       }
10780     }
10781   } else {
10782     // If no user simdlen is provided, follow the AAVFABI rules for
10783     // generating the vector length.
10784     if (ISA == 's') {
10785       // SVE, section 3.4.1, item 1.
10786       addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName,
10787                            OutputBecomesInput, Fn);
10788     } else {
10789       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
10790       // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or
10791       // two vector names depending on the use of the clause
10792       // `[not]inbranch`.
10793       switch (State) {
10794       case OMPDeclareSimdDeclAttr::BS_Undefined:
10795         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10796                                   OutputBecomesInput, Fn);
10797         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
10798                                   OutputBecomesInput, Fn);
10799         break;
10800       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
10801         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
10802                                   OutputBecomesInput, Fn);
10803         break;
10804       case OMPDeclareSimdDeclAttr::BS_Inbranch:
10805         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
10806                                   OutputBecomesInput, Fn);
10807         break;
10808       }
10809     }
10810   }
10811 }
10812 
10813 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD,
10814                                               llvm::Function *Fn) {
10815   ASTContext &C = CGM.getContext();
10816   FD = FD->getMostRecentDecl();
10817   // Map params to their positions in function decl.
10818   llvm::DenseMap<const Decl *, unsigned> ParamPositions;
10819   if (isa<CXXMethodDecl>(FD))
10820     ParamPositions.try_emplace(FD, 0);
10821   unsigned ParamPos = ParamPositions.size();
10822   for (const ParmVarDecl *P : FD->parameters()) {
10823     ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos);
10824     ++ParamPos;
10825   }
10826   while (FD) {
10827     for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) {
10828       llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size());
10829       // Mark uniform parameters.
10830       for (const Expr *E : Attr->uniforms()) {
10831         E = E->IgnoreParenImpCasts();
10832         unsigned Pos;
10833         if (isa<CXXThisExpr>(E)) {
10834           Pos = ParamPositions[FD];
10835         } else {
10836           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
10837                                 ->getCanonicalDecl();
10838           Pos = ParamPositions[PVD];
10839         }
10840         ParamAttrs[Pos].Kind = Uniform;
10841       }
10842       // Get alignment info.
10843       auto NI = Attr->alignments_begin();
10844       for (const Expr *E : Attr->aligneds()) {
10845         E = E->IgnoreParenImpCasts();
10846         unsigned Pos;
10847         QualType ParmTy;
10848         if (isa<CXXThisExpr>(E)) {
10849           Pos = ParamPositions[FD];
10850           ParmTy = E->getType();
10851         } else {
10852           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
10853                                 ->getCanonicalDecl();
10854           Pos = ParamPositions[PVD];
10855           ParmTy = PVD->getType();
10856         }
10857         ParamAttrs[Pos].Alignment =
10858             (*NI)
10859                 ? (*NI)->EvaluateKnownConstInt(C)
10860                 : llvm::APSInt::getUnsigned(
10861                       C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy))
10862                           .getQuantity());
10863         ++NI;
10864       }
10865       // Mark linear parameters.
10866       auto SI = Attr->steps_begin();
10867       auto MI = Attr->modifiers_begin();
10868       for (const Expr *E : Attr->linears()) {
10869         E = E->IgnoreParenImpCasts();
10870         unsigned Pos;
10871         if (isa<CXXThisExpr>(E)) {
10872           Pos = ParamPositions[FD];
10873         } else {
10874           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
10875                                 ->getCanonicalDecl();
10876           Pos = ParamPositions[PVD];
10877         }
10878         ParamAttrTy &ParamAttr = ParamAttrs[Pos];
10879         ParamAttr.Kind = Linear;
10880         if (*SI) {
10881           Expr::EvalResult Result;
10882           if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) {
10883             if (const auto *DRE =
10884                     cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) {
10885               if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) {
10886                 ParamAttr.Kind = LinearWithVarStride;
10887                 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(
10888                     ParamPositions[StridePVD->getCanonicalDecl()]);
10889               }
10890             }
10891           } else {
10892             ParamAttr.StrideOrArg = Result.Val.getInt();
10893           }
10894         }
10895         ++SI;
10896         ++MI;
10897       }
10898       llvm::APSInt VLENVal;
10899       SourceLocation ExprLoc;
10900       const Expr *VLENExpr = Attr->getSimdlen();
10901       if (VLENExpr) {
10902         VLENVal = VLENExpr->EvaluateKnownConstInt(C);
10903         ExprLoc = VLENExpr->getExprLoc();
10904       }
10905       OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState();
10906       if (CGM.getTriple().getArch() == llvm::Triple::x86 ||
10907           CGM.getTriple().getArch() == llvm::Triple::x86_64) {
10908         emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State);
10909       } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) {
10910         unsigned VLEN = VLENVal.getExtValue();
10911         StringRef MangledName = Fn->getName();
10912         if (CGM.getTarget().hasFeature("sve"))
10913           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
10914                                          MangledName, 's', 128, Fn, ExprLoc);
10915         if (CGM.getTarget().hasFeature("neon"))
10916           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
10917                                          MangledName, 'n', 128, Fn, ExprLoc);
10918       }
10919     }
10920     FD = FD->getPreviousDecl();
10921   }
10922 }
10923 
10924 namespace {
10925 /// Cleanup action for doacross support.
10926 class DoacrossCleanupTy final : public EHScopeStack::Cleanup {
10927 public:
10928   static const int DoacrossFinArgs = 2;
10929 
10930 private:
10931   llvm::FunctionCallee RTLFn;
10932   llvm::Value *Args[DoacrossFinArgs];
10933 
10934 public:
10935   DoacrossCleanupTy(llvm::FunctionCallee RTLFn,
10936                     ArrayRef<llvm::Value *> CallArgs)
10937       : RTLFn(RTLFn) {
10938     assert(CallArgs.size() == DoacrossFinArgs);
10939     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
10940   }
10941   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
10942     if (!CGF.HaveInsertPoint())
10943       return;
10944     CGF.EmitRuntimeCall(RTLFn, Args);
10945   }
10946 };
10947 } // namespace
10948 
10949 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF,
10950                                        const OMPLoopDirective &D,
10951                                        ArrayRef<Expr *> NumIterations) {
10952   if (!CGF.HaveInsertPoint())
10953     return;
10954 
10955   ASTContext &C = CGM.getContext();
10956   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
10957   RecordDecl *RD;
10958   if (KmpDimTy.isNull()) {
10959     // Build struct kmp_dim {  // loop bounds info casted to kmp_int64
10960     //  kmp_int64 lo; // lower
10961     //  kmp_int64 up; // upper
10962     //  kmp_int64 st; // stride
10963     // };
10964     RD = C.buildImplicitRecord("kmp_dim");
10965     RD->startDefinition();
10966     addFieldToRecordDecl(C, RD, Int64Ty);
10967     addFieldToRecordDecl(C, RD, Int64Ty);
10968     addFieldToRecordDecl(C, RD, Int64Ty);
10969     RD->completeDefinition();
10970     KmpDimTy = C.getRecordType(RD);
10971   } else {
10972     RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl());
10973   }
10974   llvm::APInt Size(/*numBits=*/32, NumIterations.size());
10975   QualType ArrayTy =
10976       C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0);
10977 
10978   Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims");
10979   CGF.EmitNullInitialization(DimsAddr, ArrayTy);
10980   enum { LowerFD = 0, UpperFD, StrideFD };
10981   // Fill dims with data.
10982   for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) {
10983     LValue DimsLVal = CGF.MakeAddrLValue(
10984         CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy);
10985     // dims.upper = num_iterations;
10986     LValue UpperLVal = CGF.EmitLValueForField(
10987         DimsLVal, *std::next(RD->field_begin(), UpperFD));
10988     llvm::Value *NumIterVal =
10989         CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]),
10990                                  D.getNumIterations()->getType(), Int64Ty,
10991                                  D.getNumIterations()->getExprLoc());
10992     CGF.EmitStoreOfScalar(NumIterVal, UpperLVal);
10993     // dims.stride = 1;
10994     LValue StrideLVal = CGF.EmitLValueForField(
10995         DimsLVal, *std::next(RD->field_begin(), StrideFD));
10996     CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1),
10997                           StrideLVal);
10998   }
10999 
11000   // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid,
11001   // kmp_int32 num_dims, struct kmp_dim * dims);
11002   llvm::Value *Args[] = {
11003       emitUpdateLocation(CGF, D.getBeginLoc()),
11004       getThreadID(CGF, D.getBeginLoc()),
11005       llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()),
11006       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11007           CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(),
11008           CGM.VoidPtrTy)};
11009 
11010   llvm::FunctionCallee RTLFn =
11011       createRuntimeFunction(OMPRTL__kmpc_doacross_init);
11012   CGF.EmitRuntimeCall(RTLFn, Args);
11013   llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = {
11014       emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())};
11015   llvm::FunctionCallee FiniRTLFn =
11016       createRuntimeFunction(OMPRTL__kmpc_doacross_fini);
11017   CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
11018                                              llvm::makeArrayRef(FiniArgs));
11019 }
11020 
11021 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
11022                                           const OMPDependClause *C) {
11023   QualType Int64Ty =
11024       CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
11025   llvm::APInt Size(/*numBits=*/32, C->getNumLoops());
11026   QualType ArrayTy = CGM.getContext().getConstantArrayType(
11027       Int64Ty, Size, nullptr, ArrayType::Normal, 0);
11028   Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr");
11029   for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) {
11030     const Expr *CounterVal = C->getLoopData(I);
11031     assert(CounterVal);
11032     llvm::Value *CntVal = CGF.EmitScalarConversion(
11033         CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty,
11034         CounterVal->getExprLoc());
11035     CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I),
11036                           /*Volatile=*/false, Int64Ty);
11037   }
11038   llvm::Value *Args[] = {
11039       emitUpdateLocation(CGF, C->getBeginLoc()),
11040       getThreadID(CGF, C->getBeginLoc()),
11041       CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()};
11042   llvm::FunctionCallee RTLFn;
11043   if (C->getDependencyKind() == OMPC_DEPEND_source) {
11044     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post);
11045   } else {
11046     assert(C->getDependencyKind() == OMPC_DEPEND_sink);
11047     RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait);
11048   }
11049   CGF.EmitRuntimeCall(RTLFn, Args);
11050 }
11051 
11052 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc,
11053                                llvm::FunctionCallee Callee,
11054                                ArrayRef<llvm::Value *> Args) const {
11055   assert(Loc.isValid() && "Outlined function call location must be valid.");
11056   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
11057 
11058   if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) {
11059     if (Fn->doesNotThrow()) {
11060       CGF.EmitNounwindRuntimeCall(Fn, Args);
11061       return;
11062     }
11063   }
11064   CGF.EmitRuntimeCall(Callee, Args);
11065 }
11066 
11067 void CGOpenMPRuntime::emitOutlinedFunctionCall(
11068     CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn,
11069     ArrayRef<llvm::Value *> Args) const {
11070   emitCall(CGF, Loc, OutlinedFn, Args);
11071 }
11072 
11073 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) {
11074   if (const auto *FD = dyn_cast<FunctionDecl>(D))
11075     if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD))
11076       HasEmittedDeclareTargetRegion = true;
11077 }
11078 
11079 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF,
11080                                              const VarDecl *NativeParam,
11081                                              const VarDecl *TargetParam) const {
11082   return CGF.GetAddrOfLocalVar(NativeParam);
11083 }
11084 
11085 namespace {
11086 /// Cleanup action for allocate support.
11087 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup {
11088 public:
11089   static const int CleanupArgs = 3;
11090 
11091 private:
11092   llvm::FunctionCallee RTLFn;
11093   llvm::Value *Args[CleanupArgs];
11094 
11095 public:
11096   OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn,
11097                        ArrayRef<llvm::Value *> CallArgs)
11098       : RTLFn(RTLFn) {
11099     assert(CallArgs.size() == CleanupArgs &&
11100            "Size of arguments does not match.");
11101     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
11102   }
11103   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
11104     if (!CGF.HaveInsertPoint())
11105       return;
11106     CGF.EmitRuntimeCall(RTLFn, Args);
11107   }
11108 };
11109 } // namespace
11110 
11111 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF,
11112                                                    const VarDecl *VD) {
11113   if (!VD)
11114     return Address::invalid();
11115   const VarDecl *CVD = VD->getCanonicalDecl();
11116   if (!CVD->hasAttr<OMPAllocateDeclAttr>())
11117     return Address::invalid();
11118   const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
11119   // Use the default allocation.
11120   if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc &&
11121       !AA->getAllocator())
11122     return Address::invalid();
11123   llvm::Value *Size;
11124   CharUnits Align = CGM.getContext().getDeclAlign(CVD);
11125   if (CVD->getType()->isVariablyModifiedType()) {
11126     Size = CGF.getTypeSize(CVD->getType());
11127     // Align the size: ((size + align - 1) / align) * align
11128     Size = CGF.Builder.CreateNUWAdd(
11129         Size, CGM.getSize(Align - CharUnits::fromQuantity(1)));
11130     Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align));
11131     Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align));
11132   } else {
11133     CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType());
11134     Size = CGM.getSize(Sz.alignTo(Align));
11135   }
11136   llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc());
11137   assert(AA->getAllocator() &&
11138          "Expected allocator expression for non-default allocator.");
11139   llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator());
11140   // According to the standard, the original allocator type is a enum (integer).
11141   // Convert to pointer type, if required.
11142   if (Allocator->getType()->isIntegerTy())
11143     Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy);
11144   else if (Allocator->getType()->isPointerTy())
11145     Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator,
11146                                                                 CGM.VoidPtrTy);
11147   llvm::Value *Args[] = {ThreadID, Size, Allocator};
11148 
11149   llvm::Value *Addr =
11150       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args,
11151                           CVD->getName() + ".void.addr");
11152   llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr,
11153                                                               Allocator};
11154   llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free);
11155 
11156   CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
11157                                                 llvm::makeArrayRef(FiniArgs));
11158   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
11159       Addr,
11160       CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())),
11161       CVD->getName() + ".addr");
11162   return Address(Addr, Align);
11163 }
11164 
11165 /// Checks current context and returns true if it matches the context selector.
11166 template <OMPDeclareVariantAttr::CtxSelectorSetType CtxSet,
11167           OMPDeclareVariantAttr::CtxSelectorType Ctx>
11168 static bool checkContext(const OMPDeclareVariantAttr *A) {
11169   assert(CtxSet != OMPDeclareVariantAttr::CtxSetUnknown &&
11170          Ctx != OMPDeclareVariantAttr::CtxUnknown &&
11171          "Unknown context selector or context selector set.");
11172   return false;
11173 }
11174 
11175 /// Checks for implementation={vendor(<vendor>)} context selector.
11176 /// \returns true iff <vendor>="llvm", false otherwise.
11177 template <>
11178 bool checkContext<OMPDeclareVariantAttr::CtxSetImplementation,
11179                   OMPDeclareVariantAttr::CtxVendor>(
11180     const OMPDeclareVariantAttr *A) {
11181   return llvm::all_of(A->implVendors(),
11182                       [](StringRef S) { return !S.compare_lower("llvm"); });
11183 }
11184 
11185 static bool greaterCtxScore(ASTContext &Ctx, const Expr *LHS, const Expr *RHS) {
11186   // If both scores are unknown, choose the very first one.
11187   if (!LHS && !RHS)
11188     return true;
11189   // If only one is known, return this one.
11190   if (LHS && !RHS)
11191     return true;
11192   if (!LHS && RHS)
11193     return false;
11194   llvm::APSInt LHSVal = LHS->EvaluateKnownConstInt(Ctx);
11195   llvm::APSInt RHSVal = RHS->EvaluateKnownConstInt(Ctx);
11196   return llvm::APSInt::compareValues(LHSVal, RHSVal) >= 0;
11197 }
11198 
11199 namespace {
11200 /// Comparator for the priority queue for context selector.
11201 class OMPDeclareVariantAttrComparer
11202     : public std::greater<const OMPDeclareVariantAttr *> {
11203 private:
11204   ASTContext &Ctx;
11205 
11206 public:
11207   OMPDeclareVariantAttrComparer(ASTContext &Ctx) : Ctx(Ctx) {}
11208   bool operator()(const OMPDeclareVariantAttr *LHS,
11209                   const OMPDeclareVariantAttr *RHS) const {
11210     const Expr *LHSExpr = nullptr;
11211     const Expr *RHSExpr = nullptr;
11212     if (LHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified)
11213       LHSExpr = LHS->getScore();
11214     if (RHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified)
11215       RHSExpr = RHS->getScore();
11216     return greaterCtxScore(Ctx, LHSExpr, RHSExpr);
11217   }
11218 };
11219 } // anonymous namespace
11220 
11221 /// Finds the variant function that matches current context with its context
11222 /// selector.
11223 static const FunctionDecl *getDeclareVariantFunction(ASTContext &Ctx,
11224                                                      const FunctionDecl *FD) {
11225   if (!FD->hasAttrs() || !FD->hasAttr<OMPDeclareVariantAttr>())
11226     return FD;
11227   // Iterate through all DeclareVariant attributes and check context selectors.
11228   auto &&Comparer = [&Ctx](const OMPDeclareVariantAttr *LHS,
11229                            const OMPDeclareVariantAttr *RHS) {
11230     const Expr *LHSExpr = nullptr;
11231     const Expr *RHSExpr = nullptr;
11232     if (LHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified)
11233       LHSExpr = LHS->getScore();
11234     if (RHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified)
11235       RHSExpr = RHS->getScore();
11236     return greaterCtxScore(Ctx, LHSExpr, RHSExpr);
11237   };
11238   const OMPDeclareVariantAttr *TopMostAttr = nullptr;
11239   for (const auto *A : FD->specific_attrs<OMPDeclareVariantAttr>()) {
11240     const OMPDeclareVariantAttr *SelectedAttr = nullptr;
11241     switch (A->getCtxSelectorSet()) {
11242     case OMPDeclareVariantAttr::CtxSetImplementation:
11243       switch (A->getCtxSelector()) {
11244       case OMPDeclareVariantAttr::CtxVendor:
11245         if (checkContext<OMPDeclareVariantAttr::CtxSetImplementation,
11246                          OMPDeclareVariantAttr::CtxVendor>(A))
11247           SelectedAttr = A;
11248         break;
11249       case OMPDeclareVariantAttr::CtxUnknown:
11250         llvm_unreachable(
11251             "Unknown context selector in implementation selector set.");
11252       }
11253       break;
11254     case OMPDeclareVariantAttr::CtxSetUnknown:
11255       llvm_unreachable("Unknown context selector set.");
11256     }
11257     // If the attribute matches the context, find the attribute with the highest
11258     // score.
11259     if (SelectedAttr && (!TopMostAttr || !Comparer(TopMostAttr, SelectedAttr)))
11260       TopMostAttr = SelectedAttr;
11261   }
11262   if (!TopMostAttr)
11263     return FD;
11264   return cast<FunctionDecl>(
11265       cast<DeclRefExpr>(TopMostAttr->getVariantFuncRef()->IgnoreParenImpCasts())
11266           ->getDecl());
11267 }
11268 
11269 bool CGOpenMPRuntime::emitDeclareVariant(GlobalDecl GD, bool IsForDefinition) {
11270   const auto *D = cast<FunctionDecl>(GD.getDecl());
11271   // If the original function is defined already, use its definition.
11272   StringRef MangledName = CGM.getMangledName(GD);
11273   llvm::GlobalValue *Orig = CGM.GetGlobalValue(MangledName);
11274   if (Orig && !Orig->isDeclaration())
11275     return false;
11276   const FunctionDecl *NewFD = getDeclareVariantFunction(CGM.getContext(), D);
11277   // Emit original function if it does not have declare variant attribute or the
11278   // context does not match.
11279   if (NewFD == D)
11280     return false;
11281   GlobalDecl NewGD = GD.getWithDecl(NewFD);
11282   if (tryEmitDeclareVariant(NewGD, GD, Orig, IsForDefinition)) {
11283     DeferredVariantFunction.erase(D);
11284     return true;
11285   }
11286   DeferredVariantFunction.insert(std::make_pair(D, std::make_pair(NewGD, GD)));
11287   return true;
11288 }
11289 
11290 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction(
11291     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11292     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11293   llvm_unreachable("Not supported in SIMD-only mode");
11294 }
11295 
11296 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction(
11297     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11298     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
11299   llvm_unreachable("Not supported in SIMD-only mode");
11300 }
11301 
11302 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction(
11303     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
11304     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
11305     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
11306     bool Tied, unsigned &NumberOfParts) {
11307   llvm_unreachable("Not supported in SIMD-only mode");
11308 }
11309 
11310 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF,
11311                                            SourceLocation Loc,
11312                                            llvm::Function *OutlinedFn,
11313                                            ArrayRef<llvm::Value *> CapturedVars,
11314                                            const Expr *IfCond) {
11315   llvm_unreachable("Not supported in SIMD-only mode");
11316 }
11317 
11318 void CGOpenMPSIMDRuntime::emitCriticalRegion(
11319     CodeGenFunction &CGF, StringRef CriticalName,
11320     const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc,
11321     const Expr *Hint) {
11322   llvm_unreachable("Not supported in SIMD-only mode");
11323 }
11324 
11325 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF,
11326                                            const RegionCodeGenTy &MasterOpGen,
11327                                            SourceLocation Loc) {
11328   llvm_unreachable("Not supported in SIMD-only mode");
11329 }
11330 
11331 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
11332                                             SourceLocation Loc) {
11333   llvm_unreachable("Not supported in SIMD-only mode");
11334 }
11335 
11336 void CGOpenMPSIMDRuntime::emitTaskgroupRegion(
11337     CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen,
11338     SourceLocation Loc) {
11339   llvm_unreachable("Not supported in SIMD-only mode");
11340 }
11341 
11342 void CGOpenMPSIMDRuntime::emitSingleRegion(
11343     CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen,
11344     SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars,
11345     ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs,
11346     ArrayRef<const Expr *> AssignmentOps) {
11347   llvm_unreachable("Not supported in SIMD-only mode");
11348 }
11349 
11350 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF,
11351                                             const RegionCodeGenTy &OrderedOpGen,
11352                                             SourceLocation Loc,
11353                                             bool IsThreads) {
11354   llvm_unreachable("Not supported in SIMD-only mode");
11355 }
11356 
11357 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF,
11358                                           SourceLocation Loc,
11359                                           OpenMPDirectiveKind Kind,
11360                                           bool EmitChecks,
11361                                           bool ForceSimpleCall) {
11362   llvm_unreachable("Not supported in SIMD-only mode");
11363 }
11364 
11365 void CGOpenMPSIMDRuntime::emitForDispatchInit(
11366     CodeGenFunction &CGF, SourceLocation Loc,
11367     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
11368     bool Ordered, const DispatchRTInput &DispatchValues) {
11369   llvm_unreachable("Not supported in SIMD-only mode");
11370 }
11371 
11372 void CGOpenMPSIMDRuntime::emitForStaticInit(
11373     CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind,
11374     const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) {
11375   llvm_unreachable("Not supported in SIMD-only mode");
11376 }
11377 
11378 void CGOpenMPSIMDRuntime::emitDistributeStaticInit(
11379     CodeGenFunction &CGF, SourceLocation Loc,
11380     OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) {
11381   llvm_unreachable("Not supported in SIMD-only mode");
11382 }
11383 
11384 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
11385                                                      SourceLocation Loc,
11386                                                      unsigned IVSize,
11387                                                      bool IVSigned) {
11388   llvm_unreachable("Not supported in SIMD-only mode");
11389 }
11390 
11391 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF,
11392                                               SourceLocation Loc,
11393                                               OpenMPDirectiveKind DKind) {
11394   llvm_unreachable("Not supported in SIMD-only mode");
11395 }
11396 
11397 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF,
11398                                               SourceLocation Loc,
11399                                               unsigned IVSize, bool IVSigned,
11400                                               Address IL, Address LB,
11401                                               Address UB, Address ST) {
11402   llvm_unreachable("Not supported in SIMD-only mode");
11403 }
11404 
11405 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
11406                                                llvm::Value *NumThreads,
11407                                                SourceLocation Loc) {
11408   llvm_unreachable("Not supported in SIMD-only mode");
11409 }
11410 
11411 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF,
11412                                              OpenMPProcBindClauseKind ProcBind,
11413                                              SourceLocation Loc) {
11414   llvm_unreachable("Not supported in SIMD-only mode");
11415 }
11416 
11417 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
11418                                                     const VarDecl *VD,
11419                                                     Address VDAddr,
11420                                                     SourceLocation Loc) {
11421   llvm_unreachable("Not supported in SIMD-only mode");
11422 }
11423 
11424 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition(
11425     const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit,
11426     CodeGenFunction *CGF) {
11427   llvm_unreachable("Not supported in SIMD-only mode");
11428 }
11429 
11430 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate(
11431     CodeGenFunction &CGF, QualType VarType, StringRef Name) {
11432   llvm_unreachable("Not supported in SIMD-only mode");
11433 }
11434 
11435 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF,
11436                                     ArrayRef<const Expr *> Vars,
11437                                     SourceLocation Loc) {
11438   llvm_unreachable("Not supported in SIMD-only mode");
11439 }
11440 
11441 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
11442                                        const OMPExecutableDirective &D,
11443                                        llvm::Function *TaskFunction,
11444                                        QualType SharedsTy, Address Shareds,
11445                                        const Expr *IfCond,
11446                                        const OMPTaskDataTy &Data) {
11447   llvm_unreachable("Not supported in SIMD-only mode");
11448 }
11449 
11450 void CGOpenMPSIMDRuntime::emitTaskLoopCall(
11451     CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D,
11452     llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds,
11453     const Expr *IfCond, const OMPTaskDataTy &Data) {
11454   llvm_unreachable("Not supported in SIMD-only mode");
11455 }
11456 
11457 void CGOpenMPSIMDRuntime::emitReduction(
11458     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates,
11459     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
11460     ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) {
11461   assert(Options.SimpleReduction && "Only simple reduction is expected.");
11462   CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs,
11463                                  ReductionOps, Options);
11464 }
11465 
11466 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit(
11467     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
11468     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
11469   llvm_unreachable("Not supported in SIMD-only mode");
11470 }
11471 
11472 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
11473                                                   SourceLocation Loc,
11474                                                   ReductionCodeGen &RCG,
11475                                                   unsigned N) {
11476   llvm_unreachable("Not supported in SIMD-only mode");
11477 }
11478 
11479 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF,
11480                                                   SourceLocation Loc,
11481                                                   llvm::Value *ReductionsPtr,
11482                                                   LValue SharedLVal) {
11483   llvm_unreachable("Not supported in SIMD-only mode");
11484 }
11485 
11486 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
11487                                            SourceLocation Loc) {
11488   llvm_unreachable("Not supported in SIMD-only mode");
11489 }
11490 
11491 void CGOpenMPSIMDRuntime::emitCancellationPointCall(
11492     CodeGenFunction &CGF, SourceLocation Loc,
11493     OpenMPDirectiveKind CancelRegion) {
11494   llvm_unreachable("Not supported in SIMD-only mode");
11495 }
11496 
11497 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF,
11498                                          SourceLocation Loc, const Expr *IfCond,
11499                                          OpenMPDirectiveKind CancelRegion) {
11500   llvm_unreachable("Not supported in SIMD-only mode");
11501 }
11502 
11503 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction(
11504     const OMPExecutableDirective &D, StringRef ParentName,
11505     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
11506     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
11507   llvm_unreachable("Not supported in SIMD-only mode");
11508 }
11509 
11510 void CGOpenMPSIMDRuntime::emitTargetCall(
11511     CodeGenFunction &CGF, const OMPExecutableDirective &D,
11512     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
11513     const Expr *Device,
11514     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
11515                                      const OMPLoopDirective &D)>
11516         SizeEmitter) {
11517   llvm_unreachable("Not supported in SIMD-only mode");
11518 }
11519 
11520 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) {
11521   llvm_unreachable("Not supported in SIMD-only mode");
11522 }
11523 
11524 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
11525   llvm_unreachable("Not supported in SIMD-only mode");
11526 }
11527 
11528 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) {
11529   return false;
11530 }
11531 
11532 llvm::Function *CGOpenMPSIMDRuntime::emitRegistrationFunction() {
11533   return nullptr;
11534 }
11535 
11536 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF,
11537                                         const OMPExecutableDirective &D,
11538                                         SourceLocation Loc,
11539                                         llvm::Function *OutlinedFn,
11540                                         ArrayRef<llvm::Value *> CapturedVars) {
11541   llvm_unreachable("Not supported in SIMD-only mode");
11542 }
11543 
11544 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
11545                                              const Expr *NumTeams,
11546                                              const Expr *ThreadLimit,
11547                                              SourceLocation Loc) {
11548   llvm_unreachable("Not supported in SIMD-only mode");
11549 }
11550 
11551 void CGOpenMPSIMDRuntime::emitTargetDataCalls(
11552     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
11553     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
11554   llvm_unreachable("Not supported in SIMD-only mode");
11555 }
11556 
11557 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall(
11558     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
11559     const Expr *Device) {
11560   llvm_unreachable("Not supported in SIMD-only mode");
11561 }
11562 
11563 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF,
11564                                            const OMPLoopDirective &D,
11565                                            ArrayRef<Expr *> NumIterations) {
11566   llvm_unreachable("Not supported in SIMD-only mode");
11567 }
11568 
11569 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
11570                                               const OMPDependClause *C) {
11571   llvm_unreachable("Not supported in SIMD-only mode");
11572 }
11573 
11574 const VarDecl *
11575 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD,
11576                                         const VarDecl *NativeParam) const {
11577   llvm_unreachable("Not supported in SIMD-only mode");
11578 }
11579 
11580 Address
11581 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF,
11582                                          const VarDecl *NativeParam,
11583                                          const VarDecl *TargetParam) const {
11584   llvm_unreachable("Not supported in SIMD-only mode");
11585 }
11586