1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This provides a class for OpenMP runtime code generation.
10 //
11 //===----------------------------------------------------------------------===//
12 
13 #include "CGOpenMPRuntime.h"
14 #include "CGCXXABI.h"
15 #include "CGCleanup.h"
16 #include "CGRecordLayout.h"
17 #include "CodeGenFunction.h"
18 #include "TargetInfo.h"
19 #include "clang/AST/APValue.h"
20 #include "clang/AST/Attr.h"
21 #include "clang/AST/Decl.h"
22 #include "clang/AST/OpenMPClause.h"
23 #include "clang/AST/StmtOpenMP.h"
24 #include "clang/AST/StmtVisitor.h"
25 #include "clang/Basic/BitmaskEnum.h"
26 #include "clang/Basic/FileManager.h"
27 #include "clang/Basic/OpenMPKinds.h"
28 #include "clang/Basic/SourceManager.h"
29 #include "clang/CodeGen/ConstantInitBuilder.h"
30 #include "llvm/ADT/ArrayRef.h"
31 #include "llvm/ADT/SetOperations.h"
32 #include "llvm/ADT/SmallBitVector.h"
33 #include "llvm/ADT/StringExtras.h"
34 #include "llvm/Bitcode/BitcodeReader.h"
35 #include "llvm/IR/Constants.h"
36 #include "llvm/IR/DerivedTypes.h"
37 #include "llvm/IR/GlobalValue.h"
38 #include "llvm/IR/InstrTypes.h"
39 #include "llvm/IR/Value.h"
40 #include "llvm/Support/AtomicOrdering.h"
41 #include "llvm/Support/Format.h"
42 #include "llvm/Support/raw_ostream.h"
43 #include <cassert>
44 #include <numeric>
45 
46 using namespace clang;
47 using namespace CodeGen;
48 using namespace llvm::omp;
49 
50 namespace {
51 /// Base class for handling code generation inside OpenMP regions.
52 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
53 public:
54   /// Kinds of OpenMP regions used in codegen.
55   enum CGOpenMPRegionKind {
56     /// Region with outlined function for standalone 'parallel'
57     /// directive.
58     ParallelOutlinedRegion,
59     /// Region with outlined function for standalone 'task' directive.
60     TaskOutlinedRegion,
61     /// Region for constructs that do not require function outlining,
62     /// like 'for', 'sections', 'atomic' etc. directives.
63     InlinedRegion,
64     /// Region with outlined function for standalone 'target' directive.
65     TargetRegion,
66   };
67 
68   CGOpenMPRegionInfo(const CapturedStmt &CS,
69                      const CGOpenMPRegionKind RegionKind,
70                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
71                      bool HasCancel)
72       : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
73         CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {}
74 
75   CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
76                      const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind,
77                      bool HasCancel)
78       : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen),
79         Kind(Kind), HasCancel(HasCancel) {}
80 
81   /// Get a variable or parameter for storing global thread id
82   /// inside OpenMP construct.
83   virtual const VarDecl *getThreadIDVariable() const = 0;
84 
85   /// Emit the captured statement body.
86   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
87 
88   /// Get an LValue for the current ThreadID variable.
89   /// \return LValue for thread id variable. This LValue always has type int32*.
90   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
91 
92   virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {}
93 
94   CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
95 
96   OpenMPDirectiveKind getDirectiveKind() const { return Kind; }
97 
98   bool hasCancel() const { return HasCancel; }
99 
100   static bool classof(const CGCapturedStmtInfo *Info) {
101     return Info->getKind() == CR_OpenMP;
102   }
103 
104   ~CGOpenMPRegionInfo() override = default;
105 
106 protected:
107   CGOpenMPRegionKind RegionKind;
108   RegionCodeGenTy CodeGen;
109   OpenMPDirectiveKind Kind;
110   bool HasCancel;
111 };
112 
113 /// API for captured statement code generation in OpenMP constructs.
114 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo {
115 public:
116   CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
117                              const RegionCodeGenTy &CodeGen,
118                              OpenMPDirectiveKind Kind, bool HasCancel,
119                              StringRef HelperName)
120       : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind,
121                            HasCancel),
122         ThreadIDVar(ThreadIDVar), HelperName(HelperName) {
123     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
124   }
125 
126   /// Get a variable or parameter for storing global thread id
127   /// inside OpenMP construct.
128   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
129 
130   /// Get the name of the capture helper.
131   StringRef getHelperName() const override { return HelperName; }
132 
133   static bool classof(const CGCapturedStmtInfo *Info) {
134     return CGOpenMPRegionInfo::classof(Info) &&
135            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
136                ParallelOutlinedRegion;
137   }
138 
139 private:
140   /// A variable or parameter storing global thread id for OpenMP
141   /// constructs.
142   const VarDecl *ThreadIDVar;
143   StringRef HelperName;
144 };
145 
146 /// API for captured statement code generation in OpenMP constructs.
147 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo {
148 public:
149   class UntiedTaskActionTy final : public PrePostActionTy {
150     bool Untied;
151     const VarDecl *PartIDVar;
152     const RegionCodeGenTy UntiedCodeGen;
153     llvm::SwitchInst *UntiedSwitch = nullptr;
154 
155   public:
156     UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar,
157                        const RegionCodeGenTy &UntiedCodeGen)
158         : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {}
159     void Enter(CodeGenFunction &CGF) override {
160       if (Untied) {
161         // Emit task switching point.
162         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
163             CGF.GetAddrOfLocalVar(PartIDVar),
164             PartIDVar->getType()->castAs<PointerType>());
165         llvm::Value *Res =
166             CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation());
167         llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done.");
168         UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB);
169         CGF.EmitBlock(DoneBB);
170         CGF.EmitBranchThroughCleanup(CGF.ReturnBlock);
171         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
172         UntiedSwitch->addCase(CGF.Builder.getInt32(0),
173                               CGF.Builder.GetInsertBlock());
174         emitUntiedSwitch(CGF);
175       }
176     }
177     void emitUntiedSwitch(CodeGenFunction &CGF) const {
178       if (Untied) {
179         LValue PartIdLVal = CGF.EmitLoadOfPointerLValue(
180             CGF.GetAddrOfLocalVar(PartIDVar),
181             PartIDVar->getType()->castAs<PointerType>());
182         CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
183                               PartIdLVal);
184         UntiedCodeGen(CGF);
185         CodeGenFunction::JumpDest CurPoint =
186             CGF.getJumpDestInCurrentScope(".untied.next.");
187         CGF.EmitBranch(CGF.ReturnBlock.getBlock());
188         CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp."));
189         UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()),
190                               CGF.Builder.GetInsertBlock());
191         CGF.EmitBranchThroughCleanup(CurPoint);
192         CGF.EmitBlock(CurPoint.getBlock());
193       }
194     }
195     unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); }
196   };
197   CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
198                                  const VarDecl *ThreadIDVar,
199                                  const RegionCodeGenTy &CodeGen,
200                                  OpenMPDirectiveKind Kind, bool HasCancel,
201                                  const UntiedTaskActionTy &Action)
202       : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel),
203         ThreadIDVar(ThreadIDVar), Action(Action) {
204     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
205   }
206 
207   /// Get a variable or parameter for storing global thread id
208   /// inside OpenMP construct.
209   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
210 
211   /// Get an LValue for the current ThreadID variable.
212   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
213 
214   /// Get the name of the capture helper.
215   StringRef getHelperName() const override { return ".omp_outlined."; }
216 
217   void emitUntiedSwitch(CodeGenFunction &CGF) override {
218     Action.emitUntiedSwitch(CGF);
219   }
220 
221   static bool classof(const CGCapturedStmtInfo *Info) {
222     return CGOpenMPRegionInfo::classof(Info) &&
223            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
224                TaskOutlinedRegion;
225   }
226 
227 private:
228   /// A variable or parameter storing global thread id for OpenMP
229   /// constructs.
230   const VarDecl *ThreadIDVar;
231   /// Action for emitting code for untied tasks.
232   const UntiedTaskActionTy &Action;
233 };
234 
235 /// API for inlined captured statement code generation in OpenMP
236 /// constructs.
237 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
238 public:
239   CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
240                             const RegionCodeGenTy &CodeGen,
241                             OpenMPDirectiveKind Kind, bool HasCancel)
242       : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel),
243         OldCSI(OldCSI),
244         OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
245 
246   // Retrieve the value of the context parameter.
247   llvm::Value *getContextValue() const override {
248     if (OuterRegionInfo)
249       return OuterRegionInfo->getContextValue();
250     llvm_unreachable("No context value for inlined OpenMP region");
251   }
252 
253   void setContextValue(llvm::Value *V) override {
254     if (OuterRegionInfo) {
255       OuterRegionInfo->setContextValue(V);
256       return;
257     }
258     llvm_unreachable("No context value for inlined OpenMP region");
259   }
260 
261   /// Lookup the captured field decl for a variable.
262   const FieldDecl *lookup(const VarDecl *VD) const override {
263     if (OuterRegionInfo)
264       return OuterRegionInfo->lookup(VD);
265     // If there is no outer outlined region,no need to lookup in a list of
266     // captured variables, we can use the original one.
267     return nullptr;
268   }
269 
270   FieldDecl *getThisFieldDecl() const override {
271     if (OuterRegionInfo)
272       return OuterRegionInfo->getThisFieldDecl();
273     return nullptr;
274   }
275 
276   /// Get a variable or parameter for storing global thread id
277   /// inside OpenMP construct.
278   const VarDecl *getThreadIDVariable() const override {
279     if (OuterRegionInfo)
280       return OuterRegionInfo->getThreadIDVariable();
281     return nullptr;
282   }
283 
284   /// Get an LValue for the current ThreadID variable.
285   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override {
286     if (OuterRegionInfo)
287       return OuterRegionInfo->getThreadIDVariableLValue(CGF);
288     llvm_unreachable("No LValue for inlined OpenMP construct");
289   }
290 
291   /// Get the name of the capture helper.
292   StringRef getHelperName() const override {
293     if (auto *OuterRegionInfo = getOldCSI())
294       return OuterRegionInfo->getHelperName();
295     llvm_unreachable("No helper name for inlined OpenMP construct");
296   }
297 
298   void emitUntiedSwitch(CodeGenFunction &CGF) override {
299     if (OuterRegionInfo)
300       OuterRegionInfo->emitUntiedSwitch(CGF);
301   }
302 
303   CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
304 
305   static bool classof(const CGCapturedStmtInfo *Info) {
306     return CGOpenMPRegionInfo::classof(Info) &&
307            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion;
308   }
309 
310   ~CGOpenMPInlinedRegionInfo() override = default;
311 
312 private:
313   /// CodeGen info about outer OpenMP region.
314   CodeGenFunction::CGCapturedStmtInfo *OldCSI;
315   CGOpenMPRegionInfo *OuterRegionInfo;
316 };
317 
318 /// API for captured statement code generation in OpenMP target
319 /// constructs. For this captures, implicit parameters are used instead of the
320 /// captured fields. The name of the target region has to be unique in a given
321 /// application so it is provided by the client, because only the client has
322 /// the information to generate that.
323 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo {
324 public:
325   CGOpenMPTargetRegionInfo(const CapturedStmt &CS,
326                            const RegionCodeGenTy &CodeGen, StringRef HelperName)
327       : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target,
328                            /*HasCancel=*/false),
329         HelperName(HelperName) {}
330 
331   /// This is unused for target regions because each starts executing
332   /// with a single thread.
333   const VarDecl *getThreadIDVariable() const override { return nullptr; }
334 
335   /// Get the name of the capture helper.
336   StringRef getHelperName() const override { return HelperName; }
337 
338   static bool classof(const CGCapturedStmtInfo *Info) {
339     return CGOpenMPRegionInfo::classof(Info) &&
340            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion;
341   }
342 
343 private:
344   StringRef HelperName;
345 };
346 
347 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) {
348   llvm_unreachable("No codegen for expressions");
349 }
350 /// API for generation of expressions captured in a innermost OpenMP
351 /// region.
352 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo {
353 public:
354   CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS)
355       : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen,
356                                   OMPD_unknown,
357                                   /*HasCancel=*/false),
358         PrivScope(CGF) {
359     // Make sure the globals captured in the provided statement are local by
360     // using the privatization logic. We assume the same variable is not
361     // captured more than once.
362     for (const auto &C : CS.captures()) {
363       if (!C.capturesVariable() && !C.capturesVariableByCopy())
364         continue;
365 
366       const VarDecl *VD = C.getCapturedVar();
367       if (VD->isLocalVarDeclOrParm())
368         continue;
369 
370       DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD),
371                       /*RefersToEnclosingVariableOrCapture=*/false,
372                       VD->getType().getNonReferenceType(), VK_LValue,
373                       C.getLocation());
374       PrivScope.addPrivate(VD, CGF.EmitLValue(&DRE).getAddress(CGF));
375     }
376     (void)PrivScope.Privatize();
377   }
378 
379   /// Lookup the captured field decl for a variable.
380   const FieldDecl *lookup(const VarDecl *VD) const override {
381     if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD))
382       return FD;
383     return nullptr;
384   }
385 
386   /// Emit the captured statement body.
387   void EmitBody(CodeGenFunction &CGF, const Stmt *S) override {
388     llvm_unreachable("No body for expressions");
389   }
390 
391   /// Get a variable or parameter for storing global thread id
392   /// inside OpenMP construct.
393   const VarDecl *getThreadIDVariable() const override {
394     llvm_unreachable("No thread id for expressions");
395   }
396 
397   /// Get the name of the capture helper.
398   StringRef getHelperName() const override {
399     llvm_unreachable("No helper name for expressions");
400   }
401 
402   static bool classof(const CGCapturedStmtInfo *Info) { return false; }
403 
404 private:
405   /// Private scope to capture global variables.
406   CodeGenFunction::OMPPrivateScope PrivScope;
407 };
408 
409 /// RAII for emitting code of OpenMP constructs.
410 class InlinedOpenMPRegionRAII {
411   CodeGenFunction &CGF;
412   llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields;
413   FieldDecl *LambdaThisCaptureField = nullptr;
414   const CodeGen::CGBlockInfo *BlockInfo = nullptr;
415   bool NoInheritance = false;
416 
417 public:
418   /// Constructs region for combined constructs.
419   /// \param CodeGen Code generation sequence for combined directives. Includes
420   /// a list of functions used for code generation of implicitly inlined
421   /// regions.
422   InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen,
423                           OpenMPDirectiveKind Kind, bool HasCancel,
424                           bool NoInheritance = true)
425       : CGF(CGF), NoInheritance(NoInheritance) {
426     // Start emission for the construct.
427     CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(
428         CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel);
429     if (NoInheritance) {
430       std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
431       LambdaThisCaptureField = CGF.LambdaThisCaptureField;
432       CGF.LambdaThisCaptureField = nullptr;
433       BlockInfo = CGF.BlockInfo;
434       CGF.BlockInfo = nullptr;
435     }
436   }
437 
438   ~InlinedOpenMPRegionRAII() {
439     // Restore original CapturedStmtInfo only if we're done with code emission.
440     auto *OldCSI =
441         cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
442     delete CGF.CapturedStmtInfo;
443     CGF.CapturedStmtInfo = OldCSI;
444     if (NoInheritance) {
445       std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields);
446       CGF.LambdaThisCaptureField = LambdaThisCaptureField;
447       CGF.BlockInfo = BlockInfo;
448     }
449   }
450 };
451 
452 /// Values for bit flags used in the ident_t to describe the fields.
453 /// All enumeric elements are named and described in accordance with the code
454 /// from https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h
455 enum OpenMPLocationFlags : unsigned {
456   /// Use trampoline for internal microtask.
457   OMP_IDENT_IMD = 0x01,
458   /// Use c-style ident structure.
459   OMP_IDENT_KMPC = 0x02,
460   /// Atomic reduction option for kmpc_reduce.
461   OMP_ATOMIC_REDUCE = 0x10,
462   /// Explicit 'barrier' directive.
463   OMP_IDENT_BARRIER_EXPL = 0x20,
464   /// Implicit barrier in code.
465   OMP_IDENT_BARRIER_IMPL = 0x40,
466   /// Implicit barrier in 'for' directive.
467   OMP_IDENT_BARRIER_IMPL_FOR = 0x40,
468   /// Implicit barrier in 'sections' directive.
469   OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0,
470   /// Implicit barrier in 'single' directive.
471   OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140,
472   /// Call of __kmp_for_static_init for static loop.
473   OMP_IDENT_WORK_LOOP = 0x200,
474   /// Call of __kmp_for_static_init for sections.
475   OMP_IDENT_WORK_SECTIONS = 0x400,
476   /// Call of __kmp_for_static_init for distribute.
477   OMP_IDENT_WORK_DISTRIBUTE = 0x800,
478   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE)
479 };
480 
481 namespace {
482 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
483 /// Values for bit flags for marking which requires clauses have been used.
484 enum OpenMPOffloadingRequiresDirFlags : int64_t {
485   /// flag undefined.
486   OMP_REQ_UNDEFINED               = 0x000,
487   /// no requires clause present.
488   OMP_REQ_NONE                    = 0x001,
489   /// reverse_offload clause.
490   OMP_REQ_REVERSE_OFFLOAD         = 0x002,
491   /// unified_address clause.
492   OMP_REQ_UNIFIED_ADDRESS         = 0x004,
493   /// unified_shared_memory clause.
494   OMP_REQ_UNIFIED_SHARED_MEMORY   = 0x008,
495   /// dynamic_allocators clause.
496   OMP_REQ_DYNAMIC_ALLOCATORS      = 0x010,
497   LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS)
498 };
499 
500 enum OpenMPOffloadingReservedDeviceIDs {
501   /// Device ID if the device was not defined, runtime should get it
502   /// from environment variables in the spec.
503   OMP_DEVICEID_UNDEF = -1,
504 };
505 } // anonymous namespace
506 
507 /// Describes ident structure that describes a source location.
508 /// All descriptions are taken from
509 /// https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h
510 /// Original structure:
511 /// typedef struct ident {
512 ///    kmp_int32 reserved_1;   /**<  might be used in Fortran;
513 ///                                  see above  */
514 ///    kmp_int32 flags;        /**<  also f.flags; KMP_IDENT_xxx flags;
515 ///                                  KMP_IDENT_KMPC identifies this union
516 ///                                  member  */
517 ///    kmp_int32 reserved_2;   /**<  not really used in Fortran any more;
518 ///                                  see above */
519 ///#if USE_ITT_BUILD
520 ///                            /*  but currently used for storing
521 ///                                region-specific ITT */
522 ///                            /*  contextual information. */
523 ///#endif /* USE_ITT_BUILD */
524 ///    kmp_int32 reserved_3;   /**< source[4] in Fortran, do not use for
525 ///                                 C++  */
526 ///    char const *psource;    /**< String describing the source location.
527 ///                            The string is composed of semi-colon separated
528 //                             fields which describe the source file,
529 ///                            the function and a pair of line numbers that
530 ///                            delimit the construct.
531 ///                             */
532 /// } ident_t;
533 enum IdentFieldIndex {
534   /// might be used in Fortran
535   IdentField_Reserved_1,
536   /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member.
537   IdentField_Flags,
538   /// Not really used in Fortran any more
539   IdentField_Reserved_2,
540   /// Source[4] in Fortran, do not use for C++
541   IdentField_Reserved_3,
542   /// String describing the source location. The string is composed of
543   /// semi-colon separated fields which describe the source file, the function
544   /// and a pair of line numbers that delimit the construct.
545   IdentField_PSource
546 };
547 
548 /// Schedule types for 'omp for' loops (these enumerators are taken from
549 /// the enum sched_type in kmp.h).
550 enum OpenMPSchedType {
551   /// Lower bound for default (unordered) versions.
552   OMP_sch_lower = 32,
553   OMP_sch_static_chunked = 33,
554   OMP_sch_static = 34,
555   OMP_sch_dynamic_chunked = 35,
556   OMP_sch_guided_chunked = 36,
557   OMP_sch_runtime = 37,
558   OMP_sch_auto = 38,
559   /// static with chunk adjustment (e.g., simd)
560   OMP_sch_static_balanced_chunked = 45,
561   /// Lower bound for 'ordered' versions.
562   OMP_ord_lower = 64,
563   OMP_ord_static_chunked = 65,
564   OMP_ord_static = 66,
565   OMP_ord_dynamic_chunked = 67,
566   OMP_ord_guided_chunked = 68,
567   OMP_ord_runtime = 69,
568   OMP_ord_auto = 70,
569   OMP_sch_default = OMP_sch_static,
570   /// dist_schedule types
571   OMP_dist_sch_static_chunked = 91,
572   OMP_dist_sch_static = 92,
573   /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers.
574   /// Set if the monotonic schedule modifier was present.
575   OMP_sch_modifier_monotonic = (1 << 29),
576   /// Set if the nonmonotonic schedule modifier was present.
577   OMP_sch_modifier_nonmonotonic = (1 << 30),
578 };
579 
580 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP
581 /// region.
582 class CleanupTy final : public EHScopeStack::Cleanup {
583   PrePostActionTy *Action;
584 
585 public:
586   explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {}
587   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
588     if (!CGF.HaveInsertPoint())
589       return;
590     Action->Exit(CGF);
591   }
592 };
593 
594 } // anonymous namespace
595 
596 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const {
597   CodeGenFunction::RunCleanupsScope Scope(CGF);
598   if (PrePostAction) {
599     CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction);
600     Callback(CodeGen, CGF, *PrePostAction);
601   } else {
602     PrePostActionTy Action;
603     Callback(CodeGen, CGF, Action);
604   }
605 }
606 
607 /// Check if the combiner is a call to UDR combiner and if it is so return the
608 /// UDR decl used for reduction.
609 static const OMPDeclareReductionDecl *
610 getReductionInit(const Expr *ReductionOp) {
611   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
612     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
613       if (const auto *DRE =
614               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
615         if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl()))
616           return DRD;
617   return nullptr;
618 }
619 
620 static void emitInitWithReductionInitializer(CodeGenFunction &CGF,
621                                              const OMPDeclareReductionDecl *DRD,
622                                              const Expr *InitOp,
623                                              Address Private, Address Original,
624                                              QualType Ty) {
625   if (DRD->getInitializer()) {
626     std::pair<llvm::Function *, llvm::Function *> Reduction =
627         CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
628     const auto *CE = cast<CallExpr>(InitOp);
629     const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee());
630     const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts();
631     const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts();
632     const auto *LHSDRE =
633         cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr());
634     const auto *RHSDRE =
635         cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr());
636     CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
637     PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), Private);
638     PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), Original);
639     (void)PrivateScope.Privatize();
640     RValue Func = RValue::get(Reduction.second);
641     CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
642     CGF.EmitIgnoredExpr(InitOp);
643   } else {
644     llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty);
645     std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"});
646     auto *GV = new llvm::GlobalVariable(
647         CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true,
648         llvm::GlobalValue::PrivateLinkage, Init, Name);
649     LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty);
650     RValue InitRVal;
651     switch (CGF.getEvaluationKind(Ty)) {
652     case TEK_Scalar:
653       InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation());
654       break;
655     case TEK_Complex:
656       InitRVal =
657           RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation()));
658       break;
659     case TEK_Aggregate: {
660       OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_LValue);
661       CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, LV);
662       CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
663                            /*IsInitializer=*/false);
664       return;
665     }
666     }
667     OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_PRValue);
668     CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal);
669     CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(),
670                          /*IsInitializer=*/false);
671   }
672 }
673 
674 /// Emit initialization of arrays of complex types.
675 /// \param DestAddr Address of the array.
676 /// \param Type Type of array.
677 /// \param Init Initial expression of array.
678 /// \param SrcAddr Address of the original array.
679 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr,
680                                  QualType Type, bool EmitDeclareReductionInit,
681                                  const Expr *Init,
682                                  const OMPDeclareReductionDecl *DRD,
683                                  Address SrcAddr = Address::invalid()) {
684   // Perform element-by-element initialization.
685   QualType ElementTy;
686 
687   // Drill down to the base element type on both arrays.
688   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
689   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr);
690   if (DRD)
691     SrcAddr =
692         CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType());
693 
694   llvm::Value *SrcBegin = nullptr;
695   if (DRD)
696     SrcBegin = SrcAddr.getPointer();
697   llvm::Value *DestBegin = DestAddr.getPointer();
698   // Cast from pointer to array type to pointer to single element.
699   llvm::Value *DestEnd =
700       CGF.Builder.CreateGEP(DestAddr.getElementType(), DestBegin, NumElements);
701   // The basic structure here is a while-do loop.
702   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body");
703   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done");
704   llvm::Value *IsEmpty =
705       CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty");
706   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
707 
708   // Enter the loop body, making that address the current address.
709   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
710   CGF.EmitBlock(BodyBB);
711 
712   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
713 
714   llvm::PHINode *SrcElementPHI = nullptr;
715   Address SrcElementCurrent = Address::invalid();
716   if (DRD) {
717     SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2,
718                                           "omp.arraycpy.srcElementPast");
719     SrcElementPHI->addIncoming(SrcBegin, EntryBB);
720     SrcElementCurrent =
721         Address(SrcElementPHI, SrcAddr.getElementType(),
722                 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize));
723   }
724   llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI(
725       DestBegin->getType(), 2, "omp.arraycpy.destElementPast");
726   DestElementPHI->addIncoming(DestBegin, EntryBB);
727   Address DestElementCurrent =
728       Address(DestElementPHI, DestAddr.getElementType(),
729               DestAddr.getAlignment().alignmentOfArrayElement(ElementSize));
730 
731   // Emit copy.
732   {
733     CodeGenFunction::RunCleanupsScope InitScope(CGF);
734     if (EmitDeclareReductionInit) {
735       emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent,
736                                        SrcElementCurrent, ElementTy);
737     } else
738       CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(),
739                            /*IsInitializer=*/false);
740   }
741 
742   if (DRD) {
743     // Shift the address forward by one element.
744     llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32(
745         SrcAddr.getElementType(), SrcElementPHI, /*Idx0=*/1,
746         "omp.arraycpy.dest.element");
747     SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock());
748   }
749 
750   // Shift the address forward by one element.
751   llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32(
752       DestAddr.getElementType(), DestElementPHI, /*Idx0=*/1,
753       "omp.arraycpy.dest.element");
754   // Check whether we've reached the end.
755   llvm::Value *Done =
756       CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done");
757   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
758   DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock());
759 
760   // Done.
761   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
762 }
763 
764 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) {
765   return CGF.EmitOMPSharedLValue(E);
766 }
767 
768 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF,
769                                             const Expr *E) {
770   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E))
771     return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false);
772   return LValue();
773 }
774 
775 void ReductionCodeGen::emitAggregateInitialization(
776     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr,
777     const OMPDeclareReductionDecl *DRD) {
778   // Emit VarDecl with copy init for arrays.
779   // Get the address of the original variable captured in current
780   // captured region.
781   const auto *PrivateVD =
782       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
783   bool EmitDeclareReductionInit =
784       DRD && (DRD->getInitializer() || !PrivateVD->hasInit());
785   EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(),
786                        EmitDeclareReductionInit,
787                        EmitDeclareReductionInit ? ClausesData[N].ReductionOp
788                                                 : PrivateVD->getInit(),
789                        DRD, SharedAddr);
790 }
791 
792 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds,
793                                    ArrayRef<const Expr *> Origs,
794                                    ArrayRef<const Expr *> Privates,
795                                    ArrayRef<const Expr *> ReductionOps) {
796   ClausesData.reserve(Shareds.size());
797   SharedAddresses.reserve(Shareds.size());
798   Sizes.reserve(Shareds.size());
799   BaseDecls.reserve(Shareds.size());
800   const auto *IOrig = Origs.begin();
801   const auto *IPriv = Privates.begin();
802   const auto *IRed = ReductionOps.begin();
803   for (const Expr *Ref : Shareds) {
804     ClausesData.emplace_back(Ref, *IOrig, *IPriv, *IRed);
805     std::advance(IOrig, 1);
806     std::advance(IPriv, 1);
807     std::advance(IRed, 1);
808   }
809 }
810 
811 void ReductionCodeGen::emitSharedOrigLValue(CodeGenFunction &CGF, unsigned N) {
812   assert(SharedAddresses.size() == N && OrigAddresses.size() == N &&
813          "Number of generated lvalues must be exactly N.");
814   LValue First = emitSharedLValue(CGF, ClausesData[N].Shared);
815   LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Shared);
816   SharedAddresses.emplace_back(First, Second);
817   if (ClausesData[N].Shared == ClausesData[N].Ref) {
818     OrigAddresses.emplace_back(First, Second);
819   } else {
820     LValue First = emitSharedLValue(CGF, ClausesData[N].Ref);
821     LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref);
822     OrigAddresses.emplace_back(First, Second);
823   }
824 }
825 
826 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) {
827   QualType PrivateType = getPrivateType(N);
828   bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref);
829   if (!PrivateType->isVariablyModifiedType()) {
830     Sizes.emplace_back(
831         CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType()),
832         nullptr);
833     return;
834   }
835   llvm::Value *Size;
836   llvm::Value *SizeInChars;
837   auto *ElemType = OrigAddresses[N].first.getAddress(CGF).getElementType();
838   auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType);
839   if (AsArraySection) {
840     Size = CGF.Builder.CreatePtrDiff(ElemType,
841                                      OrigAddresses[N].second.getPointer(CGF),
842                                      OrigAddresses[N].first.getPointer(CGF));
843     Size = CGF.Builder.CreateNUWAdd(
844         Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1));
845     SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf);
846   } else {
847     SizeInChars =
848         CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType());
849     Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf);
850   }
851   Sizes.emplace_back(SizeInChars, Size);
852   CodeGenFunction::OpaqueValueMapping OpaqueMap(
853       CGF,
854       cast<OpaqueValueExpr>(
855           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
856       RValue::get(Size));
857   CGF.EmitVariablyModifiedType(PrivateType);
858 }
859 
860 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N,
861                                          llvm::Value *Size) {
862   QualType PrivateType = getPrivateType(N);
863   if (!PrivateType->isVariablyModifiedType()) {
864     assert(!Size && !Sizes[N].second &&
865            "Size should be nullptr for non-variably modified reduction "
866            "items.");
867     return;
868   }
869   CodeGenFunction::OpaqueValueMapping OpaqueMap(
870       CGF,
871       cast<OpaqueValueExpr>(
872           CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()),
873       RValue::get(Size));
874   CGF.EmitVariablyModifiedType(PrivateType);
875 }
876 
877 void ReductionCodeGen::emitInitialization(
878     CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr,
879     llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) {
880   assert(SharedAddresses.size() > N && "No variable was generated");
881   const auto *PrivateVD =
882       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl());
883   const OMPDeclareReductionDecl *DRD =
884       getReductionInit(ClausesData[N].ReductionOp);
885   if (CGF.getContext().getAsArrayType(PrivateVD->getType())) {
886     if (DRD && DRD->getInitializer())
887       (void)DefaultInit(CGF);
888     emitAggregateInitialization(CGF, N, PrivateAddr, SharedAddr, DRD);
889   } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) {
890     (void)DefaultInit(CGF);
891     QualType SharedType = SharedAddresses[N].first.getType();
892     emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp,
893                                      PrivateAddr, SharedAddr, SharedType);
894   } else if (!DefaultInit(CGF) && PrivateVD->hasInit() &&
895              !CGF.isTrivialInitializer(PrivateVD->getInit())) {
896     CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr,
897                          PrivateVD->getType().getQualifiers(),
898                          /*IsInitializer=*/false);
899   }
900 }
901 
902 bool ReductionCodeGen::needCleanups(unsigned N) {
903   QualType PrivateType = getPrivateType(N);
904   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
905   return DTorKind != QualType::DK_none;
906 }
907 
908 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N,
909                                     Address PrivateAddr) {
910   QualType PrivateType = getPrivateType(N);
911   QualType::DestructionKind DTorKind = PrivateType.isDestructedType();
912   if (needCleanups(N)) {
913     PrivateAddr = CGF.Builder.CreateElementBitCast(
914         PrivateAddr, CGF.ConvertTypeForMem(PrivateType));
915     CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType);
916   }
917 }
918 
919 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
920                           LValue BaseLV) {
921   BaseTy = BaseTy.getNonReferenceType();
922   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
923          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
924     if (const auto *PtrTy = BaseTy->getAs<PointerType>()) {
925       BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy);
926     } else {
927       LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy);
928       BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal);
929     }
930     BaseTy = BaseTy->getPointeeType();
931   }
932   return CGF.MakeAddrLValue(
933       CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF),
934                                        CGF.ConvertTypeForMem(ElTy)),
935       BaseLV.getType(), BaseLV.getBaseInfo(),
936       CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType()));
937 }
938 
939 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy,
940                           llvm::Type *BaseLVType, CharUnits BaseLVAlignment,
941                           llvm::Value *Addr) {
942   Address Tmp = Address::invalid();
943   Address TopTmp = Address::invalid();
944   Address MostTopTmp = Address::invalid();
945   BaseTy = BaseTy.getNonReferenceType();
946   while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) &&
947          !CGF.getContext().hasSameType(BaseTy, ElTy)) {
948     Tmp = CGF.CreateMemTemp(BaseTy);
949     if (TopTmp.isValid())
950       CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp);
951     else
952       MostTopTmp = Tmp;
953     TopTmp = Tmp;
954     BaseTy = BaseTy->getPointeeType();
955   }
956   llvm::Type *Ty = BaseLVType;
957   if (Tmp.isValid())
958     Ty = Tmp.getElementType();
959   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty);
960   if (Tmp.isValid()) {
961     CGF.Builder.CreateStore(Addr, Tmp);
962     return MostTopTmp;
963   }
964   return Address::deprecated(Addr, BaseLVAlignment);
965 }
966 
967 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) {
968   const VarDecl *OrigVD = nullptr;
969   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) {
970     const Expr *Base = OASE->getBase()->IgnoreParenImpCasts();
971     while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base))
972       Base = TempOASE->getBase()->IgnoreParenImpCasts();
973     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
974       Base = TempASE->getBase()->IgnoreParenImpCasts();
975     DE = cast<DeclRefExpr>(Base);
976     OrigVD = cast<VarDecl>(DE->getDecl());
977   } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) {
978     const Expr *Base = ASE->getBase()->IgnoreParenImpCasts();
979     while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base))
980       Base = TempASE->getBase()->IgnoreParenImpCasts();
981     DE = cast<DeclRefExpr>(Base);
982     OrigVD = cast<VarDecl>(DE->getDecl());
983   }
984   return OrigVD;
985 }
986 
987 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N,
988                                                Address PrivateAddr) {
989   const DeclRefExpr *DE;
990   if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) {
991     BaseDecls.emplace_back(OrigVD);
992     LValue OriginalBaseLValue = CGF.EmitLValue(DE);
993     LValue BaseLValue =
994         loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(),
995                     OriginalBaseLValue);
996     Address SharedAddr = SharedAddresses[N].first.getAddress(CGF);
997     llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff(
998         SharedAddr.getElementType(), BaseLValue.getPointer(CGF),
999         SharedAddr.getPointer());
1000     llvm::Value *PrivatePointer =
1001         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1002             PrivateAddr.getPointer(), SharedAddr.getType());
1003     llvm::Value *Ptr = CGF.Builder.CreateGEP(
1004         SharedAddr.getElementType(), PrivatePointer, Adjustment);
1005     return castToBase(CGF, OrigVD->getType(),
1006                       SharedAddresses[N].first.getType(),
1007                       OriginalBaseLValue.getAddress(CGF).getType(),
1008                       OriginalBaseLValue.getAlignment(), Ptr);
1009   }
1010   BaseDecls.emplace_back(
1011       cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl()));
1012   return PrivateAddr;
1013 }
1014 
1015 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const {
1016   const OMPDeclareReductionDecl *DRD =
1017       getReductionInit(ClausesData[N].ReductionOp);
1018   return DRD && DRD->getInitializer();
1019 }
1020 
1021 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
1022   return CGF.EmitLoadOfPointerLValue(
1023       CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1024       getThreadIDVariable()->getType()->castAs<PointerType>());
1025 }
1026 
1027 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt *S) {
1028   if (!CGF.HaveInsertPoint())
1029     return;
1030   // 1.2.2 OpenMP Language Terminology
1031   // Structured block - An executable statement with a single entry at the
1032   // top and a single exit at the bottom.
1033   // The point of exit cannot be a branch out of the structured block.
1034   // longjmp() and throw() must not violate the entry/exit criteria.
1035   CGF.EHStack.pushTerminate();
1036   if (S)
1037     CGF.incrementProfileCounter(S);
1038   CodeGen(CGF);
1039   CGF.EHStack.popTerminate();
1040 }
1041 
1042 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
1043     CodeGenFunction &CGF) {
1044   return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()),
1045                             getThreadIDVariable()->getType(),
1046                             AlignmentSource::Decl);
1047 }
1048 
1049 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1050                                        QualType FieldTy) {
1051   auto *Field = FieldDecl::Create(
1052       C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1053       C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1054       /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1055   Field->setAccess(AS_public);
1056   DC->addDecl(Field);
1057   return Field;
1058 }
1059 
1060 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator,
1061                                  StringRef Separator)
1062     : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator),
1063       OMPBuilder(CGM.getModule()), OffloadEntriesInfoManager(CGM) {
1064   KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
1065 
1066   // Initialize Types used in OpenMPIRBuilder from OMPKinds.def
1067   OMPBuilder.initialize();
1068   loadOffloadInfoMetadata();
1069 }
1070 
1071 void CGOpenMPRuntime::clear() {
1072   InternalVars.clear();
1073   // Clean non-target variable declarations possibly used only in debug info.
1074   for (const auto &Data : EmittedNonTargetVariables) {
1075     if (!Data.getValue().pointsToAliveValue())
1076       continue;
1077     auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue());
1078     if (!GV)
1079       continue;
1080     if (!GV->isDeclaration() || GV->getNumUses() > 0)
1081       continue;
1082     GV->eraseFromParent();
1083   }
1084 }
1085 
1086 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const {
1087   SmallString<128> Buffer;
1088   llvm::raw_svector_ostream OS(Buffer);
1089   StringRef Sep = FirstSeparator;
1090   for (StringRef Part : Parts) {
1091     OS << Sep << Part;
1092     Sep = Separator;
1093   }
1094   return std::string(OS.str());
1095 }
1096 
1097 static llvm::Function *
1098 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty,
1099                           const Expr *CombinerInitializer, const VarDecl *In,
1100                           const VarDecl *Out, bool IsCombiner) {
1101   // void .omp_combiner.(Ty *in, Ty *out);
1102   ASTContext &C = CGM.getContext();
1103   QualType PtrTy = C.getPointerType(Ty).withRestrict();
1104   FunctionArgList Args;
1105   ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(),
1106                                /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1107   ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(),
1108                               /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other);
1109   Args.push_back(&OmpOutParm);
1110   Args.push_back(&OmpInParm);
1111   const CGFunctionInfo &FnInfo =
1112       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
1113   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
1114   std::string Name = CGM.getOpenMPRuntime().getName(
1115       {IsCombiner ? "omp_combiner" : "omp_initializer", ""});
1116   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
1117                                     Name, &CGM.getModule());
1118   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
1119   if (CGM.getLangOpts().Optimize) {
1120     Fn->removeFnAttr(llvm::Attribute::NoInline);
1121     Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
1122     Fn->addFnAttr(llvm::Attribute::AlwaysInline);
1123   }
1124   CodeGenFunction CGF(CGM);
1125   // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions.
1126   // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions.
1127   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(),
1128                     Out->getLocation());
1129   CodeGenFunction::OMPPrivateScope Scope(CGF);
1130   Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm);
1131   Scope.addPrivate(
1132       In, CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>())
1133               .getAddress(CGF));
1134   Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm);
1135   Scope.addPrivate(
1136       Out, CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>())
1137                .getAddress(CGF));
1138   (void)Scope.Privatize();
1139   if (!IsCombiner && Out->hasInit() &&
1140       !CGF.isTrivialInitializer(Out->getInit())) {
1141     CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out),
1142                          Out->getType().getQualifiers(),
1143                          /*IsInitializer=*/true);
1144   }
1145   if (CombinerInitializer)
1146     CGF.EmitIgnoredExpr(CombinerInitializer);
1147   Scope.ForceCleanup();
1148   CGF.FinishFunction();
1149   return Fn;
1150 }
1151 
1152 void CGOpenMPRuntime::emitUserDefinedReduction(
1153     CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) {
1154   if (UDRMap.count(D) > 0)
1155     return;
1156   llvm::Function *Combiner = emitCombinerOrInitializer(
1157       CGM, D->getType(), D->getCombiner(),
1158       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()),
1159       cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()),
1160       /*IsCombiner=*/true);
1161   llvm::Function *Initializer = nullptr;
1162   if (const Expr *Init = D->getInitializer()) {
1163     Initializer = emitCombinerOrInitializer(
1164         CGM, D->getType(),
1165         D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init
1166                                                                      : nullptr,
1167         cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()),
1168         cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()),
1169         /*IsCombiner=*/false);
1170   }
1171   UDRMap.try_emplace(D, Combiner, Initializer);
1172   if (CGF) {
1173     auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn);
1174     Decls.second.push_back(D);
1175   }
1176 }
1177 
1178 std::pair<llvm::Function *, llvm::Function *>
1179 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) {
1180   auto I = UDRMap.find(D);
1181   if (I != UDRMap.end())
1182     return I->second;
1183   emitUserDefinedReduction(/*CGF=*/nullptr, D);
1184   return UDRMap.lookup(D);
1185 }
1186 
1187 namespace {
1188 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR
1189 // Builder if one is present.
1190 struct PushAndPopStackRAII {
1191   PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF,
1192                       bool HasCancel, llvm::omp::Directive Kind)
1193       : OMPBuilder(OMPBuilder) {
1194     if (!OMPBuilder)
1195       return;
1196 
1197     // The following callback is the crucial part of clangs cleanup process.
1198     //
1199     // NOTE:
1200     // Once the OpenMPIRBuilder is used to create parallel regions (and
1201     // similar), the cancellation destination (Dest below) is determined via
1202     // IP. That means if we have variables to finalize we split the block at IP,
1203     // use the new block (=BB) as destination to build a JumpDest (via
1204     // getJumpDestInCurrentScope(BB)) which then is fed to
1205     // EmitBranchThroughCleanup. Furthermore, there will not be the need
1206     // to push & pop an FinalizationInfo object.
1207     // The FiniCB will still be needed but at the point where the
1208     // OpenMPIRBuilder is asked to construct a parallel (or similar) construct.
1209     auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) {
1210       assert(IP.getBlock()->end() == IP.getPoint() &&
1211              "Clang CG should cause non-terminated block!");
1212       CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1213       CGF.Builder.restoreIP(IP);
1214       CodeGenFunction::JumpDest Dest =
1215           CGF.getOMPCancelDestination(OMPD_parallel);
1216       CGF.EmitBranchThroughCleanup(Dest);
1217     };
1218 
1219     // TODO: Remove this once we emit parallel regions through the
1220     //       OpenMPIRBuilder as it can do this setup internally.
1221     llvm::OpenMPIRBuilder::FinalizationInfo FI({FiniCB, Kind, HasCancel});
1222     OMPBuilder->pushFinalizationCB(std::move(FI));
1223   }
1224   ~PushAndPopStackRAII() {
1225     if (OMPBuilder)
1226       OMPBuilder->popFinalizationCB();
1227   }
1228   llvm::OpenMPIRBuilder *OMPBuilder;
1229 };
1230 } // namespace
1231 
1232 static llvm::Function *emitParallelOrTeamsOutlinedFunction(
1233     CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS,
1234     const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind,
1235     const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) {
1236   assert(ThreadIDVar->getType()->isPointerType() &&
1237          "thread id variable must be of type kmp_int32 *");
1238   CodeGenFunction CGF(CGM, true);
1239   bool HasCancel = false;
1240   if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D))
1241     HasCancel = OPD->hasCancel();
1242   else if (const auto *OPD = dyn_cast<OMPTargetParallelDirective>(&D))
1243     HasCancel = OPD->hasCancel();
1244   else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D))
1245     HasCancel = OPSD->hasCancel();
1246   else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D))
1247     HasCancel = OPFD->hasCancel();
1248   else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D))
1249     HasCancel = OPFD->hasCancel();
1250   else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D))
1251     HasCancel = OPFD->hasCancel();
1252   else if (const auto *OPFD =
1253                dyn_cast<OMPTeamsDistributeParallelForDirective>(&D))
1254     HasCancel = OPFD->hasCancel();
1255   else if (const auto *OPFD =
1256                dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D))
1257     HasCancel = OPFD->hasCancel();
1258 
1259   // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new
1260   //       parallel region to make cancellation barriers work properly.
1261   llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder();
1262   PushAndPopStackRAII PSR(&OMPBuilder, CGF, HasCancel, InnermostKind);
1263   CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind,
1264                                     HasCancel, OutlinedHelperName);
1265   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1266   return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc());
1267 }
1268 
1269 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction(
1270     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1271     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1272   const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel);
1273   return emitParallelOrTeamsOutlinedFunction(
1274       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1275 }
1276 
1277 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction(
1278     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1279     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
1280   const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams);
1281   return emitParallelOrTeamsOutlinedFunction(
1282       CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen);
1283 }
1284 
1285 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction(
1286     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
1287     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
1288     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
1289     bool Tied, unsigned &NumberOfParts) {
1290   auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF,
1291                                               PrePostActionTy &) {
1292     llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc());
1293     llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc());
1294     llvm::Value *TaskArgs[] = {
1295         UpLoc, ThreadID,
1296         CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar),
1297                                     TaskTVar->getType()->castAs<PointerType>())
1298             .getPointer(CGF)};
1299     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
1300                             CGM.getModule(), OMPRTL___kmpc_omp_task),
1301                         TaskArgs);
1302   };
1303   CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar,
1304                                                             UntiedCodeGen);
1305   CodeGen.setAction(Action);
1306   assert(!ThreadIDVar->getType()->isPointerType() &&
1307          "thread id variable must be of type kmp_int32 for tasks");
1308   const OpenMPDirectiveKind Region =
1309       isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop
1310                                                       : OMPD_task;
1311   const CapturedStmt *CS = D.getCapturedStmt(Region);
1312   bool HasCancel = false;
1313   if (const auto *TD = dyn_cast<OMPTaskDirective>(&D))
1314     HasCancel = TD->hasCancel();
1315   else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D))
1316     HasCancel = TD->hasCancel();
1317   else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D))
1318     HasCancel = TD->hasCancel();
1319   else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D))
1320     HasCancel = TD->hasCancel();
1321 
1322   CodeGenFunction CGF(CGM, true);
1323   CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen,
1324                                         InnermostKind, HasCancel, Action);
1325   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
1326   llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS);
1327   if (!Tied)
1328     NumberOfParts = Action.getNumberOfParts();
1329   return Res;
1330 }
1331 
1332 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM,
1333                              const RecordDecl *RD, const CGRecordLayout &RL,
1334                              ArrayRef<llvm::Constant *> Data) {
1335   llvm::StructType *StructTy = RL.getLLVMType();
1336   unsigned PrevIdx = 0;
1337   ConstantInitBuilder CIBuilder(CGM);
1338   const auto *DI = Data.begin();
1339   for (const FieldDecl *FD : RD->fields()) {
1340     unsigned Idx = RL.getLLVMFieldNo(FD);
1341     // Fill the alignment.
1342     for (unsigned I = PrevIdx; I < Idx; ++I)
1343       Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I)));
1344     PrevIdx = Idx + 1;
1345     Fields.add(*DI);
1346     ++DI;
1347   }
1348 }
1349 
1350 template <class... As>
1351 static llvm::GlobalVariable *
1352 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant,
1353                    ArrayRef<llvm::Constant *> Data, const Twine &Name,
1354                    As &&... Args) {
1355   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1356   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1357   ConstantInitBuilder CIBuilder(CGM);
1358   ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType());
1359   buildStructValue(Fields, CGM, RD, RL, Data);
1360   return Fields.finishAndCreateGlobal(
1361       Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant,
1362       std::forward<As>(Args)...);
1363 }
1364 
1365 template <typename T>
1366 static void
1367 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty,
1368                                          ArrayRef<llvm::Constant *> Data,
1369                                          T &Parent) {
1370   const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl());
1371   const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD);
1372   ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType());
1373   buildStructValue(Fields, CGM, RD, RL, Data);
1374   Fields.finishAndAddTo(Parent);
1375 }
1376 
1377 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF,
1378                                              bool AtCurrentPoint) {
1379   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1380   assert(!Elem.second.ServiceInsertPt && "Insert point is set already.");
1381 
1382   llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty);
1383   if (AtCurrentPoint) {
1384     Elem.second.ServiceInsertPt = new llvm::BitCastInst(
1385         Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock());
1386   } else {
1387     Elem.second.ServiceInsertPt =
1388         new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt");
1389     Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt);
1390   }
1391 }
1392 
1393 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) {
1394   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1395   if (Elem.second.ServiceInsertPt) {
1396     llvm::Instruction *Ptr = Elem.second.ServiceInsertPt;
1397     Elem.second.ServiceInsertPt = nullptr;
1398     Ptr->eraseFromParent();
1399   }
1400 }
1401 
1402 static StringRef getIdentStringFromSourceLocation(CodeGenFunction &CGF,
1403                                                   SourceLocation Loc,
1404                                                   SmallString<128> &Buffer) {
1405   llvm::raw_svector_ostream OS(Buffer);
1406   // Build debug location
1407   PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1408   OS << ";" << PLoc.getFilename() << ";";
1409   if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1410     OS << FD->getQualifiedNameAsString();
1411   OS << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
1412   return OS.str();
1413 }
1414 
1415 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
1416                                                  SourceLocation Loc,
1417                                                  unsigned Flags) {
1418   uint32_t SrcLocStrSize;
1419   llvm::Constant *SrcLocStr;
1420   if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo ||
1421       Loc.isInvalid()) {
1422     SrcLocStr = OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize);
1423   } else {
1424     std::string FunctionName;
1425     if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl))
1426       FunctionName = FD->getQualifiedNameAsString();
1427     PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
1428     const char *FileName = PLoc.getFilename();
1429     unsigned Line = PLoc.getLine();
1430     unsigned Column = PLoc.getColumn();
1431     SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(FunctionName, FileName, Line,
1432                                                 Column, SrcLocStrSize);
1433   }
1434   unsigned Reserved2Flags = getDefaultLocationReserved2Flags();
1435   return OMPBuilder.getOrCreateIdent(
1436       SrcLocStr, SrcLocStrSize, llvm::omp::IdentFlag(Flags), Reserved2Flags);
1437 }
1438 
1439 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
1440                                           SourceLocation Loc) {
1441   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1442   // If the OpenMPIRBuilder is used we need to use it for all thread id calls as
1443   // the clang invariants used below might be broken.
1444   if (CGM.getLangOpts().OpenMPIRBuilder) {
1445     SmallString<128> Buffer;
1446     OMPBuilder.updateToLocation(CGF.Builder.saveIP());
1447     uint32_t SrcLocStrSize;
1448     auto *SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(
1449         getIdentStringFromSourceLocation(CGF, Loc, Buffer), SrcLocStrSize);
1450     return OMPBuilder.getOrCreateThreadID(
1451         OMPBuilder.getOrCreateIdent(SrcLocStr, SrcLocStrSize));
1452   }
1453 
1454   llvm::Value *ThreadID = nullptr;
1455   // Check whether we've already cached a load of the thread id in this
1456   // function.
1457   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
1458   if (I != OpenMPLocThreadIDMap.end()) {
1459     ThreadID = I->second.ThreadID;
1460     if (ThreadID != nullptr)
1461       return ThreadID;
1462   }
1463   // If exceptions are enabled, do not use parameter to avoid possible crash.
1464   if (auto *OMPRegionInfo =
1465           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
1466     if (OMPRegionInfo->getThreadIDVariable()) {
1467       // Check if this an outlined function with thread id passed as argument.
1468       LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
1469       llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent();
1470       if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions ||
1471           !CGF.getLangOpts().CXXExceptions ||
1472           CGF.Builder.GetInsertBlock() == TopBlock ||
1473           !isa<llvm::Instruction>(LVal.getPointer(CGF)) ||
1474           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1475               TopBlock ||
1476           cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() ==
1477               CGF.Builder.GetInsertBlock()) {
1478         ThreadID = CGF.EmitLoadOfScalar(LVal, Loc);
1479         // If value loaded in entry block, cache it and use it everywhere in
1480         // function.
1481         if (CGF.Builder.GetInsertBlock() == TopBlock) {
1482           auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1483           Elem.second.ThreadID = ThreadID;
1484         }
1485         return ThreadID;
1486       }
1487     }
1488   }
1489 
1490   // This is not an outlined function region - need to call __kmpc_int32
1491   // kmpc_global_thread_num(ident_t *loc).
1492   // Generate thread id value and cache this value for use across the
1493   // function.
1494   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
1495   if (!Elem.second.ServiceInsertPt)
1496     setLocThreadIdInsertPt(CGF);
1497   CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
1498   CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt);
1499   llvm::CallInst *Call = CGF.Builder.CreateCall(
1500       OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
1501                                             OMPRTL___kmpc_global_thread_num),
1502       emitUpdateLocation(CGF, Loc));
1503   Call->setCallingConv(CGF.getRuntimeCC());
1504   Elem.second.ThreadID = Call;
1505   return Call;
1506 }
1507 
1508 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
1509   assert(CGF.CurFn && "No function in current CodeGenFunction.");
1510   if (OpenMPLocThreadIDMap.count(CGF.CurFn)) {
1511     clearLocThreadIdInsertPt(CGF);
1512     OpenMPLocThreadIDMap.erase(CGF.CurFn);
1513   }
1514   if (FunctionUDRMap.count(CGF.CurFn) > 0) {
1515     for(const auto *D : FunctionUDRMap[CGF.CurFn])
1516       UDRMap.erase(D);
1517     FunctionUDRMap.erase(CGF.CurFn);
1518   }
1519   auto I = FunctionUDMMap.find(CGF.CurFn);
1520   if (I != FunctionUDMMap.end()) {
1521     for(const auto *D : I->second)
1522       UDMMap.erase(D);
1523     FunctionUDMMap.erase(I);
1524   }
1525   LastprivateConditionalToTypes.erase(CGF.CurFn);
1526   FunctionToUntiedTaskStackMap.erase(CGF.CurFn);
1527 }
1528 
1529 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
1530   return OMPBuilder.IdentPtr;
1531 }
1532 
1533 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() {
1534   if (!Kmpc_MicroTy) {
1535     // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...)
1536     llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty),
1537                                  llvm::PointerType::getUnqual(CGM.Int32Ty)};
1538     Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true);
1539   }
1540   return llvm::PointerType::getUnqual(Kmpc_MicroTy);
1541 }
1542 
1543 llvm::FunctionCallee
1544 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned,
1545                                              bool IsGPUDistribute) {
1546   assert((IVSize == 32 || IVSize == 64) &&
1547          "IV size is not compatible with the omp runtime");
1548   StringRef Name;
1549   if (IsGPUDistribute)
1550     Name = IVSize == 32 ? (IVSigned ? "__kmpc_distribute_static_init_4"
1551                                     : "__kmpc_distribute_static_init_4u")
1552                         : (IVSigned ? "__kmpc_distribute_static_init_8"
1553                                     : "__kmpc_distribute_static_init_8u");
1554   else
1555     Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4"
1556                                     : "__kmpc_for_static_init_4u")
1557                         : (IVSigned ? "__kmpc_for_static_init_8"
1558                                     : "__kmpc_for_static_init_8u");
1559 
1560   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
1561   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
1562   llvm::Type *TypeParams[] = {
1563     getIdentTyPointerTy(),                     // loc
1564     CGM.Int32Ty,                               // tid
1565     CGM.Int32Ty,                               // schedtype
1566     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
1567     PtrTy,                                     // p_lower
1568     PtrTy,                                     // p_upper
1569     PtrTy,                                     // p_stride
1570     ITy,                                       // incr
1571     ITy                                        // chunk
1572   };
1573   auto *FnTy =
1574       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1575   return CGM.CreateRuntimeFunction(FnTy, Name);
1576 }
1577 
1578 llvm::FunctionCallee
1579 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) {
1580   assert((IVSize == 32 || IVSize == 64) &&
1581          "IV size is not compatible with the omp runtime");
1582   StringRef Name =
1583       IVSize == 32
1584           ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u")
1585           : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u");
1586   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
1587   llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc
1588                                CGM.Int32Ty,           // tid
1589                                CGM.Int32Ty,           // schedtype
1590                                ITy,                   // lower
1591                                ITy,                   // upper
1592                                ITy,                   // stride
1593                                ITy                    // chunk
1594   };
1595   auto *FnTy =
1596       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
1597   return CGM.CreateRuntimeFunction(FnTy, Name);
1598 }
1599 
1600 llvm::FunctionCallee
1601 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) {
1602   assert((IVSize == 32 || IVSize == 64) &&
1603          "IV size is not compatible with the omp runtime");
1604   StringRef Name =
1605       IVSize == 32
1606           ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u")
1607           : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u");
1608   llvm::Type *TypeParams[] = {
1609       getIdentTyPointerTy(), // loc
1610       CGM.Int32Ty,           // tid
1611   };
1612   auto *FnTy =
1613       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
1614   return CGM.CreateRuntimeFunction(FnTy, Name);
1615 }
1616 
1617 llvm::FunctionCallee
1618 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) {
1619   assert((IVSize == 32 || IVSize == 64) &&
1620          "IV size is not compatible with the omp runtime");
1621   StringRef Name =
1622       IVSize == 32
1623           ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u")
1624           : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u");
1625   llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
1626   auto *PtrTy = llvm::PointerType::getUnqual(ITy);
1627   llvm::Type *TypeParams[] = {
1628     getIdentTyPointerTy(),                     // loc
1629     CGM.Int32Ty,                               // tid
1630     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
1631     PtrTy,                                     // p_lower
1632     PtrTy,                                     // p_upper
1633     PtrTy                                      // p_stride
1634   };
1635   auto *FnTy =
1636       llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
1637   return CGM.CreateRuntimeFunction(FnTy, Name);
1638 }
1639 
1640 /// Obtain information that uniquely identifies a target entry. This
1641 /// consists of the file and device IDs as well as line number associated with
1642 /// the relevant entry source location.
1643 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc,
1644                                      unsigned &DeviceID, unsigned &FileID,
1645                                      unsigned &LineNum) {
1646   SourceManager &SM = C.getSourceManager();
1647 
1648   // The loc should be always valid and have a file ID (the user cannot use
1649   // #pragma directives in macros)
1650 
1651   assert(Loc.isValid() && "Source location is expected to be always valid.");
1652 
1653   PresumedLoc PLoc = SM.getPresumedLoc(Loc);
1654   assert(PLoc.isValid() && "Source location is expected to be always valid.");
1655 
1656   llvm::sys::fs::UniqueID ID;
1657   if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) {
1658     PLoc = SM.getPresumedLoc(Loc, /*UseLineDirectives=*/false);
1659     assert(PLoc.isValid() && "Source location is expected to be always valid.");
1660     if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID))
1661       SM.getDiagnostics().Report(diag::err_cannot_open_file)
1662           << PLoc.getFilename() << EC.message();
1663   }
1664 
1665   DeviceID = ID.getDevice();
1666   FileID = ID.getFile();
1667   LineNum = PLoc.getLine();
1668 }
1669 
1670 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) {
1671   if (CGM.getLangOpts().OpenMPSimd)
1672     return Address::invalid();
1673   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
1674       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
1675   if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link ||
1676               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
1677                HasRequiresUnifiedSharedMemory))) {
1678     SmallString<64> PtrName;
1679     {
1680       llvm::raw_svector_ostream OS(PtrName);
1681       OS << CGM.getMangledName(GlobalDecl(VD));
1682       if (!VD->isExternallyVisible()) {
1683         unsigned DeviceID, FileID, Line;
1684         getTargetEntryUniqueInfo(CGM.getContext(),
1685                                  VD->getCanonicalDecl()->getBeginLoc(),
1686                                  DeviceID, FileID, Line);
1687         OS << llvm::format("_%x", FileID);
1688       }
1689       OS << "_decl_tgt_ref_ptr";
1690     }
1691     llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName);
1692     QualType PtrTy = CGM.getContext().getPointerType(VD->getType());
1693     llvm::Type *LlvmPtrTy = CGM.getTypes().ConvertTypeForMem(PtrTy);
1694     if (!Ptr) {
1695       Ptr = getOrCreateInternalVariable(LlvmPtrTy, PtrName);
1696 
1697       auto *GV = cast<llvm::GlobalVariable>(Ptr);
1698       GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
1699 
1700       if (!CGM.getLangOpts().OpenMPIsDevice)
1701         GV->setInitializer(CGM.GetAddrOfGlobal(VD));
1702       registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr));
1703     }
1704     return Address(Ptr, LlvmPtrTy, CGM.getContext().getDeclAlign(VD));
1705   }
1706   return Address::invalid();
1707 }
1708 
1709 llvm::Constant *
1710 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
1711   assert(!CGM.getLangOpts().OpenMPUseTLS ||
1712          !CGM.getContext().getTargetInfo().isTLSSupported());
1713   // Lookup the entry, lazily creating it if necessary.
1714   std::string Suffix = getName({"cache", ""});
1715   return getOrCreateInternalVariable(
1716       CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix));
1717 }
1718 
1719 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
1720                                                 const VarDecl *VD,
1721                                                 Address VDAddr,
1722                                                 SourceLocation Loc) {
1723   if (CGM.getLangOpts().OpenMPUseTLS &&
1724       CGM.getContext().getTargetInfo().isTLSSupported())
1725     return VDAddr;
1726 
1727   llvm::Type *VarTy = VDAddr.getElementType();
1728   llvm::Value *Args[] = {
1729       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
1730       CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.Int8PtrTy),
1731       CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
1732       getOrCreateThreadPrivateCache(VD)};
1733   return Address(
1734       CGF.EmitRuntimeCall(
1735           OMPBuilder.getOrCreateRuntimeFunction(
1736               CGM.getModule(), OMPRTL___kmpc_threadprivate_cached),
1737           Args),
1738       CGF.Int8Ty, VDAddr.getAlignment());
1739 }
1740 
1741 void CGOpenMPRuntime::emitThreadPrivateVarInit(
1742     CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor,
1743     llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
1744   // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
1745   // library.
1746   llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc);
1747   CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
1748                           CGM.getModule(), OMPRTL___kmpc_global_thread_num),
1749                       OMPLoc);
1750   // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
1751   // to register constructor/destructor for variable.
1752   llvm::Value *Args[] = {
1753       OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy),
1754       Ctor, CopyCtor, Dtor};
1755   CGF.EmitRuntimeCall(
1756       OMPBuilder.getOrCreateRuntimeFunction(
1757           CGM.getModule(), OMPRTL___kmpc_threadprivate_register),
1758       Args);
1759 }
1760 
1761 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
1762     const VarDecl *VD, Address VDAddr, SourceLocation Loc,
1763     bool PerformInit, CodeGenFunction *CGF) {
1764   if (CGM.getLangOpts().OpenMPUseTLS &&
1765       CGM.getContext().getTargetInfo().isTLSSupported())
1766     return nullptr;
1767 
1768   VD = VD->getDefinition(CGM.getContext());
1769   if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) {
1770     QualType ASTTy = VD->getType();
1771 
1772     llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
1773     const Expr *Init = VD->getAnyInitializer();
1774     if (CGM.getLangOpts().CPlusPlus && PerformInit) {
1775       // Generate function that re-emits the declaration's initializer into the
1776       // threadprivate copy of the variable VD
1777       CodeGenFunction CtorCGF(CGM);
1778       FunctionArgList Args;
1779       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
1780                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
1781                             ImplicitParamDecl::Other);
1782       Args.push_back(&Dst);
1783 
1784       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
1785           CGM.getContext().VoidPtrTy, Args);
1786       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
1787       std::string Name = getName({"__kmpc_global_ctor_", ""});
1788       llvm::Function *Fn =
1789           CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc);
1790       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
1791                             Args, Loc, Loc);
1792       llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar(
1793           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
1794           CGM.getContext().VoidPtrTy, Dst.getLocation());
1795       Address Arg(ArgVal, CtorCGF.Int8Ty, VDAddr.getAlignment());
1796       Arg = CtorCGF.Builder.CreateElementBitCast(
1797           Arg, CtorCGF.ConvertTypeForMem(ASTTy));
1798       CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
1799                                /*IsInitializer=*/true);
1800       ArgVal = CtorCGF.EmitLoadOfScalar(
1801           CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false,
1802           CGM.getContext().VoidPtrTy, Dst.getLocation());
1803       CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
1804       CtorCGF.FinishFunction();
1805       Ctor = Fn;
1806     }
1807     if (VD->getType().isDestructedType() != QualType::DK_none) {
1808       // Generate function that emits destructor call for the threadprivate copy
1809       // of the variable VD
1810       CodeGenFunction DtorCGF(CGM);
1811       FunctionArgList Args;
1812       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc,
1813                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy,
1814                             ImplicitParamDecl::Other);
1815       Args.push_back(&Dst);
1816 
1817       const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration(
1818           CGM.getContext().VoidTy, Args);
1819       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
1820       std::string Name = getName({"__kmpc_global_dtor_", ""});
1821       llvm::Function *Fn =
1822           CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc);
1823       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
1824       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
1825                             Loc, Loc);
1826       // Create a scope with an artificial location for the body of this function.
1827       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
1828       llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar(
1829           DtorCGF.GetAddrOfLocalVar(&Dst),
1830           /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation());
1831       DtorCGF.emitDestroy(
1832           Address(ArgVal, DtorCGF.Int8Ty, VDAddr.getAlignment()), ASTTy,
1833           DtorCGF.getDestroyer(ASTTy.isDestructedType()),
1834           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
1835       DtorCGF.FinishFunction();
1836       Dtor = Fn;
1837     }
1838     // Do not emit init function if it is not required.
1839     if (!Ctor && !Dtor)
1840       return nullptr;
1841 
1842     llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
1843     auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs,
1844                                                /*isVarArg=*/false)
1845                            ->getPointerTo();
1846     // Copying constructor for the threadprivate variable.
1847     // Must be NULL - reserved by runtime, but currently it requires that this
1848     // parameter is always NULL. Otherwise it fires assertion.
1849     CopyCtor = llvm::Constant::getNullValue(CopyCtorTy);
1850     if (Ctor == nullptr) {
1851       auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
1852                                              /*isVarArg=*/false)
1853                          ->getPointerTo();
1854       Ctor = llvm::Constant::getNullValue(CtorTy);
1855     }
1856     if (Dtor == nullptr) {
1857       auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy,
1858                                              /*isVarArg=*/false)
1859                          ->getPointerTo();
1860       Dtor = llvm::Constant::getNullValue(DtorTy);
1861     }
1862     if (!CGF) {
1863       auto *InitFunctionTy =
1864           llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
1865       std::string Name = getName({"__omp_threadprivate_init_", ""});
1866       llvm::Function *InitFunction = CGM.CreateGlobalInitOrCleanUpFunction(
1867           InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction());
1868       CodeGenFunction InitCGF(CGM);
1869       FunctionArgList ArgList;
1870       InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
1871                             CGM.getTypes().arrangeNullaryFunction(), ArgList,
1872                             Loc, Loc);
1873       emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
1874       InitCGF.FinishFunction();
1875       return InitFunction;
1876     }
1877     emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
1878   }
1879   return nullptr;
1880 }
1881 
1882 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD,
1883                                                      llvm::GlobalVariable *Addr,
1884                                                      bool PerformInit) {
1885   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
1886       !CGM.getLangOpts().OpenMPIsDevice)
1887     return false;
1888   Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
1889       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
1890   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
1891       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
1892        HasRequiresUnifiedSharedMemory))
1893     return CGM.getLangOpts().OpenMPIsDevice;
1894   VD = VD->getDefinition(CGM.getContext());
1895   assert(VD && "Unknown VarDecl");
1896 
1897   if (!DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second)
1898     return CGM.getLangOpts().OpenMPIsDevice;
1899 
1900   QualType ASTTy = VD->getType();
1901   SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc();
1902 
1903   // Produce the unique prefix to identify the new target regions. We use
1904   // the source location of the variable declaration which we know to not
1905   // conflict with any target region.
1906   unsigned DeviceID;
1907   unsigned FileID;
1908   unsigned Line;
1909   getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line);
1910   SmallString<128> Buffer, Out;
1911   {
1912     llvm::raw_svector_ostream OS(Buffer);
1913     OS << "__omp_offloading_" << llvm::format("_%x", DeviceID)
1914        << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line;
1915   }
1916 
1917   const Expr *Init = VD->getAnyInitializer();
1918   if (CGM.getLangOpts().CPlusPlus && PerformInit) {
1919     llvm::Constant *Ctor;
1920     llvm::Constant *ID;
1921     if (CGM.getLangOpts().OpenMPIsDevice) {
1922       // Generate function that re-emits the declaration's initializer into
1923       // the threadprivate copy of the variable VD
1924       CodeGenFunction CtorCGF(CGM);
1925 
1926       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
1927       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
1928       llvm::Function *Fn = CGM.CreateGlobalInitOrCleanUpFunction(
1929           FTy, Twine(Buffer, "_ctor"), FI, Loc);
1930       auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF);
1931       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
1932                             FunctionArgList(), Loc, Loc);
1933       auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF);
1934       llvm::Constant *AddrInAS0 = Addr;
1935       if (Addr->getAddressSpace() != 0)
1936         AddrInAS0 = llvm::ConstantExpr::getAddrSpaceCast(
1937             Addr, llvm::PointerType::getWithSamePointeeType(
1938                       cast<llvm::PointerType>(Addr->getType()), 0));
1939       CtorCGF.EmitAnyExprToMem(Init,
1940                                Address(AddrInAS0, Addr->getValueType(),
1941                                        CGM.getContext().getDeclAlign(VD)),
1942                                Init->getType().getQualifiers(),
1943                                /*IsInitializer=*/true);
1944       CtorCGF.FinishFunction();
1945       Ctor = Fn;
1946       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
1947       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor));
1948     } else {
1949       Ctor = new llvm::GlobalVariable(
1950           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
1951           llvm::GlobalValue::PrivateLinkage,
1952           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor"));
1953       ID = Ctor;
1954     }
1955 
1956     // Register the information for the entry associated with the constructor.
1957     Out.clear();
1958     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
1959         DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor,
1960         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor);
1961   }
1962   if (VD->getType().isDestructedType() != QualType::DK_none) {
1963     llvm::Constant *Dtor;
1964     llvm::Constant *ID;
1965     if (CGM.getLangOpts().OpenMPIsDevice) {
1966       // Generate function that emits destructor call for the threadprivate
1967       // copy of the variable VD
1968       CodeGenFunction DtorCGF(CGM);
1969 
1970       const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction();
1971       llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
1972       llvm::Function *Fn = CGM.CreateGlobalInitOrCleanUpFunction(
1973           FTy, Twine(Buffer, "_dtor"), FI, Loc);
1974       auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF);
1975       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI,
1976                             FunctionArgList(), Loc, Loc);
1977       // Create a scope with an artificial location for the body of this
1978       // function.
1979       auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF);
1980       llvm::Constant *AddrInAS0 = Addr;
1981       if (Addr->getAddressSpace() != 0)
1982         AddrInAS0 = llvm::ConstantExpr::getAddrSpaceCast(
1983             Addr, llvm::PointerType::getWithSamePointeeType(
1984                       cast<llvm::PointerType>(Addr->getType()), 0));
1985       DtorCGF.emitDestroy(Address(AddrInAS0, Addr->getValueType(),
1986                                   CGM.getContext().getDeclAlign(VD)),
1987                           ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()),
1988                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
1989       DtorCGF.FinishFunction();
1990       Dtor = Fn;
1991       ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy);
1992       CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor));
1993     } else {
1994       Dtor = new llvm::GlobalVariable(
1995           CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
1996           llvm::GlobalValue::PrivateLinkage,
1997           llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor"));
1998       ID = Dtor;
1999     }
2000     // Register the information for the entry associated with the destructor.
2001     Out.clear();
2002     OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
2003         DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor,
2004         ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor);
2005   }
2006   return CGM.getLangOpts().OpenMPIsDevice;
2007 }
2008 
2009 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF,
2010                                                           QualType VarType,
2011                                                           StringRef Name) {
2012   std::string Suffix = getName({"artificial", ""});
2013   llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType);
2014   llvm::GlobalVariable *GAddr =
2015       getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix));
2016   if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS &&
2017       CGM.getTarget().isTLSSupported()) {
2018     GAddr->setThreadLocal(/*Val=*/true);
2019     return Address(GAddr, GAddr->getValueType(),
2020                    CGM.getContext().getTypeAlignInChars(VarType));
2021   }
2022   std::string CacheSuffix = getName({"cache", ""});
2023   llvm::Value *Args[] = {
2024       emitUpdateLocation(CGF, SourceLocation()),
2025       getThreadID(CGF, SourceLocation()),
2026       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy),
2027       CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy,
2028                                 /*isSigned=*/false),
2029       getOrCreateInternalVariable(
2030           CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))};
2031   return Address(
2032       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2033           CGF.EmitRuntimeCall(
2034               OMPBuilder.getOrCreateRuntimeFunction(
2035                   CGM.getModule(), OMPRTL___kmpc_threadprivate_cached),
2036               Args),
2037           VarLVType->getPointerTo(/*AddrSpace=*/0)),
2038       VarLVType, CGM.getContext().getTypeAlignInChars(VarType));
2039 }
2040 
2041 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond,
2042                                    const RegionCodeGenTy &ThenGen,
2043                                    const RegionCodeGenTy &ElseGen) {
2044   CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
2045 
2046   // If the condition constant folds and can be elided, try to avoid emitting
2047   // the condition and the dead arm of the if/else.
2048   bool CondConstant;
2049   if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
2050     if (CondConstant)
2051       ThenGen(CGF);
2052     else
2053       ElseGen(CGF);
2054     return;
2055   }
2056 
2057   // Otherwise, the condition did not fold, or we couldn't elide it.  Just
2058   // emit the conditional branch.
2059   llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then");
2060   llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else");
2061   llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end");
2062   CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0);
2063 
2064   // Emit the 'then' code.
2065   CGF.EmitBlock(ThenBlock);
2066   ThenGen(CGF);
2067   CGF.EmitBranch(ContBlock);
2068   // Emit the 'else' code if present.
2069   // There is no need to emit line number for unconditional branch.
2070   (void)ApplyDebugLocation::CreateEmpty(CGF);
2071   CGF.EmitBlock(ElseBlock);
2072   ElseGen(CGF);
2073   // There is no need to emit line number for unconditional branch.
2074   (void)ApplyDebugLocation::CreateEmpty(CGF);
2075   CGF.EmitBranch(ContBlock);
2076   // Emit the continuation block for code after the if.
2077   CGF.EmitBlock(ContBlock, /*IsFinished=*/true);
2078 }
2079 
2080 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc,
2081                                        llvm::Function *OutlinedFn,
2082                                        ArrayRef<llvm::Value *> CapturedVars,
2083                                        const Expr *IfCond,
2084                                        llvm::Value *NumThreads) {
2085   if (!CGF.HaveInsertPoint())
2086     return;
2087   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
2088   auto &M = CGM.getModule();
2089   auto &&ThenGen = [&M, OutlinedFn, CapturedVars, RTLoc,
2090                     this](CodeGenFunction &CGF, PrePostActionTy &) {
2091     // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn);
2092     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
2093     llvm::Value *Args[] = {
2094         RTLoc,
2095         CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
2096         CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())};
2097     llvm::SmallVector<llvm::Value *, 16> RealArgs;
2098     RealArgs.append(std::begin(Args), std::end(Args));
2099     RealArgs.append(CapturedVars.begin(), CapturedVars.end());
2100 
2101     llvm::FunctionCallee RTLFn =
2102         OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_fork_call);
2103     CGF.EmitRuntimeCall(RTLFn, RealArgs);
2104   };
2105   auto &&ElseGen = [&M, OutlinedFn, CapturedVars, RTLoc, Loc,
2106                     this](CodeGenFunction &CGF, PrePostActionTy &) {
2107     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
2108     llvm::Value *ThreadID = RT.getThreadID(CGF, Loc);
2109     // Build calls:
2110     // __kmpc_serialized_parallel(&Loc, GTid);
2111     llvm::Value *Args[] = {RTLoc, ThreadID};
2112     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2113                             M, OMPRTL___kmpc_serialized_parallel),
2114                         Args);
2115 
2116     // OutlinedFn(&GTid, &zero_bound, CapturedStruct);
2117     Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc);
2118     Address ZeroAddrBound =
2119         CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty,
2120                                          /*Name=*/".bound.zero.addr");
2121     CGF.Builder.CreateStore(CGF.Builder.getInt32(/*C*/ 0), ZeroAddrBound);
2122     llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs;
2123     // ThreadId for serialized parallels is 0.
2124     OutlinedFnArgs.push_back(ThreadIDAddr.getPointer());
2125     OutlinedFnArgs.push_back(ZeroAddrBound.getPointer());
2126     OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end());
2127 
2128     // Ensure we do not inline the function. This is trivially true for the ones
2129     // passed to __kmpc_fork_call but the ones called in serialized regions
2130     // could be inlined. This is not a perfect but it is closer to the invariant
2131     // we want, namely, every data environment starts with a new function.
2132     // TODO: We should pass the if condition to the runtime function and do the
2133     //       handling there. Much cleaner code.
2134     OutlinedFn->removeFnAttr(llvm::Attribute::AlwaysInline);
2135     OutlinedFn->addFnAttr(llvm::Attribute::NoInline);
2136     RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs);
2137 
2138     // __kmpc_end_serialized_parallel(&Loc, GTid);
2139     llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID};
2140     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2141                             M, OMPRTL___kmpc_end_serialized_parallel),
2142                         EndArgs);
2143   };
2144   if (IfCond) {
2145     emitIfClause(CGF, IfCond, ThenGen, ElseGen);
2146   } else {
2147     RegionCodeGenTy ThenRCG(ThenGen);
2148     ThenRCG(CGF);
2149   }
2150 }
2151 
2152 // If we're inside an (outlined) parallel region, use the region info's
2153 // thread-ID variable (it is passed in a first argument of the outlined function
2154 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
2155 // regular serial code region, get thread ID by calling kmp_int32
2156 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
2157 // return the address of that temp.
2158 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
2159                                              SourceLocation Loc) {
2160   if (auto *OMPRegionInfo =
2161           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
2162     if (OMPRegionInfo->getThreadIDVariable())
2163       return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF);
2164 
2165   llvm::Value *ThreadID = getThreadID(CGF, Loc);
2166   QualType Int32Ty =
2167       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
2168   Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp.");
2169   CGF.EmitStoreOfScalar(ThreadID,
2170                         CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty));
2171 
2172   return ThreadIDTemp;
2173 }
2174 
2175 llvm::GlobalVariable *CGOpenMPRuntime::getOrCreateInternalVariable(
2176     llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) {
2177   SmallString<256> Buffer;
2178   llvm::raw_svector_ostream Out(Buffer);
2179   Out << Name;
2180   StringRef RuntimeName = Out.str();
2181   auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first;
2182   if (Elem.second) {
2183     assert(Elem.second->getType()->isOpaqueOrPointeeTypeMatches(Ty) &&
2184            "OMP internal variable has different type than requested");
2185     return &*Elem.second;
2186   }
2187 
2188   return Elem.second = new llvm::GlobalVariable(
2189              CGM.getModule(), Ty, /*IsConstant*/ false,
2190              llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty),
2191              Elem.first(), /*InsertBefore=*/nullptr,
2192              llvm::GlobalValue::NotThreadLocal, AddressSpace);
2193 }
2194 
2195 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
2196   std::string Prefix = Twine("gomp_critical_user_", CriticalName).str();
2197   std::string Name = getName({Prefix, "var"});
2198   return getOrCreateInternalVariable(KmpCriticalNameTy, Name);
2199 }
2200 
2201 namespace {
2202 /// Common pre(post)-action for different OpenMP constructs.
2203 class CommonActionTy final : public PrePostActionTy {
2204   llvm::FunctionCallee EnterCallee;
2205   ArrayRef<llvm::Value *> EnterArgs;
2206   llvm::FunctionCallee ExitCallee;
2207   ArrayRef<llvm::Value *> ExitArgs;
2208   bool Conditional;
2209   llvm::BasicBlock *ContBlock = nullptr;
2210 
2211 public:
2212   CommonActionTy(llvm::FunctionCallee EnterCallee,
2213                  ArrayRef<llvm::Value *> EnterArgs,
2214                  llvm::FunctionCallee ExitCallee,
2215                  ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false)
2216       : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee),
2217         ExitArgs(ExitArgs), Conditional(Conditional) {}
2218   void Enter(CodeGenFunction &CGF) override {
2219     llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs);
2220     if (Conditional) {
2221       llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes);
2222       auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
2223       ContBlock = CGF.createBasicBlock("omp_if.end");
2224       // Generate the branch (If-stmt)
2225       CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
2226       CGF.EmitBlock(ThenBlock);
2227     }
2228   }
2229   void Done(CodeGenFunction &CGF) {
2230     // Emit the rest of blocks/branches
2231     CGF.EmitBranch(ContBlock);
2232     CGF.EmitBlock(ContBlock, true);
2233   }
2234   void Exit(CodeGenFunction &CGF) override {
2235     CGF.EmitRuntimeCall(ExitCallee, ExitArgs);
2236   }
2237 };
2238 } // anonymous namespace
2239 
2240 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF,
2241                                          StringRef CriticalName,
2242                                          const RegionCodeGenTy &CriticalOpGen,
2243                                          SourceLocation Loc, const Expr *Hint) {
2244   // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]);
2245   // CriticalOpGen();
2246   // __kmpc_end_critical(ident_t *, gtid, Lock);
2247   // Prepare arguments and build a call to __kmpc_critical
2248   if (!CGF.HaveInsertPoint())
2249     return;
2250   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2251                          getCriticalRegionLock(CriticalName)};
2252   llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args),
2253                                                 std::end(Args));
2254   if (Hint) {
2255     EnterArgs.push_back(CGF.Builder.CreateIntCast(
2256         CGF.EmitScalarExpr(Hint), CGM.Int32Ty, /*isSigned=*/false));
2257   }
2258   CommonActionTy Action(
2259       OMPBuilder.getOrCreateRuntimeFunction(
2260           CGM.getModule(),
2261           Hint ? OMPRTL___kmpc_critical_with_hint : OMPRTL___kmpc_critical),
2262       EnterArgs,
2263       OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
2264                                             OMPRTL___kmpc_end_critical),
2265       Args);
2266   CriticalOpGen.setAction(Action);
2267   emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen);
2268 }
2269 
2270 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
2271                                        const RegionCodeGenTy &MasterOpGen,
2272                                        SourceLocation Loc) {
2273   if (!CGF.HaveInsertPoint())
2274     return;
2275   // if(__kmpc_master(ident_t *, gtid)) {
2276   //   MasterOpGen();
2277   //   __kmpc_end_master(ident_t *, gtid);
2278   // }
2279   // Prepare arguments and build a call to __kmpc_master
2280   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2281   CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2282                             CGM.getModule(), OMPRTL___kmpc_master),
2283                         Args,
2284                         OMPBuilder.getOrCreateRuntimeFunction(
2285                             CGM.getModule(), OMPRTL___kmpc_end_master),
2286                         Args,
2287                         /*Conditional=*/true);
2288   MasterOpGen.setAction(Action);
2289   emitInlinedDirective(CGF, OMPD_master, MasterOpGen);
2290   Action.Done(CGF);
2291 }
2292 
2293 void CGOpenMPRuntime::emitMaskedRegion(CodeGenFunction &CGF,
2294                                        const RegionCodeGenTy &MaskedOpGen,
2295                                        SourceLocation Loc, const Expr *Filter) {
2296   if (!CGF.HaveInsertPoint())
2297     return;
2298   // if(__kmpc_masked(ident_t *, gtid, filter)) {
2299   //   MaskedOpGen();
2300   //   __kmpc_end_masked(iden_t *, gtid);
2301   // }
2302   // Prepare arguments and build a call to __kmpc_masked
2303   llvm::Value *FilterVal = Filter
2304                                ? CGF.EmitScalarExpr(Filter, CGF.Int32Ty)
2305                                : llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/0);
2306   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2307                          FilterVal};
2308   llvm::Value *ArgsEnd[] = {emitUpdateLocation(CGF, Loc),
2309                             getThreadID(CGF, Loc)};
2310   CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2311                             CGM.getModule(), OMPRTL___kmpc_masked),
2312                         Args,
2313                         OMPBuilder.getOrCreateRuntimeFunction(
2314                             CGM.getModule(), OMPRTL___kmpc_end_masked),
2315                         ArgsEnd,
2316                         /*Conditional=*/true);
2317   MaskedOpGen.setAction(Action);
2318   emitInlinedDirective(CGF, OMPD_masked, MaskedOpGen);
2319   Action.Done(CGF);
2320 }
2321 
2322 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
2323                                         SourceLocation Loc) {
2324   if (!CGF.HaveInsertPoint())
2325     return;
2326   if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2327     OMPBuilder.createTaskyield(CGF.Builder);
2328   } else {
2329     // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
2330     llvm::Value *Args[] = {
2331         emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2332         llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
2333     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2334                             CGM.getModule(), OMPRTL___kmpc_omp_taskyield),
2335                         Args);
2336   }
2337 
2338   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
2339     Region->emitUntiedSwitch(CGF);
2340 }
2341 
2342 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF,
2343                                           const RegionCodeGenTy &TaskgroupOpGen,
2344                                           SourceLocation Loc) {
2345   if (!CGF.HaveInsertPoint())
2346     return;
2347   // __kmpc_taskgroup(ident_t *, gtid);
2348   // TaskgroupOpGen();
2349   // __kmpc_end_taskgroup(ident_t *, gtid);
2350   // Prepare arguments and build a call to __kmpc_taskgroup
2351   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2352   CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2353                             CGM.getModule(), OMPRTL___kmpc_taskgroup),
2354                         Args,
2355                         OMPBuilder.getOrCreateRuntimeFunction(
2356                             CGM.getModule(), OMPRTL___kmpc_end_taskgroup),
2357                         Args);
2358   TaskgroupOpGen.setAction(Action);
2359   emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen);
2360 }
2361 
2362 /// Given an array of pointers to variables, project the address of a
2363 /// given variable.
2364 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array,
2365                                       unsigned Index, const VarDecl *Var) {
2366   // Pull out the pointer to the variable.
2367   Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index);
2368   llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr);
2369 
2370   llvm::Type *ElemTy = CGF.ConvertTypeForMem(Var->getType());
2371   return Address(
2372       CGF.Builder.CreateBitCast(
2373           Ptr, ElemTy->getPointerTo(Ptr->getType()->getPointerAddressSpace())),
2374       ElemTy, CGF.getContext().getDeclAlign(Var));
2375 }
2376 
2377 static llvm::Value *emitCopyprivateCopyFunction(
2378     CodeGenModule &CGM, llvm::Type *ArgsElemType,
2379     ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
2380     ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps,
2381     SourceLocation Loc) {
2382   ASTContext &C = CGM.getContext();
2383   // void copy_func(void *LHSArg, void *RHSArg);
2384   FunctionArgList Args;
2385   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
2386                            ImplicitParamDecl::Other);
2387   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
2388                            ImplicitParamDecl::Other);
2389   Args.push_back(&LHSArg);
2390   Args.push_back(&RHSArg);
2391   const auto &CGFI =
2392       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
2393   std::string Name =
2394       CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"});
2395   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
2396                                     llvm::GlobalValue::InternalLinkage, Name,
2397                                     &CGM.getModule());
2398   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
2399   Fn->setDoesNotRecurse();
2400   CodeGenFunction CGF(CGM);
2401   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
2402   // Dest = (void*[n])(LHSArg);
2403   // Src = (void*[n])(RHSArg);
2404   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2405                   CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
2406                   ArgsElemType->getPointerTo()),
2407               ArgsElemType, CGF.getPointerAlign());
2408   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2409                   CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
2410                   ArgsElemType->getPointerTo()),
2411               ArgsElemType, CGF.getPointerAlign());
2412   // *(Type0*)Dst[0] = *(Type0*)Src[0];
2413   // *(Type1*)Dst[1] = *(Type1*)Src[1];
2414   // ...
2415   // *(Typen*)Dst[n] = *(Typen*)Src[n];
2416   for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
2417     const auto *DestVar =
2418         cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl());
2419     Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar);
2420 
2421     const auto *SrcVar =
2422         cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl());
2423     Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar);
2424 
2425     const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl();
2426     QualType Type = VD->getType();
2427     CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]);
2428   }
2429   CGF.FinishFunction();
2430   return Fn;
2431 }
2432 
2433 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
2434                                        const RegionCodeGenTy &SingleOpGen,
2435                                        SourceLocation Loc,
2436                                        ArrayRef<const Expr *> CopyprivateVars,
2437                                        ArrayRef<const Expr *> SrcExprs,
2438                                        ArrayRef<const Expr *> DstExprs,
2439                                        ArrayRef<const Expr *> AssignmentOps) {
2440   if (!CGF.HaveInsertPoint())
2441     return;
2442   assert(CopyprivateVars.size() == SrcExprs.size() &&
2443          CopyprivateVars.size() == DstExprs.size() &&
2444          CopyprivateVars.size() == AssignmentOps.size());
2445   ASTContext &C = CGM.getContext();
2446   // int32 did_it = 0;
2447   // if(__kmpc_single(ident_t *, gtid)) {
2448   //   SingleOpGen();
2449   //   __kmpc_end_single(ident_t *, gtid);
2450   //   did_it = 1;
2451   // }
2452   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
2453   // <copy_func>, did_it);
2454 
2455   Address DidIt = Address::invalid();
2456   if (!CopyprivateVars.empty()) {
2457     // int32 did_it = 0;
2458     QualType KmpInt32Ty =
2459         C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
2460     DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it");
2461     CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt);
2462   }
2463   // Prepare arguments and build a call to __kmpc_single
2464   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2465   CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2466                             CGM.getModule(), OMPRTL___kmpc_single),
2467                         Args,
2468                         OMPBuilder.getOrCreateRuntimeFunction(
2469                             CGM.getModule(), OMPRTL___kmpc_end_single),
2470                         Args,
2471                         /*Conditional=*/true);
2472   SingleOpGen.setAction(Action);
2473   emitInlinedDirective(CGF, OMPD_single, SingleOpGen);
2474   if (DidIt.isValid()) {
2475     // did_it = 1;
2476     CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt);
2477   }
2478   Action.Done(CGF);
2479   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
2480   // <copy_func>, did_it);
2481   if (DidIt.isValid()) {
2482     llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
2483     QualType CopyprivateArrayTy = C.getConstantArrayType(
2484         C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
2485         /*IndexTypeQuals=*/0);
2486     // Create a list of all private variables for copyprivate.
2487     Address CopyprivateList =
2488         CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list");
2489     for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
2490       Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I);
2491       CGF.Builder.CreateStore(
2492           CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2493               CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF),
2494               CGF.VoidPtrTy),
2495           Elem);
2496     }
2497     // Build function that copies private values from single region to all other
2498     // threads in the corresponding parallel region.
2499     llvm::Value *CpyFn = emitCopyprivateCopyFunction(
2500         CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy), CopyprivateVars,
2501         SrcExprs, DstExprs, AssignmentOps, Loc);
2502     llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy);
2503     Address CL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2504         CopyprivateList, CGF.VoidPtrTy, CGF.Int8Ty);
2505     llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt);
2506     llvm::Value *Args[] = {
2507         emitUpdateLocation(CGF, Loc), // ident_t *<loc>
2508         getThreadID(CGF, Loc),        // i32 <gtid>
2509         BufSize,                      // size_t <buf_size>
2510         CL.getPointer(),              // void *<copyprivate list>
2511         CpyFn,                        // void (*) (void *, void *) <copy_func>
2512         DidItVal                      // i32 did_it
2513     };
2514     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2515                             CGM.getModule(), OMPRTL___kmpc_copyprivate),
2516                         Args);
2517   }
2518 }
2519 
2520 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF,
2521                                         const RegionCodeGenTy &OrderedOpGen,
2522                                         SourceLocation Loc, bool IsThreads) {
2523   if (!CGF.HaveInsertPoint())
2524     return;
2525   // __kmpc_ordered(ident_t *, gtid);
2526   // OrderedOpGen();
2527   // __kmpc_end_ordered(ident_t *, gtid);
2528   // Prepare arguments and build a call to __kmpc_ordered
2529   if (IsThreads) {
2530     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2531     CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
2532                               CGM.getModule(), OMPRTL___kmpc_ordered),
2533                           Args,
2534                           OMPBuilder.getOrCreateRuntimeFunction(
2535                               CGM.getModule(), OMPRTL___kmpc_end_ordered),
2536                           Args);
2537     OrderedOpGen.setAction(Action);
2538     emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
2539     return;
2540   }
2541   emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen);
2542 }
2543 
2544 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) {
2545   unsigned Flags;
2546   if (Kind == OMPD_for)
2547     Flags = OMP_IDENT_BARRIER_IMPL_FOR;
2548   else if (Kind == OMPD_sections)
2549     Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS;
2550   else if (Kind == OMPD_single)
2551     Flags = OMP_IDENT_BARRIER_IMPL_SINGLE;
2552   else if (Kind == OMPD_barrier)
2553     Flags = OMP_IDENT_BARRIER_EXPL;
2554   else
2555     Flags = OMP_IDENT_BARRIER_IMPL;
2556   return Flags;
2557 }
2558 
2559 void CGOpenMPRuntime::getDefaultScheduleAndChunk(
2560     CodeGenFunction &CGF, const OMPLoopDirective &S,
2561     OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const {
2562   // Check if the loop directive is actually a doacross loop directive. In this
2563   // case choose static, 1 schedule.
2564   if (llvm::any_of(
2565           S.getClausesOfKind<OMPOrderedClause>(),
2566           [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) {
2567     ScheduleKind = OMPC_SCHEDULE_static;
2568     // Chunk size is 1 in this case.
2569     llvm::APInt ChunkSize(32, 1);
2570     ChunkExpr = IntegerLiteral::Create(
2571         CGF.getContext(), ChunkSize,
2572         CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0),
2573         SourceLocation());
2574   }
2575 }
2576 
2577 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
2578                                       OpenMPDirectiveKind Kind, bool EmitChecks,
2579                                       bool ForceSimpleCall) {
2580   // Check if we should use the OMPBuilder
2581   auto *OMPRegionInfo =
2582       dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo);
2583   if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2584     CGF.Builder.restoreIP(OMPBuilder.createBarrier(
2585         CGF.Builder, Kind, ForceSimpleCall, EmitChecks));
2586     return;
2587   }
2588 
2589   if (!CGF.HaveInsertPoint())
2590     return;
2591   // Build call __kmpc_cancel_barrier(loc, thread_id);
2592   // Build call __kmpc_barrier(loc, thread_id);
2593   unsigned Flags = getDefaultFlagsForBarriers(Kind);
2594   // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc,
2595   // thread_id);
2596   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
2597                          getThreadID(CGF, Loc)};
2598   if (OMPRegionInfo) {
2599     if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) {
2600       llvm::Value *Result = CGF.EmitRuntimeCall(
2601           OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
2602                                                 OMPRTL___kmpc_cancel_barrier),
2603           Args);
2604       if (EmitChecks) {
2605         // if (__kmpc_cancel_barrier()) {
2606         //   exit from construct;
2607         // }
2608         llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
2609         llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
2610         llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
2611         CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
2612         CGF.EmitBlock(ExitBB);
2613         //   exit from construct;
2614         CodeGenFunction::JumpDest CancelDestination =
2615             CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
2616         CGF.EmitBranchThroughCleanup(CancelDestination);
2617         CGF.EmitBlock(ContBB, /*IsFinished=*/true);
2618       }
2619       return;
2620     }
2621   }
2622   CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2623                           CGM.getModule(), OMPRTL___kmpc_barrier),
2624                       Args);
2625 }
2626 
2627 /// Map the OpenMP loop schedule to the runtime enumeration.
2628 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
2629                                           bool Chunked, bool Ordered) {
2630   switch (ScheduleKind) {
2631   case OMPC_SCHEDULE_static:
2632     return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked)
2633                    : (Ordered ? OMP_ord_static : OMP_sch_static);
2634   case OMPC_SCHEDULE_dynamic:
2635     return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked;
2636   case OMPC_SCHEDULE_guided:
2637     return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked;
2638   case OMPC_SCHEDULE_runtime:
2639     return Ordered ? OMP_ord_runtime : OMP_sch_runtime;
2640   case OMPC_SCHEDULE_auto:
2641     return Ordered ? OMP_ord_auto : OMP_sch_auto;
2642   case OMPC_SCHEDULE_unknown:
2643     assert(!Chunked && "chunk was specified but schedule kind not known");
2644     return Ordered ? OMP_ord_static : OMP_sch_static;
2645   }
2646   llvm_unreachable("Unexpected runtime schedule");
2647 }
2648 
2649 /// Map the OpenMP distribute schedule to the runtime enumeration.
2650 static OpenMPSchedType
2651 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) {
2652   // only static is allowed for dist_schedule
2653   return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static;
2654 }
2655 
2656 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
2657                                          bool Chunked) const {
2658   OpenMPSchedType Schedule =
2659       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
2660   return Schedule == OMP_sch_static;
2661 }
2662 
2663 bool CGOpenMPRuntime::isStaticNonchunked(
2664     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
2665   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
2666   return Schedule == OMP_dist_sch_static;
2667 }
2668 
2669 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind,
2670                                       bool Chunked) const {
2671   OpenMPSchedType Schedule =
2672       getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false);
2673   return Schedule == OMP_sch_static_chunked;
2674 }
2675 
2676 bool CGOpenMPRuntime::isStaticChunked(
2677     OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const {
2678   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
2679   return Schedule == OMP_dist_sch_static_chunked;
2680 }
2681 
2682 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
2683   OpenMPSchedType Schedule =
2684       getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false);
2685   assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
2686   return Schedule != OMP_sch_static;
2687 }
2688 
2689 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule,
2690                                   OpenMPScheduleClauseModifier M1,
2691                                   OpenMPScheduleClauseModifier M2) {
2692   int Modifier = 0;
2693   switch (M1) {
2694   case OMPC_SCHEDULE_MODIFIER_monotonic:
2695     Modifier = OMP_sch_modifier_monotonic;
2696     break;
2697   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
2698     Modifier = OMP_sch_modifier_nonmonotonic;
2699     break;
2700   case OMPC_SCHEDULE_MODIFIER_simd:
2701     if (Schedule == OMP_sch_static_chunked)
2702       Schedule = OMP_sch_static_balanced_chunked;
2703     break;
2704   case OMPC_SCHEDULE_MODIFIER_last:
2705   case OMPC_SCHEDULE_MODIFIER_unknown:
2706     break;
2707   }
2708   switch (M2) {
2709   case OMPC_SCHEDULE_MODIFIER_monotonic:
2710     Modifier = OMP_sch_modifier_monotonic;
2711     break;
2712   case OMPC_SCHEDULE_MODIFIER_nonmonotonic:
2713     Modifier = OMP_sch_modifier_nonmonotonic;
2714     break;
2715   case OMPC_SCHEDULE_MODIFIER_simd:
2716     if (Schedule == OMP_sch_static_chunked)
2717       Schedule = OMP_sch_static_balanced_chunked;
2718     break;
2719   case OMPC_SCHEDULE_MODIFIER_last:
2720   case OMPC_SCHEDULE_MODIFIER_unknown:
2721     break;
2722   }
2723   // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription.
2724   // If the static schedule kind is specified or if the ordered clause is
2725   // specified, and if the nonmonotonic modifier is not specified, the effect is
2726   // as if the monotonic modifier is specified. Otherwise, unless the monotonic
2727   // modifier is specified, the effect is as if the nonmonotonic modifier is
2728   // specified.
2729   if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) {
2730     if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static ||
2731           Schedule == OMP_sch_static_balanced_chunked ||
2732           Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static ||
2733           Schedule == OMP_dist_sch_static_chunked ||
2734           Schedule == OMP_dist_sch_static))
2735       Modifier = OMP_sch_modifier_nonmonotonic;
2736   }
2737   return Schedule | Modifier;
2738 }
2739 
2740 void CGOpenMPRuntime::emitForDispatchInit(
2741     CodeGenFunction &CGF, SourceLocation Loc,
2742     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
2743     bool Ordered, const DispatchRTInput &DispatchValues) {
2744   if (!CGF.HaveInsertPoint())
2745     return;
2746   OpenMPSchedType Schedule = getRuntimeSchedule(
2747       ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered);
2748   assert(Ordered ||
2749          (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked &&
2750           Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked &&
2751           Schedule != OMP_sch_static_balanced_chunked));
2752   // Call __kmpc_dispatch_init(
2753   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
2754   //          kmp_int[32|64] lower, kmp_int[32|64] upper,
2755   //          kmp_int[32|64] stride, kmp_int[32|64] chunk);
2756 
2757   // If the Chunk was not specified in the clause - use default value 1.
2758   llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk
2759                                             : CGF.Builder.getIntN(IVSize, 1);
2760   llvm::Value *Args[] = {
2761       emitUpdateLocation(CGF, Loc),
2762       getThreadID(CGF, Loc),
2763       CGF.Builder.getInt32(addMonoNonMonoModifier(
2764           CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type
2765       DispatchValues.LB,                                     // Lower
2766       DispatchValues.UB,                                     // Upper
2767       CGF.Builder.getIntN(IVSize, 1),                        // Stride
2768       Chunk                                                  // Chunk
2769   };
2770   CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args);
2771 }
2772 
2773 static void emitForStaticInitCall(
2774     CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId,
2775     llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule,
2776     OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2,
2777     const CGOpenMPRuntime::StaticRTInput &Values) {
2778   if (!CGF.HaveInsertPoint())
2779     return;
2780 
2781   assert(!Values.Ordered);
2782   assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked ||
2783          Schedule == OMP_sch_static_balanced_chunked ||
2784          Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked ||
2785          Schedule == OMP_dist_sch_static ||
2786          Schedule == OMP_dist_sch_static_chunked);
2787 
2788   // Call __kmpc_for_static_init(
2789   //          ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
2790   //          kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
2791   //          kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
2792   //          kmp_int[32|64] incr, kmp_int[32|64] chunk);
2793   llvm::Value *Chunk = Values.Chunk;
2794   if (Chunk == nullptr) {
2795     assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static ||
2796             Schedule == OMP_dist_sch_static) &&
2797            "expected static non-chunked schedule");
2798     // If the Chunk was not specified in the clause - use default value 1.
2799     Chunk = CGF.Builder.getIntN(Values.IVSize, 1);
2800   } else {
2801     assert((Schedule == OMP_sch_static_chunked ||
2802             Schedule == OMP_sch_static_balanced_chunked ||
2803             Schedule == OMP_ord_static_chunked ||
2804             Schedule == OMP_dist_sch_static_chunked) &&
2805            "expected static chunked schedule");
2806   }
2807   llvm::Value *Args[] = {
2808       UpdateLocation,
2809       ThreadId,
2810       CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1,
2811                                                   M2)), // Schedule type
2812       Values.IL.getPointer(),                           // &isLastIter
2813       Values.LB.getPointer(),                           // &LB
2814       Values.UB.getPointer(),                           // &UB
2815       Values.ST.getPointer(),                           // &Stride
2816       CGF.Builder.getIntN(Values.IVSize, 1),            // Incr
2817       Chunk                                             // Chunk
2818   };
2819   CGF.EmitRuntimeCall(ForStaticInitFunction, Args);
2820 }
2821 
2822 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF,
2823                                         SourceLocation Loc,
2824                                         OpenMPDirectiveKind DKind,
2825                                         const OpenMPScheduleTy &ScheduleKind,
2826                                         const StaticRTInput &Values) {
2827   OpenMPSchedType ScheduleNum = getRuntimeSchedule(
2828       ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered);
2829   assert(isOpenMPWorksharingDirective(DKind) &&
2830          "Expected loop-based or sections-based directive.");
2831   llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc,
2832                                              isOpenMPLoopDirective(DKind)
2833                                                  ? OMP_IDENT_WORK_LOOP
2834                                                  : OMP_IDENT_WORK_SECTIONS);
2835   llvm::Value *ThreadId = getThreadID(CGF, Loc);
2836   llvm::FunctionCallee StaticInitFunction =
2837       createForStaticInitFunction(Values.IVSize, Values.IVSigned, false);
2838   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
2839   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
2840                         ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values);
2841 }
2842 
2843 void CGOpenMPRuntime::emitDistributeStaticInit(
2844     CodeGenFunction &CGF, SourceLocation Loc,
2845     OpenMPDistScheduleClauseKind SchedKind,
2846     const CGOpenMPRuntime::StaticRTInput &Values) {
2847   OpenMPSchedType ScheduleNum =
2848       getRuntimeSchedule(SchedKind, Values.Chunk != nullptr);
2849   llvm::Value *UpdatedLocation =
2850       emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE);
2851   llvm::Value *ThreadId = getThreadID(CGF, Loc);
2852   llvm::FunctionCallee StaticInitFunction;
2853   bool isGPUDistribute =
2854       CGM.getLangOpts().OpenMPIsDevice &&
2855       (CGM.getTriple().isAMDGCN() || CGM.getTriple().isNVPTX());
2856   StaticInitFunction = createForStaticInitFunction(
2857       Values.IVSize, Values.IVSigned, isGPUDistribute);
2858 
2859   emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction,
2860                         ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown,
2861                         OMPC_SCHEDULE_MODIFIER_unknown, Values);
2862 }
2863 
2864 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF,
2865                                           SourceLocation Loc,
2866                                           OpenMPDirectiveKind DKind) {
2867   if (!CGF.HaveInsertPoint())
2868     return;
2869   // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
2870   llvm::Value *Args[] = {
2871       emitUpdateLocation(CGF, Loc,
2872                          isOpenMPDistributeDirective(DKind)
2873                              ? OMP_IDENT_WORK_DISTRIBUTE
2874                              : isOpenMPLoopDirective(DKind)
2875                                    ? OMP_IDENT_WORK_LOOP
2876                                    : OMP_IDENT_WORK_SECTIONS),
2877       getThreadID(CGF, Loc)};
2878   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
2879   if (isOpenMPDistributeDirective(DKind) && CGM.getLangOpts().OpenMPIsDevice &&
2880       (CGM.getTriple().isAMDGCN() || CGM.getTriple().isNVPTX()))
2881     CGF.EmitRuntimeCall(
2882         OMPBuilder.getOrCreateRuntimeFunction(
2883             CGM.getModule(), OMPRTL___kmpc_distribute_static_fini),
2884         Args);
2885   else
2886     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2887                             CGM.getModule(), OMPRTL___kmpc_for_static_fini),
2888                         Args);
2889 }
2890 
2891 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
2892                                                  SourceLocation Loc,
2893                                                  unsigned IVSize,
2894                                                  bool IVSigned) {
2895   if (!CGF.HaveInsertPoint())
2896     return;
2897   // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
2898   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2899   CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args);
2900 }
2901 
2902 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
2903                                           SourceLocation Loc, unsigned IVSize,
2904                                           bool IVSigned, Address IL,
2905                                           Address LB, Address UB,
2906                                           Address ST) {
2907   // Call __kmpc_dispatch_next(
2908   //          ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
2909   //          kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
2910   //          kmp_int[32|64] *p_stride);
2911   llvm::Value *Args[] = {
2912       emitUpdateLocation(CGF, Loc),
2913       getThreadID(CGF, Loc),
2914       IL.getPointer(), // &isLastIter
2915       LB.getPointer(), // &Lower
2916       UB.getPointer(), // &Upper
2917       ST.getPointer()  // &Stride
2918   };
2919   llvm::Value *Call =
2920       CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args);
2921   return CGF.EmitScalarConversion(
2922       Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1),
2923       CGF.getContext().BoolTy, Loc);
2924 }
2925 
2926 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
2927                                            llvm::Value *NumThreads,
2928                                            SourceLocation Loc) {
2929   if (!CGF.HaveInsertPoint())
2930     return;
2931   // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
2932   llvm::Value *Args[] = {
2933       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2934       CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)};
2935   CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2936                           CGM.getModule(), OMPRTL___kmpc_push_num_threads),
2937                       Args);
2938 }
2939 
2940 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF,
2941                                          ProcBindKind ProcBind,
2942                                          SourceLocation Loc) {
2943   if (!CGF.HaveInsertPoint())
2944     return;
2945   assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value.");
2946   // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind)
2947   llvm::Value *Args[] = {
2948       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
2949       llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)};
2950   CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2951                           CGM.getModule(), OMPRTL___kmpc_push_proc_bind),
2952                       Args);
2953 }
2954 
2955 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
2956                                 SourceLocation Loc, llvm::AtomicOrdering AO) {
2957   if (CGF.CGM.getLangOpts().OpenMPIRBuilder) {
2958     OMPBuilder.createFlush(CGF.Builder);
2959   } else {
2960     if (!CGF.HaveInsertPoint())
2961       return;
2962     // Build call void __kmpc_flush(ident_t *loc)
2963     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
2964                             CGM.getModule(), OMPRTL___kmpc_flush),
2965                         emitUpdateLocation(CGF, Loc));
2966   }
2967 }
2968 
2969 namespace {
2970 /// Indexes of fields for type kmp_task_t.
2971 enum KmpTaskTFields {
2972   /// List of shared variables.
2973   KmpTaskTShareds,
2974   /// Task routine.
2975   KmpTaskTRoutine,
2976   /// Partition id for the untied tasks.
2977   KmpTaskTPartId,
2978   /// Function with call of destructors for private variables.
2979   Data1,
2980   /// Task priority.
2981   Data2,
2982   /// (Taskloops only) Lower bound.
2983   KmpTaskTLowerBound,
2984   /// (Taskloops only) Upper bound.
2985   KmpTaskTUpperBound,
2986   /// (Taskloops only) Stride.
2987   KmpTaskTStride,
2988   /// (Taskloops only) Is last iteration flag.
2989   KmpTaskTLastIter,
2990   /// (Taskloops only) Reduction data.
2991   KmpTaskTReductions,
2992 };
2993 } // anonymous namespace
2994 
2995 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const {
2996   return OffloadEntriesTargetRegion.empty() &&
2997          OffloadEntriesDeviceGlobalVar.empty();
2998 }
2999 
3000 /// Initialize target region entry.
3001 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3002     initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3003                                     StringRef ParentName, unsigned LineNum,
3004                                     unsigned Order) {
3005   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3006                                              "only required for the device "
3007                                              "code generation.");
3008   OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] =
3009       OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr,
3010                                    OMPTargetRegionEntryTargetRegion);
3011   ++OffloadingEntriesNum;
3012 }
3013 
3014 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3015     registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID,
3016                                   StringRef ParentName, unsigned LineNum,
3017                                   llvm::Constant *Addr, llvm::Constant *ID,
3018                                   OMPTargetRegionEntryKind Flags) {
3019   // If we are emitting code for a target, the entry is already initialized,
3020   // only has to be registered.
3021   if (CGM.getLangOpts().OpenMPIsDevice) {
3022     // This could happen if the device compilation is invoked standalone.
3023     if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum))
3024       return;
3025     auto &Entry =
3026         OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum];
3027     Entry.setAddress(Addr);
3028     Entry.setID(ID);
3029     Entry.setFlags(Flags);
3030   } else {
3031     if (Flags ==
3032             OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion &&
3033         hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum,
3034                                  /*IgnoreAddressId*/ true))
3035       return;
3036     assert(!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum) &&
3037            "Target region entry already registered!");
3038     OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags);
3039     OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry;
3040     ++OffloadingEntriesNum;
3041   }
3042 }
3043 
3044 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo(
3045     unsigned DeviceID, unsigned FileID, StringRef ParentName, unsigned LineNum,
3046     bool IgnoreAddressId) const {
3047   auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID);
3048   if (PerDevice == OffloadEntriesTargetRegion.end())
3049     return false;
3050   auto PerFile = PerDevice->second.find(FileID);
3051   if (PerFile == PerDevice->second.end())
3052     return false;
3053   auto PerParentName = PerFile->second.find(ParentName);
3054   if (PerParentName == PerFile->second.end())
3055     return false;
3056   auto PerLine = PerParentName->second.find(LineNum);
3057   if (PerLine == PerParentName->second.end())
3058     return false;
3059   // Fail if this entry is already registered.
3060   if (!IgnoreAddressId &&
3061       (PerLine->second.getAddress() || PerLine->second.getID()))
3062     return false;
3063   return true;
3064 }
3065 
3066 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo(
3067     const OffloadTargetRegionEntryInfoActTy &Action) {
3068   // Scan all target region entries and perform the provided action.
3069   for (const auto &D : OffloadEntriesTargetRegion)
3070     for (const auto &F : D.second)
3071       for (const auto &P : F.second)
3072         for (const auto &L : P.second)
3073           Action(D.first, F.first, P.first(), L.first, L.second);
3074 }
3075 
3076 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3077     initializeDeviceGlobalVarEntryInfo(StringRef Name,
3078                                        OMPTargetGlobalVarEntryKind Flags,
3079                                        unsigned Order) {
3080   assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is "
3081                                              "only required for the device "
3082                                              "code generation.");
3083   OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
3084   ++OffloadingEntriesNum;
3085 }
3086 
3087 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3088     registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr,
3089                                      CharUnits VarSize,
3090                                      OMPTargetGlobalVarEntryKind Flags,
3091                                      llvm::GlobalValue::LinkageTypes Linkage) {
3092   if (CGM.getLangOpts().OpenMPIsDevice) {
3093     // This could happen if the device compilation is invoked standalone.
3094     if (!hasDeviceGlobalVarEntryInfo(VarName))
3095       return;
3096     auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
3097     if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) {
3098       if (Entry.getVarSize().isZero()) {
3099         Entry.setVarSize(VarSize);
3100         Entry.setLinkage(Linkage);
3101       }
3102       return;
3103     }
3104     Entry.setVarSize(VarSize);
3105     Entry.setLinkage(Linkage);
3106     Entry.setAddress(Addr);
3107   } else {
3108     if (hasDeviceGlobalVarEntryInfo(VarName)) {
3109       auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
3110       assert(Entry.isValid() && Entry.getFlags() == Flags &&
3111              "Entry not initialized!");
3112       if (Entry.getVarSize().isZero()) {
3113         Entry.setVarSize(VarSize);
3114         Entry.setLinkage(Linkage);
3115       }
3116       return;
3117     }
3118     OffloadEntriesDeviceGlobalVar.try_emplace(
3119         VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage);
3120     ++OffloadingEntriesNum;
3121   }
3122 }
3123 
3124 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::
3125     actOnDeviceGlobalVarEntriesInfo(
3126         const OffloadDeviceGlobalVarEntryInfoActTy &Action) {
3127   // Scan all target region entries and perform the provided action.
3128   for (const auto &E : OffloadEntriesDeviceGlobalVar)
3129     Action(E.getKey(), E.getValue());
3130 }
3131 
3132 void CGOpenMPRuntime::createOffloadEntry(
3133     llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags,
3134     llvm::GlobalValue::LinkageTypes Linkage) {
3135   StringRef Name = Addr->getName();
3136   llvm::Module &M = CGM.getModule();
3137   llvm::LLVMContext &C = M.getContext();
3138 
3139   // Create constant string with the name.
3140   llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name);
3141 
3142   std::string StringName = getName({"omp_offloading", "entry_name"});
3143   auto *Str = new llvm::GlobalVariable(
3144       M, StrPtrInit->getType(), /*isConstant=*/true,
3145       llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName);
3146   Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
3147 
3148   llvm::Constant *Data[] = {
3149       llvm::ConstantExpr::getPointerBitCastOrAddrSpaceCast(ID, CGM.VoidPtrTy),
3150       llvm::ConstantExpr::getPointerBitCastOrAddrSpaceCast(Str, CGM.Int8PtrTy),
3151       llvm::ConstantInt::get(CGM.SizeTy, Size),
3152       llvm::ConstantInt::get(CGM.Int32Ty, Flags),
3153       llvm::ConstantInt::get(CGM.Int32Ty, 0)};
3154   std::string EntryName = getName({"omp_offloading", "entry", ""});
3155   llvm::GlobalVariable *Entry = createGlobalStruct(
3156       CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data,
3157       Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage);
3158 
3159   // The entry has to be created in the section the linker expects it to be.
3160   Entry->setSection("omp_offloading_entries");
3161 }
3162 
3163 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() {
3164   // Emit the offloading entries and metadata so that the device codegen side
3165   // can easily figure out what to emit. The produced metadata looks like
3166   // this:
3167   //
3168   // !omp_offload.info = !{!1, ...}
3169   //
3170   // Right now we only generate metadata for function that contain target
3171   // regions.
3172 
3173   // If we are in simd mode or there are no entries, we don't need to do
3174   // anything.
3175   if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty())
3176     return;
3177 
3178   llvm::Module &M = CGM.getModule();
3179   llvm::LLVMContext &C = M.getContext();
3180   SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *,
3181                          SourceLocation, StringRef>,
3182               16>
3183       OrderedEntries(OffloadEntriesInfoManager.size());
3184   llvm::SmallVector<StringRef, 16> ParentFunctions(
3185       OffloadEntriesInfoManager.size());
3186 
3187   // Auxiliary methods to create metadata values and strings.
3188   auto &&GetMDInt = [this](unsigned V) {
3189     return llvm::ConstantAsMetadata::get(
3190         llvm::ConstantInt::get(CGM.Int32Ty, V));
3191   };
3192 
3193   auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); };
3194 
3195   // Create the offloading info metadata node.
3196   llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info");
3197 
3198   // Create function that emits metadata for each target region entry;
3199   auto &&TargetRegionMetadataEmitter =
3200       [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt,
3201        &GetMDString](
3202           unsigned DeviceID, unsigned FileID, StringRef ParentName,
3203           unsigned Line,
3204           const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) {
3205         // Generate metadata for target regions. Each entry of this metadata
3206         // contains:
3207         // - Entry 0 -> Kind of this type of metadata (0).
3208         // - Entry 1 -> Device ID of the file where the entry was identified.
3209         // - Entry 2 -> File ID of the file where the entry was identified.
3210         // - Entry 3 -> Mangled name of the function where the entry was
3211         // identified.
3212         // - Entry 4 -> Line in the file where the entry was identified.
3213         // - Entry 5 -> Order the entry was created.
3214         // The first element of the metadata node is the kind.
3215         llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID),
3216                                  GetMDInt(FileID),      GetMDString(ParentName),
3217                                  GetMDInt(Line),        GetMDInt(E.getOrder())};
3218 
3219         SourceLocation Loc;
3220         for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(),
3221                   E = CGM.getContext().getSourceManager().fileinfo_end();
3222              I != E; ++I) {
3223           if (I->getFirst()->getUniqueID().getDevice() == DeviceID &&
3224               I->getFirst()->getUniqueID().getFile() == FileID) {
3225             Loc = CGM.getContext().getSourceManager().translateFileLineCol(
3226                 I->getFirst(), Line, 1);
3227             break;
3228           }
3229         }
3230         // Save this entry in the right position of the ordered entries array.
3231         OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName);
3232         ParentFunctions[E.getOrder()] = ParentName;
3233 
3234         // Add metadata to the named metadata node.
3235         MD->addOperand(llvm::MDNode::get(C, Ops));
3236       };
3237 
3238   OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo(
3239       TargetRegionMetadataEmitter);
3240 
3241   // Create function that emits metadata for each device global variable entry;
3242   auto &&DeviceGlobalVarMetadataEmitter =
3243       [&C, &OrderedEntries, &GetMDInt, &GetMDString,
3244        MD](StringRef MangledName,
3245            const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar
3246                &E) {
3247         // Generate metadata for global variables. Each entry of this metadata
3248         // contains:
3249         // - Entry 0 -> Kind of this type of metadata (1).
3250         // - Entry 1 -> Mangled name of the variable.
3251         // - Entry 2 -> Declare target kind.
3252         // - Entry 3 -> Order the entry was created.
3253         // The first element of the metadata node is the kind.
3254         llvm::Metadata *Ops[] = {
3255             GetMDInt(E.getKind()), GetMDString(MangledName),
3256             GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
3257 
3258         // Save this entry in the right position of the ordered entries array.
3259         OrderedEntries[E.getOrder()] =
3260             std::make_tuple(&E, SourceLocation(), MangledName);
3261 
3262         // Add metadata to the named metadata node.
3263         MD->addOperand(llvm::MDNode::get(C, Ops));
3264       };
3265 
3266   OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo(
3267       DeviceGlobalVarMetadataEmitter);
3268 
3269   for (const auto &E : OrderedEntries) {
3270     assert(std::get<0>(E) && "All ordered entries must exist!");
3271     if (const auto *CE =
3272             dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>(
3273                 std::get<0>(E))) {
3274       if (!CE->getID() || !CE->getAddress()) {
3275         // Do not blame the entry if the parent funtion is not emitted.
3276         StringRef FnName = ParentFunctions[CE->getOrder()];
3277         if (!CGM.GetGlobalValue(FnName))
3278           continue;
3279         unsigned DiagID = CGM.getDiags().getCustomDiagID(
3280             DiagnosticsEngine::Error,
3281             "Offloading entry for target region in %0 is incorrect: either the "
3282             "address or the ID is invalid.");
3283         CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName;
3284         continue;
3285       }
3286       createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0,
3287                          CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage);
3288     } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy::
3289                                              OffloadEntryInfoDeviceGlobalVar>(
3290                    std::get<0>(E))) {
3291       OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags =
3292           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
3293               CE->getFlags());
3294       switch (Flags) {
3295       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: {
3296         if (CGM.getLangOpts().OpenMPIsDevice &&
3297             CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())
3298           continue;
3299         if (!CE->getAddress()) {
3300           unsigned DiagID = CGM.getDiags().getCustomDiagID(
3301               DiagnosticsEngine::Error, "Offloading entry for declare target "
3302                                         "variable %0 is incorrect: the "
3303                                         "address is invalid.");
3304           CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E);
3305           continue;
3306         }
3307         // The vaiable has no definition - no need to add the entry.
3308         if (CE->getVarSize().isZero())
3309           continue;
3310         break;
3311       }
3312       case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink:
3313         assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) ||
3314                 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) &&
3315                "Declaret target link address is set.");
3316         if (CGM.getLangOpts().OpenMPIsDevice)
3317           continue;
3318         if (!CE->getAddress()) {
3319           unsigned DiagID = CGM.getDiags().getCustomDiagID(
3320               DiagnosticsEngine::Error,
3321               "Offloading entry for declare target variable is incorrect: the "
3322               "address is invalid.");
3323           CGM.getDiags().Report(DiagID);
3324           continue;
3325         }
3326         break;
3327       }
3328       createOffloadEntry(CE->getAddress(), CE->getAddress(),
3329                          CE->getVarSize().getQuantity(), Flags,
3330                          CE->getLinkage());
3331     } else {
3332       llvm_unreachable("Unsupported entry kind.");
3333     }
3334   }
3335 }
3336 
3337 /// Loads all the offload entries information from the host IR
3338 /// metadata.
3339 void CGOpenMPRuntime::loadOffloadInfoMetadata() {
3340   // If we are in target mode, load the metadata from the host IR. This code has
3341   // to match the metadaata creation in createOffloadEntriesAndInfoMetadata().
3342 
3343   if (!CGM.getLangOpts().OpenMPIsDevice)
3344     return;
3345 
3346   if (CGM.getLangOpts().OMPHostIRFile.empty())
3347     return;
3348 
3349   auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile);
3350   if (auto EC = Buf.getError()) {
3351     CGM.getDiags().Report(diag::err_cannot_open_file)
3352         << CGM.getLangOpts().OMPHostIRFile << EC.message();
3353     return;
3354   }
3355 
3356   llvm::LLVMContext C;
3357   auto ME = expectedToErrorOrAndEmitErrors(
3358       C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C));
3359 
3360   if (auto EC = ME.getError()) {
3361     unsigned DiagID = CGM.getDiags().getCustomDiagID(
3362         DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'");
3363     CGM.getDiags().Report(DiagID)
3364         << CGM.getLangOpts().OMPHostIRFile << EC.message();
3365     return;
3366   }
3367 
3368   llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info");
3369   if (!MD)
3370     return;
3371 
3372   for (llvm::MDNode *MN : MD->operands()) {
3373     auto &&GetMDInt = [MN](unsigned Idx) {
3374       auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx));
3375       return cast<llvm::ConstantInt>(V->getValue())->getZExtValue();
3376     };
3377 
3378     auto &&GetMDString = [MN](unsigned Idx) {
3379       auto *V = cast<llvm::MDString>(MN->getOperand(Idx));
3380       return V->getString();
3381     };
3382 
3383     switch (GetMDInt(0)) {
3384     default:
3385       llvm_unreachable("Unexpected metadata!");
3386       break;
3387     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
3388         OffloadingEntryInfoTargetRegion:
3389       OffloadEntriesInfoManager.initializeTargetRegionEntryInfo(
3390           /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2),
3391           /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4),
3392           /*Order=*/GetMDInt(5));
3393       break;
3394     case OffloadEntriesInfoManagerTy::OffloadEntryInfo::
3395         OffloadingEntryInfoDeviceGlobalVar:
3396       OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo(
3397           /*MangledName=*/GetMDString(1),
3398           static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>(
3399               /*Flags=*/GetMDInt(2)),
3400           /*Order=*/GetMDInt(3));
3401       break;
3402     }
3403   }
3404 }
3405 
3406 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
3407   if (!KmpRoutineEntryPtrTy) {
3408     // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
3409     ASTContext &C = CGM.getContext();
3410     QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
3411     FunctionProtoType::ExtProtoInfo EPI;
3412     KmpRoutineEntryPtrQTy = C.getPointerType(
3413         C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
3414     KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
3415   }
3416 }
3417 
3418 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() {
3419   // Make sure the type of the entry is already created. This is the type we
3420   // have to create:
3421   // struct __tgt_offload_entry{
3422   //   void      *addr;       // Pointer to the offload entry info.
3423   //                          // (function or global)
3424   //   char      *name;       // Name of the function or global.
3425   //   size_t     size;       // Size of the entry info (0 if it a function).
3426   //   int32_t    flags;      // Flags associated with the entry, e.g. 'link'.
3427   //   int32_t    reserved;   // Reserved, to use by the runtime library.
3428   // };
3429   if (TgtOffloadEntryQTy.isNull()) {
3430     ASTContext &C = CGM.getContext();
3431     RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry");
3432     RD->startDefinition();
3433     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
3434     addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy));
3435     addFieldToRecordDecl(C, RD, C.getSizeType());
3436     addFieldToRecordDecl(
3437         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
3438     addFieldToRecordDecl(
3439         C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true));
3440     RD->completeDefinition();
3441     RD->addAttr(PackedAttr::CreateImplicit(C));
3442     TgtOffloadEntryQTy = C.getRecordType(RD);
3443   }
3444   return TgtOffloadEntryQTy;
3445 }
3446 
3447 namespace {
3448 struct PrivateHelpersTy {
3449   PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original,
3450                    const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit)
3451       : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy),
3452         PrivateElemInit(PrivateElemInit) {}
3453   PrivateHelpersTy(const VarDecl *Original) : Original(Original) {}
3454   const Expr *OriginalRef = nullptr;
3455   const VarDecl *Original = nullptr;
3456   const VarDecl *PrivateCopy = nullptr;
3457   const VarDecl *PrivateElemInit = nullptr;
3458   bool isLocalPrivate() const {
3459     return !OriginalRef && !PrivateCopy && !PrivateElemInit;
3460   }
3461 };
3462 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
3463 } // anonymous namespace
3464 
3465 static bool isAllocatableDecl(const VarDecl *VD) {
3466   const VarDecl *CVD = VD->getCanonicalDecl();
3467   if (!CVD->hasAttr<OMPAllocateDeclAttr>())
3468     return false;
3469   const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
3470   // Use the default allocation.
3471   return !(AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc &&
3472            !AA->getAllocator());
3473 }
3474 
3475 static RecordDecl *
3476 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) {
3477   if (!Privates.empty()) {
3478     ASTContext &C = CGM.getContext();
3479     // Build struct .kmp_privates_t. {
3480     //         /*  private vars  */
3481     //       };
3482     RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t");
3483     RD->startDefinition();
3484     for (const auto &Pair : Privates) {
3485       const VarDecl *VD = Pair.second.Original;
3486       QualType Type = VD->getType().getNonReferenceType();
3487       // If the private variable is a local variable with lvalue ref type,
3488       // allocate the pointer instead of the pointee type.
3489       if (Pair.second.isLocalPrivate()) {
3490         if (VD->getType()->isLValueReferenceType())
3491           Type = C.getPointerType(Type);
3492         if (isAllocatableDecl(VD))
3493           Type = C.getPointerType(Type);
3494       }
3495       FieldDecl *FD = addFieldToRecordDecl(C, RD, Type);
3496       if (VD->hasAttrs()) {
3497         for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()),
3498              E(VD->getAttrs().end());
3499              I != E; ++I)
3500           FD->addAttr(*I);
3501       }
3502     }
3503     RD->completeDefinition();
3504     return RD;
3505   }
3506   return nullptr;
3507 }
3508 
3509 static RecordDecl *
3510 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind,
3511                          QualType KmpInt32Ty,
3512                          QualType KmpRoutineEntryPointerQTy) {
3513   ASTContext &C = CGM.getContext();
3514   // Build struct kmp_task_t {
3515   //         void *              shareds;
3516   //         kmp_routine_entry_t routine;
3517   //         kmp_int32           part_id;
3518   //         kmp_cmplrdata_t data1;
3519   //         kmp_cmplrdata_t data2;
3520   // For taskloops additional fields:
3521   //         kmp_uint64          lb;
3522   //         kmp_uint64          ub;
3523   //         kmp_int64           st;
3524   //         kmp_int32           liter;
3525   //         void *              reductions;
3526   //       };
3527   RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union);
3528   UD->startDefinition();
3529   addFieldToRecordDecl(C, UD, KmpInt32Ty);
3530   addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy);
3531   UD->completeDefinition();
3532   QualType KmpCmplrdataTy = C.getRecordType(UD);
3533   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t");
3534   RD->startDefinition();
3535   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
3536   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
3537   addFieldToRecordDecl(C, RD, KmpInt32Ty);
3538   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
3539   addFieldToRecordDecl(C, RD, KmpCmplrdataTy);
3540   if (isOpenMPTaskLoopDirective(Kind)) {
3541     QualType KmpUInt64Ty =
3542         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
3543     QualType KmpInt64Ty =
3544         CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
3545     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
3546     addFieldToRecordDecl(C, RD, KmpUInt64Ty);
3547     addFieldToRecordDecl(C, RD, KmpInt64Ty);
3548     addFieldToRecordDecl(C, RD, KmpInt32Ty);
3549     addFieldToRecordDecl(C, RD, C.VoidPtrTy);
3550   }
3551   RD->completeDefinition();
3552   return RD;
3553 }
3554 
3555 static RecordDecl *
3556 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy,
3557                                      ArrayRef<PrivateDataTy> Privates) {
3558   ASTContext &C = CGM.getContext();
3559   // Build struct kmp_task_t_with_privates {
3560   //         kmp_task_t task_data;
3561   //         .kmp_privates_t. privates;
3562   //       };
3563   RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates");
3564   RD->startDefinition();
3565   addFieldToRecordDecl(C, RD, KmpTaskTQTy);
3566   if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates))
3567     addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD));
3568   RD->completeDefinition();
3569   return RD;
3570 }
3571 
3572 /// Emit a proxy function which accepts kmp_task_t as the second
3573 /// argument.
3574 /// \code
3575 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
3576 ///   TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt,
3577 ///   For taskloops:
3578 ///   tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
3579 ///   tt->reductions, tt->shareds);
3580 ///   return 0;
3581 /// }
3582 /// \endcode
3583 static llvm::Function *
3584 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
3585                       OpenMPDirectiveKind Kind, QualType KmpInt32Ty,
3586                       QualType KmpTaskTWithPrivatesPtrQTy,
3587                       QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
3588                       QualType SharedsPtrTy, llvm::Function *TaskFunction,
3589                       llvm::Value *TaskPrivatesMap) {
3590   ASTContext &C = CGM.getContext();
3591   FunctionArgList Args;
3592   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
3593                             ImplicitParamDecl::Other);
3594   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3595                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
3596                                 ImplicitParamDecl::Other);
3597   Args.push_back(&GtidArg);
3598   Args.push_back(&TaskTypeArg);
3599   const auto &TaskEntryFnInfo =
3600       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
3601   llvm::FunctionType *TaskEntryTy =
3602       CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
3603   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""});
3604   auto *TaskEntry = llvm::Function::Create(
3605       TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
3606   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo);
3607   TaskEntry->setDoesNotRecurse();
3608   CodeGenFunction CGF(CGM);
3609   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args,
3610                     Loc, Loc);
3611 
3612   // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map,
3613   // tt,
3614   // For taskloops:
3615   // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter,
3616   // tt->task_data.shareds);
3617   llvm::Value *GtidParam = CGF.EmitLoadOfScalar(
3618       CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc);
3619   LValue TDBase = CGF.EmitLoadOfPointerLValue(
3620       CGF.GetAddrOfLocalVar(&TaskTypeArg),
3621       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3622   const auto *KmpTaskTWithPrivatesQTyRD =
3623       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
3624   LValue Base =
3625       CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
3626   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
3627   auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
3628   LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI);
3629   llvm::Value *PartidParam = PartIdLVal.getPointer(CGF);
3630 
3631   auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds);
3632   LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI);
3633   llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3634       CGF.EmitLoadOfScalar(SharedsLVal, Loc),
3635       CGF.ConvertTypeForMem(SharedsPtrTy));
3636 
3637   auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1);
3638   llvm::Value *PrivatesParam;
3639   if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) {
3640     LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI);
3641     PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3642         PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy);
3643   } else {
3644     PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
3645   }
3646 
3647   llvm::Value *CommonArgs[] = {
3648       GtidParam, PartidParam, PrivatesParam, TaskPrivatesMap,
3649       CGF.Builder
3650           .CreatePointerBitCastOrAddrSpaceCast(TDBase.getAddress(CGF),
3651                                                CGF.VoidPtrTy, CGF.Int8Ty)
3652           .getPointer()};
3653   SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs),
3654                                           std::end(CommonArgs));
3655   if (isOpenMPTaskLoopDirective(Kind)) {
3656     auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound);
3657     LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI);
3658     llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc);
3659     auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound);
3660     LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI);
3661     llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc);
3662     auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride);
3663     LValue StLVal = CGF.EmitLValueForField(Base, *StFI);
3664     llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc);
3665     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
3666     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
3667     llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc);
3668     auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions);
3669     LValue RLVal = CGF.EmitLValueForField(Base, *RFI);
3670     llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc);
3671     CallArgs.push_back(LBParam);
3672     CallArgs.push_back(UBParam);
3673     CallArgs.push_back(StParam);
3674     CallArgs.push_back(LIParam);
3675     CallArgs.push_back(RParam);
3676   }
3677   CallArgs.push_back(SharedsParam);
3678 
3679   CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction,
3680                                                   CallArgs);
3681   CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)),
3682                              CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty));
3683   CGF.FinishFunction();
3684   return TaskEntry;
3685 }
3686 
3687 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM,
3688                                             SourceLocation Loc,
3689                                             QualType KmpInt32Ty,
3690                                             QualType KmpTaskTWithPrivatesPtrQTy,
3691                                             QualType KmpTaskTWithPrivatesQTy) {
3692   ASTContext &C = CGM.getContext();
3693   FunctionArgList Args;
3694   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty,
3695                             ImplicitParamDecl::Other);
3696   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3697                                 KmpTaskTWithPrivatesPtrQTy.withRestrict(),
3698                                 ImplicitParamDecl::Other);
3699   Args.push_back(&GtidArg);
3700   Args.push_back(&TaskTypeArg);
3701   const auto &DestructorFnInfo =
3702       CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args);
3703   llvm::FunctionType *DestructorFnTy =
3704       CGM.getTypes().GetFunctionType(DestructorFnInfo);
3705   std::string Name =
3706       CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""});
3707   auto *DestructorFn =
3708       llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage,
3709                              Name, &CGM.getModule());
3710   CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn,
3711                                     DestructorFnInfo);
3712   DestructorFn->setDoesNotRecurse();
3713   CodeGenFunction CGF(CGM);
3714   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo,
3715                     Args, Loc, Loc);
3716 
3717   LValue Base = CGF.EmitLoadOfPointerLValue(
3718       CGF.GetAddrOfLocalVar(&TaskTypeArg),
3719       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
3720   const auto *KmpTaskTWithPrivatesQTyRD =
3721       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
3722   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
3723   Base = CGF.EmitLValueForField(Base, *FI);
3724   for (const auto *Field :
3725        cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) {
3726     if (QualType::DestructionKind DtorKind =
3727             Field->getType().isDestructedType()) {
3728       LValue FieldLValue = CGF.EmitLValueForField(Base, Field);
3729       CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType());
3730     }
3731   }
3732   CGF.FinishFunction();
3733   return DestructorFn;
3734 }
3735 
3736 /// Emit a privates mapping function for correct handling of private and
3737 /// firstprivate variables.
3738 /// \code
3739 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1>
3740 /// **noalias priv1,...,  <tyn> **noalias privn) {
3741 ///   *priv1 = &.privates.priv1;
3742 ///   ...;
3743 ///   *privn = &.privates.privn;
3744 /// }
3745 /// \endcode
3746 static llvm::Value *
3747 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc,
3748                                const OMPTaskDataTy &Data, QualType PrivatesQTy,
3749                                ArrayRef<PrivateDataTy> Privates) {
3750   ASTContext &C = CGM.getContext();
3751   FunctionArgList Args;
3752   ImplicitParamDecl TaskPrivatesArg(
3753       C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3754       C.getPointerType(PrivatesQTy).withConst().withRestrict(),
3755       ImplicitParamDecl::Other);
3756   Args.push_back(&TaskPrivatesArg);
3757   llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, unsigned> PrivateVarsPos;
3758   unsigned Counter = 1;
3759   for (const Expr *E : Data.PrivateVars) {
3760     Args.push_back(ImplicitParamDecl::Create(
3761         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3762         C.getPointerType(C.getPointerType(E->getType()))
3763             .withConst()
3764             .withRestrict(),
3765         ImplicitParamDecl::Other));
3766     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
3767     PrivateVarsPos[VD] = Counter;
3768     ++Counter;
3769   }
3770   for (const Expr *E : Data.FirstprivateVars) {
3771     Args.push_back(ImplicitParamDecl::Create(
3772         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3773         C.getPointerType(C.getPointerType(E->getType()))
3774             .withConst()
3775             .withRestrict(),
3776         ImplicitParamDecl::Other));
3777     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
3778     PrivateVarsPos[VD] = Counter;
3779     ++Counter;
3780   }
3781   for (const Expr *E : Data.LastprivateVars) {
3782     Args.push_back(ImplicitParamDecl::Create(
3783         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3784         C.getPointerType(C.getPointerType(E->getType()))
3785             .withConst()
3786             .withRestrict(),
3787         ImplicitParamDecl::Other));
3788     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
3789     PrivateVarsPos[VD] = Counter;
3790     ++Counter;
3791   }
3792   for (const VarDecl *VD : Data.PrivateLocals) {
3793     QualType Ty = VD->getType().getNonReferenceType();
3794     if (VD->getType()->isLValueReferenceType())
3795       Ty = C.getPointerType(Ty);
3796     if (isAllocatableDecl(VD))
3797       Ty = C.getPointerType(Ty);
3798     Args.push_back(ImplicitParamDecl::Create(
3799         C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
3800         C.getPointerType(C.getPointerType(Ty)).withConst().withRestrict(),
3801         ImplicitParamDecl::Other));
3802     PrivateVarsPos[VD] = Counter;
3803     ++Counter;
3804   }
3805   const auto &TaskPrivatesMapFnInfo =
3806       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
3807   llvm::FunctionType *TaskPrivatesMapTy =
3808       CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo);
3809   std::string Name =
3810       CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""});
3811   auto *TaskPrivatesMap = llvm::Function::Create(
3812       TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name,
3813       &CGM.getModule());
3814   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap,
3815                                     TaskPrivatesMapFnInfo);
3816   if (CGM.getLangOpts().Optimize) {
3817     TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline);
3818     TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone);
3819     TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline);
3820   }
3821   CodeGenFunction CGF(CGM);
3822   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap,
3823                     TaskPrivatesMapFnInfo, Args, Loc, Loc);
3824 
3825   // *privi = &.privates.privi;
3826   LValue Base = CGF.EmitLoadOfPointerLValue(
3827       CGF.GetAddrOfLocalVar(&TaskPrivatesArg),
3828       TaskPrivatesArg.getType()->castAs<PointerType>());
3829   const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl());
3830   Counter = 0;
3831   for (const FieldDecl *Field : PrivatesQTyRD->fields()) {
3832     LValue FieldLVal = CGF.EmitLValueForField(Base, Field);
3833     const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]];
3834     LValue RefLVal =
3835         CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType());
3836     LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue(
3837         RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>());
3838     CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal);
3839     ++Counter;
3840   }
3841   CGF.FinishFunction();
3842   return TaskPrivatesMap;
3843 }
3844 
3845 /// Emit initialization for private variables in task-based directives.
3846 static void emitPrivatesInit(CodeGenFunction &CGF,
3847                              const OMPExecutableDirective &D,
3848                              Address KmpTaskSharedsPtr, LValue TDBase,
3849                              const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3850                              QualType SharedsTy, QualType SharedsPtrTy,
3851                              const OMPTaskDataTy &Data,
3852                              ArrayRef<PrivateDataTy> Privates, bool ForDup) {
3853   ASTContext &C = CGF.getContext();
3854   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
3855   LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI);
3856   OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind())
3857                                  ? OMPD_taskloop
3858                                  : OMPD_task;
3859   const CapturedStmt &CS = *D.getCapturedStmt(Kind);
3860   CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS);
3861   LValue SrcBase;
3862   bool IsTargetTask =
3863       isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) ||
3864       isOpenMPTargetExecutionDirective(D.getDirectiveKind());
3865   // For target-based directives skip 4 firstprivate arrays BasePointersArray,
3866   // PointersArray, SizesArray, and MappersArray. The original variables for
3867   // these arrays are not captured and we get their addresses explicitly.
3868   if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) ||
3869       (IsTargetTask && KmpTaskSharedsPtr.isValid())) {
3870     SrcBase = CGF.MakeAddrLValue(
3871         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
3872             KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy),
3873             CGF.ConvertTypeForMem(SharedsTy)),
3874         SharedsTy);
3875   }
3876   FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin();
3877   for (const PrivateDataTy &Pair : Privates) {
3878     // Do not initialize private locals.
3879     if (Pair.second.isLocalPrivate()) {
3880       ++FI;
3881       continue;
3882     }
3883     const VarDecl *VD = Pair.second.PrivateCopy;
3884     const Expr *Init = VD->getAnyInitializer();
3885     if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) &&
3886                              !CGF.isTrivialInitializer(Init)))) {
3887       LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
3888       if (const VarDecl *Elem = Pair.second.PrivateElemInit) {
3889         const VarDecl *OriginalVD = Pair.second.Original;
3890         // Check if the variable is the target-based BasePointersArray,
3891         // PointersArray, SizesArray, or MappersArray.
3892         LValue SharedRefLValue;
3893         QualType Type = PrivateLValue.getType();
3894         const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD);
3895         if (IsTargetTask && !SharedField) {
3896           assert(isa<ImplicitParamDecl>(OriginalVD) &&
3897                  isa<CapturedDecl>(OriginalVD->getDeclContext()) &&
3898                  cast<CapturedDecl>(OriginalVD->getDeclContext())
3899                          ->getNumParams() == 0 &&
3900                  isa<TranslationUnitDecl>(
3901                      cast<CapturedDecl>(OriginalVD->getDeclContext())
3902                          ->getDeclContext()) &&
3903                  "Expected artificial target data variable.");
3904           SharedRefLValue =
3905               CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type);
3906         } else if (ForDup) {
3907           SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField);
3908           SharedRefLValue = CGF.MakeAddrLValue(
3909               SharedRefLValue.getAddress(CGF).withAlignment(
3910                   C.getDeclAlign(OriginalVD)),
3911               SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl),
3912               SharedRefLValue.getTBAAInfo());
3913         } else if (CGF.LambdaCaptureFields.count(
3914                        Pair.second.Original->getCanonicalDecl()) > 0 ||
3915                    isa_and_nonnull<BlockDecl>(CGF.CurCodeDecl)) {
3916           SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef);
3917         } else {
3918           // Processing for implicitly captured variables.
3919           InlinedOpenMPRegionRAII Region(
3920               CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown,
3921               /*HasCancel=*/false, /*NoInheritance=*/true);
3922           SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef);
3923         }
3924         if (Type->isArrayType()) {
3925           // Initialize firstprivate array.
3926           if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) {
3927             // Perform simple memcpy.
3928             CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type);
3929           } else {
3930             // Initialize firstprivate array using element-by-element
3931             // initialization.
3932             CGF.EmitOMPAggregateAssign(
3933                 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF),
3934                 Type,
3935                 [&CGF, Elem, Init, &CapturesInfo](Address DestElement,
3936                                                   Address SrcElement) {
3937                   // Clean up any temporaries needed by the initialization.
3938                   CodeGenFunction::OMPPrivateScope InitScope(CGF);
3939                   InitScope.addPrivate(Elem, SrcElement);
3940                   (void)InitScope.Privatize();
3941                   // Emit initialization for single element.
3942                   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(
3943                       CGF, &CapturesInfo);
3944                   CGF.EmitAnyExprToMem(Init, DestElement,
3945                                        Init->getType().getQualifiers(),
3946                                        /*IsInitializer=*/false);
3947                 });
3948           }
3949         } else {
3950           CodeGenFunction::OMPPrivateScope InitScope(CGF);
3951           InitScope.addPrivate(Elem, SharedRefLValue.getAddress(CGF));
3952           (void)InitScope.Privatize();
3953           CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo);
3954           CGF.EmitExprAsInit(Init, VD, PrivateLValue,
3955                              /*capturedByInit=*/false);
3956         }
3957       } else {
3958         CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false);
3959       }
3960     }
3961     ++FI;
3962   }
3963 }
3964 
3965 /// Check if duplication function is required for taskloops.
3966 static bool checkInitIsRequired(CodeGenFunction &CGF,
3967                                 ArrayRef<PrivateDataTy> Privates) {
3968   bool InitRequired = false;
3969   for (const PrivateDataTy &Pair : Privates) {
3970     if (Pair.second.isLocalPrivate())
3971       continue;
3972     const VarDecl *VD = Pair.second.PrivateCopy;
3973     const Expr *Init = VD->getAnyInitializer();
3974     InitRequired = InitRequired || (isa_and_nonnull<CXXConstructExpr>(Init) &&
3975                                     !CGF.isTrivialInitializer(Init));
3976     if (InitRequired)
3977       break;
3978   }
3979   return InitRequired;
3980 }
3981 
3982 
3983 /// Emit task_dup function (for initialization of
3984 /// private/firstprivate/lastprivate vars and last_iter flag)
3985 /// \code
3986 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int
3987 /// lastpriv) {
3988 /// // setup lastprivate flag
3989 ///    task_dst->last = lastpriv;
3990 /// // could be constructor calls here...
3991 /// }
3992 /// \endcode
3993 static llvm::Value *
3994 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc,
3995                     const OMPExecutableDirective &D,
3996                     QualType KmpTaskTWithPrivatesPtrQTy,
3997                     const RecordDecl *KmpTaskTWithPrivatesQTyRD,
3998                     const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy,
3999                     QualType SharedsPtrTy, const OMPTaskDataTy &Data,
4000                     ArrayRef<PrivateDataTy> Privates, bool WithLastIter) {
4001   ASTContext &C = CGM.getContext();
4002   FunctionArgList Args;
4003   ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4004                            KmpTaskTWithPrivatesPtrQTy,
4005                            ImplicitParamDecl::Other);
4006   ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
4007                            KmpTaskTWithPrivatesPtrQTy,
4008                            ImplicitParamDecl::Other);
4009   ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy,
4010                                 ImplicitParamDecl::Other);
4011   Args.push_back(&DstArg);
4012   Args.push_back(&SrcArg);
4013   Args.push_back(&LastprivArg);
4014   const auto &TaskDupFnInfo =
4015       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
4016   llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo);
4017   std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""});
4018   auto *TaskDup = llvm::Function::Create(
4019       TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule());
4020   CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo);
4021   TaskDup->setDoesNotRecurse();
4022   CodeGenFunction CGF(CGM);
4023   CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc,
4024                     Loc);
4025 
4026   LValue TDBase = CGF.EmitLoadOfPointerLValue(
4027       CGF.GetAddrOfLocalVar(&DstArg),
4028       KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4029   // task_dst->liter = lastpriv;
4030   if (WithLastIter) {
4031     auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter);
4032     LValue Base = CGF.EmitLValueForField(
4033         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4034     LValue LILVal = CGF.EmitLValueForField(Base, *LIFI);
4035     llvm::Value *Lastpriv = CGF.EmitLoadOfScalar(
4036         CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc);
4037     CGF.EmitStoreOfScalar(Lastpriv, LILVal);
4038   }
4039 
4040   // Emit initial values for private copies (if any).
4041   assert(!Privates.empty());
4042   Address KmpTaskSharedsPtr = Address::invalid();
4043   if (!Data.FirstprivateVars.empty()) {
4044     LValue TDBase = CGF.EmitLoadOfPointerLValue(
4045         CGF.GetAddrOfLocalVar(&SrcArg),
4046         KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>());
4047     LValue Base = CGF.EmitLValueForField(
4048         TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin());
4049     KmpTaskSharedsPtr = Address::deprecated(
4050         CGF.EmitLoadOfScalar(CGF.EmitLValueForField(
4051                                  Base, *std::next(KmpTaskTQTyRD->field_begin(),
4052                                                   KmpTaskTShareds)),
4053                              Loc),
4054         CGM.getNaturalTypeAlignment(SharedsTy));
4055   }
4056   emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD,
4057                    SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true);
4058   CGF.FinishFunction();
4059   return TaskDup;
4060 }
4061 
4062 /// Checks if destructor function is required to be generated.
4063 /// \return true if cleanups are required, false otherwise.
4064 static bool
4065 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD,
4066                          ArrayRef<PrivateDataTy> Privates) {
4067   for (const PrivateDataTy &P : Privates) {
4068     if (P.second.isLocalPrivate())
4069       continue;
4070     QualType Ty = P.second.Original->getType().getNonReferenceType();
4071     if (Ty.isDestructedType())
4072       return true;
4073   }
4074   return false;
4075 }
4076 
4077 namespace {
4078 /// Loop generator for OpenMP iterator expression.
4079 class OMPIteratorGeneratorScope final
4080     : public CodeGenFunction::OMPPrivateScope {
4081   CodeGenFunction &CGF;
4082   const OMPIteratorExpr *E = nullptr;
4083   SmallVector<CodeGenFunction::JumpDest, 4> ContDests;
4084   SmallVector<CodeGenFunction::JumpDest, 4> ExitDests;
4085   OMPIteratorGeneratorScope() = delete;
4086   OMPIteratorGeneratorScope(OMPIteratorGeneratorScope &) = delete;
4087 
4088 public:
4089   OMPIteratorGeneratorScope(CodeGenFunction &CGF, const OMPIteratorExpr *E)
4090       : CodeGenFunction::OMPPrivateScope(CGF), CGF(CGF), E(E) {
4091     if (!E)
4092       return;
4093     SmallVector<llvm::Value *, 4> Uppers;
4094     for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) {
4095       Uppers.push_back(CGF.EmitScalarExpr(E->getHelper(I).Upper));
4096       const auto *VD = cast<VarDecl>(E->getIteratorDecl(I));
4097       addPrivate(VD, CGF.CreateMemTemp(VD->getType(), VD->getName()));
4098       const OMPIteratorHelperData &HelperData = E->getHelper(I);
4099       addPrivate(
4100           HelperData.CounterVD,
4101           CGF.CreateMemTemp(HelperData.CounterVD->getType(), "counter.addr"));
4102     }
4103     Privatize();
4104 
4105     for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) {
4106       const OMPIteratorHelperData &HelperData = E->getHelper(I);
4107       LValue CLVal =
4108           CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(HelperData.CounterVD),
4109                              HelperData.CounterVD->getType());
4110       // Counter = 0;
4111       CGF.EmitStoreOfScalar(
4112           llvm::ConstantInt::get(CLVal.getAddress(CGF).getElementType(), 0),
4113           CLVal);
4114       CodeGenFunction::JumpDest &ContDest =
4115           ContDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.cont"));
4116       CodeGenFunction::JumpDest &ExitDest =
4117           ExitDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.exit"));
4118       // N = <number-of_iterations>;
4119       llvm::Value *N = Uppers[I];
4120       // cont:
4121       // if (Counter < N) goto body; else goto exit;
4122       CGF.EmitBlock(ContDest.getBlock());
4123       auto *CVal =
4124           CGF.EmitLoadOfScalar(CLVal, HelperData.CounterVD->getLocation());
4125       llvm::Value *Cmp =
4126           HelperData.CounterVD->getType()->isSignedIntegerOrEnumerationType()
4127               ? CGF.Builder.CreateICmpSLT(CVal, N)
4128               : CGF.Builder.CreateICmpULT(CVal, N);
4129       llvm::BasicBlock *BodyBB = CGF.createBasicBlock("iter.body");
4130       CGF.Builder.CreateCondBr(Cmp, BodyBB, ExitDest.getBlock());
4131       // body:
4132       CGF.EmitBlock(BodyBB);
4133       // Iteri = Begini + Counter * Stepi;
4134       CGF.EmitIgnoredExpr(HelperData.Update);
4135     }
4136   }
4137   ~OMPIteratorGeneratorScope() {
4138     if (!E)
4139       return;
4140     for (unsigned I = E->numOfIterators(); I > 0; --I) {
4141       // Counter = Counter + 1;
4142       const OMPIteratorHelperData &HelperData = E->getHelper(I - 1);
4143       CGF.EmitIgnoredExpr(HelperData.CounterUpdate);
4144       // goto cont;
4145       CGF.EmitBranchThroughCleanup(ContDests[I - 1]);
4146       // exit:
4147       CGF.EmitBlock(ExitDests[I - 1].getBlock(), /*IsFinished=*/I == 1);
4148     }
4149   }
4150 };
4151 } // namespace
4152 
4153 static std::pair<llvm::Value *, llvm::Value *>
4154 getPointerAndSize(CodeGenFunction &CGF, const Expr *E) {
4155   const auto *OASE = dyn_cast<OMPArrayShapingExpr>(E);
4156   llvm::Value *Addr;
4157   if (OASE) {
4158     const Expr *Base = OASE->getBase();
4159     Addr = CGF.EmitScalarExpr(Base);
4160   } else {
4161     Addr = CGF.EmitLValue(E).getPointer(CGF);
4162   }
4163   llvm::Value *SizeVal;
4164   QualType Ty = E->getType();
4165   if (OASE) {
4166     SizeVal = CGF.getTypeSize(OASE->getBase()->getType()->getPointeeType());
4167     for (const Expr *SE : OASE->getDimensions()) {
4168       llvm::Value *Sz = CGF.EmitScalarExpr(SE);
4169       Sz = CGF.EmitScalarConversion(
4170           Sz, SE->getType(), CGF.getContext().getSizeType(), SE->getExprLoc());
4171       SizeVal = CGF.Builder.CreateNUWMul(SizeVal, Sz);
4172     }
4173   } else if (const auto *ASE =
4174                  dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) {
4175     LValue UpAddrLVal =
4176         CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false);
4177     Address UpAddrAddress = UpAddrLVal.getAddress(CGF);
4178     llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32(
4179         UpAddrAddress.getElementType(), UpAddrAddress.getPointer(), /*Idx0=*/1);
4180     llvm::Value *LowIntPtr = CGF.Builder.CreatePtrToInt(Addr, CGF.SizeTy);
4181     llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGF.SizeTy);
4182     SizeVal = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr);
4183   } else {
4184     SizeVal = CGF.getTypeSize(Ty);
4185   }
4186   return std::make_pair(Addr, SizeVal);
4187 }
4188 
4189 /// Builds kmp_depend_info, if it is not built yet, and builds flags type.
4190 static void getKmpAffinityType(ASTContext &C, QualType &KmpTaskAffinityInfoTy) {
4191   QualType FlagsTy = C.getIntTypeForBitwidth(32, /*Signed=*/false);
4192   if (KmpTaskAffinityInfoTy.isNull()) {
4193     RecordDecl *KmpAffinityInfoRD =
4194         C.buildImplicitRecord("kmp_task_affinity_info_t");
4195     KmpAffinityInfoRD->startDefinition();
4196     addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getIntPtrType());
4197     addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getSizeType());
4198     addFieldToRecordDecl(C, KmpAffinityInfoRD, FlagsTy);
4199     KmpAffinityInfoRD->completeDefinition();
4200     KmpTaskAffinityInfoTy = C.getRecordType(KmpAffinityInfoRD);
4201   }
4202 }
4203 
4204 CGOpenMPRuntime::TaskResultTy
4205 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc,
4206                               const OMPExecutableDirective &D,
4207                               llvm::Function *TaskFunction, QualType SharedsTy,
4208                               Address Shareds, const OMPTaskDataTy &Data) {
4209   ASTContext &C = CGM.getContext();
4210   llvm::SmallVector<PrivateDataTy, 4> Privates;
4211   // Aggregate privates and sort them by the alignment.
4212   const auto *I = Data.PrivateCopies.begin();
4213   for (const Expr *E : Data.PrivateVars) {
4214     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4215     Privates.emplace_back(
4216         C.getDeclAlign(VD),
4217         PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4218                          /*PrivateElemInit=*/nullptr));
4219     ++I;
4220   }
4221   I = Data.FirstprivateCopies.begin();
4222   const auto *IElemInitRef = Data.FirstprivateInits.begin();
4223   for (const Expr *E : Data.FirstprivateVars) {
4224     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4225     Privates.emplace_back(
4226         C.getDeclAlign(VD),
4227         PrivateHelpersTy(
4228             E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4229             cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl())));
4230     ++I;
4231     ++IElemInitRef;
4232   }
4233   I = Data.LastprivateCopies.begin();
4234   for (const Expr *E : Data.LastprivateVars) {
4235     const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
4236     Privates.emplace_back(
4237         C.getDeclAlign(VD),
4238         PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
4239                          /*PrivateElemInit=*/nullptr));
4240     ++I;
4241   }
4242   for (const VarDecl *VD : Data.PrivateLocals) {
4243     if (isAllocatableDecl(VD))
4244       Privates.emplace_back(CGM.getPointerAlign(), PrivateHelpersTy(VD));
4245     else
4246       Privates.emplace_back(C.getDeclAlign(VD), PrivateHelpersTy(VD));
4247   }
4248   llvm::stable_sort(Privates,
4249                     [](const PrivateDataTy &L, const PrivateDataTy &R) {
4250                       return L.first > R.first;
4251                     });
4252   QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
4253   // Build type kmp_routine_entry_t (if not built yet).
4254   emitKmpRoutineEntryT(KmpInt32Ty);
4255   // Build type kmp_task_t (if not built yet).
4256   if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) {
4257     if (SavedKmpTaskloopTQTy.isNull()) {
4258       SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl(
4259           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
4260     }
4261     KmpTaskTQTy = SavedKmpTaskloopTQTy;
4262   } else {
4263     assert((D.getDirectiveKind() == OMPD_task ||
4264             isOpenMPTargetExecutionDirective(D.getDirectiveKind()) ||
4265             isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) &&
4266            "Expected taskloop, task or target directive");
4267     if (SavedKmpTaskTQTy.isNull()) {
4268       SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl(
4269           CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy));
4270     }
4271     KmpTaskTQTy = SavedKmpTaskTQTy;
4272   }
4273   const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
4274   // Build particular struct kmp_task_t for the given task.
4275   const RecordDecl *KmpTaskTWithPrivatesQTyRD =
4276       createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates);
4277   QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD);
4278   QualType KmpTaskTWithPrivatesPtrQTy =
4279       C.getPointerType(KmpTaskTWithPrivatesQTy);
4280   llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy);
4281   llvm::Type *KmpTaskTWithPrivatesPtrTy =
4282       KmpTaskTWithPrivatesTy->getPointerTo();
4283   llvm::Value *KmpTaskTWithPrivatesTySize =
4284       CGF.getTypeSize(KmpTaskTWithPrivatesQTy);
4285   QualType SharedsPtrTy = C.getPointerType(SharedsTy);
4286 
4287   // Emit initial values for private copies (if any).
4288   llvm::Value *TaskPrivatesMap = nullptr;
4289   llvm::Type *TaskPrivatesMapTy =
4290       std::next(TaskFunction->arg_begin(), 3)->getType();
4291   if (!Privates.empty()) {
4292     auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
4293     TaskPrivatesMap =
4294         emitTaskPrivateMappingFunction(CGM, Loc, Data, FI->getType(), Privates);
4295     TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4296         TaskPrivatesMap, TaskPrivatesMapTy);
4297   } else {
4298     TaskPrivatesMap = llvm::ConstantPointerNull::get(
4299         cast<llvm::PointerType>(TaskPrivatesMapTy));
4300   }
4301   // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
4302   // kmp_task_t *tt);
4303   llvm::Function *TaskEntry = emitProxyTaskFunction(
4304       CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
4305       KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction,
4306       TaskPrivatesMap);
4307 
4308   // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
4309   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
4310   // kmp_routine_entry_t *task_entry);
4311   // Task flags. Format is taken from
4312   // https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h,
4313   // description of kmp_tasking_flags struct.
4314   enum {
4315     TiedFlag = 0x1,
4316     FinalFlag = 0x2,
4317     DestructorsFlag = 0x8,
4318     PriorityFlag = 0x20,
4319     DetachableFlag = 0x40,
4320   };
4321   unsigned Flags = Data.Tied ? TiedFlag : 0;
4322   bool NeedsCleanup = false;
4323   if (!Privates.empty()) {
4324     NeedsCleanup =
4325         checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD, Privates);
4326     if (NeedsCleanup)
4327       Flags = Flags | DestructorsFlag;
4328   }
4329   if (Data.Priority.getInt())
4330     Flags = Flags | PriorityFlag;
4331   if (D.hasClausesOfKind<OMPDetachClause>())
4332     Flags = Flags | DetachableFlag;
4333   llvm::Value *TaskFlags =
4334       Data.Final.getPointer()
4335           ? CGF.Builder.CreateSelect(Data.Final.getPointer(),
4336                                      CGF.Builder.getInt32(FinalFlag),
4337                                      CGF.Builder.getInt32(/*C=*/0))
4338           : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0);
4339   TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
4340   llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy));
4341   SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc),
4342       getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize,
4343       SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4344           TaskEntry, KmpRoutineEntryPtrTy)};
4345   llvm::Value *NewTask;
4346   if (D.hasClausesOfKind<OMPNowaitClause>()) {
4347     // Check if we have any device clause associated with the directive.
4348     const Expr *Device = nullptr;
4349     if (auto *C = D.getSingleClause<OMPDeviceClause>())
4350       Device = C->getDevice();
4351     // Emit device ID if any otherwise use default value.
4352     llvm::Value *DeviceID;
4353     if (Device)
4354       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
4355                                            CGF.Int64Ty, /*isSigned=*/true);
4356     else
4357       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
4358     AllocArgs.push_back(DeviceID);
4359     NewTask = CGF.EmitRuntimeCall(
4360         OMPBuilder.getOrCreateRuntimeFunction(
4361             CGM.getModule(), OMPRTL___kmpc_omp_target_task_alloc),
4362         AllocArgs);
4363   } else {
4364     NewTask =
4365         CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
4366                                 CGM.getModule(), OMPRTL___kmpc_omp_task_alloc),
4367                             AllocArgs);
4368   }
4369   // Emit detach clause initialization.
4370   // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid,
4371   // task_descriptor);
4372   if (const auto *DC = D.getSingleClause<OMPDetachClause>()) {
4373     const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts();
4374     LValue EvtLVal = CGF.EmitLValue(Evt);
4375 
4376     // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref,
4377     // int gtid, kmp_task_t *task);
4378     llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc());
4379     llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc());
4380     Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false);
4381     llvm::Value *EvtVal = CGF.EmitRuntimeCall(
4382         OMPBuilder.getOrCreateRuntimeFunction(
4383             CGM.getModule(), OMPRTL___kmpc_task_allow_completion_event),
4384         {Loc, Tid, NewTask});
4385     EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(),
4386                                       Evt->getExprLoc());
4387     CGF.EmitStoreOfScalar(EvtVal, EvtLVal);
4388   }
4389   // Process affinity clauses.
4390   if (D.hasClausesOfKind<OMPAffinityClause>()) {
4391     // Process list of affinity data.
4392     ASTContext &C = CGM.getContext();
4393     Address AffinitiesArray = Address::invalid();
4394     // Calculate number of elements to form the array of affinity data.
4395     llvm::Value *NumOfElements = nullptr;
4396     unsigned NumAffinities = 0;
4397     for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4398       if (const Expr *Modifier = C->getModifier()) {
4399         const auto *IE = cast<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts());
4400         for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4401           llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper);
4402           Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false);
4403           NumOfElements =
4404               NumOfElements ? CGF.Builder.CreateNUWMul(NumOfElements, Sz) : Sz;
4405         }
4406       } else {
4407         NumAffinities += C->varlist_size();
4408       }
4409     }
4410     getKmpAffinityType(CGM.getContext(), KmpTaskAffinityInfoTy);
4411     // Fields ids in kmp_task_affinity_info record.
4412     enum RTLAffinityInfoFieldsTy { BaseAddr, Len, Flags };
4413 
4414     QualType KmpTaskAffinityInfoArrayTy;
4415     if (NumOfElements) {
4416       NumOfElements = CGF.Builder.CreateNUWAdd(
4417           llvm::ConstantInt::get(CGF.SizeTy, NumAffinities), NumOfElements);
4418       auto *OVE = new (C) OpaqueValueExpr(
4419           Loc,
4420           C.getIntTypeForBitwidth(C.getTypeSize(C.getSizeType()), /*Signed=*/0),
4421           VK_PRValue);
4422       CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE,
4423                                                     RValue::get(NumOfElements));
4424       KmpTaskAffinityInfoArrayTy =
4425           C.getVariableArrayType(KmpTaskAffinityInfoTy, OVE, ArrayType::Normal,
4426                                  /*IndexTypeQuals=*/0, SourceRange(Loc, Loc));
4427       // Properly emit variable-sized array.
4428       auto *PD = ImplicitParamDecl::Create(C, KmpTaskAffinityInfoArrayTy,
4429                                            ImplicitParamDecl::Other);
4430       CGF.EmitVarDecl(*PD);
4431       AffinitiesArray = CGF.GetAddrOfLocalVar(PD);
4432       NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty,
4433                                                 /*isSigned=*/false);
4434     } else {
4435       KmpTaskAffinityInfoArrayTy = C.getConstantArrayType(
4436           KmpTaskAffinityInfoTy,
4437           llvm::APInt(C.getTypeSize(C.getSizeType()), NumAffinities), nullptr,
4438           ArrayType::Normal, /*IndexTypeQuals=*/0);
4439       AffinitiesArray =
4440           CGF.CreateMemTemp(KmpTaskAffinityInfoArrayTy, ".affs.arr.addr");
4441       AffinitiesArray = CGF.Builder.CreateConstArrayGEP(AffinitiesArray, 0);
4442       NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumAffinities,
4443                                              /*isSigned=*/false);
4444     }
4445 
4446     const auto *KmpAffinityInfoRD = KmpTaskAffinityInfoTy->getAsRecordDecl();
4447     // Fill array by elements without iterators.
4448     unsigned Pos = 0;
4449     bool HasIterator = false;
4450     for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4451       if (C->getModifier()) {
4452         HasIterator = true;
4453         continue;
4454       }
4455       for (const Expr *E : C->varlists()) {
4456         llvm::Value *Addr;
4457         llvm::Value *Size;
4458         std::tie(Addr, Size) = getPointerAndSize(CGF, E);
4459         LValue Base =
4460             CGF.MakeAddrLValue(CGF.Builder.CreateConstGEP(AffinitiesArray, Pos),
4461                                KmpTaskAffinityInfoTy);
4462         // affs[i].base_addr = &<Affinities[i].second>;
4463         LValue BaseAddrLVal = CGF.EmitLValueForField(
4464             Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr));
4465         CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy),
4466                               BaseAddrLVal);
4467         // affs[i].len = sizeof(<Affinities[i].second>);
4468         LValue LenLVal = CGF.EmitLValueForField(
4469             Base, *std::next(KmpAffinityInfoRD->field_begin(), Len));
4470         CGF.EmitStoreOfScalar(Size, LenLVal);
4471         ++Pos;
4472       }
4473     }
4474     LValue PosLVal;
4475     if (HasIterator) {
4476       PosLVal = CGF.MakeAddrLValue(
4477           CGF.CreateMemTemp(C.getSizeType(), "affs.counter.addr"),
4478           C.getSizeType());
4479       CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal);
4480     }
4481     // Process elements with iterators.
4482     for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) {
4483       const Expr *Modifier = C->getModifier();
4484       if (!Modifier)
4485         continue;
4486       OMPIteratorGeneratorScope IteratorScope(
4487           CGF, cast_or_null<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts()));
4488       for (const Expr *E : C->varlists()) {
4489         llvm::Value *Addr;
4490         llvm::Value *Size;
4491         std::tie(Addr, Size) = getPointerAndSize(CGF, E);
4492         llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4493         LValue Base = CGF.MakeAddrLValue(
4494             CGF.Builder.CreateGEP(AffinitiesArray, Idx), KmpTaskAffinityInfoTy);
4495         // affs[i].base_addr = &<Affinities[i].second>;
4496         LValue BaseAddrLVal = CGF.EmitLValueForField(
4497             Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr));
4498         CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy),
4499                               BaseAddrLVal);
4500         // affs[i].len = sizeof(<Affinities[i].second>);
4501         LValue LenLVal = CGF.EmitLValueForField(
4502             Base, *std::next(KmpAffinityInfoRD->field_begin(), Len));
4503         CGF.EmitStoreOfScalar(Size, LenLVal);
4504         Idx = CGF.Builder.CreateNUWAdd(
4505             Idx, llvm::ConstantInt::get(Idx->getType(), 1));
4506         CGF.EmitStoreOfScalar(Idx, PosLVal);
4507       }
4508     }
4509     // Call to kmp_int32 __kmpc_omp_reg_task_with_affinity(ident_t *loc_ref,
4510     // kmp_int32 gtid, kmp_task_t *new_task, kmp_int32
4511     // naffins, kmp_task_affinity_info_t *affin_list);
4512     llvm::Value *LocRef = emitUpdateLocation(CGF, Loc);
4513     llvm::Value *GTid = getThreadID(CGF, Loc);
4514     llvm::Value *AffinListPtr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4515         AffinitiesArray.getPointer(), CGM.VoidPtrTy);
4516     // FIXME: Emit the function and ignore its result for now unless the
4517     // runtime function is properly implemented.
4518     (void)CGF.EmitRuntimeCall(
4519         OMPBuilder.getOrCreateRuntimeFunction(
4520             CGM.getModule(), OMPRTL___kmpc_omp_reg_task_with_affinity),
4521         {LocRef, GTid, NewTask, NumOfElements, AffinListPtr});
4522   }
4523   llvm::Value *NewTaskNewTaskTTy =
4524       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4525           NewTask, KmpTaskTWithPrivatesPtrTy);
4526   LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy,
4527                                                KmpTaskTWithPrivatesQTy);
4528   LValue TDBase =
4529       CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
4530   // Fill the data in the resulting kmp_task_t record.
4531   // Copy shareds if there are any.
4532   Address KmpTaskSharedsPtr = Address::invalid();
4533   if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) {
4534     KmpTaskSharedsPtr = Address::deprecated(
4535         CGF.EmitLoadOfScalar(
4536             CGF.EmitLValueForField(
4537                 TDBase,
4538                 *std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds)),
4539             Loc),
4540         CGM.getNaturalTypeAlignment(SharedsTy));
4541     LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy);
4542     LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy);
4543     CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap);
4544   }
4545   // Emit initial values for private copies (if any).
4546   TaskResultTy Result;
4547   if (!Privates.empty()) {
4548     emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD,
4549                      SharedsTy, SharedsPtrTy, Data, Privates,
4550                      /*ForDup=*/false);
4551     if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) &&
4552         (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) {
4553       Result.TaskDupFn = emitTaskDupFunction(
4554           CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD,
4555           KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates,
4556           /*WithLastIter=*/!Data.LastprivateVars.empty());
4557     }
4558   }
4559   // Fields of union "kmp_cmplrdata_t" for destructors and priority.
4560   enum { Priority = 0, Destructors = 1 };
4561   // Provide pointer to function with destructors for privates.
4562   auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1);
4563   const RecordDecl *KmpCmplrdataUD =
4564       (*FI)->getType()->getAsUnionType()->getDecl();
4565   if (NeedsCleanup) {
4566     llvm::Value *DestructorFn = emitDestructorsFunction(
4567         CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy,
4568         KmpTaskTWithPrivatesQTy);
4569     LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI);
4570     LValue DestructorsLV = CGF.EmitLValueForField(
4571         Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors));
4572     CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4573                               DestructorFn, KmpRoutineEntryPtrTy),
4574                           DestructorsLV);
4575   }
4576   // Set priority.
4577   if (Data.Priority.getInt()) {
4578     LValue Data2LV = CGF.EmitLValueForField(
4579         TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2));
4580     LValue PriorityLV = CGF.EmitLValueForField(
4581         Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority));
4582     CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV);
4583   }
4584   Result.NewTask = NewTask;
4585   Result.TaskEntry = TaskEntry;
4586   Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy;
4587   Result.TDBase = TDBase;
4588   Result.KmpTaskTQTyRD = KmpTaskTQTyRD;
4589   return Result;
4590 }
4591 
4592 namespace {
4593 /// Dependence kind for RTL.
4594 enum RTLDependenceKindTy {
4595   DepIn = 0x01,
4596   DepInOut = 0x3,
4597   DepMutexInOutSet = 0x4,
4598   DepInOutSet = 0x8
4599 };
4600 /// Fields ids in kmp_depend_info record.
4601 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags };
4602 } // namespace
4603 
4604 /// Translates internal dependency kind into the runtime kind.
4605 static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) {
4606   RTLDependenceKindTy DepKind;
4607   switch (K) {
4608   case OMPC_DEPEND_in:
4609     DepKind = DepIn;
4610     break;
4611   // Out and InOut dependencies must use the same code.
4612   case OMPC_DEPEND_out:
4613   case OMPC_DEPEND_inout:
4614     DepKind = DepInOut;
4615     break;
4616   case OMPC_DEPEND_mutexinoutset:
4617     DepKind = DepMutexInOutSet;
4618     break;
4619   case OMPC_DEPEND_inoutset:
4620     DepKind = DepInOutSet;
4621     break;
4622   case OMPC_DEPEND_source:
4623   case OMPC_DEPEND_sink:
4624   case OMPC_DEPEND_depobj:
4625   case OMPC_DEPEND_unknown:
4626     llvm_unreachable("Unknown task dependence type");
4627   }
4628   return DepKind;
4629 }
4630 
4631 /// Builds kmp_depend_info, if it is not built yet, and builds flags type.
4632 static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy,
4633                            QualType &FlagsTy) {
4634   FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false);
4635   if (KmpDependInfoTy.isNull()) {
4636     RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info");
4637     KmpDependInfoRD->startDefinition();
4638     addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType());
4639     addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType());
4640     addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy);
4641     KmpDependInfoRD->completeDefinition();
4642     KmpDependInfoTy = C.getRecordType(KmpDependInfoRD);
4643   }
4644 }
4645 
4646 std::pair<llvm::Value *, LValue>
4647 CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal,
4648                                    SourceLocation Loc) {
4649   ASTContext &C = CGM.getContext();
4650   QualType FlagsTy;
4651   getDependTypes(C, KmpDependInfoTy, FlagsTy);
4652   RecordDecl *KmpDependInfoRD =
4653       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
4654   LValue Base = CGF.EmitLoadOfPointerLValue(
4655       DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>());
4656   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
4657   Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4658       Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy),
4659       CGF.ConvertTypeForMem(KmpDependInfoTy));
4660   Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(),
4661                             Base.getTBAAInfo());
4662   Address DepObjAddr = CGF.Builder.CreateGEP(
4663       Addr, llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
4664   LValue NumDepsBase = CGF.MakeAddrLValue(
4665       DepObjAddr, KmpDependInfoTy, Base.getBaseInfo(), Base.getTBAAInfo());
4666   // NumDeps = deps[i].base_addr;
4667   LValue BaseAddrLVal = CGF.EmitLValueForField(
4668       NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
4669   llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc);
4670   return std::make_pair(NumDeps, Base);
4671 }
4672 
4673 static void emitDependData(CodeGenFunction &CGF, QualType &KmpDependInfoTy,
4674                            llvm::PointerUnion<unsigned *, LValue *> Pos,
4675                            const OMPTaskDataTy::DependData &Data,
4676                            Address DependenciesArray) {
4677   CodeGenModule &CGM = CGF.CGM;
4678   ASTContext &C = CGM.getContext();
4679   QualType FlagsTy;
4680   getDependTypes(C, KmpDependInfoTy, FlagsTy);
4681   RecordDecl *KmpDependInfoRD =
4682       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
4683   llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
4684 
4685   OMPIteratorGeneratorScope IteratorScope(
4686       CGF, cast_or_null<OMPIteratorExpr>(
4687                Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4688                                  : nullptr));
4689   for (const Expr *E : Data.DepExprs) {
4690     llvm::Value *Addr;
4691     llvm::Value *Size;
4692     std::tie(Addr, Size) = getPointerAndSize(CGF, E);
4693     LValue Base;
4694     if (unsigned *P = Pos.dyn_cast<unsigned *>()) {
4695       Base = CGF.MakeAddrLValue(
4696           CGF.Builder.CreateConstGEP(DependenciesArray, *P), KmpDependInfoTy);
4697     } else {
4698       LValue &PosLVal = *Pos.get<LValue *>();
4699       llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4700       Base = CGF.MakeAddrLValue(
4701           CGF.Builder.CreateGEP(DependenciesArray, Idx), KmpDependInfoTy);
4702     }
4703     // deps[i].base_addr = &<Dependencies[i].second>;
4704     LValue BaseAddrLVal = CGF.EmitLValueForField(
4705         Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
4706     CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy),
4707                           BaseAddrLVal);
4708     // deps[i].len = sizeof(<Dependencies[i].second>);
4709     LValue LenLVal = CGF.EmitLValueForField(
4710         Base, *std::next(KmpDependInfoRD->field_begin(), Len));
4711     CGF.EmitStoreOfScalar(Size, LenLVal);
4712     // deps[i].flags = <Dependencies[i].first>;
4713     RTLDependenceKindTy DepKind = translateDependencyKind(Data.DepKind);
4714     LValue FlagsLVal = CGF.EmitLValueForField(
4715         Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
4716     CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
4717                           FlagsLVal);
4718     if (unsigned *P = Pos.dyn_cast<unsigned *>()) {
4719       ++(*P);
4720     } else {
4721       LValue &PosLVal = *Pos.get<LValue *>();
4722       llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4723       Idx = CGF.Builder.CreateNUWAdd(Idx,
4724                                      llvm::ConstantInt::get(Idx->getType(), 1));
4725       CGF.EmitStoreOfScalar(Idx, PosLVal);
4726     }
4727   }
4728 }
4729 
4730 static SmallVector<llvm::Value *, 4>
4731 emitDepobjElementsSizes(CodeGenFunction &CGF, QualType &KmpDependInfoTy,
4732                         const OMPTaskDataTy::DependData &Data) {
4733   assert(Data.DepKind == OMPC_DEPEND_depobj &&
4734          "Expected depobj dependecy kind.");
4735   SmallVector<llvm::Value *, 4> Sizes;
4736   SmallVector<LValue, 4> SizeLVals;
4737   ASTContext &C = CGF.getContext();
4738   QualType FlagsTy;
4739   getDependTypes(C, KmpDependInfoTy, FlagsTy);
4740   RecordDecl *KmpDependInfoRD =
4741       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
4742   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
4743   llvm::Type *KmpDependInfoPtrT = CGF.ConvertTypeForMem(KmpDependInfoPtrTy);
4744   {
4745     OMPIteratorGeneratorScope IteratorScope(
4746         CGF, cast_or_null<OMPIteratorExpr>(
4747                  Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4748                                    : nullptr));
4749     for (const Expr *E : Data.DepExprs) {
4750       LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts());
4751       LValue Base = CGF.EmitLoadOfPointerLValue(
4752           DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>());
4753       Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4754           Base.getAddress(CGF), KmpDependInfoPtrT,
4755           CGF.ConvertTypeForMem(KmpDependInfoTy));
4756       Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(),
4757                                 Base.getTBAAInfo());
4758       Address DepObjAddr = CGF.Builder.CreateGEP(
4759           Addr, llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
4760       LValue NumDepsBase = CGF.MakeAddrLValue(
4761           DepObjAddr, KmpDependInfoTy, Base.getBaseInfo(), Base.getTBAAInfo());
4762       // NumDeps = deps[i].base_addr;
4763       LValue BaseAddrLVal = CGF.EmitLValueForField(
4764           NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
4765       llvm::Value *NumDeps =
4766           CGF.EmitLoadOfScalar(BaseAddrLVal, E->getExprLoc());
4767       LValue NumLVal = CGF.MakeAddrLValue(
4768           CGF.CreateMemTemp(C.getUIntPtrType(), "depobj.size.addr"),
4769           C.getUIntPtrType());
4770       CGF.Builder.CreateStore(llvm::ConstantInt::get(CGF.IntPtrTy, 0),
4771                               NumLVal.getAddress(CGF));
4772       llvm::Value *PrevVal = CGF.EmitLoadOfScalar(NumLVal, E->getExprLoc());
4773       llvm::Value *Add = CGF.Builder.CreateNUWAdd(PrevVal, NumDeps);
4774       CGF.EmitStoreOfScalar(Add, NumLVal);
4775       SizeLVals.push_back(NumLVal);
4776     }
4777   }
4778   for (unsigned I = 0, E = SizeLVals.size(); I < E; ++I) {
4779     llvm::Value *Size =
4780         CGF.EmitLoadOfScalar(SizeLVals[I], Data.DepExprs[I]->getExprLoc());
4781     Sizes.push_back(Size);
4782   }
4783   return Sizes;
4784 }
4785 
4786 static void emitDepobjElements(CodeGenFunction &CGF, QualType &KmpDependInfoTy,
4787                                LValue PosLVal,
4788                                const OMPTaskDataTy::DependData &Data,
4789                                Address DependenciesArray) {
4790   assert(Data.DepKind == OMPC_DEPEND_depobj &&
4791          "Expected depobj dependecy kind.");
4792   ASTContext &C = CGF.getContext();
4793   QualType FlagsTy;
4794   getDependTypes(C, KmpDependInfoTy, FlagsTy);
4795   RecordDecl *KmpDependInfoRD =
4796       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
4797   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
4798   llvm::Type *KmpDependInfoPtrT = CGF.ConvertTypeForMem(KmpDependInfoPtrTy);
4799   llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy);
4800   {
4801     OMPIteratorGeneratorScope IteratorScope(
4802         CGF, cast_or_null<OMPIteratorExpr>(
4803                  Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts()
4804                                    : nullptr));
4805     for (unsigned I = 0, End = Data.DepExprs.size(); I < End; ++I) {
4806       const Expr *E = Data.DepExprs[I];
4807       LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts());
4808       LValue Base = CGF.EmitLoadOfPointerLValue(
4809           DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>());
4810       Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4811           Base.getAddress(CGF), KmpDependInfoPtrT,
4812           CGF.ConvertTypeForMem(KmpDependInfoTy));
4813       Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(),
4814                                 Base.getTBAAInfo());
4815 
4816       // Get number of elements in a single depobj.
4817       Address DepObjAddr = CGF.Builder.CreateGEP(
4818           Addr, llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
4819       LValue NumDepsBase = CGF.MakeAddrLValue(
4820           DepObjAddr, KmpDependInfoTy, Base.getBaseInfo(), Base.getTBAAInfo());
4821       // NumDeps = deps[i].base_addr;
4822       LValue BaseAddrLVal = CGF.EmitLValueForField(
4823           NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
4824       llvm::Value *NumDeps =
4825           CGF.EmitLoadOfScalar(BaseAddrLVal, E->getExprLoc());
4826 
4827       // memcopy dependency data.
4828       llvm::Value *Size = CGF.Builder.CreateNUWMul(
4829           ElSize,
4830           CGF.Builder.CreateIntCast(NumDeps, CGF.SizeTy, /*isSigned=*/false));
4831       llvm::Value *Pos = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc());
4832       Address DepAddr = CGF.Builder.CreateGEP(DependenciesArray, Pos);
4833       CGF.Builder.CreateMemCpy(DepAddr, Base.getAddress(CGF), Size);
4834 
4835       // Increase pos.
4836       // pos += size;
4837       llvm::Value *Add = CGF.Builder.CreateNUWAdd(Pos, NumDeps);
4838       CGF.EmitStoreOfScalar(Add, PosLVal);
4839     }
4840   }
4841 }
4842 
4843 std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause(
4844     CodeGenFunction &CGF, ArrayRef<OMPTaskDataTy::DependData> Dependencies,
4845     SourceLocation Loc) {
4846   if (llvm::all_of(Dependencies, [](const OMPTaskDataTy::DependData &D) {
4847         return D.DepExprs.empty();
4848       }))
4849     return std::make_pair(nullptr, Address::invalid());
4850   // Process list of dependencies.
4851   ASTContext &C = CGM.getContext();
4852   Address DependenciesArray = Address::invalid();
4853   llvm::Value *NumOfElements = nullptr;
4854   unsigned NumDependencies = std::accumulate(
4855       Dependencies.begin(), Dependencies.end(), 0,
4856       [](unsigned V, const OMPTaskDataTy::DependData &D) {
4857         return D.DepKind == OMPC_DEPEND_depobj
4858                    ? V
4859                    : (V + (D.IteratorExpr ? 0 : D.DepExprs.size()));
4860       });
4861   QualType FlagsTy;
4862   getDependTypes(C, KmpDependInfoTy, FlagsTy);
4863   bool HasDepobjDeps = false;
4864   bool HasRegularWithIterators = false;
4865   llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0);
4866   llvm::Value *NumOfRegularWithIterators =
4867       llvm::ConstantInt::get(CGF.IntPtrTy, 0);
4868   // Calculate number of depobj dependecies and regular deps with the iterators.
4869   for (const OMPTaskDataTy::DependData &D : Dependencies) {
4870     if (D.DepKind == OMPC_DEPEND_depobj) {
4871       SmallVector<llvm::Value *, 4> Sizes =
4872           emitDepobjElementsSizes(CGF, KmpDependInfoTy, D);
4873       for (llvm::Value *Size : Sizes) {
4874         NumOfDepobjElements =
4875             CGF.Builder.CreateNUWAdd(NumOfDepobjElements, Size);
4876       }
4877       HasDepobjDeps = true;
4878       continue;
4879     }
4880     // Include number of iterations, if any.
4881 
4882     if (const auto *IE = cast_or_null<OMPIteratorExpr>(D.IteratorExpr)) {
4883       for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4884         llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper);
4885         Sz = CGF.Builder.CreateIntCast(Sz, CGF.IntPtrTy, /*isSigned=*/false);
4886         llvm::Value *NumClauseDeps = CGF.Builder.CreateNUWMul(
4887             Sz, llvm::ConstantInt::get(CGF.IntPtrTy, D.DepExprs.size()));
4888         NumOfRegularWithIterators =
4889             CGF.Builder.CreateNUWAdd(NumOfRegularWithIterators, NumClauseDeps);
4890       }
4891       HasRegularWithIterators = true;
4892       continue;
4893     }
4894   }
4895 
4896   QualType KmpDependInfoArrayTy;
4897   if (HasDepobjDeps || HasRegularWithIterators) {
4898     NumOfElements = llvm::ConstantInt::get(CGM.IntPtrTy, NumDependencies,
4899                                            /*isSigned=*/false);
4900     if (HasDepobjDeps) {
4901       NumOfElements =
4902           CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumOfElements);
4903     }
4904     if (HasRegularWithIterators) {
4905       NumOfElements =
4906           CGF.Builder.CreateNUWAdd(NumOfRegularWithIterators, NumOfElements);
4907     }
4908     auto *OVE = new (C) OpaqueValueExpr(
4909         Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0),
4910         VK_PRValue);
4911     CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE,
4912                                                   RValue::get(NumOfElements));
4913     KmpDependInfoArrayTy =
4914         C.getVariableArrayType(KmpDependInfoTy, OVE, ArrayType::Normal,
4915                                /*IndexTypeQuals=*/0, SourceRange(Loc, Loc));
4916     // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy);
4917     // Properly emit variable-sized array.
4918     auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy,
4919                                          ImplicitParamDecl::Other);
4920     CGF.EmitVarDecl(*PD);
4921     DependenciesArray = CGF.GetAddrOfLocalVar(PD);
4922     NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty,
4923                                               /*isSigned=*/false);
4924   } else {
4925     KmpDependInfoArrayTy = C.getConstantArrayType(
4926         KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), nullptr,
4927         ArrayType::Normal, /*IndexTypeQuals=*/0);
4928     DependenciesArray =
4929         CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr");
4930     DependenciesArray = CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0);
4931     NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies,
4932                                            /*isSigned=*/false);
4933   }
4934   unsigned Pos = 0;
4935   for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) {
4936     if (Dependencies[I].DepKind == OMPC_DEPEND_depobj ||
4937         Dependencies[I].IteratorExpr)
4938       continue;
4939     emitDependData(CGF, KmpDependInfoTy, &Pos, Dependencies[I],
4940                    DependenciesArray);
4941   }
4942   // Copy regular dependecies with iterators.
4943   LValue PosLVal = CGF.MakeAddrLValue(
4944       CGF.CreateMemTemp(C.getSizeType(), "dep.counter.addr"), C.getSizeType());
4945   CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal);
4946   for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) {
4947     if (Dependencies[I].DepKind == OMPC_DEPEND_depobj ||
4948         !Dependencies[I].IteratorExpr)
4949       continue;
4950     emitDependData(CGF, KmpDependInfoTy, &PosLVal, Dependencies[I],
4951                    DependenciesArray);
4952   }
4953   // Copy final depobj arrays without iterators.
4954   if (HasDepobjDeps) {
4955     for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) {
4956       if (Dependencies[I].DepKind != OMPC_DEPEND_depobj)
4957         continue;
4958       emitDepobjElements(CGF, KmpDependInfoTy, PosLVal, Dependencies[I],
4959                          DependenciesArray);
4960     }
4961   }
4962   DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
4963       DependenciesArray, CGF.VoidPtrTy, CGF.Int8Ty);
4964   return std::make_pair(NumOfElements, DependenciesArray);
4965 }
4966 
4967 Address CGOpenMPRuntime::emitDepobjDependClause(
4968     CodeGenFunction &CGF, const OMPTaskDataTy::DependData &Dependencies,
4969     SourceLocation Loc) {
4970   if (Dependencies.DepExprs.empty())
4971     return Address::invalid();
4972   // Process list of dependencies.
4973   ASTContext &C = CGM.getContext();
4974   Address DependenciesArray = Address::invalid();
4975   unsigned NumDependencies = Dependencies.DepExprs.size();
4976   QualType FlagsTy;
4977   getDependTypes(C, KmpDependInfoTy, FlagsTy);
4978   RecordDecl *KmpDependInfoRD =
4979       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
4980 
4981   llvm::Value *Size;
4982   // Define type kmp_depend_info[<Dependencies.size()>];
4983   // For depobj reserve one extra element to store the number of elements.
4984   // It is required to handle depobj(x) update(in) construct.
4985   // kmp_depend_info[<Dependencies.size()>] deps;
4986   llvm::Value *NumDepsVal;
4987   CharUnits Align = C.getTypeAlignInChars(KmpDependInfoTy);
4988   if (const auto *IE =
4989           cast_or_null<OMPIteratorExpr>(Dependencies.IteratorExpr)) {
4990     NumDepsVal = llvm::ConstantInt::get(CGF.SizeTy, 1);
4991     for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) {
4992       llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper);
4993       Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false);
4994       NumDepsVal = CGF.Builder.CreateNUWMul(NumDepsVal, Sz);
4995     }
4996     Size = CGF.Builder.CreateNUWAdd(llvm::ConstantInt::get(CGF.SizeTy, 1),
4997                                     NumDepsVal);
4998     CharUnits SizeInBytes =
4999         C.getTypeSizeInChars(KmpDependInfoTy).alignTo(Align);
5000     llvm::Value *RecSize = CGM.getSize(SizeInBytes);
5001     Size = CGF.Builder.CreateNUWMul(Size, RecSize);
5002     NumDepsVal =
5003         CGF.Builder.CreateIntCast(NumDepsVal, CGF.IntPtrTy, /*isSigned=*/false);
5004   } else {
5005     QualType KmpDependInfoArrayTy = C.getConstantArrayType(
5006         KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1),
5007         nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
5008     CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy);
5009     Size = CGM.getSize(Sz.alignTo(Align));
5010     NumDepsVal = llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies);
5011   }
5012   // Need to allocate on the dynamic memory.
5013   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5014   // Use default allocator.
5015   llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5016   llvm::Value *Args[] = {ThreadID, Size, Allocator};
5017 
5018   llvm::Value *Addr =
5019       CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5020                               CGM.getModule(), OMPRTL___kmpc_alloc),
5021                           Args, ".dep.arr.addr");
5022   llvm::Type *KmpDependInfoLlvmTy = CGF.ConvertTypeForMem(KmpDependInfoTy);
5023   Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5024       Addr, KmpDependInfoLlvmTy->getPointerTo());
5025   DependenciesArray = Address(Addr, KmpDependInfoLlvmTy, Align);
5026   // Write number of elements in the first element of array for depobj.
5027   LValue Base = CGF.MakeAddrLValue(DependenciesArray, KmpDependInfoTy);
5028   // deps[i].base_addr = NumDependencies;
5029   LValue BaseAddrLVal = CGF.EmitLValueForField(
5030       Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr));
5031   CGF.EmitStoreOfScalar(NumDepsVal, BaseAddrLVal);
5032   llvm::PointerUnion<unsigned *, LValue *> Pos;
5033   unsigned Idx = 1;
5034   LValue PosLVal;
5035   if (Dependencies.IteratorExpr) {
5036     PosLVal = CGF.MakeAddrLValue(
5037         CGF.CreateMemTemp(C.getSizeType(), "iterator.counter.addr"),
5038         C.getSizeType());
5039     CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Idx), PosLVal,
5040                           /*IsInit=*/true);
5041     Pos = &PosLVal;
5042   } else {
5043     Pos = &Idx;
5044   }
5045   emitDependData(CGF, KmpDependInfoTy, Pos, Dependencies, DependenciesArray);
5046   DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5047       CGF.Builder.CreateConstGEP(DependenciesArray, 1), CGF.VoidPtrTy,
5048       CGF.Int8Ty);
5049   return DependenciesArray;
5050 }
5051 
5052 void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal,
5053                                         SourceLocation Loc) {
5054   ASTContext &C = CGM.getContext();
5055   QualType FlagsTy;
5056   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5057   LValue Base = CGF.EmitLoadOfPointerLValue(
5058       DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>());
5059   QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy);
5060   Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5061       Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy),
5062       CGF.ConvertTypeForMem(KmpDependInfoTy));
5063   llvm::Value *DepObjAddr = CGF.Builder.CreateGEP(
5064       Addr.getElementType(), Addr.getPointer(),
5065       llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true));
5066   DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr,
5067                                                                CGF.VoidPtrTy);
5068   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5069   // Use default allocator.
5070   llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5071   llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator};
5072 
5073   // _kmpc_free(gtid, addr, nullptr);
5074   (void)CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5075                                 CGM.getModule(), OMPRTL___kmpc_free),
5076                             Args);
5077 }
5078 
5079 void CGOpenMPRuntime::emitUpdateClause(CodeGenFunction &CGF, LValue DepobjLVal,
5080                                        OpenMPDependClauseKind NewDepKind,
5081                                        SourceLocation Loc) {
5082   ASTContext &C = CGM.getContext();
5083   QualType FlagsTy;
5084   getDependTypes(C, KmpDependInfoTy, FlagsTy);
5085   RecordDecl *KmpDependInfoRD =
5086       cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl());
5087   llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy);
5088   llvm::Value *NumDeps;
5089   LValue Base;
5090   std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc);
5091 
5092   Address Begin = Base.getAddress(CGF);
5093   // Cast from pointer to array type to pointer to single element.
5094   llvm::Value *End = CGF.Builder.CreateGEP(
5095       Begin.getElementType(), Begin.getPointer(), NumDeps);
5096   // The basic structure here is a while-do loop.
5097   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body");
5098   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done");
5099   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5100   CGF.EmitBlock(BodyBB);
5101   llvm::PHINode *ElementPHI =
5102       CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast");
5103   ElementPHI->addIncoming(Begin.getPointer(), EntryBB);
5104   Begin = Begin.withPointer(ElementPHI);
5105   Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(),
5106                             Base.getTBAAInfo());
5107   // deps[i].flags = NewDepKind;
5108   RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind);
5109   LValue FlagsLVal = CGF.EmitLValueForField(
5110       Base, *std::next(KmpDependInfoRD->field_begin(), Flags));
5111   CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind),
5112                         FlagsLVal);
5113 
5114   // Shift the address forward by one element.
5115   Address ElementNext =
5116       CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext");
5117   ElementPHI->addIncoming(ElementNext.getPointer(),
5118                           CGF.Builder.GetInsertBlock());
5119   llvm::Value *IsEmpty =
5120       CGF.Builder.CreateICmpEQ(ElementNext.getPointer(), End, "omp.isempty");
5121   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5122   // Done.
5123   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5124 }
5125 
5126 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
5127                                    const OMPExecutableDirective &D,
5128                                    llvm::Function *TaskFunction,
5129                                    QualType SharedsTy, Address Shareds,
5130                                    const Expr *IfCond,
5131                                    const OMPTaskDataTy &Data) {
5132   if (!CGF.HaveInsertPoint())
5133     return;
5134 
5135   TaskResultTy Result =
5136       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5137   llvm::Value *NewTask = Result.NewTask;
5138   llvm::Function *TaskEntry = Result.TaskEntry;
5139   llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy;
5140   LValue TDBase = Result.TDBase;
5141   const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD;
5142   // Process list of dependences.
5143   Address DependenciesArray = Address::invalid();
5144   llvm::Value *NumOfElements;
5145   std::tie(NumOfElements, DependenciesArray) =
5146       emitDependClause(CGF, Data.Dependences, Loc);
5147 
5148   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5149   // libcall.
5150   // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid,
5151   // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list,
5152   // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence
5153   // list is not empty
5154   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5155   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5156   llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask };
5157   llvm::Value *DepTaskArgs[7];
5158   if (!Data.Dependences.empty()) {
5159     DepTaskArgs[0] = UpLoc;
5160     DepTaskArgs[1] = ThreadID;
5161     DepTaskArgs[2] = NewTask;
5162     DepTaskArgs[3] = NumOfElements;
5163     DepTaskArgs[4] = DependenciesArray.getPointer();
5164     DepTaskArgs[5] = CGF.Builder.getInt32(0);
5165     DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5166   }
5167   auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs,
5168                         &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) {
5169     if (!Data.Tied) {
5170       auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
5171       LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI);
5172       CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal);
5173     }
5174     if (!Data.Dependences.empty()) {
5175       CGF.EmitRuntimeCall(
5176           OMPBuilder.getOrCreateRuntimeFunction(
5177               CGM.getModule(), OMPRTL___kmpc_omp_task_with_deps),
5178           DepTaskArgs);
5179     } else {
5180       CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5181                               CGM.getModule(), OMPRTL___kmpc_omp_task),
5182                           TaskArgs);
5183     }
5184     // Check if parent region is untied and build return for untied task;
5185     if (auto *Region =
5186             dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
5187       Region->emitUntiedSwitch(CGF);
5188   };
5189 
5190   llvm::Value *DepWaitTaskArgs[6];
5191   if (!Data.Dependences.empty()) {
5192     DepWaitTaskArgs[0] = UpLoc;
5193     DepWaitTaskArgs[1] = ThreadID;
5194     DepWaitTaskArgs[2] = NumOfElements;
5195     DepWaitTaskArgs[3] = DependenciesArray.getPointer();
5196     DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
5197     DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
5198   }
5199   auto &M = CGM.getModule();
5200   auto &&ElseCodeGen = [this, &M, &TaskArgs, ThreadID, NewTaskNewTaskTTy,
5201                         TaskEntry, &Data, &DepWaitTaskArgs,
5202                         Loc](CodeGenFunction &CGF, PrePostActionTy &) {
5203     CodeGenFunction::RunCleanupsScope LocalScope(CGF);
5204     // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
5205     // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
5206     // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
5207     // is specified.
5208     if (!Data.Dependences.empty())
5209       CGF.EmitRuntimeCall(
5210           OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_wait_deps),
5211           DepWaitTaskArgs);
5212     // Call proxy_task_entry(gtid, new_task);
5213     auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy,
5214                       Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
5215       Action.Enter(CGF);
5216       llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
5217       CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry,
5218                                                           OutlinedFnArgs);
5219     };
5220 
5221     // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid,
5222     // kmp_task_t *new_task);
5223     // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
5224     // kmp_task_t *new_task);
5225     RegionCodeGenTy RCG(CodeGen);
5226     CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction(
5227                               M, OMPRTL___kmpc_omp_task_begin_if0),
5228                           TaskArgs,
5229                           OMPBuilder.getOrCreateRuntimeFunction(
5230                               M, OMPRTL___kmpc_omp_task_complete_if0),
5231                           TaskArgs);
5232     RCG.setAction(Action);
5233     RCG(CGF);
5234   };
5235 
5236   if (IfCond) {
5237     emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen);
5238   } else {
5239     RegionCodeGenTy ThenRCG(ThenCodeGen);
5240     ThenRCG(CGF);
5241   }
5242 }
5243 
5244 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc,
5245                                        const OMPLoopDirective &D,
5246                                        llvm::Function *TaskFunction,
5247                                        QualType SharedsTy, Address Shareds,
5248                                        const Expr *IfCond,
5249                                        const OMPTaskDataTy &Data) {
5250   if (!CGF.HaveInsertPoint())
5251     return;
5252   TaskResultTy Result =
5253       emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data);
5254   // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc()
5255   // libcall.
5256   // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int
5257   // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int
5258   // sched, kmp_uint64 grainsize, void *task_dup);
5259   llvm::Value *ThreadID = getThreadID(CGF, Loc);
5260   llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
5261   llvm::Value *IfVal;
5262   if (IfCond) {
5263     IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy,
5264                                       /*isSigned=*/true);
5265   } else {
5266     IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1);
5267   }
5268 
5269   LValue LBLVal = CGF.EmitLValueForField(
5270       Result.TDBase,
5271       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound));
5272   const auto *LBVar =
5273       cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl());
5274   CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF),
5275                        LBLVal.getQuals(),
5276                        /*IsInitializer=*/true);
5277   LValue UBLVal = CGF.EmitLValueForField(
5278       Result.TDBase,
5279       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound));
5280   const auto *UBVar =
5281       cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl());
5282   CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF),
5283                        UBLVal.getQuals(),
5284                        /*IsInitializer=*/true);
5285   LValue StLVal = CGF.EmitLValueForField(
5286       Result.TDBase,
5287       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride));
5288   const auto *StVar =
5289       cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl());
5290   CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF),
5291                        StLVal.getQuals(),
5292                        /*IsInitializer=*/true);
5293   // Store reductions address.
5294   LValue RedLVal = CGF.EmitLValueForField(
5295       Result.TDBase,
5296       *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions));
5297   if (Data.Reductions) {
5298     CGF.EmitStoreOfScalar(Data.Reductions, RedLVal);
5299   } else {
5300     CGF.EmitNullInitialization(RedLVal.getAddress(CGF),
5301                                CGF.getContext().VoidPtrTy);
5302   }
5303   enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 };
5304   llvm::Value *TaskArgs[] = {
5305       UpLoc,
5306       ThreadID,
5307       Result.NewTask,
5308       IfVal,
5309       LBLVal.getPointer(CGF),
5310       UBLVal.getPointer(CGF),
5311       CGF.EmitLoadOfScalar(StLVal, Loc),
5312       llvm::ConstantInt::getSigned(
5313           CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler
5314       llvm::ConstantInt::getSigned(
5315           CGF.IntTy, Data.Schedule.getPointer()
5316                          ? Data.Schedule.getInt() ? NumTasks : Grainsize
5317                          : NoSchedule),
5318       Data.Schedule.getPointer()
5319           ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty,
5320                                       /*isSigned=*/false)
5321           : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0),
5322       Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5323                              Result.TaskDupFn, CGF.VoidPtrTy)
5324                        : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)};
5325   CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
5326                           CGM.getModule(), OMPRTL___kmpc_taskloop),
5327                       TaskArgs);
5328 }
5329 
5330 /// Emit reduction operation for each element of array (required for
5331 /// array sections) LHS op = RHS.
5332 /// \param Type Type of array.
5333 /// \param LHSVar Variable on the left side of the reduction operation
5334 /// (references element of array in original variable).
5335 /// \param RHSVar Variable on the right side of the reduction operation
5336 /// (references element of array in original variable).
5337 /// \param RedOpGen Generator of reduction operation with use of LHSVar and
5338 /// RHSVar.
5339 static void EmitOMPAggregateReduction(
5340     CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar,
5341     const VarDecl *RHSVar,
5342     const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *,
5343                                   const Expr *, const Expr *)> &RedOpGen,
5344     const Expr *XExpr = nullptr, const Expr *EExpr = nullptr,
5345     const Expr *UpExpr = nullptr) {
5346   // Perform element-by-element initialization.
5347   QualType ElementTy;
5348   Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar);
5349   Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar);
5350 
5351   // Drill down to the base element type on both arrays.
5352   const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe();
5353   llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr);
5354 
5355   llvm::Value *RHSBegin = RHSAddr.getPointer();
5356   llvm::Value *LHSBegin = LHSAddr.getPointer();
5357   // Cast from pointer to array type to pointer to single element.
5358   llvm::Value *LHSEnd =
5359       CGF.Builder.CreateGEP(LHSAddr.getElementType(), LHSBegin, NumElements);
5360   // The basic structure here is a while-do loop.
5361   llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body");
5362   llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done");
5363   llvm::Value *IsEmpty =
5364       CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty");
5365   CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
5366 
5367   // Enter the loop body, making that address the current address.
5368   llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock();
5369   CGF.EmitBlock(BodyBB);
5370 
5371   CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy);
5372 
5373   llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI(
5374       RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast");
5375   RHSElementPHI->addIncoming(RHSBegin, EntryBB);
5376   Address RHSElementCurrent(
5377       RHSElementPHI, RHSAddr.getElementType(),
5378       RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5379 
5380   llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI(
5381       LHSBegin->getType(), 2, "omp.arraycpy.destElementPast");
5382   LHSElementPHI->addIncoming(LHSBegin, EntryBB);
5383   Address LHSElementCurrent(
5384       LHSElementPHI, LHSAddr.getElementType(),
5385       LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize));
5386 
5387   // Emit copy.
5388   CodeGenFunction::OMPPrivateScope Scope(CGF);
5389   Scope.addPrivate(LHSVar, LHSElementCurrent);
5390   Scope.addPrivate(RHSVar, RHSElementCurrent);
5391   Scope.Privatize();
5392   RedOpGen(CGF, XExpr, EExpr, UpExpr);
5393   Scope.ForceCleanup();
5394 
5395   // Shift the address forward by one element.
5396   llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32(
5397       LHSAddr.getElementType(), LHSElementPHI, /*Idx0=*/1,
5398       "omp.arraycpy.dest.element");
5399   llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32(
5400       RHSAddr.getElementType(), RHSElementPHI, /*Idx0=*/1,
5401       "omp.arraycpy.src.element");
5402   // Check whether we've reached the end.
5403   llvm::Value *Done =
5404       CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done");
5405   CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB);
5406   LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock());
5407   RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock());
5408 
5409   // Done.
5410   CGF.EmitBlock(DoneBB, /*IsFinished=*/true);
5411 }
5412 
5413 /// Emit reduction combiner. If the combiner is a simple expression emit it as
5414 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of
5415 /// UDR combiner function.
5416 static void emitReductionCombiner(CodeGenFunction &CGF,
5417                                   const Expr *ReductionOp) {
5418   if (const auto *CE = dyn_cast<CallExpr>(ReductionOp))
5419     if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee()))
5420       if (const auto *DRE =
5421               dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts()))
5422         if (const auto *DRD =
5423                 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) {
5424           std::pair<llvm::Function *, llvm::Function *> Reduction =
5425               CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD);
5426           RValue Func = RValue::get(Reduction.first);
5427           CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func);
5428           CGF.EmitIgnoredExpr(ReductionOp);
5429           return;
5430         }
5431   CGF.EmitIgnoredExpr(ReductionOp);
5432 }
5433 
5434 llvm::Function *CGOpenMPRuntime::emitReductionFunction(
5435     SourceLocation Loc, llvm::Type *ArgsElemType,
5436     ArrayRef<const Expr *> Privates, ArrayRef<const Expr *> LHSExprs,
5437     ArrayRef<const Expr *> RHSExprs, ArrayRef<const Expr *> ReductionOps) {
5438   ASTContext &C = CGM.getContext();
5439 
5440   // void reduction_func(void *LHSArg, void *RHSArg);
5441   FunctionArgList Args;
5442   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5443                            ImplicitParamDecl::Other);
5444   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5445                            ImplicitParamDecl::Other);
5446   Args.push_back(&LHSArg);
5447   Args.push_back(&RHSArg);
5448   const auto &CGFI =
5449       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5450   std::string Name = getName({"omp", "reduction", "reduction_func"});
5451   auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI),
5452                                     llvm::GlobalValue::InternalLinkage, Name,
5453                                     &CGM.getModule());
5454   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI);
5455   Fn->setDoesNotRecurse();
5456   CodeGenFunction CGF(CGM);
5457   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc);
5458 
5459   // Dst = (void*[n])(LHSArg);
5460   // Src = (void*[n])(RHSArg);
5461   Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5462                   CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)),
5463                   ArgsElemType->getPointerTo()),
5464               ArgsElemType, CGF.getPointerAlign());
5465   Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5466                   CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)),
5467                   ArgsElemType->getPointerTo()),
5468               ArgsElemType, CGF.getPointerAlign());
5469 
5470   //  ...
5471   //  *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
5472   //  ...
5473   CodeGenFunction::OMPPrivateScope Scope(CGF);
5474   const auto *IPriv = Privates.begin();
5475   unsigned Idx = 0;
5476   for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) {
5477     const auto *RHSVar =
5478         cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl());
5479     Scope.addPrivate(RHSVar, emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar));
5480     const auto *LHSVar =
5481         cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl());
5482     Scope.addPrivate(LHSVar, emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar));
5483     QualType PrivTy = (*IPriv)->getType();
5484     if (PrivTy->isVariablyModifiedType()) {
5485       // Get array size and emit VLA type.
5486       ++Idx;
5487       Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx);
5488       llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem);
5489       const VariableArrayType *VLA =
5490           CGF.getContext().getAsVariableArrayType(PrivTy);
5491       const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr());
5492       CodeGenFunction::OpaqueValueMapping OpaqueMap(
5493           CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy)));
5494       CGF.EmitVariablyModifiedType(PrivTy);
5495     }
5496   }
5497   Scope.Privatize();
5498   IPriv = Privates.begin();
5499   const auto *ILHS = LHSExprs.begin();
5500   const auto *IRHS = RHSExprs.begin();
5501   for (const Expr *E : ReductionOps) {
5502     if ((*IPriv)->getType()->isArrayType()) {
5503       // Emit reduction for array section.
5504       const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5505       const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5506       EmitOMPAggregateReduction(
5507           CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5508           [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5509             emitReductionCombiner(CGF, E);
5510           });
5511     } else {
5512       // Emit reduction for array subscript or single variable.
5513       emitReductionCombiner(CGF, E);
5514     }
5515     ++IPriv;
5516     ++ILHS;
5517     ++IRHS;
5518   }
5519   Scope.ForceCleanup();
5520   CGF.FinishFunction();
5521   return Fn;
5522 }
5523 
5524 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF,
5525                                                   const Expr *ReductionOp,
5526                                                   const Expr *PrivateRef,
5527                                                   const DeclRefExpr *LHS,
5528                                                   const DeclRefExpr *RHS) {
5529   if (PrivateRef->getType()->isArrayType()) {
5530     // Emit reduction for array section.
5531     const auto *LHSVar = cast<VarDecl>(LHS->getDecl());
5532     const auto *RHSVar = cast<VarDecl>(RHS->getDecl());
5533     EmitOMPAggregateReduction(
5534         CGF, PrivateRef->getType(), LHSVar, RHSVar,
5535         [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) {
5536           emitReductionCombiner(CGF, ReductionOp);
5537         });
5538   } else {
5539     // Emit reduction for array subscript or single variable.
5540     emitReductionCombiner(CGF, ReductionOp);
5541   }
5542 }
5543 
5544 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc,
5545                                     ArrayRef<const Expr *> Privates,
5546                                     ArrayRef<const Expr *> LHSExprs,
5547                                     ArrayRef<const Expr *> RHSExprs,
5548                                     ArrayRef<const Expr *> ReductionOps,
5549                                     ReductionOptionsTy Options) {
5550   if (!CGF.HaveInsertPoint())
5551     return;
5552 
5553   bool WithNowait = Options.WithNowait;
5554   bool SimpleReduction = Options.SimpleReduction;
5555 
5556   // Next code should be emitted for reduction:
5557   //
5558   // static kmp_critical_name lock = { 0 };
5559   //
5560   // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
5561   //  *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
5562   //  ...
5563   //  *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
5564   //  *(Type<n>-1*)rhs[<n>-1]);
5565   // }
5566   //
5567   // ...
5568   // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
5569   // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5570   // RedList, reduce_func, &<lock>)) {
5571   // case 1:
5572   //  ...
5573   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5574   //  ...
5575   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5576   // break;
5577   // case 2:
5578   //  ...
5579   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5580   //  ...
5581   // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
5582   // break;
5583   // default:;
5584   // }
5585   //
5586   // if SimpleReduction is true, only the next code is generated:
5587   //  ...
5588   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5589   //  ...
5590 
5591   ASTContext &C = CGM.getContext();
5592 
5593   if (SimpleReduction) {
5594     CodeGenFunction::RunCleanupsScope Scope(CGF);
5595     const auto *IPriv = Privates.begin();
5596     const auto *ILHS = LHSExprs.begin();
5597     const auto *IRHS = RHSExprs.begin();
5598     for (const Expr *E : ReductionOps) {
5599       emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5600                                   cast<DeclRefExpr>(*IRHS));
5601       ++IPriv;
5602       ++ILHS;
5603       ++IRHS;
5604     }
5605     return;
5606   }
5607 
5608   // 1. Build a list of reduction variables.
5609   // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
5610   auto Size = RHSExprs.size();
5611   for (const Expr *E : Privates) {
5612     if (E->getType()->isVariablyModifiedType())
5613       // Reserve place for array size.
5614       ++Size;
5615   }
5616   llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size);
5617   QualType ReductionArrayTy =
5618       C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal,
5619                              /*IndexTypeQuals=*/0);
5620   Address ReductionList =
5621       CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list");
5622   const auto *IPriv = Privates.begin();
5623   unsigned Idx = 0;
5624   for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) {
5625     Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5626     CGF.Builder.CreateStore(
5627         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5628             CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy),
5629         Elem);
5630     if ((*IPriv)->getType()->isVariablyModifiedType()) {
5631       // Store array size.
5632       ++Idx;
5633       Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx);
5634       llvm::Value *Size = CGF.Builder.CreateIntCast(
5635           CGF.getVLASize(
5636                  CGF.getContext().getAsVariableArrayType((*IPriv)->getType()))
5637               .NumElts,
5638           CGF.SizeTy, /*isSigned=*/false);
5639       CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy),
5640                               Elem);
5641     }
5642   }
5643 
5644   // 2. Emit reduce_func().
5645   llvm::Function *ReductionFn =
5646       emitReductionFunction(Loc, CGF.ConvertTypeForMem(ReductionArrayTy),
5647                             Privates, LHSExprs, RHSExprs, ReductionOps);
5648 
5649   // 3. Create static kmp_critical_name lock = { 0 };
5650   std::string Name = getName({"reduction"});
5651   llvm::Value *Lock = getCriticalRegionLock(Name);
5652 
5653   // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
5654   // RedList, reduce_func, &<lock>);
5655   llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE);
5656   llvm::Value *ThreadId = getThreadID(CGF, Loc);
5657   llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy);
5658   llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
5659       ReductionList.getPointer(), CGF.VoidPtrTy);
5660   llvm::Value *Args[] = {
5661       IdentTLoc,                             // ident_t *<loc>
5662       ThreadId,                              // i32 <gtid>
5663       CGF.Builder.getInt32(RHSExprs.size()), // i32 <n>
5664       ReductionArrayTySize,                  // size_type sizeof(RedList)
5665       RL,                                    // void *RedList
5666       ReductionFn, // void (*) (void *, void *) <reduce_func>
5667       Lock         // kmp_critical_name *&<lock>
5668   };
5669   llvm::Value *Res = CGF.EmitRuntimeCall(
5670       OMPBuilder.getOrCreateRuntimeFunction(
5671           CGM.getModule(),
5672           WithNowait ? OMPRTL___kmpc_reduce_nowait : OMPRTL___kmpc_reduce),
5673       Args);
5674 
5675   // 5. Build switch(res)
5676   llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default");
5677   llvm::SwitchInst *SwInst =
5678       CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2);
5679 
5680   // 6. Build case 1:
5681   //  ...
5682   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
5683   //  ...
5684   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5685   // break;
5686   llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1");
5687   SwInst->addCase(CGF.Builder.getInt32(1), Case1BB);
5688   CGF.EmitBlock(Case1BB);
5689 
5690   // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
5691   llvm::Value *EndArgs[] = {
5692       IdentTLoc, // ident_t *<loc>
5693       ThreadId,  // i32 <gtid>
5694       Lock       // kmp_critical_name *&<lock>
5695   };
5696   auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps](
5697                        CodeGenFunction &CGF, PrePostActionTy &Action) {
5698     CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5699     const auto *IPriv = Privates.begin();
5700     const auto *ILHS = LHSExprs.begin();
5701     const auto *IRHS = RHSExprs.begin();
5702     for (const Expr *E : ReductionOps) {
5703       RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS),
5704                                      cast<DeclRefExpr>(*IRHS));
5705       ++IPriv;
5706       ++ILHS;
5707       ++IRHS;
5708     }
5709   };
5710   RegionCodeGenTy RCG(CodeGen);
5711   CommonActionTy Action(
5712       nullptr, llvm::None,
5713       OMPBuilder.getOrCreateRuntimeFunction(
5714           CGM.getModule(), WithNowait ? OMPRTL___kmpc_end_reduce_nowait
5715                                       : OMPRTL___kmpc_end_reduce),
5716       EndArgs);
5717   RCG.setAction(Action);
5718   RCG(CGF);
5719 
5720   CGF.EmitBranch(DefaultBB);
5721 
5722   // 7. Build case 2:
5723   //  ...
5724   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
5725   //  ...
5726   // break;
5727   llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2");
5728   SwInst->addCase(CGF.Builder.getInt32(2), Case2BB);
5729   CGF.EmitBlock(Case2BB);
5730 
5731   auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps](
5732                              CodeGenFunction &CGF, PrePostActionTy &Action) {
5733     const auto *ILHS = LHSExprs.begin();
5734     const auto *IRHS = RHSExprs.begin();
5735     const auto *IPriv = Privates.begin();
5736     for (const Expr *E : ReductionOps) {
5737       const Expr *XExpr = nullptr;
5738       const Expr *EExpr = nullptr;
5739       const Expr *UpExpr = nullptr;
5740       BinaryOperatorKind BO = BO_Comma;
5741       if (const auto *BO = dyn_cast<BinaryOperator>(E)) {
5742         if (BO->getOpcode() == BO_Assign) {
5743           XExpr = BO->getLHS();
5744           UpExpr = BO->getRHS();
5745         }
5746       }
5747       // Try to emit update expression as a simple atomic.
5748       const Expr *RHSExpr = UpExpr;
5749       if (RHSExpr) {
5750         // Analyze RHS part of the whole expression.
5751         if (const auto *ACO = dyn_cast<AbstractConditionalOperator>(
5752                 RHSExpr->IgnoreParenImpCasts())) {
5753           // If this is a conditional operator, analyze its condition for
5754           // min/max reduction operator.
5755           RHSExpr = ACO->getCond();
5756         }
5757         if (const auto *BORHS =
5758                 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) {
5759           EExpr = BORHS->getRHS();
5760           BO = BORHS->getOpcode();
5761         }
5762       }
5763       if (XExpr) {
5764         const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5765         auto &&AtomicRedGen = [BO, VD,
5766                                Loc](CodeGenFunction &CGF, const Expr *XExpr,
5767                                     const Expr *EExpr, const Expr *UpExpr) {
5768           LValue X = CGF.EmitLValue(XExpr);
5769           RValue E;
5770           if (EExpr)
5771             E = CGF.EmitAnyExpr(EExpr);
5772           CGF.EmitOMPAtomicSimpleUpdateExpr(
5773               X, E, BO, /*IsXLHSInRHSPart=*/true,
5774               llvm::AtomicOrdering::Monotonic, Loc,
5775               [&CGF, UpExpr, VD, Loc](RValue XRValue) {
5776                 CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
5777                 Address LHSTemp = CGF.CreateMemTemp(VD->getType());
5778                 CGF.emitOMPSimpleStore(
5779                     CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue,
5780                     VD->getType().getNonReferenceType(), Loc);
5781                 PrivateScope.addPrivate(VD, LHSTemp);
5782                 (void)PrivateScope.Privatize();
5783                 return CGF.EmitAnyExpr(UpExpr);
5784               });
5785         };
5786         if ((*IPriv)->getType()->isArrayType()) {
5787           // Emit atomic reduction for array section.
5788           const auto *RHSVar =
5789               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5790           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar,
5791                                     AtomicRedGen, XExpr, EExpr, UpExpr);
5792         } else {
5793           // Emit atomic reduction for array subscript or single variable.
5794           AtomicRedGen(CGF, XExpr, EExpr, UpExpr);
5795         }
5796       } else {
5797         // Emit as a critical region.
5798         auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *,
5799                                            const Expr *, const Expr *) {
5800           CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
5801           std::string Name = RT.getName({"atomic_reduction"});
5802           RT.emitCriticalRegion(
5803               CGF, Name,
5804               [=](CodeGenFunction &CGF, PrePostActionTy &Action) {
5805                 Action.Enter(CGF);
5806                 emitReductionCombiner(CGF, E);
5807               },
5808               Loc);
5809         };
5810         if ((*IPriv)->getType()->isArrayType()) {
5811           const auto *LHSVar =
5812               cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl());
5813           const auto *RHSVar =
5814               cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl());
5815           EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar,
5816                                     CritRedGen);
5817         } else {
5818           CritRedGen(CGF, nullptr, nullptr, nullptr);
5819         }
5820       }
5821       ++ILHS;
5822       ++IRHS;
5823       ++IPriv;
5824     }
5825   };
5826   RegionCodeGenTy AtomicRCG(AtomicCodeGen);
5827   if (!WithNowait) {
5828     // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
5829     llvm::Value *EndArgs[] = {
5830         IdentTLoc, // ident_t *<loc>
5831         ThreadId,  // i32 <gtid>
5832         Lock       // kmp_critical_name *&<lock>
5833     };
5834     CommonActionTy Action(nullptr, llvm::None,
5835                           OMPBuilder.getOrCreateRuntimeFunction(
5836                               CGM.getModule(), OMPRTL___kmpc_end_reduce),
5837                           EndArgs);
5838     AtomicRCG.setAction(Action);
5839     AtomicRCG(CGF);
5840   } else {
5841     AtomicRCG(CGF);
5842   }
5843 
5844   CGF.EmitBranch(DefaultBB);
5845   CGF.EmitBlock(DefaultBB, /*IsFinished=*/true);
5846 }
5847 
5848 /// Generates unique name for artificial threadprivate variables.
5849 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>"
5850 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix,
5851                                       const Expr *Ref) {
5852   SmallString<256> Buffer;
5853   llvm::raw_svector_ostream Out(Buffer);
5854   const clang::DeclRefExpr *DE;
5855   const VarDecl *D = ::getBaseDecl(Ref, DE);
5856   if (!D)
5857     D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl());
5858   D = D->getCanonicalDecl();
5859   std::string Name = CGM.getOpenMPRuntime().getName(
5860       {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)});
5861   Out << Prefix << Name << "_"
5862       << D->getCanonicalDecl()->getBeginLoc().getRawEncoding();
5863   return std::string(Out.str());
5864 }
5865 
5866 /// Emits reduction initializer function:
5867 /// \code
5868 /// void @.red_init(void* %arg, void* %orig) {
5869 /// %0 = bitcast void* %arg to <type>*
5870 /// store <type> <init>, <type>* %0
5871 /// ret void
5872 /// }
5873 /// \endcode
5874 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM,
5875                                            SourceLocation Loc,
5876                                            ReductionCodeGen &RCG, unsigned N) {
5877   ASTContext &C = CGM.getContext();
5878   QualType VoidPtrTy = C.VoidPtrTy;
5879   VoidPtrTy.addRestrict();
5880   FunctionArgList Args;
5881   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, VoidPtrTy,
5882                           ImplicitParamDecl::Other);
5883   ImplicitParamDecl ParamOrig(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, VoidPtrTy,
5884                               ImplicitParamDecl::Other);
5885   Args.emplace_back(&Param);
5886   Args.emplace_back(&ParamOrig);
5887   const auto &FnInfo =
5888       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5889   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
5890   std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""});
5891   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
5892                                     Name, &CGM.getModule());
5893   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
5894   Fn->setDoesNotRecurse();
5895   CodeGenFunction CGF(CGM);
5896   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
5897   QualType PrivateType = RCG.getPrivateType(N);
5898   Address PrivateAddr = CGF.EmitLoadOfPointer(
5899       CGF.Builder.CreateElementBitCast(
5900           CGF.GetAddrOfLocalVar(&Param),
5901           CGF.ConvertTypeForMem(PrivateType)->getPointerTo()),
5902       C.getPointerType(PrivateType)->castAs<PointerType>());
5903   llvm::Value *Size = nullptr;
5904   // If the size of the reduction item is non-constant, load it from global
5905   // threadprivate variable.
5906   if (RCG.getSizes(N).second) {
5907     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
5908         CGF, CGM.getContext().getSizeType(),
5909         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
5910     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
5911                                 CGM.getContext().getSizeType(), Loc);
5912   }
5913   RCG.emitAggregateType(CGF, N, Size);
5914   Address OrigAddr = Address::invalid();
5915   // If initializer uses initializer from declare reduction construct, emit a
5916   // pointer to the address of the original reduction item (reuired by reduction
5917   // initializer)
5918   if (RCG.usesReductionInitializer(N)) {
5919     Address SharedAddr = CGF.GetAddrOfLocalVar(&ParamOrig);
5920     OrigAddr = CGF.EmitLoadOfPointer(
5921         SharedAddr,
5922         CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr());
5923   }
5924   // Emit the initializer:
5925   // %0 = bitcast void* %arg to <type>*
5926   // store <type> <init>, <type>* %0
5927   RCG.emitInitialization(CGF, N, PrivateAddr, OrigAddr,
5928                          [](CodeGenFunction &) { return false; });
5929   CGF.FinishFunction();
5930   return Fn;
5931 }
5932 
5933 /// Emits reduction combiner function:
5934 /// \code
5935 /// void @.red_comb(void* %arg0, void* %arg1) {
5936 /// %lhs = bitcast void* %arg0 to <type>*
5937 /// %rhs = bitcast void* %arg1 to <type>*
5938 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs)
5939 /// store <type> %2, <type>* %lhs
5940 /// ret void
5941 /// }
5942 /// \endcode
5943 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM,
5944                                            SourceLocation Loc,
5945                                            ReductionCodeGen &RCG, unsigned N,
5946                                            const Expr *ReductionOp,
5947                                            const Expr *LHS, const Expr *RHS,
5948                                            const Expr *PrivateRef) {
5949   ASTContext &C = CGM.getContext();
5950   const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl());
5951   const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl());
5952   FunctionArgList Args;
5953   ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
5954                                C.VoidPtrTy, ImplicitParamDecl::Other);
5955   ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
5956                             ImplicitParamDecl::Other);
5957   Args.emplace_back(&ParamInOut);
5958   Args.emplace_back(&ParamIn);
5959   const auto &FnInfo =
5960       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
5961   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
5962   std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""});
5963   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
5964                                     Name, &CGM.getModule());
5965   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
5966   Fn->setDoesNotRecurse();
5967   CodeGenFunction CGF(CGM);
5968   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
5969   llvm::Value *Size = nullptr;
5970   // If the size of the reduction item is non-constant, load it from global
5971   // threadprivate variable.
5972   if (RCG.getSizes(N).second) {
5973     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
5974         CGF, CGM.getContext().getSizeType(),
5975         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
5976     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
5977                                 CGM.getContext().getSizeType(), Loc);
5978   }
5979   RCG.emitAggregateType(CGF, N, Size);
5980   // Remap lhs and rhs variables to the addresses of the function arguments.
5981   // %lhs = bitcast void* %arg0 to <type>*
5982   // %rhs = bitcast void* %arg1 to <type>*
5983   CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
5984   PrivateScope.addPrivate(
5985       LHSVD,
5986       // Pull out the pointer to the variable.
5987       CGF.EmitLoadOfPointer(
5988           CGF.Builder.CreateElementBitCast(
5989               CGF.GetAddrOfLocalVar(&ParamInOut),
5990               CGF.ConvertTypeForMem(LHSVD->getType())->getPointerTo()),
5991           C.getPointerType(LHSVD->getType())->castAs<PointerType>()));
5992   PrivateScope.addPrivate(
5993       RHSVD,
5994       // Pull out the pointer to the variable.
5995       CGF.EmitLoadOfPointer(
5996           CGF.Builder.CreateElementBitCast(
5997             CGF.GetAddrOfLocalVar(&ParamIn),
5998             CGF.ConvertTypeForMem(RHSVD->getType())->getPointerTo()),
5999           C.getPointerType(RHSVD->getType())->castAs<PointerType>()));
6000   PrivateScope.Privatize();
6001   // Emit the combiner body:
6002   // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs)
6003   // store <type> %2, <type>* %lhs
6004   CGM.getOpenMPRuntime().emitSingleReductionCombiner(
6005       CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS),
6006       cast<DeclRefExpr>(RHS));
6007   CGF.FinishFunction();
6008   return Fn;
6009 }
6010 
6011 /// Emits reduction finalizer function:
6012 /// \code
6013 /// void @.red_fini(void* %arg) {
6014 /// %0 = bitcast void* %arg to <type>*
6015 /// <destroy>(<type>* %0)
6016 /// ret void
6017 /// }
6018 /// \endcode
6019 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM,
6020                                            SourceLocation Loc,
6021                                            ReductionCodeGen &RCG, unsigned N) {
6022   if (!RCG.needCleanups(N))
6023     return nullptr;
6024   ASTContext &C = CGM.getContext();
6025   FunctionArgList Args;
6026   ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
6027                           ImplicitParamDecl::Other);
6028   Args.emplace_back(&Param);
6029   const auto &FnInfo =
6030       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
6031   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
6032   std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""});
6033   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
6034                                     Name, &CGM.getModule());
6035   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
6036   Fn->setDoesNotRecurse();
6037   CodeGenFunction CGF(CGM);
6038   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
6039   Address PrivateAddr = CGF.EmitLoadOfPointer(
6040       CGF.GetAddrOfLocalVar(&Param), C.VoidPtrTy.castAs<PointerType>());
6041   llvm::Value *Size = nullptr;
6042   // If the size of the reduction item is non-constant, load it from global
6043   // threadprivate variable.
6044   if (RCG.getSizes(N).second) {
6045     Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate(
6046         CGF, CGM.getContext().getSizeType(),
6047         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6048     Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false,
6049                                 CGM.getContext().getSizeType(), Loc);
6050   }
6051   RCG.emitAggregateType(CGF, N, Size);
6052   // Emit the finalizer body:
6053   // <destroy>(<type>* %0)
6054   RCG.emitCleanups(CGF, N, PrivateAddr);
6055   CGF.FinishFunction(Loc);
6056   return Fn;
6057 }
6058 
6059 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit(
6060     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
6061     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
6062   if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty())
6063     return nullptr;
6064 
6065   // Build typedef struct:
6066   // kmp_taskred_input {
6067   //   void *reduce_shar; // shared reduction item
6068   //   void *reduce_orig; // original reduction item used for initialization
6069   //   size_t reduce_size; // size of data item
6070   //   void *reduce_init; // data initialization routine
6071   //   void *reduce_fini; // data finalization routine
6072   //   void *reduce_comb; // data combiner routine
6073   //   kmp_task_red_flags_t flags; // flags for additional info from compiler
6074   // } kmp_taskred_input_t;
6075   ASTContext &C = CGM.getContext();
6076   RecordDecl *RD = C.buildImplicitRecord("kmp_taskred_input_t");
6077   RD->startDefinition();
6078   const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6079   const FieldDecl *OrigFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6080   const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType());
6081   const FieldDecl *InitFD  = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6082   const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6083   const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy);
6084   const FieldDecl *FlagsFD = addFieldToRecordDecl(
6085       C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false));
6086   RD->completeDefinition();
6087   QualType RDType = C.getRecordType(RD);
6088   unsigned Size = Data.ReductionVars.size();
6089   llvm::APInt ArraySize(/*numBits=*/64, Size);
6090   QualType ArrayRDType = C.getConstantArrayType(
6091       RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0);
6092   // kmp_task_red_input_t .rd_input.[Size];
6093   Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input.");
6094   ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionOrigs,
6095                        Data.ReductionCopies, Data.ReductionOps);
6096   for (unsigned Cnt = 0; Cnt < Size; ++Cnt) {
6097     // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt];
6098     llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0),
6099                            llvm::ConstantInt::get(CGM.SizeTy, Cnt)};
6100     llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP(
6101         TaskRedInput.getElementType(), TaskRedInput.getPointer(), Idxs,
6102         /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc,
6103         ".rd_input.gep.");
6104     LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType);
6105     // ElemLVal.reduce_shar = &Shareds[Cnt];
6106     LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD);
6107     RCG.emitSharedOrigLValue(CGF, Cnt);
6108     llvm::Value *CastedShared =
6109         CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF));
6110     CGF.EmitStoreOfScalar(CastedShared, SharedLVal);
6111     // ElemLVal.reduce_orig = &Origs[Cnt];
6112     LValue OrigLVal = CGF.EmitLValueForField(ElemLVal, OrigFD);
6113     llvm::Value *CastedOrig =
6114         CGF.EmitCastToVoidPtr(RCG.getOrigLValue(Cnt).getPointer(CGF));
6115     CGF.EmitStoreOfScalar(CastedOrig, OrigLVal);
6116     RCG.emitAggregateType(CGF, Cnt);
6117     llvm::Value *SizeValInChars;
6118     llvm::Value *SizeVal;
6119     std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt);
6120     // We use delayed creation/initialization for VLAs and array sections. It is
6121     // required because runtime does not provide the way to pass the sizes of
6122     // VLAs/array sections to initializer/combiner/finalizer functions. Instead
6123     // threadprivate global variables are used to store these values and use
6124     // them in the functions.
6125     bool DelayedCreation = !!SizeVal;
6126     SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy,
6127                                                /*isSigned=*/false);
6128     LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD);
6129     CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal);
6130     // ElemLVal.reduce_init = init;
6131     LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD);
6132     llvm::Value *InitAddr =
6133         CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt));
6134     CGF.EmitStoreOfScalar(InitAddr, InitLVal);
6135     // ElemLVal.reduce_fini = fini;
6136     LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD);
6137     llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt);
6138     llvm::Value *FiniAddr = Fini
6139                                 ? CGF.EmitCastToVoidPtr(Fini)
6140                                 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy);
6141     CGF.EmitStoreOfScalar(FiniAddr, FiniLVal);
6142     // ElemLVal.reduce_comb = comb;
6143     LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD);
6144     llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction(
6145         CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt],
6146         RHSExprs[Cnt], Data.ReductionCopies[Cnt]));
6147     CGF.EmitStoreOfScalar(CombAddr, CombLVal);
6148     // ElemLVal.flags = 0;
6149     LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD);
6150     if (DelayedCreation) {
6151       CGF.EmitStoreOfScalar(
6152           llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true),
6153           FlagsLVal);
6154     } else
6155       CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF),
6156                                  FlagsLVal.getType());
6157   }
6158   if (Data.IsReductionWithTaskMod) {
6159     // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int
6160     // is_ws, int num, void *data);
6161     llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc);
6162     llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6163                                                   CGM.IntTy, /*isSigned=*/true);
6164     llvm::Value *Args[] = {
6165         IdentTLoc, GTid,
6166         llvm::ConstantInt::get(CGM.IntTy, Data.IsWorksharingReduction ? 1 : 0,
6167                                /*isSigned=*/true),
6168         llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6169         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6170             TaskRedInput.getPointer(), CGM.VoidPtrTy)};
6171     return CGF.EmitRuntimeCall(
6172         OMPBuilder.getOrCreateRuntimeFunction(
6173             CGM.getModule(), OMPRTL___kmpc_taskred_modifier_init),
6174         Args);
6175   }
6176   // Build call void *__kmpc_taskred_init(int gtid, int num_data, void *data);
6177   llvm::Value *Args[] = {
6178       CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy,
6179                                 /*isSigned=*/true),
6180       llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true),
6181       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(),
6182                                                       CGM.VoidPtrTy)};
6183   return CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
6184                                  CGM.getModule(), OMPRTL___kmpc_taskred_init),
6185                              Args);
6186 }
6187 
6188 void CGOpenMPRuntime::emitTaskReductionFini(CodeGenFunction &CGF,
6189                                             SourceLocation Loc,
6190                                             bool IsWorksharingReduction) {
6191   // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int
6192   // is_ws, int num, void *data);
6193   llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc);
6194   llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6195                                                 CGM.IntTy, /*isSigned=*/true);
6196   llvm::Value *Args[] = {IdentTLoc, GTid,
6197                          llvm::ConstantInt::get(CGM.IntTy,
6198                                                 IsWorksharingReduction ? 1 : 0,
6199                                                 /*isSigned=*/true)};
6200   (void)CGF.EmitRuntimeCall(
6201       OMPBuilder.getOrCreateRuntimeFunction(
6202           CGM.getModule(), OMPRTL___kmpc_task_reduction_modifier_fini),
6203       Args);
6204 }
6205 
6206 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
6207                                               SourceLocation Loc,
6208                                               ReductionCodeGen &RCG,
6209                                               unsigned N) {
6210   auto Sizes = RCG.getSizes(N);
6211   // Emit threadprivate global variable if the type is non-constant
6212   // (Sizes.second = nullptr).
6213   if (Sizes.second) {
6214     llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy,
6215                                                      /*isSigned=*/false);
6216     Address SizeAddr = getAddrOfArtificialThreadPrivate(
6217         CGF, CGM.getContext().getSizeType(),
6218         generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N)));
6219     CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false);
6220   }
6221 }
6222 
6223 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF,
6224                                               SourceLocation Loc,
6225                                               llvm::Value *ReductionsPtr,
6226                                               LValue SharedLVal) {
6227   // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void
6228   // *d);
6229   llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc),
6230                                                    CGM.IntTy,
6231                                                    /*isSigned=*/true),
6232                          ReductionsPtr,
6233                          CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6234                              SharedLVal.getPointer(CGF), CGM.VoidPtrTy)};
6235   return Address(
6236       CGF.EmitRuntimeCall(
6237           OMPBuilder.getOrCreateRuntimeFunction(
6238               CGM.getModule(), OMPRTL___kmpc_task_reduction_get_th_data),
6239           Args),
6240       CGF.Int8Ty, SharedLVal.getAlignment());
6241 }
6242 
6243 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, SourceLocation Loc,
6244                                        const OMPTaskDataTy &Data) {
6245   if (!CGF.HaveInsertPoint())
6246     return;
6247 
6248   if (CGF.CGM.getLangOpts().OpenMPIRBuilder && Data.Dependences.empty()) {
6249     // TODO: Need to support taskwait with dependences in the OpenMPIRBuilder.
6250     OMPBuilder.createTaskwait(CGF.Builder);
6251   } else {
6252     llvm::Value *ThreadID = getThreadID(CGF, Loc);
6253     llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc);
6254     auto &M = CGM.getModule();
6255     Address DependenciesArray = Address::invalid();
6256     llvm::Value *NumOfElements;
6257     std::tie(NumOfElements, DependenciesArray) =
6258         emitDependClause(CGF, Data.Dependences, Loc);
6259     llvm::Value *DepWaitTaskArgs[6];
6260     if (!Data.Dependences.empty()) {
6261       DepWaitTaskArgs[0] = UpLoc;
6262       DepWaitTaskArgs[1] = ThreadID;
6263       DepWaitTaskArgs[2] = NumOfElements;
6264       DepWaitTaskArgs[3] = DependenciesArray.getPointer();
6265       DepWaitTaskArgs[4] = CGF.Builder.getInt32(0);
6266       DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
6267 
6268       CodeGenFunction::RunCleanupsScope LocalScope(CGF);
6269 
6270       // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid,
6271       // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32
6272       // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info
6273       // is specified.
6274       CGF.EmitRuntimeCall(
6275           OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_wait_deps),
6276           DepWaitTaskArgs);
6277 
6278     } else {
6279 
6280       // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
6281       // global_tid);
6282       llvm::Value *Args[] = {UpLoc, ThreadID};
6283       // Ignore return result until untied tasks are supported.
6284       CGF.EmitRuntimeCall(
6285           OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_taskwait),
6286           Args);
6287     }
6288   }
6289 
6290   if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
6291     Region->emitUntiedSwitch(CGF);
6292 }
6293 
6294 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF,
6295                                            OpenMPDirectiveKind InnerKind,
6296                                            const RegionCodeGenTy &CodeGen,
6297                                            bool HasCancel) {
6298   if (!CGF.HaveInsertPoint())
6299     return;
6300   InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel,
6301                                  InnerKind != OMPD_critical &&
6302                                      InnerKind != OMPD_master &&
6303                                      InnerKind != OMPD_masked);
6304   CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
6305 }
6306 
6307 namespace {
6308 enum RTCancelKind {
6309   CancelNoreq = 0,
6310   CancelParallel = 1,
6311   CancelLoop = 2,
6312   CancelSections = 3,
6313   CancelTaskgroup = 4
6314 };
6315 } // anonymous namespace
6316 
6317 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) {
6318   RTCancelKind CancelKind = CancelNoreq;
6319   if (CancelRegion == OMPD_parallel)
6320     CancelKind = CancelParallel;
6321   else if (CancelRegion == OMPD_for)
6322     CancelKind = CancelLoop;
6323   else if (CancelRegion == OMPD_sections)
6324     CancelKind = CancelSections;
6325   else {
6326     assert(CancelRegion == OMPD_taskgroup);
6327     CancelKind = CancelTaskgroup;
6328   }
6329   return CancelKind;
6330 }
6331 
6332 void CGOpenMPRuntime::emitCancellationPointCall(
6333     CodeGenFunction &CGF, SourceLocation Loc,
6334     OpenMPDirectiveKind CancelRegion) {
6335   if (!CGF.HaveInsertPoint())
6336     return;
6337   // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32
6338   // global_tid, kmp_int32 cncl_kind);
6339   if (auto *OMPRegionInfo =
6340           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6341     // For 'cancellation point taskgroup', the task region info may not have a
6342     // cancel. This may instead happen in another adjacent task.
6343     if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) {
6344       llvm::Value *Args[] = {
6345           emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
6346           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6347       // Ignore return result until untied tasks are supported.
6348       llvm::Value *Result = CGF.EmitRuntimeCall(
6349           OMPBuilder.getOrCreateRuntimeFunction(
6350               CGM.getModule(), OMPRTL___kmpc_cancellationpoint),
6351           Args);
6352       // if (__kmpc_cancellationpoint()) {
6353       //   call i32 @__kmpc_cancel_barrier( // for parallel cancellation only
6354       //   exit from construct;
6355       // }
6356       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6357       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6358       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6359       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6360       CGF.EmitBlock(ExitBB);
6361       if (CancelRegion == OMPD_parallel)
6362         emitBarrierCall(CGF, Loc, OMPD_unknown, /*EmitChecks=*/false);
6363       // exit from construct;
6364       CodeGenFunction::JumpDest CancelDest =
6365           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6366       CGF.EmitBranchThroughCleanup(CancelDest);
6367       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6368     }
6369   }
6370 }
6371 
6372 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc,
6373                                      const Expr *IfCond,
6374                                      OpenMPDirectiveKind CancelRegion) {
6375   if (!CGF.HaveInsertPoint())
6376     return;
6377   // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid,
6378   // kmp_int32 cncl_kind);
6379   auto &M = CGM.getModule();
6380   if (auto *OMPRegionInfo =
6381           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
6382     auto &&ThenGen = [this, &M, Loc, CancelRegion,
6383                       OMPRegionInfo](CodeGenFunction &CGF, PrePostActionTy &) {
6384       CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime();
6385       llvm::Value *Args[] = {
6386           RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc),
6387           CGF.Builder.getInt32(getCancellationKind(CancelRegion))};
6388       // Ignore return result until untied tasks are supported.
6389       llvm::Value *Result = CGF.EmitRuntimeCall(
6390           OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_cancel), Args);
6391       // if (__kmpc_cancel()) {
6392       //   call i32 @__kmpc_cancel_barrier( // for parallel cancellation only
6393       //   exit from construct;
6394       // }
6395       llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit");
6396       llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue");
6397       llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result);
6398       CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB);
6399       CGF.EmitBlock(ExitBB);
6400       if (CancelRegion == OMPD_parallel)
6401         RT.emitBarrierCall(CGF, Loc, OMPD_unknown, /*EmitChecks=*/false);
6402       // exit from construct;
6403       CodeGenFunction::JumpDest CancelDest =
6404           CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind());
6405       CGF.EmitBranchThroughCleanup(CancelDest);
6406       CGF.EmitBlock(ContBB, /*IsFinished=*/true);
6407     };
6408     if (IfCond) {
6409       emitIfClause(CGF, IfCond, ThenGen,
6410                    [](CodeGenFunction &, PrePostActionTy &) {});
6411     } else {
6412       RegionCodeGenTy ThenRCG(ThenGen);
6413       ThenRCG(CGF);
6414     }
6415   }
6416 }
6417 
6418 namespace {
6419 /// Cleanup action for uses_allocators support.
6420 class OMPUsesAllocatorsActionTy final : public PrePostActionTy {
6421   ArrayRef<std::pair<const Expr *, const Expr *>> Allocators;
6422 
6423 public:
6424   OMPUsesAllocatorsActionTy(
6425       ArrayRef<std::pair<const Expr *, const Expr *>> Allocators)
6426       : Allocators(Allocators) {}
6427   void Enter(CodeGenFunction &CGF) override {
6428     if (!CGF.HaveInsertPoint())
6429       return;
6430     for (const auto &AllocatorData : Allocators) {
6431       CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsInit(
6432           CGF, AllocatorData.first, AllocatorData.second);
6433     }
6434   }
6435   void Exit(CodeGenFunction &CGF) override {
6436     if (!CGF.HaveInsertPoint())
6437       return;
6438     for (const auto &AllocatorData : Allocators) {
6439       CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsFini(CGF,
6440                                                         AllocatorData.first);
6441     }
6442   }
6443 };
6444 } // namespace
6445 
6446 void CGOpenMPRuntime::emitTargetOutlinedFunction(
6447     const OMPExecutableDirective &D, StringRef ParentName,
6448     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6449     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6450   assert(!ParentName.empty() && "Invalid target region parent name!");
6451   HasEmittedTargetRegion = true;
6452   SmallVector<std::pair<const Expr *, const Expr *>, 4> Allocators;
6453   for (const auto *C : D.getClausesOfKind<OMPUsesAllocatorsClause>()) {
6454     for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) {
6455       const OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I);
6456       if (!D.AllocatorTraits)
6457         continue;
6458       Allocators.emplace_back(D.Allocator, D.AllocatorTraits);
6459     }
6460   }
6461   OMPUsesAllocatorsActionTy UsesAllocatorAction(Allocators);
6462   CodeGen.setAction(UsesAllocatorAction);
6463   emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID,
6464                                    IsOffloadEntry, CodeGen);
6465 }
6466 
6467 void CGOpenMPRuntime::emitUsesAllocatorsInit(CodeGenFunction &CGF,
6468                                              const Expr *Allocator,
6469                                              const Expr *AllocatorTraits) {
6470   llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc());
6471   ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true);
6472   // Use default memspace handle.
6473   llvm::Value *MemSpaceHandle = llvm::ConstantPointerNull::get(CGF.VoidPtrTy);
6474   llvm::Value *NumTraits = llvm::ConstantInt::get(
6475       CGF.IntTy, cast<ConstantArrayType>(
6476                      AllocatorTraits->getType()->getAsArrayTypeUnsafe())
6477                      ->getSize()
6478                      .getLimitedValue());
6479   LValue AllocatorTraitsLVal = CGF.EmitLValue(AllocatorTraits);
6480   Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
6481       AllocatorTraitsLVal.getAddress(CGF), CGF.VoidPtrPtrTy, CGF.VoidPtrTy);
6482   AllocatorTraitsLVal = CGF.MakeAddrLValue(Addr, CGF.getContext().VoidPtrTy,
6483                                            AllocatorTraitsLVal.getBaseInfo(),
6484                                            AllocatorTraitsLVal.getTBAAInfo());
6485   llvm::Value *Traits =
6486       CGF.EmitLoadOfScalar(AllocatorTraitsLVal, AllocatorTraits->getExprLoc());
6487 
6488   llvm::Value *AllocatorVal =
6489       CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
6490                               CGM.getModule(), OMPRTL___kmpc_init_allocator),
6491                           {ThreadId, MemSpaceHandle, NumTraits, Traits});
6492   // Store to allocator.
6493   CGF.EmitVarDecl(*cast<VarDecl>(
6494       cast<DeclRefExpr>(Allocator->IgnoreParenImpCasts())->getDecl()));
6495   LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts());
6496   AllocatorVal =
6497       CGF.EmitScalarConversion(AllocatorVal, CGF.getContext().VoidPtrTy,
6498                                Allocator->getType(), Allocator->getExprLoc());
6499   CGF.EmitStoreOfScalar(AllocatorVal, AllocatorLVal);
6500 }
6501 
6502 void CGOpenMPRuntime::emitUsesAllocatorsFini(CodeGenFunction &CGF,
6503                                              const Expr *Allocator) {
6504   llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc());
6505   ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true);
6506   LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts());
6507   llvm::Value *AllocatorVal =
6508       CGF.EmitLoadOfScalar(AllocatorLVal, Allocator->getExprLoc());
6509   AllocatorVal = CGF.EmitScalarConversion(AllocatorVal, Allocator->getType(),
6510                                           CGF.getContext().VoidPtrTy,
6511                                           Allocator->getExprLoc());
6512   (void)CGF.EmitRuntimeCall(
6513       OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
6514                                             OMPRTL___kmpc_destroy_allocator),
6515       {ThreadId, AllocatorVal});
6516 }
6517 
6518 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper(
6519     const OMPExecutableDirective &D, StringRef ParentName,
6520     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
6521     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
6522   // Create a unique name for the entry function using the source location
6523   // information of the current target region. The name will be something like:
6524   //
6525   // __omp_offloading_DD_FFFF_PP_lBB
6526   //
6527   // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the
6528   // mangled name of the function that encloses the target region and BB is the
6529   // line number of the target region.
6530 
6531   const bool BuildOutlinedFn = CGM.getLangOpts().OpenMPIsDevice ||
6532                                !CGM.getLangOpts().OpenMPOffloadMandatory;
6533   unsigned DeviceID;
6534   unsigned FileID;
6535   unsigned Line;
6536   getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID,
6537                            Line);
6538   SmallString<64> EntryFnName;
6539   {
6540     llvm::raw_svector_ostream OS(EntryFnName);
6541     OS << "__omp_offloading" << llvm::format("_%x", DeviceID)
6542        << llvm::format("_%x_", FileID) << ParentName << "_l" << Line;
6543   }
6544 
6545   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
6546 
6547   CodeGenFunction CGF(CGM, true);
6548   CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName);
6549   CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6550 
6551   if (BuildOutlinedFn)
6552     OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc());
6553 
6554   // If this target outline function is not an offload entry, we don't need to
6555   // register it.
6556   if (!IsOffloadEntry)
6557     return;
6558 
6559   // The target region ID is used by the runtime library to identify the current
6560   // target region, so it only has to be unique and not necessarily point to
6561   // anything. It could be the pointer to the outlined function that implements
6562   // the target region, but we aren't using that so that the compiler doesn't
6563   // need to keep that, and could therefore inline the host function if proven
6564   // worthwhile during optimization. In the other hand, if emitting code for the
6565   // device, the ID has to be the function address so that it can retrieved from
6566   // the offloading entry and launched by the runtime library. We also mark the
6567   // outlined function to have external linkage in case we are emitting code for
6568   // the device, because these functions will be entry points to the device.
6569 
6570   if (CGM.getLangOpts().OpenMPIsDevice) {
6571     OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy);
6572     OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage);
6573     OutlinedFn->setDSOLocal(false);
6574     if (CGM.getTriple().isAMDGCN())
6575       OutlinedFn->setCallingConv(llvm::CallingConv::AMDGPU_KERNEL);
6576   } else {
6577     std::string Name = getName({EntryFnName, "region_id"});
6578     OutlinedFnID = new llvm::GlobalVariable(
6579         CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
6580         llvm::GlobalValue::WeakAnyLinkage,
6581         llvm::Constant::getNullValue(CGM.Int8Ty), Name);
6582   }
6583 
6584   // If we do not allow host fallback we still need a named address to use.
6585   llvm::Constant *TargetRegionEntryAddr = OutlinedFn;
6586   if (!BuildOutlinedFn) {
6587     assert(!CGM.getModule().getGlobalVariable(EntryFnName, true) &&
6588            "Named kernel already exists?");
6589     TargetRegionEntryAddr = new llvm::GlobalVariable(
6590         CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true,
6591         llvm::GlobalValue::InternalLinkage,
6592         llvm::Constant::getNullValue(CGM.Int8Ty), EntryFnName);
6593   }
6594 
6595   // Register the information for the entry associated with this target region.
6596   OffloadEntriesInfoManager.registerTargetRegionEntryInfo(
6597       DeviceID, FileID, ParentName, Line, TargetRegionEntryAddr, OutlinedFnID,
6598       OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion);
6599 
6600   // Add NumTeams and ThreadLimit attributes to the outlined GPU function
6601   int32_t DefaultValTeams = -1;
6602   getNumTeamsExprForTargetDirective(CGF, D, DefaultValTeams);
6603   if (DefaultValTeams > 0 && OutlinedFn) {
6604     OutlinedFn->addFnAttr("omp_target_num_teams",
6605                           std::to_string(DefaultValTeams));
6606   }
6607   int32_t DefaultValThreads = -1;
6608   getNumThreadsExprForTargetDirective(CGF, D, DefaultValThreads);
6609   if (DefaultValThreads > 0 && OutlinedFn) {
6610     OutlinedFn->addFnAttr("omp_target_thread_limit",
6611                           std::to_string(DefaultValThreads));
6612   }
6613 
6614   if (BuildOutlinedFn)
6615     CGM.getTargetCodeGenInfo().setTargetAttributes(nullptr, OutlinedFn, CGM);
6616 }
6617 
6618 /// Checks if the expression is constant or does not have non-trivial function
6619 /// calls.
6620 static bool isTrivial(ASTContext &Ctx, const Expr * E) {
6621   // We can skip constant expressions.
6622   // We can skip expressions with trivial calls or simple expressions.
6623   return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) ||
6624           !E->hasNonTrivialCall(Ctx)) &&
6625          !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true);
6626 }
6627 
6628 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx,
6629                                                     const Stmt *Body) {
6630   const Stmt *Child = Body->IgnoreContainers();
6631   while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) {
6632     Child = nullptr;
6633     for (const Stmt *S : C->body()) {
6634       if (const auto *E = dyn_cast<Expr>(S)) {
6635         if (isTrivial(Ctx, E))
6636           continue;
6637       }
6638       // Some of the statements can be ignored.
6639       if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) ||
6640           isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S))
6641         continue;
6642       // Analyze declarations.
6643       if (const auto *DS = dyn_cast<DeclStmt>(S)) {
6644         if (llvm::all_of(DS->decls(), [](const Decl *D) {
6645               if (isa<EmptyDecl>(D) || isa<DeclContext>(D) ||
6646                   isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) ||
6647                   isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) ||
6648                   isa<UsingDirectiveDecl>(D) ||
6649                   isa<OMPDeclareReductionDecl>(D) ||
6650                   isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D))
6651                 return true;
6652               const auto *VD = dyn_cast<VarDecl>(D);
6653               if (!VD)
6654                 return false;
6655               return VD->hasGlobalStorage() || !VD->isUsed();
6656             }))
6657           continue;
6658       }
6659       // Found multiple children - cannot get the one child only.
6660       if (Child)
6661         return nullptr;
6662       Child = S;
6663     }
6664     if (Child)
6665       Child = Child->IgnoreContainers();
6666   }
6667   return Child;
6668 }
6669 
6670 const Expr *CGOpenMPRuntime::getNumTeamsExprForTargetDirective(
6671     CodeGenFunction &CGF, const OMPExecutableDirective &D,
6672     int32_t &DefaultVal) {
6673 
6674   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6675   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6676          "Expected target-based executable directive.");
6677   switch (DirectiveKind) {
6678   case OMPD_target: {
6679     const auto *CS = D.getInnermostCapturedStmt();
6680     const auto *Body =
6681         CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
6682     const Stmt *ChildStmt =
6683         CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body);
6684     if (const auto *NestedDir =
6685             dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
6686       if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) {
6687         if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) {
6688           const Expr *NumTeams =
6689               NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6690           if (NumTeams->isIntegerConstantExpr(CGF.getContext()))
6691             if (auto Constant =
6692                     NumTeams->getIntegerConstantExpr(CGF.getContext()))
6693               DefaultVal = Constant->getExtValue();
6694           return NumTeams;
6695         }
6696         DefaultVal = 0;
6697         return nullptr;
6698       }
6699       if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) ||
6700           isOpenMPSimdDirective(NestedDir->getDirectiveKind())) {
6701         DefaultVal = 1;
6702         return nullptr;
6703       }
6704       DefaultVal = 1;
6705       return nullptr;
6706     }
6707     // A value of -1 is used to check if we need to emit no teams region
6708     DefaultVal = -1;
6709     return nullptr;
6710   }
6711   case OMPD_target_teams:
6712   case OMPD_target_teams_distribute:
6713   case OMPD_target_teams_distribute_simd:
6714   case OMPD_target_teams_distribute_parallel_for:
6715   case OMPD_target_teams_distribute_parallel_for_simd: {
6716     if (D.hasClausesOfKind<OMPNumTeamsClause>()) {
6717       const Expr *NumTeams =
6718           D.getSingleClause<OMPNumTeamsClause>()->getNumTeams();
6719       if (NumTeams->isIntegerConstantExpr(CGF.getContext()))
6720         if (auto Constant = NumTeams->getIntegerConstantExpr(CGF.getContext()))
6721           DefaultVal = Constant->getExtValue();
6722       return NumTeams;
6723     }
6724     DefaultVal = 0;
6725     return nullptr;
6726   }
6727   case OMPD_target_parallel:
6728   case OMPD_target_parallel_for:
6729   case OMPD_target_parallel_for_simd:
6730   case OMPD_target_simd:
6731     DefaultVal = 1;
6732     return nullptr;
6733   case OMPD_parallel:
6734   case OMPD_for:
6735   case OMPD_parallel_for:
6736   case OMPD_parallel_master:
6737   case OMPD_parallel_sections:
6738   case OMPD_for_simd:
6739   case OMPD_parallel_for_simd:
6740   case OMPD_cancel:
6741   case OMPD_cancellation_point:
6742   case OMPD_ordered:
6743   case OMPD_threadprivate:
6744   case OMPD_allocate:
6745   case OMPD_task:
6746   case OMPD_simd:
6747   case OMPD_tile:
6748   case OMPD_unroll:
6749   case OMPD_sections:
6750   case OMPD_section:
6751   case OMPD_single:
6752   case OMPD_master:
6753   case OMPD_critical:
6754   case OMPD_taskyield:
6755   case OMPD_barrier:
6756   case OMPD_taskwait:
6757   case OMPD_taskgroup:
6758   case OMPD_atomic:
6759   case OMPD_flush:
6760   case OMPD_depobj:
6761   case OMPD_scan:
6762   case OMPD_teams:
6763   case OMPD_target_data:
6764   case OMPD_target_exit_data:
6765   case OMPD_target_enter_data:
6766   case OMPD_distribute:
6767   case OMPD_distribute_simd:
6768   case OMPD_distribute_parallel_for:
6769   case OMPD_distribute_parallel_for_simd:
6770   case OMPD_teams_distribute:
6771   case OMPD_teams_distribute_simd:
6772   case OMPD_teams_distribute_parallel_for:
6773   case OMPD_teams_distribute_parallel_for_simd:
6774   case OMPD_target_update:
6775   case OMPD_declare_simd:
6776   case OMPD_declare_variant:
6777   case OMPD_begin_declare_variant:
6778   case OMPD_end_declare_variant:
6779   case OMPD_declare_target:
6780   case OMPD_end_declare_target:
6781   case OMPD_declare_reduction:
6782   case OMPD_declare_mapper:
6783   case OMPD_taskloop:
6784   case OMPD_taskloop_simd:
6785   case OMPD_master_taskloop:
6786   case OMPD_master_taskloop_simd:
6787   case OMPD_parallel_master_taskloop:
6788   case OMPD_parallel_master_taskloop_simd:
6789   case OMPD_requires:
6790   case OMPD_metadirective:
6791   case OMPD_unknown:
6792     break;
6793   default:
6794     break;
6795   }
6796   llvm_unreachable("Unexpected directive kind.");
6797 }
6798 
6799 llvm::Value *CGOpenMPRuntime::emitNumTeamsForTargetDirective(
6800     CodeGenFunction &CGF, const OMPExecutableDirective &D) {
6801   assert(!CGF.getLangOpts().OpenMPIsDevice &&
6802          "Clauses associated with the teams directive expected to be emitted "
6803          "only for the host!");
6804   CGBuilderTy &Bld = CGF.Builder;
6805   int32_t DefaultNT = -1;
6806   const Expr *NumTeams = getNumTeamsExprForTargetDirective(CGF, D, DefaultNT);
6807   if (NumTeams != nullptr) {
6808     OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6809 
6810     switch (DirectiveKind) {
6811     case OMPD_target: {
6812       const auto *CS = D.getInnermostCapturedStmt();
6813       CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6814       CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6815       llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(NumTeams,
6816                                                   /*IgnoreResultAssign*/ true);
6817       return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6818                              /*isSigned=*/true);
6819     }
6820     case OMPD_target_teams:
6821     case OMPD_target_teams_distribute:
6822     case OMPD_target_teams_distribute_simd:
6823     case OMPD_target_teams_distribute_parallel_for:
6824     case OMPD_target_teams_distribute_parallel_for_simd: {
6825       CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF);
6826       llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(NumTeams,
6827                                                   /*IgnoreResultAssign*/ true);
6828       return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty,
6829                              /*isSigned=*/true);
6830     }
6831     default:
6832       break;
6833     }
6834   } else if (DefaultNT == -1) {
6835     return nullptr;
6836   }
6837 
6838   return Bld.getInt32(DefaultNT);
6839 }
6840 
6841 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS,
6842                                   llvm::Value *DefaultThreadLimitVal) {
6843   const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
6844       CGF.getContext(), CS->getCapturedStmt());
6845   if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
6846     if (isOpenMPParallelDirective(Dir->getDirectiveKind())) {
6847       llvm::Value *NumThreads = nullptr;
6848       llvm::Value *CondVal = nullptr;
6849       // Handle if clause. If if clause present, the number of threads is
6850       // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
6851       if (Dir->hasClausesOfKind<OMPIfClause>()) {
6852         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6853         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6854         const OMPIfClause *IfClause = nullptr;
6855         for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) {
6856           if (C->getNameModifier() == OMPD_unknown ||
6857               C->getNameModifier() == OMPD_parallel) {
6858             IfClause = C;
6859             break;
6860           }
6861         }
6862         if (IfClause) {
6863           const Expr *Cond = IfClause->getCondition();
6864           bool Result;
6865           if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
6866             if (!Result)
6867               return CGF.Builder.getInt32(1);
6868           } else {
6869             CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange());
6870             if (const auto *PreInit =
6871                     cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) {
6872               for (const auto *I : PreInit->decls()) {
6873                 if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6874                   CGF.EmitVarDecl(cast<VarDecl>(*I));
6875                 } else {
6876                   CodeGenFunction::AutoVarEmission Emission =
6877                       CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6878                   CGF.EmitAutoVarCleanups(Emission);
6879                 }
6880               }
6881             }
6882             CondVal = CGF.EvaluateExprAsBool(Cond);
6883           }
6884         }
6885       }
6886       // Check the value of num_threads clause iff if clause was not specified
6887       // or is not evaluated to false.
6888       if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) {
6889         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
6890         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
6891         const auto *NumThreadsClause =
6892             Dir->getSingleClause<OMPNumThreadsClause>();
6893         CodeGenFunction::LexicalScope Scope(
6894             CGF, NumThreadsClause->getNumThreads()->getSourceRange());
6895         if (const auto *PreInit =
6896                 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) {
6897           for (const auto *I : PreInit->decls()) {
6898             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
6899               CGF.EmitVarDecl(cast<VarDecl>(*I));
6900             } else {
6901               CodeGenFunction::AutoVarEmission Emission =
6902                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
6903               CGF.EmitAutoVarCleanups(Emission);
6904             }
6905           }
6906         }
6907         NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads());
6908         NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty,
6909                                                /*isSigned=*/false);
6910         if (DefaultThreadLimitVal)
6911           NumThreads = CGF.Builder.CreateSelect(
6912               CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads),
6913               DefaultThreadLimitVal, NumThreads);
6914       } else {
6915         NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal
6916                                            : CGF.Builder.getInt32(0);
6917       }
6918       // Process condition of the if clause.
6919       if (CondVal) {
6920         NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads,
6921                                               CGF.Builder.getInt32(1));
6922       }
6923       return NumThreads;
6924     }
6925     if (isOpenMPSimdDirective(Dir->getDirectiveKind()))
6926       return CGF.Builder.getInt32(1);
6927     return DefaultThreadLimitVal;
6928   }
6929   return DefaultThreadLimitVal ? DefaultThreadLimitVal
6930                                : CGF.Builder.getInt32(0);
6931 }
6932 
6933 const Expr *CGOpenMPRuntime::getNumThreadsExprForTargetDirective(
6934     CodeGenFunction &CGF, const OMPExecutableDirective &D,
6935     int32_t &DefaultVal) {
6936   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
6937   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
6938          "Expected target-based executable directive.");
6939 
6940   switch (DirectiveKind) {
6941   case OMPD_target:
6942     // Teams have no clause thread_limit
6943     return nullptr;
6944   case OMPD_target_teams:
6945   case OMPD_target_teams_distribute:
6946     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
6947       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
6948       const Expr *ThreadLimit = ThreadLimitClause->getThreadLimit();
6949       if (ThreadLimit->isIntegerConstantExpr(CGF.getContext()))
6950         if (auto Constant =
6951                 ThreadLimit->getIntegerConstantExpr(CGF.getContext()))
6952           DefaultVal = Constant->getExtValue();
6953       return ThreadLimit;
6954     }
6955     return nullptr;
6956   case OMPD_target_parallel:
6957   case OMPD_target_parallel_for:
6958   case OMPD_target_parallel_for_simd:
6959   case OMPD_target_teams_distribute_parallel_for:
6960   case OMPD_target_teams_distribute_parallel_for_simd: {
6961     Expr *ThreadLimit = nullptr;
6962     Expr *NumThreads = nullptr;
6963     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
6964       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
6965       ThreadLimit = ThreadLimitClause->getThreadLimit();
6966       if (ThreadLimit->isIntegerConstantExpr(CGF.getContext()))
6967         if (auto Constant =
6968                 ThreadLimit->getIntegerConstantExpr(CGF.getContext()))
6969           DefaultVal = Constant->getExtValue();
6970     }
6971     if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
6972       const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
6973       NumThreads = NumThreadsClause->getNumThreads();
6974       if (NumThreads->isIntegerConstantExpr(CGF.getContext())) {
6975         if (auto Constant =
6976                 NumThreads->getIntegerConstantExpr(CGF.getContext())) {
6977           if (Constant->getExtValue() < DefaultVal) {
6978             DefaultVal = Constant->getExtValue();
6979             ThreadLimit = NumThreads;
6980           }
6981         }
6982       }
6983     }
6984     return ThreadLimit;
6985   }
6986   case OMPD_target_teams_distribute_simd:
6987   case OMPD_target_simd:
6988     DefaultVal = 1;
6989     return nullptr;
6990   case OMPD_parallel:
6991   case OMPD_for:
6992   case OMPD_parallel_for:
6993   case OMPD_parallel_master:
6994   case OMPD_parallel_sections:
6995   case OMPD_for_simd:
6996   case OMPD_parallel_for_simd:
6997   case OMPD_cancel:
6998   case OMPD_cancellation_point:
6999   case OMPD_ordered:
7000   case OMPD_threadprivate:
7001   case OMPD_allocate:
7002   case OMPD_task:
7003   case OMPD_simd:
7004   case OMPD_tile:
7005   case OMPD_unroll:
7006   case OMPD_sections:
7007   case OMPD_section:
7008   case OMPD_single:
7009   case OMPD_master:
7010   case OMPD_critical:
7011   case OMPD_taskyield:
7012   case OMPD_barrier:
7013   case OMPD_taskwait:
7014   case OMPD_taskgroup:
7015   case OMPD_atomic:
7016   case OMPD_flush:
7017   case OMPD_depobj:
7018   case OMPD_scan:
7019   case OMPD_teams:
7020   case OMPD_target_data:
7021   case OMPD_target_exit_data:
7022   case OMPD_target_enter_data:
7023   case OMPD_distribute:
7024   case OMPD_distribute_simd:
7025   case OMPD_distribute_parallel_for:
7026   case OMPD_distribute_parallel_for_simd:
7027   case OMPD_teams_distribute:
7028   case OMPD_teams_distribute_simd:
7029   case OMPD_teams_distribute_parallel_for:
7030   case OMPD_teams_distribute_parallel_for_simd:
7031   case OMPD_target_update:
7032   case OMPD_declare_simd:
7033   case OMPD_declare_variant:
7034   case OMPD_begin_declare_variant:
7035   case OMPD_end_declare_variant:
7036   case OMPD_declare_target:
7037   case OMPD_end_declare_target:
7038   case OMPD_declare_reduction:
7039   case OMPD_declare_mapper:
7040   case OMPD_taskloop:
7041   case OMPD_taskloop_simd:
7042   case OMPD_master_taskloop:
7043   case OMPD_master_taskloop_simd:
7044   case OMPD_parallel_master_taskloop:
7045   case OMPD_parallel_master_taskloop_simd:
7046   case OMPD_requires:
7047   case OMPD_unknown:
7048     break;
7049   default:
7050     break;
7051   }
7052   llvm_unreachable("Unsupported directive kind.");
7053 }
7054 
7055 llvm::Value *CGOpenMPRuntime::emitNumThreadsForTargetDirective(
7056     CodeGenFunction &CGF, const OMPExecutableDirective &D) {
7057   assert(!CGF.getLangOpts().OpenMPIsDevice &&
7058          "Clauses associated with the teams directive expected to be emitted "
7059          "only for the host!");
7060   OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind();
7061   assert(isOpenMPTargetExecutionDirective(DirectiveKind) &&
7062          "Expected target-based executable directive.");
7063   CGBuilderTy &Bld = CGF.Builder;
7064   llvm::Value *ThreadLimitVal = nullptr;
7065   llvm::Value *NumThreadsVal = nullptr;
7066   switch (DirectiveKind) {
7067   case OMPD_target: {
7068     const CapturedStmt *CS = D.getInnermostCapturedStmt();
7069     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7070       return NumThreads;
7071     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7072         CGF.getContext(), CS->getCapturedStmt());
7073     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7074       if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) {
7075         CGOpenMPInnerExprInfo CGInfo(CGF, *CS);
7076         CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo);
7077         const auto *ThreadLimitClause =
7078             Dir->getSingleClause<OMPThreadLimitClause>();
7079         CodeGenFunction::LexicalScope Scope(
7080             CGF, ThreadLimitClause->getThreadLimit()->getSourceRange());
7081         if (const auto *PreInit =
7082                 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) {
7083           for (const auto *I : PreInit->decls()) {
7084             if (!I->hasAttr<OMPCaptureNoInitAttr>()) {
7085               CGF.EmitVarDecl(cast<VarDecl>(*I));
7086             } else {
7087               CodeGenFunction::AutoVarEmission Emission =
7088                   CGF.EmitAutoVarAlloca(cast<VarDecl>(*I));
7089               CGF.EmitAutoVarCleanups(Emission);
7090             }
7091           }
7092         }
7093         llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7094             ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7095         ThreadLimitVal =
7096             Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7097       }
7098       if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) &&
7099           !isOpenMPDistributeDirective(Dir->getDirectiveKind())) {
7100         CS = Dir->getInnermostCapturedStmt();
7101         const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7102             CGF.getContext(), CS->getCapturedStmt());
7103         Dir = dyn_cast_or_null<OMPExecutableDirective>(Child);
7104       }
7105       if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) &&
7106           !isOpenMPSimdDirective(Dir->getDirectiveKind())) {
7107         CS = Dir->getInnermostCapturedStmt();
7108         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7109           return NumThreads;
7110       }
7111       if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind()))
7112         return Bld.getInt32(1);
7113     }
7114     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7115   }
7116   case OMPD_target_teams: {
7117     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7118       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7119       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7120       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7121           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7122       ThreadLimitVal =
7123           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7124     }
7125     const CapturedStmt *CS = D.getInnermostCapturedStmt();
7126     if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7127       return NumThreads;
7128     const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild(
7129         CGF.getContext(), CS->getCapturedStmt());
7130     if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) {
7131       if (Dir->getDirectiveKind() == OMPD_distribute) {
7132         CS = Dir->getInnermostCapturedStmt();
7133         if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal))
7134           return NumThreads;
7135       }
7136     }
7137     return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0);
7138   }
7139   case OMPD_target_teams_distribute:
7140     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7141       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7142       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7143       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7144           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7145       ThreadLimitVal =
7146           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7147     }
7148     return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal);
7149   case OMPD_target_parallel:
7150   case OMPD_target_parallel_for:
7151   case OMPD_target_parallel_for_simd:
7152   case OMPD_target_teams_distribute_parallel_for:
7153   case OMPD_target_teams_distribute_parallel_for_simd: {
7154     llvm::Value *CondVal = nullptr;
7155     // Handle if clause. If if clause present, the number of threads is
7156     // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1.
7157     if (D.hasClausesOfKind<OMPIfClause>()) {
7158       const OMPIfClause *IfClause = nullptr;
7159       for (const auto *C : D.getClausesOfKind<OMPIfClause>()) {
7160         if (C->getNameModifier() == OMPD_unknown ||
7161             C->getNameModifier() == OMPD_parallel) {
7162           IfClause = C;
7163           break;
7164         }
7165       }
7166       if (IfClause) {
7167         const Expr *Cond = IfClause->getCondition();
7168         bool Result;
7169         if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) {
7170           if (!Result)
7171             return Bld.getInt32(1);
7172         } else {
7173           CodeGenFunction::RunCleanupsScope Scope(CGF);
7174           CondVal = CGF.EvaluateExprAsBool(Cond);
7175         }
7176       }
7177     }
7178     if (D.hasClausesOfKind<OMPThreadLimitClause>()) {
7179       CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF);
7180       const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>();
7181       llvm::Value *ThreadLimit = CGF.EmitScalarExpr(
7182           ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true);
7183       ThreadLimitVal =
7184           Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false);
7185     }
7186     if (D.hasClausesOfKind<OMPNumThreadsClause>()) {
7187       CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
7188       const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>();
7189       llvm::Value *NumThreads = CGF.EmitScalarExpr(
7190           NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true);
7191       NumThreadsVal =
7192           Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false);
7193       ThreadLimitVal = ThreadLimitVal
7194                            ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal,
7195                                                                 ThreadLimitVal),
7196                                               NumThreadsVal, ThreadLimitVal)
7197                            : NumThreadsVal;
7198     }
7199     if (!ThreadLimitVal)
7200       ThreadLimitVal = Bld.getInt32(0);
7201     if (CondVal)
7202       return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1));
7203     return ThreadLimitVal;
7204   }
7205   case OMPD_target_teams_distribute_simd:
7206   case OMPD_target_simd:
7207     return Bld.getInt32(1);
7208   case OMPD_parallel:
7209   case OMPD_for:
7210   case OMPD_parallel_for:
7211   case OMPD_parallel_master:
7212   case OMPD_parallel_sections:
7213   case OMPD_for_simd:
7214   case OMPD_parallel_for_simd:
7215   case OMPD_cancel:
7216   case OMPD_cancellation_point:
7217   case OMPD_ordered:
7218   case OMPD_threadprivate:
7219   case OMPD_allocate:
7220   case OMPD_task:
7221   case OMPD_simd:
7222   case OMPD_tile:
7223   case OMPD_unroll:
7224   case OMPD_sections:
7225   case OMPD_section:
7226   case OMPD_single:
7227   case OMPD_master:
7228   case OMPD_critical:
7229   case OMPD_taskyield:
7230   case OMPD_barrier:
7231   case OMPD_taskwait:
7232   case OMPD_taskgroup:
7233   case OMPD_atomic:
7234   case OMPD_flush:
7235   case OMPD_depobj:
7236   case OMPD_scan:
7237   case OMPD_teams:
7238   case OMPD_target_data:
7239   case OMPD_target_exit_data:
7240   case OMPD_target_enter_data:
7241   case OMPD_distribute:
7242   case OMPD_distribute_simd:
7243   case OMPD_distribute_parallel_for:
7244   case OMPD_distribute_parallel_for_simd:
7245   case OMPD_teams_distribute:
7246   case OMPD_teams_distribute_simd:
7247   case OMPD_teams_distribute_parallel_for:
7248   case OMPD_teams_distribute_parallel_for_simd:
7249   case OMPD_target_update:
7250   case OMPD_declare_simd:
7251   case OMPD_declare_variant:
7252   case OMPD_begin_declare_variant:
7253   case OMPD_end_declare_variant:
7254   case OMPD_declare_target:
7255   case OMPD_end_declare_target:
7256   case OMPD_declare_reduction:
7257   case OMPD_declare_mapper:
7258   case OMPD_taskloop:
7259   case OMPD_taskloop_simd:
7260   case OMPD_master_taskloop:
7261   case OMPD_master_taskloop_simd:
7262   case OMPD_parallel_master_taskloop:
7263   case OMPD_parallel_master_taskloop_simd:
7264   case OMPD_requires:
7265   case OMPD_metadirective:
7266   case OMPD_unknown:
7267     break;
7268   default:
7269     break;
7270   }
7271   llvm_unreachable("Unsupported directive kind.");
7272 }
7273 
7274 namespace {
7275 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE();
7276 
7277 // Utility to handle information from clauses associated with a given
7278 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause).
7279 // It provides a convenient interface to obtain the information and generate
7280 // code for that information.
7281 class MappableExprsHandler {
7282 public:
7283   /// Values for bit flags used to specify the mapping type for
7284   /// offloading.
7285   enum OpenMPOffloadMappingFlags : uint64_t {
7286     /// No flags
7287     OMP_MAP_NONE = 0x0,
7288     /// Allocate memory on the device and move data from host to device.
7289     OMP_MAP_TO = 0x01,
7290     /// Allocate memory on the device and move data from device to host.
7291     OMP_MAP_FROM = 0x02,
7292     /// Always perform the requested mapping action on the element, even
7293     /// if it was already mapped before.
7294     OMP_MAP_ALWAYS = 0x04,
7295     /// Delete the element from the device environment, ignoring the
7296     /// current reference count associated with the element.
7297     OMP_MAP_DELETE = 0x08,
7298     /// The element being mapped is a pointer-pointee pair; both the
7299     /// pointer and the pointee should be mapped.
7300     OMP_MAP_PTR_AND_OBJ = 0x10,
7301     /// This flags signals that the base address of an entry should be
7302     /// passed to the target kernel as an argument.
7303     OMP_MAP_TARGET_PARAM = 0x20,
7304     /// Signal that the runtime library has to return the device pointer
7305     /// in the current position for the data being mapped. Used when we have the
7306     /// use_device_ptr or use_device_addr clause.
7307     OMP_MAP_RETURN_PARAM = 0x40,
7308     /// This flag signals that the reference being passed is a pointer to
7309     /// private data.
7310     OMP_MAP_PRIVATE = 0x80,
7311     /// Pass the element to the device by value.
7312     OMP_MAP_LITERAL = 0x100,
7313     /// Implicit map
7314     OMP_MAP_IMPLICIT = 0x200,
7315     /// Close is a hint to the runtime to allocate memory close to
7316     /// the target device.
7317     OMP_MAP_CLOSE = 0x400,
7318     /// 0x800 is reserved for compatibility with XLC.
7319     /// Produce a runtime error if the data is not already allocated.
7320     OMP_MAP_PRESENT = 0x1000,
7321     // Increment and decrement a separate reference counter so that the data
7322     // cannot be unmapped within the associated region.  Thus, this flag is
7323     // intended to be used on 'target' and 'target data' directives because they
7324     // are inherently structured.  It is not intended to be used on 'target
7325     // enter data' and 'target exit data' directives because they are inherently
7326     // dynamic.
7327     // This is an OpenMP extension for the sake of OpenACC support.
7328     OMP_MAP_OMPX_HOLD = 0x2000,
7329     /// Signal that the runtime library should use args as an array of
7330     /// descriptor_dim pointers and use args_size as dims. Used when we have
7331     /// non-contiguous list items in target update directive
7332     OMP_MAP_NON_CONTIG = 0x100000000000,
7333     /// The 16 MSBs of the flags indicate whether the entry is member of some
7334     /// struct/class.
7335     OMP_MAP_MEMBER_OF = 0xffff000000000000,
7336     LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF),
7337   };
7338 
7339   /// Get the offset of the OMP_MAP_MEMBER_OF field.
7340   static unsigned getFlagMemberOffset() {
7341     unsigned Offset = 0;
7342     for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1);
7343          Remain = Remain >> 1)
7344       Offset++;
7345     return Offset;
7346   }
7347 
7348   /// Class that holds debugging information for a data mapping to be passed to
7349   /// the runtime library.
7350   class MappingExprInfo {
7351     /// The variable declaration used for the data mapping.
7352     const ValueDecl *MapDecl = nullptr;
7353     /// The original expression used in the map clause, or null if there is
7354     /// none.
7355     const Expr *MapExpr = nullptr;
7356 
7357   public:
7358     MappingExprInfo(const ValueDecl *MapDecl, const Expr *MapExpr = nullptr)
7359         : MapDecl(MapDecl), MapExpr(MapExpr) {}
7360 
7361     const ValueDecl *getMapDecl() const { return MapDecl; }
7362     const Expr *getMapExpr() const { return MapExpr; }
7363   };
7364 
7365   /// Class that associates information with a base pointer to be passed to the
7366   /// runtime library.
7367   class BasePointerInfo {
7368     /// The base pointer.
7369     llvm::Value *Ptr = nullptr;
7370     /// The base declaration that refers to this device pointer, or null if
7371     /// there is none.
7372     const ValueDecl *DevPtrDecl = nullptr;
7373 
7374   public:
7375     BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr)
7376         : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {}
7377     llvm::Value *operator*() const { return Ptr; }
7378     const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; }
7379     void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; }
7380   };
7381 
7382   using MapExprsArrayTy = SmallVector<MappingExprInfo, 4>;
7383   using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>;
7384   using MapValuesArrayTy = SmallVector<llvm::Value *, 4>;
7385   using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>;
7386   using MapMappersArrayTy = SmallVector<const ValueDecl *, 4>;
7387   using MapDimArrayTy = SmallVector<uint64_t, 4>;
7388   using MapNonContiguousArrayTy = SmallVector<MapValuesArrayTy, 4>;
7389 
7390   /// This structure contains combined information generated for mappable
7391   /// clauses, including base pointers, pointers, sizes, map types, user-defined
7392   /// mappers, and non-contiguous information.
7393   struct MapCombinedInfoTy {
7394     struct StructNonContiguousInfo {
7395       bool IsNonContiguous = false;
7396       MapDimArrayTy Dims;
7397       MapNonContiguousArrayTy Offsets;
7398       MapNonContiguousArrayTy Counts;
7399       MapNonContiguousArrayTy Strides;
7400     };
7401     MapExprsArrayTy Exprs;
7402     MapBaseValuesArrayTy BasePointers;
7403     MapValuesArrayTy Pointers;
7404     MapValuesArrayTy Sizes;
7405     MapFlagsArrayTy Types;
7406     MapMappersArrayTy Mappers;
7407     StructNonContiguousInfo NonContigInfo;
7408 
7409     /// Append arrays in \a CurInfo.
7410     void append(MapCombinedInfoTy &CurInfo) {
7411       Exprs.append(CurInfo.Exprs.begin(), CurInfo.Exprs.end());
7412       BasePointers.append(CurInfo.BasePointers.begin(),
7413                           CurInfo.BasePointers.end());
7414       Pointers.append(CurInfo.Pointers.begin(), CurInfo.Pointers.end());
7415       Sizes.append(CurInfo.Sizes.begin(), CurInfo.Sizes.end());
7416       Types.append(CurInfo.Types.begin(), CurInfo.Types.end());
7417       Mappers.append(CurInfo.Mappers.begin(), CurInfo.Mappers.end());
7418       NonContigInfo.Dims.append(CurInfo.NonContigInfo.Dims.begin(),
7419                                  CurInfo.NonContigInfo.Dims.end());
7420       NonContigInfo.Offsets.append(CurInfo.NonContigInfo.Offsets.begin(),
7421                                     CurInfo.NonContigInfo.Offsets.end());
7422       NonContigInfo.Counts.append(CurInfo.NonContigInfo.Counts.begin(),
7423                                    CurInfo.NonContigInfo.Counts.end());
7424       NonContigInfo.Strides.append(CurInfo.NonContigInfo.Strides.begin(),
7425                                     CurInfo.NonContigInfo.Strides.end());
7426     }
7427   };
7428 
7429   /// Map between a struct and the its lowest & highest elements which have been
7430   /// mapped.
7431   /// [ValueDecl *] --> {LE(FieldIndex, Pointer),
7432   ///                    HE(FieldIndex, Pointer)}
7433   struct StructRangeInfoTy {
7434     MapCombinedInfoTy PreliminaryMapData;
7435     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = {
7436         0, Address::invalid()};
7437     std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = {
7438         0, Address::invalid()};
7439     Address Base = Address::invalid();
7440     Address LB = Address::invalid();
7441     bool IsArraySection = false;
7442     bool HasCompleteRecord = false;
7443   };
7444 
7445 private:
7446   /// Kind that defines how a device pointer has to be returned.
7447   struct MapInfo {
7448     OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
7449     OpenMPMapClauseKind MapType = OMPC_MAP_unknown;
7450     ArrayRef<OpenMPMapModifierKind> MapModifiers;
7451     ArrayRef<OpenMPMotionModifierKind> MotionModifiers;
7452     bool ReturnDevicePointer = false;
7453     bool IsImplicit = false;
7454     const ValueDecl *Mapper = nullptr;
7455     const Expr *VarRef = nullptr;
7456     bool ForDeviceAddr = false;
7457 
7458     MapInfo() = default;
7459     MapInfo(
7460         OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7461         OpenMPMapClauseKind MapType,
7462         ArrayRef<OpenMPMapModifierKind> MapModifiers,
7463         ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
7464         bool ReturnDevicePointer, bool IsImplicit,
7465         const ValueDecl *Mapper = nullptr, const Expr *VarRef = nullptr,
7466         bool ForDeviceAddr = false)
7467         : Components(Components), MapType(MapType), MapModifiers(MapModifiers),
7468           MotionModifiers(MotionModifiers),
7469           ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit),
7470           Mapper(Mapper), VarRef(VarRef), ForDeviceAddr(ForDeviceAddr) {}
7471   };
7472 
7473   /// If use_device_ptr or use_device_addr is used on a decl which is a struct
7474   /// member and there is no map information about it, then emission of that
7475   /// entry is deferred until the whole struct has been processed.
7476   struct DeferredDevicePtrEntryTy {
7477     const Expr *IE = nullptr;
7478     const ValueDecl *VD = nullptr;
7479     bool ForDeviceAddr = false;
7480 
7481     DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD,
7482                              bool ForDeviceAddr)
7483         : IE(IE), VD(VD), ForDeviceAddr(ForDeviceAddr) {}
7484   };
7485 
7486   /// The target directive from where the mappable clauses were extracted. It
7487   /// is either a executable directive or a user-defined mapper directive.
7488   llvm::PointerUnion<const OMPExecutableDirective *,
7489                      const OMPDeclareMapperDecl *>
7490       CurDir;
7491 
7492   /// Function the directive is being generated for.
7493   CodeGenFunction &CGF;
7494 
7495   /// Set of all first private variables in the current directive.
7496   /// bool data is set to true if the variable is implicitly marked as
7497   /// firstprivate, false otherwise.
7498   llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls;
7499 
7500   /// Map between device pointer declarations and their expression components.
7501   /// The key value for declarations in 'this' is null.
7502   llvm::DenseMap<
7503       const ValueDecl *,
7504       SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>>
7505       DevPointersMap;
7506 
7507   /// Map between lambda declarations and their map type.
7508   llvm::DenseMap<const ValueDecl *, const OMPMapClause *> LambdasMap;
7509 
7510   llvm::Value *getExprTypeSize(const Expr *E) const {
7511     QualType ExprTy = E->getType().getCanonicalType();
7512 
7513     // Calculate the size for array shaping expression.
7514     if (const auto *OAE = dyn_cast<OMPArrayShapingExpr>(E)) {
7515       llvm::Value *Size =
7516           CGF.getTypeSize(OAE->getBase()->getType()->getPointeeType());
7517       for (const Expr *SE : OAE->getDimensions()) {
7518         llvm::Value *Sz = CGF.EmitScalarExpr(SE);
7519         Sz = CGF.EmitScalarConversion(Sz, SE->getType(),
7520                                       CGF.getContext().getSizeType(),
7521                                       SE->getExprLoc());
7522         Size = CGF.Builder.CreateNUWMul(Size, Sz);
7523       }
7524       return Size;
7525     }
7526 
7527     // Reference types are ignored for mapping purposes.
7528     if (const auto *RefTy = ExprTy->getAs<ReferenceType>())
7529       ExprTy = RefTy->getPointeeType().getCanonicalType();
7530 
7531     // Given that an array section is considered a built-in type, we need to
7532     // do the calculation based on the length of the section instead of relying
7533     // on CGF.getTypeSize(E->getType()).
7534     if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) {
7535       QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType(
7536                             OAE->getBase()->IgnoreParenImpCasts())
7537                             .getCanonicalType();
7538 
7539       // If there is no length associated with the expression and lower bound is
7540       // not specified too, that means we are using the whole length of the
7541       // base.
7542       if (!OAE->getLength() && OAE->getColonLocFirst().isValid() &&
7543           !OAE->getLowerBound())
7544         return CGF.getTypeSize(BaseTy);
7545 
7546       llvm::Value *ElemSize;
7547       if (const auto *PTy = BaseTy->getAs<PointerType>()) {
7548         ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType());
7549       } else {
7550         const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr());
7551         assert(ATy && "Expecting array type if not a pointer type.");
7552         ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType());
7553       }
7554 
7555       // If we don't have a length at this point, that is because we have an
7556       // array section with a single element.
7557       if (!OAE->getLength() && OAE->getColonLocFirst().isInvalid())
7558         return ElemSize;
7559 
7560       if (const Expr *LenExpr = OAE->getLength()) {
7561         llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr);
7562         LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(),
7563                                              CGF.getContext().getSizeType(),
7564                                              LenExpr->getExprLoc());
7565         return CGF.Builder.CreateNUWMul(LengthVal, ElemSize);
7566       }
7567       assert(!OAE->getLength() && OAE->getColonLocFirst().isValid() &&
7568              OAE->getLowerBound() && "expected array_section[lb:].");
7569       // Size = sizetype - lb * elemtype;
7570       llvm::Value *LengthVal = CGF.getTypeSize(BaseTy);
7571       llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound());
7572       LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(),
7573                                        CGF.getContext().getSizeType(),
7574                                        OAE->getLowerBound()->getExprLoc());
7575       LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize);
7576       llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal);
7577       llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal);
7578       LengthVal = CGF.Builder.CreateSelect(
7579           Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0));
7580       return LengthVal;
7581     }
7582     return CGF.getTypeSize(ExprTy);
7583   }
7584 
7585   /// Return the corresponding bits for a given map clause modifier. Add
7586   /// a flag marking the map as a pointer if requested. Add a flag marking the
7587   /// map as the first one of a series of maps that relate to the same map
7588   /// expression.
7589   OpenMPOffloadMappingFlags getMapTypeBits(
7590       OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7591       ArrayRef<OpenMPMotionModifierKind> MotionModifiers, bool IsImplicit,
7592       bool AddPtrFlag, bool AddIsTargetParamFlag, bool IsNonContiguous) const {
7593     OpenMPOffloadMappingFlags Bits =
7594         IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE;
7595     switch (MapType) {
7596     case OMPC_MAP_alloc:
7597     case OMPC_MAP_release:
7598       // alloc and release is the default behavior in the runtime library,  i.e.
7599       // if we don't pass any bits alloc/release that is what the runtime is
7600       // going to do. Therefore, we don't need to signal anything for these two
7601       // type modifiers.
7602       break;
7603     case OMPC_MAP_to:
7604       Bits |= OMP_MAP_TO;
7605       break;
7606     case OMPC_MAP_from:
7607       Bits |= OMP_MAP_FROM;
7608       break;
7609     case OMPC_MAP_tofrom:
7610       Bits |= OMP_MAP_TO | OMP_MAP_FROM;
7611       break;
7612     case OMPC_MAP_delete:
7613       Bits |= OMP_MAP_DELETE;
7614       break;
7615     case OMPC_MAP_unknown:
7616       llvm_unreachable("Unexpected map type!");
7617     }
7618     if (AddPtrFlag)
7619       Bits |= OMP_MAP_PTR_AND_OBJ;
7620     if (AddIsTargetParamFlag)
7621       Bits |= OMP_MAP_TARGET_PARAM;
7622     if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_always))
7623       Bits |= OMP_MAP_ALWAYS;
7624     if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_close))
7625       Bits |= OMP_MAP_CLOSE;
7626     if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_present) ||
7627         llvm::is_contained(MotionModifiers, OMPC_MOTION_MODIFIER_present))
7628       Bits |= OMP_MAP_PRESENT;
7629     if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_ompx_hold))
7630       Bits |= OMP_MAP_OMPX_HOLD;
7631     if (IsNonContiguous)
7632       Bits |= OMP_MAP_NON_CONTIG;
7633     return Bits;
7634   }
7635 
7636   /// Return true if the provided expression is a final array section. A
7637   /// final array section, is one whose length can't be proved to be one.
7638   bool isFinalArraySectionExpression(const Expr *E) const {
7639     const auto *OASE = dyn_cast<OMPArraySectionExpr>(E);
7640 
7641     // It is not an array section and therefore not a unity-size one.
7642     if (!OASE)
7643       return false;
7644 
7645     // An array section with no colon always refer to a single element.
7646     if (OASE->getColonLocFirst().isInvalid())
7647       return false;
7648 
7649     const Expr *Length = OASE->getLength();
7650 
7651     // If we don't have a length we have to check if the array has size 1
7652     // for this dimension. Also, we should always expect a length if the
7653     // base type is pointer.
7654     if (!Length) {
7655       QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType(
7656                              OASE->getBase()->IgnoreParenImpCasts())
7657                              .getCanonicalType();
7658       if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr()))
7659         return ATy->getSize().getSExtValue() != 1;
7660       // If we don't have a constant dimension length, we have to consider
7661       // the current section as having any size, so it is not necessarily
7662       // unitary. If it happen to be unity size, that's user fault.
7663       return true;
7664     }
7665 
7666     // Check if the length evaluates to 1.
7667     Expr::EvalResult Result;
7668     if (!Length->EvaluateAsInt(Result, CGF.getContext()))
7669       return true; // Can have more that size 1.
7670 
7671     llvm::APSInt ConstLength = Result.Val.getInt();
7672     return ConstLength.getSExtValue() != 1;
7673   }
7674 
7675   /// Generate the base pointers, section pointers, sizes, map type bits, and
7676   /// user-defined mappers (all included in \a CombinedInfo) for the provided
7677   /// map type, map or motion modifiers, and expression components.
7678   /// \a IsFirstComponent should be set to true if the provided set of
7679   /// components is the first associated with a capture.
7680   void generateInfoForComponentList(
7681       OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers,
7682       ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
7683       OMPClauseMappableExprCommon::MappableExprComponentListRef Components,
7684       MapCombinedInfoTy &CombinedInfo, StructRangeInfoTy &PartialStruct,
7685       bool IsFirstComponentList, bool IsImplicit,
7686       const ValueDecl *Mapper = nullptr, bool ForDeviceAddr = false,
7687       const ValueDecl *BaseDecl = nullptr, const Expr *MapExpr = nullptr,
7688       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
7689           OverlappedElements = llvm::None) const {
7690     // The following summarizes what has to be generated for each map and the
7691     // types below. The generated information is expressed in this order:
7692     // base pointer, section pointer, size, flags
7693     // (to add to the ones that come from the map type and modifier).
7694     //
7695     // double d;
7696     // int i[100];
7697     // float *p;
7698     //
7699     // struct S1 {
7700     //   int i;
7701     //   float f[50];
7702     // }
7703     // struct S2 {
7704     //   int i;
7705     //   float f[50];
7706     //   S1 s;
7707     //   double *p;
7708     //   struct S2 *ps;
7709     //   int &ref;
7710     // }
7711     // S2 s;
7712     // S2 *ps;
7713     //
7714     // map(d)
7715     // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM
7716     //
7717     // map(i)
7718     // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM
7719     //
7720     // map(i[1:23])
7721     // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM
7722     //
7723     // map(p)
7724     // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM
7725     //
7726     // map(p[1:24])
7727     // &p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM | PTR_AND_OBJ
7728     // in unified shared memory mode or for local pointers
7729     // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM
7730     //
7731     // map(s)
7732     // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM
7733     //
7734     // map(s.i)
7735     // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM
7736     //
7737     // map(s.s.f)
7738     // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7739     //
7740     // map(s.p)
7741     // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM
7742     //
7743     // map(to: s.p[:22])
7744     // &s, &(s.p), sizeof(double*), TARGET_PARAM (*)
7745     // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**)
7746     // &(s.p), &(s.p[0]), 22*sizeof(double),
7747     //   MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
7748     // (*) alloc space for struct members, only this is a target parameter
7749     // (**) map the pointer (nothing to be mapped in this example) (the compiler
7750     //      optimizes this entry out, same in the examples below)
7751     // (***) map the pointee (map: to)
7752     //
7753     // map(to: s.ref)
7754     // &s, &(s.ref), sizeof(int*), TARGET_PARAM (*)
7755     // &s, &(s.ref), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | TO (***)
7756     // (*) alloc space for struct members, only this is a target parameter
7757     // (**) map the pointer (nothing to be mapped in this example) (the compiler
7758     //      optimizes this entry out, same in the examples below)
7759     // (***) map the pointee (map: to)
7760     //
7761     // map(s.ps)
7762     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7763     //
7764     // map(from: s.ps->s.i)
7765     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7766     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7767     // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ  | FROM
7768     //
7769     // map(to: s.ps->ps)
7770     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7771     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7772     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ  | TO
7773     //
7774     // map(s.ps->ps->ps)
7775     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7776     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7777     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7778     // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7779     //
7780     // map(to: s.ps->ps->s.f[:22])
7781     // &s, &(s.ps), sizeof(S2*), TARGET_PARAM
7782     // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1)
7783     // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7784     // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7785     //
7786     // map(ps)
7787     // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM
7788     //
7789     // map(ps->i)
7790     // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM
7791     //
7792     // map(ps->s.f)
7793     // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM
7794     //
7795     // map(from: ps->p)
7796     // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM
7797     //
7798     // map(to: ps->p[:22])
7799     // ps, &(ps->p), sizeof(double*), TARGET_PARAM
7800     // ps, &(ps->p), sizeof(double*), MEMBER_OF(1)
7801     // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO
7802     //
7803     // map(ps->ps)
7804     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM
7805     //
7806     // map(from: ps->ps->s.i)
7807     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7808     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7809     // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7810     //
7811     // map(from: ps->ps->ps)
7812     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7813     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7814     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7815     //
7816     // map(ps->ps->ps->ps)
7817     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7818     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7819     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7820     // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM
7821     //
7822     // map(to: ps->ps->ps->s.f[:22])
7823     // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM
7824     // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1)
7825     // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ
7826     // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO
7827     //
7828     // map(to: s.f[:22]) map(from: s.p[:33])
7829     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) +
7830     //     sizeof(double*) (**), TARGET_PARAM
7831     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO
7832     // &s, &(s.p), sizeof(double*), MEMBER_OF(1)
7833     // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM
7834     // (*) allocate contiguous space needed to fit all mapped members even if
7835     //     we allocate space for members not mapped (in this example,
7836     //     s.f[22..49] and s.s are not mapped, yet we must allocate space for
7837     //     them as well because they fall between &s.f[0] and &s.p)
7838     //
7839     // map(from: s.f[:22]) map(to: ps->p[:33])
7840     // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM
7841     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7842     // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*)
7843     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO
7844     // (*) the struct this entry pertains to is the 2nd element in the list of
7845     //     arguments, hence MEMBER_OF(2)
7846     //
7847     // map(from: s.f[:22], s.s) map(to: ps->p[:33])
7848     // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM
7849     // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM
7850     // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM
7851     // ps, &(ps->p), sizeof(S2*), TARGET_PARAM
7852     // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*)
7853     // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO
7854     // (*) the struct this entry pertains to is the 4th element in the list
7855     //     of arguments, hence MEMBER_OF(4)
7856 
7857     // Track if the map information being generated is the first for a capture.
7858     bool IsCaptureFirstInfo = IsFirstComponentList;
7859     // When the variable is on a declare target link or in a to clause with
7860     // unified memory, a reference is needed to hold the host/device address
7861     // of the variable.
7862     bool RequiresReference = false;
7863 
7864     // Scan the components from the base to the complete expression.
7865     auto CI = Components.rbegin();
7866     auto CE = Components.rend();
7867     auto I = CI;
7868 
7869     // Track if the map information being generated is the first for a list of
7870     // components.
7871     bool IsExpressionFirstInfo = true;
7872     bool FirstPointerInComplexData = false;
7873     Address BP = Address::invalid();
7874     const Expr *AssocExpr = I->getAssociatedExpression();
7875     const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr);
7876     const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr);
7877     const auto *OAShE = dyn_cast<OMPArrayShapingExpr>(AssocExpr);
7878 
7879     if (isa<MemberExpr>(AssocExpr)) {
7880       // The base is the 'this' pointer. The content of the pointer is going
7881       // to be the base of the field being mapped.
7882       BP = CGF.LoadCXXThisAddress();
7883     } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) ||
7884                (OASE &&
7885                 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) {
7886       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7887     } else if (OAShE &&
7888                isa<CXXThisExpr>(OAShE->getBase()->IgnoreParenCasts())) {
7889       BP = Address::deprecated(
7890           CGF.EmitScalarExpr(OAShE->getBase()),
7891           CGF.getContext().getTypeAlignInChars(OAShE->getBase()->getType()));
7892     } else {
7893       // The base is the reference to the variable.
7894       // BP = &Var.
7895       BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF);
7896       if (const auto *VD =
7897               dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) {
7898         if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
7899                 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) {
7900           if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
7901               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
7902                CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) {
7903             RequiresReference = true;
7904             BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
7905           }
7906         }
7907       }
7908 
7909       // If the variable is a pointer and is being dereferenced (i.e. is not
7910       // the last component), the base has to be the pointer itself, not its
7911       // reference. References are ignored for mapping purposes.
7912       QualType Ty =
7913           I->getAssociatedDeclaration()->getType().getNonReferenceType();
7914       if (Ty->isAnyPointerType() && std::next(I) != CE) {
7915         // No need to generate individual map information for the pointer, it
7916         // can be associated with the combined storage if shared memory mode is
7917         // active or the base declaration is not global variable.
7918         const auto *VD = dyn_cast<VarDecl>(I->getAssociatedDeclaration());
7919         if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
7920             !VD || VD->hasLocalStorage())
7921           BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
7922         else
7923           FirstPointerInComplexData = true;
7924         ++I;
7925       }
7926     }
7927 
7928     // Track whether a component of the list should be marked as MEMBER_OF some
7929     // combined entry (for partial structs). Only the first PTR_AND_OBJ entry
7930     // in a component list should be marked as MEMBER_OF, all subsequent entries
7931     // do not belong to the base struct. E.g.
7932     // struct S2 s;
7933     // s.ps->ps->ps->f[:]
7934     //   (1) (2) (3) (4)
7935     // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a
7936     // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3)
7937     // is the pointee of ps(2) which is not member of struct s, so it should not
7938     // be marked as such (it is still PTR_AND_OBJ).
7939     // The variable is initialized to false so that PTR_AND_OBJ entries which
7940     // are not struct members are not considered (e.g. array of pointers to
7941     // data).
7942     bool ShouldBeMemberOf = false;
7943 
7944     // Variable keeping track of whether or not we have encountered a component
7945     // in the component list which is a member expression. Useful when we have a
7946     // pointer or a final array section, in which case it is the previous
7947     // component in the list which tells us whether we have a member expression.
7948     // E.g. X.f[:]
7949     // While processing the final array section "[:]" it is "f" which tells us
7950     // whether we are dealing with a member of a declared struct.
7951     const MemberExpr *EncounteredME = nullptr;
7952 
7953     // Track for the total number of dimension. Start from one for the dummy
7954     // dimension.
7955     uint64_t DimSize = 1;
7956 
7957     bool IsNonContiguous = CombinedInfo.NonContigInfo.IsNonContiguous;
7958     bool IsPrevMemberReference = false;
7959 
7960     for (; I != CE; ++I) {
7961       // If the current component is member of a struct (parent struct) mark it.
7962       if (!EncounteredME) {
7963         EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression());
7964         // If we encounter a PTR_AND_OBJ entry from now on it should be marked
7965         // as MEMBER_OF the parent struct.
7966         if (EncounteredME) {
7967           ShouldBeMemberOf = true;
7968           // Do not emit as complex pointer if this is actually not array-like
7969           // expression.
7970           if (FirstPointerInComplexData) {
7971             QualType Ty = std::prev(I)
7972                               ->getAssociatedDeclaration()
7973                               ->getType()
7974                               .getNonReferenceType();
7975             BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
7976             FirstPointerInComplexData = false;
7977           }
7978         }
7979       }
7980 
7981       auto Next = std::next(I);
7982 
7983       // We need to generate the addresses and sizes if this is the last
7984       // component, if the component is a pointer or if it is an array section
7985       // whose length can't be proved to be one. If this is a pointer, it
7986       // becomes the base address for the following components.
7987 
7988       // A final array section, is one whose length can't be proved to be one.
7989       // If the map item is non-contiguous then we don't treat any array section
7990       // as final array section.
7991       bool IsFinalArraySection =
7992           !IsNonContiguous &&
7993           isFinalArraySectionExpression(I->getAssociatedExpression());
7994 
7995       // If we have a declaration for the mapping use that, otherwise use
7996       // the base declaration of the map clause.
7997       const ValueDecl *MapDecl = (I->getAssociatedDeclaration())
7998                                      ? I->getAssociatedDeclaration()
7999                                      : BaseDecl;
8000       MapExpr = (I->getAssociatedExpression()) ? I->getAssociatedExpression()
8001                                                : MapExpr;
8002 
8003       // Get information on whether the element is a pointer. Have to do a
8004       // special treatment for array sections given that they are built-in
8005       // types.
8006       const auto *OASE =
8007           dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression());
8008       const auto *OAShE =
8009           dyn_cast<OMPArrayShapingExpr>(I->getAssociatedExpression());
8010       const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression());
8011       const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression());
8012       bool IsPointer =
8013           OAShE ||
8014           (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE)
8015                        .getCanonicalType()
8016                        ->isAnyPointerType()) ||
8017           I->getAssociatedExpression()->getType()->isAnyPointerType();
8018       bool IsMemberReference = isa<MemberExpr>(I->getAssociatedExpression()) &&
8019                                MapDecl &&
8020                                MapDecl->getType()->isLValueReferenceType();
8021       bool IsNonDerefPointer = IsPointer && !UO && !BO && !IsNonContiguous;
8022 
8023       if (OASE)
8024         ++DimSize;
8025 
8026       if (Next == CE || IsMemberReference || IsNonDerefPointer ||
8027           IsFinalArraySection) {
8028         // If this is not the last component, we expect the pointer to be
8029         // associated with an array expression or member expression.
8030         assert((Next == CE ||
8031                 isa<MemberExpr>(Next->getAssociatedExpression()) ||
8032                 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) ||
8033                 isa<OMPArraySectionExpr>(Next->getAssociatedExpression()) ||
8034                 isa<OMPArrayShapingExpr>(Next->getAssociatedExpression()) ||
8035                 isa<UnaryOperator>(Next->getAssociatedExpression()) ||
8036                 isa<BinaryOperator>(Next->getAssociatedExpression())) &&
8037                "Unexpected expression");
8038 
8039         Address LB = Address::invalid();
8040         Address LowestElem = Address::invalid();
8041         auto &&EmitMemberExprBase = [](CodeGenFunction &CGF,
8042                                        const MemberExpr *E) {
8043           const Expr *BaseExpr = E->getBase();
8044           // If this is s.x, emit s as an lvalue.  If it is s->x, emit s as a
8045           // scalar.
8046           LValue BaseLV;
8047           if (E->isArrow()) {
8048             LValueBaseInfo BaseInfo;
8049             TBAAAccessInfo TBAAInfo;
8050             Address Addr =
8051                 CGF.EmitPointerWithAlignment(BaseExpr, &BaseInfo, &TBAAInfo);
8052             QualType PtrTy = BaseExpr->getType()->getPointeeType();
8053             BaseLV = CGF.MakeAddrLValue(Addr, PtrTy, BaseInfo, TBAAInfo);
8054           } else {
8055             BaseLV = CGF.EmitOMPSharedLValue(BaseExpr);
8056           }
8057           return BaseLV;
8058         };
8059         if (OAShE) {
8060           LowestElem = LB =
8061               Address::deprecated(CGF.EmitScalarExpr(OAShE->getBase()),
8062                                   CGF.getContext().getTypeAlignInChars(
8063                                       OAShE->getBase()->getType()));
8064         } else if (IsMemberReference) {
8065           const auto *ME = cast<MemberExpr>(I->getAssociatedExpression());
8066           LValue BaseLVal = EmitMemberExprBase(CGF, ME);
8067           LowestElem = CGF.EmitLValueForFieldInitialization(
8068                               BaseLVal, cast<FieldDecl>(MapDecl))
8069                            .getAddress(CGF);
8070           LB = CGF.EmitLoadOfReferenceLValue(LowestElem, MapDecl->getType())
8071                    .getAddress(CGF);
8072         } else {
8073           LowestElem = LB =
8074               CGF.EmitOMPSharedLValue(I->getAssociatedExpression())
8075                   .getAddress(CGF);
8076         }
8077 
8078         // If this component is a pointer inside the base struct then we don't
8079         // need to create any entry for it - it will be combined with the object
8080         // it is pointing to into a single PTR_AND_OBJ entry.
8081         bool IsMemberPointerOrAddr =
8082             EncounteredME &&
8083             (((IsPointer || ForDeviceAddr) &&
8084               I->getAssociatedExpression() == EncounteredME) ||
8085              (IsPrevMemberReference && !IsPointer) ||
8086              (IsMemberReference && Next != CE &&
8087               !Next->getAssociatedExpression()->getType()->isPointerType()));
8088         if (!OverlappedElements.empty() && Next == CE) {
8089           // Handle base element with the info for overlapped elements.
8090           assert(!PartialStruct.Base.isValid() && "The base element is set.");
8091           assert(!IsPointer &&
8092                  "Unexpected base element with the pointer type.");
8093           // Mark the whole struct as the struct that requires allocation on the
8094           // device.
8095           PartialStruct.LowestElem = {0, LowestElem};
8096           CharUnits TypeSize = CGF.getContext().getTypeSizeInChars(
8097               I->getAssociatedExpression()->getType());
8098           Address HB = CGF.Builder.CreateConstGEP(
8099               CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
8100                   LowestElem, CGF.VoidPtrTy, CGF.Int8Ty),
8101               TypeSize.getQuantity() - 1);
8102           PartialStruct.HighestElem = {
8103               std::numeric_limits<decltype(
8104                   PartialStruct.HighestElem.first)>::max(),
8105               HB};
8106           PartialStruct.Base = BP;
8107           PartialStruct.LB = LB;
8108           assert(
8109               PartialStruct.PreliminaryMapData.BasePointers.empty() &&
8110               "Overlapped elements must be used only once for the variable.");
8111           std::swap(PartialStruct.PreliminaryMapData, CombinedInfo);
8112           // Emit data for non-overlapped data.
8113           OpenMPOffloadMappingFlags Flags =
8114               OMP_MAP_MEMBER_OF |
8115               getMapTypeBits(MapType, MapModifiers, MotionModifiers, IsImplicit,
8116                              /*AddPtrFlag=*/false,
8117                              /*AddIsTargetParamFlag=*/false, IsNonContiguous);
8118           llvm::Value *Size = nullptr;
8119           // Do bitcopy of all non-overlapped structure elements.
8120           for (OMPClauseMappableExprCommon::MappableExprComponentListRef
8121                    Component : OverlappedElements) {
8122             Address ComponentLB = Address::invalid();
8123             for (const OMPClauseMappableExprCommon::MappableComponent &MC :
8124                  Component) {
8125               if (const ValueDecl *VD = MC.getAssociatedDeclaration()) {
8126                 const auto *FD = dyn_cast<FieldDecl>(VD);
8127                 if (FD && FD->getType()->isLValueReferenceType()) {
8128                   const auto *ME =
8129                       cast<MemberExpr>(MC.getAssociatedExpression());
8130                   LValue BaseLVal = EmitMemberExprBase(CGF, ME);
8131                   ComponentLB =
8132                       CGF.EmitLValueForFieldInitialization(BaseLVal, FD)
8133                           .getAddress(CGF);
8134                 } else {
8135                   ComponentLB =
8136                       CGF.EmitOMPSharedLValue(MC.getAssociatedExpression())
8137                           .getAddress(CGF);
8138                 }
8139                 Size = CGF.Builder.CreatePtrDiff(
8140                     CGF.Int8Ty, CGF.EmitCastToVoidPtr(ComponentLB.getPointer()),
8141                     CGF.EmitCastToVoidPtr(LB.getPointer()));
8142                 break;
8143               }
8144             }
8145             assert(Size && "Failed to determine structure size");
8146             CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr);
8147             CombinedInfo.BasePointers.push_back(BP.getPointer());
8148             CombinedInfo.Pointers.push_back(LB.getPointer());
8149             CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
8150                 Size, CGF.Int64Ty, /*isSigned=*/true));
8151             CombinedInfo.Types.push_back(Flags);
8152             CombinedInfo.Mappers.push_back(nullptr);
8153             CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize
8154                                                                       : 1);
8155             LB = CGF.Builder.CreateConstGEP(ComponentLB, 1);
8156           }
8157           CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr);
8158           CombinedInfo.BasePointers.push_back(BP.getPointer());
8159           CombinedInfo.Pointers.push_back(LB.getPointer());
8160           Size = CGF.Builder.CreatePtrDiff(
8161               CGF.Int8Ty, CGF.Builder.CreateConstGEP(HB, 1).getPointer(),
8162               CGF.EmitCastToVoidPtr(LB.getPointer()));
8163           CombinedInfo.Sizes.push_back(
8164               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
8165           CombinedInfo.Types.push_back(Flags);
8166           CombinedInfo.Mappers.push_back(nullptr);
8167           CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize
8168                                                                     : 1);
8169           break;
8170         }
8171         llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression());
8172         if (!IsMemberPointerOrAddr ||
8173             (Next == CE && MapType != OMPC_MAP_unknown)) {
8174           CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr);
8175           CombinedInfo.BasePointers.push_back(BP.getPointer());
8176           CombinedInfo.Pointers.push_back(LB.getPointer());
8177           CombinedInfo.Sizes.push_back(
8178               CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true));
8179           CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize
8180                                                                     : 1);
8181 
8182           // If Mapper is valid, the last component inherits the mapper.
8183           bool HasMapper = Mapper && Next == CE;
8184           CombinedInfo.Mappers.push_back(HasMapper ? Mapper : nullptr);
8185 
8186           // We need to add a pointer flag for each map that comes from the
8187           // same expression except for the first one. We also need to signal
8188           // this map is the first one that relates with the current capture
8189           // (there is a set of entries for each capture).
8190           OpenMPOffloadMappingFlags Flags = getMapTypeBits(
8191               MapType, MapModifiers, MotionModifiers, IsImplicit,
8192               !IsExpressionFirstInfo || RequiresReference ||
8193                   FirstPointerInComplexData || IsMemberReference,
8194               IsCaptureFirstInfo && !RequiresReference, IsNonContiguous);
8195 
8196           if (!IsExpressionFirstInfo || IsMemberReference) {
8197             // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well,
8198             // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags.
8199             if (IsPointer || (IsMemberReference && Next != CE))
8200               Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS |
8201                          OMP_MAP_DELETE | OMP_MAP_CLOSE);
8202 
8203             if (ShouldBeMemberOf) {
8204               // Set placeholder value MEMBER_OF=FFFF to indicate that the flag
8205               // should be later updated with the correct value of MEMBER_OF.
8206               Flags |= OMP_MAP_MEMBER_OF;
8207               // From now on, all subsequent PTR_AND_OBJ entries should not be
8208               // marked as MEMBER_OF.
8209               ShouldBeMemberOf = false;
8210             }
8211           }
8212 
8213           CombinedInfo.Types.push_back(Flags);
8214         }
8215 
8216         // If we have encountered a member expression so far, keep track of the
8217         // mapped member. If the parent is "*this", then the value declaration
8218         // is nullptr.
8219         if (EncounteredME) {
8220           const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl());
8221           unsigned FieldIndex = FD->getFieldIndex();
8222 
8223           // Update info about the lowest and highest elements for this struct
8224           if (!PartialStruct.Base.isValid()) {
8225             PartialStruct.LowestElem = {FieldIndex, LowestElem};
8226             if (IsFinalArraySection) {
8227               Address HB =
8228                   CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false)
8229                       .getAddress(CGF);
8230               PartialStruct.HighestElem = {FieldIndex, HB};
8231             } else {
8232               PartialStruct.HighestElem = {FieldIndex, LowestElem};
8233             }
8234             PartialStruct.Base = BP;
8235             PartialStruct.LB = BP;
8236           } else if (FieldIndex < PartialStruct.LowestElem.first) {
8237             PartialStruct.LowestElem = {FieldIndex, LowestElem};
8238           } else if (FieldIndex > PartialStruct.HighestElem.first) {
8239             PartialStruct.HighestElem = {FieldIndex, LowestElem};
8240           }
8241         }
8242 
8243         // Need to emit combined struct for array sections.
8244         if (IsFinalArraySection || IsNonContiguous)
8245           PartialStruct.IsArraySection = true;
8246 
8247         // If we have a final array section, we are done with this expression.
8248         if (IsFinalArraySection)
8249           break;
8250 
8251         // The pointer becomes the base for the next element.
8252         if (Next != CE)
8253           BP = IsMemberReference ? LowestElem : LB;
8254 
8255         IsExpressionFirstInfo = false;
8256         IsCaptureFirstInfo = false;
8257         FirstPointerInComplexData = false;
8258         IsPrevMemberReference = IsMemberReference;
8259       } else if (FirstPointerInComplexData) {
8260         QualType Ty = Components.rbegin()
8261                           ->getAssociatedDeclaration()
8262                           ->getType()
8263                           .getNonReferenceType();
8264         BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>());
8265         FirstPointerInComplexData = false;
8266       }
8267     }
8268     // If ran into the whole component - allocate the space for the whole
8269     // record.
8270     if (!EncounteredME)
8271       PartialStruct.HasCompleteRecord = true;
8272 
8273     if (!IsNonContiguous)
8274       return;
8275 
8276     const ASTContext &Context = CGF.getContext();
8277 
8278     // For supporting stride in array section, we need to initialize the first
8279     // dimension size as 1, first offset as 0, and first count as 1
8280     MapValuesArrayTy CurOffsets = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 0)};
8281     MapValuesArrayTy CurCounts = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)};
8282     MapValuesArrayTy CurStrides;
8283     MapValuesArrayTy DimSizes{llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)};
8284     uint64_t ElementTypeSize;
8285 
8286     // Collect Size information for each dimension and get the element size as
8287     // the first Stride. For example, for `int arr[10][10]`, the DimSizes
8288     // should be [10, 10] and the first stride is 4 btyes.
8289     for (const OMPClauseMappableExprCommon::MappableComponent &Component :
8290          Components) {
8291       const Expr *AssocExpr = Component.getAssociatedExpression();
8292       const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr);
8293 
8294       if (!OASE)
8295         continue;
8296 
8297       QualType Ty = OMPArraySectionExpr::getBaseOriginalType(OASE->getBase());
8298       auto *CAT = Context.getAsConstantArrayType(Ty);
8299       auto *VAT = Context.getAsVariableArrayType(Ty);
8300 
8301       // We need all the dimension size except for the last dimension.
8302       assert((VAT || CAT || &Component == &*Components.begin()) &&
8303              "Should be either ConstantArray or VariableArray if not the "
8304              "first Component");
8305 
8306       // Get element size if CurStrides is empty.
8307       if (CurStrides.empty()) {
8308         const Type *ElementType = nullptr;
8309         if (CAT)
8310           ElementType = CAT->getElementType().getTypePtr();
8311         else if (VAT)
8312           ElementType = VAT->getElementType().getTypePtr();
8313         else
8314           assert(&Component == &*Components.begin() &&
8315                  "Only expect pointer (non CAT or VAT) when this is the "
8316                  "first Component");
8317         // If ElementType is null, then it means the base is a pointer
8318         // (neither CAT nor VAT) and we'll attempt to get ElementType again
8319         // for next iteration.
8320         if (ElementType) {
8321           // For the case that having pointer as base, we need to remove one
8322           // level of indirection.
8323           if (&Component != &*Components.begin())
8324             ElementType = ElementType->getPointeeOrArrayElementType();
8325           ElementTypeSize =
8326               Context.getTypeSizeInChars(ElementType).getQuantity();
8327           CurStrides.push_back(
8328               llvm::ConstantInt::get(CGF.Int64Ty, ElementTypeSize));
8329         }
8330       }
8331       // Get dimension value except for the last dimension since we don't need
8332       // it.
8333       if (DimSizes.size() < Components.size() - 1) {
8334         if (CAT)
8335           DimSizes.push_back(llvm::ConstantInt::get(
8336               CGF.Int64Ty, CAT->getSize().getZExtValue()));
8337         else if (VAT)
8338           DimSizes.push_back(CGF.Builder.CreateIntCast(
8339               CGF.EmitScalarExpr(VAT->getSizeExpr()), CGF.Int64Ty,
8340               /*IsSigned=*/false));
8341       }
8342     }
8343 
8344     // Skip the dummy dimension since we have already have its information.
8345     auto *DI = DimSizes.begin() + 1;
8346     // Product of dimension.
8347     llvm::Value *DimProd =
8348         llvm::ConstantInt::get(CGF.CGM.Int64Ty, ElementTypeSize);
8349 
8350     // Collect info for non-contiguous. Notice that offset, count, and stride
8351     // are only meaningful for array-section, so we insert a null for anything
8352     // other than array-section.
8353     // Also, the size of offset, count, and stride are not the same as
8354     // pointers, base_pointers, sizes, or dims. Instead, the size of offset,
8355     // count, and stride are the same as the number of non-contiguous
8356     // declaration in target update to/from clause.
8357     for (const OMPClauseMappableExprCommon::MappableComponent &Component :
8358          Components) {
8359       const Expr *AssocExpr = Component.getAssociatedExpression();
8360 
8361       if (const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr)) {
8362         llvm::Value *Offset = CGF.Builder.CreateIntCast(
8363             CGF.EmitScalarExpr(AE->getIdx()), CGF.Int64Ty,
8364             /*isSigned=*/false);
8365         CurOffsets.push_back(Offset);
8366         CurCounts.push_back(llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/1));
8367         CurStrides.push_back(CurStrides.back());
8368         continue;
8369       }
8370 
8371       const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr);
8372 
8373       if (!OASE)
8374         continue;
8375 
8376       // Offset
8377       const Expr *OffsetExpr = OASE->getLowerBound();
8378       llvm::Value *Offset = nullptr;
8379       if (!OffsetExpr) {
8380         // If offset is absent, then we just set it to zero.
8381         Offset = llvm::ConstantInt::get(CGF.Int64Ty, 0);
8382       } else {
8383         Offset = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(OffsetExpr),
8384                                            CGF.Int64Ty,
8385                                            /*isSigned=*/false);
8386       }
8387       CurOffsets.push_back(Offset);
8388 
8389       // Count
8390       const Expr *CountExpr = OASE->getLength();
8391       llvm::Value *Count = nullptr;
8392       if (!CountExpr) {
8393         // In Clang, once a high dimension is an array section, we construct all
8394         // the lower dimension as array section, however, for case like
8395         // arr[0:2][2], Clang construct the inner dimension as an array section
8396         // but it actually is not in an array section form according to spec.
8397         if (!OASE->getColonLocFirst().isValid() &&
8398             !OASE->getColonLocSecond().isValid()) {
8399           Count = llvm::ConstantInt::get(CGF.Int64Ty, 1);
8400         } else {
8401           // OpenMP 5.0, 2.1.5 Array Sections, Description.
8402           // When the length is absent it defaults to ⌈(size −
8403           // lower-bound)/stride⌉, where size is the size of the array
8404           // dimension.
8405           const Expr *StrideExpr = OASE->getStride();
8406           llvm::Value *Stride =
8407               StrideExpr
8408                   ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr),
8409                                               CGF.Int64Ty, /*isSigned=*/false)
8410                   : nullptr;
8411           if (Stride)
8412             Count = CGF.Builder.CreateUDiv(
8413                 CGF.Builder.CreateNUWSub(*DI, Offset), Stride);
8414           else
8415             Count = CGF.Builder.CreateNUWSub(*DI, Offset);
8416         }
8417       } else {
8418         Count = CGF.EmitScalarExpr(CountExpr);
8419       }
8420       Count = CGF.Builder.CreateIntCast(Count, CGF.Int64Ty, /*isSigned=*/false);
8421       CurCounts.push_back(Count);
8422 
8423       // Stride_n' = Stride_n * (D_0 * D_1 ... * D_n-1) * Unit size
8424       // Take `int arr[5][5][5]` and `arr[0:2:2][1:2:1][0:2:2]` as an example:
8425       //              Offset      Count     Stride
8426       //    D0          0           1         4    (int)    <- dummy dimension
8427       //    D1          0           2         8    (2 * (1) * 4)
8428       //    D2          1           2         20   (1 * (1 * 5) * 4)
8429       //    D3          0           2         200  (2 * (1 * 5 * 4) * 4)
8430       const Expr *StrideExpr = OASE->getStride();
8431       llvm::Value *Stride =
8432           StrideExpr
8433               ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr),
8434                                           CGF.Int64Ty, /*isSigned=*/false)
8435               : nullptr;
8436       DimProd = CGF.Builder.CreateNUWMul(DimProd, *(DI - 1));
8437       if (Stride)
8438         CurStrides.push_back(CGF.Builder.CreateNUWMul(DimProd, Stride));
8439       else
8440         CurStrides.push_back(DimProd);
8441       if (DI != DimSizes.end())
8442         ++DI;
8443     }
8444 
8445     CombinedInfo.NonContigInfo.Offsets.push_back(CurOffsets);
8446     CombinedInfo.NonContigInfo.Counts.push_back(CurCounts);
8447     CombinedInfo.NonContigInfo.Strides.push_back(CurStrides);
8448   }
8449 
8450   /// Return the adjusted map modifiers if the declaration a capture refers to
8451   /// appears in a first-private clause. This is expected to be used only with
8452   /// directives that start with 'target'.
8453   MappableExprsHandler::OpenMPOffloadMappingFlags
8454   getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const {
8455     assert(Cap.capturesVariable() && "Expected capture by reference only!");
8456 
8457     // A first private variable captured by reference will use only the
8458     // 'private ptr' and 'map to' flag. Return the right flags if the captured
8459     // declaration is known as first-private in this handler.
8460     if (FirstPrivateDecls.count(Cap.getCapturedVar())) {
8461       if (Cap.getCapturedVar()->getType()->isAnyPointerType())
8462         return MappableExprsHandler::OMP_MAP_TO |
8463                MappableExprsHandler::OMP_MAP_PTR_AND_OBJ;
8464       return MappableExprsHandler::OMP_MAP_PRIVATE |
8465              MappableExprsHandler::OMP_MAP_TO;
8466     }
8467     auto I = LambdasMap.find(Cap.getCapturedVar()->getCanonicalDecl());
8468     if (I != LambdasMap.end())
8469       // for map(to: lambda): using user specified map type.
8470       return getMapTypeBits(
8471           I->getSecond()->getMapType(), I->getSecond()->getMapTypeModifiers(),
8472           /*MotionModifiers=*/llvm::None, I->getSecond()->isImplicit(),
8473           /*AddPtrFlag=*/false,
8474           /*AddIsTargetParamFlag=*/false,
8475           /*isNonContiguous=*/false);
8476     return MappableExprsHandler::OMP_MAP_TO |
8477            MappableExprsHandler::OMP_MAP_FROM;
8478   }
8479 
8480   static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) {
8481     // Rotate by getFlagMemberOffset() bits.
8482     return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1)
8483                                                   << getFlagMemberOffset());
8484   }
8485 
8486   static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags,
8487                                      OpenMPOffloadMappingFlags MemberOfFlag) {
8488     // If the entry is PTR_AND_OBJ but has not been marked with the special
8489     // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be
8490     // marked as MEMBER_OF.
8491     if ((Flags & OMP_MAP_PTR_AND_OBJ) &&
8492         ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF))
8493       return;
8494 
8495     // Reset the placeholder value to prepare the flag for the assignment of the
8496     // proper MEMBER_OF value.
8497     Flags &= ~OMP_MAP_MEMBER_OF;
8498     Flags |= MemberOfFlag;
8499   }
8500 
8501   void getPlainLayout(const CXXRecordDecl *RD,
8502                       llvm::SmallVectorImpl<const FieldDecl *> &Layout,
8503                       bool AsBase) const {
8504     const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD);
8505 
8506     llvm::StructType *St =
8507         AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType();
8508 
8509     unsigned NumElements = St->getNumElements();
8510     llvm::SmallVector<
8511         llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4>
8512         RecordLayout(NumElements);
8513 
8514     // Fill bases.
8515     for (const auto &I : RD->bases()) {
8516       if (I.isVirtual())
8517         continue;
8518       const auto *Base = I.getType()->getAsCXXRecordDecl();
8519       // Ignore empty bases.
8520       if (Base->isEmpty() || CGF.getContext()
8521                                  .getASTRecordLayout(Base)
8522                                  .getNonVirtualSize()
8523                                  .isZero())
8524         continue;
8525 
8526       unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base);
8527       RecordLayout[FieldIndex] = Base;
8528     }
8529     // Fill in virtual bases.
8530     for (const auto &I : RD->vbases()) {
8531       const auto *Base = I.getType()->getAsCXXRecordDecl();
8532       // Ignore empty bases.
8533       if (Base->isEmpty())
8534         continue;
8535       unsigned FieldIndex = RL.getVirtualBaseIndex(Base);
8536       if (RecordLayout[FieldIndex])
8537         continue;
8538       RecordLayout[FieldIndex] = Base;
8539     }
8540     // Fill in all the fields.
8541     assert(!RD->isUnion() && "Unexpected union.");
8542     for (const auto *Field : RD->fields()) {
8543       // Fill in non-bitfields. (Bitfields always use a zero pattern, which we
8544       // will fill in later.)
8545       if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) {
8546         unsigned FieldIndex = RL.getLLVMFieldNo(Field);
8547         RecordLayout[FieldIndex] = Field;
8548       }
8549     }
8550     for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>
8551              &Data : RecordLayout) {
8552       if (Data.isNull())
8553         continue;
8554       if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>())
8555         getPlainLayout(Base, Layout, /*AsBase=*/true);
8556       else
8557         Layout.push_back(Data.get<const FieldDecl *>());
8558     }
8559   }
8560 
8561   /// Generate all the base pointers, section pointers, sizes, map types, and
8562   /// mappers for the extracted mappable expressions (all included in \a
8563   /// CombinedInfo). Also, for each item that relates with a device pointer, a
8564   /// pair of the relevant declaration and index where it occurs is appended to
8565   /// the device pointers info array.
8566   void generateAllInfoForClauses(
8567       ArrayRef<const OMPClause *> Clauses, MapCombinedInfoTy &CombinedInfo,
8568       const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet =
8569           llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const {
8570     // We have to process the component lists that relate with the same
8571     // declaration in a single chunk so that we can generate the map flags
8572     // correctly. Therefore, we organize all lists in a map.
8573     enum MapKind { Present, Allocs, Other, Total };
8574     llvm::MapVector<CanonicalDeclPtr<const Decl>,
8575                     SmallVector<SmallVector<MapInfo, 8>, 4>>
8576         Info;
8577 
8578     // Helper function to fill the information map for the different supported
8579     // clauses.
8580     auto &&InfoGen =
8581         [&Info, &SkipVarSet](
8582             const ValueDecl *D, MapKind Kind,
8583             OMPClauseMappableExprCommon::MappableExprComponentListRef L,
8584             OpenMPMapClauseKind MapType,
8585             ArrayRef<OpenMPMapModifierKind> MapModifiers,
8586             ArrayRef<OpenMPMotionModifierKind> MotionModifiers,
8587             bool ReturnDevicePointer, bool IsImplicit, const ValueDecl *Mapper,
8588             const Expr *VarRef = nullptr, bool ForDeviceAddr = false) {
8589           if (SkipVarSet.contains(D))
8590             return;
8591           auto It = Info.find(D);
8592           if (It == Info.end())
8593             It = Info
8594                      .insert(std::make_pair(
8595                          D, SmallVector<SmallVector<MapInfo, 8>, 4>(Total)))
8596                      .first;
8597           It->second[Kind].emplace_back(
8598               L, MapType, MapModifiers, MotionModifiers, ReturnDevicePointer,
8599               IsImplicit, Mapper, VarRef, ForDeviceAddr);
8600         };
8601 
8602     for (const auto *Cl : Clauses) {
8603       const auto *C = dyn_cast<OMPMapClause>(Cl);
8604       if (!C)
8605         continue;
8606       MapKind Kind = Other;
8607       if (llvm::is_contained(C->getMapTypeModifiers(),
8608                              OMPC_MAP_MODIFIER_present))
8609         Kind = Present;
8610       else if (C->getMapType() == OMPC_MAP_alloc)
8611         Kind = Allocs;
8612       const auto *EI = C->getVarRefs().begin();
8613       for (const auto L : C->component_lists()) {
8614         const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr;
8615         InfoGen(std::get<0>(L), Kind, std::get<1>(L), C->getMapType(),
8616                 C->getMapTypeModifiers(), llvm::None,
8617                 /*ReturnDevicePointer=*/false, C->isImplicit(), std::get<2>(L),
8618                 E);
8619         ++EI;
8620       }
8621     }
8622     for (const auto *Cl : Clauses) {
8623       const auto *C = dyn_cast<OMPToClause>(Cl);
8624       if (!C)
8625         continue;
8626       MapKind Kind = Other;
8627       if (llvm::is_contained(C->getMotionModifiers(),
8628                              OMPC_MOTION_MODIFIER_present))
8629         Kind = Present;
8630       const auto *EI = C->getVarRefs().begin();
8631       for (const auto L : C->component_lists()) {
8632         InfoGen(std::get<0>(L), Kind, std::get<1>(L), OMPC_MAP_to, llvm::None,
8633                 C->getMotionModifiers(), /*ReturnDevicePointer=*/false,
8634                 C->isImplicit(), std::get<2>(L), *EI);
8635         ++EI;
8636       }
8637     }
8638     for (const auto *Cl : Clauses) {
8639       const auto *C = dyn_cast<OMPFromClause>(Cl);
8640       if (!C)
8641         continue;
8642       MapKind Kind = Other;
8643       if (llvm::is_contained(C->getMotionModifiers(),
8644                              OMPC_MOTION_MODIFIER_present))
8645         Kind = Present;
8646       const auto *EI = C->getVarRefs().begin();
8647       for (const auto L : C->component_lists()) {
8648         InfoGen(std::get<0>(L), Kind, std::get<1>(L), OMPC_MAP_from, llvm::None,
8649                 C->getMotionModifiers(), /*ReturnDevicePointer=*/false,
8650                 C->isImplicit(), std::get<2>(L), *EI);
8651         ++EI;
8652       }
8653     }
8654 
8655     // Look at the use_device_ptr clause information and mark the existing map
8656     // entries as such. If there is no map information for an entry in the
8657     // use_device_ptr list, we create one with map type 'alloc' and zero size
8658     // section. It is the user fault if that was not mapped before. If there is
8659     // no map information and the pointer is a struct member, then we defer the
8660     // emission of that entry until the whole struct has been processed.
8661     llvm::MapVector<CanonicalDeclPtr<const Decl>,
8662                     SmallVector<DeferredDevicePtrEntryTy, 4>>
8663         DeferredInfo;
8664     MapCombinedInfoTy UseDevicePtrCombinedInfo;
8665 
8666     for (const auto *Cl : Clauses) {
8667       const auto *C = dyn_cast<OMPUseDevicePtrClause>(Cl);
8668       if (!C)
8669         continue;
8670       for (const auto L : C->component_lists()) {
8671         OMPClauseMappableExprCommon::MappableExprComponentListRef Components =
8672             std::get<1>(L);
8673         assert(!Components.empty() &&
8674                "Not expecting empty list of components!");
8675         const ValueDecl *VD = Components.back().getAssociatedDeclaration();
8676         VD = cast<ValueDecl>(VD->getCanonicalDecl());
8677         const Expr *IE = Components.back().getAssociatedExpression();
8678         // If the first component is a member expression, we have to look into
8679         // 'this', which maps to null in the map of map information. Otherwise
8680         // look directly for the information.
8681         auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD);
8682 
8683         // We potentially have map information for this declaration already.
8684         // Look for the first set of components that refer to it.
8685         if (It != Info.end()) {
8686           bool Found = false;
8687           for (auto &Data : It->second) {
8688             auto *CI = llvm::find_if(Data, [VD](const MapInfo &MI) {
8689               return MI.Components.back().getAssociatedDeclaration() == VD;
8690             });
8691             // If we found a map entry, signal that the pointer has to be
8692             // returned and move on to the next declaration. Exclude cases where
8693             // the base pointer is mapped as array subscript, array section or
8694             // array shaping. The base address is passed as a pointer to base in
8695             // this case and cannot be used as a base for use_device_ptr list
8696             // item.
8697             if (CI != Data.end()) {
8698               auto PrevCI = std::next(CI->Components.rbegin());
8699               const auto *VarD = dyn_cast<VarDecl>(VD);
8700               if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() ||
8701                   isa<MemberExpr>(IE) ||
8702                   !VD->getType().getNonReferenceType()->isPointerType() ||
8703                   PrevCI == CI->Components.rend() ||
8704                   isa<MemberExpr>(PrevCI->getAssociatedExpression()) || !VarD ||
8705                   VarD->hasLocalStorage()) {
8706                 CI->ReturnDevicePointer = true;
8707                 Found = true;
8708                 break;
8709               }
8710             }
8711           }
8712           if (Found)
8713             continue;
8714         }
8715 
8716         // We didn't find any match in our map information - generate a zero
8717         // size array section - if the pointer is a struct member we defer this
8718         // action until the whole struct has been processed.
8719         if (isa<MemberExpr>(IE)) {
8720           // Insert the pointer into Info to be processed by
8721           // generateInfoForComponentList. Because it is a member pointer
8722           // without a pointee, no entry will be generated for it, therefore
8723           // we need to generate one after the whole struct has been processed.
8724           // Nonetheless, generateInfoForComponentList must be called to take
8725           // the pointer into account for the calculation of the range of the
8726           // partial struct.
8727           InfoGen(nullptr, Other, Components, OMPC_MAP_unknown, llvm::None,
8728                   llvm::None, /*ReturnDevicePointer=*/false, C->isImplicit(),
8729                   nullptr);
8730           DeferredInfo[nullptr].emplace_back(IE, VD, /*ForDeviceAddr=*/false);
8731         } else {
8732           llvm::Value *Ptr =
8733               CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc());
8734           UseDevicePtrCombinedInfo.Exprs.push_back(VD);
8735           UseDevicePtrCombinedInfo.BasePointers.emplace_back(Ptr, VD);
8736           UseDevicePtrCombinedInfo.Pointers.push_back(Ptr);
8737           UseDevicePtrCombinedInfo.Sizes.push_back(
8738               llvm::Constant::getNullValue(CGF.Int64Ty));
8739           UseDevicePtrCombinedInfo.Types.push_back(OMP_MAP_RETURN_PARAM);
8740           UseDevicePtrCombinedInfo.Mappers.push_back(nullptr);
8741         }
8742       }
8743     }
8744 
8745     // Look at the use_device_addr clause information and mark the existing map
8746     // entries as such. If there is no map information for an entry in the
8747     // use_device_addr list, we create one with map type 'alloc' and zero size
8748     // section. It is the user fault if that was not mapped before. If there is
8749     // no map information and the pointer is a struct member, then we defer the
8750     // emission of that entry until the whole struct has been processed.
8751     llvm::SmallDenseSet<CanonicalDeclPtr<const Decl>, 4> Processed;
8752     for (const auto *Cl : Clauses) {
8753       const auto *C = dyn_cast<OMPUseDeviceAddrClause>(Cl);
8754       if (!C)
8755         continue;
8756       for (const auto L : C->component_lists()) {
8757         assert(!std::get<1>(L).empty() &&
8758                "Not expecting empty list of components!");
8759         const ValueDecl *VD = std::get<1>(L).back().getAssociatedDeclaration();
8760         if (!Processed.insert(VD).second)
8761           continue;
8762         VD = cast<ValueDecl>(VD->getCanonicalDecl());
8763         const Expr *IE = std::get<1>(L).back().getAssociatedExpression();
8764         // If the first component is a member expression, we have to look into
8765         // 'this', which maps to null in the map of map information. Otherwise
8766         // look directly for the information.
8767         auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD);
8768 
8769         // We potentially have map information for this declaration already.
8770         // Look for the first set of components that refer to it.
8771         if (It != Info.end()) {
8772           bool Found = false;
8773           for (auto &Data : It->second) {
8774             auto *CI = llvm::find_if(Data, [VD](const MapInfo &MI) {
8775               return MI.Components.back().getAssociatedDeclaration() == VD;
8776             });
8777             // If we found a map entry, signal that the pointer has to be
8778             // returned and move on to the next declaration.
8779             if (CI != Data.end()) {
8780               CI->ReturnDevicePointer = true;
8781               Found = true;
8782               break;
8783             }
8784           }
8785           if (Found)
8786             continue;
8787         }
8788 
8789         // We didn't find any match in our map information - generate a zero
8790         // size array section - if the pointer is a struct member we defer this
8791         // action until the whole struct has been processed.
8792         if (isa<MemberExpr>(IE)) {
8793           // Insert the pointer into Info to be processed by
8794           // generateInfoForComponentList. Because it is a member pointer
8795           // without a pointee, no entry will be generated for it, therefore
8796           // we need to generate one after the whole struct has been processed.
8797           // Nonetheless, generateInfoForComponentList must be called to take
8798           // the pointer into account for the calculation of the range of the
8799           // partial struct.
8800           InfoGen(nullptr, Other, std::get<1>(L), OMPC_MAP_unknown, llvm::None,
8801                   llvm::None, /*ReturnDevicePointer=*/false, C->isImplicit(),
8802                   nullptr, nullptr, /*ForDeviceAddr=*/true);
8803           DeferredInfo[nullptr].emplace_back(IE, VD, /*ForDeviceAddr=*/true);
8804         } else {
8805           llvm::Value *Ptr;
8806           if (IE->isGLValue())
8807             Ptr = CGF.EmitLValue(IE).getPointer(CGF);
8808           else
8809             Ptr = CGF.EmitScalarExpr(IE);
8810           CombinedInfo.Exprs.push_back(VD);
8811           CombinedInfo.BasePointers.emplace_back(Ptr, VD);
8812           CombinedInfo.Pointers.push_back(Ptr);
8813           CombinedInfo.Sizes.push_back(
8814               llvm::Constant::getNullValue(CGF.Int64Ty));
8815           CombinedInfo.Types.push_back(OMP_MAP_RETURN_PARAM);
8816           CombinedInfo.Mappers.push_back(nullptr);
8817         }
8818       }
8819     }
8820 
8821     for (const auto &Data : Info) {
8822       StructRangeInfoTy PartialStruct;
8823       // Temporary generated information.
8824       MapCombinedInfoTy CurInfo;
8825       const Decl *D = Data.first;
8826       const ValueDecl *VD = cast_or_null<ValueDecl>(D);
8827       for (const auto &M : Data.second) {
8828         for (const MapInfo &L : M) {
8829           assert(!L.Components.empty() &&
8830                  "Not expecting declaration with no component lists.");
8831 
8832           // Remember the current base pointer index.
8833           unsigned CurrentBasePointersIdx = CurInfo.BasePointers.size();
8834           CurInfo.NonContigInfo.IsNonContiguous =
8835               L.Components.back().isNonContiguous();
8836           generateInfoForComponentList(
8837               L.MapType, L.MapModifiers, L.MotionModifiers, L.Components,
8838               CurInfo, PartialStruct, /*IsFirstComponentList=*/false,
8839               L.IsImplicit, L.Mapper, L.ForDeviceAddr, VD, L.VarRef);
8840 
8841           // If this entry relates with a device pointer, set the relevant
8842           // declaration and add the 'return pointer' flag.
8843           if (L.ReturnDevicePointer) {
8844             assert(CurInfo.BasePointers.size() > CurrentBasePointersIdx &&
8845                    "Unexpected number of mapped base pointers.");
8846 
8847             const ValueDecl *RelevantVD =
8848                 L.Components.back().getAssociatedDeclaration();
8849             assert(RelevantVD &&
8850                    "No relevant declaration related with device pointer??");
8851 
8852             CurInfo.BasePointers[CurrentBasePointersIdx].setDevicePtrDecl(
8853                 RelevantVD);
8854             CurInfo.Types[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM;
8855           }
8856         }
8857       }
8858 
8859       // Append any pending zero-length pointers which are struct members and
8860       // used with use_device_ptr or use_device_addr.
8861       auto CI = DeferredInfo.find(Data.first);
8862       if (CI != DeferredInfo.end()) {
8863         for (const DeferredDevicePtrEntryTy &L : CI->second) {
8864           llvm::Value *BasePtr;
8865           llvm::Value *Ptr;
8866           if (L.ForDeviceAddr) {
8867             if (L.IE->isGLValue())
8868               Ptr = this->CGF.EmitLValue(L.IE).getPointer(CGF);
8869             else
8870               Ptr = this->CGF.EmitScalarExpr(L.IE);
8871             BasePtr = Ptr;
8872             // Entry is RETURN_PARAM. Also, set the placeholder value
8873             // MEMBER_OF=FFFF so that the entry is later updated with the
8874             // correct value of MEMBER_OF.
8875             CurInfo.Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_MEMBER_OF);
8876           } else {
8877             BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF);
8878             Ptr = this->CGF.EmitLoadOfScalar(this->CGF.EmitLValue(L.IE),
8879                                              L.IE->getExprLoc());
8880             // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the
8881             // placeholder value MEMBER_OF=FFFF so that the entry is later
8882             // updated with the correct value of MEMBER_OF.
8883             CurInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM |
8884                                     OMP_MAP_MEMBER_OF);
8885           }
8886           CurInfo.Exprs.push_back(L.VD);
8887           CurInfo.BasePointers.emplace_back(BasePtr, L.VD);
8888           CurInfo.Pointers.push_back(Ptr);
8889           CurInfo.Sizes.push_back(
8890               llvm::Constant::getNullValue(this->CGF.Int64Ty));
8891           CurInfo.Mappers.push_back(nullptr);
8892         }
8893       }
8894       // If there is an entry in PartialStruct it means we have a struct with
8895       // individual members mapped. Emit an extra combined entry.
8896       if (PartialStruct.Base.isValid()) {
8897         CurInfo.NonContigInfo.Dims.push_back(0);
8898         emitCombinedEntry(CombinedInfo, CurInfo.Types, PartialStruct, VD);
8899       }
8900 
8901       // We need to append the results of this capture to what we already
8902       // have.
8903       CombinedInfo.append(CurInfo);
8904     }
8905     // Append data for use_device_ptr clauses.
8906     CombinedInfo.append(UseDevicePtrCombinedInfo);
8907   }
8908 
8909 public:
8910   MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF)
8911       : CurDir(&Dir), CGF(CGF) {
8912     // Extract firstprivate clause information.
8913     for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>())
8914       for (const auto *D : C->varlists())
8915         FirstPrivateDecls.try_emplace(
8916             cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit());
8917     // Extract implicit firstprivates from uses_allocators clauses.
8918     for (const auto *C : Dir.getClausesOfKind<OMPUsesAllocatorsClause>()) {
8919       for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) {
8920         OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I);
8921         if (const auto *DRE = dyn_cast_or_null<DeclRefExpr>(D.AllocatorTraits))
8922           FirstPrivateDecls.try_emplace(cast<VarDecl>(DRE->getDecl()),
8923                                         /*Implicit=*/true);
8924         else if (const auto *VD = dyn_cast<VarDecl>(
8925                      cast<DeclRefExpr>(D.Allocator->IgnoreParenImpCasts())
8926                          ->getDecl()))
8927           FirstPrivateDecls.try_emplace(VD, /*Implicit=*/true);
8928       }
8929     }
8930     // Extract device pointer clause information.
8931     for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>())
8932       for (auto L : C->component_lists())
8933         DevPointersMap[std::get<0>(L)].push_back(std::get<1>(L));
8934     // Extract map information.
8935     for (const auto *C : Dir.getClausesOfKind<OMPMapClause>()) {
8936       if (C->getMapType() != OMPC_MAP_to)
8937         continue;
8938       for (auto L : C->component_lists()) {
8939         const ValueDecl *VD = std::get<0>(L);
8940         const auto *RD = VD ? VD->getType()
8941                                   .getCanonicalType()
8942                                   .getNonReferenceType()
8943                                   ->getAsCXXRecordDecl()
8944                             : nullptr;
8945         if (RD && RD->isLambda())
8946           LambdasMap.try_emplace(std::get<0>(L), C);
8947       }
8948     }
8949   }
8950 
8951   /// Constructor for the declare mapper directive.
8952   MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF)
8953       : CurDir(&Dir), CGF(CGF) {}
8954 
8955   /// Generate code for the combined entry if we have a partially mapped struct
8956   /// and take care of the mapping flags of the arguments corresponding to
8957   /// individual struct members.
8958   void emitCombinedEntry(MapCombinedInfoTy &CombinedInfo,
8959                          MapFlagsArrayTy &CurTypes,
8960                          const StructRangeInfoTy &PartialStruct,
8961                          const ValueDecl *VD = nullptr,
8962                          bool NotTargetParams = true) const {
8963     if (CurTypes.size() == 1 &&
8964         ((CurTypes.back() & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF) &&
8965         !PartialStruct.IsArraySection)
8966       return;
8967     Address LBAddr = PartialStruct.LowestElem.second;
8968     Address HBAddr = PartialStruct.HighestElem.second;
8969     if (PartialStruct.HasCompleteRecord) {
8970       LBAddr = PartialStruct.LB;
8971       HBAddr = PartialStruct.LB;
8972     }
8973     CombinedInfo.Exprs.push_back(VD);
8974     // Base is the base of the struct
8975     CombinedInfo.BasePointers.push_back(PartialStruct.Base.getPointer());
8976     // Pointer is the address of the lowest element
8977     llvm::Value *LB = LBAddr.getPointer();
8978     CombinedInfo.Pointers.push_back(LB);
8979     // There should not be a mapper for a combined entry.
8980     CombinedInfo.Mappers.push_back(nullptr);
8981     // Size is (addr of {highest+1} element) - (addr of lowest element)
8982     llvm::Value *HB = HBAddr.getPointer();
8983     llvm::Value *HAddr =
8984         CGF.Builder.CreateConstGEP1_32(HBAddr.getElementType(), HB, /*Idx0=*/1);
8985     llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy);
8986     llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy);
8987     llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CGF.Int8Ty, CHAddr, CLAddr);
8988     llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty,
8989                                                   /*isSigned=*/false);
8990     CombinedInfo.Sizes.push_back(Size);
8991     // Map type is always TARGET_PARAM, if generate info for captures.
8992     CombinedInfo.Types.push_back(NotTargetParams ? OMP_MAP_NONE
8993                                                  : OMP_MAP_TARGET_PARAM);
8994     // If any element has the present modifier, then make sure the runtime
8995     // doesn't attempt to allocate the struct.
8996     if (CurTypes.end() !=
8997         llvm::find_if(CurTypes, [](OpenMPOffloadMappingFlags Type) {
8998           return Type & OMP_MAP_PRESENT;
8999         }))
9000       CombinedInfo.Types.back() |= OMP_MAP_PRESENT;
9001     // Remove TARGET_PARAM flag from the first element
9002     (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM;
9003     // If any element has the ompx_hold modifier, then make sure the runtime
9004     // uses the hold reference count for the struct as a whole so that it won't
9005     // be unmapped by an extra dynamic reference count decrement.  Add it to all
9006     // elements as well so the runtime knows which reference count to check
9007     // when determining whether it's time for device-to-host transfers of
9008     // individual elements.
9009     if (CurTypes.end() !=
9010         llvm::find_if(CurTypes, [](OpenMPOffloadMappingFlags Type) {
9011           return Type & OMP_MAP_OMPX_HOLD;
9012         })) {
9013       CombinedInfo.Types.back() |= OMP_MAP_OMPX_HOLD;
9014       for (auto &M : CurTypes)
9015         M |= OMP_MAP_OMPX_HOLD;
9016     }
9017 
9018     // All other current entries will be MEMBER_OF the combined entry
9019     // (except for PTR_AND_OBJ entries which do not have a placeholder value
9020     // 0xFFFF in the MEMBER_OF field).
9021     OpenMPOffloadMappingFlags MemberOfFlag =
9022         getMemberOfFlag(CombinedInfo.BasePointers.size() - 1);
9023     for (auto &M : CurTypes)
9024       setCorrectMemberOfFlag(M, MemberOfFlag);
9025   }
9026 
9027   /// Generate all the base pointers, section pointers, sizes, map types, and
9028   /// mappers for the extracted mappable expressions (all included in \a
9029   /// CombinedInfo). Also, for each item that relates with a device pointer, a
9030   /// pair of the relevant declaration and index where it occurs is appended to
9031   /// the device pointers info array.
9032   void generateAllInfo(
9033       MapCombinedInfoTy &CombinedInfo,
9034       const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet =
9035           llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const {
9036     assert(CurDir.is<const OMPExecutableDirective *>() &&
9037            "Expect a executable directive");
9038     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
9039     generateAllInfoForClauses(CurExecDir->clauses(), CombinedInfo, SkipVarSet);
9040   }
9041 
9042   /// Generate all the base pointers, section pointers, sizes, map types, and
9043   /// mappers for the extracted map clauses of user-defined mapper (all included
9044   /// in \a CombinedInfo).
9045   void generateAllInfoForMapper(MapCombinedInfoTy &CombinedInfo) const {
9046     assert(CurDir.is<const OMPDeclareMapperDecl *>() &&
9047            "Expect a declare mapper directive");
9048     const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>();
9049     generateAllInfoForClauses(CurMapperDir->clauses(), CombinedInfo);
9050   }
9051 
9052   /// Emit capture info for lambdas for variables captured by reference.
9053   void generateInfoForLambdaCaptures(
9054       const ValueDecl *VD, llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo,
9055       llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const {
9056     QualType VDType = VD->getType().getCanonicalType().getNonReferenceType();
9057     const auto *RD = VDType->getAsCXXRecordDecl();
9058     if (!RD || !RD->isLambda())
9059       return;
9060     Address VDAddr(Arg, CGF.ConvertTypeForMem(VDType),
9061                    CGF.getContext().getDeclAlign(VD));
9062     LValue VDLVal = CGF.MakeAddrLValue(VDAddr, VDType);
9063     llvm::DenseMap<const VarDecl *, FieldDecl *> Captures;
9064     FieldDecl *ThisCapture = nullptr;
9065     RD->getCaptureFields(Captures, ThisCapture);
9066     if (ThisCapture) {
9067       LValue ThisLVal =
9068           CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture);
9069       LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture);
9070       LambdaPointers.try_emplace(ThisLVal.getPointer(CGF),
9071                                  VDLVal.getPointer(CGF));
9072       CombinedInfo.Exprs.push_back(VD);
9073       CombinedInfo.BasePointers.push_back(ThisLVal.getPointer(CGF));
9074       CombinedInfo.Pointers.push_back(ThisLValVal.getPointer(CGF));
9075       CombinedInfo.Sizes.push_back(
9076           CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy),
9077                                     CGF.Int64Ty, /*isSigned=*/true));
9078       CombinedInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
9079                                    OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
9080       CombinedInfo.Mappers.push_back(nullptr);
9081     }
9082     for (const LambdaCapture &LC : RD->captures()) {
9083       if (!LC.capturesVariable())
9084         continue;
9085       const VarDecl *VD = LC.getCapturedVar();
9086       if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType())
9087         continue;
9088       auto It = Captures.find(VD);
9089       assert(It != Captures.end() && "Found lambda capture without field.");
9090       LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second);
9091       if (LC.getCaptureKind() == LCK_ByRef) {
9092         LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second);
9093         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
9094                                    VDLVal.getPointer(CGF));
9095         CombinedInfo.Exprs.push_back(VD);
9096         CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF));
9097         CombinedInfo.Pointers.push_back(VarLValVal.getPointer(CGF));
9098         CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
9099             CGF.getTypeSize(
9100                 VD->getType().getCanonicalType().getNonReferenceType()),
9101             CGF.Int64Ty, /*isSigned=*/true));
9102       } else {
9103         RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation());
9104         LambdaPointers.try_emplace(VarLVal.getPointer(CGF),
9105                                    VDLVal.getPointer(CGF));
9106         CombinedInfo.Exprs.push_back(VD);
9107         CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF));
9108         CombinedInfo.Pointers.push_back(VarRVal.getScalarVal());
9109         CombinedInfo.Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0));
9110       }
9111       CombinedInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
9112                                    OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT);
9113       CombinedInfo.Mappers.push_back(nullptr);
9114     }
9115   }
9116 
9117   /// Set correct indices for lambdas captures.
9118   void adjustMemberOfForLambdaCaptures(
9119       const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers,
9120       MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers,
9121       MapFlagsArrayTy &Types) const {
9122     for (unsigned I = 0, E = Types.size(); I < E; ++I) {
9123       // Set correct member_of idx for all implicit lambda captures.
9124       if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL |
9125                        OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT))
9126         continue;
9127       llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]);
9128       assert(BasePtr && "Unable to find base lambda address.");
9129       int TgtIdx = -1;
9130       for (unsigned J = I; J > 0; --J) {
9131         unsigned Idx = J - 1;
9132         if (Pointers[Idx] != BasePtr)
9133           continue;
9134         TgtIdx = Idx;
9135         break;
9136       }
9137       assert(TgtIdx != -1 && "Unable to find parent lambda.");
9138       // All other current entries will be MEMBER_OF the combined entry
9139       // (except for PTR_AND_OBJ entries which do not have a placeholder value
9140       // 0xFFFF in the MEMBER_OF field).
9141       OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx);
9142       setCorrectMemberOfFlag(Types[I], MemberOfFlag);
9143     }
9144   }
9145 
9146   /// Generate the base pointers, section pointers, sizes, map types, and
9147   /// mappers associated to a given capture (all included in \a CombinedInfo).
9148   void generateInfoForCapture(const CapturedStmt::Capture *Cap,
9149                               llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo,
9150                               StructRangeInfoTy &PartialStruct) const {
9151     assert(!Cap->capturesVariableArrayType() &&
9152            "Not expecting to generate map info for a variable array type!");
9153 
9154     // We need to know when we generating information for the first component
9155     const ValueDecl *VD = Cap->capturesThis()
9156                               ? nullptr
9157                               : Cap->getCapturedVar()->getCanonicalDecl();
9158 
9159     // for map(to: lambda): skip here, processing it in
9160     // generateDefaultMapInfo
9161     if (LambdasMap.count(VD))
9162       return;
9163 
9164     // If this declaration appears in a is_device_ptr clause we just have to
9165     // pass the pointer by value. If it is a reference to a declaration, we just
9166     // pass its value.
9167     if (DevPointersMap.count(VD)) {
9168       CombinedInfo.Exprs.push_back(VD);
9169       CombinedInfo.BasePointers.emplace_back(Arg, VD);
9170       CombinedInfo.Pointers.push_back(Arg);
9171       CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
9172           CGF.getTypeSize(CGF.getContext().VoidPtrTy), CGF.Int64Ty,
9173           /*isSigned=*/true));
9174       CombinedInfo.Types.push_back(
9175           (Cap->capturesVariable() ? OMP_MAP_TO : OMP_MAP_LITERAL) |
9176           OMP_MAP_TARGET_PARAM);
9177       CombinedInfo.Mappers.push_back(nullptr);
9178       return;
9179     }
9180 
9181     using MapData =
9182         std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef,
9183                    OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool,
9184                    const ValueDecl *, const Expr *>;
9185     SmallVector<MapData, 4> DeclComponentLists;
9186     assert(CurDir.is<const OMPExecutableDirective *>() &&
9187            "Expect a executable directive");
9188     const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>();
9189     for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) {
9190       const auto *EI = C->getVarRefs().begin();
9191       for (const auto L : C->decl_component_lists(VD)) {
9192         const ValueDecl *VDecl, *Mapper;
9193         // The Expression is not correct if the mapping is implicit
9194         const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr;
9195         OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
9196         std::tie(VDecl, Components, Mapper) = L;
9197         assert(VDecl == VD && "We got information for the wrong declaration??");
9198         assert(!Components.empty() &&
9199                "Not expecting declaration with no component lists.");
9200         DeclComponentLists.emplace_back(Components, C->getMapType(),
9201                                         C->getMapTypeModifiers(),
9202                                         C->isImplicit(), Mapper, E);
9203         ++EI;
9204       }
9205     }
9206     llvm::stable_sort(DeclComponentLists, [](const MapData &LHS,
9207                                              const MapData &RHS) {
9208       ArrayRef<OpenMPMapModifierKind> MapModifiers = std::get<2>(LHS);
9209       OpenMPMapClauseKind MapType = std::get<1>(RHS);
9210       bool HasPresent =
9211           llvm::is_contained(MapModifiers, clang::OMPC_MAP_MODIFIER_present);
9212       bool HasAllocs = MapType == OMPC_MAP_alloc;
9213       MapModifiers = std::get<2>(RHS);
9214       MapType = std::get<1>(LHS);
9215       bool HasPresentR =
9216           llvm::is_contained(MapModifiers, clang::OMPC_MAP_MODIFIER_present);
9217       bool HasAllocsR = MapType == OMPC_MAP_alloc;
9218       return (HasPresent && !HasPresentR) || (HasAllocs && !HasAllocsR);
9219     });
9220 
9221     // Find overlapping elements (including the offset from the base element).
9222     llvm::SmallDenseMap<
9223         const MapData *,
9224         llvm::SmallVector<
9225             OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>,
9226         4>
9227         OverlappedData;
9228     size_t Count = 0;
9229     for (const MapData &L : DeclComponentLists) {
9230       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
9231       OpenMPMapClauseKind MapType;
9232       ArrayRef<OpenMPMapModifierKind> MapModifiers;
9233       bool IsImplicit;
9234       const ValueDecl *Mapper;
9235       const Expr *VarRef;
9236       std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) =
9237           L;
9238       ++Count;
9239       for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) {
9240         OMPClauseMappableExprCommon::MappableExprComponentListRef Components1;
9241         std::tie(Components1, MapType, MapModifiers, IsImplicit, Mapper,
9242                  VarRef) = L1;
9243         auto CI = Components.rbegin();
9244         auto CE = Components.rend();
9245         auto SI = Components1.rbegin();
9246         auto SE = Components1.rend();
9247         for (; CI != CE && SI != SE; ++CI, ++SI) {
9248           if (CI->getAssociatedExpression()->getStmtClass() !=
9249               SI->getAssociatedExpression()->getStmtClass())
9250             break;
9251           // Are we dealing with different variables/fields?
9252           if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration())
9253             break;
9254         }
9255         // Found overlapping if, at least for one component, reached the head
9256         // of the components list.
9257         if (CI == CE || SI == SE) {
9258           // Ignore it if it is the same component.
9259           if (CI == CE && SI == SE)
9260             continue;
9261           const auto It = (SI == SE) ? CI : SI;
9262           // If one component is a pointer and another one is a kind of
9263           // dereference of this pointer (array subscript, section, dereference,
9264           // etc.), it is not an overlapping.
9265           // Same, if one component is a base and another component is a
9266           // dereferenced pointer memberexpr with the same base.
9267           if (!isa<MemberExpr>(It->getAssociatedExpression()) ||
9268               (std::prev(It)->getAssociatedDeclaration() &&
9269                std::prev(It)
9270                    ->getAssociatedDeclaration()
9271                    ->getType()
9272                    ->isPointerType()) ||
9273               (It->getAssociatedDeclaration() &&
9274                It->getAssociatedDeclaration()->getType()->isPointerType() &&
9275                std::next(It) != CE && std::next(It) != SE))
9276             continue;
9277           const MapData &BaseData = CI == CE ? L : L1;
9278           OMPClauseMappableExprCommon::MappableExprComponentListRef SubData =
9279               SI == SE ? Components : Components1;
9280           auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData);
9281           OverlappedElements.getSecond().push_back(SubData);
9282         }
9283       }
9284     }
9285     // Sort the overlapped elements for each item.
9286     llvm::SmallVector<const FieldDecl *, 4> Layout;
9287     if (!OverlappedData.empty()) {
9288       const Type *BaseType = VD->getType().getCanonicalType().getTypePtr();
9289       const Type *OrigType = BaseType->getPointeeOrArrayElementType();
9290       while (BaseType != OrigType) {
9291         BaseType = OrigType->getCanonicalTypeInternal().getTypePtr();
9292         OrigType = BaseType->getPointeeOrArrayElementType();
9293       }
9294 
9295       if (const auto *CRD = BaseType->getAsCXXRecordDecl())
9296         getPlainLayout(CRD, Layout, /*AsBase=*/false);
9297       else {
9298         const auto *RD = BaseType->getAsRecordDecl();
9299         Layout.append(RD->field_begin(), RD->field_end());
9300       }
9301     }
9302     for (auto &Pair : OverlappedData) {
9303       llvm::stable_sort(
9304           Pair.getSecond(),
9305           [&Layout](
9306               OMPClauseMappableExprCommon::MappableExprComponentListRef First,
9307               OMPClauseMappableExprCommon::MappableExprComponentListRef
9308                   Second) {
9309             auto CI = First.rbegin();
9310             auto CE = First.rend();
9311             auto SI = Second.rbegin();
9312             auto SE = Second.rend();
9313             for (; CI != CE && SI != SE; ++CI, ++SI) {
9314               if (CI->getAssociatedExpression()->getStmtClass() !=
9315                   SI->getAssociatedExpression()->getStmtClass())
9316                 break;
9317               // Are we dealing with different variables/fields?
9318               if (CI->getAssociatedDeclaration() !=
9319                   SI->getAssociatedDeclaration())
9320                 break;
9321             }
9322 
9323             // Lists contain the same elements.
9324             if (CI == CE && SI == SE)
9325               return false;
9326 
9327             // List with less elements is less than list with more elements.
9328             if (CI == CE || SI == SE)
9329               return CI == CE;
9330 
9331             const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration());
9332             const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration());
9333             if (FD1->getParent() == FD2->getParent())
9334               return FD1->getFieldIndex() < FD2->getFieldIndex();
9335             const auto *It =
9336                 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) {
9337                   return FD == FD1 || FD == FD2;
9338                 });
9339             return *It == FD1;
9340           });
9341     }
9342 
9343     // Associated with a capture, because the mapping flags depend on it.
9344     // Go through all of the elements with the overlapped elements.
9345     bool IsFirstComponentList = true;
9346     for (const auto &Pair : OverlappedData) {
9347       const MapData &L = *Pair.getFirst();
9348       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
9349       OpenMPMapClauseKind MapType;
9350       ArrayRef<OpenMPMapModifierKind> MapModifiers;
9351       bool IsImplicit;
9352       const ValueDecl *Mapper;
9353       const Expr *VarRef;
9354       std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) =
9355           L;
9356       ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef>
9357           OverlappedComponents = Pair.getSecond();
9358       generateInfoForComponentList(
9359           MapType, MapModifiers, llvm::None, Components, CombinedInfo,
9360           PartialStruct, IsFirstComponentList, IsImplicit, Mapper,
9361           /*ForDeviceAddr=*/false, VD, VarRef, OverlappedComponents);
9362       IsFirstComponentList = false;
9363     }
9364     // Go through other elements without overlapped elements.
9365     for (const MapData &L : DeclComponentLists) {
9366       OMPClauseMappableExprCommon::MappableExprComponentListRef Components;
9367       OpenMPMapClauseKind MapType;
9368       ArrayRef<OpenMPMapModifierKind> MapModifiers;
9369       bool IsImplicit;
9370       const ValueDecl *Mapper;
9371       const Expr *VarRef;
9372       std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) =
9373           L;
9374       auto It = OverlappedData.find(&L);
9375       if (It == OverlappedData.end())
9376         generateInfoForComponentList(MapType, MapModifiers, llvm::None,
9377                                      Components, CombinedInfo, PartialStruct,
9378                                      IsFirstComponentList, IsImplicit, Mapper,
9379                                      /*ForDeviceAddr=*/false, VD, VarRef);
9380       IsFirstComponentList = false;
9381     }
9382   }
9383 
9384   /// Generate the default map information for a given capture \a CI,
9385   /// record field declaration \a RI and captured value \a CV.
9386   void generateDefaultMapInfo(const CapturedStmt::Capture &CI,
9387                               const FieldDecl &RI, llvm::Value *CV,
9388                               MapCombinedInfoTy &CombinedInfo) const {
9389     bool IsImplicit = true;
9390     // Do the default mapping.
9391     if (CI.capturesThis()) {
9392       CombinedInfo.Exprs.push_back(nullptr);
9393       CombinedInfo.BasePointers.push_back(CV);
9394       CombinedInfo.Pointers.push_back(CV);
9395       const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr());
9396       CombinedInfo.Sizes.push_back(
9397           CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()),
9398                                     CGF.Int64Ty, /*isSigned=*/true));
9399       // Default map type.
9400       CombinedInfo.Types.push_back(OMP_MAP_TO | OMP_MAP_FROM);
9401     } else if (CI.capturesVariableByCopy()) {
9402       const VarDecl *VD = CI.getCapturedVar();
9403       CombinedInfo.Exprs.push_back(VD->getCanonicalDecl());
9404       CombinedInfo.BasePointers.push_back(CV);
9405       CombinedInfo.Pointers.push_back(CV);
9406       if (!RI.getType()->isAnyPointerType()) {
9407         // We have to signal to the runtime captures passed by value that are
9408         // not pointers.
9409         CombinedInfo.Types.push_back(OMP_MAP_LITERAL);
9410         CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
9411             CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true));
9412       } else {
9413         // Pointers are implicitly mapped with a zero size and no flags
9414         // (other than first map that is added for all implicit maps).
9415         CombinedInfo.Types.push_back(OMP_MAP_NONE);
9416         CombinedInfo.Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty));
9417       }
9418       auto I = FirstPrivateDecls.find(VD);
9419       if (I != FirstPrivateDecls.end())
9420         IsImplicit = I->getSecond();
9421     } else {
9422       assert(CI.capturesVariable() && "Expected captured reference.");
9423       const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr());
9424       QualType ElementType = PtrTy->getPointeeType();
9425       CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
9426           CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true));
9427       // The default map type for a scalar/complex type is 'to' because by
9428       // default the value doesn't have to be retrieved. For an aggregate
9429       // type, the default is 'tofrom'.
9430       CombinedInfo.Types.push_back(getMapModifiersForPrivateClauses(CI));
9431       const VarDecl *VD = CI.getCapturedVar();
9432       auto I = FirstPrivateDecls.find(VD);
9433       CombinedInfo.Exprs.push_back(VD->getCanonicalDecl());
9434       CombinedInfo.BasePointers.push_back(CV);
9435       if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) {
9436         Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue(
9437             CV, ElementType, CGF.getContext().getDeclAlign(VD),
9438             AlignmentSource::Decl));
9439         CombinedInfo.Pointers.push_back(PtrAddr.getPointer());
9440       } else {
9441         CombinedInfo.Pointers.push_back(CV);
9442       }
9443       if (I != FirstPrivateDecls.end())
9444         IsImplicit = I->getSecond();
9445     }
9446     // Every default map produces a single argument which is a target parameter.
9447     CombinedInfo.Types.back() |= OMP_MAP_TARGET_PARAM;
9448 
9449     // Add flag stating this is an implicit map.
9450     if (IsImplicit)
9451       CombinedInfo.Types.back() |= OMP_MAP_IMPLICIT;
9452 
9453     // No user-defined mapper for default mapping.
9454     CombinedInfo.Mappers.push_back(nullptr);
9455   }
9456 };
9457 } // anonymous namespace
9458 
9459 static void emitNonContiguousDescriptor(
9460     CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo,
9461     CGOpenMPRuntime::TargetDataInfo &Info) {
9462   CodeGenModule &CGM = CGF.CGM;
9463   MappableExprsHandler::MapCombinedInfoTy::StructNonContiguousInfo
9464       &NonContigInfo = CombinedInfo.NonContigInfo;
9465 
9466   // Build an array of struct descriptor_dim and then assign it to
9467   // offload_args.
9468   //
9469   // struct descriptor_dim {
9470   //  uint64_t offset;
9471   //  uint64_t count;
9472   //  uint64_t stride
9473   // };
9474   ASTContext &C = CGF.getContext();
9475   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0);
9476   RecordDecl *RD;
9477   RD = C.buildImplicitRecord("descriptor_dim");
9478   RD->startDefinition();
9479   addFieldToRecordDecl(C, RD, Int64Ty);
9480   addFieldToRecordDecl(C, RD, Int64Ty);
9481   addFieldToRecordDecl(C, RD, Int64Ty);
9482   RD->completeDefinition();
9483   QualType DimTy = C.getRecordType(RD);
9484 
9485   enum { OffsetFD = 0, CountFD, StrideFD };
9486   // We need two index variable here since the size of "Dims" is the same as the
9487   // size of Components, however, the size of offset, count, and stride is equal
9488   // to the size of base declaration that is non-contiguous.
9489   for (unsigned I = 0, L = 0, E = NonContigInfo.Dims.size(); I < E; ++I) {
9490     // Skip emitting ir if dimension size is 1 since it cannot be
9491     // non-contiguous.
9492     if (NonContigInfo.Dims[I] == 1)
9493       continue;
9494     llvm::APInt Size(/*numBits=*/32, NonContigInfo.Dims[I]);
9495     QualType ArrayTy =
9496         C.getConstantArrayType(DimTy, Size, nullptr, ArrayType::Normal, 0);
9497     Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims");
9498     for (unsigned II = 0, EE = NonContigInfo.Dims[I]; II < EE; ++II) {
9499       unsigned RevIdx = EE - II - 1;
9500       LValue DimsLVal = CGF.MakeAddrLValue(
9501           CGF.Builder.CreateConstArrayGEP(DimsAddr, II), DimTy);
9502       // Offset
9503       LValue OffsetLVal = CGF.EmitLValueForField(
9504           DimsLVal, *std::next(RD->field_begin(), OffsetFD));
9505       CGF.EmitStoreOfScalar(NonContigInfo.Offsets[L][RevIdx], OffsetLVal);
9506       // Count
9507       LValue CountLVal = CGF.EmitLValueForField(
9508           DimsLVal, *std::next(RD->field_begin(), CountFD));
9509       CGF.EmitStoreOfScalar(NonContigInfo.Counts[L][RevIdx], CountLVal);
9510       // Stride
9511       LValue StrideLVal = CGF.EmitLValueForField(
9512           DimsLVal, *std::next(RD->field_begin(), StrideFD));
9513       CGF.EmitStoreOfScalar(NonContigInfo.Strides[L][RevIdx], StrideLVal);
9514     }
9515     // args[I] = &dims
9516     Address DAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
9517         DimsAddr, CGM.Int8PtrTy, CGM.Int8Ty);
9518     llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32(
9519         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
9520         Info.PointersArray, 0, I);
9521     Address PAddr(P, CGM.VoidPtrTy, CGF.getPointerAlign());
9522     CGF.Builder.CreateStore(DAddr.getPointer(), PAddr);
9523     ++L;
9524   }
9525 }
9526 
9527 // Try to extract the base declaration from a `this->x` expression if possible.
9528 static ValueDecl *getDeclFromThisExpr(const Expr *E) {
9529   if (!E)
9530     return nullptr;
9531 
9532   if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E->IgnoreParenCasts()))
9533     if (const MemberExpr *ME =
9534             dyn_cast<MemberExpr>(OASE->getBase()->IgnoreParenImpCasts()))
9535       return ME->getMemberDecl();
9536   return nullptr;
9537 }
9538 
9539 /// Emit a string constant containing the names of the values mapped to the
9540 /// offloading runtime library.
9541 llvm::Constant *
9542 emitMappingInformation(CodeGenFunction &CGF, llvm::OpenMPIRBuilder &OMPBuilder,
9543                        MappableExprsHandler::MappingExprInfo &MapExprs) {
9544 
9545   uint32_t SrcLocStrSize;
9546   if (!MapExprs.getMapDecl() && !MapExprs.getMapExpr())
9547     return OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize);
9548 
9549   SourceLocation Loc;
9550   if (!MapExprs.getMapDecl() && MapExprs.getMapExpr()) {
9551     if (const ValueDecl *VD = getDeclFromThisExpr(MapExprs.getMapExpr()))
9552       Loc = VD->getLocation();
9553     else
9554       Loc = MapExprs.getMapExpr()->getExprLoc();
9555   } else {
9556     Loc = MapExprs.getMapDecl()->getLocation();
9557   }
9558 
9559   std::string ExprName;
9560   if (MapExprs.getMapExpr()) {
9561     PrintingPolicy P(CGF.getContext().getLangOpts());
9562     llvm::raw_string_ostream OS(ExprName);
9563     MapExprs.getMapExpr()->printPretty(OS, nullptr, P);
9564     OS.flush();
9565   } else {
9566     ExprName = MapExprs.getMapDecl()->getNameAsString();
9567   }
9568 
9569   PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
9570   return OMPBuilder.getOrCreateSrcLocStr(PLoc.getFilename(), ExprName,
9571                                          PLoc.getLine(), PLoc.getColumn(),
9572                                          SrcLocStrSize);
9573 }
9574 
9575 /// Emit the arrays used to pass the captures and map information to the
9576 /// offloading runtime library. If there is no map or capture information,
9577 /// return nullptr by reference.
9578 static void emitOffloadingArrays(
9579     CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo,
9580     CGOpenMPRuntime::TargetDataInfo &Info, llvm::OpenMPIRBuilder &OMPBuilder,
9581     bool IsNonContiguous = false) {
9582   CodeGenModule &CGM = CGF.CGM;
9583   ASTContext &Ctx = CGF.getContext();
9584 
9585   // Reset the array information.
9586   Info.clearArrayInfo();
9587   Info.NumberOfPtrs = CombinedInfo.BasePointers.size();
9588 
9589   if (Info.NumberOfPtrs) {
9590     // Detect if we have any capture size requiring runtime evaluation of the
9591     // size so that a constant array could be eventually used.
9592 
9593     llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true);
9594     QualType PointerArrayType = Ctx.getConstantArrayType(
9595         Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal,
9596         /*IndexTypeQuals=*/0);
9597 
9598     Info.BasePointersArray =
9599         CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer();
9600     Info.PointersArray =
9601         CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer();
9602     Address MappersArray =
9603         CGF.CreateMemTemp(PointerArrayType, ".offload_mappers");
9604     Info.MappersArray = MappersArray.getPointer();
9605 
9606     // If we don't have any VLA types or other types that require runtime
9607     // evaluation, we can use a constant array for the map sizes, otherwise we
9608     // need to fill up the arrays as we do for the pointers.
9609     QualType Int64Ty =
9610         Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
9611     SmallVector<llvm::Constant *> ConstSizes(
9612         CombinedInfo.Sizes.size(), llvm::ConstantInt::get(CGF.Int64Ty, 0));
9613     llvm::SmallBitVector RuntimeSizes(CombinedInfo.Sizes.size());
9614     for (unsigned I = 0, E = CombinedInfo.Sizes.size(); I < E; ++I) {
9615       if (auto *CI = dyn_cast<llvm::Constant>(CombinedInfo.Sizes[I])) {
9616         if (!isa<llvm::ConstantExpr>(CI) && !isa<llvm::GlobalValue>(CI)) {
9617           if (IsNonContiguous && (CombinedInfo.Types[I] &
9618                                   MappableExprsHandler::OMP_MAP_NON_CONTIG))
9619             ConstSizes[I] = llvm::ConstantInt::get(
9620                 CGF.Int64Ty, CombinedInfo.NonContigInfo.Dims[I]);
9621           else
9622             ConstSizes[I] = CI;
9623           continue;
9624         }
9625       }
9626       RuntimeSizes.set(I);
9627     }
9628 
9629     if (RuntimeSizes.all()) {
9630       QualType SizeArrayType = Ctx.getConstantArrayType(
9631           Int64Ty, PointerNumAP, nullptr, ArrayType::Normal,
9632           /*IndexTypeQuals=*/0);
9633       Info.SizesArray =
9634           CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer();
9635     } else {
9636       auto *SizesArrayInit = llvm::ConstantArray::get(
9637           llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes);
9638       std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"});
9639       auto *SizesArrayGbl = new llvm::GlobalVariable(
9640           CGM.getModule(), SizesArrayInit->getType(), /*isConstant=*/true,
9641           llvm::GlobalValue::PrivateLinkage, SizesArrayInit, Name);
9642       SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global);
9643       if (RuntimeSizes.any()) {
9644         QualType SizeArrayType = Ctx.getConstantArrayType(
9645             Int64Ty, PointerNumAP, nullptr, ArrayType::Normal,
9646             /*IndexTypeQuals=*/0);
9647         Address Buffer = CGF.CreateMemTemp(SizeArrayType, ".offload_sizes");
9648         llvm::Value *GblConstPtr =
9649             CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
9650                 SizesArrayGbl, CGM.Int64Ty->getPointerTo());
9651         CGF.Builder.CreateMemCpy(
9652             Buffer,
9653             Address(GblConstPtr, CGM.Int64Ty,
9654                     CGM.getNaturalTypeAlignment(Ctx.getIntTypeForBitwidth(
9655                         /*DestWidth=*/64, /*Signed=*/false))),
9656             CGF.getTypeSize(SizeArrayType));
9657         Info.SizesArray = Buffer.getPointer();
9658       } else {
9659         Info.SizesArray = SizesArrayGbl;
9660       }
9661     }
9662 
9663     // The map types are always constant so we don't need to generate code to
9664     // fill arrays. Instead, we create an array constant.
9665     SmallVector<uint64_t, 4> Mapping(CombinedInfo.Types.size(), 0);
9666     llvm::copy(CombinedInfo.Types, Mapping.begin());
9667     std::string MaptypesName =
9668         CGM.getOpenMPRuntime().getName({"offload_maptypes"});
9669     auto *MapTypesArrayGbl =
9670         OMPBuilder.createOffloadMaptypes(Mapping, MaptypesName);
9671     Info.MapTypesArray = MapTypesArrayGbl;
9672 
9673     // The information types are only built if there is debug information
9674     // requested.
9675     if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo) {
9676       Info.MapNamesArray = llvm::Constant::getNullValue(
9677           llvm::Type::getInt8Ty(CGF.Builder.getContext())->getPointerTo());
9678     } else {
9679       auto fillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) {
9680         return emitMappingInformation(CGF, OMPBuilder, MapExpr);
9681       };
9682       SmallVector<llvm::Constant *, 4> InfoMap(CombinedInfo.Exprs.size());
9683       llvm::transform(CombinedInfo.Exprs, InfoMap.begin(), fillInfoMap);
9684       std::string MapnamesName =
9685           CGM.getOpenMPRuntime().getName({"offload_mapnames"});
9686       auto *MapNamesArrayGbl =
9687           OMPBuilder.createOffloadMapnames(InfoMap, MapnamesName);
9688       Info.MapNamesArray = MapNamesArrayGbl;
9689     }
9690 
9691     // If there's a present map type modifier, it must not be applied to the end
9692     // of a region, so generate a separate map type array in that case.
9693     if (Info.separateBeginEndCalls()) {
9694       bool EndMapTypesDiffer = false;
9695       for (uint64_t &Type : Mapping) {
9696         if (Type & MappableExprsHandler::OMP_MAP_PRESENT) {
9697           Type &= ~MappableExprsHandler::OMP_MAP_PRESENT;
9698           EndMapTypesDiffer = true;
9699         }
9700       }
9701       if (EndMapTypesDiffer) {
9702         MapTypesArrayGbl =
9703             OMPBuilder.createOffloadMaptypes(Mapping, MaptypesName);
9704         Info.MapTypesArrayEnd = MapTypesArrayGbl;
9705       }
9706     }
9707 
9708     for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) {
9709       llvm::Value *BPVal = *CombinedInfo.BasePointers[I];
9710       llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32(
9711           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
9712           Info.BasePointersArray, 0, I);
9713       BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
9714           BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0));
9715       Address BPAddr(BP, BPVal->getType(),
9716                      Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
9717       CGF.Builder.CreateStore(BPVal, BPAddr);
9718 
9719       if (Info.requiresDevicePointerInfo())
9720         if (const ValueDecl *DevVD =
9721                 CombinedInfo.BasePointers[I].getDevicePtrDecl())
9722           Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr);
9723 
9724       llvm::Value *PVal = CombinedInfo.Pointers[I];
9725       llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32(
9726           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
9727           Info.PointersArray, 0, I);
9728       P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
9729           P, PVal->getType()->getPointerTo(/*AddrSpace=*/0));
9730       Address PAddr(P, PVal->getType(), Ctx.getTypeAlignInChars(Ctx.VoidPtrTy));
9731       CGF.Builder.CreateStore(PVal, PAddr);
9732 
9733       if (RuntimeSizes.test(I)) {
9734         llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32(
9735             llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
9736             Info.SizesArray,
9737             /*Idx0=*/0,
9738             /*Idx1=*/I);
9739         Address SAddr(S, CGM.Int64Ty, Ctx.getTypeAlignInChars(Int64Ty));
9740         CGF.Builder.CreateStore(CGF.Builder.CreateIntCast(CombinedInfo.Sizes[I],
9741                                                           CGM.Int64Ty,
9742                                                           /*isSigned=*/true),
9743                                 SAddr);
9744       }
9745 
9746       // Fill up the mapper array.
9747       llvm::Value *MFunc = llvm::ConstantPointerNull::get(CGM.VoidPtrTy);
9748       if (CombinedInfo.Mappers[I]) {
9749         MFunc = CGM.getOpenMPRuntime().getOrCreateUserDefinedMapperFunc(
9750             cast<OMPDeclareMapperDecl>(CombinedInfo.Mappers[I]));
9751         MFunc = CGF.Builder.CreatePointerCast(MFunc, CGM.VoidPtrTy);
9752         Info.HasMapper = true;
9753       }
9754       Address MAddr = CGF.Builder.CreateConstArrayGEP(MappersArray, I);
9755       CGF.Builder.CreateStore(MFunc, MAddr);
9756     }
9757   }
9758 
9759   if (!IsNonContiguous || CombinedInfo.NonContigInfo.Offsets.empty() ||
9760       Info.NumberOfPtrs == 0)
9761     return;
9762 
9763   emitNonContiguousDescriptor(CGF, CombinedInfo, Info);
9764 }
9765 
9766 namespace {
9767 /// Additional arguments for emitOffloadingArraysArgument function.
9768 struct ArgumentsOptions {
9769   bool ForEndCall = false;
9770   ArgumentsOptions() = default;
9771   ArgumentsOptions(bool ForEndCall) : ForEndCall(ForEndCall) {}
9772 };
9773 } // namespace
9774 
9775 /// Emit the arguments to be passed to the runtime library based on the
9776 /// arrays of base pointers, pointers, sizes, map types, and mappers.  If
9777 /// ForEndCall, emit map types to be passed for the end of the region instead of
9778 /// the beginning.
9779 static void emitOffloadingArraysArgument(
9780     CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg,
9781     llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg,
9782     llvm::Value *&MapTypesArrayArg, llvm::Value *&MapNamesArrayArg,
9783     llvm::Value *&MappersArrayArg, CGOpenMPRuntime::TargetDataInfo &Info,
9784     const ArgumentsOptions &Options = ArgumentsOptions()) {
9785   assert((!Options.ForEndCall || Info.separateBeginEndCalls()) &&
9786          "expected region end call to runtime only when end call is separate");
9787   CodeGenModule &CGM = CGF.CGM;
9788   if (Info.NumberOfPtrs) {
9789     BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
9790         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
9791         Info.BasePointersArray,
9792         /*Idx0=*/0, /*Idx1=*/0);
9793     PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
9794         llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
9795         Info.PointersArray,
9796         /*Idx0=*/0,
9797         /*Idx1=*/0);
9798     SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
9799         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray,
9800         /*Idx0=*/0, /*Idx1=*/0);
9801     MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
9802         llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs),
9803         Options.ForEndCall && Info.MapTypesArrayEnd ? Info.MapTypesArrayEnd
9804                                                     : Info.MapTypesArray,
9805         /*Idx0=*/0,
9806         /*Idx1=*/0);
9807 
9808     // Only emit the mapper information arrays if debug information is
9809     // requested.
9810     if (CGF.CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo)
9811       MapNamesArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
9812     else
9813       MapNamesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32(
9814           llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs),
9815           Info.MapNamesArray,
9816           /*Idx0=*/0,
9817           /*Idx1=*/0);
9818     // If there is no user-defined mapper, set the mapper array to nullptr to
9819     // avoid an unnecessary data privatization
9820     if (!Info.HasMapper)
9821       MappersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
9822     else
9823       MappersArrayArg =
9824           CGF.Builder.CreatePointerCast(Info.MappersArray, CGM.VoidPtrPtrTy);
9825   } else {
9826     BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
9827     PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
9828     SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
9829     MapTypesArrayArg =
9830         llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo());
9831     MapNamesArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
9832     MappersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy);
9833   }
9834 }
9835 
9836 /// Check for inner distribute directive.
9837 static const OMPExecutableDirective *
9838 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) {
9839   const auto *CS = D.getInnermostCapturedStmt();
9840   const auto *Body =
9841       CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true);
9842   const Stmt *ChildStmt =
9843       CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
9844 
9845   if (const auto *NestedDir =
9846           dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
9847     OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind();
9848     switch (D.getDirectiveKind()) {
9849     case OMPD_target:
9850       if (isOpenMPDistributeDirective(DKind))
9851         return NestedDir;
9852       if (DKind == OMPD_teams) {
9853         Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers(
9854             /*IgnoreCaptured=*/true);
9855         if (!Body)
9856           return nullptr;
9857         ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body);
9858         if (const auto *NND =
9859                 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) {
9860           DKind = NND->getDirectiveKind();
9861           if (isOpenMPDistributeDirective(DKind))
9862             return NND;
9863         }
9864       }
9865       return nullptr;
9866     case OMPD_target_teams:
9867       if (isOpenMPDistributeDirective(DKind))
9868         return NestedDir;
9869       return nullptr;
9870     case OMPD_target_parallel:
9871     case OMPD_target_simd:
9872     case OMPD_target_parallel_for:
9873     case OMPD_target_parallel_for_simd:
9874       return nullptr;
9875     case OMPD_target_teams_distribute:
9876     case OMPD_target_teams_distribute_simd:
9877     case OMPD_target_teams_distribute_parallel_for:
9878     case OMPD_target_teams_distribute_parallel_for_simd:
9879     case OMPD_parallel:
9880     case OMPD_for:
9881     case OMPD_parallel_for:
9882     case OMPD_parallel_master:
9883     case OMPD_parallel_sections:
9884     case OMPD_for_simd:
9885     case OMPD_parallel_for_simd:
9886     case OMPD_cancel:
9887     case OMPD_cancellation_point:
9888     case OMPD_ordered:
9889     case OMPD_threadprivate:
9890     case OMPD_allocate:
9891     case OMPD_task:
9892     case OMPD_simd:
9893     case OMPD_tile:
9894     case OMPD_unroll:
9895     case OMPD_sections:
9896     case OMPD_section:
9897     case OMPD_single:
9898     case OMPD_master:
9899     case OMPD_critical:
9900     case OMPD_taskyield:
9901     case OMPD_barrier:
9902     case OMPD_taskwait:
9903     case OMPD_taskgroup:
9904     case OMPD_atomic:
9905     case OMPD_flush:
9906     case OMPD_depobj:
9907     case OMPD_scan:
9908     case OMPD_teams:
9909     case OMPD_target_data:
9910     case OMPD_target_exit_data:
9911     case OMPD_target_enter_data:
9912     case OMPD_distribute:
9913     case OMPD_distribute_simd:
9914     case OMPD_distribute_parallel_for:
9915     case OMPD_distribute_parallel_for_simd:
9916     case OMPD_teams_distribute:
9917     case OMPD_teams_distribute_simd:
9918     case OMPD_teams_distribute_parallel_for:
9919     case OMPD_teams_distribute_parallel_for_simd:
9920     case OMPD_target_update:
9921     case OMPD_declare_simd:
9922     case OMPD_declare_variant:
9923     case OMPD_begin_declare_variant:
9924     case OMPD_end_declare_variant:
9925     case OMPD_declare_target:
9926     case OMPD_end_declare_target:
9927     case OMPD_declare_reduction:
9928     case OMPD_declare_mapper:
9929     case OMPD_taskloop:
9930     case OMPD_taskloop_simd:
9931     case OMPD_master_taskloop:
9932     case OMPD_master_taskloop_simd:
9933     case OMPD_parallel_master_taskloop:
9934     case OMPD_parallel_master_taskloop_simd:
9935     case OMPD_requires:
9936     case OMPD_metadirective:
9937     case OMPD_unknown:
9938     default:
9939       llvm_unreachable("Unexpected directive.");
9940     }
9941   }
9942 
9943   return nullptr;
9944 }
9945 
9946 /// Emit the user-defined mapper function. The code generation follows the
9947 /// pattern in the example below.
9948 /// \code
9949 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle,
9950 ///                                           void *base, void *begin,
9951 ///                                           int64_t size, int64_t type,
9952 ///                                           void *name = nullptr) {
9953 ///   // Allocate space for an array section first or add a base/begin for
9954 ///   // pointer dereference.
9955 ///   if ((size > 1 || (base != begin && maptype.IsPtrAndObj)) &&
9956 ///       !maptype.IsDelete)
9957 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
9958 ///                                 size*sizeof(Ty), clearToFromMember(type));
9959 ///   // Map members.
9960 ///   for (unsigned i = 0; i < size; i++) {
9961 ///     // For each component specified by this mapper:
9962 ///     for (auto c : begin[i]->all_components) {
9963 ///       if (c.hasMapper())
9964 ///         (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size,
9965 ///                       c.arg_type, c.arg_name);
9966 ///       else
9967 ///         __tgt_push_mapper_component(rt_mapper_handle, c.arg_base,
9968 ///                                     c.arg_begin, c.arg_size, c.arg_type,
9969 ///                                     c.arg_name);
9970 ///     }
9971 ///   }
9972 ///   // Delete the array section.
9973 ///   if (size > 1 && maptype.IsDelete)
9974 ///     __tgt_push_mapper_component(rt_mapper_handle, base, begin,
9975 ///                                 size*sizeof(Ty), clearToFromMember(type));
9976 /// }
9977 /// \endcode
9978 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D,
9979                                             CodeGenFunction *CGF) {
9980   if (UDMMap.count(D) > 0)
9981     return;
9982   ASTContext &C = CGM.getContext();
9983   QualType Ty = D->getType();
9984   QualType PtrTy = C.getPointerType(Ty).withRestrict();
9985   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
9986   auto *MapperVarDecl =
9987       cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl());
9988   SourceLocation Loc = D->getLocation();
9989   CharUnits ElementSize = C.getTypeSizeInChars(Ty);
9990   llvm::Type *ElemTy = CGM.getTypes().ConvertTypeForMem(Ty);
9991 
9992   // Prepare mapper function arguments and attributes.
9993   ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
9994                               C.VoidPtrTy, ImplicitParamDecl::Other);
9995   ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
9996                             ImplicitParamDecl::Other);
9997   ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr,
9998                              C.VoidPtrTy, ImplicitParamDecl::Other);
9999   ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
10000                             ImplicitParamDecl::Other);
10001   ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty,
10002                             ImplicitParamDecl::Other);
10003   ImplicitParamDecl NameArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy,
10004                             ImplicitParamDecl::Other);
10005   FunctionArgList Args;
10006   Args.push_back(&HandleArg);
10007   Args.push_back(&BaseArg);
10008   Args.push_back(&BeginArg);
10009   Args.push_back(&SizeArg);
10010   Args.push_back(&TypeArg);
10011   Args.push_back(&NameArg);
10012   const CGFunctionInfo &FnInfo =
10013       CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args);
10014   llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo);
10015   SmallString<64> TyStr;
10016   llvm::raw_svector_ostream Out(TyStr);
10017   CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out);
10018   std::string Name = getName({"omp_mapper", TyStr, D->getName()});
10019   auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage,
10020                                     Name, &CGM.getModule());
10021   CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo);
10022   Fn->removeFnAttr(llvm::Attribute::OptimizeNone);
10023   // Start the mapper function code generation.
10024   CodeGenFunction MapperCGF(CGM);
10025   MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc);
10026   // Compute the starting and end addresses of array elements.
10027   llvm::Value *Size = MapperCGF.EmitLoadOfScalar(
10028       MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false,
10029       C.getPointerType(Int64Ty), Loc);
10030   // Prepare common arguments for array initiation and deletion.
10031   llvm::Value *Handle = MapperCGF.EmitLoadOfScalar(
10032       MapperCGF.GetAddrOfLocalVar(&HandleArg),
10033       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
10034   llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar(
10035       MapperCGF.GetAddrOfLocalVar(&BaseArg),
10036       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
10037   llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar(
10038       MapperCGF.GetAddrOfLocalVar(&BeginArg),
10039       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
10040   // Convert the size in bytes into the number of array elements.
10041   Size = MapperCGF.Builder.CreateExactUDiv(
10042       Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity()));
10043   llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast(
10044       BeginIn, CGM.getTypes().ConvertTypeForMem(PtrTy));
10045   llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(ElemTy, PtrBegin, Size);
10046   llvm::Value *MapType = MapperCGF.EmitLoadOfScalar(
10047       MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false,
10048       C.getPointerType(Int64Ty), Loc);
10049   llvm::Value *MapName = MapperCGF.EmitLoadOfScalar(
10050       MapperCGF.GetAddrOfLocalVar(&NameArg),
10051       /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc);
10052 
10053   // Emit array initiation if this is an array section and \p MapType indicates
10054   // that memory allocation is required.
10055   llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head");
10056   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
10057                              MapName, ElementSize, HeadBB, /*IsInit=*/true);
10058 
10059   // Emit a for loop to iterate through SizeArg of elements and map all of them.
10060 
10061   // Emit the loop header block.
10062   MapperCGF.EmitBlock(HeadBB);
10063   llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body");
10064   llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done");
10065   // Evaluate whether the initial condition is satisfied.
10066   llvm::Value *IsEmpty =
10067       MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty");
10068   MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
10069   llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock();
10070 
10071   // Emit the loop body block.
10072   MapperCGF.EmitBlock(BodyBB);
10073   llvm::BasicBlock *LastBB = BodyBB;
10074   llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI(
10075       PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent");
10076   PtrPHI->addIncoming(PtrBegin, EntryBB);
10077   Address PtrCurrent(PtrPHI, ElemTy,
10078                      MapperCGF.GetAddrOfLocalVar(&BeginArg)
10079                          .getAlignment()
10080                          .alignmentOfArrayElement(ElementSize));
10081   // Privatize the declared variable of mapper to be the current array element.
10082   CodeGenFunction::OMPPrivateScope Scope(MapperCGF);
10083   Scope.addPrivate(MapperVarDecl, PtrCurrent);
10084   (void)Scope.Privatize();
10085 
10086   // Get map clause information. Fill up the arrays with all mapped variables.
10087   MappableExprsHandler::MapCombinedInfoTy Info;
10088   MappableExprsHandler MEHandler(*D, MapperCGF);
10089   MEHandler.generateAllInfoForMapper(Info);
10090 
10091   // Call the runtime API __tgt_mapper_num_components to get the number of
10092   // pre-existing components.
10093   llvm::Value *OffloadingArgs[] = {Handle};
10094   llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall(
10095       OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
10096                                             OMPRTL___tgt_mapper_num_components),
10097       OffloadingArgs);
10098   llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl(
10099       PreviousSize,
10100       MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset()));
10101 
10102   // Fill up the runtime mapper handle for all components.
10103   for (unsigned I = 0; I < Info.BasePointers.size(); ++I) {
10104     llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast(
10105         *Info.BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
10106     llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast(
10107         Info.Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy));
10108     llvm::Value *CurSizeArg = Info.Sizes[I];
10109     llvm::Value *CurNameArg =
10110         (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo)
10111             ? llvm::ConstantPointerNull::get(CGM.VoidPtrTy)
10112             : emitMappingInformation(MapperCGF, OMPBuilder, Info.Exprs[I]);
10113 
10114     // Extract the MEMBER_OF field from the map type.
10115     llvm::Value *OriMapType = MapperCGF.Builder.getInt64(Info.Types[I]);
10116     llvm::Value *MemberMapType =
10117         MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
10118 
10119     // Combine the map type inherited from user-defined mapper with that
10120     // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM
10121     // bits of the \a MapType, which is the input argument of the mapper
10122     // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM
10123     // bits of MemberMapType.
10124     // [OpenMP 5.0], 1.2.6. map-type decay.
10125     //        | alloc |  to   | from  | tofrom | release | delete
10126     // ----------------------------------------------------------
10127     // alloc  | alloc | alloc | alloc | alloc  | release | delete
10128     // to     | alloc |  to   | alloc |   to   | release | delete
10129     // from   | alloc | alloc | from  |  from  | release | delete
10130     // tofrom | alloc |  to   | from  | tofrom | release | delete
10131     llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd(
10132         MapType,
10133         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO |
10134                                    MappableExprsHandler::OMP_MAP_FROM));
10135     llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc");
10136     llvm::BasicBlock *AllocElseBB =
10137         MapperCGF.createBasicBlock("omp.type.alloc.else");
10138     llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to");
10139     llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else");
10140     llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from");
10141     llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end");
10142     llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom);
10143     MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
10144     // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM.
10145     MapperCGF.EmitBlock(AllocBB);
10146     llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd(
10147         MemberMapType,
10148         MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
10149                                      MappableExprsHandler::OMP_MAP_FROM)));
10150     MapperCGF.Builder.CreateBr(EndBB);
10151     MapperCGF.EmitBlock(AllocElseBB);
10152     llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ(
10153         LeftToFrom,
10154         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO));
10155     MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
10156     // In case of to, clear OMP_MAP_FROM.
10157     MapperCGF.EmitBlock(ToBB);
10158     llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd(
10159         MemberMapType,
10160         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM));
10161     MapperCGF.Builder.CreateBr(EndBB);
10162     MapperCGF.EmitBlock(ToElseBB);
10163     llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ(
10164         LeftToFrom,
10165         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM));
10166     MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB);
10167     // In case of from, clear OMP_MAP_TO.
10168     MapperCGF.EmitBlock(FromBB);
10169     llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd(
10170         MemberMapType,
10171         MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO));
10172     // In case of tofrom, do nothing.
10173     MapperCGF.EmitBlock(EndBB);
10174     LastBB = EndBB;
10175     llvm::PHINode *CurMapType =
10176         MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype");
10177     CurMapType->addIncoming(AllocMapType, AllocBB);
10178     CurMapType->addIncoming(ToMapType, ToBB);
10179     CurMapType->addIncoming(FromMapType, FromBB);
10180     CurMapType->addIncoming(MemberMapType, ToElseBB);
10181 
10182     llvm::Value *OffloadingArgs[] = {Handle,     CurBaseArg, CurBeginArg,
10183                                      CurSizeArg, CurMapType, CurNameArg};
10184     if (Info.Mappers[I]) {
10185       // Call the corresponding mapper function.
10186       llvm::Function *MapperFunc = getOrCreateUserDefinedMapperFunc(
10187           cast<OMPDeclareMapperDecl>(Info.Mappers[I]));
10188       assert(MapperFunc && "Expect a valid mapper function is available.");
10189       MapperCGF.EmitNounwindRuntimeCall(MapperFunc, OffloadingArgs);
10190     } else {
10191       // Call the runtime API __tgt_push_mapper_component to fill up the runtime
10192       // data structure.
10193       MapperCGF.EmitRuntimeCall(
10194           OMPBuilder.getOrCreateRuntimeFunction(
10195               CGM.getModule(), OMPRTL___tgt_push_mapper_component),
10196           OffloadingArgs);
10197     }
10198   }
10199 
10200   // Update the pointer to point to the next element that needs to be mapped,
10201   // and check whether we have mapped all elements.
10202   llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32(
10203       ElemTy, PtrPHI, /*Idx0=*/1, "omp.arraymap.next");
10204   PtrPHI->addIncoming(PtrNext, LastBB);
10205   llvm::Value *IsDone =
10206       MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone");
10207   llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit");
10208   MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
10209 
10210   MapperCGF.EmitBlock(ExitBB);
10211   // Emit array deletion if this is an array section and \p MapType indicates
10212   // that deletion is required.
10213   emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType,
10214                              MapName, ElementSize, DoneBB, /*IsInit=*/false);
10215 
10216   // Emit the function exit block.
10217   MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true);
10218   MapperCGF.FinishFunction();
10219   UDMMap.try_emplace(D, Fn);
10220   if (CGF) {
10221     auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn);
10222     Decls.second.push_back(D);
10223   }
10224 }
10225 
10226 /// Emit the array initialization or deletion portion for user-defined mapper
10227 /// code generation. First, it evaluates whether an array section is mapped and
10228 /// whether the \a MapType instructs to delete this section. If \a IsInit is
10229 /// true, and \a MapType indicates to not delete this array, array
10230 /// initialization code is generated. If \a IsInit is false, and \a MapType
10231 /// indicates to not this array, array deletion code is generated.
10232 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel(
10233     CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base,
10234     llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType,
10235     llvm::Value *MapName, CharUnits ElementSize, llvm::BasicBlock *ExitBB,
10236     bool IsInit) {
10237   StringRef Prefix = IsInit ? ".init" : ".del";
10238 
10239   // Evaluate if this is an array section.
10240   llvm::BasicBlock *BodyBB =
10241       MapperCGF.createBasicBlock(getName({"omp.array", Prefix}));
10242   llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGT(
10243       Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray");
10244   llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd(
10245       MapType,
10246       MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE));
10247   llvm::Value *DeleteCond;
10248   llvm::Value *Cond;
10249   if (IsInit) {
10250     // base != begin?
10251     llvm::Value *BaseIsBegin = MapperCGF.Builder.CreateICmpNE(Base, Begin);
10252     // IsPtrAndObj?
10253     llvm::Value *PtrAndObjBit = MapperCGF.Builder.CreateAnd(
10254         MapType,
10255         MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_PTR_AND_OBJ));
10256     PtrAndObjBit = MapperCGF.Builder.CreateIsNotNull(PtrAndObjBit);
10257     BaseIsBegin = MapperCGF.Builder.CreateAnd(BaseIsBegin, PtrAndObjBit);
10258     Cond = MapperCGF.Builder.CreateOr(IsArray, BaseIsBegin);
10259     DeleteCond = MapperCGF.Builder.CreateIsNull(
10260         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
10261   } else {
10262     Cond = IsArray;
10263     DeleteCond = MapperCGF.Builder.CreateIsNotNull(
10264         DeleteBit, getName({"omp.array", Prefix, ".delete"}));
10265   }
10266   Cond = MapperCGF.Builder.CreateAnd(Cond, DeleteCond);
10267   MapperCGF.Builder.CreateCondBr(Cond, BodyBB, ExitBB);
10268 
10269   MapperCGF.EmitBlock(BodyBB);
10270   // Get the array size by multiplying element size and element number (i.e., \p
10271   // Size).
10272   llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul(
10273       Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity()));
10274   // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves
10275   // memory allocation/deletion purpose only.
10276   llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd(
10277       MapType,
10278       MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO |
10279                                    MappableExprsHandler::OMP_MAP_FROM)));
10280   MapTypeArg = MapperCGF.Builder.CreateOr(
10281       MapTypeArg,
10282       MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_IMPLICIT));
10283 
10284   // Call the runtime API __tgt_push_mapper_component to fill up the runtime
10285   // data structure.
10286   llvm::Value *OffloadingArgs[] = {Handle,    Base,       Begin,
10287                                    ArraySize, MapTypeArg, MapName};
10288   MapperCGF.EmitRuntimeCall(
10289       OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
10290                                             OMPRTL___tgt_push_mapper_component),
10291       OffloadingArgs);
10292 }
10293 
10294 llvm::Function *CGOpenMPRuntime::getOrCreateUserDefinedMapperFunc(
10295     const OMPDeclareMapperDecl *D) {
10296   auto I = UDMMap.find(D);
10297   if (I != UDMMap.end())
10298     return I->second;
10299   emitUserDefinedMapper(D);
10300   return UDMMap.lookup(D);
10301 }
10302 
10303 void CGOpenMPRuntime::emitTargetNumIterationsCall(
10304     CodeGenFunction &CGF, const OMPExecutableDirective &D,
10305     llvm::Value *DeviceID,
10306     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
10307                                      const OMPLoopDirective &D)>
10308         SizeEmitter) {
10309   OpenMPDirectiveKind Kind = D.getDirectiveKind();
10310   const OMPExecutableDirective *TD = &D;
10311   // Get nested teams distribute kind directive, if any.
10312   if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind))
10313     TD = getNestedDistributeDirective(CGM.getContext(), D);
10314   if (!TD)
10315     return;
10316   const auto *LD = cast<OMPLoopDirective>(TD);
10317   auto &&CodeGen = [LD, DeviceID, SizeEmitter, &D, this](CodeGenFunction &CGF,
10318                                                          PrePostActionTy &) {
10319     if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) {
10320       llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc());
10321       llvm::Value *Args[] = {RTLoc, DeviceID, NumIterations};
10322       CGF.EmitRuntimeCall(
10323           OMPBuilder.getOrCreateRuntimeFunction(
10324               CGM.getModule(), OMPRTL___kmpc_push_target_tripcount_mapper),
10325           Args);
10326     }
10327   };
10328   emitInlinedDirective(CGF, OMPD_unknown, CodeGen);
10329 }
10330 
10331 void CGOpenMPRuntime::emitTargetCall(
10332     CodeGenFunction &CGF, const OMPExecutableDirective &D,
10333     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
10334     llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
10335     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
10336                                      const OMPLoopDirective &D)>
10337         SizeEmitter) {
10338   if (!CGF.HaveInsertPoint())
10339     return;
10340 
10341   const bool OffloadingMandatory = !CGM.getLangOpts().OpenMPIsDevice &&
10342                                    CGM.getLangOpts().OpenMPOffloadMandatory;
10343 
10344   assert((OffloadingMandatory || OutlinedFn) && "Invalid outlined function!");
10345 
10346   const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() ||
10347                                  D.hasClausesOfKind<OMPNowaitClause>();
10348   llvm::SmallVector<llvm::Value *, 16> CapturedVars;
10349   const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target);
10350   auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF,
10351                                             PrePostActionTy &) {
10352     CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
10353   };
10354   emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen);
10355 
10356   CodeGenFunction::OMPTargetDataInfo InputInfo;
10357   llvm::Value *MapTypesArray = nullptr;
10358   llvm::Value *MapNamesArray = nullptr;
10359   // Generate code for the host fallback function.
10360   auto &&FallbackGen = [this, OutlinedFn, &D, &CapturedVars, RequiresOuterTask,
10361                         &CS, OffloadingMandatory](CodeGenFunction &CGF) {
10362     if (OffloadingMandatory) {
10363       CGF.Builder.CreateUnreachable();
10364     } else {
10365       if (RequiresOuterTask) {
10366         CapturedVars.clear();
10367         CGF.GenerateOpenMPCapturedVars(CS, CapturedVars);
10368       }
10369       emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars);
10370     }
10371   };
10372   // Fill up the pointer arrays and transfer execution to the device.
10373   auto &&ThenGen = [this, Device, OutlinedFnID, &D, &InputInfo, &MapTypesArray,
10374                     &MapNamesArray, SizeEmitter,
10375                     FallbackGen](CodeGenFunction &CGF, PrePostActionTy &) {
10376     if (Device.getInt() == OMPC_DEVICE_ancestor) {
10377       // Reverse offloading is not supported, so just execute on the host.
10378       FallbackGen(CGF);
10379       return;
10380     }
10381 
10382     // On top of the arrays that were filled up, the target offloading call
10383     // takes as arguments the device id as well as the host pointer. The host
10384     // pointer is used by the runtime library to identify the current target
10385     // region, so it only has to be unique and not necessarily point to
10386     // anything. It could be the pointer to the outlined function that
10387     // implements the target region, but we aren't using that so that the
10388     // compiler doesn't need to keep that, and could therefore inline the host
10389     // function if proven worthwhile during optimization.
10390 
10391     // From this point on, we need to have an ID of the target region defined.
10392     assert(OutlinedFnID && "Invalid outlined function ID!");
10393     (void)OutlinedFnID;
10394 
10395     // Emit device ID if any.
10396     llvm::Value *DeviceID;
10397     if (Device.getPointer()) {
10398       assert((Device.getInt() == OMPC_DEVICE_unknown ||
10399               Device.getInt() == OMPC_DEVICE_device_num) &&
10400              "Expected device_num modifier.");
10401       llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer());
10402       DeviceID =
10403           CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true);
10404     } else {
10405       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
10406     }
10407 
10408     // Emit the number of elements in the offloading arrays.
10409     llvm::Value *PointerNum =
10410         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
10411 
10412     // Return value of the runtime offloading call.
10413     llvm::Value *Return;
10414 
10415     llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D);
10416     llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D);
10417 
10418     // Source location for the ident struct
10419     llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc());
10420 
10421     // Emit tripcount for the target loop-based directive.
10422     emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter);
10423 
10424     bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
10425     // The target region is an outlined function launched by the runtime
10426     // via calls __tgt_target() or __tgt_target_teams().
10427     //
10428     // __tgt_target() launches a target region with one team and one thread,
10429     // executing a serial region.  This master thread may in turn launch
10430     // more threads within its team upon encountering a parallel region,
10431     // however, no additional teams can be launched on the device.
10432     //
10433     // __tgt_target_teams() launches a target region with one or more teams,
10434     // each with one or more threads.  This call is required for target
10435     // constructs such as:
10436     //  'target teams'
10437     //  'target' / 'teams'
10438     //  'target teams distribute parallel for'
10439     //  'target parallel'
10440     // and so on.
10441     //
10442     // Note that on the host and CPU targets, the runtime implementation of
10443     // these calls simply call the outlined function without forking threads.
10444     // The outlined functions themselves have runtime calls to
10445     // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by
10446     // the compiler in emitTeamsCall() and emitParallelCall().
10447     //
10448     // In contrast, on the NVPTX target, the implementation of
10449     // __tgt_target_teams() launches a GPU kernel with the requested number
10450     // of teams and threads so no additional calls to the runtime are required.
10451     if (NumTeams) {
10452       // If we have NumTeams defined this means that we have an enclosed teams
10453       // region. Therefore we also expect to have NumThreads defined. These two
10454       // values should be defined in the presence of a teams directive,
10455       // regardless of having any clauses associated. If the user is using teams
10456       // but no clauses, these two values will be the default that should be
10457       // passed to the runtime library - a 32-bit integer with the value zero.
10458       assert(NumThreads && "Thread limit expression should be available along "
10459                            "with number of teams.");
10460       SmallVector<llvm::Value *> OffloadingArgs = {
10461           RTLoc,
10462           DeviceID,
10463           OutlinedFnID,
10464           PointerNum,
10465           InputInfo.BasePointersArray.getPointer(),
10466           InputInfo.PointersArray.getPointer(),
10467           InputInfo.SizesArray.getPointer(),
10468           MapTypesArray,
10469           MapNamesArray,
10470           InputInfo.MappersArray.getPointer(),
10471           NumTeams,
10472           NumThreads};
10473       if (HasNowait) {
10474         // Add int32_t depNum = 0, void *depList = nullptr, int32_t
10475         // noAliasDepNum = 0, void *noAliasDepList = nullptr.
10476         OffloadingArgs.push_back(CGF.Builder.getInt32(0));
10477         OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy));
10478         OffloadingArgs.push_back(CGF.Builder.getInt32(0));
10479         OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy));
10480       }
10481       Return = CGF.EmitRuntimeCall(
10482           OMPBuilder.getOrCreateRuntimeFunction(
10483               CGM.getModule(), HasNowait
10484                                    ? OMPRTL___tgt_target_teams_nowait_mapper
10485                                    : OMPRTL___tgt_target_teams_mapper),
10486           OffloadingArgs);
10487     } else {
10488       SmallVector<llvm::Value *> OffloadingArgs = {
10489           RTLoc,
10490           DeviceID,
10491           OutlinedFnID,
10492           PointerNum,
10493           InputInfo.BasePointersArray.getPointer(),
10494           InputInfo.PointersArray.getPointer(),
10495           InputInfo.SizesArray.getPointer(),
10496           MapTypesArray,
10497           MapNamesArray,
10498           InputInfo.MappersArray.getPointer()};
10499       if (HasNowait) {
10500         // Add int32_t depNum = 0, void *depList = nullptr, int32_t
10501         // noAliasDepNum = 0, void *noAliasDepList = nullptr.
10502         OffloadingArgs.push_back(CGF.Builder.getInt32(0));
10503         OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy));
10504         OffloadingArgs.push_back(CGF.Builder.getInt32(0));
10505         OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy));
10506       }
10507       Return = CGF.EmitRuntimeCall(
10508           OMPBuilder.getOrCreateRuntimeFunction(
10509               CGM.getModule(), HasNowait ? OMPRTL___tgt_target_nowait_mapper
10510                                          : OMPRTL___tgt_target_mapper),
10511           OffloadingArgs);
10512     }
10513 
10514     // Check the error code and execute the host version if required.
10515     llvm::BasicBlock *OffloadFailedBlock =
10516         CGF.createBasicBlock("omp_offload.failed");
10517     llvm::BasicBlock *OffloadContBlock =
10518         CGF.createBasicBlock("omp_offload.cont");
10519     llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return);
10520     CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock);
10521 
10522     CGF.EmitBlock(OffloadFailedBlock);
10523     FallbackGen(CGF);
10524 
10525     CGF.EmitBranch(OffloadContBlock);
10526 
10527     CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true);
10528   };
10529 
10530   // Notify that the host version must be executed.
10531   auto &&ElseGen = [FallbackGen](CodeGenFunction &CGF, PrePostActionTy &) {
10532     FallbackGen(CGF);
10533   };
10534 
10535   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
10536                           &MapNamesArray, &CapturedVars, RequiresOuterTask,
10537                           &CS](CodeGenFunction &CGF, PrePostActionTy &) {
10538     // Fill up the arrays with all the captured variables.
10539     MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
10540 
10541     // Get mappable expression information.
10542     MappableExprsHandler MEHandler(D, CGF);
10543     llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers;
10544     llvm::DenseSet<CanonicalDeclPtr<const Decl>> MappedVarSet;
10545 
10546     auto RI = CS.getCapturedRecordDecl()->field_begin();
10547     auto *CV = CapturedVars.begin();
10548     for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(),
10549                                               CE = CS.capture_end();
10550          CI != CE; ++CI, ++RI, ++CV) {
10551       MappableExprsHandler::MapCombinedInfoTy CurInfo;
10552       MappableExprsHandler::StructRangeInfoTy PartialStruct;
10553 
10554       // VLA sizes are passed to the outlined region by copy and do not have map
10555       // information associated.
10556       if (CI->capturesVariableArrayType()) {
10557         CurInfo.Exprs.push_back(nullptr);
10558         CurInfo.BasePointers.push_back(*CV);
10559         CurInfo.Pointers.push_back(*CV);
10560         CurInfo.Sizes.push_back(CGF.Builder.CreateIntCast(
10561             CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true));
10562         // Copy to the device as an argument. No need to retrieve it.
10563         CurInfo.Types.push_back(MappableExprsHandler::OMP_MAP_LITERAL |
10564                                 MappableExprsHandler::OMP_MAP_TARGET_PARAM |
10565                                 MappableExprsHandler::OMP_MAP_IMPLICIT);
10566         CurInfo.Mappers.push_back(nullptr);
10567       } else {
10568         // If we have any information in the map clause, we use it, otherwise we
10569         // just do a default mapping.
10570         MEHandler.generateInfoForCapture(CI, *CV, CurInfo, PartialStruct);
10571         if (!CI->capturesThis())
10572           MappedVarSet.insert(CI->getCapturedVar());
10573         else
10574           MappedVarSet.insert(nullptr);
10575         if (CurInfo.BasePointers.empty() && !PartialStruct.Base.isValid())
10576           MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurInfo);
10577         // Generate correct mapping for variables captured by reference in
10578         // lambdas.
10579         if (CI->capturesVariable())
10580           MEHandler.generateInfoForLambdaCaptures(CI->getCapturedVar(), *CV,
10581                                                   CurInfo, LambdaPointers);
10582       }
10583       // We expect to have at least an element of information for this capture.
10584       assert((!CurInfo.BasePointers.empty() || PartialStruct.Base.isValid()) &&
10585              "Non-existing map pointer for capture!");
10586       assert(CurInfo.BasePointers.size() == CurInfo.Pointers.size() &&
10587              CurInfo.BasePointers.size() == CurInfo.Sizes.size() &&
10588              CurInfo.BasePointers.size() == CurInfo.Types.size() &&
10589              CurInfo.BasePointers.size() == CurInfo.Mappers.size() &&
10590              "Inconsistent map information sizes!");
10591 
10592       // If there is an entry in PartialStruct it means we have a struct with
10593       // individual members mapped. Emit an extra combined entry.
10594       if (PartialStruct.Base.isValid()) {
10595         CombinedInfo.append(PartialStruct.PreliminaryMapData);
10596         MEHandler.emitCombinedEntry(
10597             CombinedInfo, CurInfo.Types, PartialStruct, nullptr,
10598             !PartialStruct.PreliminaryMapData.BasePointers.empty());
10599       }
10600 
10601       // We need to append the results of this capture to what we already have.
10602       CombinedInfo.append(CurInfo);
10603     }
10604     // Adjust MEMBER_OF flags for the lambdas captures.
10605     MEHandler.adjustMemberOfForLambdaCaptures(
10606         LambdaPointers, CombinedInfo.BasePointers, CombinedInfo.Pointers,
10607         CombinedInfo.Types);
10608     // Map any list items in a map clause that were not captures because they
10609     // weren't referenced within the construct.
10610     MEHandler.generateAllInfo(CombinedInfo, MappedVarSet);
10611 
10612     TargetDataInfo Info;
10613     // Fill up the arrays and create the arguments.
10614     emitOffloadingArrays(CGF, CombinedInfo, Info, OMPBuilder);
10615     emitOffloadingArraysArgument(
10616         CGF, Info.BasePointersArray, Info.PointersArray, Info.SizesArray,
10617         Info.MapTypesArray, Info.MapNamesArray, Info.MappersArray, Info,
10618         {/*ForEndCall=*/false});
10619 
10620     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
10621     InputInfo.BasePointersArray =
10622         Address(Info.BasePointersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
10623     InputInfo.PointersArray =
10624         Address(Info.PointersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
10625     InputInfo.SizesArray =
10626         Address(Info.SizesArray, CGF.Int64Ty, CGM.getPointerAlign());
10627     InputInfo.MappersArray =
10628         Address(Info.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
10629     MapTypesArray = Info.MapTypesArray;
10630     MapNamesArray = Info.MapNamesArray;
10631     if (RequiresOuterTask)
10632       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
10633     else
10634       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
10635   };
10636 
10637   auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask](
10638                              CodeGenFunction &CGF, PrePostActionTy &) {
10639     if (RequiresOuterTask) {
10640       CodeGenFunction::OMPTargetDataInfo InputInfo;
10641       CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo);
10642     } else {
10643       emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen);
10644     }
10645   };
10646 
10647   // If we have a target function ID it means that we need to support
10648   // offloading, otherwise, just execute on the host. We need to execute on host
10649   // regardless of the conditional in the if clause if, e.g., the user do not
10650   // specify target triples.
10651   if (OutlinedFnID) {
10652     if (IfCond) {
10653       emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen);
10654     } else {
10655       RegionCodeGenTy ThenRCG(TargetThenGen);
10656       ThenRCG(CGF);
10657     }
10658   } else {
10659     RegionCodeGenTy ElseRCG(TargetElseGen);
10660     ElseRCG(CGF);
10661   }
10662 }
10663 
10664 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S,
10665                                                     StringRef ParentName) {
10666   if (!S)
10667     return;
10668 
10669   // Codegen OMP target directives that offload compute to the device.
10670   bool RequiresDeviceCodegen =
10671       isa<OMPExecutableDirective>(S) &&
10672       isOpenMPTargetExecutionDirective(
10673           cast<OMPExecutableDirective>(S)->getDirectiveKind());
10674 
10675   if (RequiresDeviceCodegen) {
10676     const auto &E = *cast<OMPExecutableDirective>(S);
10677     unsigned DeviceID;
10678     unsigned FileID;
10679     unsigned Line;
10680     getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID,
10681                              FileID, Line);
10682 
10683     // Is this a target region that should not be emitted as an entry point? If
10684     // so just signal we are done with this target region.
10685     if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID,
10686                                                             ParentName, Line))
10687       return;
10688 
10689     switch (E.getDirectiveKind()) {
10690     case OMPD_target:
10691       CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName,
10692                                                    cast<OMPTargetDirective>(E));
10693       break;
10694     case OMPD_target_parallel:
10695       CodeGenFunction::EmitOMPTargetParallelDeviceFunction(
10696           CGM, ParentName, cast<OMPTargetParallelDirective>(E));
10697       break;
10698     case OMPD_target_teams:
10699       CodeGenFunction::EmitOMPTargetTeamsDeviceFunction(
10700           CGM, ParentName, cast<OMPTargetTeamsDirective>(E));
10701       break;
10702     case OMPD_target_teams_distribute:
10703       CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction(
10704           CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E));
10705       break;
10706     case OMPD_target_teams_distribute_simd:
10707       CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction(
10708           CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E));
10709       break;
10710     case OMPD_target_parallel_for:
10711       CodeGenFunction::EmitOMPTargetParallelForDeviceFunction(
10712           CGM, ParentName, cast<OMPTargetParallelForDirective>(E));
10713       break;
10714     case OMPD_target_parallel_for_simd:
10715       CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction(
10716           CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E));
10717       break;
10718     case OMPD_target_simd:
10719       CodeGenFunction::EmitOMPTargetSimdDeviceFunction(
10720           CGM, ParentName, cast<OMPTargetSimdDirective>(E));
10721       break;
10722     case OMPD_target_teams_distribute_parallel_for:
10723       CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction(
10724           CGM, ParentName,
10725           cast<OMPTargetTeamsDistributeParallelForDirective>(E));
10726       break;
10727     case OMPD_target_teams_distribute_parallel_for_simd:
10728       CodeGenFunction::
10729           EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction(
10730               CGM, ParentName,
10731               cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E));
10732       break;
10733     case OMPD_parallel:
10734     case OMPD_for:
10735     case OMPD_parallel_for:
10736     case OMPD_parallel_master:
10737     case OMPD_parallel_sections:
10738     case OMPD_for_simd:
10739     case OMPD_parallel_for_simd:
10740     case OMPD_cancel:
10741     case OMPD_cancellation_point:
10742     case OMPD_ordered:
10743     case OMPD_threadprivate:
10744     case OMPD_allocate:
10745     case OMPD_task:
10746     case OMPD_simd:
10747     case OMPD_tile:
10748     case OMPD_unroll:
10749     case OMPD_sections:
10750     case OMPD_section:
10751     case OMPD_single:
10752     case OMPD_master:
10753     case OMPD_critical:
10754     case OMPD_taskyield:
10755     case OMPD_barrier:
10756     case OMPD_taskwait:
10757     case OMPD_taskgroup:
10758     case OMPD_atomic:
10759     case OMPD_flush:
10760     case OMPD_depobj:
10761     case OMPD_scan:
10762     case OMPD_teams:
10763     case OMPD_target_data:
10764     case OMPD_target_exit_data:
10765     case OMPD_target_enter_data:
10766     case OMPD_distribute:
10767     case OMPD_distribute_simd:
10768     case OMPD_distribute_parallel_for:
10769     case OMPD_distribute_parallel_for_simd:
10770     case OMPD_teams_distribute:
10771     case OMPD_teams_distribute_simd:
10772     case OMPD_teams_distribute_parallel_for:
10773     case OMPD_teams_distribute_parallel_for_simd:
10774     case OMPD_target_update:
10775     case OMPD_declare_simd:
10776     case OMPD_declare_variant:
10777     case OMPD_begin_declare_variant:
10778     case OMPD_end_declare_variant:
10779     case OMPD_declare_target:
10780     case OMPD_end_declare_target:
10781     case OMPD_declare_reduction:
10782     case OMPD_declare_mapper:
10783     case OMPD_taskloop:
10784     case OMPD_taskloop_simd:
10785     case OMPD_master_taskloop:
10786     case OMPD_master_taskloop_simd:
10787     case OMPD_parallel_master_taskloop:
10788     case OMPD_parallel_master_taskloop_simd:
10789     case OMPD_requires:
10790     case OMPD_metadirective:
10791     case OMPD_unknown:
10792     default:
10793       llvm_unreachable("Unknown target directive for OpenMP device codegen.");
10794     }
10795     return;
10796   }
10797 
10798   if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) {
10799     if (!E->hasAssociatedStmt() || !E->getAssociatedStmt())
10800       return;
10801 
10802     scanForTargetRegionsFunctions(E->getRawStmt(), ParentName);
10803     return;
10804   }
10805 
10806   // If this is a lambda function, look into its body.
10807   if (const auto *L = dyn_cast<LambdaExpr>(S))
10808     S = L->getBody();
10809 
10810   // Keep looking for target regions recursively.
10811   for (const Stmt *II : S->children())
10812     scanForTargetRegionsFunctions(II, ParentName);
10813 }
10814 
10815 static bool isAssumedToBeNotEmitted(const ValueDecl *VD, bool IsDevice) {
10816   Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
10817       OMPDeclareTargetDeclAttr::getDeviceType(VD);
10818   if (!DevTy)
10819     return false;
10820   // Do not emit device_type(nohost) functions for the host.
10821   if (!IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_NoHost)
10822     return true;
10823   // Do not emit device_type(host) functions for the device.
10824   if (IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_Host)
10825     return true;
10826   return false;
10827 }
10828 
10829 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) {
10830   // If emitting code for the host, we do not process FD here. Instead we do
10831   // the normal code generation.
10832   if (!CGM.getLangOpts().OpenMPIsDevice) {
10833     if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl()))
10834       if (isAssumedToBeNotEmitted(cast<ValueDecl>(FD),
10835                                   CGM.getLangOpts().OpenMPIsDevice))
10836         return true;
10837     return false;
10838   }
10839 
10840   const ValueDecl *VD = cast<ValueDecl>(GD.getDecl());
10841   // Try to detect target regions in the function.
10842   if (const auto *FD = dyn_cast<FunctionDecl>(VD)) {
10843     StringRef Name = CGM.getMangledName(GD);
10844     scanForTargetRegionsFunctions(FD->getBody(), Name);
10845     if (isAssumedToBeNotEmitted(cast<ValueDecl>(FD),
10846                                 CGM.getLangOpts().OpenMPIsDevice))
10847       return true;
10848   }
10849 
10850   // Do not to emit function if it is not marked as declare target.
10851   return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) &&
10852          AlreadyEmittedTargetDecls.count(VD) == 0;
10853 }
10854 
10855 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
10856   if (isAssumedToBeNotEmitted(cast<ValueDecl>(GD.getDecl()),
10857                               CGM.getLangOpts().OpenMPIsDevice))
10858     return true;
10859 
10860   if (!CGM.getLangOpts().OpenMPIsDevice)
10861     return false;
10862 
10863   // Check if there are Ctors/Dtors in this declaration and look for target
10864   // regions in it. We use the complete variant to produce the kernel name
10865   // mangling.
10866   QualType RDTy = cast<VarDecl>(GD.getDecl())->getType();
10867   if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) {
10868     for (const CXXConstructorDecl *Ctor : RD->ctors()) {
10869       StringRef ParentName =
10870           CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete));
10871       scanForTargetRegionsFunctions(Ctor->getBody(), ParentName);
10872     }
10873     if (const CXXDestructorDecl *Dtor = RD->getDestructor()) {
10874       StringRef ParentName =
10875           CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete));
10876       scanForTargetRegionsFunctions(Dtor->getBody(), ParentName);
10877     }
10878   }
10879 
10880   // Do not to emit variable if it is not marked as declare target.
10881   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
10882       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(
10883           cast<VarDecl>(GD.getDecl()));
10884   if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link ||
10885       (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10886        HasRequiresUnifiedSharedMemory)) {
10887     DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl()));
10888     return true;
10889   }
10890   return false;
10891 }
10892 
10893 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD,
10894                                                    llvm::Constant *Addr) {
10895   if (CGM.getLangOpts().OMPTargetTriples.empty() &&
10896       !CGM.getLangOpts().OpenMPIsDevice)
10897     return;
10898 
10899   // If we have host/nohost variables, they do not need to be registered.
10900   Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy =
10901       OMPDeclareTargetDeclAttr::getDeviceType(VD);
10902   if (DevTy && DevTy.getValue() != OMPDeclareTargetDeclAttr::DT_Any)
10903     return;
10904 
10905   llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
10906       OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
10907   if (!Res) {
10908     if (CGM.getLangOpts().OpenMPIsDevice) {
10909       // Register non-target variables being emitted in device code (debug info
10910       // may cause this).
10911       StringRef VarName = CGM.getMangledName(VD);
10912       EmittedNonTargetVariables.try_emplace(VarName, Addr);
10913     }
10914     return;
10915   }
10916   // Register declare target variables.
10917   OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags;
10918   StringRef VarName;
10919   CharUnits VarSize;
10920   llvm::GlobalValue::LinkageTypes Linkage;
10921 
10922   if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10923       !HasRequiresUnifiedSharedMemory) {
10924     Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
10925     VarName = CGM.getMangledName(VD);
10926     if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) {
10927       VarSize = CGM.getContext().getTypeSizeInChars(VD->getType());
10928       assert(!VarSize.isZero() && "Expected non-zero size of the variable");
10929     } else {
10930       VarSize = CharUnits::Zero();
10931     }
10932     Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false);
10933     // Temp solution to prevent optimizations of the internal variables.
10934     if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) {
10935       // Do not create a "ref-variable" if the original is not also available
10936       // on the host.
10937       if (!OffloadEntriesInfoManager.hasDeviceGlobalVarEntryInfo(VarName))
10938         return;
10939       std::string RefName = getName({VarName, "ref"});
10940       if (!CGM.GetGlobalValue(RefName)) {
10941         llvm::Constant *AddrRef =
10942             getOrCreateInternalVariable(Addr->getType(), RefName);
10943         auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef);
10944         GVAddrRef->setConstant(/*Val=*/true);
10945         GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage);
10946         GVAddrRef->setInitializer(Addr);
10947         CGM.addCompilerUsedGlobal(GVAddrRef);
10948       }
10949     }
10950   } else {
10951     assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) ||
10952             (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10953              HasRequiresUnifiedSharedMemory)) &&
10954            "Declare target attribute must link or to with unified memory.");
10955     if (*Res == OMPDeclareTargetDeclAttr::MT_Link)
10956       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink;
10957     else
10958       Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo;
10959 
10960     if (CGM.getLangOpts().OpenMPIsDevice) {
10961       VarName = Addr->getName();
10962       Addr = nullptr;
10963     } else {
10964       VarName = getAddrOfDeclareTargetVar(VD).getName();
10965       Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer());
10966     }
10967     VarSize = CGM.getPointerSize();
10968     Linkage = llvm::GlobalValue::WeakAnyLinkage;
10969   }
10970 
10971   OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo(
10972       VarName, Addr, VarSize, Flags, Linkage);
10973 }
10974 
10975 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) {
10976   if (isa<FunctionDecl>(GD.getDecl()) ||
10977       isa<OMPDeclareReductionDecl>(GD.getDecl()))
10978     return emitTargetFunctions(GD);
10979 
10980   return emitTargetGlobalVariable(GD);
10981 }
10982 
10983 void CGOpenMPRuntime::emitDeferredTargetDecls() const {
10984   for (const VarDecl *VD : DeferredGlobalVariables) {
10985     llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res =
10986         OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD);
10987     if (!Res)
10988       continue;
10989     if (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10990         !HasRequiresUnifiedSharedMemory) {
10991       CGM.EmitGlobal(VD);
10992     } else {
10993       assert((*Res == OMPDeclareTargetDeclAttr::MT_Link ||
10994               (*Res == OMPDeclareTargetDeclAttr::MT_To &&
10995                HasRequiresUnifiedSharedMemory)) &&
10996              "Expected link clause or to clause with unified memory.");
10997       (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD);
10998     }
10999   }
11000 }
11001 
11002 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas(
11003     CodeGenFunction &CGF, const OMPExecutableDirective &D) const {
11004   assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) &&
11005          " Expected target-based directive.");
11006 }
11007 
11008 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) {
11009   for (const OMPClause *Clause : D->clauselists()) {
11010     if (Clause->getClauseKind() == OMPC_unified_shared_memory) {
11011       HasRequiresUnifiedSharedMemory = true;
11012     } else if (const auto *AC =
11013                    dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) {
11014       switch (AC->getAtomicDefaultMemOrderKind()) {
11015       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel:
11016         RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease;
11017         break;
11018       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst:
11019         RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent;
11020         break;
11021       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed:
11022         RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic;
11023         break;
11024       case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown:
11025         break;
11026       }
11027     }
11028   }
11029 }
11030 
11031 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const {
11032   return RequiresAtomicOrdering;
11033 }
11034 
11035 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD,
11036                                                        LangAS &AS) {
11037   if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>())
11038     return false;
11039   const auto *A = VD->getAttr<OMPAllocateDeclAttr>();
11040   switch(A->getAllocatorType()) {
11041   case OMPAllocateDeclAttr::OMPNullMemAlloc:
11042   case OMPAllocateDeclAttr::OMPDefaultMemAlloc:
11043   // Not supported, fallback to the default mem space.
11044   case OMPAllocateDeclAttr::OMPLargeCapMemAlloc:
11045   case OMPAllocateDeclAttr::OMPCGroupMemAlloc:
11046   case OMPAllocateDeclAttr::OMPHighBWMemAlloc:
11047   case OMPAllocateDeclAttr::OMPLowLatMemAlloc:
11048   case OMPAllocateDeclAttr::OMPThreadMemAlloc:
11049   case OMPAllocateDeclAttr::OMPConstMemAlloc:
11050   case OMPAllocateDeclAttr::OMPPTeamMemAlloc:
11051     AS = LangAS::Default;
11052     return true;
11053   case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc:
11054     llvm_unreachable("Expected predefined allocator for the variables with the "
11055                      "static storage.");
11056   }
11057   return false;
11058 }
11059 
11060 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const {
11061   return HasRequiresUnifiedSharedMemory;
11062 }
11063 
11064 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII(
11065     CodeGenModule &CGM)
11066     : CGM(CGM) {
11067   if (CGM.getLangOpts().OpenMPIsDevice) {
11068     SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal;
11069     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false;
11070   }
11071 }
11072 
11073 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() {
11074   if (CGM.getLangOpts().OpenMPIsDevice)
11075     CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal;
11076 }
11077 
11078 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) {
11079   if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal)
11080     return true;
11081 
11082   const auto *D = cast<FunctionDecl>(GD.getDecl());
11083   // Do not to emit function if it is marked as declare target as it was already
11084   // emitted.
11085   if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) {
11086     if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) {
11087       if (auto *F = dyn_cast_or_null<llvm::Function>(
11088               CGM.GetGlobalValue(CGM.getMangledName(GD))))
11089         return !F->isDeclaration();
11090       return false;
11091     }
11092     return true;
11093   }
11094 
11095   return !AlreadyEmittedTargetDecls.insert(D).second;
11096 }
11097 
11098 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() {
11099   // If we don't have entries or if we are emitting code for the device, we
11100   // don't need to do anything.
11101   if (CGM.getLangOpts().OMPTargetTriples.empty() ||
11102       CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice ||
11103       (OffloadEntriesInfoManager.empty() &&
11104        !HasEmittedDeclareTargetRegion &&
11105        !HasEmittedTargetRegion))
11106     return nullptr;
11107 
11108   // Create and register the function that handles the requires directives.
11109   ASTContext &C = CGM.getContext();
11110 
11111   llvm::Function *RequiresRegFn;
11112   {
11113     CodeGenFunction CGF(CGM);
11114     const auto &FI = CGM.getTypes().arrangeNullaryFunction();
11115     llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI);
11116     std::string ReqName = getName({"omp_offloading", "requires_reg"});
11117     RequiresRegFn = CGM.CreateGlobalInitOrCleanUpFunction(FTy, ReqName, FI);
11118     CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {});
11119     OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE;
11120     // TODO: check for other requires clauses.
11121     // The requires directive takes effect only when a target region is
11122     // present in the compilation unit. Otherwise it is ignored and not
11123     // passed to the runtime. This avoids the runtime from throwing an error
11124     // for mismatching requires clauses across compilation units that don't
11125     // contain at least 1 target region.
11126     assert((HasEmittedTargetRegion ||
11127             HasEmittedDeclareTargetRegion ||
11128             !OffloadEntriesInfoManager.empty()) &&
11129            "Target or declare target region expected.");
11130     if (HasRequiresUnifiedSharedMemory)
11131       Flags = OMP_REQ_UNIFIED_SHARED_MEMORY;
11132     CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
11133                             CGM.getModule(), OMPRTL___tgt_register_requires),
11134                         llvm::ConstantInt::get(CGM.Int64Ty, Flags));
11135     CGF.FinishFunction();
11136   }
11137   return RequiresRegFn;
11138 }
11139 
11140 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF,
11141                                     const OMPExecutableDirective &D,
11142                                     SourceLocation Loc,
11143                                     llvm::Function *OutlinedFn,
11144                                     ArrayRef<llvm::Value *> CapturedVars) {
11145   if (!CGF.HaveInsertPoint())
11146     return;
11147 
11148   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11149   CodeGenFunction::RunCleanupsScope Scope(CGF);
11150 
11151   // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn);
11152   llvm::Value *Args[] = {
11153       RTLoc,
11154       CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars
11155       CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())};
11156   llvm::SmallVector<llvm::Value *, 16> RealArgs;
11157   RealArgs.append(std::begin(Args), std::end(Args));
11158   RealArgs.append(CapturedVars.begin(), CapturedVars.end());
11159 
11160   llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction(
11161       CGM.getModule(), OMPRTL___kmpc_fork_teams);
11162   CGF.EmitRuntimeCall(RTLFn, RealArgs);
11163 }
11164 
11165 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
11166                                          const Expr *NumTeams,
11167                                          const Expr *ThreadLimit,
11168                                          SourceLocation Loc) {
11169   if (!CGF.HaveInsertPoint())
11170     return;
11171 
11172   llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc);
11173 
11174   llvm::Value *NumTeamsVal =
11175       NumTeams
11176           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams),
11177                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
11178           : CGF.Builder.getInt32(0);
11179 
11180   llvm::Value *ThreadLimitVal =
11181       ThreadLimit
11182           ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit),
11183                                       CGF.CGM.Int32Ty, /* isSigned = */ true)
11184           : CGF.Builder.getInt32(0);
11185 
11186   // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit)
11187   llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal,
11188                                      ThreadLimitVal};
11189   CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction(
11190                           CGM.getModule(), OMPRTL___kmpc_push_num_teams),
11191                       PushNumTeamsArgs);
11192 }
11193 
11194 void CGOpenMPRuntime::emitTargetDataCalls(
11195     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
11196     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
11197   if (!CGF.HaveInsertPoint())
11198     return;
11199 
11200   // Action used to replace the default codegen action and turn privatization
11201   // off.
11202   PrePostActionTy NoPrivAction;
11203 
11204   // Generate the code for the opening of the data environment. Capture all the
11205   // arguments of the runtime call by reference because they are used in the
11206   // closing of the region.
11207   auto &&BeginThenGen = [this, &D, Device, &Info,
11208                          &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) {
11209     // Fill up the arrays with all the mapped variables.
11210     MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
11211 
11212     // Get map clause information.
11213     MappableExprsHandler MEHandler(D, CGF);
11214     MEHandler.generateAllInfo(CombinedInfo);
11215 
11216     // Fill up the arrays and create the arguments.
11217     emitOffloadingArrays(CGF, CombinedInfo, Info, OMPBuilder,
11218                          /*IsNonContiguous=*/true);
11219 
11220     llvm::Value *BasePointersArrayArg = nullptr;
11221     llvm::Value *PointersArrayArg = nullptr;
11222     llvm::Value *SizesArrayArg = nullptr;
11223     llvm::Value *MapTypesArrayArg = nullptr;
11224     llvm::Value *MapNamesArrayArg = nullptr;
11225     llvm::Value *MappersArrayArg = nullptr;
11226     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
11227                                  SizesArrayArg, MapTypesArrayArg,
11228                                  MapNamesArrayArg, MappersArrayArg, Info);
11229 
11230     // Emit device ID if any.
11231     llvm::Value *DeviceID = nullptr;
11232     if (Device) {
11233       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
11234                                            CGF.Int64Ty, /*isSigned=*/true);
11235     } else {
11236       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
11237     }
11238 
11239     // Emit the number of elements in the offloading arrays.
11240     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
11241     //
11242     // Source location for the ident struct
11243     llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc());
11244 
11245     llvm::Value *OffloadingArgs[] = {RTLoc,
11246                                      DeviceID,
11247                                      PointerNum,
11248                                      BasePointersArrayArg,
11249                                      PointersArrayArg,
11250                                      SizesArrayArg,
11251                                      MapTypesArrayArg,
11252                                      MapNamesArrayArg,
11253                                      MappersArrayArg};
11254     CGF.EmitRuntimeCall(
11255         OMPBuilder.getOrCreateRuntimeFunction(
11256             CGM.getModule(), OMPRTL___tgt_target_data_begin_mapper),
11257         OffloadingArgs);
11258 
11259     // If device pointer privatization is required, emit the body of the region
11260     // here. It will have to be duplicated: with and without privatization.
11261     if (!Info.CaptureDeviceAddrMap.empty())
11262       CodeGen(CGF);
11263   };
11264 
11265   // Generate code for the closing of the data region.
11266   auto &&EndThenGen = [this, Device, &Info, &D](CodeGenFunction &CGF,
11267                                                 PrePostActionTy &) {
11268     assert(Info.isValid() && "Invalid data environment closing arguments.");
11269 
11270     llvm::Value *BasePointersArrayArg = nullptr;
11271     llvm::Value *PointersArrayArg = nullptr;
11272     llvm::Value *SizesArrayArg = nullptr;
11273     llvm::Value *MapTypesArrayArg = nullptr;
11274     llvm::Value *MapNamesArrayArg = nullptr;
11275     llvm::Value *MappersArrayArg = nullptr;
11276     emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg,
11277                                  SizesArrayArg, MapTypesArrayArg,
11278                                  MapNamesArrayArg, MappersArrayArg, Info,
11279                                  {/*ForEndCall=*/true});
11280 
11281     // Emit device ID if any.
11282     llvm::Value *DeviceID = nullptr;
11283     if (Device) {
11284       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
11285                                            CGF.Int64Ty, /*isSigned=*/true);
11286     } else {
11287       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
11288     }
11289 
11290     // Emit the number of elements in the offloading arrays.
11291     llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs);
11292 
11293     // Source location for the ident struct
11294     llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc());
11295 
11296     llvm::Value *OffloadingArgs[] = {RTLoc,
11297                                      DeviceID,
11298                                      PointerNum,
11299                                      BasePointersArrayArg,
11300                                      PointersArrayArg,
11301                                      SizesArrayArg,
11302                                      MapTypesArrayArg,
11303                                      MapNamesArrayArg,
11304                                      MappersArrayArg};
11305     CGF.EmitRuntimeCall(
11306         OMPBuilder.getOrCreateRuntimeFunction(
11307             CGM.getModule(), OMPRTL___tgt_target_data_end_mapper),
11308         OffloadingArgs);
11309   };
11310 
11311   // If we need device pointer privatization, we need to emit the body of the
11312   // region with no privatization in the 'else' branch of the conditional.
11313   // Otherwise, we don't have to do anything.
11314   auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF,
11315                                                          PrePostActionTy &) {
11316     if (!Info.CaptureDeviceAddrMap.empty()) {
11317       CodeGen.setAction(NoPrivAction);
11318       CodeGen(CGF);
11319     }
11320   };
11321 
11322   // We don't have to do anything to close the region if the if clause evaluates
11323   // to false.
11324   auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {};
11325 
11326   if (IfCond) {
11327     emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen);
11328   } else {
11329     RegionCodeGenTy RCG(BeginThenGen);
11330     RCG(CGF);
11331   }
11332 
11333   // If we don't require privatization of device pointers, we emit the body in
11334   // between the runtime calls. This avoids duplicating the body code.
11335   if (Info.CaptureDeviceAddrMap.empty()) {
11336     CodeGen.setAction(NoPrivAction);
11337     CodeGen(CGF);
11338   }
11339 
11340   if (IfCond) {
11341     emitIfClause(CGF, IfCond, EndThenGen, EndElseGen);
11342   } else {
11343     RegionCodeGenTy RCG(EndThenGen);
11344     RCG(CGF);
11345   }
11346 }
11347 
11348 void CGOpenMPRuntime::emitTargetDataStandAloneCall(
11349     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
11350     const Expr *Device) {
11351   if (!CGF.HaveInsertPoint())
11352     return;
11353 
11354   assert((isa<OMPTargetEnterDataDirective>(D) ||
11355           isa<OMPTargetExitDataDirective>(D) ||
11356           isa<OMPTargetUpdateDirective>(D)) &&
11357          "Expecting either target enter, exit data, or update directives.");
11358 
11359   CodeGenFunction::OMPTargetDataInfo InputInfo;
11360   llvm::Value *MapTypesArray = nullptr;
11361   llvm::Value *MapNamesArray = nullptr;
11362   // Generate the code for the opening of the data environment.
11363   auto &&ThenGen = [this, &D, Device, &InputInfo, &MapTypesArray,
11364                     &MapNamesArray](CodeGenFunction &CGF, PrePostActionTy &) {
11365     // Emit device ID if any.
11366     llvm::Value *DeviceID = nullptr;
11367     if (Device) {
11368       DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device),
11369                                            CGF.Int64Ty, /*isSigned=*/true);
11370     } else {
11371       DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF);
11372     }
11373 
11374     // Emit the number of elements in the offloading arrays.
11375     llvm::Constant *PointerNum =
11376         CGF.Builder.getInt32(InputInfo.NumberOfTargetItems);
11377 
11378     // Source location for the ident struct
11379     llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc());
11380 
11381     llvm::Value *OffloadingArgs[] = {RTLoc,
11382                                      DeviceID,
11383                                      PointerNum,
11384                                      InputInfo.BasePointersArray.getPointer(),
11385                                      InputInfo.PointersArray.getPointer(),
11386                                      InputInfo.SizesArray.getPointer(),
11387                                      MapTypesArray,
11388                                      MapNamesArray,
11389                                      InputInfo.MappersArray.getPointer()};
11390 
11391     // Select the right runtime function call for each standalone
11392     // directive.
11393     const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>();
11394     RuntimeFunction RTLFn;
11395     switch (D.getDirectiveKind()) {
11396     case OMPD_target_enter_data:
11397       RTLFn = HasNowait ? OMPRTL___tgt_target_data_begin_nowait_mapper
11398                         : OMPRTL___tgt_target_data_begin_mapper;
11399       break;
11400     case OMPD_target_exit_data:
11401       RTLFn = HasNowait ? OMPRTL___tgt_target_data_end_nowait_mapper
11402                         : OMPRTL___tgt_target_data_end_mapper;
11403       break;
11404     case OMPD_target_update:
11405       RTLFn = HasNowait ? OMPRTL___tgt_target_data_update_nowait_mapper
11406                         : OMPRTL___tgt_target_data_update_mapper;
11407       break;
11408     case OMPD_parallel:
11409     case OMPD_for:
11410     case OMPD_parallel_for:
11411     case OMPD_parallel_master:
11412     case OMPD_parallel_sections:
11413     case OMPD_for_simd:
11414     case OMPD_parallel_for_simd:
11415     case OMPD_cancel:
11416     case OMPD_cancellation_point:
11417     case OMPD_ordered:
11418     case OMPD_threadprivate:
11419     case OMPD_allocate:
11420     case OMPD_task:
11421     case OMPD_simd:
11422     case OMPD_tile:
11423     case OMPD_unroll:
11424     case OMPD_sections:
11425     case OMPD_section:
11426     case OMPD_single:
11427     case OMPD_master:
11428     case OMPD_critical:
11429     case OMPD_taskyield:
11430     case OMPD_barrier:
11431     case OMPD_taskwait:
11432     case OMPD_taskgroup:
11433     case OMPD_atomic:
11434     case OMPD_flush:
11435     case OMPD_depobj:
11436     case OMPD_scan:
11437     case OMPD_teams:
11438     case OMPD_target_data:
11439     case OMPD_distribute:
11440     case OMPD_distribute_simd:
11441     case OMPD_distribute_parallel_for:
11442     case OMPD_distribute_parallel_for_simd:
11443     case OMPD_teams_distribute:
11444     case OMPD_teams_distribute_simd:
11445     case OMPD_teams_distribute_parallel_for:
11446     case OMPD_teams_distribute_parallel_for_simd:
11447     case OMPD_declare_simd:
11448     case OMPD_declare_variant:
11449     case OMPD_begin_declare_variant:
11450     case OMPD_end_declare_variant:
11451     case OMPD_declare_target:
11452     case OMPD_end_declare_target:
11453     case OMPD_declare_reduction:
11454     case OMPD_declare_mapper:
11455     case OMPD_taskloop:
11456     case OMPD_taskloop_simd:
11457     case OMPD_master_taskloop:
11458     case OMPD_master_taskloop_simd:
11459     case OMPD_parallel_master_taskloop:
11460     case OMPD_parallel_master_taskloop_simd:
11461     case OMPD_target:
11462     case OMPD_target_simd:
11463     case OMPD_target_teams_distribute:
11464     case OMPD_target_teams_distribute_simd:
11465     case OMPD_target_teams_distribute_parallel_for:
11466     case OMPD_target_teams_distribute_parallel_for_simd:
11467     case OMPD_target_teams:
11468     case OMPD_target_parallel:
11469     case OMPD_target_parallel_for:
11470     case OMPD_target_parallel_for_simd:
11471     case OMPD_requires:
11472     case OMPD_metadirective:
11473     case OMPD_unknown:
11474     default:
11475       llvm_unreachable("Unexpected standalone target data directive.");
11476       break;
11477     }
11478     CGF.EmitRuntimeCall(
11479         OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), RTLFn),
11480         OffloadingArgs);
11481   };
11482 
11483   auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray,
11484                           &MapNamesArray](CodeGenFunction &CGF,
11485                                           PrePostActionTy &) {
11486     // Fill up the arrays with all the mapped variables.
11487     MappableExprsHandler::MapCombinedInfoTy CombinedInfo;
11488 
11489     // Get map clause information.
11490     MappableExprsHandler MEHandler(D, CGF);
11491     MEHandler.generateAllInfo(CombinedInfo);
11492 
11493     TargetDataInfo Info;
11494     // Fill up the arrays and create the arguments.
11495     emitOffloadingArrays(CGF, CombinedInfo, Info, OMPBuilder,
11496                          /*IsNonContiguous=*/true);
11497     bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() ||
11498                              D.hasClausesOfKind<OMPNowaitClause>();
11499     emitOffloadingArraysArgument(
11500         CGF, Info.BasePointersArray, Info.PointersArray, Info.SizesArray,
11501         Info.MapTypesArray, Info.MapNamesArray, Info.MappersArray, Info,
11502         {/*ForEndCall=*/false});
11503     InputInfo.NumberOfTargetItems = Info.NumberOfPtrs;
11504     InputInfo.BasePointersArray =
11505         Address(Info.BasePointersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
11506     InputInfo.PointersArray =
11507         Address(Info.PointersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
11508     InputInfo.SizesArray =
11509         Address(Info.SizesArray, CGF.Int64Ty, CGM.getPointerAlign());
11510     InputInfo.MappersArray =
11511         Address(Info.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign());
11512     MapTypesArray = Info.MapTypesArray;
11513     MapNamesArray = Info.MapNamesArray;
11514     if (RequiresOuterTask)
11515       CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo);
11516     else
11517       emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen);
11518   };
11519 
11520   if (IfCond) {
11521     emitIfClause(CGF, IfCond, TargetThenGen,
11522                  [](CodeGenFunction &CGF, PrePostActionTy &) {});
11523   } else {
11524     RegionCodeGenTy ThenRCG(TargetThenGen);
11525     ThenRCG(CGF);
11526   }
11527 }
11528 
11529 namespace {
11530   /// Kind of parameter in a function with 'declare simd' directive.
11531   enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector };
11532   /// Attribute set of the parameter.
11533   struct ParamAttrTy {
11534     ParamKindTy Kind = Vector;
11535     llvm::APSInt StrideOrArg;
11536     llvm::APSInt Alignment;
11537   };
11538 } // namespace
11539 
11540 static unsigned evaluateCDTSize(const FunctionDecl *FD,
11541                                 ArrayRef<ParamAttrTy> ParamAttrs) {
11542   // Every vector variant of a SIMD-enabled function has a vector length (VLEN).
11543   // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument
11544   // of that clause. The VLEN value must be power of 2.
11545   // In other case the notion of the function`s "characteristic data type" (CDT)
11546   // is used to compute the vector length.
11547   // CDT is defined in the following order:
11548   //   a) For non-void function, the CDT is the return type.
11549   //   b) If the function has any non-uniform, non-linear parameters, then the
11550   //   CDT is the type of the first such parameter.
11551   //   c) If the CDT determined by a) or b) above is struct, union, or class
11552   //   type which is pass-by-value (except for the type that maps to the
11553   //   built-in complex data type), the characteristic data type is int.
11554   //   d) If none of the above three cases is applicable, the CDT is int.
11555   // The VLEN is then determined based on the CDT and the size of vector
11556   // register of that ISA for which current vector version is generated. The
11557   // VLEN is computed using the formula below:
11558   //   VLEN  = sizeof(vector_register) / sizeof(CDT),
11559   // where vector register size specified in section 3.2.1 Registers and the
11560   // Stack Frame of original AMD64 ABI document.
11561   QualType RetType = FD->getReturnType();
11562   if (RetType.isNull())
11563     return 0;
11564   ASTContext &C = FD->getASTContext();
11565   QualType CDT;
11566   if (!RetType.isNull() && !RetType->isVoidType()) {
11567     CDT = RetType;
11568   } else {
11569     unsigned Offset = 0;
11570     if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) {
11571       if (ParamAttrs[Offset].Kind == Vector)
11572         CDT = C.getPointerType(C.getRecordType(MD->getParent()));
11573       ++Offset;
11574     }
11575     if (CDT.isNull()) {
11576       for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
11577         if (ParamAttrs[I + Offset].Kind == Vector) {
11578           CDT = FD->getParamDecl(I)->getType();
11579           break;
11580         }
11581       }
11582     }
11583   }
11584   if (CDT.isNull())
11585     CDT = C.IntTy;
11586   CDT = CDT->getCanonicalTypeUnqualified();
11587   if (CDT->isRecordType() || CDT->isUnionType())
11588     CDT = C.IntTy;
11589   return C.getTypeSize(CDT);
11590 }
11591 
11592 static void
11593 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn,
11594                            const llvm::APSInt &VLENVal,
11595                            ArrayRef<ParamAttrTy> ParamAttrs,
11596                            OMPDeclareSimdDeclAttr::BranchStateTy State) {
11597   struct ISADataTy {
11598     char ISA;
11599     unsigned VecRegSize;
11600   };
11601   ISADataTy ISAData[] = {
11602       {
11603           'b', 128
11604       }, // SSE
11605       {
11606           'c', 256
11607       }, // AVX
11608       {
11609           'd', 256
11610       }, // AVX2
11611       {
11612           'e', 512
11613       }, // AVX512
11614   };
11615   llvm::SmallVector<char, 2> Masked;
11616   switch (State) {
11617   case OMPDeclareSimdDeclAttr::BS_Undefined:
11618     Masked.push_back('N');
11619     Masked.push_back('M');
11620     break;
11621   case OMPDeclareSimdDeclAttr::BS_Notinbranch:
11622     Masked.push_back('N');
11623     break;
11624   case OMPDeclareSimdDeclAttr::BS_Inbranch:
11625     Masked.push_back('M');
11626     break;
11627   }
11628   for (char Mask : Masked) {
11629     for (const ISADataTy &Data : ISAData) {
11630       SmallString<256> Buffer;
11631       llvm::raw_svector_ostream Out(Buffer);
11632       Out << "_ZGV" << Data.ISA << Mask;
11633       if (!VLENVal) {
11634         unsigned NumElts = evaluateCDTSize(FD, ParamAttrs);
11635         assert(NumElts && "Non-zero simdlen/cdtsize expected");
11636         Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts);
11637       } else {
11638         Out << VLENVal;
11639       }
11640       for (const ParamAttrTy &ParamAttr : ParamAttrs) {
11641         switch (ParamAttr.Kind){
11642         case LinearWithVarStride:
11643           Out << 's' << ParamAttr.StrideOrArg;
11644           break;
11645         case Linear:
11646           Out << 'l';
11647           if (ParamAttr.StrideOrArg != 1)
11648             Out << ParamAttr.StrideOrArg;
11649           break;
11650         case Uniform:
11651           Out << 'u';
11652           break;
11653         case Vector:
11654           Out << 'v';
11655           break;
11656         }
11657         if (!!ParamAttr.Alignment)
11658           Out << 'a' << ParamAttr.Alignment;
11659       }
11660       Out << '_' << Fn->getName();
11661       Fn->addFnAttr(Out.str());
11662     }
11663   }
11664 }
11665 
11666 // This are the Functions that are needed to mangle the name of the
11667 // vector functions generated by the compiler, according to the rules
11668 // defined in the "Vector Function ABI specifications for AArch64",
11669 // available at
11670 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi.
11671 
11672 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI.
11673 ///
11674 /// TODO: Need to implement the behavior for reference marked with a
11675 /// var or no linear modifiers (1.b in the section). For this, we
11676 /// need to extend ParamKindTy to support the linear modifiers.
11677 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) {
11678   QT = QT.getCanonicalType();
11679 
11680   if (QT->isVoidType())
11681     return false;
11682 
11683   if (Kind == ParamKindTy::Uniform)
11684     return false;
11685 
11686   if (Kind == ParamKindTy::Linear)
11687     return false;
11688 
11689   // TODO: Handle linear references with modifiers
11690 
11691   if (Kind == ParamKindTy::LinearWithVarStride)
11692     return false;
11693 
11694   return true;
11695 }
11696 
11697 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI.
11698 static bool getAArch64PBV(QualType QT, ASTContext &C) {
11699   QT = QT.getCanonicalType();
11700   unsigned Size = C.getTypeSize(QT);
11701 
11702   // Only scalars and complex within 16 bytes wide set PVB to true.
11703   if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128)
11704     return false;
11705 
11706   if (QT->isFloatingType())
11707     return true;
11708 
11709   if (QT->isIntegerType())
11710     return true;
11711 
11712   if (QT->isPointerType())
11713     return true;
11714 
11715   // TODO: Add support for complex types (section 3.1.2, item 2).
11716 
11717   return false;
11718 }
11719 
11720 /// Computes the lane size (LS) of a return type or of an input parameter,
11721 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI.
11722 /// TODO: Add support for references, section 3.2.1, item 1.
11723 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) {
11724   if (!getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) {
11725     QualType PTy = QT.getCanonicalType()->getPointeeType();
11726     if (getAArch64PBV(PTy, C))
11727       return C.getTypeSize(PTy);
11728   }
11729   if (getAArch64PBV(QT, C))
11730     return C.getTypeSize(QT);
11731 
11732   return C.getTypeSize(C.getUIntPtrType());
11733 }
11734 
11735 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the
11736 // signature of the scalar function, as defined in 3.2.2 of the
11737 // AAVFABI.
11738 static std::tuple<unsigned, unsigned, bool>
11739 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) {
11740   QualType RetType = FD->getReturnType().getCanonicalType();
11741 
11742   ASTContext &C = FD->getASTContext();
11743 
11744   bool OutputBecomesInput = false;
11745 
11746   llvm::SmallVector<unsigned, 8> Sizes;
11747   if (!RetType->isVoidType()) {
11748     Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C));
11749     if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {}))
11750       OutputBecomesInput = true;
11751   }
11752   for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) {
11753     QualType QT = FD->getParamDecl(I)->getType().getCanonicalType();
11754     Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C));
11755   }
11756 
11757   assert(!Sizes.empty() && "Unable to determine NDS and WDS.");
11758   // The LS of a function parameter / return value can only be a power
11759   // of 2, starting from 8 bits, up to 128.
11760   assert(llvm::all_of(Sizes,
11761                       [](unsigned Size) {
11762                         return Size == 8 || Size == 16 || Size == 32 ||
11763                                Size == 64 || Size == 128;
11764                       }) &&
11765          "Invalid size");
11766 
11767   return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)),
11768                          *std::max_element(std::begin(Sizes), std::end(Sizes)),
11769                          OutputBecomesInput);
11770 }
11771 
11772 /// Mangle the parameter part of the vector function name according to
11773 /// their OpenMP classification. The mangling function is defined in
11774 /// section 3.5 of the AAVFABI.
11775 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) {
11776   SmallString<256> Buffer;
11777   llvm::raw_svector_ostream Out(Buffer);
11778   for (const auto &ParamAttr : ParamAttrs) {
11779     switch (ParamAttr.Kind) {
11780     case LinearWithVarStride:
11781       Out << "ls" << ParamAttr.StrideOrArg;
11782       break;
11783     case Linear:
11784       Out << 'l';
11785       // Don't print the step value if it is not present or if it is
11786       // equal to 1.
11787       if (ParamAttr.StrideOrArg != 1)
11788         Out << ParamAttr.StrideOrArg;
11789       break;
11790     case Uniform:
11791       Out << 'u';
11792       break;
11793     case Vector:
11794       Out << 'v';
11795       break;
11796     }
11797 
11798     if (!!ParamAttr.Alignment)
11799       Out << 'a' << ParamAttr.Alignment;
11800   }
11801 
11802   return std::string(Out.str());
11803 }
11804 
11805 // Function used to add the attribute. The parameter `VLEN` is
11806 // templated to allow the use of "x" when targeting scalable functions
11807 // for SVE.
11808 template <typename T>
11809 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix,
11810                                  char ISA, StringRef ParSeq,
11811                                  StringRef MangledName, bool OutputBecomesInput,
11812                                  llvm::Function *Fn) {
11813   SmallString<256> Buffer;
11814   llvm::raw_svector_ostream Out(Buffer);
11815   Out << Prefix << ISA << LMask << VLEN;
11816   if (OutputBecomesInput)
11817     Out << "v";
11818   Out << ParSeq << "_" << MangledName;
11819   Fn->addFnAttr(Out.str());
11820 }
11821 
11822 // Helper function to generate the Advanced SIMD names depending on
11823 // the value of the NDS when simdlen is not present.
11824 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask,
11825                                       StringRef Prefix, char ISA,
11826                                       StringRef ParSeq, StringRef MangledName,
11827                                       bool OutputBecomesInput,
11828                                       llvm::Function *Fn) {
11829   switch (NDS) {
11830   case 8:
11831     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
11832                          OutputBecomesInput, Fn);
11833     addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName,
11834                          OutputBecomesInput, Fn);
11835     break;
11836   case 16:
11837     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
11838                          OutputBecomesInput, Fn);
11839     addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName,
11840                          OutputBecomesInput, Fn);
11841     break;
11842   case 32:
11843     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
11844                          OutputBecomesInput, Fn);
11845     addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName,
11846                          OutputBecomesInput, Fn);
11847     break;
11848   case 64:
11849   case 128:
11850     addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName,
11851                          OutputBecomesInput, Fn);
11852     break;
11853   default:
11854     llvm_unreachable("Scalar type is too wide.");
11855   }
11856 }
11857 
11858 /// Emit vector function attributes for AArch64, as defined in the AAVFABI.
11859 static void emitAArch64DeclareSimdFunction(
11860     CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN,
11861     ArrayRef<ParamAttrTy> ParamAttrs,
11862     OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName,
11863     char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) {
11864 
11865   // Get basic data for building the vector signature.
11866   const auto Data = getNDSWDS(FD, ParamAttrs);
11867   const unsigned NDS = std::get<0>(Data);
11868   const unsigned WDS = std::get<1>(Data);
11869   const bool OutputBecomesInput = std::get<2>(Data);
11870 
11871   // Check the values provided via `simdlen` by the user.
11872   // 1. A `simdlen(1)` doesn't produce vector signatures,
11873   if (UserVLEN == 1) {
11874     unsigned DiagID = CGM.getDiags().getCustomDiagID(
11875         DiagnosticsEngine::Warning,
11876         "The clause simdlen(1) has no effect when targeting aarch64.");
11877     CGM.getDiags().Report(SLoc, DiagID);
11878     return;
11879   }
11880 
11881   // 2. Section 3.3.1, item 1: user input must be a power of 2 for
11882   // Advanced SIMD output.
11883   if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) {
11884     unsigned DiagID = CGM.getDiags().getCustomDiagID(
11885         DiagnosticsEngine::Warning, "The value specified in simdlen must be a "
11886                                     "power of 2 when targeting Advanced SIMD.");
11887     CGM.getDiags().Report(SLoc, DiagID);
11888     return;
11889   }
11890 
11891   // 3. Section 3.4.1. SVE fixed lengh must obey the architectural
11892   // limits.
11893   if (ISA == 's' && UserVLEN != 0) {
11894     if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) {
11895       unsigned DiagID = CGM.getDiags().getCustomDiagID(
11896           DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit "
11897                                       "lanes in the architectural constraints "
11898                                       "for SVE (min is 128-bit, max is "
11899                                       "2048-bit, by steps of 128-bit)");
11900       CGM.getDiags().Report(SLoc, DiagID) << WDS;
11901       return;
11902     }
11903   }
11904 
11905   // Sort out parameter sequence.
11906   const std::string ParSeq = mangleVectorParameters(ParamAttrs);
11907   StringRef Prefix = "_ZGV";
11908   // Generate simdlen from user input (if any).
11909   if (UserVLEN) {
11910     if (ISA == 's') {
11911       // SVE generates only a masked function.
11912       addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
11913                            OutputBecomesInput, Fn);
11914     } else {
11915       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
11916       // Advanced SIMD generates one or two functions, depending on
11917       // the `[not]inbranch` clause.
11918       switch (State) {
11919       case OMPDeclareSimdDeclAttr::BS_Undefined:
11920         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
11921                              OutputBecomesInput, Fn);
11922         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
11923                              OutputBecomesInput, Fn);
11924         break;
11925       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
11926         addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName,
11927                              OutputBecomesInput, Fn);
11928         break;
11929       case OMPDeclareSimdDeclAttr::BS_Inbranch:
11930         addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName,
11931                              OutputBecomesInput, Fn);
11932         break;
11933       }
11934     }
11935   } else {
11936     // If no user simdlen is provided, follow the AAVFABI rules for
11937     // generating the vector length.
11938     if (ISA == 's') {
11939       // SVE, section 3.4.1, item 1.
11940       addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName,
11941                            OutputBecomesInput, Fn);
11942     } else {
11943       assert(ISA == 'n' && "Expected ISA either 's' or 'n'.");
11944       // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or
11945       // two vector names depending on the use of the clause
11946       // `[not]inbranch`.
11947       switch (State) {
11948       case OMPDeclareSimdDeclAttr::BS_Undefined:
11949         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
11950                                   OutputBecomesInput, Fn);
11951         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
11952                                   OutputBecomesInput, Fn);
11953         break;
11954       case OMPDeclareSimdDeclAttr::BS_Notinbranch:
11955         addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName,
11956                                   OutputBecomesInput, Fn);
11957         break;
11958       case OMPDeclareSimdDeclAttr::BS_Inbranch:
11959         addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName,
11960                                   OutputBecomesInput, Fn);
11961         break;
11962       }
11963     }
11964   }
11965 }
11966 
11967 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD,
11968                                               llvm::Function *Fn) {
11969   ASTContext &C = CGM.getContext();
11970   FD = FD->getMostRecentDecl();
11971   // Map params to their positions in function decl.
11972   llvm::DenseMap<const Decl *, unsigned> ParamPositions;
11973   if (isa<CXXMethodDecl>(FD))
11974     ParamPositions.try_emplace(FD, 0);
11975   unsigned ParamPos = ParamPositions.size();
11976   for (const ParmVarDecl *P : FD->parameters()) {
11977     ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos);
11978     ++ParamPos;
11979   }
11980   while (FD) {
11981     for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) {
11982       llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size());
11983       // Mark uniform parameters.
11984       for (const Expr *E : Attr->uniforms()) {
11985         E = E->IgnoreParenImpCasts();
11986         unsigned Pos;
11987         if (isa<CXXThisExpr>(E)) {
11988           Pos = ParamPositions[FD];
11989         } else {
11990           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
11991                                 ->getCanonicalDecl();
11992           Pos = ParamPositions[PVD];
11993         }
11994         ParamAttrs[Pos].Kind = Uniform;
11995       }
11996       // Get alignment info.
11997       auto *NI = Attr->alignments_begin();
11998       for (const Expr *E : Attr->aligneds()) {
11999         E = E->IgnoreParenImpCasts();
12000         unsigned Pos;
12001         QualType ParmTy;
12002         if (isa<CXXThisExpr>(E)) {
12003           Pos = ParamPositions[FD];
12004           ParmTy = E->getType();
12005         } else {
12006           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
12007                                 ->getCanonicalDecl();
12008           Pos = ParamPositions[PVD];
12009           ParmTy = PVD->getType();
12010         }
12011         ParamAttrs[Pos].Alignment =
12012             (*NI)
12013                 ? (*NI)->EvaluateKnownConstInt(C)
12014                 : llvm::APSInt::getUnsigned(
12015                       C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy))
12016                           .getQuantity());
12017         ++NI;
12018       }
12019       // Mark linear parameters.
12020       auto *SI = Attr->steps_begin();
12021       auto *MI = Attr->modifiers_begin();
12022       for (const Expr *E : Attr->linears()) {
12023         E = E->IgnoreParenImpCasts();
12024         unsigned Pos;
12025         // Rescaling factor needed to compute the linear parameter
12026         // value in the mangled name.
12027         unsigned PtrRescalingFactor = 1;
12028         if (isa<CXXThisExpr>(E)) {
12029           Pos = ParamPositions[FD];
12030         } else {
12031           const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl())
12032                                 ->getCanonicalDecl();
12033           Pos = ParamPositions[PVD];
12034           if (auto *P = dyn_cast<PointerType>(PVD->getType()))
12035             PtrRescalingFactor = CGM.getContext()
12036                                      .getTypeSizeInChars(P->getPointeeType())
12037                                      .getQuantity();
12038         }
12039         ParamAttrTy &ParamAttr = ParamAttrs[Pos];
12040         ParamAttr.Kind = Linear;
12041         // Assuming a stride of 1, for `linear` without modifiers.
12042         ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(1);
12043         if (*SI) {
12044           Expr::EvalResult Result;
12045           if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) {
12046             if (const auto *DRE =
12047                     cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) {
12048               if (const auto *StridePVD =
12049                       dyn_cast<ParmVarDecl>(DRE->getDecl())) {
12050                 ParamAttr.Kind = LinearWithVarStride;
12051                 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(
12052                     ParamPositions[StridePVD->getCanonicalDecl()]);
12053               }
12054             }
12055           } else {
12056             ParamAttr.StrideOrArg = Result.Val.getInt();
12057           }
12058         }
12059         // If we are using a linear clause on a pointer, we need to
12060         // rescale the value of linear_step with the byte size of the
12061         // pointee type.
12062         if (Linear == ParamAttr.Kind)
12063           ParamAttr.StrideOrArg = ParamAttr.StrideOrArg * PtrRescalingFactor;
12064         ++SI;
12065         ++MI;
12066       }
12067       llvm::APSInt VLENVal;
12068       SourceLocation ExprLoc;
12069       const Expr *VLENExpr = Attr->getSimdlen();
12070       if (VLENExpr) {
12071         VLENVal = VLENExpr->EvaluateKnownConstInt(C);
12072         ExprLoc = VLENExpr->getExprLoc();
12073       }
12074       OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState();
12075       if (CGM.getTriple().isX86()) {
12076         emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State);
12077       } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) {
12078         unsigned VLEN = VLENVal.getExtValue();
12079         StringRef MangledName = Fn->getName();
12080         if (CGM.getTarget().hasFeature("sve"))
12081           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
12082                                          MangledName, 's', 128, Fn, ExprLoc);
12083         if (CGM.getTarget().hasFeature("neon"))
12084           emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State,
12085                                          MangledName, 'n', 128, Fn, ExprLoc);
12086       }
12087     }
12088     FD = FD->getPreviousDecl();
12089   }
12090 }
12091 
12092 namespace {
12093 /// Cleanup action for doacross support.
12094 class DoacrossCleanupTy final : public EHScopeStack::Cleanup {
12095 public:
12096   static const int DoacrossFinArgs = 2;
12097 
12098 private:
12099   llvm::FunctionCallee RTLFn;
12100   llvm::Value *Args[DoacrossFinArgs];
12101 
12102 public:
12103   DoacrossCleanupTy(llvm::FunctionCallee RTLFn,
12104                     ArrayRef<llvm::Value *> CallArgs)
12105       : RTLFn(RTLFn) {
12106     assert(CallArgs.size() == DoacrossFinArgs);
12107     std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args));
12108   }
12109   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
12110     if (!CGF.HaveInsertPoint())
12111       return;
12112     CGF.EmitRuntimeCall(RTLFn, Args);
12113   }
12114 };
12115 } // namespace
12116 
12117 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF,
12118                                        const OMPLoopDirective &D,
12119                                        ArrayRef<Expr *> NumIterations) {
12120   if (!CGF.HaveInsertPoint())
12121     return;
12122 
12123   ASTContext &C = CGM.getContext();
12124   QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true);
12125   RecordDecl *RD;
12126   if (KmpDimTy.isNull()) {
12127     // Build struct kmp_dim {  // loop bounds info casted to kmp_int64
12128     //  kmp_int64 lo; // lower
12129     //  kmp_int64 up; // upper
12130     //  kmp_int64 st; // stride
12131     // };
12132     RD = C.buildImplicitRecord("kmp_dim");
12133     RD->startDefinition();
12134     addFieldToRecordDecl(C, RD, Int64Ty);
12135     addFieldToRecordDecl(C, RD, Int64Ty);
12136     addFieldToRecordDecl(C, RD, Int64Ty);
12137     RD->completeDefinition();
12138     KmpDimTy = C.getRecordType(RD);
12139   } else {
12140     RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl());
12141   }
12142   llvm::APInt Size(/*numBits=*/32, NumIterations.size());
12143   QualType ArrayTy =
12144       C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0);
12145 
12146   Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims");
12147   CGF.EmitNullInitialization(DimsAddr, ArrayTy);
12148   enum { LowerFD = 0, UpperFD, StrideFD };
12149   // Fill dims with data.
12150   for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) {
12151     LValue DimsLVal = CGF.MakeAddrLValue(
12152         CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy);
12153     // dims.upper = num_iterations;
12154     LValue UpperLVal = CGF.EmitLValueForField(
12155         DimsLVal, *std::next(RD->field_begin(), UpperFD));
12156     llvm::Value *NumIterVal = CGF.EmitScalarConversion(
12157         CGF.EmitScalarExpr(NumIterations[I]), NumIterations[I]->getType(),
12158         Int64Ty, NumIterations[I]->getExprLoc());
12159     CGF.EmitStoreOfScalar(NumIterVal, UpperLVal);
12160     // dims.stride = 1;
12161     LValue StrideLVal = CGF.EmitLValueForField(
12162         DimsLVal, *std::next(RD->field_begin(), StrideFD));
12163     CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1),
12164                           StrideLVal);
12165   }
12166 
12167   // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid,
12168   // kmp_int32 num_dims, struct kmp_dim * dims);
12169   llvm::Value *Args[] = {
12170       emitUpdateLocation(CGF, D.getBeginLoc()),
12171       getThreadID(CGF, D.getBeginLoc()),
12172       llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()),
12173       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
12174           CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(),
12175           CGM.VoidPtrTy)};
12176 
12177   llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12178       CGM.getModule(), OMPRTL___kmpc_doacross_init);
12179   CGF.EmitRuntimeCall(RTLFn, Args);
12180   llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = {
12181       emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())};
12182   llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12183       CGM.getModule(), OMPRTL___kmpc_doacross_fini);
12184   CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn,
12185                                              llvm::makeArrayRef(FiniArgs));
12186 }
12187 
12188 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
12189                                           const OMPDependClause *C) {
12190   QualType Int64Ty =
12191       CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1);
12192   llvm::APInt Size(/*numBits=*/32, C->getNumLoops());
12193   QualType ArrayTy = CGM.getContext().getConstantArrayType(
12194       Int64Ty, Size, nullptr, ArrayType::Normal, 0);
12195   Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr");
12196   for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) {
12197     const Expr *CounterVal = C->getLoopData(I);
12198     assert(CounterVal);
12199     llvm::Value *CntVal = CGF.EmitScalarConversion(
12200         CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty,
12201         CounterVal->getExprLoc());
12202     CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I),
12203                           /*Volatile=*/false, Int64Ty);
12204   }
12205   llvm::Value *Args[] = {
12206       emitUpdateLocation(CGF, C->getBeginLoc()),
12207       getThreadID(CGF, C->getBeginLoc()),
12208       CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()};
12209   llvm::FunctionCallee RTLFn;
12210   if (C->getDependencyKind() == OMPC_DEPEND_source) {
12211     RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
12212                                                   OMPRTL___kmpc_doacross_post);
12213   } else {
12214     assert(C->getDependencyKind() == OMPC_DEPEND_sink);
12215     RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(),
12216                                                   OMPRTL___kmpc_doacross_wait);
12217   }
12218   CGF.EmitRuntimeCall(RTLFn, Args);
12219 }
12220 
12221 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc,
12222                                llvm::FunctionCallee Callee,
12223                                ArrayRef<llvm::Value *> Args) const {
12224   assert(Loc.isValid() && "Outlined function call location must be valid.");
12225   auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc);
12226 
12227   if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) {
12228     if (Fn->doesNotThrow()) {
12229       CGF.EmitNounwindRuntimeCall(Fn, Args);
12230       return;
12231     }
12232   }
12233   CGF.EmitRuntimeCall(Callee, Args);
12234 }
12235 
12236 void CGOpenMPRuntime::emitOutlinedFunctionCall(
12237     CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn,
12238     ArrayRef<llvm::Value *> Args) const {
12239   emitCall(CGF, Loc, OutlinedFn, Args);
12240 }
12241 
12242 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) {
12243   if (const auto *FD = dyn_cast<FunctionDecl>(D))
12244     if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD))
12245       HasEmittedDeclareTargetRegion = true;
12246 }
12247 
12248 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF,
12249                                              const VarDecl *NativeParam,
12250                                              const VarDecl *TargetParam) const {
12251   return CGF.GetAddrOfLocalVar(NativeParam);
12252 }
12253 
12254 /// Return allocator value from expression, or return a null allocator (default
12255 /// when no allocator specified).
12256 static llvm::Value *getAllocatorVal(CodeGenFunction &CGF,
12257                                     const Expr *Allocator) {
12258   llvm::Value *AllocVal;
12259   if (Allocator) {
12260     AllocVal = CGF.EmitScalarExpr(Allocator);
12261     // According to the standard, the original allocator type is a enum
12262     // (integer). Convert to pointer type, if required.
12263     AllocVal = CGF.EmitScalarConversion(AllocVal, Allocator->getType(),
12264                                         CGF.getContext().VoidPtrTy,
12265                                         Allocator->getExprLoc());
12266   } else {
12267     // If no allocator specified, it defaults to the null allocator.
12268     AllocVal = llvm::Constant::getNullValue(
12269         CGF.CGM.getTypes().ConvertType(CGF.getContext().VoidPtrTy));
12270   }
12271   return AllocVal;
12272 }
12273 
12274 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF,
12275                                                    const VarDecl *VD) {
12276   if (!VD)
12277     return Address::invalid();
12278   Address UntiedAddr = Address::invalid();
12279   Address UntiedRealAddr = Address::invalid();
12280   auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn);
12281   if (It != FunctionToUntiedTaskStackMap.end()) {
12282     const UntiedLocalVarsAddressesMap &UntiedData =
12283         UntiedLocalVarsStack[It->second];
12284     auto I = UntiedData.find(VD);
12285     if (I != UntiedData.end()) {
12286       UntiedAddr = I->second.first;
12287       UntiedRealAddr = I->second.second;
12288     }
12289   }
12290   const VarDecl *CVD = VD->getCanonicalDecl();
12291   if (CVD->hasAttr<OMPAllocateDeclAttr>()) {
12292     // Use the default allocation.
12293     if (!isAllocatableDecl(VD))
12294       return UntiedAddr;
12295     llvm::Value *Size;
12296     CharUnits Align = CGM.getContext().getDeclAlign(CVD);
12297     if (CVD->getType()->isVariablyModifiedType()) {
12298       Size = CGF.getTypeSize(CVD->getType());
12299       // Align the size: ((size + align - 1) / align) * align
12300       Size = CGF.Builder.CreateNUWAdd(
12301           Size, CGM.getSize(Align - CharUnits::fromQuantity(1)));
12302       Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align));
12303       Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align));
12304     } else {
12305       CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType());
12306       Size = CGM.getSize(Sz.alignTo(Align));
12307     }
12308     llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc());
12309     const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>();
12310     const Expr *Allocator = AA->getAllocator();
12311     llvm::Value *AllocVal = getAllocatorVal(CGF, Allocator);
12312     llvm::Value *Alignment =
12313         AA->getAlignment()
12314             ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(AA->getAlignment()),
12315                                         CGM.SizeTy, /*isSigned=*/false)
12316             : nullptr;
12317     SmallVector<llvm::Value *, 4> Args;
12318     Args.push_back(ThreadID);
12319     if (Alignment)
12320       Args.push_back(Alignment);
12321     Args.push_back(Size);
12322     Args.push_back(AllocVal);
12323     llvm::omp::RuntimeFunction FnID =
12324         Alignment ? OMPRTL___kmpc_aligned_alloc : OMPRTL___kmpc_alloc;
12325     llvm::Value *Addr = CGF.EmitRuntimeCall(
12326         OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), FnID), Args,
12327         getName({CVD->getName(), ".void.addr"}));
12328     llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction(
12329         CGM.getModule(), OMPRTL___kmpc_free);
12330     QualType Ty = CGM.getContext().getPointerType(CVD->getType());
12331     Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
12332         Addr, CGF.ConvertTypeForMem(Ty), getName({CVD->getName(), ".addr"}));
12333     if (UntiedAddr.isValid())
12334       CGF.EmitStoreOfScalar(Addr, UntiedAddr, /*Volatile=*/false, Ty);
12335 
12336     // Cleanup action for allocate support.
12337     class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup {
12338       llvm::FunctionCallee RTLFn;
12339       SourceLocation::UIntTy LocEncoding;
12340       Address Addr;
12341       const Expr *AllocExpr;
12342 
12343     public:
12344       OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn,
12345                            SourceLocation::UIntTy LocEncoding, Address Addr,
12346                            const Expr *AllocExpr)
12347           : RTLFn(RTLFn), LocEncoding(LocEncoding), Addr(Addr),
12348             AllocExpr(AllocExpr) {}
12349       void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
12350         if (!CGF.HaveInsertPoint())
12351           return;
12352         llvm::Value *Args[3];
12353         Args[0] = CGF.CGM.getOpenMPRuntime().getThreadID(
12354             CGF, SourceLocation::getFromRawEncoding(LocEncoding));
12355         Args[1] = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
12356             Addr.getPointer(), CGF.VoidPtrTy);
12357         llvm::Value *AllocVal = getAllocatorVal(CGF, AllocExpr);
12358         Args[2] = AllocVal;
12359         CGF.EmitRuntimeCall(RTLFn, Args);
12360       }
12361     };
12362     Address VDAddr =
12363         UntiedRealAddr.isValid()
12364             ? UntiedRealAddr
12365             : Address(Addr, CGF.ConvertTypeForMem(CVD->getType()), Align);
12366     CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(
12367         NormalAndEHCleanup, FiniRTLFn, CVD->getLocation().getRawEncoding(),
12368         VDAddr, Allocator);
12369     if (UntiedRealAddr.isValid())
12370       if (auto *Region =
12371               dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
12372         Region->emitUntiedSwitch(CGF);
12373     return VDAddr;
12374   }
12375   return UntiedAddr;
12376 }
12377 
12378 bool CGOpenMPRuntime::isLocalVarInUntiedTask(CodeGenFunction &CGF,
12379                                              const VarDecl *VD) const {
12380   auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn);
12381   if (It == FunctionToUntiedTaskStackMap.end())
12382     return false;
12383   return UntiedLocalVarsStack[It->second].count(VD) > 0;
12384 }
12385 
12386 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII(
12387     CodeGenModule &CGM, const OMPLoopDirective &S)
12388     : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) {
12389   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12390   if (!NeedToPush)
12391     return;
12392   NontemporalDeclsSet &DS =
12393       CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back();
12394   for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) {
12395     for (const Stmt *Ref : C->private_refs()) {
12396       const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts();
12397       const ValueDecl *VD;
12398       if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) {
12399         VD = DRE->getDecl();
12400       } else {
12401         const auto *ME = cast<MemberExpr>(SimpleRefExpr);
12402         assert((ME->isImplicitCXXThis() ||
12403                 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) &&
12404                "Expected member of current class.");
12405         VD = ME->getMemberDecl();
12406       }
12407       DS.insert(VD);
12408     }
12409   }
12410 }
12411 
12412 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() {
12413   if (!NeedToPush)
12414     return;
12415   CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back();
12416 }
12417 
12418 CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::UntiedTaskLocalDeclsRAII(
12419     CodeGenFunction &CGF,
12420     const llvm::MapVector<CanonicalDeclPtr<const VarDecl>,
12421                           std::pair<Address, Address>> &LocalVars)
12422     : CGM(CGF.CGM), NeedToPush(!LocalVars.empty()) {
12423   if (!NeedToPush)
12424     return;
12425   CGM.getOpenMPRuntime().FunctionToUntiedTaskStackMap.try_emplace(
12426       CGF.CurFn, CGM.getOpenMPRuntime().UntiedLocalVarsStack.size());
12427   CGM.getOpenMPRuntime().UntiedLocalVarsStack.push_back(LocalVars);
12428 }
12429 
12430 CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::~UntiedTaskLocalDeclsRAII() {
12431   if (!NeedToPush)
12432     return;
12433   CGM.getOpenMPRuntime().UntiedLocalVarsStack.pop_back();
12434 }
12435 
12436 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const {
12437   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12438 
12439   return llvm::any_of(
12440       CGM.getOpenMPRuntime().NontemporalDeclsStack,
12441       [VD](const NontemporalDeclsSet &Set) { return Set.contains(VD); });
12442 }
12443 
12444 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis(
12445     const OMPExecutableDirective &S,
12446     llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled)
12447     const {
12448   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs;
12449   // Vars in target/task regions must be excluded completely.
12450   if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) ||
12451       isOpenMPTaskingDirective(S.getDirectiveKind())) {
12452     SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
12453     getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind());
12454     const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front());
12455     for (const CapturedStmt::Capture &Cap : CS->captures()) {
12456       if (Cap.capturesVariable() || Cap.capturesVariableByCopy())
12457         NeedToCheckForLPCs.insert(Cap.getCapturedVar());
12458     }
12459   }
12460   // Exclude vars in private clauses.
12461   for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) {
12462     for (const Expr *Ref : C->varlists()) {
12463       if (!Ref->getType()->isScalarType())
12464         continue;
12465       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12466       if (!DRE)
12467         continue;
12468       NeedToCheckForLPCs.insert(DRE->getDecl());
12469     }
12470   }
12471   for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) {
12472     for (const Expr *Ref : C->varlists()) {
12473       if (!Ref->getType()->isScalarType())
12474         continue;
12475       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12476       if (!DRE)
12477         continue;
12478       NeedToCheckForLPCs.insert(DRE->getDecl());
12479     }
12480   }
12481   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
12482     for (const Expr *Ref : C->varlists()) {
12483       if (!Ref->getType()->isScalarType())
12484         continue;
12485       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12486       if (!DRE)
12487         continue;
12488       NeedToCheckForLPCs.insert(DRE->getDecl());
12489     }
12490   }
12491   for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) {
12492     for (const Expr *Ref : C->varlists()) {
12493       if (!Ref->getType()->isScalarType())
12494         continue;
12495       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12496       if (!DRE)
12497         continue;
12498       NeedToCheckForLPCs.insert(DRE->getDecl());
12499     }
12500   }
12501   for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) {
12502     for (const Expr *Ref : C->varlists()) {
12503       if (!Ref->getType()->isScalarType())
12504         continue;
12505       const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts());
12506       if (!DRE)
12507         continue;
12508       NeedToCheckForLPCs.insert(DRE->getDecl());
12509     }
12510   }
12511   for (const Decl *VD : NeedToCheckForLPCs) {
12512     for (const LastprivateConditionalData &Data :
12513          llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) {
12514       if (Data.DeclToUniqueName.count(VD) > 0) {
12515         if (!Data.Disabled)
12516           NeedToAddForLPCsAsDisabled.insert(VD);
12517         break;
12518       }
12519     }
12520   }
12521 }
12522 
12523 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
12524     CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal)
12525     : CGM(CGF.CGM),
12526       Action((CGM.getLangOpts().OpenMP >= 50 &&
12527               llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(),
12528                            [](const OMPLastprivateClause *C) {
12529                              return C->getKind() ==
12530                                     OMPC_LASTPRIVATE_conditional;
12531                            }))
12532                  ? ActionToDo::PushAsLastprivateConditional
12533                  : ActionToDo::DoNotPush) {
12534   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12535   if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush)
12536     return;
12537   assert(Action == ActionToDo::PushAsLastprivateConditional &&
12538          "Expected a push action.");
12539   LastprivateConditionalData &Data =
12540       CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
12541   for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) {
12542     if (C->getKind() != OMPC_LASTPRIVATE_conditional)
12543       continue;
12544 
12545     for (const Expr *Ref : C->varlists()) {
12546       Data.DeclToUniqueName.insert(std::make_pair(
12547           cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(),
12548           SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref))));
12549     }
12550   }
12551   Data.IVLVal = IVLVal;
12552   Data.Fn = CGF.CurFn;
12553 }
12554 
12555 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII(
12556     CodeGenFunction &CGF, const OMPExecutableDirective &S)
12557     : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) {
12558   assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode.");
12559   if (CGM.getLangOpts().OpenMP < 50)
12560     return;
12561   llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled;
12562   tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled);
12563   if (!NeedToAddForLPCsAsDisabled.empty()) {
12564     Action = ActionToDo::DisableLastprivateConditional;
12565     LastprivateConditionalData &Data =
12566         CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back();
12567     for (const Decl *VD : NeedToAddForLPCsAsDisabled)
12568       Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>()));
12569     Data.Fn = CGF.CurFn;
12570     Data.Disabled = true;
12571   }
12572 }
12573 
12574 CGOpenMPRuntime::LastprivateConditionalRAII
12575 CGOpenMPRuntime::LastprivateConditionalRAII::disable(
12576     CodeGenFunction &CGF, const OMPExecutableDirective &S) {
12577   return LastprivateConditionalRAII(CGF, S);
12578 }
12579 
12580 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() {
12581   if (CGM.getLangOpts().OpenMP < 50)
12582     return;
12583   if (Action == ActionToDo::DisableLastprivateConditional) {
12584     assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
12585            "Expected list of disabled private vars.");
12586     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
12587   }
12588   if (Action == ActionToDo::PushAsLastprivateConditional) {
12589     assert(
12590         !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled &&
12591         "Expected list of lastprivate conditional vars.");
12592     CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back();
12593   }
12594 }
12595 
12596 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF,
12597                                                         const VarDecl *VD) {
12598   ASTContext &C = CGM.getContext();
12599   auto I = LastprivateConditionalToTypes.find(CGF.CurFn);
12600   if (I == LastprivateConditionalToTypes.end())
12601     I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first;
12602   QualType NewType;
12603   const FieldDecl *VDField;
12604   const FieldDecl *FiredField;
12605   LValue BaseLVal;
12606   auto VI = I->getSecond().find(VD);
12607   if (VI == I->getSecond().end()) {
12608     RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional");
12609     RD->startDefinition();
12610     VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType());
12611     FiredField = addFieldToRecordDecl(C, RD, C.CharTy);
12612     RD->completeDefinition();
12613     NewType = C.getRecordType(RD);
12614     Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName());
12615     BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl);
12616     I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal);
12617   } else {
12618     NewType = std::get<0>(VI->getSecond());
12619     VDField = std::get<1>(VI->getSecond());
12620     FiredField = std::get<2>(VI->getSecond());
12621     BaseLVal = std::get<3>(VI->getSecond());
12622   }
12623   LValue FiredLVal =
12624       CGF.EmitLValueForField(BaseLVal, FiredField);
12625   CGF.EmitStoreOfScalar(
12626       llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)),
12627       FiredLVal);
12628   return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF);
12629 }
12630 
12631 namespace {
12632 /// Checks if the lastprivate conditional variable is referenced in LHS.
12633 class LastprivateConditionalRefChecker final
12634     : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> {
12635   ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM;
12636   const Expr *FoundE = nullptr;
12637   const Decl *FoundD = nullptr;
12638   StringRef UniqueDeclName;
12639   LValue IVLVal;
12640   llvm::Function *FoundFn = nullptr;
12641   SourceLocation Loc;
12642 
12643 public:
12644   bool VisitDeclRefExpr(const DeclRefExpr *E) {
12645     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
12646          llvm::reverse(LPM)) {
12647       auto It = D.DeclToUniqueName.find(E->getDecl());
12648       if (It == D.DeclToUniqueName.end())
12649         continue;
12650       if (D.Disabled)
12651         return false;
12652       FoundE = E;
12653       FoundD = E->getDecl()->getCanonicalDecl();
12654       UniqueDeclName = It->second;
12655       IVLVal = D.IVLVal;
12656       FoundFn = D.Fn;
12657       break;
12658     }
12659     return FoundE == E;
12660   }
12661   bool VisitMemberExpr(const MemberExpr *E) {
12662     if (!CodeGenFunction::IsWrappedCXXThis(E->getBase()))
12663       return false;
12664     for (const CGOpenMPRuntime::LastprivateConditionalData &D :
12665          llvm::reverse(LPM)) {
12666       auto It = D.DeclToUniqueName.find(E->getMemberDecl());
12667       if (It == D.DeclToUniqueName.end())
12668         continue;
12669       if (D.Disabled)
12670         return false;
12671       FoundE = E;
12672       FoundD = E->getMemberDecl()->getCanonicalDecl();
12673       UniqueDeclName = It->second;
12674       IVLVal = D.IVLVal;
12675       FoundFn = D.Fn;
12676       break;
12677     }
12678     return FoundE == E;
12679   }
12680   bool VisitStmt(const Stmt *S) {
12681     for (const Stmt *Child : S->children()) {
12682       if (!Child)
12683         continue;
12684       if (const auto *E = dyn_cast<Expr>(Child))
12685         if (!E->isGLValue())
12686           continue;
12687       if (Visit(Child))
12688         return true;
12689     }
12690     return false;
12691   }
12692   explicit LastprivateConditionalRefChecker(
12693       ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM)
12694       : LPM(LPM) {}
12695   std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *>
12696   getFoundData() const {
12697     return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn);
12698   }
12699 };
12700 } // namespace
12701 
12702 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF,
12703                                                        LValue IVLVal,
12704                                                        StringRef UniqueDeclName,
12705                                                        LValue LVal,
12706                                                        SourceLocation Loc) {
12707   // Last updated loop counter for the lastprivate conditional var.
12708   // int<xx> last_iv = 0;
12709   llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType());
12710   llvm::Constant *LastIV =
12711       getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"}));
12712   cast<llvm::GlobalVariable>(LastIV)->setAlignment(
12713       IVLVal.getAlignment().getAsAlign());
12714   LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType());
12715 
12716   // Last value of the lastprivate conditional.
12717   // decltype(priv_a) last_a;
12718   llvm::GlobalVariable *Last = getOrCreateInternalVariable(
12719       CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName);
12720   Last->setAlignment(LVal.getAlignment().getAsAlign());
12721   LValue LastLVal = CGF.MakeAddrLValue(
12722       Address(Last, Last->getValueType(), LVal.getAlignment()), LVal.getType());
12723 
12724   // Global loop counter. Required to handle inner parallel-for regions.
12725   // iv
12726   llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc);
12727 
12728   // #pragma omp critical(a)
12729   // if (last_iv <= iv) {
12730   //   last_iv = iv;
12731   //   last_a = priv_a;
12732   // }
12733   auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal,
12734                     Loc](CodeGenFunction &CGF, PrePostActionTy &Action) {
12735     Action.Enter(CGF);
12736     llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc);
12737     // (last_iv <= iv) ? Check if the variable is updated and store new
12738     // value in global var.
12739     llvm::Value *CmpRes;
12740     if (IVLVal.getType()->isSignedIntegerType()) {
12741       CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal);
12742     } else {
12743       assert(IVLVal.getType()->isUnsignedIntegerType() &&
12744              "Loop iteration variable must be integer.");
12745       CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal);
12746     }
12747     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then");
12748     llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit");
12749     CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB);
12750     // {
12751     CGF.EmitBlock(ThenBB);
12752 
12753     //   last_iv = iv;
12754     CGF.EmitStoreOfScalar(IVVal, LastIVLVal);
12755 
12756     //   last_a = priv_a;
12757     switch (CGF.getEvaluationKind(LVal.getType())) {
12758     case TEK_Scalar: {
12759       llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc);
12760       CGF.EmitStoreOfScalar(PrivVal, LastLVal);
12761       break;
12762     }
12763     case TEK_Complex: {
12764       CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc);
12765       CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false);
12766       break;
12767     }
12768     case TEK_Aggregate:
12769       llvm_unreachable(
12770           "Aggregates are not supported in lastprivate conditional.");
12771     }
12772     // }
12773     CGF.EmitBranch(ExitBB);
12774     // There is no need to emit line number for unconditional branch.
12775     (void)ApplyDebugLocation::CreateEmpty(CGF);
12776     CGF.EmitBlock(ExitBB, /*IsFinished=*/true);
12777   };
12778 
12779   if (CGM.getLangOpts().OpenMPSimd) {
12780     // Do not emit as a critical region as no parallel region could be emitted.
12781     RegionCodeGenTy ThenRCG(CodeGen);
12782     ThenRCG(CGF);
12783   } else {
12784     emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc);
12785   }
12786 }
12787 
12788 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF,
12789                                                          const Expr *LHS) {
12790   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
12791     return;
12792   LastprivateConditionalRefChecker Checker(LastprivateConditionalStack);
12793   if (!Checker.Visit(LHS))
12794     return;
12795   const Expr *FoundE;
12796   const Decl *FoundD;
12797   StringRef UniqueDeclName;
12798   LValue IVLVal;
12799   llvm::Function *FoundFn;
12800   std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) =
12801       Checker.getFoundData();
12802   if (FoundFn != CGF.CurFn) {
12803     // Special codegen for inner parallel regions.
12804     // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1;
12805     auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD);
12806     assert(It != LastprivateConditionalToTypes[FoundFn].end() &&
12807            "Lastprivate conditional is not found in outer region.");
12808     QualType StructTy = std::get<0>(It->getSecond());
12809     const FieldDecl* FiredDecl = std::get<2>(It->getSecond());
12810     LValue PrivLVal = CGF.EmitLValue(FoundE);
12811     Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
12812         PrivLVal.getAddress(CGF),
12813         CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy)),
12814         CGF.ConvertTypeForMem(StructTy));
12815     LValue BaseLVal =
12816         CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl);
12817     LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl);
12818     CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get(
12819                             CGF.ConvertTypeForMem(FiredDecl->getType()), 1)),
12820                         FiredLVal, llvm::AtomicOrdering::Unordered,
12821                         /*IsVolatile=*/true, /*isInit=*/false);
12822     return;
12823   }
12824 
12825   // Private address of the lastprivate conditional in the current context.
12826   // priv_a
12827   LValue LVal = CGF.EmitLValue(FoundE);
12828   emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal,
12829                                    FoundE->getExprLoc());
12830 }
12831 
12832 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional(
12833     CodeGenFunction &CGF, const OMPExecutableDirective &D,
12834     const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) {
12835   if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty())
12836     return;
12837   auto Range = llvm::reverse(LastprivateConditionalStack);
12838   auto It = llvm::find_if(
12839       Range, [](const LastprivateConditionalData &D) { return !D.Disabled; });
12840   if (It == Range.end() || It->Fn != CGF.CurFn)
12841     return;
12842   auto LPCI = LastprivateConditionalToTypes.find(It->Fn);
12843   assert(LPCI != LastprivateConditionalToTypes.end() &&
12844          "Lastprivates must be registered already.");
12845   SmallVector<OpenMPDirectiveKind, 4> CaptureRegions;
12846   getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind());
12847   const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back());
12848   for (const auto &Pair : It->DeclToUniqueName) {
12849     const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl());
12850     if (!CS->capturesVariable(VD) || IgnoredDecls.contains(VD))
12851       continue;
12852     auto I = LPCI->getSecond().find(Pair.first);
12853     assert(I != LPCI->getSecond().end() &&
12854            "Lastprivate must be rehistered already.");
12855     // bool Cmp = priv_a.Fired != 0;
12856     LValue BaseLVal = std::get<3>(I->getSecond());
12857     LValue FiredLVal =
12858         CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond()));
12859     llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc());
12860     llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res);
12861     llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then");
12862     llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done");
12863     // if (Cmp) {
12864     CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB);
12865     CGF.EmitBlock(ThenBB);
12866     Address Addr = CGF.GetAddrOfLocalVar(VD);
12867     LValue LVal;
12868     if (VD->getType()->isReferenceType())
12869       LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(),
12870                                            AlignmentSource::Decl);
12871     else
12872       LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(),
12873                                 AlignmentSource::Decl);
12874     emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal,
12875                                      D.getBeginLoc());
12876     auto AL = ApplyDebugLocation::CreateArtificial(CGF);
12877     CGF.EmitBlock(DoneBB, /*IsFinal=*/true);
12878     // }
12879   }
12880 }
12881 
12882 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate(
12883     CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD,
12884     SourceLocation Loc) {
12885   if (CGF.getLangOpts().OpenMP < 50)
12886     return;
12887   auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD);
12888   assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() &&
12889          "Unknown lastprivate conditional variable.");
12890   StringRef UniqueName = It->second;
12891   llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName);
12892   // The variable was not updated in the region - exit.
12893   if (!GV)
12894     return;
12895   LValue LPLVal = CGF.MakeAddrLValue(
12896       Address(GV, GV->getValueType(), PrivLVal.getAlignment()),
12897       PrivLVal.getType().getNonReferenceType());
12898   llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc);
12899   CGF.EmitStoreOfScalar(Res, PrivLVal);
12900 }
12901 
12902 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction(
12903     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
12904     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
12905   llvm_unreachable("Not supported in SIMD-only mode");
12906 }
12907 
12908 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction(
12909     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
12910     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) {
12911   llvm_unreachable("Not supported in SIMD-only mode");
12912 }
12913 
12914 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction(
12915     const OMPExecutableDirective &D, const VarDecl *ThreadIDVar,
12916     const VarDecl *PartIDVar, const VarDecl *TaskTVar,
12917     OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen,
12918     bool Tied, unsigned &NumberOfParts) {
12919   llvm_unreachable("Not supported in SIMD-only mode");
12920 }
12921 
12922 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF,
12923                                            SourceLocation Loc,
12924                                            llvm::Function *OutlinedFn,
12925                                            ArrayRef<llvm::Value *> CapturedVars,
12926                                            const Expr *IfCond,
12927                                            llvm::Value *NumThreads) {
12928   llvm_unreachable("Not supported in SIMD-only mode");
12929 }
12930 
12931 void CGOpenMPSIMDRuntime::emitCriticalRegion(
12932     CodeGenFunction &CGF, StringRef CriticalName,
12933     const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc,
12934     const Expr *Hint) {
12935   llvm_unreachable("Not supported in SIMD-only mode");
12936 }
12937 
12938 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF,
12939                                            const RegionCodeGenTy &MasterOpGen,
12940                                            SourceLocation Loc) {
12941   llvm_unreachable("Not supported in SIMD-only mode");
12942 }
12943 
12944 void CGOpenMPSIMDRuntime::emitMaskedRegion(CodeGenFunction &CGF,
12945                                            const RegionCodeGenTy &MasterOpGen,
12946                                            SourceLocation Loc,
12947                                            const Expr *Filter) {
12948   llvm_unreachable("Not supported in SIMD-only mode");
12949 }
12950 
12951 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
12952                                             SourceLocation Loc) {
12953   llvm_unreachable("Not supported in SIMD-only mode");
12954 }
12955 
12956 void CGOpenMPSIMDRuntime::emitTaskgroupRegion(
12957     CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen,
12958     SourceLocation Loc) {
12959   llvm_unreachable("Not supported in SIMD-only mode");
12960 }
12961 
12962 void CGOpenMPSIMDRuntime::emitSingleRegion(
12963     CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen,
12964     SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars,
12965     ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs,
12966     ArrayRef<const Expr *> AssignmentOps) {
12967   llvm_unreachable("Not supported in SIMD-only mode");
12968 }
12969 
12970 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF,
12971                                             const RegionCodeGenTy &OrderedOpGen,
12972                                             SourceLocation Loc,
12973                                             bool IsThreads) {
12974   llvm_unreachable("Not supported in SIMD-only mode");
12975 }
12976 
12977 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF,
12978                                           SourceLocation Loc,
12979                                           OpenMPDirectiveKind Kind,
12980                                           bool EmitChecks,
12981                                           bool ForceSimpleCall) {
12982   llvm_unreachable("Not supported in SIMD-only mode");
12983 }
12984 
12985 void CGOpenMPSIMDRuntime::emitForDispatchInit(
12986     CodeGenFunction &CGF, SourceLocation Loc,
12987     const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned,
12988     bool Ordered, const DispatchRTInput &DispatchValues) {
12989   llvm_unreachable("Not supported in SIMD-only mode");
12990 }
12991 
12992 void CGOpenMPSIMDRuntime::emitForStaticInit(
12993     CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind,
12994     const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) {
12995   llvm_unreachable("Not supported in SIMD-only mode");
12996 }
12997 
12998 void CGOpenMPSIMDRuntime::emitDistributeStaticInit(
12999     CodeGenFunction &CGF, SourceLocation Loc,
13000     OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) {
13001   llvm_unreachable("Not supported in SIMD-only mode");
13002 }
13003 
13004 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF,
13005                                                      SourceLocation Loc,
13006                                                      unsigned IVSize,
13007                                                      bool IVSigned) {
13008   llvm_unreachable("Not supported in SIMD-only mode");
13009 }
13010 
13011 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF,
13012                                               SourceLocation Loc,
13013                                               OpenMPDirectiveKind DKind) {
13014   llvm_unreachable("Not supported in SIMD-only mode");
13015 }
13016 
13017 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF,
13018                                               SourceLocation Loc,
13019                                               unsigned IVSize, bool IVSigned,
13020                                               Address IL, Address LB,
13021                                               Address UB, Address ST) {
13022   llvm_unreachable("Not supported in SIMD-only mode");
13023 }
13024 
13025 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
13026                                                llvm::Value *NumThreads,
13027                                                SourceLocation Loc) {
13028   llvm_unreachable("Not supported in SIMD-only mode");
13029 }
13030 
13031 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF,
13032                                              ProcBindKind ProcBind,
13033                                              SourceLocation Loc) {
13034   llvm_unreachable("Not supported in SIMD-only mode");
13035 }
13036 
13037 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
13038                                                     const VarDecl *VD,
13039                                                     Address VDAddr,
13040                                                     SourceLocation Loc) {
13041   llvm_unreachable("Not supported in SIMD-only mode");
13042 }
13043 
13044 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition(
13045     const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit,
13046     CodeGenFunction *CGF) {
13047   llvm_unreachable("Not supported in SIMD-only mode");
13048 }
13049 
13050 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate(
13051     CodeGenFunction &CGF, QualType VarType, StringRef Name) {
13052   llvm_unreachable("Not supported in SIMD-only mode");
13053 }
13054 
13055 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF,
13056                                     ArrayRef<const Expr *> Vars,
13057                                     SourceLocation Loc,
13058                                     llvm::AtomicOrdering AO) {
13059   llvm_unreachable("Not supported in SIMD-only mode");
13060 }
13061 
13062 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc,
13063                                        const OMPExecutableDirective &D,
13064                                        llvm::Function *TaskFunction,
13065                                        QualType SharedsTy, Address Shareds,
13066                                        const Expr *IfCond,
13067                                        const OMPTaskDataTy &Data) {
13068   llvm_unreachable("Not supported in SIMD-only mode");
13069 }
13070 
13071 void CGOpenMPSIMDRuntime::emitTaskLoopCall(
13072     CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D,
13073     llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds,
13074     const Expr *IfCond, const OMPTaskDataTy &Data) {
13075   llvm_unreachable("Not supported in SIMD-only mode");
13076 }
13077 
13078 void CGOpenMPSIMDRuntime::emitReduction(
13079     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates,
13080     ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs,
13081     ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) {
13082   assert(Options.SimpleReduction && "Only simple reduction is expected.");
13083   CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs,
13084                                  ReductionOps, Options);
13085 }
13086 
13087 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit(
13088     CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs,
13089     ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) {
13090   llvm_unreachable("Not supported in SIMD-only mode");
13091 }
13092 
13093 void CGOpenMPSIMDRuntime::emitTaskReductionFini(CodeGenFunction &CGF,
13094                                                 SourceLocation Loc,
13095                                                 bool IsWorksharingReduction) {
13096   llvm_unreachable("Not supported in SIMD-only mode");
13097 }
13098 
13099 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF,
13100                                                   SourceLocation Loc,
13101                                                   ReductionCodeGen &RCG,
13102                                                   unsigned N) {
13103   llvm_unreachable("Not supported in SIMD-only mode");
13104 }
13105 
13106 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF,
13107                                                   SourceLocation Loc,
13108                                                   llvm::Value *ReductionsPtr,
13109                                                   LValue SharedLVal) {
13110   llvm_unreachable("Not supported in SIMD-only mode");
13111 }
13112 
13113 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
13114                                            SourceLocation Loc,
13115                                            const OMPTaskDataTy &Data) {
13116   llvm_unreachable("Not supported in SIMD-only mode");
13117 }
13118 
13119 void CGOpenMPSIMDRuntime::emitCancellationPointCall(
13120     CodeGenFunction &CGF, SourceLocation Loc,
13121     OpenMPDirectiveKind CancelRegion) {
13122   llvm_unreachable("Not supported in SIMD-only mode");
13123 }
13124 
13125 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF,
13126                                          SourceLocation Loc, const Expr *IfCond,
13127                                          OpenMPDirectiveKind CancelRegion) {
13128   llvm_unreachable("Not supported in SIMD-only mode");
13129 }
13130 
13131 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction(
13132     const OMPExecutableDirective &D, StringRef ParentName,
13133     llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID,
13134     bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) {
13135   llvm_unreachable("Not supported in SIMD-only mode");
13136 }
13137 
13138 void CGOpenMPSIMDRuntime::emitTargetCall(
13139     CodeGenFunction &CGF, const OMPExecutableDirective &D,
13140     llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond,
13141     llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device,
13142     llvm::function_ref<llvm::Value *(CodeGenFunction &CGF,
13143                                      const OMPLoopDirective &D)>
13144         SizeEmitter) {
13145   llvm_unreachable("Not supported in SIMD-only mode");
13146 }
13147 
13148 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) {
13149   llvm_unreachable("Not supported in SIMD-only mode");
13150 }
13151 
13152 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) {
13153   llvm_unreachable("Not supported in SIMD-only mode");
13154 }
13155 
13156 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) {
13157   return false;
13158 }
13159 
13160 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF,
13161                                         const OMPExecutableDirective &D,
13162                                         SourceLocation Loc,
13163                                         llvm::Function *OutlinedFn,
13164                                         ArrayRef<llvm::Value *> CapturedVars) {
13165   llvm_unreachable("Not supported in SIMD-only mode");
13166 }
13167 
13168 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF,
13169                                              const Expr *NumTeams,
13170                                              const Expr *ThreadLimit,
13171                                              SourceLocation Loc) {
13172   llvm_unreachable("Not supported in SIMD-only mode");
13173 }
13174 
13175 void CGOpenMPSIMDRuntime::emitTargetDataCalls(
13176     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
13177     const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) {
13178   llvm_unreachable("Not supported in SIMD-only mode");
13179 }
13180 
13181 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall(
13182     CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond,
13183     const Expr *Device) {
13184   llvm_unreachable("Not supported in SIMD-only mode");
13185 }
13186 
13187 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF,
13188                                            const OMPLoopDirective &D,
13189                                            ArrayRef<Expr *> NumIterations) {
13190   llvm_unreachable("Not supported in SIMD-only mode");
13191 }
13192 
13193 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF,
13194                                               const OMPDependClause *C) {
13195   llvm_unreachable("Not supported in SIMD-only mode");
13196 }
13197 
13198 const VarDecl *
13199 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD,
13200                                         const VarDecl *NativeParam) const {
13201   llvm_unreachable("Not supported in SIMD-only mode");
13202 }
13203 
13204 Address
13205 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF,
13206                                          const VarDecl *NativeParam,
13207                                          const VarDecl *TargetParam) const {
13208   llvm_unreachable("Not supported in SIMD-only mode");
13209 }
13210