1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 // This provides a class for OpenMP runtime code generation.
11 //
12 //===----------------------------------------------------------------------===//
13 
14 #include "CGOpenMPRuntime.h"
15 #include "CodeGenFunction.h"
16 #include "CGCleanup.h"
17 #include "clang/AST/Decl.h"
18 #include "clang/AST/StmtOpenMP.h"
19 #include "llvm/ADT/ArrayRef.h"
20 #include "llvm/IR/CallSite.h"
21 #include "llvm/IR/DerivedTypes.h"
22 #include "llvm/IR/GlobalValue.h"
23 #include "llvm/IR/Value.h"
24 #include "llvm/Support/raw_ostream.h"
25 #include <cassert>
26 
27 using namespace clang;
28 using namespace CodeGen;
29 
30 namespace {
31 /// \brief Base class for handling code generation inside OpenMP regions.
32 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
33 public:
34   /// \brief Kinds of OpenMP regions used in codegen.
35   enum CGOpenMPRegionKind {
36     /// \brief Region with outlined function for standalone 'parallel'
37     /// directive.
38     ParallelOutlinedRegion,
39     /// \brief Region with outlined function for standalone 'task' directive.
40     TaskOutlinedRegion,
41     /// \brief Region for constructs that do not require function outlining,
42     /// like 'for', 'sections', 'atomic' etc. directives.
43     InlinedRegion,
44   };
45 
46   CGOpenMPRegionInfo(const CapturedStmt &CS,
47                      const CGOpenMPRegionKind RegionKind,
48                      const RegionCodeGenTy &CodeGen)
49       : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind),
50         CodeGen(CodeGen) {}
51 
52   CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind,
53                      const RegionCodeGenTy &CodeGen)
54       : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind),
55         CodeGen(CodeGen) {}
56 
57   /// \brief Get a variable or parameter for storing global thread id
58   /// inside OpenMP construct.
59   virtual const VarDecl *getThreadIDVariable() const = 0;
60 
61   /// \brief Emit the captured statement body.
62   virtual void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
63 
64   /// \brief Get an LValue for the current ThreadID variable.
65   /// \return LValue for thread id variable. This LValue always has type int32*.
66   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
67 
68   CGOpenMPRegionKind getRegionKind() const { return RegionKind; }
69 
70   static bool classof(const CGCapturedStmtInfo *Info) {
71     return Info->getKind() == CR_OpenMP;
72   }
73 
74 protected:
75   CGOpenMPRegionKind RegionKind;
76   const RegionCodeGenTy &CodeGen;
77 };
78 
79 /// \brief API for captured statement code generation in OpenMP constructs.
80 class CGOpenMPOutlinedRegionInfo : public CGOpenMPRegionInfo {
81 public:
82   CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar,
83                              const RegionCodeGenTy &CodeGen)
84       : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen),
85         ThreadIDVar(ThreadIDVar) {
86     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
87   }
88   /// \brief Get a variable or parameter for storing global thread id
89   /// inside OpenMP construct.
90   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
91 
92   /// \brief Get the name of the capture helper.
93   StringRef getHelperName() const override { return ".omp_outlined."; }
94 
95   static bool classof(const CGCapturedStmtInfo *Info) {
96     return CGOpenMPRegionInfo::classof(Info) &&
97            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
98                ParallelOutlinedRegion;
99   }
100 
101 private:
102   /// \brief A variable or parameter storing global thread id for OpenMP
103   /// constructs.
104   const VarDecl *ThreadIDVar;
105 };
106 
107 /// \brief API for captured statement code generation in OpenMP constructs.
108 class CGOpenMPTaskOutlinedRegionInfo : public CGOpenMPRegionInfo {
109 public:
110   CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS,
111                                  const VarDecl *ThreadIDVar,
112                                  const RegionCodeGenTy &CodeGen)
113       : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen),
114         ThreadIDVar(ThreadIDVar) {
115     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
116   }
117   /// \brief Get a variable or parameter for storing global thread id
118   /// inside OpenMP construct.
119   const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; }
120 
121   /// \brief Get an LValue for the current ThreadID variable.
122   LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
123 
124   /// \brief Get the name of the capture helper.
125   StringRef getHelperName() const override { return ".omp_outlined."; }
126 
127   static bool classof(const CGCapturedStmtInfo *Info) {
128     return CGOpenMPRegionInfo::classof(Info) &&
129            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() ==
130                TaskOutlinedRegion;
131   }
132 
133 private:
134   /// \brief A variable or parameter storing global thread id for OpenMP
135   /// constructs.
136   const VarDecl *ThreadIDVar;
137 };
138 
139 /// \brief API for inlined captured statement code generation in OpenMP
140 /// constructs.
141 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
142 public:
143   CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI,
144                             const RegionCodeGenTy &CodeGen)
145       : CGOpenMPRegionInfo(InlinedRegion, CodeGen), OldCSI(OldCSI),
146         OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
147   // \brief Retrieve the value of the context parameter.
148   llvm::Value *getContextValue() const override {
149     if (OuterRegionInfo)
150       return OuterRegionInfo->getContextValue();
151     llvm_unreachable("No context value for inlined OpenMP region");
152   }
153   virtual void setContextValue(llvm::Value *V) override {
154     if (OuterRegionInfo) {
155       OuterRegionInfo->setContextValue(V);
156       return;
157     }
158     llvm_unreachable("No context value for inlined OpenMP region");
159   }
160   /// \brief Lookup the captured field decl for a variable.
161   const FieldDecl *lookup(const VarDecl *VD) const override {
162     if (OuterRegionInfo)
163       return OuterRegionInfo->lookup(VD);
164     // If there is no outer outlined region,no need to lookup in a list of
165     // captured variables, we can use the original one.
166     return nullptr;
167   }
168   FieldDecl *getThisFieldDecl() const override {
169     if (OuterRegionInfo)
170       return OuterRegionInfo->getThisFieldDecl();
171     return nullptr;
172   }
173   /// \brief Get a variable or parameter for storing global thread id
174   /// inside OpenMP construct.
175   const VarDecl *getThreadIDVariable() const override {
176     if (OuterRegionInfo)
177       return OuterRegionInfo->getThreadIDVariable();
178     return nullptr;
179   }
180 
181   /// \brief Get the name of the capture helper.
182   StringRef getHelperName() const override {
183     if (auto *OuterRegionInfo = getOldCSI())
184       return OuterRegionInfo->getHelperName();
185     llvm_unreachable("No helper name for inlined OpenMP construct");
186   }
187 
188   CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
189 
190   static bool classof(const CGCapturedStmtInfo *Info) {
191     return CGOpenMPRegionInfo::classof(Info) &&
192            cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion;
193   }
194 
195 private:
196   /// \brief CodeGen info about outer OpenMP region.
197   CodeGenFunction::CGCapturedStmtInfo *OldCSI;
198   CGOpenMPRegionInfo *OuterRegionInfo;
199 };
200 
201 /// \brief RAII for emitting code of OpenMP constructs.
202 class InlinedOpenMPRegionRAII {
203   CodeGenFunction &CGF;
204 
205 public:
206   /// \brief Constructs region for combined constructs.
207   /// \param CodeGen Code generation sequence for combined directives. Includes
208   /// a list of functions used for code generation of implicitly inlined
209   /// regions.
210   InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen)
211       : CGF(CGF) {
212     // Start emission for the construct.
213     CGF.CapturedStmtInfo =
214         new CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, CodeGen);
215   }
216   ~InlinedOpenMPRegionRAII() {
217     // Restore original CapturedStmtInfo only if we're done with code emission.
218     auto *OldCSI =
219         cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
220     delete CGF.CapturedStmtInfo;
221     CGF.CapturedStmtInfo = OldCSI;
222   }
223 };
224 
225 } // namespace
226 
227 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
228   return CGF.MakeNaturalAlignAddrLValue(
229       CGF.Builder.CreateAlignedLoad(
230           CGF.GetAddrOfLocalVar(getThreadIDVariable()),
231           CGF.PointerAlignInBytes),
232       getThreadIDVariable()
233           ->getType()
234           ->castAs<PointerType>()
235           ->getPointeeType());
236 }
237 
238 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) {
239   // 1.2.2 OpenMP Language Terminology
240   // Structured block - An executable statement with a single entry at the
241   // top and a single exit at the bottom.
242   // The point of exit cannot be a branch out of the structured block.
243   // longjmp() and throw() must not violate the entry/exit criteria.
244   CGF.EHStack.pushTerminate();
245   {
246     CodeGenFunction::RunCleanupsScope Scope(CGF);
247     CodeGen(CGF);
248   }
249   CGF.EHStack.popTerminate();
250 }
251 
252 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
253     CodeGenFunction &CGF) {
254   return CGF.MakeNaturalAlignAddrLValue(
255       CGF.GetAddrOfLocalVar(getThreadIDVariable()),
256       getThreadIDVariable()->getType());
257 }
258 
259 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM)
260     : CGM(CGM), DefaultOpenMPPSource(nullptr), KmpRoutineEntryPtrTy(nullptr) {
261   IdentTy = llvm::StructType::create(
262       "ident_t", CGM.Int32Ty /* reserved_1 */, CGM.Int32Ty /* flags */,
263       CGM.Int32Ty /* reserved_2 */, CGM.Int32Ty /* reserved_3 */,
264       CGM.Int8PtrTy /* psource */, nullptr);
265   // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...)
266   llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty),
267                                llvm::PointerType::getUnqual(CGM.Int32Ty)};
268   Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true);
269   KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
270 }
271 
272 void CGOpenMPRuntime::clear() {
273   InternalVars.clear();
274 }
275 
276 llvm::Value *
277 CGOpenMPRuntime::emitParallelOutlinedFunction(const OMPExecutableDirective &D,
278                                               const VarDecl *ThreadIDVar,
279                                               const RegionCodeGenTy &CodeGen) {
280   assert(ThreadIDVar->getType()->isPointerType() &&
281          "thread id variable must be of type kmp_int32 *");
282   const CapturedStmt *CS = cast<CapturedStmt>(D.getAssociatedStmt());
283   CodeGenFunction CGF(CGM, true);
284   CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen);
285   CGF.CapturedStmtInfo = &CGInfo;
286   return CGF.GenerateCapturedStmtFunction(*CS);
287 }
288 
289 llvm::Value *
290 CGOpenMPRuntime::emitTaskOutlinedFunction(const OMPExecutableDirective &D,
291                                           const VarDecl *ThreadIDVar,
292                                           const RegionCodeGenTy &CodeGen) {
293   assert(!ThreadIDVar->getType()->isPointerType() &&
294          "thread id variable must be of type kmp_int32 for tasks");
295   auto *CS = cast<CapturedStmt>(D.getAssociatedStmt());
296   CodeGenFunction CGF(CGM, true);
297   CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen);
298   CGF.CapturedStmtInfo = &CGInfo;
299   return CGF.GenerateCapturedStmtFunction(*CS);
300 }
301 
302 llvm::Value *
303 CGOpenMPRuntime::getOrCreateDefaultLocation(OpenMPLocationFlags Flags) {
304   llvm::Value *Entry = OpenMPDefaultLocMap.lookup(Flags);
305   if (!Entry) {
306     if (!DefaultOpenMPPSource) {
307       // Initialize default location for psource field of ident_t structure of
308       // all ident_t objects. Format is ";file;function;line;column;;".
309       // Taken from
310       // http://llvm.org/svn/llvm-project/openmp/trunk/runtime/src/kmp_str.c
311       DefaultOpenMPPSource =
312           CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;");
313       DefaultOpenMPPSource =
314           llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy);
315     }
316     auto DefaultOpenMPLocation = new llvm::GlobalVariable(
317         CGM.getModule(), IdentTy, /*isConstant*/ true,
318         llvm::GlobalValue::PrivateLinkage, /*Initializer*/ nullptr);
319     DefaultOpenMPLocation->setUnnamedAddr(true);
320 
321     llvm::Constant *Zero = llvm::ConstantInt::get(CGM.Int32Ty, 0, true);
322     llvm::Constant *Values[] = {Zero,
323                                 llvm::ConstantInt::get(CGM.Int32Ty, Flags),
324                                 Zero, Zero, DefaultOpenMPPSource};
325     llvm::Constant *Init = llvm::ConstantStruct::get(IdentTy, Values);
326     DefaultOpenMPLocation->setInitializer(Init);
327     OpenMPDefaultLocMap[Flags] = DefaultOpenMPLocation;
328     return DefaultOpenMPLocation;
329   }
330   return Entry;
331 }
332 
333 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
334                                                  SourceLocation Loc,
335                                                  OpenMPLocationFlags Flags) {
336   // If no debug info is generated - return global default location.
337   if (CGM.getCodeGenOpts().getDebugInfo() == CodeGenOptions::NoDebugInfo ||
338       Loc.isInvalid())
339     return getOrCreateDefaultLocation(Flags);
340 
341   assert(CGF.CurFn && "No function in current CodeGenFunction.");
342 
343   llvm::Value *LocValue = nullptr;
344   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
345   if (I != OpenMPLocThreadIDMap.end())
346     LocValue = I->second.DebugLoc;
347   // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if
348   // GetOpenMPThreadID was called before this routine.
349   if (LocValue == nullptr) {
350     // Generate "ident_t .kmpc_loc.addr;"
351     llvm::AllocaInst *AI = CGF.CreateTempAlloca(IdentTy, ".kmpc_loc.addr");
352     AI->setAlignment(CGM.getDataLayout().getPrefTypeAlignment(IdentTy));
353     auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
354     Elem.second.DebugLoc = AI;
355     LocValue = AI;
356 
357     CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
358     CGF.Builder.SetInsertPoint(CGF.AllocaInsertPt);
359     CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags),
360                              llvm::ConstantExpr::getSizeOf(IdentTy),
361                              CGM.PointerAlignInBytes);
362   }
363 
364   // char **psource = &.kmpc_loc_<flags>.addr.psource;
365   auto *PSource = CGF.Builder.CreateConstInBoundsGEP2_32(IdentTy, LocValue, 0,
366                                                          IdentField_PSource);
367 
368   auto OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding());
369   if (OMPDebugLoc == nullptr) {
370     SmallString<128> Buffer2;
371     llvm::raw_svector_ostream OS2(Buffer2);
372     // Build debug location
373     PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
374     OS2 << ";" << PLoc.getFilename() << ";";
375     if (const FunctionDecl *FD =
376             dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) {
377       OS2 << FD->getQualifiedNameAsString();
378     }
379     OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
380     OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str());
381     OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc;
382   }
383   // *psource = ";<File>;<Function>;<Line>;<Column>;;";
384   CGF.Builder.CreateStore(OMPDebugLoc, PSource);
385 
386   return LocValue;
387 }
388 
389 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
390                                           SourceLocation Loc) {
391   assert(CGF.CurFn && "No function in current CodeGenFunction.");
392 
393   llvm::Value *ThreadID = nullptr;
394   // Check whether we've already cached a load of the thread id in this
395   // function.
396   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
397   if (I != OpenMPLocThreadIDMap.end()) {
398     ThreadID = I->second.ThreadID;
399     if (ThreadID != nullptr)
400       return ThreadID;
401   }
402   if (auto OMPRegionInfo =
403           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
404     if (OMPRegionInfo->getThreadIDVariable()) {
405       // Check if this an outlined function with thread id passed as argument.
406       auto LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
407       ThreadID = CGF.EmitLoadOfLValue(LVal, Loc).getScalarVal();
408       // If value loaded in entry block, cache it and use it everywhere in
409       // function.
410       if (CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) {
411         auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
412         Elem.second.ThreadID = ThreadID;
413       }
414       return ThreadID;
415     }
416   }
417 
418   // This is not an outlined function region - need to call __kmpc_int32
419   // kmpc_global_thread_num(ident_t *loc).
420   // Generate thread id value and cache this value for use across the
421   // function.
422   CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
423   CGF.Builder.SetInsertPoint(CGF.AllocaInsertPt);
424   ThreadID =
425       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
426                           emitUpdateLocation(CGF, Loc));
427   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
428   Elem.second.ThreadID = ThreadID;
429   return ThreadID;
430 }
431 
432 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
433   assert(CGF.CurFn && "No function in current CodeGenFunction.");
434   if (OpenMPLocThreadIDMap.count(CGF.CurFn))
435     OpenMPLocThreadIDMap.erase(CGF.CurFn);
436 }
437 
438 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
439   return llvm::PointerType::getUnqual(IdentTy);
440 }
441 
442 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() {
443   return llvm::PointerType::getUnqual(Kmpc_MicroTy);
444 }
445 
446 llvm::Constant *
447 CGOpenMPRuntime::createRuntimeFunction(OpenMPRTLFunction Function) {
448   llvm::Constant *RTLFn = nullptr;
449   switch (Function) {
450   case OMPRTL__kmpc_fork_call: {
451     // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro
452     // microtask, ...);
453     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
454                                 getKmpc_MicroPointerTy()};
455     llvm::FunctionType *FnTy =
456         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
457     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call");
458     break;
459   }
460   case OMPRTL__kmpc_global_thread_num: {
461     // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc);
462     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
463     llvm::FunctionType *FnTy =
464         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
465     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num");
466     break;
467   }
468   case OMPRTL__kmpc_threadprivate_cached: {
469     // Build void *__kmpc_threadprivate_cached(ident_t *loc,
470     // kmp_int32 global_tid, void *data, size_t size, void ***cache);
471     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
472                                 CGM.VoidPtrTy, CGM.SizeTy,
473                                 CGM.VoidPtrTy->getPointerTo()->getPointerTo()};
474     llvm::FunctionType *FnTy =
475         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false);
476     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached");
477     break;
478   }
479   case OMPRTL__kmpc_critical: {
480     // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
481     // kmp_critical_name *crit);
482     llvm::Type *TypeParams[] = {
483         getIdentTyPointerTy(), CGM.Int32Ty,
484         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
485     llvm::FunctionType *FnTy =
486         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
487     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical");
488     break;
489   }
490   case OMPRTL__kmpc_threadprivate_register: {
491     // Build void __kmpc_threadprivate_register(ident_t *, void *data,
492     // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
493     // typedef void *(*kmpc_ctor)(void *);
494     auto KmpcCtorTy =
495         llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
496                                 /*isVarArg*/ false)->getPointerTo();
497     // typedef void *(*kmpc_cctor)(void *, void *);
498     llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
499     auto KmpcCopyCtorTy =
500         llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs,
501                                 /*isVarArg*/ false)->getPointerTo();
502     // typedef void (*kmpc_dtor)(void *);
503     auto KmpcDtorTy =
504         llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false)
505             ->getPointerTo();
506     llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy,
507                               KmpcCopyCtorTy, KmpcDtorTy};
508     auto FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs,
509                                         /*isVarArg*/ false);
510     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register");
511     break;
512   }
513   case OMPRTL__kmpc_end_critical: {
514     // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
515     // kmp_critical_name *crit);
516     llvm::Type *TypeParams[] = {
517         getIdentTyPointerTy(), CGM.Int32Ty,
518         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
519     llvm::FunctionType *FnTy =
520         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
521     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical");
522     break;
523   }
524   case OMPRTL__kmpc_cancel_barrier: {
525     // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
526     // global_tid);
527     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
528     llvm::FunctionType *FnTy =
529         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
530     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier");
531     break;
532   }
533   case OMPRTL__kmpc_for_static_fini: {
534     // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
535     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
536     llvm::FunctionType *FnTy =
537         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
538     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini");
539     break;
540   }
541   case OMPRTL__kmpc_push_num_threads: {
542     // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
543     // kmp_int32 num_threads)
544     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
545                                 CGM.Int32Ty};
546     llvm::FunctionType *FnTy =
547         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
548     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads");
549     break;
550   }
551   case OMPRTL__kmpc_serialized_parallel: {
552     // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
553     // global_tid);
554     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
555     llvm::FunctionType *FnTy =
556         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
557     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel");
558     break;
559   }
560   case OMPRTL__kmpc_end_serialized_parallel: {
561     // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
562     // global_tid);
563     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
564     llvm::FunctionType *FnTy =
565         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
566     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel");
567     break;
568   }
569   case OMPRTL__kmpc_flush: {
570     // Build void __kmpc_flush(ident_t *loc);
571     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
572     llvm::FunctionType *FnTy =
573         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
574     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush");
575     break;
576   }
577   case OMPRTL__kmpc_master: {
578     // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid);
579     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
580     llvm::FunctionType *FnTy =
581         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
582     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master");
583     break;
584   }
585   case OMPRTL__kmpc_end_master: {
586     // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid);
587     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
588     llvm::FunctionType *FnTy =
589         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
590     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master");
591     break;
592   }
593   case OMPRTL__kmpc_omp_taskyield: {
594     // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
595     // int end_part);
596     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
597     llvm::FunctionType *FnTy =
598         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
599     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield");
600     break;
601   }
602   case OMPRTL__kmpc_single: {
603     // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid);
604     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
605     llvm::FunctionType *FnTy =
606         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
607     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single");
608     break;
609   }
610   case OMPRTL__kmpc_end_single: {
611     // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid);
612     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
613     llvm::FunctionType *FnTy =
614         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
615     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single");
616     break;
617   }
618   case OMPRTL__kmpc_omp_task_alloc: {
619     // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
620     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
621     // kmp_routine_entry_t *task_entry);
622     assert(KmpRoutineEntryPtrTy != nullptr &&
623            "Type kmp_routine_entry_t must be created.");
624     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
625                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy};
626     // Return void * and then cast to particular kmp_task_t type.
627     llvm::FunctionType *FnTy =
628         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
629     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc");
630     break;
631   }
632   case OMPRTL__kmpc_omp_task: {
633     // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
634     // *new_task);
635     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
636                                 CGM.VoidPtrTy};
637     llvm::FunctionType *FnTy =
638         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
639     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task");
640     break;
641   }
642   case OMPRTL__kmpc_copyprivate: {
643     // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid,
644     // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *),
645     // kmp_int32 didit);
646     llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
647     auto *CpyFnTy =
648         llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false);
649     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy,
650                                 CGM.VoidPtrTy, CpyFnTy->getPointerTo(),
651                                 CGM.Int32Ty};
652     llvm::FunctionType *FnTy =
653         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
654     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate");
655     break;
656   }
657   case OMPRTL__kmpc_reduce: {
658     // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid,
659     // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void
660     // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck);
661     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
662     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
663                                                /*isVarArg=*/false);
664     llvm::Type *TypeParams[] = {
665         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
666         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
667         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
668     llvm::FunctionType *FnTy =
669         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
670     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce");
671     break;
672   }
673   case OMPRTL__kmpc_reduce_nowait: {
674     // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32
675     // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data,
676     // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name
677     // *lck);
678     llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
679     auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams,
680                                                /*isVarArg=*/false);
681     llvm::Type *TypeParams[] = {
682         getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy,
683         CGM.VoidPtrTy, ReduceFnTy->getPointerTo(),
684         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
685     llvm::FunctionType *FnTy =
686         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
687     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait");
688     break;
689   }
690   case OMPRTL__kmpc_end_reduce: {
691     // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid,
692     // kmp_critical_name *lck);
693     llvm::Type *TypeParams[] = {
694         getIdentTyPointerTy(), CGM.Int32Ty,
695         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
696     llvm::FunctionType *FnTy =
697         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
698     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce");
699     break;
700   }
701   case OMPRTL__kmpc_end_reduce_nowait: {
702     // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid,
703     // kmp_critical_name *lck);
704     llvm::Type *TypeParams[] = {
705         getIdentTyPointerTy(), CGM.Int32Ty,
706         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
707     llvm::FunctionType *FnTy =
708         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
709     RTLFn =
710         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait");
711     break;
712   }
713   case OMPRTL__kmpc_omp_task_begin_if0: {
714     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
715     // *new_task);
716     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
717                                 CGM.VoidPtrTy};
718     llvm::FunctionType *FnTy =
719         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
720     RTLFn =
721         CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0");
722     break;
723   }
724   case OMPRTL__kmpc_omp_task_complete_if0: {
725     // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
726     // *new_task);
727     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
728                                 CGM.VoidPtrTy};
729     llvm::FunctionType *FnTy =
730         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
731     RTLFn = CGM.CreateRuntimeFunction(FnTy,
732                                       /*Name=*/"__kmpc_omp_task_complete_if0");
733     break;
734   }
735   case OMPRTL__kmpc_ordered: {
736     // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
737     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
738     llvm::FunctionType *FnTy =
739         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
740     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered");
741     break;
742   }
743   case OMPRTL__kmpc_end_ordered: {
744     // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid);
745     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
746     llvm::FunctionType *FnTy =
747         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
748     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered");
749     break;
750   }
751   case OMPRTL__kmpc_omp_taskwait: {
752     // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid);
753     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
754     llvm::FunctionType *FnTy =
755         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
756     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait");
757     break;
758   }
759   }
760   return RTLFn;
761 }
762 
763 llvm::Constant *CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize,
764                                                              bool IVSigned) {
765   assert((IVSize == 32 || IVSize == 64) &&
766          "IV size is not compatible with the omp runtime");
767   auto Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4"
768                                        : "__kmpc_for_static_init_4u")
769                            : (IVSigned ? "__kmpc_for_static_init_8"
770                                        : "__kmpc_for_static_init_8u");
771   auto ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
772   auto PtrTy = llvm::PointerType::getUnqual(ITy);
773   llvm::Type *TypeParams[] = {
774     getIdentTyPointerTy(),                     // loc
775     CGM.Int32Ty,                               // tid
776     CGM.Int32Ty,                               // schedtype
777     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
778     PtrTy,                                     // p_lower
779     PtrTy,                                     // p_upper
780     PtrTy,                                     // p_stride
781     ITy,                                       // incr
782     ITy                                        // chunk
783   };
784   llvm::FunctionType *FnTy =
785       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
786   return CGM.CreateRuntimeFunction(FnTy, Name);
787 }
788 
789 llvm::Constant *CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize,
790                                                             bool IVSigned) {
791   assert((IVSize == 32 || IVSize == 64) &&
792          "IV size is not compatible with the omp runtime");
793   auto Name =
794       IVSize == 32
795           ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u")
796           : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u");
797   auto ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
798   llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc
799                                CGM.Int32Ty,           // tid
800                                CGM.Int32Ty,           // schedtype
801                                ITy,                   // lower
802                                ITy,                   // upper
803                                ITy,                   // stride
804                                ITy                    // chunk
805   };
806   llvm::FunctionType *FnTy =
807       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
808   return CGM.CreateRuntimeFunction(FnTy, Name);
809 }
810 
811 llvm::Constant *CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize,
812                                                             bool IVSigned) {
813   assert((IVSize == 32 || IVSize == 64) &&
814          "IV size is not compatible with the omp runtime");
815   auto Name =
816       IVSize == 32
817           ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u")
818           : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u");
819   llvm::Type *TypeParams[] = {
820       getIdentTyPointerTy(), // loc
821       CGM.Int32Ty,           // tid
822   };
823   llvm::FunctionType *FnTy =
824       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
825   return CGM.CreateRuntimeFunction(FnTy, Name);
826 }
827 
828 llvm::Constant *CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize,
829                                                             bool IVSigned) {
830   assert((IVSize == 32 || IVSize == 64) &&
831          "IV size is not compatible with the omp runtime");
832   auto Name =
833       IVSize == 32
834           ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u")
835           : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u");
836   auto ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
837   auto PtrTy = llvm::PointerType::getUnqual(ITy);
838   llvm::Type *TypeParams[] = {
839     getIdentTyPointerTy(),                     // loc
840     CGM.Int32Ty,                               // tid
841     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
842     PtrTy,                                     // p_lower
843     PtrTy,                                     // p_upper
844     PtrTy                                      // p_stride
845   };
846   llvm::FunctionType *FnTy =
847       llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
848   return CGM.CreateRuntimeFunction(FnTy, Name);
849 }
850 
851 llvm::Constant *
852 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
853   // Lookup the entry, lazily creating it if necessary.
854   return getOrCreateInternalVariable(CGM.Int8PtrPtrTy,
855                                      Twine(CGM.getMangledName(VD)) + ".cache.");
856 }
857 
858 llvm::Value *CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
859                                                      const VarDecl *VD,
860                                                      llvm::Value *VDAddr,
861                                                      SourceLocation Loc) {
862   auto VarTy = VDAddr->getType()->getPointerElementType();
863   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
864                          CGF.Builder.CreatePointerCast(VDAddr, CGM.Int8PtrTy),
865                          CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
866                          getOrCreateThreadPrivateCache(VD)};
867   return CGF.EmitRuntimeCall(
868       createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args);
869 }
870 
871 void CGOpenMPRuntime::emitThreadPrivateVarInit(
872     CodeGenFunction &CGF, llvm::Value *VDAddr, llvm::Value *Ctor,
873     llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
874   // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
875   // library.
876   auto OMPLoc = emitUpdateLocation(CGF, Loc);
877   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
878                       OMPLoc);
879   // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
880   // to register constructor/destructor for variable.
881   llvm::Value *Args[] = {OMPLoc,
882                          CGF.Builder.CreatePointerCast(VDAddr, CGM.VoidPtrTy),
883                          Ctor, CopyCtor, Dtor};
884   CGF.EmitRuntimeCall(
885       createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args);
886 }
887 
888 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
889     const VarDecl *VD, llvm::Value *VDAddr, SourceLocation Loc,
890     bool PerformInit, CodeGenFunction *CGF) {
891   VD = VD->getDefinition(CGM.getContext());
892   if (VD && ThreadPrivateWithDefinition.count(VD) == 0) {
893     ThreadPrivateWithDefinition.insert(VD);
894     QualType ASTTy = VD->getType();
895 
896     llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
897     auto Init = VD->getAnyInitializer();
898     if (CGM.getLangOpts().CPlusPlus && PerformInit) {
899       // Generate function that re-emits the declaration's initializer into the
900       // threadprivate copy of the variable VD
901       CodeGenFunction CtorCGF(CGM);
902       FunctionArgList Args;
903       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, SourceLocation(),
904                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy);
905       Args.push_back(&Dst);
906 
907       auto &FI = CGM.getTypes().arrangeFreeFunctionDeclaration(
908           CGM.getContext().VoidPtrTy, Args, FunctionType::ExtInfo(),
909           /*isVariadic=*/false);
910       auto FTy = CGM.getTypes().GetFunctionType(FI);
911       auto Fn = CGM.CreateGlobalInitOrDestructFunction(
912           FTy, ".__kmpc_global_ctor_.", Loc);
913       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
914                             Args, SourceLocation());
915       auto ArgVal = CtorCGF.EmitLoadOfScalar(
916           CtorCGF.GetAddrOfLocalVar(&Dst),
917           /*Volatile=*/false, CGM.PointerAlignInBytes,
918           CGM.getContext().VoidPtrTy, Dst.getLocation());
919       auto Arg = CtorCGF.Builder.CreatePointerCast(
920           ArgVal,
921           CtorCGF.ConvertTypeForMem(CGM.getContext().getPointerType(ASTTy)));
922       CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
923                                /*IsInitializer=*/true);
924       ArgVal = CtorCGF.EmitLoadOfScalar(
925           CtorCGF.GetAddrOfLocalVar(&Dst),
926           /*Volatile=*/false, CGM.PointerAlignInBytes,
927           CGM.getContext().VoidPtrTy, Dst.getLocation());
928       CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
929       CtorCGF.FinishFunction();
930       Ctor = Fn;
931     }
932     if (VD->getType().isDestructedType() != QualType::DK_none) {
933       // Generate function that emits destructor call for the threadprivate copy
934       // of the variable VD
935       CodeGenFunction DtorCGF(CGM);
936       FunctionArgList Args;
937       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, SourceLocation(),
938                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy);
939       Args.push_back(&Dst);
940 
941       auto &FI = CGM.getTypes().arrangeFreeFunctionDeclaration(
942           CGM.getContext().VoidTy, Args, FunctionType::ExtInfo(),
943           /*isVariadic=*/false);
944       auto FTy = CGM.getTypes().GetFunctionType(FI);
945       auto Fn = CGM.CreateGlobalInitOrDestructFunction(
946           FTy, ".__kmpc_global_dtor_.", Loc);
947       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
948                             SourceLocation());
949       auto ArgVal = DtorCGF.EmitLoadOfScalar(
950           DtorCGF.GetAddrOfLocalVar(&Dst),
951           /*Volatile=*/false, CGM.PointerAlignInBytes,
952           CGM.getContext().VoidPtrTy, Dst.getLocation());
953       DtorCGF.emitDestroy(ArgVal, ASTTy,
954                           DtorCGF.getDestroyer(ASTTy.isDestructedType()),
955                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
956       DtorCGF.FinishFunction();
957       Dtor = Fn;
958     }
959     // Do not emit init function if it is not required.
960     if (!Ctor && !Dtor)
961       return nullptr;
962 
963     llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
964     auto CopyCtorTy =
965         llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs,
966                                 /*isVarArg=*/false)->getPointerTo();
967     // Copying constructor for the threadprivate variable.
968     // Must be NULL - reserved by runtime, but currently it requires that this
969     // parameter is always NULL. Otherwise it fires assertion.
970     CopyCtor = llvm::Constant::getNullValue(CopyCtorTy);
971     if (Ctor == nullptr) {
972       auto CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
973                                             /*isVarArg=*/false)->getPointerTo();
974       Ctor = llvm::Constant::getNullValue(CtorTy);
975     }
976     if (Dtor == nullptr) {
977       auto DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy,
978                                             /*isVarArg=*/false)->getPointerTo();
979       Dtor = llvm::Constant::getNullValue(DtorTy);
980     }
981     if (!CGF) {
982       auto InitFunctionTy =
983           llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
984       auto InitFunction = CGM.CreateGlobalInitOrDestructFunction(
985           InitFunctionTy, ".__omp_threadprivate_init_.");
986       CodeGenFunction InitCGF(CGM);
987       FunctionArgList ArgList;
988       InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
989                             CGM.getTypes().arrangeNullaryFunction(), ArgList,
990                             Loc);
991       emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
992       InitCGF.FinishFunction();
993       return InitFunction;
994     }
995     emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
996   }
997   return nullptr;
998 }
999 
1000 /// \brief Emits code for OpenMP 'if' clause using specified \a CodeGen
1001 /// function. Here is the logic:
1002 /// if (Cond) {
1003 ///   ThenGen();
1004 /// } else {
1005 ///   ElseGen();
1006 /// }
1007 static void emitOMPIfClause(CodeGenFunction &CGF, const Expr *Cond,
1008                             const RegionCodeGenTy &ThenGen,
1009                             const RegionCodeGenTy &ElseGen) {
1010   CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
1011 
1012   // If the condition constant folds and can be elided, try to avoid emitting
1013   // the condition and the dead arm of the if/else.
1014   bool CondConstant;
1015   if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
1016     CodeGenFunction::RunCleanupsScope Scope(CGF);
1017     if (CondConstant) {
1018       ThenGen(CGF);
1019     } else {
1020       ElseGen(CGF);
1021     }
1022     return;
1023   }
1024 
1025   // Otherwise, the condition did not fold, or we couldn't elide it.  Just
1026   // emit the conditional branch.
1027   auto ThenBlock = CGF.createBasicBlock("omp_if.then");
1028   auto ElseBlock = CGF.createBasicBlock("omp_if.else");
1029   auto ContBlock = CGF.createBasicBlock("omp_if.end");
1030   CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0);
1031 
1032   // Emit the 'then' code.
1033   CGF.EmitBlock(ThenBlock);
1034   {
1035     CodeGenFunction::RunCleanupsScope ThenScope(CGF);
1036     ThenGen(CGF);
1037   }
1038   CGF.EmitBranch(ContBlock);
1039   // Emit the 'else' code if present.
1040   {
1041     // There is no need to emit line number for unconditional branch.
1042     auto NL = ApplyDebugLocation::CreateEmpty(CGF);
1043     CGF.EmitBlock(ElseBlock);
1044   }
1045   {
1046     CodeGenFunction::RunCleanupsScope ThenScope(CGF);
1047     ElseGen(CGF);
1048   }
1049   {
1050     // There is no need to emit line number for unconditional branch.
1051     auto NL = ApplyDebugLocation::CreateEmpty(CGF);
1052     CGF.EmitBranch(ContBlock);
1053   }
1054   // Emit the continuation block for code after the if.
1055   CGF.EmitBlock(ContBlock, /*IsFinished=*/true);
1056 }
1057 
1058 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc,
1059                                        llvm::Value *OutlinedFn,
1060                                        llvm::Value *CapturedStruct,
1061                                        const Expr *IfCond) {
1062   auto *RTLoc = emitUpdateLocation(CGF, Loc);
1063   auto &&ThenGen =
1064       [this, OutlinedFn, CapturedStruct, RTLoc](CodeGenFunction &CGF) {
1065         // Build call __kmpc_fork_call(loc, 1, microtask,
1066         // captured_struct/*context*/)
1067         llvm::Value *Args[] = {
1068             RTLoc,
1069             CGF.Builder.getInt32(
1070                 1), // Number of arguments after 'microtask' argument
1071             // (there is only one additional argument - 'context')
1072             CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy()),
1073             CGF.EmitCastToVoidPtr(CapturedStruct)};
1074         auto RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_call);
1075         CGF.EmitRuntimeCall(RTLFn, Args);
1076       };
1077   auto &&ElseGen = [this, OutlinedFn, CapturedStruct, RTLoc, Loc](
1078       CodeGenFunction &CGF) {
1079     auto ThreadID = getThreadID(CGF, Loc);
1080     // Build calls:
1081     // __kmpc_serialized_parallel(&Loc, GTid);
1082     llvm::Value *Args[] = {RTLoc, ThreadID};
1083     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_serialized_parallel),
1084                         Args);
1085 
1086     // OutlinedFn(&GTid, &zero, CapturedStruct);
1087     auto ThreadIDAddr = emitThreadIDAddress(CGF, Loc);
1088     auto Int32Ty = CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32,
1089                                                           /*Signed*/ true);
1090     auto ZeroAddr = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".zero.addr");
1091     CGF.InitTempAlloca(ZeroAddr, CGF.Builder.getInt32(/*C*/ 0));
1092     llvm::Value *OutlinedFnArgs[] = {ThreadIDAddr, ZeroAddr, CapturedStruct};
1093     CGF.EmitCallOrInvoke(OutlinedFn, OutlinedFnArgs);
1094 
1095     // __kmpc_end_serialized_parallel(&Loc, GTid);
1096     llvm::Value *EndArgs[] = {emitUpdateLocation(CGF, Loc), ThreadID};
1097     CGF.EmitRuntimeCall(
1098         createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel), EndArgs);
1099   };
1100   if (IfCond) {
1101     emitOMPIfClause(CGF, IfCond, ThenGen, ElseGen);
1102   } else {
1103     CodeGenFunction::RunCleanupsScope Scope(CGF);
1104     ThenGen(CGF);
1105   }
1106 }
1107 
1108 // If we're inside an (outlined) parallel region, use the region info's
1109 // thread-ID variable (it is passed in a first argument of the outlined function
1110 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
1111 // regular serial code region, get thread ID by calling kmp_int32
1112 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
1113 // return the address of that temp.
1114 llvm::Value *CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
1115                                                   SourceLocation Loc) {
1116   if (auto OMPRegionInfo =
1117           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
1118     if (OMPRegionInfo->getThreadIDVariable())
1119       return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress();
1120 
1121   auto ThreadID = getThreadID(CGF, Loc);
1122   auto Int32Ty =
1123       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
1124   auto ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp.");
1125   CGF.EmitStoreOfScalar(ThreadID,
1126                         CGF.MakeNaturalAlignAddrLValue(ThreadIDTemp, Int32Ty));
1127 
1128   return ThreadIDTemp;
1129 }
1130 
1131 llvm::Constant *
1132 CGOpenMPRuntime::getOrCreateInternalVariable(llvm::Type *Ty,
1133                                              const llvm::Twine &Name) {
1134   SmallString<256> Buffer;
1135   llvm::raw_svector_ostream Out(Buffer);
1136   Out << Name;
1137   auto RuntimeName = Out.str();
1138   auto &Elem = *InternalVars.insert(std::make_pair(RuntimeName, nullptr)).first;
1139   if (Elem.second) {
1140     assert(Elem.second->getType()->getPointerElementType() == Ty &&
1141            "OMP internal variable has different type than requested");
1142     return &*Elem.second;
1143   }
1144 
1145   return Elem.second = new llvm::GlobalVariable(
1146              CGM.getModule(), Ty, /*IsConstant*/ false,
1147              llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty),
1148              Elem.first());
1149 }
1150 
1151 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
1152   llvm::Twine Name(".gomp_critical_user_", CriticalName);
1153   return getOrCreateInternalVariable(KmpCriticalNameTy, Name.concat(".var"));
1154 }
1155 
1156 namespace {
1157 template <size_t N> class CallEndCleanup : public EHScopeStack::Cleanup {
1158   llvm::Value *Callee;
1159   llvm::Value *Args[N];
1160 
1161 public:
1162   CallEndCleanup(llvm::Value *Callee, ArrayRef<llvm::Value *> CleanupArgs)
1163       : Callee(Callee) {
1164     assert(CleanupArgs.size() == N);
1165     std::copy(CleanupArgs.begin(), CleanupArgs.end(), std::begin(Args));
1166   }
1167   void Emit(CodeGenFunction &CGF, Flags /*flags*/) override {
1168     CGF.EmitRuntimeCall(Callee, Args);
1169   }
1170 };
1171 } // namespace
1172 
1173 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF,
1174                                          StringRef CriticalName,
1175                                          const RegionCodeGenTy &CriticalOpGen,
1176                                          SourceLocation Loc) {
1177   // __kmpc_critical(ident_t *, gtid, Lock);
1178   // CriticalOpGen();
1179   // __kmpc_end_critical(ident_t *, gtid, Lock);
1180   // Prepare arguments and build a call to __kmpc_critical
1181   {
1182     CodeGenFunction::RunCleanupsScope Scope(CGF);
1183     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
1184                            getCriticalRegionLock(CriticalName)};
1185     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_critical), Args);
1186     // Build a call to __kmpc_end_critical
1187     CGF.EHStack.pushCleanup<CallEndCleanup<std::extent<decltype(Args)>::value>>(
1188         NormalAndEHCleanup, createRuntimeFunction(OMPRTL__kmpc_end_critical),
1189         llvm::makeArrayRef(Args));
1190     emitInlinedDirective(CGF, CriticalOpGen);
1191   }
1192 }
1193 
1194 static void emitIfStmt(CodeGenFunction &CGF, llvm::Value *IfCond,
1195                        const RegionCodeGenTy &BodyOpGen) {
1196   llvm::Value *CallBool = CGF.EmitScalarConversion(
1197       IfCond,
1198       CGF.getContext().getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true),
1199       CGF.getContext().BoolTy);
1200 
1201   auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
1202   auto *ContBlock = CGF.createBasicBlock("omp_if.end");
1203   // Generate the branch (If-stmt)
1204   CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
1205   CGF.EmitBlock(ThenBlock);
1206   CGF.CGM.getOpenMPRuntime().emitInlinedDirective(CGF, BodyOpGen);
1207   // Emit the rest of bblocks/branches
1208   CGF.EmitBranch(ContBlock);
1209   CGF.EmitBlock(ContBlock, true);
1210 }
1211 
1212 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
1213                                        const RegionCodeGenTy &MasterOpGen,
1214                                        SourceLocation Loc) {
1215   // if(__kmpc_master(ident_t *, gtid)) {
1216   //   MasterOpGen();
1217   //   __kmpc_end_master(ident_t *, gtid);
1218   // }
1219   // Prepare arguments and build a call to __kmpc_master
1220   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
1221   auto *IsMaster =
1222       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_master), Args);
1223   typedef CallEndCleanup<std::extent<decltype(Args)>::value>
1224       MasterCallEndCleanup;
1225   emitIfStmt(CGF, IsMaster, [&](CodeGenFunction &CGF) -> void {
1226     CodeGenFunction::RunCleanupsScope Scope(CGF);
1227     CGF.EHStack.pushCleanup<MasterCallEndCleanup>(
1228         NormalAndEHCleanup, createRuntimeFunction(OMPRTL__kmpc_end_master),
1229         llvm::makeArrayRef(Args));
1230     MasterOpGen(CGF);
1231   });
1232 }
1233 
1234 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
1235                                         SourceLocation Loc) {
1236   // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
1237   llvm::Value *Args[] = {
1238       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
1239       llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
1240   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), Args);
1241 }
1242 
1243 static llvm::Value *emitCopyprivateCopyFunction(
1244     CodeGenModule &CGM, llvm::Type *ArgsType,
1245     ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs,
1246     ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps) {
1247   auto &C = CGM.getContext();
1248   // void copy_func(void *LHSArg, void *RHSArg);
1249   FunctionArgList Args;
1250   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, SourceLocation(), /*Id=*/nullptr,
1251                            C.VoidPtrTy);
1252   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, SourceLocation(), /*Id=*/nullptr,
1253                            C.VoidPtrTy);
1254   Args.push_back(&LHSArg);
1255   Args.push_back(&RHSArg);
1256   FunctionType::ExtInfo EI;
1257   auto &CGFI = CGM.getTypes().arrangeFreeFunctionDeclaration(
1258       C.VoidTy, Args, EI, /*isVariadic=*/false);
1259   auto *Fn = llvm::Function::Create(
1260       CGM.getTypes().GetFunctionType(CGFI), llvm::GlobalValue::InternalLinkage,
1261       ".omp.copyprivate.copy_func", &CGM.getModule());
1262   CGM.SetLLVMFunctionAttributes(/*D=*/nullptr, CGFI, Fn);
1263   CodeGenFunction CGF(CGM);
1264   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args);
1265   // Dest = (void*[n])(LHSArg);
1266   // Src = (void*[n])(RHSArg);
1267   auto *LHS = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1268       CGF.Builder.CreateAlignedLoad(CGF.GetAddrOfLocalVar(&LHSArg),
1269                                     CGF.PointerAlignInBytes),
1270       ArgsType);
1271   auto *RHS = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1272       CGF.Builder.CreateAlignedLoad(CGF.GetAddrOfLocalVar(&RHSArg),
1273                                     CGF.PointerAlignInBytes),
1274       ArgsType);
1275   // *(Type0*)Dst[0] = *(Type0*)Src[0];
1276   // *(Type1*)Dst[1] = *(Type1*)Src[1];
1277   // ...
1278   // *(Typen*)Dst[n] = *(Typen*)Src[n];
1279   for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) {
1280     auto *DestAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1281         CGF.Builder.CreateAlignedLoad(
1282             CGF.Builder.CreateStructGEP(nullptr, LHS, I),
1283             CGM.PointerAlignInBytes),
1284         CGF.ConvertTypeForMem(C.getPointerType(SrcExprs[I]->getType())));
1285     auto *SrcAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1286         CGF.Builder.CreateAlignedLoad(
1287             CGF.Builder.CreateStructGEP(nullptr, RHS, I),
1288             CGM.PointerAlignInBytes),
1289         CGF.ConvertTypeForMem(C.getPointerType(SrcExprs[I]->getType())));
1290     auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl();
1291     QualType Type = VD->getType();
1292     if (auto *PVD = dyn_cast<ParmVarDecl>(VD)) {
1293       Type = PVD->getOriginalType();
1294     }
1295     CGF.EmitOMPCopy(CGF, Type, DestAddr, SrcAddr,
1296                     cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()),
1297                     cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()),
1298                     AssignmentOps[I]);
1299   }
1300   CGF.FinishFunction();
1301   return Fn;
1302 }
1303 
1304 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
1305                                        const RegionCodeGenTy &SingleOpGen,
1306                                        SourceLocation Loc,
1307                                        ArrayRef<const Expr *> CopyprivateVars,
1308                                        ArrayRef<const Expr *> SrcExprs,
1309                                        ArrayRef<const Expr *> DstExprs,
1310                                        ArrayRef<const Expr *> AssignmentOps) {
1311   assert(CopyprivateVars.size() == SrcExprs.size() &&
1312          CopyprivateVars.size() == DstExprs.size() &&
1313          CopyprivateVars.size() == AssignmentOps.size());
1314   auto &C = CGM.getContext();
1315   // int32 did_it = 0;
1316   // if(__kmpc_single(ident_t *, gtid)) {
1317   //   SingleOpGen();
1318   //   __kmpc_end_single(ident_t *, gtid);
1319   //   did_it = 1;
1320   // }
1321   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
1322   // <copy_func>, did_it);
1323 
1324   llvm::AllocaInst *DidIt = nullptr;
1325   if (!CopyprivateVars.empty()) {
1326     // int32 did_it = 0;
1327     auto KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
1328     DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it");
1329     CGF.Builder.CreateAlignedStore(CGF.Builder.getInt32(0), DidIt,
1330                                    DidIt->getAlignment());
1331   }
1332   // Prepare arguments and build a call to __kmpc_single
1333   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
1334   auto *IsSingle =
1335       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_single), Args);
1336   typedef CallEndCleanup<std::extent<decltype(Args)>::value>
1337       SingleCallEndCleanup;
1338   emitIfStmt(CGF, IsSingle, [&](CodeGenFunction &CGF) -> void {
1339     CodeGenFunction::RunCleanupsScope Scope(CGF);
1340     CGF.EHStack.pushCleanup<SingleCallEndCleanup>(
1341         NormalAndEHCleanup, createRuntimeFunction(OMPRTL__kmpc_end_single),
1342         llvm::makeArrayRef(Args));
1343     SingleOpGen(CGF);
1344     if (DidIt) {
1345       // did_it = 1;
1346       CGF.Builder.CreateAlignedStore(CGF.Builder.getInt32(1), DidIt,
1347                                      DidIt->getAlignment());
1348     }
1349   });
1350   // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>,
1351   // <copy_func>, did_it);
1352   if (DidIt) {
1353     llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size());
1354     auto CopyprivateArrayTy =
1355         C.getConstantArrayType(C.VoidPtrTy, ArraySize, ArrayType::Normal,
1356                                /*IndexTypeQuals=*/0);
1357     // Create a list of all private variables for copyprivate.
1358     auto *CopyprivateList =
1359         CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list");
1360     for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) {
1361       auto *Elem = CGF.Builder.CreateStructGEP(
1362           CopyprivateList->getAllocatedType(), CopyprivateList, I);
1363       CGF.Builder.CreateAlignedStore(
1364           CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1365               CGF.EmitLValue(CopyprivateVars[I]).getAddress(), CGF.VoidPtrTy),
1366           Elem, CGM.PointerAlignInBytes);
1367     }
1368     // Build function that copies private values from single region to all other
1369     // threads in the corresponding parallel region.
1370     auto *CpyFn = emitCopyprivateCopyFunction(
1371         CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(),
1372         CopyprivateVars, SrcExprs, DstExprs, AssignmentOps);
1373     auto *BufSize = llvm::ConstantInt::get(
1374         CGM.SizeTy, C.getTypeSizeInChars(CopyprivateArrayTy).getQuantity());
1375     auto *CL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList,
1376                                                                CGF.VoidPtrTy);
1377     auto *DidItVal =
1378         CGF.Builder.CreateAlignedLoad(DidIt, CGF.PointerAlignInBytes);
1379     llvm::Value *Args[] = {
1380         emitUpdateLocation(CGF, Loc), // ident_t *<loc>
1381         getThreadID(CGF, Loc),        // i32 <gtid>
1382         BufSize,                      // size_t <buf_size>
1383         CL,                           // void *<copyprivate list>
1384         CpyFn,                        // void (*) (void *, void *) <copy_func>
1385         DidItVal                      // i32 did_it
1386     };
1387     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args);
1388   }
1389 }
1390 
1391 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF,
1392                                         const RegionCodeGenTy &OrderedOpGen,
1393                                         SourceLocation Loc) {
1394   // __kmpc_ordered(ident_t *, gtid);
1395   // OrderedOpGen();
1396   // __kmpc_end_ordered(ident_t *, gtid);
1397   // Prepare arguments and build a call to __kmpc_ordered
1398   {
1399     CodeGenFunction::RunCleanupsScope Scope(CGF);
1400     llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
1401     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_ordered), Args);
1402     // Build a call to __kmpc_end_ordered
1403     CGF.EHStack.pushCleanup<CallEndCleanup<std::extent<decltype(Args)>::value>>(
1404         NormalAndEHCleanup, createRuntimeFunction(OMPRTL__kmpc_end_ordered),
1405         llvm::makeArrayRef(Args));
1406     emitInlinedDirective(CGF, OrderedOpGen);
1407   }
1408 }
1409 
1410 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
1411                                       OpenMPDirectiveKind Kind) {
1412   // Build call __kmpc_cancel_barrier(loc, thread_id);
1413   OpenMPLocationFlags Flags = OMP_IDENT_KMPC;
1414   if (Kind == OMPD_for) {
1415     Flags =
1416         static_cast<OpenMPLocationFlags>(Flags | OMP_IDENT_BARRIER_IMPL_FOR);
1417   } else if (Kind == OMPD_sections) {
1418     Flags = static_cast<OpenMPLocationFlags>(Flags |
1419                                              OMP_IDENT_BARRIER_IMPL_SECTIONS);
1420   } else if (Kind == OMPD_single) {
1421     Flags =
1422         static_cast<OpenMPLocationFlags>(Flags | OMP_IDENT_BARRIER_IMPL_SINGLE);
1423   } else if (Kind == OMPD_barrier) {
1424     Flags = static_cast<OpenMPLocationFlags>(Flags | OMP_IDENT_BARRIER_EXPL);
1425   } else {
1426     Flags = static_cast<OpenMPLocationFlags>(Flags | OMP_IDENT_BARRIER_IMPL);
1427   }
1428   // Build call __kmpc_cancel_barrier(loc, thread_id);
1429   // Replace __kmpc_barrier() function by __kmpc_cancel_barrier() because this
1430   // one provides the same functionality and adds initial support for
1431   // cancellation constructs introduced in OpenMP 4.0. __kmpc_cancel_barrier()
1432   // is provided default by the runtime library so it safe to make such
1433   // replacement.
1434   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
1435                          getThreadID(CGF, Loc)};
1436   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args);
1437 }
1438 
1439 /// \brief Schedule types for 'omp for' loops (these enumerators are taken from
1440 /// the enum sched_type in kmp.h).
1441 enum OpenMPSchedType {
1442   /// \brief Lower bound for default (unordered) versions.
1443   OMP_sch_lower = 32,
1444   OMP_sch_static_chunked = 33,
1445   OMP_sch_static = 34,
1446   OMP_sch_dynamic_chunked = 35,
1447   OMP_sch_guided_chunked = 36,
1448   OMP_sch_runtime = 37,
1449   OMP_sch_auto = 38,
1450   /// \brief Lower bound for 'ordered' versions.
1451   OMP_ord_lower = 64,
1452   /// \brief Lower bound for 'nomerge' versions.
1453   OMP_nm_lower = 160,
1454 };
1455 
1456 /// \brief Map the OpenMP loop schedule to the runtime enumeration.
1457 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
1458                                           bool Chunked) {
1459   switch (ScheduleKind) {
1460   case OMPC_SCHEDULE_static:
1461     return Chunked ? OMP_sch_static_chunked : OMP_sch_static;
1462   case OMPC_SCHEDULE_dynamic:
1463     return OMP_sch_dynamic_chunked;
1464   case OMPC_SCHEDULE_guided:
1465     return OMP_sch_guided_chunked;
1466   case OMPC_SCHEDULE_auto:
1467     return OMP_sch_auto;
1468   case OMPC_SCHEDULE_runtime:
1469     return OMP_sch_runtime;
1470   case OMPC_SCHEDULE_unknown:
1471     assert(!Chunked && "chunk was specified but schedule kind not known");
1472     return OMP_sch_static;
1473   }
1474   llvm_unreachable("Unexpected runtime schedule");
1475 }
1476 
1477 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
1478                                          bool Chunked) const {
1479   auto Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
1480   return Schedule == OMP_sch_static;
1481 }
1482 
1483 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
1484   auto Schedule = getRuntimeSchedule(ScheduleKind, /* Chunked */ false);
1485   assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
1486   return Schedule != OMP_sch_static;
1487 }
1488 
1489 void CGOpenMPRuntime::emitForInit(CodeGenFunction &CGF, SourceLocation Loc,
1490                                   OpenMPScheduleClauseKind ScheduleKind,
1491                                   unsigned IVSize, bool IVSigned,
1492                                   llvm::Value *IL, llvm::Value *LB,
1493                                   llvm::Value *UB, llvm::Value *ST,
1494                                   llvm::Value *Chunk) {
1495   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunk != nullptr);
1496   if (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked) {
1497     // Call __kmpc_dispatch_init(
1498     //          ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
1499     //          kmp_int[32|64] lower, kmp_int[32|64] upper,
1500     //          kmp_int[32|64] stride, kmp_int[32|64] chunk);
1501 
1502     // If the Chunk was not specified in the clause - use default value 1.
1503     if (Chunk == nullptr)
1504       Chunk = CGF.Builder.getIntN(IVSize, 1);
1505     llvm::Value *Args[] = { emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC),
1506                             getThreadID(CGF, Loc),
1507                             CGF.Builder.getInt32(Schedule), // Schedule type
1508                             CGF.Builder.getIntN(IVSize, 0), // Lower
1509                             UB,                             // Upper
1510                             CGF.Builder.getIntN(IVSize, 1), // Stride
1511                             Chunk                           // Chunk
1512     };
1513     CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args);
1514   } else {
1515     // Call __kmpc_for_static_init(
1516     //          ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
1517     //          kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
1518     //          kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
1519     //          kmp_int[32|64] incr, kmp_int[32|64] chunk);
1520     if (Chunk == nullptr) {
1521       assert(Schedule == OMP_sch_static &&
1522              "expected static non-chunked schedule");
1523       // If the Chunk was not specified in the clause - use default value 1.
1524       Chunk = CGF.Builder.getIntN(IVSize, 1);
1525     } else
1526       assert(Schedule == OMP_sch_static_chunked &&
1527              "expected static chunked schedule");
1528     llvm::Value *Args[] = { emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC),
1529                             getThreadID(CGF, Loc),
1530                             CGF.Builder.getInt32(Schedule), // Schedule type
1531                             IL,                             // &isLastIter
1532                             LB,                             // &LB
1533                             UB,                             // &UB
1534                             ST,                             // &Stride
1535                             CGF.Builder.getIntN(IVSize, 1), // Incr
1536                             Chunk                           // Chunk
1537     };
1538     CGF.EmitRuntimeCall(createForStaticInitFunction(IVSize, IVSigned), Args);
1539   }
1540 }
1541 
1542 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF,
1543                                           SourceLocation Loc) {
1544   // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
1545   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC),
1546                          getThreadID(CGF, Loc)};
1547   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini),
1548                       Args);
1549 }
1550 
1551 void CGOpenMPRuntime::emitForOrderedDynamicIterationEnd(CodeGenFunction &CGF,
1552                                                         SourceLocation Loc,
1553                                                         unsigned IVSize,
1554                                                         bool IVSigned) {
1555   // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid);
1556   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC),
1557                          getThreadID(CGF, Loc)};
1558   CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args);
1559 }
1560 
1561 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
1562                                           SourceLocation Loc, unsigned IVSize,
1563                                           bool IVSigned, llvm::Value *IL,
1564                                           llvm::Value *LB, llvm::Value *UB,
1565                                           llvm::Value *ST) {
1566   // Call __kmpc_dispatch_next(
1567   //          ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
1568   //          kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
1569   //          kmp_int[32|64] *p_stride);
1570   llvm::Value *Args[] = {
1571       emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC), getThreadID(CGF, Loc),
1572       IL, // &isLastIter
1573       LB, // &Lower
1574       UB, // &Upper
1575       ST  // &Stride
1576   };
1577   llvm::Value *Call =
1578       CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args);
1579   return CGF.EmitScalarConversion(
1580       Call, CGF.getContext().getIntTypeForBitwidth(32, /* Signed */ true),
1581       CGF.getContext().BoolTy);
1582 }
1583 
1584 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
1585                                            llvm::Value *NumThreads,
1586                                            SourceLocation Loc) {
1587   // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
1588   llvm::Value *Args[] = {
1589       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
1590       CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)};
1591   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads),
1592                       Args);
1593 }
1594 
1595 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
1596                                 SourceLocation Loc) {
1597   // Build call void __kmpc_flush(ident_t *loc)
1598   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush),
1599                       emitUpdateLocation(CGF, Loc));
1600 }
1601 
1602 namespace {
1603 /// \brief Indexes of fields for type kmp_task_t.
1604 enum KmpTaskTFields {
1605   /// \brief List of shared variables.
1606   KmpTaskTShareds,
1607   /// \brief Task routine.
1608   KmpTaskTRoutine,
1609   /// \brief Partition id for the untied tasks.
1610   KmpTaskTPartId,
1611   /// \brief Function with call of destructors for private variables.
1612   KmpTaskTDestructors,
1613 };
1614 } // namespace
1615 
1616 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
1617   if (!KmpRoutineEntryPtrTy) {
1618     // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
1619     auto &C = CGM.getContext();
1620     QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
1621     FunctionProtoType::ExtProtoInfo EPI;
1622     KmpRoutineEntryPtrQTy = C.getPointerType(
1623         C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
1624     KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
1625   }
1626 }
1627 
1628 static void addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1629                                  QualType FieldTy) {
1630   auto *Field = FieldDecl::Create(
1631       C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1632       C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1633       /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1634   Field->setAccess(AS_public);
1635   DC->addDecl(Field);
1636 }
1637 
1638 namespace {
1639 struct PrivateHelpersTy {
1640   PrivateHelpersTy(const VarDecl *Original, const VarDecl *PrivateCopy,
1641                    const VarDecl *PrivateElemInit)
1642       : Original(Original), PrivateCopy(PrivateCopy),
1643         PrivateElemInit(PrivateElemInit) {}
1644   const VarDecl *Original;
1645   const VarDecl *PrivateCopy;
1646   const VarDecl *PrivateElemInit;
1647 };
1648 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy;
1649 } // namespace
1650 
1651 static RecordDecl *
1652 createPrivatesRecordDecl(CodeGenModule &CGM,
1653                          const ArrayRef<PrivateDataTy> Privates) {
1654   if (!Privates.empty()) {
1655     auto &C = CGM.getContext();
1656     // Build struct .kmp_privates_t. {
1657     //         /*  private vars  */
1658     //       };
1659     auto *RD = C.buildImplicitRecord(".kmp_privates.t");
1660     RD->startDefinition();
1661     for (auto &&Pair : Privates) {
1662       auto Type = Pair.second.Original->getType();
1663       if (auto *PVD = dyn_cast<ParmVarDecl>(Pair.second.Original)) {
1664         Type = PVD->getOriginalType();
1665       }
1666       Type = Type.getNonReferenceType();
1667       addFieldToRecordDecl(C, RD, Type);
1668     }
1669     RD->completeDefinition();
1670     return RD;
1671   }
1672   return nullptr;
1673 }
1674 
1675 static RecordDecl *
1676 createKmpTaskTRecordDecl(CodeGenModule &CGM, QualType KmpInt32Ty,
1677                          QualType KmpRoutineEntryPointerQTy) {
1678   auto &C = CGM.getContext();
1679   // Build struct kmp_task_t {
1680   //         void *              shareds;
1681   //         kmp_routine_entry_t routine;
1682   //         kmp_int32           part_id;
1683   //         kmp_routine_entry_t destructors;
1684   //       };
1685   auto *RD = C.buildImplicitRecord("kmp_task_t");
1686   RD->startDefinition();
1687   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
1688   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
1689   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1690   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
1691   RD->completeDefinition();
1692   return RD;
1693 }
1694 
1695 static RecordDecl *
1696 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy,
1697                                      const ArrayRef<PrivateDataTy> Privates) {
1698   auto &C = CGM.getContext();
1699   // Build struct kmp_task_t_with_privates {
1700   //         kmp_task_t task_data;
1701   //         .kmp_privates_t. privates;
1702   //       };
1703   auto *RD = C.buildImplicitRecord("kmp_task_t_with_privates");
1704   RD->startDefinition();
1705   addFieldToRecordDecl(C, RD, KmpTaskTQTy);
1706   if (auto *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) {
1707     addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD));
1708   }
1709   RD->completeDefinition();
1710   return RD;
1711 }
1712 
1713 /// \brief Emit a proxy function which accepts kmp_task_t as the second
1714 /// argument.
1715 /// \code
1716 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
1717 ///   TaskFunction(gtid, tt->part_id, tt->shareds);
1718 ///   return 0;
1719 /// }
1720 /// \endcode
1721 static llvm::Value *
1722 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
1723                       QualType KmpInt32Ty, QualType KmpTaskTWithPrivatesPtrQTy,
1724                       QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy,
1725                       QualType SharedsPtrTy, llvm::Value *TaskFunction) {
1726   auto &C = CGM.getContext();
1727   FunctionArgList Args;
1728   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty);
1729   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc,
1730                                 /*Id=*/nullptr, KmpTaskTWithPrivatesPtrQTy);
1731   Args.push_back(&GtidArg);
1732   Args.push_back(&TaskTypeArg);
1733   FunctionType::ExtInfo Info;
1734   auto &TaskEntryFnInfo =
1735       CGM.getTypes().arrangeFreeFunctionDeclaration(KmpInt32Ty, Args, Info,
1736                                                     /*isVariadic=*/false);
1737   auto *TaskEntryTy = CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
1738   auto *TaskEntry =
1739       llvm::Function::Create(TaskEntryTy, llvm::GlobalValue::InternalLinkage,
1740                              ".omp_task_entry.", &CGM.getModule());
1741   CGM.SetLLVMFunctionAttributes(/*D=*/nullptr, TaskEntryFnInfo, TaskEntry);
1742   CodeGenFunction CGF(CGM);
1743   CGF.disableDebugInfo();
1744   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args);
1745 
1746   // TaskFunction(gtid, tt->task_data.part_id, tt->task_data.shareds);
1747   auto *GtidParam = CGF.EmitLoadOfScalar(
1748       CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false,
1749       C.getTypeAlignInChars(KmpInt32Ty).getQuantity(), KmpInt32Ty, Loc);
1750   auto *TaskTypeArgAddr = CGF.Builder.CreateAlignedLoad(
1751       CGF.GetAddrOfLocalVar(&TaskTypeArg), CGM.PointerAlignInBytes);
1752   LValue Base =
1753       CGF.MakeNaturalAlignAddrLValue(TaskTypeArgAddr, KmpTaskTWithPrivatesQTy);
1754   auto *KmpTaskTWithPrivatesQTyRD =
1755       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
1756   Base =
1757       CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
1758   auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
1759   auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId);
1760   auto PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI);
1761   auto *PartidParam = CGF.EmitLoadOfLValue(PartIdLVal, Loc).getScalarVal();
1762 
1763   auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds);
1764   auto SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI);
1765   auto *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1766       CGF.EmitLoadOfLValue(SharedsLVal, Loc).getScalarVal(),
1767       CGF.ConvertTypeForMem(SharedsPtrTy));
1768 
1769   llvm::Value *CallArgs[] = {GtidParam, PartidParam, SharedsParam};
1770   CGF.EmitCallOrInvoke(TaskFunction, CallArgs);
1771   CGF.EmitStoreThroughLValue(
1772       RValue::get(CGF.Builder.getInt32(/*C=*/0)),
1773       CGF.MakeNaturalAlignAddrLValue(CGF.ReturnValue, KmpInt32Ty));
1774   CGF.FinishFunction();
1775   return TaskEntry;
1776 }
1777 
1778 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM,
1779                                             SourceLocation Loc,
1780                                             QualType KmpInt32Ty,
1781                                             QualType KmpTaskTWithPrivatesPtrQTy,
1782                                             QualType KmpTaskTWithPrivatesQTy) {
1783   auto &C = CGM.getContext();
1784   FunctionArgList Args;
1785   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty);
1786   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc,
1787                                 /*Id=*/nullptr, KmpTaskTWithPrivatesPtrQTy);
1788   Args.push_back(&GtidArg);
1789   Args.push_back(&TaskTypeArg);
1790   FunctionType::ExtInfo Info;
1791   auto &DestructorFnInfo =
1792       CGM.getTypes().arrangeFreeFunctionDeclaration(KmpInt32Ty, Args, Info,
1793                                                     /*isVariadic=*/false);
1794   auto *DestructorFnTy = CGM.getTypes().GetFunctionType(DestructorFnInfo);
1795   auto *DestructorFn =
1796       llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage,
1797                              ".omp_task_destructor.", &CGM.getModule());
1798   CGM.SetLLVMFunctionAttributes(/*D=*/nullptr, DestructorFnInfo, DestructorFn);
1799   CodeGenFunction CGF(CGM);
1800   CGF.disableDebugInfo();
1801   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo,
1802                     Args);
1803 
1804   auto *TaskTypeArgAddr = CGF.Builder.CreateAlignedLoad(
1805       CGF.GetAddrOfLocalVar(&TaskTypeArg), CGM.PointerAlignInBytes);
1806   LValue Base =
1807       CGF.MakeNaturalAlignAddrLValue(TaskTypeArgAddr, KmpTaskTWithPrivatesQTy);
1808   auto *KmpTaskTWithPrivatesQTyRD =
1809       cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl());
1810   auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
1811   Base = CGF.EmitLValueForField(Base, *FI);
1812   for (auto *Field :
1813        cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) {
1814     if (auto DtorKind = Field->getType().isDestructedType()) {
1815       auto FieldLValue = CGF.EmitLValueForField(Base, Field);
1816       CGF.pushDestroy(DtorKind, FieldLValue.getAddress(), Field->getType());
1817     }
1818   }
1819   CGF.FinishFunction();
1820   return DestructorFn;
1821 }
1822 
1823 static int array_pod_sort_comparator(const PrivateDataTy *P1,
1824                                      const PrivateDataTy *P2) {
1825   return P1->first < P2->first ? 1 : (P2->first < P1->first ? -1 : 0);
1826 }
1827 
1828 void CGOpenMPRuntime::emitTaskCall(
1829     CodeGenFunction &CGF, SourceLocation Loc, const OMPExecutableDirective &D,
1830     bool Tied, llvm::PointerIntPair<llvm::Value *, 1, bool> Final,
1831     llvm::Value *TaskFunction, QualType SharedsTy, llvm::Value *Shareds,
1832     const Expr *IfCond, const ArrayRef<const Expr *> PrivateVars,
1833     const ArrayRef<const Expr *> PrivateCopies,
1834     const ArrayRef<const Expr *> FirstprivateVars,
1835     const ArrayRef<const Expr *> FirstprivateCopies,
1836     const ArrayRef<const Expr *> FirstprivateInits) {
1837   auto &C = CGM.getContext();
1838   llvm::SmallVector<PrivateDataTy, 8> Privates;
1839   // Aggeregate privates and sort them by the alignment.
1840   auto I = PrivateCopies.begin();
1841   for (auto *E : PrivateVars) {
1842     auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
1843     Privates.push_back(std::make_pair(
1844         C.getTypeAlignInChars(VD->getType()),
1845         PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
1846                          /*PrivateElemInit=*/nullptr)));
1847     ++I;
1848   }
1849   I = FirstprivateCopies.begin();
1850   auto IElemInitRef = FirstprivateInits.begin();
1851   for (auto *E : FirstprivateVars) {
1852     auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
1853     Privates.push_back(std::make_pair(
1854         C.getTypeAlignInChars(VD->getType()),
1855         PrivateHelpersTy(
1856             VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()),
1857             cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))));
1858     ++I, ++IElemInitRef;
1859   }
1860   llvm::array_pod_sort(Privates.begin(), Privates.end(),
1861                        array_pod_sort_comparator);
1862   auto KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
1863   // Build type kmp_routine_entry_t (if not built yet).
1864   emitKmpRoutineEntryT(KmpInt32Ty);
1865   // Build type kmp_task_t (if not built yet).
1866   if (KmpTaskTQTy.isNull()) {
1867     KmpTaskTQTy = C.getRecordType(
1868         createKmpTaskTRecordDecl(CGM, KmpInt32Ty, KmpRoutineEntryPtrQTy));
1869   }
1870   auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl());
1871   // Build particular struct kmp_task_t for the given task.
1872   auto *KmpTaskTWithPrivatesQTyRD =
1873       createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates);
1874   auto KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD);
1875   QualType KmpTaskTWithPrivatesPtrQTy =
1876       C.getPointerType(KmpTaskTWithPrivatesQTy);
1877   auto *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy);
1878   auto *KmpTaskTWithPrivatesPtrTy = KmpTaskTWithPrivatesTy->getPointerTo();
1879   auto KmpTaskTWithPrivatesTySize =
1880       CGM.getSize(C.getTypeSizeInChars(KmpTaskTWithPrivatesQTy));
1881   QualType SharedsPtrTy = C.getPointerType(SharedsTy);
1882 
1883   // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
1884   // kmp_task_t *tt);
1885   auto *TaskEntry = emitProxyTaskFunction(
1886       CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTy,
1887       KmpTaskTQTy, SharedsPtrTy, TaskFunction);
1888 
1889   // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
1890   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
1891   // kmp_routine_entry_t *task_entry);
1892   // Task flags. Format is taken from
1893   // http://llvm.org/svn/llvm-project/openmp/trunk/runtime/src/kmp.h,
1894   // description of kmp_tasking_flags struct.
1895   const unsigned TiedFlag = 0x1;
1896   const unsigned FinalFlag = 0x2;
1897   unsigned Flags = Tied ? TiedFlag : 0;
1898   auto *TaskFlags =
1899       Final.getPointer()
1900           ? CGF.Builder.CreateSelect(Final.getPointer(),
1901                                      CGF.Builder.getInt32(FinalFlag),
1902                                      CGF.Builder.getInt32(/*C=*/0))
1903           : CGF.Builder.getInt32(Final.getInt() ? FinalFlag : 0);
1904   TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
1905   auto SharedsSize = C.getTypeSizeInChars(SharedsTy);
1906   llvm::Value *AllocArgs[] = {
1907       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), TaskFlags,
1908       KmpTaskTWithPrivatesTySize, CGM.getSize(SharedsSize),
1909       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskEntry,
1910                                                       KmpRoutineEntryPtrTy)};
1911   auto *NewTask = CGF.EmitRuntimeCall(
1912       createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs);
1913   auto *NewTaskNewTaskTTy = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1914       NewTask, KmpTaskTWithPrivatesPtrTy);
1915   LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy,
1916                                                KmpTaskTWithPrivatesQTy);
1917   LValue TDBase =
1918       CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin());
1919   // Fill the data in the resulting kmp_task_t record.
1920   // Copy shareds if there are any.
1921   llvm::Value *KmpTaskSharedsPtr = nullptr;
1922   if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) {
1923     KmpTaskSharedsPtr = CGF.EmitLoadOfScalar(
1924         CGF.EmitLValueForField(
1925             TDBase, *std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds)),
1926         Loc);
1927     CGF.EmitAggregateCopy(KmpTaskSharedsPtr, Shareds, SharedsTy);
1928   }
1929   // Emit initial values for private copies (if any).
1930   bool NeedsCleanup = false;
1931   if (!Privates.empty()) {
1932     auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin());
1933     auto PrivatesBase = CGF.EmitLValueForField(Base, *FI);
1934     FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin();
1935     LValue SharedsBase = CGF.MakeNaturalAlignAddrLValue(
1936         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1937             KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)),
1938         SharedsTy);
1939     CodeGenFunction::CGCapturedStmtInfo CapturesInfo(
1940         cast<CapturedStmt>(*D.getAssociatedStmt()));
1941     for (auto &&Pair : Privates) {
1942       auto *VD = Pair.second.PrivateCopy;
1943       auto *Init = VD->getAnyInitializer();
1944       LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI);
1945       if (Init) {
1946         if (auto *Elem = Pair.second.PrivateElemInit) {
1947           auto *OriginalVD = Pair.second.Original;
1948           auto *SharedField = CapturesInfo.lookup(OriginalVD);
1949           auto SharedRefLValue =
1950               CGF.EmitLValueForField(SharedsBase, SharedField);
1951           QualType Type = OriginalVD->getType();
1952           if (auto *PVD = dyn_cast<ParmVarDecl>(OriginalVD)) {
1953             Type = PVD->getOriginalType();
1954           }
1955           if (Type->isArrayType()) {
1956             // Initialize firstprivate array.
1957             if (!isa<CXXConstructExpr>(Init) ||
1958                 CGF.isTrivialInitializer(Init)) {
1959               // Perform simple memcpy.
1960               CGF.EmitAggregateAssign(PrivateLValue.getAddress(),
1961                                       SharedRefLValue.getAddress(), Type);
1962             } else {
1963               // Initialize firstprivate array using element-by-element
1964               // intialization.
1965               CGF.EmitOMPAggregateAssign(
1966                   PrivateLValue.getAddress(), SharedRefLValue.getAddress(),
1967                   Type, [&CGF, Elem, Init, &CapturesInfo](
1968                             llvm::Value *DestElement, llvm::Value *SrcElement) {
1969                     // Clean up any temporaries needed by the initialization.
1970                     CodeGenFunction::OMPPrivateScope InitScope(CGF);
1971                     InitScope.addPrivate(Elem, [SrcElement]() -> llvm::Value *{
1972                       return SrcElement;
1973                     });
1974                     (void)InitScope.Privatize();
1975                     // Emit initialization for single element.
1976                     auto *OldCapturedStmtInfo = CGF.CapturedStmtInfo;
1977                     CGF.CapturedStmtInfo = &CapturesInfo;
1978                     CGF.EmitAnyExprToMem(Init, DestElement,
1979                                          Init->getType().getQualifiers(),
1980                                          /*IsInitializer=*/false);
1981                     CGF.CapturedStmtInfo = OldCapturedStmtInfo;
1982                   });
1983             }
1984           } else {
1985             CodeGenFunction::OMPPrivateScope InitScope(CGF);
1986             InitScope.addPrivate(Elem, [SharedRefLValue]() -> llvm::Value *{
1987               return SharedRefLValue.getAddress();
1988             });
1989             (void)InitScope.Privatize();
1990             auto *OldCapturedStmtInfo = CGF.CapturedStmtInfo;
1991             CGF.CapturedStmtInfo = &CapturesInfo;
1992             CGF.EmitExprAsInit(Init, VD, PrivateLValue,
1993                                /*capturedByInit=*/false);
1994             CGF.CapturedStmtInfo = OldCapturedStmtInfo;
1995           }
1996         } else {
1997           CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false);
1998         }
1999       }
2000       NeedsCleanup = NeedsCleanup || FI->getType().isDestructedType();
2001       // Copy addresses of privates to corresponding references in the list of
2002       // captured variables.
2003       //   ...
2004       //   tt->shareds.var_addr = &tt->privates.private_var;
2005       //   ...
2006       auto *OriginalVD = Pair.second.Original;
2007       auto *SharedField = CapturesInfo.lookup(OriginalVD);
2008       auto SharedRefLValue =
2009           CGF.EmitLValueForFieldInitialization(SharedsBase, SharedField);
2010       CGF.EmitStoreThroughLValue(
2011           RValue::get(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2012               PrivateLValue.getAddress(), SharedRefLValue.getAddress()
2013                                               ->getType()
2014                                               ->getPointerElementType())),
2015           SharedRefLValue);
2016       ++FI, ++I;
2017     }
2018   }
2019   // Provide pointer to function with destructors for privates.
2020   llvm::Value *DestructorFn =
2021       NeedsCleanup ? emitDestructorsFunction(CGM, Loc, KmpInt32Ty,
2022                                              KmpTaskTWithPrivatesPtrQTy,
2023                                              KmpTaskTWithPrivatesQTy)
2024                    : llvm::ConstantPointerNull::get(
2025                          cast<llvm::PointerType>(KmpRoutineEntryPtrTy));
2026   LValue Destructor = CGF.EmitLValueForField(
2027       TDBase, *std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTDestructors));
2028   CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2029                             DestructorFn, KmpRoutineEntryPtrTy),
2030                         Destructor);
2031   // NOTE: routine and part_id fields are intialized by __kmpc_omp_task_alloc()
2032   // libcall.
2033   // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
2034   // *new_task);
2035   auto *ThreadID = getThreadID(CGF, Loc);
2036   llvm::Value *TaskArgs[] = {emitUpdateLocation(CGF, Loc), ThreadID, NewTask};
2037   auto &&ThenCodeGen = [this, &TaskArgs](CodeGenFunction &CGF) {
2038     // TODO: add check for untied tasks.
2039     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs);
2040   };
2041   typedef CallEndCleanup<std::extent<decltype(TaskArgs)>::value>
2042       IfCallEndCleanup;
2043   auto &&ElseCodeGen =
2044       [this, &TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry](
2045           CodeGenFunction &CGF) {
2046         CodeGenFunction::RunCleanupsScope LocalScope(CGF);
2047         CGF.EmitRuntimeCall(
2048             createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs);
2049         // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid,
2050         // kmp_task_t *new_task);
2051         CGF.EHStack.pushCleanup<IfCallEndCleanup>(
2052             NormalAndEHCleanup,
2053             createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0),
2054             llvm::makeArrayRef(TaskArgs));
2055 
2056         // Call proxy_task_entry(gtid, new_task);
2057         llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy};
2058         CGF.EmitCallOrInvoke(TaskEntry, OutlinedFnArgs);
2059       };
2060   if (IfCond) {
2061     emitOMPIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen);
2062   } else {
2063     CodeGenFunction::RunCleanupsScope Scope(CGF);
2064     ThenCodeGen(CGF);
2065   }
2066 }
2067 
2068 static llvm::Value *emitReductionFunction(CodeGenModule &CGM,
2069                                           llvm::Type *ArgsType,
2070                                           ArrayRef<const Expr *> LHSExprs,
2071                                           ArrayRef<const Expr *> RHSExprs,
2072                                           ArrayRef<const Expr *> ReductionOps) {
2073   auto &C = CGM.getContext();
2074 
2075   // void reduction_func(void *LHSArg, void *RHSArg);
2076   FunctionArgList Args;
2077   ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, SourceLocation(), /*Id=*/nullptr,
2078                            C.VoidPtrTy);
2079   ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, SourceLocation(), /*Id=*/nullptr,
2080                            C.VoidPtrTy);
2081   Args.push_back(&LHSArg);
2082   Args.push_back(&RHSArg);
2083   FunctionType::ExtInfo EI;
2084   auto &CGFI = CGM.getTypes().arrangeFreeFunctionDeclaration(
2085       C.VoidTy, Args, EI, /*isVariadic=*/false);
2086   auto *Fn = llvm::Function::Create(
2087       CGM.getTypes().GetFunctionType(CGFI), llvm::GlobalValue::InternalLinkage,
2088       ".omp.reduction.reduction_func", &CGM.getModule());
2089   CGM.SetLLVMFunctionAttributes(/*D=*/nullptr, CGFI, Fn);
2090   CodeGenFunction CGF(CGM);
2091   CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args);
2092 
2093   // Dst = (void*[n])(LHSArg);
2094   // Src = (void*[n])(RHSArg);
2095   auto *LHS = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2096       CGF.Builder.CreateAlignedLoad(CGF.GetAddrOfLocalVar(&LHSArg),
2097                                     CGF.PointerAlignInBytes),
2098       ArgsType);
2099   auto *RHS = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2100       CGF.Builder.CreateAlignedLoad(CGF.GetAddrOfLocalVar(&RHSArg),
2101                                     CGF.PointerAlignInBytes),
2102       ArgsType);
2103 
2104   //  ...
2105   //  *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]);
2106   //  ...
2107   CodeGenFunction::OMPPrivateScope Scope(CGF);
2108   for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I) {
2109     Scope.addPrivate(
2110         cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()),
2111         [&]() -> llvm::Value *{
2112           return CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2113               CGF.Builder.CreateAlignedLoad(
2114                   CGF.Builder.CreateStructGEP(/*Ty=*/nullptr, RHS, I),
2115                   CGM.PointerAlignInBytes),
2116               CGF.ConvertTypeForMem(C.getPointerType(RHSExprs[I]->getType())));
2117         });
2118     Scope.addPrivate(
2119         cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()),
2120         [&]() -> llvm::Value *{
2121           return CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2122               CGF.Builder.CreateAlignedLoad(
2123                   CGF.Builder.CreateStructGEP(/*Ty=*/nullptr, LHS, I),
2124                   CGM.PointerAlignInBytes),
2125               CGF.ConvertTypeForMem(C.getPointerType(LHSExprs[I]->getType())));
2126         });
2127   }
2128   Scope.Privatize();
2129   for (auto *E : ReductionOps) {
2130     CGF.EmitIgnoredExpr(E);
2131   }
2132   Scope.ForceCleanup();
2133   CGF.FinishFunction();
2134   return Fn;
2135 }
2136 
2137 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc,
2138                                     ArrayRef<const Expr *> LHSExprs,
2139                                     ArrayRef<const Expr *> RHSExprs,
2140                                     ArrayRef<const Expr *> ReductionOps,
2141                                     bool WithNowait) {
2142   // Next code should be emitted for reduction:
2143   //
2144   // static kmp_critical_name lock = { 0 };
2145   //
2146   // void reduce_func(void *lhs[<n>], void *rhs[<n>]) {
2147   //  *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]);
2148   //  ...
2149   //  *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1],
2150   //  *(Type<n>-1*)rhs[<n>-1]);
2151   // }
2152   //
2153   // ...
2154   // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]};
2155   // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
2156   // RedList, reduce_func, &<lock>)) {
2157   // case 1:
2158   //  ...
2159   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
2160   //  ...
2161   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
2162   // break;
2163   // case 2:
2164   //  ...
2165   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
2166   //  ...
2167   // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);]
2168   // break;
2169   // default:;
2170   // }
2171 
2172   auto &C = CGM.getContext();
2173 
2174   // 1. Build a list of reduction variables.
2175   // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]};
2176   llvm::APInt ArraySize(/*unsigned int numBits=*/32, RHSExprs.size());
2177   QualType ReductionArrayTy =
2178       C.getConstantArrayType(C.VoidPtrTy, ArraySize, ArrayType::Normal,
2179                              /*IndexTypeQuals=*/0);
2180   auto *ReductionList =
2181       CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list");
2182   for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I) {
2183     auto *Elem = CGF.Builder.CreateStructGEP(/*Ty=*/nullptr, ReductionList, I);
2184     CGF.Builder.CreateAlignedStore(
2185         CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
2186             CGF.EmitLValue(RHSExprs[I]).getAddress(), CGF.VoidPtrTy),
2187         Elem, CGM.PointerAlignInBytes);
2188   }
2189 
2190   // 2. Emit reduce_func().
2191   auto *ReductionFn = emitReductionFunction(
2192       CGM, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), LHSExprs,
2193       RHSExprs, ReductionOps);
2194 
2195   // 3. Create static kmp_critical_name lock = { 0 };
2196   auto *Lock = getCriticalRegionLock(".reduction");
2197 
2198   // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList),
2199   // RedList, reduce_func, &<lock>);
2200   auto *IdentTLoc = emitUpdateLocation(
2201       CGF, Loc,
2202       static_cast<OpenMPLocationFlags>(OMP_IDENT_KMPC | OMP_ATOMIC_REDUCE));
2203   auto *ThreadId = getThreadID(CGF, Loc);
2204   auto *ReductionArrayTySize = llvm::ConstantInt::get(
2205       CGM.SizeTy, C.getTypeSizeInChars(ReductionArrayTy).getQuantity());
2206   auto *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(ReductionList,
2207                                                              CGF.VoidPtrTy);
2208   llvm::Value *Args[] = {
2209       IdentTLoc,                             // ident_t *<loc>
2210       ThreadId,                              // i32 <gtid>
2211       CGF.Builder.getInt32(RHSExprs.size()), // i32 <n>
2212       ReductionArrayTySize,                  // size_type sizeof(RedList)
2213       RL,                                    // void *RedList
2214       ReductionFn, // void (*) (void *, void *) <reduce_func>
2215       Lock         // kmp_critical_name *&<lock>
2216   };
2217   auto Res = CGF.EmitRuntimeCall(
2218       createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait
2219                                        : OMPRTL__kmpc_reduce),
2220       Args);
2221 
2222   // 5. Build switch(res)
2223   auto *DefaultBB = CGF.createBasicBlock(".omp.reduction.default");
2224   auto *SwInst = CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2);
2225 
2226   // 6. Build case 1:
2227   //  ...
2228   //  <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]);
2229   //  ...
2230   // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
2231   // break;
2232   auto *Case1BB = CGF.createBasicBlock(".omp.reduction.case1");
2233   SwInst->addCase(CGF.Builder.getInt32(1), Case1BB);
2234   CGF.EmitBlock(Case1BB);
2235 
2236   {
2237     CodeGenFunction::RunCleanupsScope Scope(CGF);
2238     // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>);
2239     llvm::Value *EndArgs[] = {
2240         IdentTLoc, // ident_t *<loc>
2241         ThreadId,  // i32 <gtid>
2242         Lock       // kmp_critical_name *&<lock>
2243     };
2244     CGF.EHStack
2245         .pushCleanup<CallEndCleanup<std::extent<decltype(EndArgs)>::value>>(
2246             NormalAndEHCleanup,
2247             createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait
2248                                              : OMPRTL__kmpc_end_reduce),
2249             llvm::makeArrayRef(EndArgs));
2250     for (auto *E : ReductionOps) {
2251       CGF.EmitIgnoredExpr(E);
2252     }
2253   }
2254 
2255   CGF.EmitBranch(DefaultBB);
2256 
2257   // 7. Build case 2:
2258   //  ...
2259   //  Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]));
2260   //  ...
2261   // break;
2262   auto *Case2BB = CGF.createBasicBlock(".omp.reduction.case2");
2263   SwInst->addCase(CGF.Builder.getInt32(2), Case2BB);
2264   CGF.EmitBlock(Case2BB);
2265 
2266   {
2267     CodeGenFunction::RunCleanupsScope Scope(CGF);
2268     if (!WithNowait) {
2269       // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>);
2270       llvm::Value *EndArgs[] = {
2271           IdentTLoc, // ident_t *<loc>
2272           ThreadId,  // i32 <gtid>
2273           Lock       // kmp_critical_name *&<lock>
2274       };
2275       CGF.EHStack
2276           .pushCleanup<CallEndCleanup<std::extent<decltype(EndArgs)>::value>>(
2277               NormalAndEHCleanup,
2278               createRuntimeFunction(OMPRTL__kmpc_end_reduce),
2279               llvm::makeArrayRef(EndArgs));
2280     }
2281     auto I = LHSExprs.begin();
2282     for (auto *E : ReductionOps) {
2283       const Expr *XExpr = nullptr;
2284       const Expr *EExpr = nullptr;
2285       const Expr *UpExpr = nullptr;
2286       BinaryOperatorKind BO = BO_Comma;
2287       if (auto *BO = dyn_cast<BinaryOperator>(E)) {
2288         if (BO->getOpcode() == BO_Assign) {
2289           XExpr = BO->getLHS();
2290           UpExpr = BO->getRHS();
2291         }
2292       }
2293       // Try to emit update expression as a simple atomic.
2294       auto *RHSExpr = UpExpr;
2295       if (RHSExpr) {
2296         // Analyze RHS part of the whole expression.
2297         if (auto *ACO = dyn_cast<AbstractConditionalOperator>(
2298                 RHSExpr->IgnoreParenImpCasts())) {
2299           // If this is a conditional operator, analyze its condition for
2300           // min/max reduction operator.
2301           RHSExpr = ACO->getCond();
2302         }
2303         if (auto *BORHS =
2304                 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) {
2305           EExpr = BORHS->getRHS();
2306           BO = BORHS->getOpcode();
2307         }
2308       }
2309       if (XExpr) {
2310         auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl());
2311         LValue X = CGF.EmitLValue(XExpr);
2312         RValue E;
2313         if (EExpr)
2314           E = CGF.EmitAnyExpr(EExpr);
2315         CGF.EmitOMPAtomicSimpleUpdateExpr(
2316             X, E, BO, /*IsXLHSInRHSPart=*/true, llvm::Monotonic, Loc,
2317             [&CGF, UpExpr, VD](RValue XRValue) {
2318               CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
2319               PrivateScope.addPrivate(
2320                   VD, [&CGF, VD, XRValue]() -> llvm::Value *{
2321                     auto *LHSTemp = CGF.CreateMemTemp(VD->getType());
2322                     CGF.EmitStoreThroughLValue(
2323                         XRValue,
2324                         CGF.MakeNaturalAlignAddrLValue(LHSTemp, VD->getType()));
2325                     return LHSTemp;
2326                   });
2327               (void)PrivateScope.Privatize();
2328               return CGF.EmitAnyExpr(UpExpr);
2329             });
2330       } else {
2331         // Emit as a critical region.
2332         emitCriticalRegion(CGF, ".atomic_reduction", [E](CodeGenFunction &CGF) {
2333           CGF.EmitIgnoredExpr(E);
2334         }, Loc);
2335       }
2336       ++I;
2337     }
2338   }
2339 
2340   CGF.EmitBranch(DefaultBB);
2341   CGF.EmitBlock(DefaultBB, /*IsFinished=*/true);
2342 }
2343 
2344 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF,
2345                                        SourceLocation Loc) {
2346   // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32
2347   // global_tid);
2348   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
2349   // Ignore return result until untied tasks are supported.
2350   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args);
2351 }
2352 
2353 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF,
2354                                            const RegionCodeGenTy &CodeGen) {
2355   InlinedOpenMPRegionRAII Region(CGF, CodeGen);
2356   CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr);
2357 }
2358 
2359