1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 // This provides a class for OpenMP runtime code generation.
11 //
12 //===----------------------------------------------------------------------===//
13 
14 #include "CGOpenMPRuntime.h"
15 #include "CodeGenFunction.h"
16 #include "CGCleanup.h"
17 #include "clang/AST/Decl.h"
18 #include "clang/AST/StmtOpenMP.h"
19 #include "llvm/ADT/ArrayRef.h"
20 #include "llvm/IR/CallSite.h"
21 #include "llvm/IR/DerivedTypes.h"
22 #include "llvm/IR/GlobalValue.h"
23 #include "llvm/IR/Value.h"
24 #include "llvm/Support/raw_ostream.h"
25 #include <cassert>
26 
27 using namespace clang;
28 using namespace CodeGen;
29 
30 namespace {
31 /// \brief Base class for handling code generation inside OpenMP regions.
32 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo {
33 public:
34   CGOpenMPRegionInfo(const OMPExecutableDirective &D, const CapturedStmt &CS)
35       : CGCapturedStmtInfo(CS, CR_OpenMP), Directive(D) {}
36 
37   CGOpenMPRegionInfo(const OMPExecutableDirective &D)
38       : CGCapturedStmtInfo(CR_OpenMP), Directive(D) {}
39 
40   /// \brief Get a variable or parameter for storing global thread id
41   /// inside OpenMP construct.
42   virtual const VarDecl *getThreadIDVariable() const = 0;
43 
44   /// \brief Get an LValue for the current ThreadID variable.
45   /// \return LValue for thread id variable. This LValue always has type int32*.
46   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF);
47 
48     /// \brief Emit the captured statement body.
49   virtual void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
50 
51   static bool classof(const CGCapturedStmtInfo *Info) {
52     return Info->getKind() == CR_OpenMP;
53   }
54 protected:
55   /// \brief OpenMP executable directive associated with the region.
56   const OMPExecutableDirective &Directive;
57 };
58 
59 /// \brief API for captured statement code generation in OpenMP constructs.
60 class CGOpenMPOutlinedRegionInfo : public CGOpenMPRegionInfo {
61 public:
62   CGOpenMPOutlinedRegionInfo(const OMPExecutableDirective &D,
63                              const CapturedStmt &CS, const VarDecl *ThreadIDVar)
64       : CGOpenMPRegionInfo(D, CS), ThreadIDVar(ThreadIDVar) {
65     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
66   }
67   /// \brief Get a variable or parameter for storing global thread id
68   /// inside OpenMP construct.
69   virtual const VarDecl *getThreadIDVariable() const override {
70     return ThreadIDVar;
71   }
72   /// \brief Get the name of the capture helper.
73   StringRef getHelperName() const override { return ".omp_outlined."; }
74 
75 private:
76   /// \brief A variable or parameter storing global thread id for OpenMP
77   /// constructs.
78   const VarDecl *ThreadIDVar;
79 };
80 
81 /// \brief API for captured statement code generation in OpenMP constructs.
82 class CGOpenMPTaskOutlinedRegionInfo : public CGOpenMPRegionInfo {
83 public:
84   CGOpenMPTaskOutlinedRegionInfo(const OMPExecutableDirective &D,
85                                  const CapturedStmt &CS,
86                                  const VarDecl *ThreadIDVar,
87                                  const VarDecl *PartIDVar)
88       : CGOpenMPRegionInfo(D, CS), ThreadIDVar(ThreadIDVar),
89         PartIDVar(PartIDVar) {
90     assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region.");
91   }
92   /// \brief Get a variable or parameter for storing global thread id
93   /// inside OpenMP construct.
94   virtual const VarDecl *getThreadIDVariable() const override {
95     return ThreadIDVar;
96   }
97 
98   /// \brief Get an LValue for the current ThreadID variable.
99   virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override;
100 
101   /// \brief Emit the captured statement body.
102   virtual void EmitBody(CodeGenFunction &CGF, const Stmt *S) override;
103 
104   /// \brief Get the name of the capture helper.
105   StringRef getHelperName() const override { return ".omp_outlined."; }
106 
107 private:
108   /// \brief A variable or parameter storing global thread id for OpenMP
109   /// constructs.
110   const VarDecl *ThreadIDVar;
111   /// \brief A variable or parameter storing part id for OpenMP tasking
112   /// constructs.
113   const VarDecl *PartIDVar;
114 };
115 
116 /// \brief API for inlined captured statement code generation in OpenMP
117 /// constructs.
118 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo {
119 public:
120   CGOpenMPInlinedRegionInfo(const OMPExecutableDirective &D,
121                             CodeGenFunction::CGCapturedStmtInfo *OldCSI)
122       : CGOpenMPRegionInfo(D), OldCSI(OldCSI),
123         OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {}
124   // \brief Retrieve the value of the context parameter.
125   virtual llvm::Value *getContextValue() const override {
126     if (OuterRegionInfo)
127       return OuterRegionInfo->getContextValue();
128     llvm_unreachable("No context value for inlined OpenMP region");
129   }
130   /// \brief Lookup the captured field decl for a variable.
131   virtual const FieldDecl *lookup(const VarDecl *VD) const override {
132     if (OuterRegionInfo)
133       return OuterRegionInfo->lookup(VD);
134     llvm_unreachable("Trying to reference VarDecl that is neither local nor "
135                      "captured in outer OpenMP region");
136   }
137   virtual FieldDecl *getThisFieldDecl() const override {
138     if (OuterRegionInfo)
139       return OuterRegionInfo->getThisFieldDecl();
140     return nullptr;
141   }
142   /// \brief Get a variable or parameter for storing global thread id
143   /// inside OpenMP construct.
144   virtual const VarDecl *getThreadIDVariable() const override {
145     if (OuterRegionInfo)
146       return OuterRegionInfo->getThreadIDVariable();
147     return nullptr;
148   }
149 
150   /// \brief Get the name of the capture helper.
151   virtual StringRef getHelperName() const override {
152     llvm_unreachable("No helper name for inlined OpenMP construct");
153   }
154 
155   CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; }
156 
157 private:
158   /// \brief CodeGen info about outer OpenMP region.
159   CodeGenFunction::CGCapturedStmtInfo *OldCSI;
160   CGOpenMPRegionInfo *OuterRegionInfo;
161 };
162 } // namespace
163 
164 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) {
165   return CGF.MakeNaturalAlignAddrLValue(
166       CGF.Builder.CreateAlignedLoad(
167           CGF.GetAddrOfLocalVar(getThreadIDVariable()),
168           CGF.PointerAlignInBytes),
169       getThreadIDVariable()
170           ->getType()
171           ->castAs<PointerType>()
172           ->getPointeeType());
173 }
174 
175 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt *S) {
176   CodeGenFunction::OMPPrivateScope PrivateScope(CGF);
177   CGF.EmitOMPPrivateClause(Directive, PrivateScope);
178   CGF.EmitOMPFirstprivateClause(Directive, PrivateScope);
179   if (PrivateScope.Privatize())
180     // Emit implicit barrier to synchronize threads and avoid data races.
181     CGF.CGM.getOpenMPRuntime().emitBarrierCall(CGF, Directive.getLocStart(),
182                                                /*IsExplicit=*/false);
183   CGCapturedStmtInfo::EmitBody(CGF, S);
184 }
185 
186 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue(
187     CodeGenFunction &CGF) {
188   return CGF.MakeNaturalAlignAddrLValue(
189       CGF.GetAddrOfLocalVar(getThreadIDVariable()),
190       getThreadIDVariable()->getType());
191 }
192 
193 void CGOpenMPTaskOutlinedRegionInfo::EmitBody(CodeGenFunction &CGF,
194                                               const Stmt *S) {
195   if (PartIDVar) {
196     // TODO: emit code for untied tasks.
197   }
198   CGCapturedStmtInfo::EmitBody(CGF, S);
199 }
200 
201 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM)
202     : CGM(CGM), DefaultOpenMPPSource(nullptr), KmpRoutineEntryPtrTy(nullptr) {
203   IdentTy = llvm::StructType::create(
204       "ident_t", CGM.Int32Ty /* reserved_1 */, CGM.Int32Ty /* flags */,
205       CGM.Int32Ty /* reserved_2 */, CGM.Int32Ty /* reserved_3 */,
206       CGM.Int8PtrTy /* psource */, nullptr);
207   // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...)
208   llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty),
209                                llvm::PointerType::getUnqual(CGM.Int32Ty)};
210   Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true);
211   KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8);
212 }
213 
214 void CGOpenMPRuntime::clear() {
215   InternalVars.clear();
216 }
217 
218 llvm::Value *
219 CGOpenMPRuntime::emitOutlinedFunction(const OMPExecutableDirective &D,
220                                       const VarDecl *ThreadIDVar) {
221   assert(ThreadIDVar->getType()->isPointerType() &&
222          "thread id variable must be of type kmp_int32 *");
223   const CapturedStmt *CS = cast<CapturedStmt>(D.getAssociatedStmt());
224   CodeGenFunction CGF(CGM, true);
225   CGOpenMPOutlinedRegionInfo CGInfo(D, *CS, ThreadIDVar);
226   CGF.CapturedStmtInfo = &CGInfo;
227   return CGF.GenerateCapturedStmtFunction(*CS);
228 }
229 
230 llvm::Value *
231 CGOpenMPRuntime::emitTaskOutlinedFunction(const OMPExecutableDirective &D,
232                                           const VarDecl *ThreadIDVar,
233                                           const VarDecl *PartIDVar) {
234   assert(!ThreadIDVar->getType()->isPointerType() &&
235          "thread id variable must be of type kmp_int32 for tasks");
236   auto *CS = cast<CapturedStmt>(D.getAssociatedStmt());
237   CodeGenFunction CGF(CGM, true);
238   CGOpenMPTaskOutlinedRegionInfo CGInfo(D, *CS, ThreadIDVar, PartIDVar);
239   CGF.CapturedStmtInfo = &CGInfo;
240   return CGF.GenerateCapturedStmtFunction(*CS);
241 }
242 
243 llvm::Value *
244 CGOpenMPRuntime::getOrCreateDefaultLocation(OpenMPLocationFlags Flags) {
245   llvm::Value *Entry = OpenMPDefaultLocMap.lookup(Flags);
246   if (!Entry) {
247     if (!DefaultOpenMPPSource) {
248       // Initialize default location for psource field of ident_t structure of
249       // all ident_t objects. Format is ";file;function;line;column;;".
250       // Taken from
251       // http://llvm.org/svn/llvm-project/openmp/trunk/runtime/src/kmp_str.c
252       DefaultOpenMPPSource =
253           CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;");
254       DefaultOpenMPPSource =
255           llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy);
256     }
257     auto DefaultOpenMPLocation = new llvm::GlobalVariable(
258         CGM.getModule(), IdentTy, /*isConstant*/ true,
259         llvm::GlobalValue::PrivateLinkage, /*Initializer*/ nullptr);
260     DefaultOpenMPLocation->setUnnamedAddr(true);
261 
262     llvm::Constant *Zero = llvm::ConstantInt::get(CGM.Int32Ty, 0, true);
263     llvm::Constant *Values[] = {Zero,
264                                 llvm::ConstantInt::get(CGM.Int32Ty, Flags),
265                                 Zero, Zero, DefaultOpenMPPSource};
266     llvm::Constant *Init = llvm::ConstantStruct::get(IdentTy, Values);
267     DefaultOpenMPLocation->setInitializer(Init);
268     OpenMPDefaultLocMap[Flags] = DefaultOpenMPLocation;
269     return DefaultOpenMPLocation;
270   }
271   return Entry;
272 }
273 
274 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF,
275                                                  SourceLocation Loc,
276                                                  OpenMPLocationFlags Flags) {
277   // If no debug info is generated - return global default location.
278   if (CGM.getCodeGenOpts().getDebugInfo() == CodeGenOptions::NoDebugInfo ||
279       Loc.isInvalid())
280     return getOrCreateDefaultLocation(Flags);
281 
282   assert(CGF.CurFn && "No function in current CodeGenFunction.");
283 
284   llvm::Value *LocValue = nullptr;
285   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
286   if (I != OpenMPLocThreadIDMap.end())
287     LocValue = I->second.DebugLoc;
288   // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if
289   // GetOpenMPThreadID was called before this routine.
290   if (LocValue == nullptr) {
291     // Generate "ident_t .kmpc_loc.addr;"
292     llvm::AllocaInst *AI = CGF.CreateTempAlloca(IdentTy, ".kmpc_loc.addr");
293     AI->setAlignment(CGM.getDataLayout().getPrefTypeAlignment(IdentTy));
294     auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
295     Elem.second.DebugLoc = AI;
296     LocValue = AI;
297 
298     CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
299     CGF.Builder.SetInsertPoint(CGF.AllocaInsertPt);
300     CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags),
301                              llvm::ConstantExpr::getSizeOf(IdentTy),
302                              CGM.PointerAlignInBytes);
303   }
304 
305   // char **psource = &.kmpc_loc_<flags>.addr.psource;
306   auto *PSource =
307       CGF.Builder.CreateConstInBoundsGEP2_32(LocValue, 0, IdentField_PSource);
308 
309   auto OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding());
310   if (OMPDebugLoc == nullptr) {
311     SmallString<128> Buffer2;
312     llvm::raw_svector_ostream OS2(Buffer2);
313     // Build debug location
314     PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc);
315     OS2 << ";" << PLoc.getFilename() << ";";
316     if (const FunctionDecl *FD =
317             dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) {
318       OS2 << FD->getQualifiedNameAsString();
319     }
320     OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;";
321     OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str());
322     OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc;
323   }
324   // *psource = ";<File>;<Function>;<Line>;<Column>;;";
325   CGF.Builder.CreateStore(OMPDebugLoc, PSource);
326 
327   return LocValue;
328 }
329 
330 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF,
331                                           SourceLocation Loc) {
332   assert(CGF.CurFn && "No function in current CodeGenFunction.");
333 
334   llvm::Value *ThreadID = nullptr;
335   // Check whether we've already cached a load of the thread id in this
336   // function.
337   auto I = OpenMPLocThreadIDMap.find(CGF.CurFn);
338   if (I != OpenMPLocThreadIDMap.end()) {
339     ThreadID = I->second.ThreadID;
340     if (ThreadID != nullptr)
341       return ThreadID;
342   }
343   if (auto OMPRegionInfo =
344           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) {
345     if (OMPRegionInfo->getThreadIDVariable()) {
346       // Check if this an outlined function with thread id passed as argument.
347       auto LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF);
348       ThreadID = CGF.EmitLoadOfLValue(LVal, Loc).getScalarVal();
349       // If value loaded in entry block, cache it and use it everywhere in
350       // function.
351       if (CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) {
352         auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
353         Elem.second.ThreadID = ThreadID;
354       }
355       return ThreadID;
356     }
357   }
358 
359   // This is not an outlined function region - need to call __kmpc_int32
360   // kmpc_global_thread_num(ident_t *loc).
361   // Generate thread id value and cache this value for use across the
362   // function.
363   CGBuilderTy::InsertPointGuard IPG(CGF.Builder);
364   CGF.Builder.SetInsertPoint(CGF.AllocaInsertPt);
365   ThreadID =
366       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
367                           emitUpdateLocation(CGF, Loc));
368   auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn);
369   Elem.second.ThreadID = ThreadID;
370   return ThreadID;
371 }
372 
373 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) {
374   assert(CGF.CurFn && "No function in current CodeGenFunction.");
375   if (OpenMPLocThreadIDMap.count(CGF.CurFn))
376     OpenMPLocThreadIDMap.erase(CGF.CurFn);
377 }
378 
379 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() {
380   return llvm::PointerType::getUnqual(IdentTy);
381 }
382 
383 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() {
384   return llvm::PointerType::getUnqual(Kmpc_MicroTy);
385 }
386 
387 llvm::Constant *
388 CGOpenMPRuntime::createRuntimeFunction(OpenMPRTLFunction Function) {
389   llvm::Constant *RTLFn = nullptr;
390   switch (Function) {
391   case OMPRTL__kmpc_fork_call: {
392     // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro
393     // microtask, ...);
394     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
395                                 getKmpc_MicroPointerTy()};
396     llvm::FunctionType *FnTy =
397         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true);
398     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call");
399     break;
400   }
401   case OMPRTL__kmpc_global_thread_num: {
402     // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc);
403     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
404     llvm::FunctionType *FnTy =
405         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
406     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num");
407     break;
408   }
409   case OMPRTL__kmpc_threadprivate_cached: {
410     // Build void *__kmpc_threadprivate_cached(ident_t *loc,
411     // kmp_int32 global_tid, void *data, size_t size, void ***cache);
412     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
413                                 CGM.VoidPtrTy, CGM.SizeTy,
414                                 CGM.VoidPtrTy->getPointerTo()->getPointerTo()};
415     llvm::FunctionType *FnTy =
416         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false);
417     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached");
418     break;
419   }
420   case OMPRTL__kmpc_critical: {
421     // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid,
422     // kmp_critical_name *crit);
423     llvm::Type *TypeParams[] = {
424         getIdentTyPointerTy(), CGM.Int32Ty,
425         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
426     llvm::FunctionType *FnTy =
427         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
428     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical");
429     break;
430   }
431   case OMPRTL__kmpc_threadprivate_register: {
432     // Build void __kmpc_threadprivate_register(ident_t *, void *data,
433     // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor);
434     // typedef void *(*kmpc_ctor)(void *);
435     auto KmpcCtorTy =
436         llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
437                                 /*isVarArg*/ false)->getPointerTo();
438     // typedef void *(*kmpc_cctor)(void *, void *);
439     llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
440     auto KmpcCopyCtorTy =
441         llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs,
442                                 /*isVarArg*/ false)->getPointerTo();
443     // typedef void (*kmpc_dtor)(void *);
444     auto KmpcDtorTy =
445         llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false)
446             ->getPointerTo();
447     llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy,
448                               KmpcCopyCtorTy, KmpcDtorTy};
449     auto FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs,
450                                         /*isVarArg*/ false);
451     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register");
452     break;
453   }
454   case OMPRTL__kmpc_end_critical: {
455     // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid,
456     // kmp_critical_name *crit);
457     llvm::Type *TypeParams[] = {
458         getIdentTyPointerTy(), CGM.Int32Ty,
459         llvm::PointerType::getUnqual(KmpCriticalNameTy)};
460     llvm::FunctionType *FnTy =
461         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
462     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical");
463     break;
464   }
465   case OMPRTL__kmpc_cancel_barrier: {
466     // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32
467     // global_tid);
468     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
469     llvm::FunctionType *FnTy =
470         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
471     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier");
472     break;
473   }
474   case OMPRTL__kmpc_for_static_fini: {
475     // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid);
476     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
477     llvm::FunctionType *FnTy =
478         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
479     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini");
480     break;
481   }
482   case OMPRTL__kmpc_push_num_threads: {
483     // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid,
484     // kmp_int32 num_threads)
485     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
486                                 CGM.Int32Ty};
487     llvm::FunctionType *FnTy =
488         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
489     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads");
490     break;
491   }
492   case OMPRTL__kmpc_serialized_parallel: {
493     // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32
494     // global_tid);
495     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
496     llvm::FunctionType *FnTy =
497         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
498     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel");
499     break;
500   }
501   case OMPRTL__kmpc_end_serialized_parallel: {
502     // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32
503     // global_tid);
504     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
505     llvm::FunctionType *FnTy =
506         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
507     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel");
508     break;
509   }
510   case OMPRTL__kmpc_flush: {
511     // Build void __kmpc_flush(ident_t *loc);
512     llvm::Type *TypeParams[] = {getIdentTyPointerTy()};
513     llvm::FunctionType *FnTy =
514         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
515     RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush");
516     break;
517   }
518   case OMPRTL__kmpc_master: {
519     // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid);
520     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
521     llvm::FunctionType *FnTy =
522         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
523     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master");
524     break;
525   }
526   case OMPRTL__kmpc_end_master: {
527     // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid);
528     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
529     llvm::FunctionType *FnTy =
530         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
531     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master");
532     break;
533   }
534   case OMPRTL__kmpc_omp_taskyield: {
535     // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid,
536     // int end_part);
537     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy};
538     llvm::FunctionType *FnTy =
539         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
540     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield");
541     break;
542   }
543   case OMPRTL__kmpc_single: {
544     // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid);
545     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
546     llvm::FunctionType *FnTy =
547         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
548     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single");
549     break;
550   }
551   case OMPRTL__kmpc_end_single: {
552     // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid);
553     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty};
554     llvm::FunctionType *FnTy =
555         llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false);
556     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single");
557     break;
558   }
559   case OMPRTL__kmpc_omp_task_alloc: {
560     // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
561     // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
562     // kmp_routine_entry_t *task_entry);
563     assert(KmpRoutineEntryPtrTy != nullptr &&
564            "Type kmp_routine_entry_t must be created.");
565     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty,
566                                 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy};
567     // Return void * and then cast to particular kmp_task_t type.
568     llvm::FunctionType *FnTy =
569         llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false);
570     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc");
571     break;
572   }
573   case OMPRTL__kmpc_omp_task: {
574     // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
575     // *new_task);
576     llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty,
577                                 CGM.VoidPtrTy};
578     llvm::FunctionType *FnTy =
579         llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false);
580     RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task");
581     break;
582   }
583   }
584   return RTLFn;
585 }
586 
587 llvm::Constant *CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize,
588                                                              bool IVSigned) {
589   assert((IVSize == 32 || IVSize == 64) &&
590          "IV size is not compatible with the omp runtime");
591   auto Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4"
592                                        : "__kmpc_for_static_init_4u")
593                            : (IVSigned ? "__kmpc_for_static_init_8"
594                                        : "__kmpc_for_static_init_8u");
595   auto ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
596   auto PtrTy = llvm::PointerType::getUnqual(ITy);
597   llvm::Type *TypeParams[] = {
598     getIdentTyPointerTy(),                     // loc
599     CGM.Int32Ty,                               // tid
600     CGM.Int32Ty,                               // schedtype
601     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
602     PtrTy,                                     // p_lower
603     PtrTy,                                     // p_upper
604     PtrTy,                                     // p_stride
605     ITy,                                       // incr
606     ITy                                        // chunk
607   };
608   llvm::FunctionType *FnTy =
609       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
610   return CGM.CreateRuntimeFunction(FnTy, Name);
611 }
612 
613 llvm::Constant *CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize,
614                                                             bool IVSigned) {
615   assert((IVSize == 32 || IVSize == 64) &&
616          "IV size is not compatible with the omp runtime");
617   auto Name =
618       IVSize == 32
619           ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u")
620           : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u");
621   auto ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
622   llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc
623                                CGM.Int32Ty,           // tid
624                                CGM.Int32Ty,           // schedtype
625                                ITy,                   // lower
626                                ITy,                   // upper
627                                ITy,                   // stride
628                                ITy                    // chunk
629   };
630   llvm::FunctionType *FnTy =
631       llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false);
632   return CGM.CreateRuntimeFunction(FnTy, Name);
633 }
634 
635 llvm::Constant *CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize,
636                                                             bool IVSigned) {
637   assert((IVSize == 32 || IVSize == 64) &&
638          "IV size is not compatible with the omp runtime");
639   auto Name =
640       IVSize == 32
641           ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u")
642           : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u");
643   auto ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty;
644   auto PtrTy = llvm::PointerType::getUnqual(ITy);
645   llvm::Type *TypeParams[] = {
646     getIdentTyPointerTy(),                     // loc
647     CGM.Int32Ty,                               // tid
648     llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter
649     PtrTy,                                     // p_lower
650     PtrTy,                                     // p_upper
651     PtrTy                                      // p_stride
652   };
653   llvm::FunctionType *FnTy =
654       llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false);
655   return CGM.CreateRuntimeFunction(FnTy, Name);
656 }
657 
658 llvm::Constant *
659 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) {
660   // Lookup the entry, lazily creating it if necessary.
661   return getOrCreateInternalVariable(CGM.Int8PtrPtrTy,
662                                      Twine(CGM.getMangledName(VD)) + ".cache.");
663 }
664 
665 llvm::Value *CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF,
666                                                      const VarDecl *VD,
667                                                      llvm::Value *VDAddr,
668                                                      SourceLocation Loc) {
669   auto VarTy = VDAddr->getType()->getPointerElementType();
670   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
671                          CGF.Builder.CreatePointerCast(VDAddr, CGM.Int8PtrTy),
672                          CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)),
673                          getOrCreateThreadPrivateCache(VD)};
674   return CGF.EmitRuntimeCall(
675       createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args);
676 }
677 
678 void CGOpenMPRuntime::emitThreadPrivateVarInit(
679     CodeGenFunction &CGF, llvm::Value *VDAddr, llvm::Value *Ctor,
680     llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) {
681   // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime
682   // library.
683   auto OMPLoc = emitUpdateLocation(CGF, Loc);
684   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num),
685                       OMPLoc);
686   // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor)
687   // to register constructor/destructor for variable.
688   llvm::Value *Args[] = {OMPLoc,
689                          CGF.Builder.CreatePointerCast(VDAddr, CGM.VoidPtrTy),
690                          Ctor, CopyCtor, Dtor};
691   CGF.EmitRuntimeCall(
692       createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args);
693 }
694 
695 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition(
696     const VarDecl *VD, llvm::Value *VDAddr, SourceLocation Loc,
697     bool PerformInit, CodeGenFunction *CGF) {
698   VD = VD->getDefinition(CGM.getContext());
699   if (VD && ThreadPrivateWithDefinition.count(VD) == 0) {
700     ThreadPrivateWithDefinition.insert(VD);
701     QualType ASTTy = VD->getType();
702 
703     llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr;
704     auto Init = VD->getAnyInitializer();
705     if (CGM.getLangOpts().CPlusPlus && PerformInit) {
706       // Generate function that re-emits the declaration's initializer into the
707       // threadprivate copy of the variable VD
708       CodeGenFunction CtorCGF(CGM);
709       FunctionArgList Args;
710       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, SourceLocation(),
711                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy);
712       Args.push_back(&Dst);
713 
714       auto &FI = CGM.getTypes().arrangeFreeFunctionDeclaration(
715           CGM.getContext().VoidPtrTy, Args, FunctionType::ExtInfo(),
716           /*isVariadic=*/false);
717       auto FTy = CGM.getTypes().GetFunctionType(FI);
718       auto Fn = CGM.CreateGlobalInitOrDestructFunction(
719           FTy, ".__kmpc_global_ctor_.", Loc);
720       CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI,
721                             Args, SourceLocation());
722       auto ArgVal = CtorCGF.EmitLoadOfScalar(
723           CtorCGF.GetAddrOfLocalVar(&Dst),
724           /*Volatile=*/false, CGM.PointerAlignInBytes,
725           CGM.getContext().VoidPtrTy, Dst.getLocation());
726       auto Arg = CtorCGF.Builder.CreatePointerCast(
727           ArgVal,
728           CtorCGF.ConvertTypeForMem(CGM.getContext().getPointerType(ASTTy)));
729       CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(),
730                                /*IsInitializer=*/true);
731       ArgVal = CtorCGF.EmitLoadOfScalar(
732           CtorCGF.GetAddrOfLocalVar(&Dst),
733           /*Volatile=*/false, CGM.PointerAlignInBytes,
734           CGM.getContext().VoidPtrTy, Dst.getLocation());
735       CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue);
736       CtorCGF.FinishFunction();
737       Ctor = Fn;
738     }
739     if (VD->getType().isDestructedType() != QualType::DK_none) {
740       // Generate function that emits destructor call for the threadprivate copy
741       // of the variable VD
742       CodeGenFunction DtorCGF(CGM);
743       FunctionArgList Args;
744       ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, SourceLocation(),
745                             /*Id=*/nullptr, CGM.getContext().VoidPtrTy);
746       Args.push_back(&Dst);
747 
748       auto &FI = CGM.getTypes().arrangeFreeFunctionDeclaration(
749           CGM.getContext().VoidTy, Args, FunctionType::ExtInfo(),
750           /*isVariadic=*/false);
751       auto FTy = CGM.getTypes().GetFunctionType(FI);
752       auto Fn = CGM.CreateGlobalInitOrDestructFunction(
753           FTy, ".__kmpc_global_dtor_.", Loc);
754       DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args,
755                             SourceLocation());
756       auto ArgVal = DtorCGF.EmitLoadOfScalar(
757           DtorCGF.GetAddrOfLocalVar(&Dst),
758           /*Volatile=*/false, CGM.PointerAlignInBytes,
759           CGM.getContext().VoidPtrTy, Dst.getLocation());
760       DtorCGF.emitDestroy(ArgVal, ASTTy,
761                           DtorCGF.getDestroyer(ASTTy.isDestructedType()),
762                           DtorCGF.needsEHCleanup(ASTTy.isDestructedType()));
763       DtorCGF.FinishFunction();
764       Dtor = Fn;
765     }
766     // Do not emit init function if it is not required.
767     if (!Ctor && !Dtor)
768       return nullptr;
769 
770     llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy};
771     auto CopyCtorTy =
772         llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs,
773                                 /*isVarArg=*/false)->getPointerTo();
774     // Copying constructor for the threadprivate variable.
775     // Must be NULL - reserved by runtime, but currently it requires that this
776     // parameter is always NULL. Otherwise it fires assertion.
777     CopyCtor = llvm::Constant::getNullValue(CopyCtorTy);
778     if (Ctor == nullptr) {
779       auto CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy,
780                                             /*isVarArg=*/false)->getPointerTo();
781       Ctor = llvm::Constant::getNullValue(CtorTy);
782     }
783     if (Dtor == nullptr) {
784       auto DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy,
785                                             /*isVarArg=*/false)->getPointerTo();
786       Dtor = llvm::Constant::getNullValue(DtorTy);
787     }
788     if (!CGF) {
789       auto InitFunctionTy =
790           llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false);
791       auto InitFunction = CGM.CreateGlobalInitOrDestructFunction(
792           InitFunctionTy, ".__omp_threadprivate_init_.");
793       CodeGenFunction InitCGF(CGM);
794       FunctionArgList ArgList;
795       InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction,
796                             CGM.getTypes().arrangeNullaryFunction(), ArgList,
797                             Loc);
798       emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
799       InitCGF.FinishFunction();
800       return InitFunction;
801     }
802     emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc);
803   }
804   return nullptr;
805 }
806 
807 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc,
808                                        llvm::Value *OutlinedFn,
809                                        llvm::Value *CapturedStruct) {
810   // Build call __kmpc_fork_call(loc, 1, microtask, captured_struct/*context*/)
811   llvm::Value *Args[] = {
812       emitUpdateLocation(CGF, Loc),
813       CGF.Builder.getInt32(1), // Number of arguments after 'microtask' argument
814       // (there is only one additional argument - 'context')
815       CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy()),
816       CGF.EmitCastToVoidPtr(CapturedStruct)};
817   auto RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_call);
818   CGF.EmitRuntimeCall(RTLFn, Args);
819 }
820 
821 void CGOpenMPRuntime::emitSerialCall(CodeGenFunction &CGF, SourceLocation Loc,
822                                      llvm::Value *OutlinedFn,
823                                      llvm::Value *CapturedStruct) {
824   auto ThreadID = getThreadID(CGF, Loc);
825   // Build calls:
826   // __kmpc_serialized_parallel(&Loc, GTid);
827   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), ThreadID};
828   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_serialized_parallel),
829                       Args);
830 
831   // OutlinedFn(&GTid, &zero, CapturedStruct);
832   auto ThreadIDAddr = emitThreadIDAddress(CGF, Loc);
833   auto Int32Ty =
834       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
835   auto ZeroAddr = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".zero.addr");
836   CGF.InitTempAlloca(ZeroAddr, CGF.Builder.getInt32(/*C*/ 0));
837   llvm::Value *OutlinedFnArgs[] = {ThreadIDAddr, ZeroAddr, CapturedStruct};
838   CGF.EmitCallOrInvoke(OutlinedFn, OutlinedFnArgs);
839 
840   // __kmpc_end_serialized_parallel(&Loc, GTid);
841   llvm::Value *EndArgs[] = {emitUpdateLocation(CGF, Loc), ThreadID};
842   CGF.EmitRuntimeCall(
843       createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel), EndArgs);
844 }
845 
846 // If we're inside an (outlined) parallel region, use the region info's
847 // thread-ID variable (it is passed in a first argument of the outlined function
848 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in
849 // regular serial code region, get thread ID by calling kmp_int32
850 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and
851 // return the address of that temp.
852 llvm::Value *CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF,
853                                                   SourceLocation Loc) {
854   if (auto OMPRegionInfo =
855           dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo))
856     if (OMPRegionInfo->getThreadIDVariable())
857       return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress();
858 
859   auto ThreadID = getThreadID(CGF, Loc);
860   auto Int32Ty =
861       CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true);
862   auto ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp.");
863   CGF.EmitStoreOfScalar(ThreadID,
864                         CGF.MakeNaturalAlignAddrLValue(ThreadIDTemp, Int32Ty));
865 
866   return ThreadIDTemp;
867 }
868 
869 llvm::Constant *
870 CGOpenMPRuntime::getOrCreateInternalVariable(llvm::Type *Ty,
871                                              const llvm::Twine &Name) {
872   SmallString<256> Buffer;
873   llvm::raw_svector_ostream Out(Buffer);
874   Out << Name;
875   auto RuntimeName = Out.str();
876   auto &Elem = *InternalVars.insert(std::make_pair(RuntimeName, nullptr)).first;
877   if (Elem.second) {
878     assert(Elem.second->getType()->getPointerElementType() == Ty &&
879            "OMP internal variable has different type than requested");
880     return &*Elem.second;
881   }
882 
883   return Elem.second = new llvm::GlobalVariable(
884              CGM.getModule(), Ty, /*IsConstant*/ false,
885              llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty),
886              Elem.first());
887 }
888 
889 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) {
890   llvm::Twine Name(".gomp_critical_user_", CriticalName);
891   return getOrCreateInternalVariable(KmpCriticalNameTy, Name.concat(".var"));
892 }
893 
894 void CGOpenMPRuntime::emitCriticalRegion(
895     CodeGenFunction &CGF, StringRef CriticalName,
896     const std::function<void()> &CriticalOpGen, SourceLocation Loc) {
897   auto RegionLock = getCriticalRegionLock(CriticalName);
898   // __kmpc_critical(ident_t *, gtid, Lock);
899   // CriticalOpGen();
900   // __kmpc_end_critical(ident_t *, gtid, Lock);
901   // Prepare arguments and build a call to __kmpc_critical
902   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
903                          RegionLock};
904   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_critical), Args);
905   CriticalOpGen();
906   // Build a call to __kmpc_end_critical
907   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_end_critical), Args);
908 }
909 
910 static void emitIfStmt(CodeGenFunction &CGF, llvm::Value *IfCond,
911                        const std::function<void()> &BodyOpGen) {
912   llvm::Value *CallBool = CGF.EmitScalarConversion(
913       IfCond,
914       CGF.getContext().getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true),
915       CGF.getContext().BoolTy);
916 
917   auto *ThenBlock = CGF.createBasicBlock("omp_if.then");
918   auto *ContBlock = CGF.createBasicBlock("omp_if.end");
919   // Generate the branch (If-stmt)
920   CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock);
921   CGF.EmitBlock(ThenBlock);
922   BodyOpGen();
923   // Emit the rest of bblocks/branches
924   CGF.EmitBranch(ContBlock);
925   CGF.EmitBlock(ContBlock, true);
926 }
927 
928 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF,
929                                        const std::function<void()> &MasterOpGen,
930                                        SourceLocation Loc) {
931   // if(__kmpc_master(ident_t *, gtid)) {
932   //   MasterOpGen();
933   //   __kmpc_end_master(ident_t *, gtid);
934   // }
935   // Prepare arguments and build a call to __kmpc_master
936   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
937   auto *IsMaster =
938       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_master), Args);
939   emitIfStmt(CGF, IsMaster, [&]() -> void {
940     MasterOpGen();
941     // Build a call to __kmpc_end_master.
942     // OpenMP [1.2.2 OpenMP Language Terminology]
943     // For C/C++, an executable statement, possibly compound, with a single
944     // entry at the top and a single exit at the bottom, or an OpenMP construct.
945     // * Access to the structured block must not be the result of a branch.
946     // * The point of exit cannot be a branch out of the structured block.
947     // * The point of entry must not be a call to setjmp().
948     // * longjmp() and throw() must not violate the entry/exit criteria.
949     // * An expression statement, iteration statement, selection statement, or
950     // try block is considered to be a structured block if the corresponding
951     // compound statement obtained by enclosing it in { and } would be a
952     // structured block.
953     // It is analyzed in Sema, so we can just call __kmpc_end_master() on
954     // fallthrough rather than pushing a normal cleanup for it.
955     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_end_master), Args);
956   });
957 }
958 
959 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF,
960                                         SourceLocation Loc) {
961   // Build call __kmpc_omp_taskyield(loc, thread_id, 0);
962   llvm::Value *Args[] = {
963       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
964       llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)};
965   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), Args);
966 }
967 
968 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF,
969                                        const std::function<void()> &SingleOpGen,
970                                        SourceLocation Loc) {
971   // if(__kmpc_single(ident_t *, gtid)) {
972   //   SingleOpGen();
973   //   __kmpc_end_single(ident_t *, gtid);
974   // }
975   // Prepare arguments and build a call to __kmpc_single
976   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)};
977   auto *IsSingle =
978       CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_single), Args);
979   emitIfStmt(CGF, IsSingle, [&]() -> void {
980     SingleOpGen();
981     // Build a call to __kmpc_end_single.
982     // OpenMP [1.2.2 OpenMP Language Terminology]
983     // For C/C++, an executable statement, possibly compound, with a single
984     // entry at the top and a single exit at the bottom, or an OpenMP construct.
985     // * Access to the structured block must not be the result of a branch.
986     // * The point of exit cannot be a branch out of the structured block.
987     // * The point of entry must not be a call to setjmp().
988     // * longjmp() and throw() must not violate the entry/exit criteria.
989     // * An expression statement, iteration statement, selection statement, or
990     // try block is considered to be a structured block if the corresponding
991     // compound statement obtained by enclosing it in { and } would be a
992     // structured block.
993     // It is analyzed in Sema, so we can just call __kmpc_end_single() on
994     // fallthrough rather than pushing a normal cleanup for it.
995     CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_end_single), Args);
996   });
997 }
998 
999 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc,
1000                                       bool IsExplicit) {
1001   // Build call __kmpc_cancel_barrier(loc, thread_id);
1002   auto Flags = static_cast<OpenMPLocationFlags>(
1003       OMP_IDENT_KMPC |
1004       (IsExplicit ? OMP_IDENT_BARRIER_EXPL : OMP_IDENT_BARRIER_IMPL));
1005   // Build call __kmpc_cancel_barrier(loc, thread_id);
1006   // Replace __kmpc_barrier() function by __kmpc_cancel_barrier() because this
1007   // one provides the same functionality and adds initial support for
1008   // cancellation constructs introduced in OpenMP 4.0. __kmpc_cancel_barrier()
1009   // is provided default by the runtime library so it safe to make such
1010   // replacement.
1011   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags),
1012                          getThreadID(CGF, Loc)};
1013   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args);
1014 }
1015 
1016 /// \brief Schedule types for 'omp for' loops (these enumerators are taken from
1017 /// the enum sched_type in kmp.h).
1018 enum OpenMPSchedType {
1019   /// \brief Lower bound for default (unordered) versions.
1020   OMP_sch_lower = 32,
1021   OMP_sch_static_chunked = 33,
1022   OMP_sch_static = 34,
1023   OMP_sch_dynamic_chunked = 35,
1024   OMP_sch_guided_chunked = 36,
1025   OMP_sch_runtime = 37,
1026   OMP_sch_auto = 38,
1027   /// \brief Lower bound for 'ordered' versions.
1028   OMP_ord_lower = 64,
1029   /// \brief Lower bound for 'nomerge' versions.
1030   OMP_nm_lower = 160,
1031 };
1032 
1033 /// \brief Map the OpenMP loop schedule to the runtime enumeration.
1034 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind,
1035                                           bool Chunked) {
1036   switch (ScheduleKind) {
1037   case OMPC_SCHEDULE_static:
1038     return Chunked ? OMP_sch_static_chunked : OMP_sch_static;
1039   case OMPC_SCHEDULE_dynamic:
1040     return OMP_sch_dynamic_chunked;
1041   case OMPC_SCHEDULE_guided:
1042     return OMP_sch_guided_chunked;
1043   case OMPC_SCHEDULE_auto:
1044     return OMP_sch_auto;
1045   case OMPC_SCHEDULE_runtime:
1046     return OMP_sch_runtime;
1047   case OMPC_SCHEDULE_unknown:
1048     assert(!Chunked && "chunk was specified but schedule kind not known");
1049     return OMP_sch_static;
1050   }
1051   llvm_unreachable("Unexpected runtime schedule");
1052 }
1053 
1054 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind,
1055                                          bool Chunked) const {
1056   auto Schedule = getRuntimeSchedule(ScheduleKind, Chunked);
1057   return Schedule == OMP_sch_static;
1058 }
1059 
1060 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const {
1061   auto Schedule = getRuntimeSchedule(ScheduleKind, /* Chunked */ false);
1062   assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here");
1063   return Schedule != OMP_sch_static;
1064 }
1065 
1066 void CGOpenMPRuntime::emitForInit(CodeGenFunction &CGF, SourceLocation Loc,
1067                                   OpenMPScheduleClauseKind ScheduleKind,
1068                                   unsigned IVSize, bool IVSigned,
1069                                   llvm::Value *IL, llvm::Value *LB,
1070                                   llvm::Value *UB, llvm::Value *ST,
1071                                   llvm::Value *Chunk) {
1072   OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunk != nullptr);
1073   if (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked) {
1074     // Call __kmpc_dispatch_init(
1075     //          ident_t *loc, kmp_int32 tid, kmp_int32 schedule,
1076     //          kmp_int[32|64] lower, kmp_int[32|64] upper,
1077     //          kmp_int[32|64] stride, kmp_int[32|64] chunk);
1078 
1079     // If the Chunk was not specified in the clause - use default value 1.
1080     if (Chunk == nullptr)
1081       Chunk = CGF.Builder.getIntN(IVSize, 1);
1082     llvm::Value *Args[] = { emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC),
1083                             getThreadID(CGF, Loc),
1084                             CGF.Builder.getInt32(Schedule), // Schedule type
1085                             CGF.Builder.getIntN(IVSize, 0), // Lower
1086                             UB,                             // Upper
1087                             CGF.Builder.getIntN(IVSize, 1), // Stride
1088                             Chunk                           // Chunk
1089     };
1090     CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args);
1091   } else {
1092     // Call __kmpc_for_static_init(
1093     //          ident_t *loc, kmp_int32 tid, kmp_int32 schedtype,
1094     //          kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower,
1095     //          kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride,
1096     //          kmp_int[32|64] incr, kmp_int[32|64] chunk);
1097     if (Chunk == nullptr) {
1098       assert(Schedule == OMP_sch_static &&
1099              "expected static non-chunked schedule");
1100       // If the Chunk was not specified in the clause - use default value 1.
1101       Chunk = CGF.Builder.getIntN(IVSize, 1);
1102     } else
1103       assert(Schedule == OMP_sch_static_chunked &&
1104              "expected static chunked schedule");
1105     llvm::Value *Args[] = { emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC),
1106                             getThreadID(CGF, Loc),
1107                             CGF.Builder.getInt32(Schedule), // Schedule type
1108                             IL,                             // &isLastIter
1109                             LB,                             // &LB
1110                             UB,                             // &UB
1111                             ST,                             // &Stride
1112                             CGF.Builder.getIntN(IVSize, 1), // Incr
1113                             Chunk                           // Chunk
1114     };
1115     CGF.EmitRuntimeCall(createForStaticInitFunction(IVSize, IVSigned), Args);
1116   }
1117 }
1118 
1119 void CGOpenMPRuntime::emitForFinish(CodeGenFunction &CGF, SourceLocation Loc,
1120                                     OpenMPScheduleClauseKind ScheduleKind) {
1121   assert((ScheduleKind == OMPC_SCHEDULE_static ||
1122           ScheduleKind == OMPC_SCHEDULE_unknown) &&
1123          "Non-static schedule kinds are not yet implemented");
1124   // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid);
1125   llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC),
1126                          getThreadID(CGF, Loc)};
1127   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini),
1128                       Args);
1129 }
1130 
1131 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF,
1132                                           SourceLocation Loc, unsigned IVSize,
1133                                           bool IVSigned, llvm::Value *IL,
1134                                           llvm::Value *LB, llvm::Value *UB,
1135                                           llvm::Value *ST) {
1136   // Call __kmpc_dispatch_next(
1137   //          ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter,
1138   //          kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper,
1139   //          kmp_int[32|64] *p_stride);
1140   llvm::Value *Args[] = {
1141       emitUpdateLocation(CGF, Loc, OMP_IDENT_KMPC), getThreadID(CGF, Loc),
1142       IL, // &isLastIter
1143       LB, // &Lower
1144       UB, // &Upper
1145       ST  // &Stride
1146   };
1147   llvm::Value *Call =
1148       CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args);
1149   return CGF.EmitScalarConversion(
1150       Call, CGF.getContext().getIntTypeForBitwidth(32, /* Signed */ true),
1151       CGF.getContext().BoolTy);
1152 }
1153 
1154 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF,
1155                                            llvm::Value *NumThreads,
1156                                            SourceLocation Loc) {
1157   // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads)
1158   llvm::Value *Args[] = {
1159       emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc),
1160       CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)};
1161   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads),
1162                       Args);
1163 }
1164 
1165 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>,
1166                                 SourceLocation Loc) {
1167   // Build call void __kmpc_flush(ident_t *loc)
1168   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush),
1169                       emitUpdateLocation(CGF, Loc));
1170 }
1171 
1172 namespace {
1173 /// \brief Indexes of fields for type kmp_task_t.
1174 enum KmpTaskTFields {
1175   /// \brief List of shared variables.
1176   KmpTaskTShareds,
1177   /// \brief Task routine.
1178   KmpTaskTRoutine,
1179   /// \brief Partition id for the untied tasks.
1180   KmpTaskTPartId,
1181   /// \brief Function with call of destructors for private variables.
1182   KmpTaskTDestructors,
1183 };
1184 } // namespace
1185 
1186 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) {
1187   if (!KmpRoutineEntryPtrTy) {
1188     // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type.
1189     auto &C = CGM.getContext();
1190     QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy};
1191     FunctionProtoType::ExtProtoInfo EPI;
1192     KmpRoutineEntryPtrQTy = C.getPointerType(
1193         C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI));
1194     KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy);
1195   }
1196 }
1197 
1198 static void addFieldToRecordDecl(ASTContext &C, DeclContext *DC,
1199                                  QualType FieldTy) {
1200   auto *Field = FieldDecl::Create(
1201       C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy,
1202       C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()),
1203       /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit);
1204   Field->setAccess(AS_public);
1205   DC->addDecl(Field);
1206 }
1207 
1208 static QualType createKmpTaskTRecordDecl(CodeGenModule &CGM,
1209                                          QualType KmpInt32Ty,
1210                                          QualType KmpRoutineEntryPointerQTy) {
1211   auto &C = CGM.getContext();
1212   // Build struct kmp_task_t {
1213   //         void *              shareds;
1214   //         kmp_routine_entry_t routine;
1215   //         kmp_int32           part_id;
1216   //         kmp_routine_entry_t destructors;
1217   //         /*  private vars  */
1218   //       };
1219   auto *RD = C.buildImplicitRecord("kmp_task_t");
1220   RD->startDefinition();
1221   addFieldToRecordDecl(C, RD, C.VoidPtrTy);
1222   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
1223   addFieldToRecordDecl(C, RD, KmpInt32Ty);
1224   addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy);
1225   // TODO: add private fields.
1226   RD->completeDefinition();
1227   return C.getRecordType(RD);
1228 }
1229 
1230 /// \brief Emit a proxy function which accepts kmp_task_t as the second
1231 /// argument.
1232 /// \code
1233 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) {
1234 ///   TaskFunction(gtid, tt->part_id, tt->shareds);
1235 ///   return 0;
1236 /// }
1237 /// \endcode
1238 static llvm::Value *
1239 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc,
1240                       QualType KmpInt32Ty, QualType KmpTaskTPtrQTy,
1241                       QualType SharedsPtrTy, llvm::Value *TaskFunction) {
1242   auto &C = CGM.getContext();
1243   FunctionArgList Args;
1244   ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty);
1245   ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc,
1246                                 /*Id=*/nullptr, KmpTaskTPtrQTy);
1247   Args.push_back(&GtidArg);
1248   Args.push_back(&TaskTypeArg);
1249   FunctionType::ExtInfo Info;
1250   auto &TaskEntryFnInfo =
1251       CGM.getTypes().arrangeFreeFunctionDeclaration(KmpInt32Ty, Args, Info,
1252                                                     /*isVariadic=*/false);
1253   auto *TaskEntryTy = CGM.getTypes().GetFunctionType(TaskEntryFnInfo);
1254   auto *TaskEntry =
1255       llvm::Function::Create(TaskEntryTy, llvm::GlobalValue::InternalLinkage,
1256                              ".omp_task_entry.", &CGM.getModule());
1257   CGM.SetLLVMFunctionAttributes(/*D=*/nullptr, TaskEntryFnInfo, TaskEntry);
1258   CodeGenFunction CGF(CGM);
1259   CGF.disableDebugInfo();
1260   CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args);
1261 
1262   // TaskFunction(gtid, tt->part_id, tt->shareds);
1263   auto *GtidParam = CGF.EmitLoadOfScalar(
1264       CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false,
1265       C.getTypeAlignInChars(KmpInt32Ty).getQuantity(), KmpInt32Ty, Loc);
1266   auto TaskTypeArgAddr = CGF.EmitLoadOfScalar(
1267       CGF.GetAddrOfLocalVar(&TaskTypeArg), /*Volatile=*/false,
1268       CGM.PointerAlignInBytes, KmpTaskTPtrQTy, Loc);
1269   auto *PartidPtr = CGF.Builder.CreateStructGEP(TaskTypeArgAddr,
1270                                                 /*Idx=*/KmpTaskTPartId);
1271   auto *PartidParam = CGF.EmitLoadOfScalar(
1272       PartidPtr, /*Volatile=*/false,
1273       C.getTypeAlignInChars(KmpInt32Ty).getQuantity(), KmpInt32Ty, Loc);
1274   auto *SharedsPtr = CGF.Builder.CreateStructGEP(TaskTypeArgAddr,
1275                                                  /*Idx=*/KmpTaskTShareds);
1276   auto *SharedsParam =
1277       CGF.EmitLoadOfScalar(SharedsPtr, /*Volatile=*/false,
1278                            CGM.PointerAlignInBytes, C.VoidPtrTy, Loc);
1279   llvm::Value *CallArgs[] = {
1280       GtidParam, PartidParam,
1281       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1282           SharedsParam, CGF.ConvertTypeForMem(SharedsPtrTy))};
1283   CGF.EmitCallOrInvoke(TaskFunction, CallArgs);
1284   CGF.EmitStoreThroughLValue(
1285       RValue::get(CGF.Builder.getInt32(/*C=*/0)),
1286       CGF.MakeNaturalAlignAddrLValue(CGF.ReturnValue, KmpInt32Ty));
1287   CGF.FinishFunction();
1288   return TaskEntry;
1289 }
1290 
1291 void CGOpenMPRuntime::emitTaskCall(
1292     CodeGenFunction &CGF, SourceLocation Loc, bool Tied,
1293     llvm::PointerIntPair<llvm::Value *, 1, bool> Final,
1294     llvm::Value *TaskFunction, QualType SharedsTy, llvm::Value *Shareds) {
1295   auto &C = CGM.getContext();
1296   auto KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1);
1297   // Build type kmp_routine_entry_t (if not built yet).
1298   emitKmpRoutineEntryT(KmpInt32Ty);
1299   // Build particular struct kmp_task_t for the given task.
1300   auto KmpTaskQTy =
1301       createKmpTaskTRecordDecl(CGM, KmpInt32Ty, KmpRoutineEntryPtrQTy);
1302   QualType KmpTaskTPtrQTy = C.getPointerType(KmpTaskQTy);
1303   auto KmpTaskTPtrTy = CGF.ConvertType(KmpTaskQTy)->getPointerTo();
1304   auto KmpTaskTySize = CGM.getSize(C.getTypeSizeInChars(KmpTaskQTy));
1305   QualType SharedsPtrTy = C.getPointerType(SharedsTy);
1306 
1307   // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid,
1308   // kmp_task_t *tt);
1309   auto *TaskEntry = emitProxyTaskFunction(CGM, Loc, KmpInt32Ty, KmpTaskTPtrQTy,
1310                                           SharedsPtrTy, TaskFunction);
1311 
1312   // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid,
1313   // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds,
1314   // kmp_routine_entry_t *task_entry);
1315   // Task flags. Format is taken from
1316   // http://llvm.org/svn/llvm-project/openmp/trunk/runtime/src/kmp.h,
1317   // description of kmp_tasking_flags struct.
1318   const unsigned TiedFlag = 0x1;
1319   const unsigned FinalFlag = 0x2;
1320   unsigned Flags = Tied ? TiedFlag : 0;
1321   auto *TaskFlags =
1322       Final.getPointer()
1323           ? CGF.Builder.CreateSelect(Final.getPointer(),
1324                                      CGF.Builder.getInt32(FinalFlag),
1325                                      CGF.Builder.getInt32(/*C=*/0))
1326           : CGF.Builder.getInt32(Final.getInt() ? FinalFlag : 0);
1327   TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags));
1328   auto SharedsSize = C.getTypeSizeInChars(SharedsTy);
1329   llvm::Value *AllocArgs[] = {emitUpdateLocation(CGF, Loc),
1330                               getThreadID(CGF, Loc), TaskFlags, KmpTaskTySize,
1331                               CGM.getSize(SharedsSize),
1332                               CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(
1333                                   TaskEntry, KmpRoutineEntryPtrTy)};
1334   auto *NewTask = CGF.EmitRuntimeCall(
1335       createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs);
1336   auto *NewTaskNewTaskTTy =
1337       CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(NewTask, KmpTaskTPtrTy);
1338   // Fill the data in the resulting kmp_task_t record.
1339   // Copy shareds if there are any.
1340   if (!SharedsTy->getAsStructureType()->getDecl()->field_empty())
1341     CGF.EmitAggregateCopy(
1342         CGF.EmitLoadOfScalar(
1343             CGF.Builder.CreateStructGEP(NewTaskNewTaskTTy,
1344                                         /*Idx=*/KmpTaskTShareds),
1345             /*Volatile=*/false, CGM.PointerAlignInBytes, SharedsPtrTy, Loc),
1346         Shareds, SharedsTy);
1347   // TODO: generate function with destructors for privates.
1348   // Provide pointer to function with destructors for privates.
1349   CGF.Builder.CreateAlignedStore(
1350       llvm::ConstantPointerNull::get(
1351           cast<llvm::PointerType>(KmpRoutineEntryPtrTy)),
1352       CGF.Builder.CreateStructGEP(NewTaskNewTaskTTy,
1353                                   /*Idx=*/KmpTaskTDestructors),
1354       CGM.PointerAlignInBytes);
1355 
1356   // NOTE: routine and part_id fields are intialized by __kmpc_omp_task_alloc()
1357   // libcall.
1358   // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t
1359   // *new_task);
1360   llvm::Value *TaskArgs[] = {emitUpdateLocation(CGF, Loc),
1361                              getThreadID(CGF, Loc), NewTask};
1362   // TODO: add check for untied tasks.
1363   CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs);
1364 }
1365 
1366 InlinedOpenMPRegionRAII::InlinedOpenMPRegionRAII(
1367     CodeGenFunction &CGF, const OMPExecutableDirective &D)
1368     : CGF(CGF) {
1369   CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo(D, CGF.CapturedStmtInfo);
1370   // 1.2.2 OpenMP Language Terminology
1371   // Structured block - An executable statement with a single entry at the
1372   // top and a single exit at the bottom.
1373   // The point of exit cannot be a branch out of the structured block.
1374   // longjmp() and throw() must not violate the entry/exit criteria.
1375   CGF.EHStack.pushTerminate();
1376 }
1377 
1378 InlinedOpenMPRegionRAII::~InlinedOpenMPRegionRAII() {
1379   CGF.EHStack.popTerminate();
1380   auto *OldCSI =
1381       cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI();
1382   delete CGF.CapturedStmtInfo;
1383   CGF.CapturedStmtInfo = OldCSI;
1384 }
1385 
1386