1 //===--- CGStmtOpenMP.cpp - Emit LLVM Code from Statements ----------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 // This contains code to emit OpenMP nodes as LLVM code.
11 //
12 //===----------------------------------------------------------------------===//
13 
14 #include "CGOpenMPRuntime.h"
15 #include "CodeGenFunction.h"
16 #include "CodeGenModule.h"
17 #include "TargetInfo.h"
18 #include "clang/AST/Stmt.h"
19 #include "clang/AST/StmtOpenMP.h"
20 using namespace clang;
21 using namespace CodeGen;
22 
23 //===----------------------------------------------------------------------===//
24 //                              OpenMP Directive Emission
25 //===----------------------------------------------------------------------===//
26 namespace {
27 /// \brief RAII for inlined OpenMP regions (like 'omp for', 'omp simd', 'omp
28 /// critical' etc.). Helps to generate proper debug info and provides correct
29 /// code generation for such constructs.
30 class InlinedOpenMPRegionScopeRAII {
31   InlinedOpenMPRegionRAII Region;
32   CodeGenFunction::LexicalScope DirectiveScope;
33 
34 public:
35   InlinedOpenMPRegionScopeRAII(CodeGenFunction &CGF,
36                                const OMPExecutableDirective &D)
37       : Region(CGF, D), DirectiveScope(CGF, D.getSourceRange()) {}
38 };
39 } // namespace
40 
41 /// \brief Emits code for OpenMP 'if' clause using specified \a CodeGen
42 /// function. Here is the logic:
43 /// if (Cond) {
44 ///   CodeGen(true);
45 /// } else {
46 ///   CodeGen(false);
47 /// }
48 static void EmitOMPIfClause(CodeGenFunction &CGF, const Expr *Cond,
49                             const std::function<void(bool)> &CodeGen) {
50   CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange());
51 
52   // If the condition constant folds and can be elided, try to avoid emitting
53   // the condition and the dead arm of the if/else.
54   bool CondConstant;
55   if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) {
56     CodeGen(CondConstant);
57     return;
58   }
59 
60   // Otherwise, the condition did not fold, or we couldn't elide it.  Just
61   // emit the conditional branch.
62   auto ThenBlock = CGF.createBasicBlock(/*name*/ "omp_if.then");
63   auto ElseBlock = CGF.createBasicBlock(/*name*/ "omp_if.else");
64   auto ContBlock = CGF.createBasicBlock(/*name*/ "omp_if.end");
65   CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount*/ 0);
66 
67   // Emit the 'then' code.
68   CGF.EmitBlock(ThenBlock);
69   CodeGen(/*ThenBlock*/ true);
70   CGF.EmitBranch(ContBlock);
71   // Emit the 'else' code if present.
72   {
73     // There is no need to emit line number for unconditional branch.
74     auto NL = ApplyDebugLocation::CreateEmpty(CGF);
75     CGF.EmitBlock(ElseBlock);
76   }
77   CodeGen(/*ThenBlock*/ false);
78   {
79     // There is no need to emit line number for unconditional branch.
80     auto NL = ApplyDebugLocation::CreateEmpty(CGF);
81     CGF.EmitBranch(ContBlock);
82   }
83   // Emit the continuation block for code after the if.
84   CGF.EmitBlock(ContBlock, /*IsFinished*/ true);
85 }
86 
87 void CodeGenFunction::EmitOMPAggregateAssign(LValue OriginalAddr,
88                                              llvm::Value *PrivateAddr,
89                                              const Expr *AssignExpr,
90                                              QualType OriginalType,
91                                              const VarDecl *VDInit) {
92   EmitBlock(createBasicBlock(".omp.assign.begin."));
93   if (!isa<CXXConstructExpr>(AssignExpr) || isTrivialInitializer(AssignExpr)) {
94     // Perform simple memcpy.
95     EmitAggregateAssign(PrivateAddr, OriginalAddr.getAddress(),
96                         AssignExpr->getType());
97   } else {
98     // Perform element-by-element initialization.
99     QualType ElementTy;
100     auto SrcBegin = OriginalAddr.getAddress();
101     auto DestBegin = PrivateAddr;
102     auto ArrayTy = OriginalType->getAsArrayTypeUnsafe();
103     auto SrcNumElements = emitArrayLength(ArrayTy, ElementTy, SrcBegin);
104     auto DestNumElements = emitArrayLength(ArrayTy, ElementTy, DestBegin);
105     auto SrcEnd = Builder.CreateGEP(SrcBegin, SrcNumElements);
106     auto DestEnd = Builder.CreateGEP(DestBegin, DestNumElements);
107     // The basic structure here is a do-while loop, because we don't
108     // need to check for the zero-element case.
109     auto BodyBB = createBasicBlock("omp.arraycpy.body");
110     auto DoneBB = createBasicBlock("omp.arraycpy.done");
111     auto IsEmpty =
112         Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arraycpy.isempty");
113     Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
114 
115     // Enter the loop body, making that address the current address.
116     auto EntryBB = Builder.GetInsertBlock();
117     EmitBlock(BodyBB);
118     auto SrcElementPast = Builder.CreatePHI(SrcBegin->getType(), 2,
119                                             "omp.arraycpy.srcElementPast");
120     SrcElementPast->addIncoming(SrcEnd, EntryBB);
121     auto DestElementPast = Builder.CreatePHI(DestBegin->getType(), 2,
122                                              "omp.arraycpy.destElementPast");
123     DestElementPast->addIncoming(DestEnd, EntryBB);
124 
125     // Shift the address back by one element.
126     auto NegativeOne = llvm::ConstantInt::get(SizeTy, -1, true);
127     auto DestElement = Builder.CreateGEP(DestElementPast, NegativeOne,
128                                          "omp.arraycpy.dest.element");
129     auto SrcElement = Builder.CreateGEP(SrcElementPast, NegativeOne,
130                                         "omp.arraycpy.src.element");
131     {
132       // Create RunCleanScope to cleanup possible temps.
133       CodeGenFunction::RunCleanupsScope Init(*this);
134       // Emit initialization for single element.
135       LocalDeclMap[VDInit] = SrcElement;
136       EmitAnyExprToMem(AssignExpr, DestElement,
137                        AssignExpr->getType().getQualifiers(),
138                        /*IsInitializer*/ false);
139       LocalDeclMap.erase(VDInit);
140     }
141 
142     // Check whether we've reached the end.
143     auto Done =
144         Builder.CreateICmpEQ(DestElement, DestBegin, "omp.arraycpy.done");
145     Builder.CreateCondBr(Done, DoneBB, BodyBB);
146     DestElementPast->addIncoming(DestElement, Builder.GetInsertBlock());
147     SrcElementPast->addIncoming(SrcElement, Builder.GetInsertBlock());
148 
149     // Done.
150     EmitBlock(DoneBB, true);
151   }
152   EmitBlock(createBasicBlock(".omp.assign.end."));
153 }
154 
155 void CodeGenFunction::EmitOMPFirstprivateClause(
156     const OMPExecutableDirective &D,
157     CodeGenFunction::OMPPrivateScope &PrivateScope) {
158   auto PrivateFilter = [](const OMPClause *C) -> bool {
159     return C->getClauseKind() == OMPC_firstprivate;
160   };
161   for (OMPExecutableDirective::filtered_clause_iterator<decltype(PrivateFilter)>
162            I(D.clauses(), PrivateFilter); I; ++I) {
163     auto *C = cast<OMPFirstprivateClause>(*I);
164     auto IRef = C->varlist_begin();
165     auto InitsRef = C->inits().begin();
166     for (auto IInit : C->private_copies()) {
167       auto *OrigVD = cast<VarDecl>(cast<DeclRefExpr>(*IRef)->getDecl());
168       auto *VD = cast<VarDecl>(cast<DeclRefExpr>(IInit)->getDecl());
169       bool IsRegistered;
170       if (*InitsRef != nullptr) {
171         // Emit VarDecl with copy init for arrays.
172         auto *FD = CapturedStmtInfo->lookup(OrigVD);
173         LValue Base = MakeNaturalAlignAddrLValue(
174             CapturedStmtInfo->getContextValue(),
175             getContext().getTagDeclType(FD->getParent()));
176         auto OriginalAddr = EmitLValueForField(Base, FD);
177         auto VDInit = cast<VarDecl>(cast<DeclRefExpr>(*InitsRef)->getDecl());
178         IsRegistered = PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * {
179           auto Emission = EmitAutoVarAlloca(*VD);
180           // Emit initialization of aggregate firstprivate vars.
181           EmitOMPAggregateAssign(OriginalAddr, Emission.getAllocatedAddress(),
182                                  VD->getInit(), (*IRef)->getType(), VDInit);
183           EmitAutoVarCleanups(Emission);
184           return Emission.getAllocatedAddress();
185         });
186       } else
187         IsRegistered = PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * {
188           // Emit private VarDecl with copy init.
189           EmitDecl(*VD);
190           return GetAddrOfLocalVar(VD);
191         });
192       assert(IsRegistered && "counter already registered as private");
193       // Silence the warning about unused variable.
194       (void)IsRegistered;
195       ++IRef, ++InitsRef;
196     }
197   }
198 }
199 
200 void CodeGenFunction::EmitOMPPrivateClause(
201     const OMPExecutableDirective &D,
202     CodeGenFunction::OMPPrivateScope &PrivateScope) {
203   auto PrivateFilter = [](const OMPClause *C) -> bool {
204     return C->getClauseKind() == OMPC_private;
205   };
206   for (OMPExecutableDirective::filtered_clause_iterator<decltype(PrivateFilter)>
207            I(D.clauses(), PrivateFilter); I; ++I) {
208     auto *C = cast<OMPPrivateClause>(*I);
209     auto IRef = C->varlist_begin();
210     for (auto IInit : C->private_copies()) {
211       auto *OrigVD = cast<VarDecl>(cast<DeclRefExpr>(*IRef)->getDecl());
212       auto VD = cast<VarDecl>(cast<DeclRefExpr>(IInit)->getDecl());
213       bool IsRegistered =
214           PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * {
215             // Emit private VarDecl with copy init.
216             EmitDecl(*VD);
217             return GetAddrOfLocalVar(VD);
218           });
219       assert(IsRegistered && "counter already registered as private");
220       // Silence the warning about unused variable.
221       (void)IsRegistered;
222       ++IRef;
223     }
224   }
225 }
226 
227 /// \brief Emits code for OpenMP parallel directive in the parallel region.
228 static void EmitOMPParallelCall(CodeGenFunction &CGF,
229                                 const OMPParallelDirective &S,
230                                 llvm::Value *OutlinedFn,
231                                 llvm::Value *CapturedStruct) {
232   if (auto C = S.getSingleClause(/*K*/ OMPC_num_threads)) {
233     CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF);
234     auto NumThreadsClause = cast<OMPNumThreadsClause>(C);
235     auto NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads(),
236                                          /*IgnoreResultAssign*/ true);
237     CGF.CGM.getOpenMPRuntime().emitNumThreadsClause(
238         CGF, NumThreads, NumThreadsClause->getLocStart());
239   }
240   CGF.CGM.getOpenMPRuntime().emitParallelCall(CGF, S.getLocStart(), OutlinedFn,
241                                               CapturedStruct);
242 }
243 
244 void CodeGenFunction::EmitOMPParallelDirective(const OMPParallelDirective &S) {
245   auto CS = cast<CapturedStmt>(S.getAssociatedStmt());
246   auto CapturedStruct = GenerateCapturedStmtArgument(*CS);
247   auto OutlinedFn = CGM.getOpenMPRuntime().emitOutlinedFunction(
248       S, *CS->getCapturedDecl()->param_begin());
249   if (auto C = S.getSingleClause(/*K*/ OMPC_if)) {
250     auto Cond = cast<OMPIfClause>(C)->getCondition();
251     EmitOMPIfClause(*this, Cond, [&](bool ThenBlock) {
252       if (ThenBlock)
253         EmitOMPParallelCall(*this, S, OutlinedFn, CapturedStruct);
254       else
255         CGM.getOpenMPRuntime().emitSerialCall(*this, S.getLocStart(),
256                                               OutlinedFn, CapturedStruct);
257     });
258   } else
259     EmitOMPParallelCall(*this, S, OutlinedFn, CapturedStruct);
260 }
261 
262 void CodeGenFunction::EmitOMPLoopBody(const OMPLoopDirective &S,
263                                       bool SeparateIter) {
264   RunCleanupsScope BodyScope(*this);
265   // Update counters values on current iteration.
266   for (auto I : S.updates()) {
267     EmitIgnoredExpr(I);
268   }
269   // On a continue in the body, jump to the end.
270   auto Continue = getJumpDestInCurrentScope("omp.body.continue");
271   BreakContinueStack.push_back(BreakContinue(JumpDest(), Continue));
272   // Emit loop body.
273   EmitStmt(S.getBody());
274   // The end (updates/cleanups).
275   EmitBlock(Continue.getBlock());
276   BreakContinueStack.pop_back();
277   if (SeparateIter) {
278     // TODO: Update lastprivates if the SeparateIter flag is true.
279     // This will be implemented in a follow-up OMPLastprivateClause patch, but
280     // result should be still correct without it, as we do not make these
281     // variables private yet.
282   }
283 }
284 
285 void CodeGenFunction::EmitOMPInnerLoop(const OMPLoopDirective &S,
286                                        OMPPrivateScope &LoopScope,
287                                        bool SeparateIter) {
288   auto LoopExit = getJumpDestInCurrentScope("omp.inner.for.end");
289   auto Cnt = getPGORegionCounter(&S);
290 
291   // Start the loop with a block that tests the condition.
292   auto CondBlock = createBasicBlock("omp.inner.for.cond");
293   EmitBlock(CondBlock);
294   LoopStack.push(CondBlock);
295 
296   // If there are any cleanups between here and the loop-exit scope,
297   // create a block to stage a loop exit along.
298   auto ExitBlock = LoopExit.getBlock();
299   if (LoopScope.requiresCleanups())
300     ExitBlock = createBasicBlock("omp.inner.for.cond.cleanup");
301 
302   auto LoopBody = createBasicBlock("omp.inner.for.body");
303 
304   // Emit condition: "IV < LastIteration + 1 [ - 1]"
305   // ("- 1" when lastprivate clause is present - separate one iteration).
306   llvm::Value *BoolCondVal = EvaluateExprAsBool(S.getCond(SeparateIter));
307   Builder.CreateCondBr(BoolCondVal, LoopBody, ExitBlock,
308                        PGO.createLoopWeights(S.getCond(SeparateIter), Cnt));
309 
310   if (ExitBlock != LoopExit.getBlock()) {
311     EmitBlock(ExitBlock);
312     EmitBranchThroughCleanup(LoopExit);
313   }
314 
315   EmitBlock(LoopBody);
316   Cnt.beginRegion(Builder);
317 
318   // Create a block for the increment.
319   auto Continue = getJumpDestInCurrentScope("omp.inner.for.inc");
320   BreakContinueStack.push_back(BreakContinue(LoopExit, Continue));
321 
322   EmitOMPLoopBody(S);
323   EmitStopPoint(&S);
324 
325   // Emit "IV = IV + 1" and a back-edge to the condition block.
326   EmitBlock(Continue.getBlock());
327   EmitIgnoredExpr(S.getInc());
328   BreakContinueStack.pop_back();
329   EmitBranch(CondBlock);
330   LoopStack.pop();
331   // Emit the fall-through block.
332   EmitBlock(LoopExit.getBlock());
333 }
334 
335 void CodeGenFunction::EmitOMPSimdFinal(const OMPLoopDirective &S) {
336   auto IC = S.counters().begin();
337   for (auto F : S.finals()) {
338     if (LocalDeclMap.lookup(cast<DeclRefExpr>((*IC))->getDecl())) {
339       EmitIgnoredExpr(F);
340     }
341     ++IC;
342   }
343 }
344 
345 static void EmitOMPAlignedClause(CodeGenFunction &CGF, CodeGenModule &CGM,
346                                  const OMPAlignedClause &Clause) {
347   unsigned ClauseAlignment = 0;
348   if (auto AlignmentExpr = Clause.getAlignment()) {
349     auto AlignmentCI =
350         cast<llvm::ConstantInt>(CGF.EmitScalarExpr(AlignmentExpr));
351     ClauseAlignment = static_cast<unsigned>(AlignmentCI->getZExtValue());
352   }
353   for (auto E : Clause.varlists()) {
354     unsigned Alignment = ClauseAlignment;
355     if (Alignment == 0) {
356       // OpenMP [2.8.1, Description]
357       // If no optional parameter is specified, implementation-defined default
358       // alignments for SIMD instructions on the target platforms are assumed.
359       Alignment = CGM.getTargetCodeGenInfo().getOpenMPSimdDefaultAlignment(
360           E->getType());
361     }
362     assert((Alignment == 0 || llvm::isPowerOf2_32(Alignment)) &&
363            "alignment is not power of 2");
364     if (Alignment != 0) {
365       llvm::Value *PtrValue = CGF.EmitScalarExpr(E);
366       CGF.EmitAlignmentAssumption(PtrValue, Alignment);
367     }
368   }
369 }
370 
371 static void EmitPrivateLoopCounters(CodeGenFunction &CGF,
372                                     CodeGenFunction::OMPPrivateScope &LoopScope,
373                                     ArrayRef<Expr *> Counters) {
374   for (auto *E : Counters) {
375     auto VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl());
376     bool IsRegistered = LoopScope.addPrivate(VD, [&]() -> llvm::Value * {
377       // Emit var without initialization.
378       auto VarEmission = CGF.EmitAutoVarAlloca(*VD);
379       CGF.EmitAutoVarCleanups(VarEmission);
380       return VarEmission.getAllocatedAddress();
381     });
382     assert(IsRegistered && "counter already registered as private");
383     // Silence the warning about unused variable.
384     (void)IsRegistered;
385   }
386   (void)LoopScope.Privatize();
387 }
388 
389 void CodeGenFunction::EmitOMPSimdDirective(const OMPSimdDirective &S) {
390   // Pragma 'simd' code depends on presence of 'lastprivate'.
391   // If present, we have to separate last iteration of the loop:
392   //
393   // if (LastIteration != 0) {
394   //   for (IV in 0..LastIteration-1) BODY;
395   //   BODY with updates of lastprivate vars;
396   //   <Final counter/linear vars updates>;
397   // }
398   //
399   // otherwise (when there's no lastprivate):
400   //
401   //   for (IV in 0..LastIteration) BODY;
402   //   <Final counter/linear vars updates>;
403   //
404 
405   // Walk clauses and process safelen/lastprivate.
406   bool SeparateIter = false;
407   LoopStack.setParallel();
408   LoopStack.setVectorizerEnable(true);
409   for (auto C : S.clauses()) {
410     switch (C->getClauseKind()) {
411     case OMPC_safelen: {
412       RValue Len = EmitAnyExpr(cast<OMPSafelenClause>(C)->getSafelen(),
413                                AggValueSlot::ignored(), true);
414       llvm::ConstantInt *Val = cast<llvm::ConstantInt>(Len.getScalarVal());
415       LoopStack.setVectorizerWidth(Val->getZExtValue());
416       // In presence of finite 'safelen', it may be unsafe to mark all
417       // the memory instructions parallel, because loop-carried
418       // dependences of 'safelen' iterations are possible.
419       LoopStack.setParallel(false);
420       break;
421     }
422     case OMPC_aligned:
423       EmitOMPAlignedClause(*this, CGM, cast<OMPAlignedClause>(*C));
424       break;
425     case OMPC_lastprivate:
426       SeparateIter = true;
427       break;
428     default:
429       // Not handled yet
430       ;
431     }
432   }
433 
434   InlinedOpenMPRegionScopeRAII Region(*this, S);
435 
436   // Emit the loop iteration variable.
437   const Expr *IVExpr = S.getIterationVariable();
438   const VarDecl *IVDecl = cast<VarDecl>(cast<DeclRefExpr>(IVExpr)->getDecl());
439   EmitVarDecl(*IVDecl);
440   EmitIgnoredExpr(S.getInit());
441 
442   // Emit the iterations count variable.
443   // If it is not a variable, Sema decided to calculate iterations count on each
444   // iteration (e.g., it is foldable into a constant).
445   if (auto LIExpr = dyn_cast<DeclRefExpr>(S.getLastIteration())) {
446     EmitVarDecl(*cast<VarDecl>(LIExpr->getDecl()));
447     // Emit calculation of the iterations count.
448     EmitIgnoredExpr(S.getCalcLastIteration());
449   }
450 
451   if (SeparateIter) {
452     // Emit: if (LastIteration > 0) - begin.
453     RegionCounter Cnt = getPGORegionCounter(&S);
454     auto ThenBlock = createBasicBlock("simd.if.then");
455     auto ContBlock = createBasicBlock("simd.if.end");
456     EmitBranchOnBoolExpr(S.getPreCond(), ThenBlock, ContBlock, Cnt.getCount());
457     EmitBlock(ThenBlock);
458     Cnt.beginRegion(Builder);
459     // Emit 'then' code.
460     {
461       OMPPrivateScope LoopScope(*this);
462       EmitPrivateLoopCounters(*this, LoopScope, S.counters());
463       EmitOMPInnerLoop(S, LoopScope, /* SeparateIter */ true);
464       EmitOMPLoopBody(S, /* SeparateIter */ true);
465     }
466     EmitOMPSimdFinal(S);
467     // Emit: if (LastIteration != 0) - end.
468     EmitBranch(ContBlock);
469     EmitBlock(ContBlock, true);
470   } else {
471     {
472       OMPPrivateScope LoopScope(*this);
473       EmitPrivateLoopCounters(*this, LoopScope, S.counters());
474       EmitOMPInnerLoop(S, LoopScope);
475     }
476     EmitOMPSimdFinal(S);
477   }
478 }
479 
480 void CodeGenFunction::EmitOMPForOuterLoop(OpenMPScheduleClauseKind ScheduleKind,
481                                           const OMPLoopDirective &S,
482                                           OMPPrivateScope &LoopScope,
483                                           llvm::Value *LB, llvm::Value *UB,
484                                           llvm::Value *ST, llvm::Value *IL,
485                                           llvm::Value *Chunk) {
486   auto &RT = CGM.getOpenMPRuntime();
487   assert(!RT.isStaticNonchunked(ScheduleKind, /* Chunked */ Chunk != nullptr) &&
488          "static non-chunked schedule does not need outer loop");
489   if (RT.isDynamic(ScheduleKind)) {
490     ErrorUnsupported(&S, "OpenMP loop with dynamic schedule");
491     return;
492   }
493 
494   // Emit outer loop.
495   //
496   // OpenMP [2.7.1, Loop Construct, Description, table 2-1]
497   // When schedule(static, chunk_size) is specified, iterations are divided into
498   // chunks of size chunk_size, and the chunks are assigned to the threads in
499   // the team in a round-robin fashion in the order of the thread number.
500   //
501   // while(UB = min(UB, GlobalUB), idx = LB, idx < UB) {
502   //   while (idx <= UB) { BODY; ++idx; } // inner loop
503   //   LB = LB + ST;
504   //   UB = UB + ST;
505   // }
506   //
507   const Expr *IVExpr = S.getIterationVariable();
508   const unsigned IVSize = getContext().getTypeSize(IVExpr->getType());
509   const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation();
510 
511   RT.emitForInit(*this, S.getLocStart(), ScheduleKind, IVSize, IVSigned, IL, LB,
512                  UB, ST, Chunk);
513   auto LoopExit = getJumpDestInCurrentScope("omp.dispatch.end");
514 
515   // Start the loop with a block that tests the condition.
516   auto CondBlock = createBasicBlock("omp.dispatch.cond");
517   EmitBlock(CondBlock);
518   LoopStack.push(CondBlock);
519 
520   llvm::Value *BoolCondVal = nullptr;
521   // UB = min(UB, GlobalUB)
522   EmitIgnoredExpr(S.getEnsureUpperBound());
523   // IV = LB
524   EmitIgnoredExpr(S.getInit());
525   // IV < UB
526   BoolCondVal = EvaluateExprAsBool(S.getCond(false));
527 
528   // If there are any cleanups between here and the loop-exit scope,
529   // create a block to stage a loop exit along.
530   auto ExitBlock = LoopExit.getBlock();
531   if (LoopScope.requiresCleanups())
532     ExitBlock = createBasicBlock("omp.dispatch.cleanup");
533 
534   auto LoopBody = createBasicBlock("omp.dispatch.body");
535   Builder.CreateCondBr(BoolCondVal, LoopBody, ExitBlock);
536   if (ExitBlock != LoopExit.getBlock()) {
537     EmitBlock(ExitBlock);
538     EmitBranchThroughCleanup(LoopExit);
539   }
540   EmitBlock(LoopBody);
541 
542   // Create a block for the increment.
543   auto Continue = getJumpDestInCurrentScope("omp.dispatch.inc");
544   BreakContinueStack.push_back(BreakContinue(LoopExit, Continue));
545 
546   EmitOMPInnerLoop(S, LoopScope);
547 
548   EmitBlock(Continue.getBlock());
549   BreakContinueStack.pop_back();
550   // Emit "LB = LB + Stride", "UB = UB + Stride".
551   EmitIgnoredExpr(S.getNextLowerBound());
552   EmitIgnoredExpr(S.getNextUpperBound());
553 
554   EmitBranch(CondBlock);
555   LoopStack.pop();
556   // Emit the fall-through block.
557   EmitBlock(LoopExit.getBlock());
558 
559   // Tell the runtime we are done.
560   RT.emitForFinish(*this, S.getLocStart(), ScheduleKind);
561 }
562 
563 /// \brief Emit a helper variable and return corresponding lvalue.
564 static LValue EmitOMPHelperVar(CodeGenFunction &CGF,
565                                const DeclRefExpr *Helper) {
566   auto VDecl = cast<VarDecl>(Helper->getDecl());
567   CGF.EmitVarDecl(*VDecl);
568   return CGF.EmitLValue(Helper);
569 }
570 
571 void CodeGenFunction::EmitOMPWorksharingLoop(const OMPLoopDirective &S) {
572   // Emit the loop iteration variable.
573   auto IVExpr = cast<DeclRefExpr>(S.getIterationVariable());
574   auto IVDecl = cast<VarDecl>(IVExpr->getDecl());
575   EmitVarDecl(*IVDecl);
576 
577   // Emit the iterations count variable.
578   // If it is not a variable, Sema decided to calculate iterations count on each
579   // iteration (e.g., it is foldable into a constant).
580   if (auto LIExpr = dyn_cast<DeclRefExpr>(S.getLastIteration())) {
581     EmitVarDecl(*cast<VarDecl>(LIExpr->getDecl()));
582     // Emit calculation of the iterations count.
583     EmitIgnoredExpr(S.getCalcLastIteration());
584   }
585 
586   auto &RT = CGM.getOpenMPRuntime();
587 
588   // Check pre-condition.
589   {
590     // Skip the entire loop if we don't meet the precondition.
591     RegionCounter Cnt = getPGORegionCounter(&S);
592     auto ThenBlock = createBasicBlock("omp.precond.then");
593     auto ContBlock = createBasicBlock("omp.precond.end");
594     EmitBranchOnBoolExpr(S.getPreCond(), ThenBlock, ContBlock, Cnt.getCount());
595     EmitBlock(ThenBlock);
596     Cnt.beginRegion(Builder);
597     // Emit 'then' code.
598     {
599       // Emit helper vars inits.
600       LValue LB =
601           EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getLowerBoundVariable()));
602       LValue UB =
603           EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getUpperBoundVariable()));
604       LValue ST =
605           EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getStrideVariable()));
606       LValue IL =
607           EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getIsLastIterVariable()));
608 
609       OMPPrivateScope LoopScope(*this);
610       EmitPrivateLoopCounters(*this, LoopScope, S.counters());
611 
612       // Detect the loop schedule kind and chunk.
613       auto ScheduleKind = OMPC_SCHEDULE_unknown;
614       llvm::Value *Chunk = nullptr;
615       if (auto C = cast_or_null<OMPScheduleClause>(
616               S.getSingleClause(OMPC_schedule))) {
617         ScheduleKind = C->getScheduleKind();
618         if (auto Ch = C->getChunkSize()) {
619           Chunk = EmitScalarExpr(Ch);
620           Chunk = EmitScalarConversion(Chunk, Ch->getType(),
621                                        S.getIterationVariable()->getType());
622         }
623       }
624       const unsigned IVSize = getContext().getTypeSize(IVExpr->getType());
625       const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation();
626       if (RT.isStaticNonchunked(ScheduleKind,
627                                 /* Chunked */ Chunk != nullptr)) {
628         // OpenMP [2.7.1, Loop Construct, Description, table 2-1]
629         // When no chunk_size is specified, the iteration space is divided into
630         // chunks that are approximately equal in size, and at most one chunk is
631         // distributed to each thread. Note that the size of the chunks is
632         // unspecified in this case.
633         RT.emitForInit(*this, S.getLocStart(), ScheduleKind, IVSize, IVSigned,
634                        IL.getAddress(), LB.getAddress(), UB.getAddress(),
635                        ST.getAddress());
636         // UB = min(UB, GlobalUB);
637         EmitIgnoredExpr(S.getEnsureUpperBound());
638         // IV = LB;
639         EmitIgnoredExpr(S.getInit());
640         // while (idx <= UB) { BODY; ++idx; }
641         EmitOMPInnerLoop(S, LoopScope);
642         // Tell the runtime we are done.
643         RT.emitForFinish(*this, S.getLocStart(), ScheduleKind);
644       } else {
645         // Emit the outer loop, which requests its work chunk [LB..UB] from
646         // runtime and runs the inner loop to process it.
647         EmitOMPForOuterLoop(ScheduleKind, S, LoopScope, LB.getAddress(),
648                             UB.getAddress(), ST.getAddress(), IL.getAddress(),
649                             Chunk);
650       }
651     }
652     // We're now done with the loop, so jump to the continuation block.
653     EmitBranch(ContBlock);
654     EmitBlock(ContBlock, true);
655   }
656 }
657 
658 void CodeGenFunction::EmitOMPForDirective(const OMPForDirective &S) {
659   InlinedOpenMPRegionScopeRAII Region(*this, S);
660 
661   EmitOMPWorksharingLoop(S);
662 
663   // Emit an implicit barrier at the end.
664   CGM.getOpenMPRuntime().emitBarrierCall(*this, S.getLocStart(),
665                                          /*IsExplicit*/ false);
666 }
667 
668 void CodeGenFunction::EmitOMPForSimdDirective(const OMPForSimdDirective &) {
669   llvm_unreachable("CodeGen for 'omp for simd' is not supported yet.");
670 }
671 
672 void CodeGenFunction::EmitOMPSectionsDirective(const OMPSectionsDirective &) {
673   llvm_unreachable("CodeGen for 'omp sections' is not supported yet.");
674 }
675 
676 void CodeGenFunction::EmitOMPSectionDirective(const OMPSectionDirective &) {
677   llvm_unreachable("CodeGen for 'omp section' is not supported yet.");
678 }
679 
680 void CodeGenFunction::EmitOMPSingleDirective(const OMPSingleDirective &S) {
681   CGM.getOpenMPRuntime().emitSingleRegion(*this, [&]() -> void {
682     InlinedOpenMPRegionScopeRAII Region(*this, S);
683     EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt());
684     EnsureInsertPoint();
685   }, S.getLocStart());
686 }
687 
688 void CodeGenFunction::EmitOMPMasterDirective(const OMPMasterDirective &S) {
689   CGM.getOpenMPRuntime().emitMasterRegion(*this, [&]() -> void {
690     InlinedOpenMPRegionScopeRAII Region(*this, S);
691     EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt());
692     EnsureInsertPoint();
693   }, S.getLocStart());
694 }
695 
696 void CodeGenFunction::EmitOMPCriticalDirective(const OMPCriticalDirective &S) {
697   CGM.getOpenMPRuntime().emitCriticalRegion(
698       *this, S.getDirectiveName().getAsString(), [&]() -> void {
699         InlinedOpenMPRegionScopeRAII Region(*this, S);
700         EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt());
701         EnsureInsertPoint();
702       }, S.getLocStart());
703 }
704 
705 void
706 CodeGenFunction::EmitOMPParallelForDirective(const OMPParallelForDirective &) {
707   llvm_unreachable("CodeGen for 'omp parallel for' is not supported yet.");
708 }
709 
710 void CodeGenFunction::EmitOMPParallelForSimdDirective(
711     const OMPParallelForSimdDirective &) {
712   llvm_unreachable("CodeGen for 'omp parallel for simd' is not supported yet.");
713 }
714 
715 void CodeGenFunction::EmitOMPParallelSectionsDirective(
716     const OMPParallelSectionsDirective &) {
717   llvm_unreachable("CodeGen for 'omp parallel sections' is not supported yet.");
718 }
719 
720 void CodeGenFunction::EmitOMPTaskDirective(const OMPTaskDirective &S) {
721   // Emit outlined function for task construct.
722   auto CS = cast<CapturedStmt>(S.getAssociatedStmt());
723   auto CapturedStruct = GenerateCapturedStmtArgument(*CS);
724   auto *I = CS->getCapturedDecl()->param_begin();
725   // The first function argument for tasks is a thread id, the second one is a
726   // part id (0 for tied tasks, >=0 for untied task).
727   auto OutlinedFn =
728       CGM.getOpenMPRuntime().emitTaskOutlinedFunction(S, *I, *std::next(I));
729   // Check if we should emit tied or untied task.
730   bool Tied = !S.getSingleClause(OMPC_untied);
731   // Check if the task is final
732   llvm::PointerIntPair<llvm::Value *, 1, bool> Final;
733   if (auto *Clause = S.getSingleClause(OMPC_final)) {
734     // If the condition constant folds and can be elided, try to avoid emitting
735     // the condition and the dead arm of the if/else.
736     auto *Cond = cast<OMPFinalClause>(Clause)->getCondition();
737     bool CondConstant;
738     if (ConstantFoldsToSimpleInteger(Cond, CondConstant))
739       Final.setInt(CondConstant);
740     else
741       Final.setPointer(EvaluateExprAsBool(Cond));
742   } else {
743     // By default the task is not final.
744     Final.setInt(/*IntVal=*/false);
745   }
746   auto SharedsTy = getContext().getRecordType(CS->getCapturedRecordDecl());
747   CGM.getOpenMPRuntime().emitTaskCall(*this, S.getLocStart(), Tied, Final,
748                                       OutlinedFn, SharedsTy, CapturedStruct);
749 }
750 
751 void CodeGenFunction::EmitOMPTaskyieldDirective(
752     const OMPTaskyieldDirective &S) {
753   CGM.getOpenMPRuntime().emitTaskyieldCall(*this, S.getLocStart());
754 }
755 
756 void CodeGenFunction::EmitOMPBarrierDirective(const OMPBarrierDirective &S) {
757   CGM.getOpenMPRuntime().emitBarrierCall(*this, S.getLocStart());
758 }
759 
760 void CodeGenFunction::EmitOMPTaskwaitDirective(const OMPTaskwaitDirective &) {
761   llvm_unreachable("CodeGen for 'omp taskwait' is not supported yet.");
762 }
763 
764 void CodeGenFunction::EmitOMPFlushDirective(const OMPFlushDirective &S) {
765   CGM.getOpenMPRuntime().emitFlush(*this, [&]() -> ArrayRef<const Expr *> {
766     if (auto C = S.getSingleClause(/*K*/ OMPC_flush)) {
767       auto FlushClause = cast<OMPFlushClause>(C);
768       return llvm::makeArrayRef(FlushClause->varlist_begin(),
769                                 FlushClause->varlist_end());
770     }
771     return llvm::None;
772   }(), S.getLocStart());
773 }
774 
775 void CodeGenFunction::EmitOMPOrderedDirective(const OMPOrderedDirective &) {
776   llvm_unreachable("CodeGen for 'omp ordered' is not supported yet.");
777 }
778 
779 static llvm::Value *convertToScalarValue(CodeGenFunction &CGF, RValue Val,
780                                          QualType SrcType, QualType DestType) {
781   assert(CGF.hasScalarEvaluationKind(DestType) &&
782          "DestType must have scalar evaluation kind.");
783   assert(!Val.isAggregate() && "Must be a scalar or complex.");
784   return Val.isScalar()
785              ? CGF.EmitScalarConversion(Val.getScalarVal(), SrcType, DestType)
786              : CGF.EmitComplexToScalarConversion(Val.getComplexVal(), SrcType,
787                                                  DestType);
788 }
789 
790 static CodeGenFunction::ComplexPairTy
791 convertToComplexValue(CodeGenFunction &CGF, RValue Val, QualType SrcType,
792                       QualType DestType) {
793   assert(CGF.getEvaluationKind(DestType) == TEK_Complex &&
794          "DestType must have complex evaluation kind.");
795   CodeGenFunction::ComplexPairTy ComplexVal;
796   if (Val.isScalar()) {
797     // Convert the input element to the element type of the complex.
798     auto DestElementType = DestType->castAs<ComplexType>()->getElementType();
799     auto ScalarVal =
800         CGF.EmitScalarConversion(Val.getScalarVal(), SrcType, DestElementType);
801     ComplexVal = CodeGenFunction::ComplexPairTy(
802         ScalarVal, llvm::Constant::getNullValue(ScalarVal->getType()));
803   } else {
804     assert(Val.isComplex() && "Must be a scalar or complex.");
805     auto SrcElementType = SrcType->castAs<ComplexType>()->getElementType();
806     auto DestElementType = DestType->castAs<ComplexType>()->getElementType();
807     ComplexVal.first = CGF.EmitScalarConversion(
808         Val.getComplexVal().first, SrcElementType, DestElementType);
809     ComplexVal.second = CGF.EmitScalarConversion(
810         Val.getComplexVal().second, SrcElementType, DestElementType);
811   }
812   return ComplexVal;
813 }
814 
815 static void EmitOMPAtomicReadExpr(CodeGenFunction &CGF, bool IsSeqCst,
816                                   const Expr *X, const Expr *V,
817                                   SourceLocation Loc) {
818   // v = x;
819   assert(V->isLValue() && "V of 'omp atomic read' is not lvalue");
820   assert(X->isLValue() && "X of 'omp atomic read' is not lvalue");
821   LValue XLValue = CGF.EmitLValue(X);
822   LValue VLValue = CGF.EmitLValue(V);
823   RValue Res = XLValue.isGlobalReg()
824                    ? CGF.EmitLoadOfLValue(XLValue, Loc)
825                    : CGF.EmitAtomicLoad(XLValue, Loc,
826                                         IsSeqCst ? llvm::SequentiallyConsistent
827                                                  : llvm::Monotonic,
828                                         XLValue.isVolatile());
829   // OpenMP, 2.12.6, atomic Construct
830   // Any atomic construct with a seq_cst clause forces the atomically
831   // performed operation to include an implicit flush operation without a
832   // list.
833   if (IsSeqCst)
834     CGF.CGM.getOpenMPRuntime().emitFlush(CGF, llvm::None, Loc);
835   switch (CGF.getEvaluationKind(V->getType())) {
836   case TEK_Scalar:
837     CGF.EmitStoreOfScalar(
838         convertToScalarValue(CGF, Res, X->getType(), V->getType()), VLValue);
839     break;
840   case TEK_Complex:
841     CGF.EmitStoreOfComplex(
842         convertToComplexValue(CGF, Res, X->getType(), V->getType()), VLValue,
843         /*isInit=*/false);
844     break;
845   case TEK_Aggregate:
846     llvm_unreachable("Must be a scalar or complex.");
847   }
848 }
849 
850 static void EmitOMPAtomicWriteExpr(CodeGenFunction &CGF, bool IsSeqCst,
851                                    const Expr *X, const Expr *E,
852                                    SourceLocation Loc) {
853   // x = expr;
854   assert(X->isLValue() && "X of 'omp atomic write' is not lvalue");
855   LValue XLValue = CGF.EmitLValue(X);
856   RValue ExprRValue = CGF.EmitAnyExpr(E);
857   if (XLValue.isGlobalReg())
858     CGF.EmitStoreThroughGlobalRegLValue(ExprRValue, XLValue);
859   else
860     CGF.EmitAtomicStore(ExprRValue, XLValue,
861                         IsSeqCst ? llvm::SequentiallyConsistent
862                                  : llvm::Monotonic,
863                         XLValue.isVolatile(), /*IsInit=*/false);
864   // OpenMP, 2.12.6, atomic Construct
865   // Any atomic construct with a seq_cst clause forces the atomically
866   // performed operation to include an implicit flush operation without a
867   // list.
868   if (IsSeqCst)
869     CGF.CGM.getOpenMPRuntime().emitFlush(CGF, llvm::None, Loc);
870 }
871 
872 static void EmitOMPAtomicExpr(CodeGenFunction &CGF, OpenMPClauseKind Kind,
873                               bool IsSeqCst, const Expr *X, const Expr *V,
874                               const Expr *E, SourceLocation Loc) {
875   switch (Kind) {
876   case OMPC_read:
877     EmitOMPAtomicReadExpr(CGF, IsSeqCst, X, V, Loc);
878     break;
879   case OMPC_write:
880     EmitOMPAtomicWriteExpr(CGF, IsSeqCst, X, E, Loc);
881     break;
882   case OMPC_update:
883   case OMPC_capture:
884     llvm_unreachable("CodeGen for 'omp atomic clause' is not supported yet.");
885   case OMPC_if:
886   case OMPC_final:
887   case OMPC_num_threads:
888   case OMPC_private:
889   case OMPC_firstprivate:
890   case OMPC_lastprivate:
891   case OMPC_reduction:
892   case OMPC_safelen:
893   case OMPC_collapse:
894   case OMPC_default:
895   case OMPC_seq_cst:
896   case OMPC_shared:
897   case OMPC_linear:
898   case OMPC_aligned:
899   case OMPC_copyin:
900   case OMPC_copyprivate:
901   case OMPC_flush:
902   case OMPC_proc_bind:
903   case OMPC_schedule:
904   case OMPC_ordered:
905   case OMPC_nowait:
906   case OMPC_untied:
907   case OMPC_threadprivate:
908   case OMPC_mergeable:
909   case OMPC_unknown:
910     llvm_unreachable("Clause is not allowed in 'omp atomic'.");
911   }
912 }
913 
914 void CodeGenFunction::EmitOMPAtomicDirective(const OMPAtomicDirective &S) {
915   bool IsSeqCst = S.getSingleClause(/*K=*/OMPC_seq_cst);
916   OpenMPClauseKind Kind = OMPC_unknown;
917   for (auto *C : S.clauses()) {
918     // Find first clause (skip seq_cst clause, if it is first).
919     if (C->getClauseKind() != OMPC_seq_cst) {
920       Kind = C->getClauseKind();
921       break;
922     }
923   }
924   InlinedOpenMPRegionScopeRAII Region(*this, S);
925   EmitOMPAtomicExpr(*this, Kind, IsSeqCst, S.getX(), S.getV(), S.getExpr(),
926                     S.getLocStart());
927 }
928 
929 void CodeGenFunction::EmitOMPTargetDirective(const OMPTargetDirective &) {
930   llvm_unreachable("CodeGen for 'omp target' is not supported yet.");
931 }
932 
933 void CodeGenFunction::EmitOMPTeamsDirective(const OMPTeamsDirective &) {
934   llvm_unreachable("CodeGen for 'omp teams' is not supported yet.");
935 }
936 
937