1 //===--- CGStmtOpenMP.cpp - Emit LLVM Code from Statements ----------------===// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //===----------------------------------------------------------------------===// 9 // 10 // This contains code to emit OpenMP nodes as LLVM code. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "CGOpenMPRuntime.h" 15 #include "CodeGenFunction.h" 16 #include "CodeGenModule.h" 17 #include "TargetInfo.h" 18 #include "clang/AST/Stmt.h" 19 #include "clang/AST/StmtOpenMP.h" 20 using namespace clang; 21 using namespace CodeGen; 22 23 //===----------------------------------------------------------------------===// 24 // OpenMP Directive Emission 25 //===----------------------------------------------------------------------===// 26 namespace { 27 /// \brief RAII for inlined OpenMP regions (like 'omp for', 'omp simd', 'omp 28 /// critical' etc.). Helps to generate proper debug info and provides correct 29 /// code generation for such constructs. 30 class InlinedOpenMPRegionScopeRAII { 31 InlinedOpenMPRegionRAII Region; 32 CodeGenFunction::LexicalScope DirectiveScope; 33 34 public: 35 InlinedOpenMPRegionScopeRAII(CodeGenFunction &CGF, 36 const OMPExecutableDirective &D) 37 : Region(CGF, D), DirectiveScope(CGF, D.getSourceRange()) {} 38 }; 39 } // namespace 40 41 /// \brief Emits code for OpenMP 'if' clause using specified \a CodeGen 42 /// function. Here is the logic: 43 /// if (Cond) { 44 /// CodeGen(true); 45 /// } else { 46 /// CodeGen(false); 47 /// } 48 static void EmitOMPIfClause(CodeGenFunction &CGF, const Expr *Cond, 49 const std::function<void(bool)> &CodeGen) { 50 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 51 52 // If the condition constant folds and can be elided, try to avoid emitting 53 // the condition and the dead arm of the if/else. 54 bool CondConstant; 55 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 56 CodeGen(CondConstant); 57 return; 58 } 59 60 // Otherwise, the condition did not fold, or we couldn't elide it. Just 61 // emit the conditional branch. 62 auto ThenBlock = CGF.createBasicBlock(/*name*/ "omp_if.then"); 63 auto ElseBlock = CGF.createBasicBlock(/*name*/ "omp_if.else"); 64 auto ContBlock = CGF.createBasicBlock(/*name*/ "omp_if.end"); 65 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount*/ 0); 66 67 // Emit the 'then' code. 68 CGF.EmitBlock(ThenBlock); 69 CodeGen(/*ThenBlock*/ true); 70 CGF.EmitBranch(ContBlock); 71 // Emit the 'else' code if present. 72 { 73 // There is no need to emit line number for unconditional branch. 74 auto NL = ApplyDebugLocation::CreateEmpty(CGF); 75 CGF.EmitBlock(ElseBlock); 76 } 77 CodeGen(/*ThenBlock*/ false); 78 { 79 // There is no need to emit line number for unconditional branch. 80 auto NL = ApplyDebugLocation::CreateEmpty(CGF); 81 CGF.EmitBranch(ContBlock); 82 } 83 // Emit the continuation block for code after the if. 84 CGF.EmitBlock(ContBlock, /*IsFinished*/ true); 85 } 86 87 void CodeGenFunction::EmitOMPAggregateAssign(LValue OriginalAddr, 88 llvm::Value *PrivateAddr, 89 const Expr *AssignExpr, 90 QualType OriginalType, 91 const VarDecl *VDInit) { 92 EmitBlock(createBasicBlock(".omp.assign.begin.")); 93 if (!isa<CXXConstructExpr>(AssignExpr) || isTrivialInitializer(AssignExpr)) { 94 // Perform simple memcpy. 95 EmitAggregateAssign(PrivateAddr, OriginalAddr.getAddress(), 96 AssignExpr->getType()); 97 } else { 98 // Perform element-by-element initialization. 99 QualType ElementTy; 100 auto SrcBegin = OriginalAddr.getAddress(); 101 auto DestBegin = PrivateAddr; 102 auto ArrayTy = OriginalType->getAsArrayTypeUnsafe(); 103 auto SrcNumElements = emitArrayLength(ArrayTy, ElementTy, SrcBegin); 104 auto DestNumElements = emitArrayLength(ArrayTy, ElementTy, DestBegin); 105 auto SrcEnd = Builder.CreateGEP(SrcBegin, SrcNumElements); 106 auto DestEnd = Builder.CreateGEP(DestBegin, DestNumElements); 107 // The basic structure here is a do-while loop, because we don't 108 // need to check for the zero-element case. 109 auto BodyBB = createBasicBlock("omp.arraycpy.body"); 110 auto DoneBB = createBasicBlock("omp.arraycpy.done"); 111 auto IsEmpty = 112 Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arraycpy.isempty"); 113 Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 114 115 // Enter the loop body, making that address the current address. 116 auto EntryBB = Builder.GetInsertBlock(); 117 EmitBlock(BodyBB); 118 auto SrcElementPast = Builder.CreatePHI(SrcBegin->getType(), 2, 119 "omp.arraycpy.srcElementPast"); 120 SrcElementPast->addIncoming(SrcEnd, EntryBB); 121 auto DestElementPast = Builder.CreatePHI(DestBegin->getType(), 2, 122 "omp.arraycpy.destElementPast"); 123 DestElementPast->addIncoming(DestEnd, EntryBB); 124 125 // Shift the address back by one element. 126 auto NegativeOne = llvm::ConstantInt::get(SizeTy, -1, true); 127 auto DestElement = Builder.CreateGEP(DestElementPast, NegativeOne, 128 "omp.arraycpy.dest.element"); 129 auto SrcElement = Builder.CreateGEP(SrcElementPast, NegativeOne, 130 "omp.arraycpy.src.element"); 131 { 132 // Create RunCleanScope to cleanup possible temps. 133 CodeGenFunction::RunCleanupsScope Init(*this); 134 // Emit initialization for single element. 135 LocalDeclMap[VDInit] = SrcElement; 136 EmitAnyExprToMem(AssignExpr, DestElement, 137 AssignExpr->getType().getQualifiers(), 138 /*IsInitializer*/ false); 139 LocalDeclMap.erase(VDInit); 140 } 141 142 // Check whether we've reached the end. 143 auto Done = 144 Builder.CreateICmpEQ(DestElement, DestBegin, "omp.arraycpy.done"); 145 Builder.CreateCondBr(Done, DoneBB, BodyBB); 146 DestElementPast->addIncoming(DestElement, Builder.GetInsertBlock()); 147 SrcElementPast->addIncoming(SrcElement, Builder.GetInsertBlock()); 148 149 // Done. 150 EmitBlock(DoneBB, true); 151 } 152 EmitBlock(createBasicBlock(".omp.assign.end.")); 153 } 154 155 void CodeGenFunction::EmitOMPFirstprivateClause( 156 const OMPExecutableDirective &D, 157 CodeGenFunction::OMPPrivateScope &PrivateScope) { 158 auto PrivateFilter = [](const OMPClause *C) -> bool { 159 return C->getClauseKind() == OMPC_firstprivate; 160 }; 161 for (OMPExecutableDirective::filtered_clause_iterator<decltype(PrivateFilter)> 162 I(D.clauses(), PrivateFilter); I; ++I) { 163 auto *C = cast<OMPFirstprivateClause>(*I); 164 auto IRef = C->varlist_begin(); 165 auto InitsRef = C->inits().begin(); 166 for (auto IInit : C->private_copies()) { 167 auto *OrigVD = cast<VarDecl>(cast<DeclRefExpr>(*IRef)->getDecl()); 168 auto *VD = cast<VarDecl>(cast<DeclRefExpr>(IInit)->getDecl()); 169 bool IsRegistered; 170 if (*InitsRef != nullptr) { 171 // Emit VarDecl with copy init for arrays. 172 auto *FD = CapturedStmtInfo->lookup(OrigVD); 173 LValue Base = MakeNaturalAlignAddrLValue( 174 CapturedStmtInfo->getContextValue(), 175 getContext().getTagDeclType(FD->getParent())); 176 auto OriginalAddr = EmitLValueForField(Base, FD); 177 auto VDInit = cast<VarDecl>(cast<DeclRefExpr>(*InitsRef)->getDecl()); 178 IsRegistered = PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * { 179 auto Emission = EmitAutoVarAlloca(*VD); 180 // Emit initialization of aggregate firstprivate vars. 181 EmitOMPAggregateAssign(OriginalAddr, Emission.getAllocatedAddress(), 182 VD->getInit(), (*IRef)->getType(), VDInit); 183 EmitAutoVarCleanups(Emission); 184 return Emission.getAllocatedAddress(); 185 }); 186 } else 187 IsRegistered = PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * { 188 // Emit private VarDecl with copy init. 189 EmitDecl(*VD); 190 return GetAddrOfLocalVar(VD); 191 }); 192 assert(IsRegistered && "counter already registered as private"); 193 // Silence the warning about unused variable. 194 (void)IsRegistered; 195 ++IRef, ++InitsRef; 196 } 197 } 198 } 199 200 void CodeGenFunction::EmitOMPPrivateClause( 201 const OMPExecutableDirective &D, 202 CodeGenFunction::OMPPrivateScope &PrivateScope) { 203 auto PrivateFilter = [](const OMPClause *C) -> bool { 204 return C->getClauseKind() == OMPC_private; 205 }; 206 for (OMPExecutableDirective::filtered_clause_iterator<decltype(PrivateFilter)> 207 I(D.clauses(), PrivateFilter); I; ++I) { 208 auto *C = cast<OMPPrivateClause>(*I); 209 auto IRef = C->varlist_begin(); 210 for (auto IInit : C->private_copies()) { 211 auto *OrigVD = cast<VarDecl>(cast<DeclRefExpr>(*IRef)->getDecl()); 212 auto VD = cast<VarDecl>(cast<DeclRefExpr>(IInit)->getDecl()); 213 bool IsRegistered = 214 PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * { 215 // Emit private VarDecl with copy init. 216 EmitDecl(*VD); 217 return GetAddrOfLocalVar(VD); 218 }); 219 assert(IsRegistered && "counter already registered as private"); 220 // Silence the warning about unused variable. 221 (void)IsRegistered; 222 ++IRef; 223 } 224 } 225 } 226 227 /// \brief Emits code for OpenMP parallel directive in the parallel region. 228 static void EmitOMPParallelCall(CodeGenFunction &CGF, 229 const OMPParallelDirective &S, 230 llvm::Value *OutlinedFn, 231 llvm::Value *CapturedStruct) { 232 if (auto C = S.getSingleClause(/*K*/ OMPC_num_threads)) { 233 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 234 auto NumThreadsClause = cast<OMPNumThreadsClause>(C); 235 auto NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads(), 236 /*IgnoreResultAssign*/ true); 237 CGF.CGM.getOpenMPRuntime().emitNumThreadsClause( 238 CGF, NumThreads, NumThreadsClause->getLocStart()); 239 } 240 CGF.CGM.getOpenMPRuntime().emitParallelCall(CGF, S.getLocStart(), OutlinedFn, 241 CapturedStruct); 242 } 243 244 void CodeGenFunction::EmitOMPParallelDirective(const OMPParallelDirective &S) { 245 auto CS = cast<CapturedStmt>(S.getAssociatedStmt()); 246 auto CapturedStruct = GenerateCapturedStmtArgument(*CS); 247 auto OutlinedFn = CGM.getOpenMPRuntime().emitOutlinedFunction( 248 S, *CS->getCapturedDecl()->param_begin()); 249 if (auto C = S.getSingleClause(/*K*/ OMPC_if)) { 250 auto Cond = cast<OMPIfClause>(C)->getCondition(); 251 EmitOMPIfClause(*this, Cond, [&](bool ThenBlock) { 252 if (ThenBlock) 253 EmitOMPParallelCall(*this, S, OutlinedFn, CapturedStruct); 254 else 255 CGM.getOpenMPRuntime().emitSerialCall(*this, S.getLocStart(), 256 OutlinedFn, CapturedStruct); 257 }); 258 } else 259 EmitOMPParallelCall(*this, S, OutlinedFn, CapturedStruct); 260 } 261 262 void CodeGenFunction::EmitOMPLoopBody(const OMPLoopDirective &S, 263 bool SeparateIter) { 264 RunCleanupsScope BodyScope(*this); 265 // Update counters values on current iteration. 266 for (auto I : S.updates()) { 267 EmitIgnoredExpr(I); 268 } 269 // On a continue in the body, jump to the end. 270 auto Continue = getJumpDestInCurrentScope("omp.body.continue"); 271 BreakContinueStack.push_back(BreakContinue(JumpDest(), Continue)); 272 // Emit loop body. 273 EmitStmt(S.getBody()); 274 // The end (updates/cleanups). 275 EmitBlock(Continue.getBlock()); 276 BreakContinueStack.pop_back(); 277 if (SeparateIter) { 278 // TODO: Update lastprivates if the SeparateIter flag is true. 279 // This will be implemented in a follow-up OMPLastprivateClause patch, but 280 // result should be still correct without it, as we do not make these 281 // variables private yet. 282 } 283 } 284 285 void CodeGenFunction::EmitOMPInnerLoop(const OMPLoopDirective &S, 286 OMPPrivateScope &LoopScope, 287 bool SeparateIter) { 288 auto LoopExit = getJumpDestInCurrentScope("omp.inner.for.end"); 289 auto Cnt = getPGORegionCounter(&S); 290 291 // Start the loop with a block that tests the condition. 292 auto CondBlock = createBasicBlock("omp.inner.for.cond"); 293 EmitBlock(CondBlock); 294 LoopStack.push(CondBlock); 295 296 // If there are any cleanups between here and the loop-exit scope, 297 // create a block to stage a loop exit along. 298 auto ExitBlock = LoopExit.getBlock(); 299 if (LoopScope.requiresCleanups()) 300 ExitBlock = createBasicBlock("omp.inner.for.cond.cleanup"); 301 302 auto LoopBody = createBasicBlock("omp.inner.for.body"); 303 304 // Emit condition: "IV < LastIteration + 1 [ - 1]" 305 // ("- 1" when lastprivate clause is present - separate one iteration). 306 llvm::Value *BoolCondVal = EvaluateExprAsBool(S.getCond(SeparateIter)); 307 Builder.CreateCondBr(BoolCondVal, LoopBody, ExitBlock, 308 PGO.createLoopWeights(S.getCond(SeparateIter), Cnt)); 309 310 if (ExitBlock != LoopExit.getBlock()) { 311 EmitBlock(ExitBlock); 312 EmitBranchThroughCleanup(LoopExit); 313 } 314 315 EmitBlock(LoopBody); 316 Cnt.beginRegion(Builder); 317 318 // Create a block for the increment. 319 auto Continue = getJumpDestInCurrentScope("omp.inner.for.inc"); 320 BreakContinueStack.push_back(BreakContinue(LoopExit, Continue)); 321 322 EmitOMPLoopBody(S); 323 EmitStopPoint(&S); 324 325 // Emit "IV = IV + 1" and a back-edge to the condition block. 326 EmitBlock(Continue.getBlock()); 327 EmitIgnoredExpr(S.getInc()); 328 BreakContinueStack.pop_back(); 329 EmitBranch(CondBlock); 330 LoopStack.pop(); 331 // Emit the fall-through block. 332 EmitBlock(LoopExit.getBlock()); 333 } 334 335 void CodeGenFunction::EmitOMPSimdFinal(const OMPLoopDirective &S) { 336 auto IC = S.counters().begin(); 337 for (auto F : S.finals()) { 338 if (LocalDeclMap.lookup(cast<DeclRefExpr>((*IC))->getDecl())) { 339 EmitIgnoredExpr(F); 340 } 341 ++IC; 342 } 343 } 344 345 static void EmitOMPAlignedClause(CodeGenFunction &CGF, CodeGenModule &CGM, 346 const OMPAlignedClause &Clause) { 347 unsigned ClauseAlignment = 0; 348 if (auto AlignmentExpr = Clause.getAlignment()) { 349 auto AlignmentCI = 350 cast<llvm::ConstantInt>(CGF.EmitScalarExpr(AlignmentExpr)); 351 ClauseAlignment = static_cast<unsigned>(AlignmentCI->getZExtValue()); 352 } 353 for (auto E : Clause.varlists()) { 354 unsigned Alignment = ClauseAlignment; 355 if (Alignment == 0) { 356 // OpenMP [2.8.1, Description] 357 // If no optional parameter is specified, implementation-defined default 358 // alignments for SIMD instructions on the target platforms are assumed. 359 Alignment = CGM.getTargetCodeGenInfo().getOpenMPSimdDefaultAlignment( 360 E->getType()); 361 } 362 assert((Alignment == 0 || llvm::isPowerOf2_32(Alignment)) && 363 "alignment is not power of 2"); 364 if (Alignment != 0) { 365 llvm::Value *PtrValue = CGF.EmitScalarExpr(E); 366 CGF.EmitAlignmentAssumption(PtrValue, Alignment); 367 } 368 } 369 } 370 371 static void EmitPrivateLoopCounters(CodeGenFunction &CGF, 372 CodeGenFunction::OMPPrivateScope &LoopScope, 373 ArrayRef<Expr *> Counters) { 374 for (auto *E : Counters) { 375 auto VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 376 bool IsRegistered = LoopScope.addPrivate(VD, [&]() -> llvm::Value * { 377 // Emit var without initialization. 378 auto VarEmission = CGF.EmitAutoVarAlloca(*VD); 379 CGF.EmitAutoVarCleanups(VarEmission); 380 return VarEmission.getAllocatedAddress(); 381 }); 382 assert(IsRegistered && "counter already registered as private"); 383 // Silence the warning about unused variable. 384 (void)IsRegistered; 385 } 386 (void)LoopScope.Privatize(); 387 } 388 389 void CodeGenFunction::EmitOMPSimdDirective(const OMPSimdDirective &S) { 390 // Pragma 'simd' code depends on presence of 'lastprivate'. 391 // If present, we have to separate last iteration of the loop: 392 // 393 // if (LastIteration != 0) { 394 // for (IV in 0..LastIteration-1) BODY; 395 // BODY with updates of lastprivate vars; 396 // <Final counter/linear vars updates>; 397 // } 398 // 399 // otherwise (when there's no lastprivate): 400 // 401 // for (IV in 0..LastIteration) BODY; 402 // <Final counter/linear vars updates>; 403 // 404 405 // Walk clauses and process safelen/lastprivate. 406 bool SeparateIter = false; 407 LoopStack.setParallel(); 408 LoopStack.setVectorizerEnable(true); 409 for (auto C : S.clauses()) { 410 switch (C->getClauseKind()) { 411 case OMPC_safelen: { 412 RValue Len = EmitAnyExpr(cast<OMPSafelenClause>(C)->getSafelen(), 413 AggValueSlot::ignored(), true); 414 llvm::ConstantInt *Val = cast<llvm::ConstantInt>(Len.getScalarVal()); 415 LoopStack.setVectorizerWidth(Val->getZExtValue()); 416 // In presence of finite 'safelen', it may be unsafe to mark all 417 // the memory instructions parallel, because loop-carried 418 // dependences of 'safelen' iterations are possible. 419 LoopStack.setParallel(false); 420 break; 421 } 422 case OMPC_aligned: 423 EmitOMPAlignedClause(*this, CGM, cast<OMPAlignedClause>(*C)); 424 break; 425 case OMPC_lastprivate: 426 SeparateIter = true; 427 break; 428 default: 429 // Not handled yet 430 ; 431 } 432 } 433 434 InlinedOpenMPRegionScopeRAII Region(*this, S); 435 436 // Emit the loop iteration variable. 437 const Expr *IVExpr = S.getIterationVariable(); 438 const VarDecl *IVDecl = cast<VarDecl>(cast<DeclRefExpr>(IVExpr)->getDecl()); 439 EmitVarDecl(*IVDecl); 440 EmitIgnoredExpr(S.getInit()); 441 442 // Emit the iterations count variable. 443 // If it is not a variable, Sema decided to calculate iterations count on each 444 // iteration (e.g., it is foldable into a constant). 445 if (auto LIExpr = dyn_cast<DeclRefExpr>(S.getLastIteration())) { 446 EmitVarDecl(*cast<VarDecl>(LIExpr->getDecl())); 447 // Emit calculation of the iterations count. 448 EmitIgnoredExpr(S.getCalcLastIteration()); 449 } 450 451 if (SeparateIter) { 452 // Emit: if (LastIteration > 0) - begin. 453 RegionCounter Cnt = getPGORegionCounter(&S); 454 auto ThenBlock = createBasicBlock("simd.if.then"); 455 auto ContBlock = createBasicBlock("simd.if.end"); 456 EmitBranchOnBoolExpr(S.getPreCond(), ThenBlock, ContBlock, Cnt.getCount()); 457 EmitBlock(ThenBlock); 458 Cnt.beginRegion(Builder); 459 // Emit 'then' code. 460 { 461 OMPPrivateScope LoopScope(*this); 462 EmitPrivateLoopCounters(*this, LoopScope, S.counters()); 463 EmitOMPInnerLoop(S, LoopScope, /* SeparateIter */ true); 464 EmitOMPLoopBody(S, /* SeparateIter */ true); 465 } 466 EmitOMPSimdFinal(S); 467 // Emit: if (LastIteration != 0) - end. 468 EmitBranch(ContBlock); 469 EmitBlock(ContBlock, true); 470 } else { 471 { 472 OMPPrivateScope LoopScope(*this); 473 EmitPrivateLoopCounters(*this, LoopScope, S.counters()); 474 EmitOMPInnerLoop(S, LoopScope); 475 } 476 EmitOMPSimdFinal(S); 477 } 478 } 479 480 void CodeGenFunction::EmitOMPForOuterLoop(OpenMPScheduleClauseKind ScheduleKind, 481 const OMPLoopDirective &S, 482 OMPPrivateScope &LoopScope, 483 llvm::Value *LB, llvm::Value *UB, 484 llvm::Value *ST, llvm::Value *IL, 485 llvm::Value *Chunk) { 486 auto &RT = CGM.getOpenMPRuntime(); 487 assert(!RT.isStaticNonchunked(ScheduleKind, /* Chunked */ Chunk != nullptr) && 488 "static non-chunked schedule does not need outer loop"); 489 if (RT.isDynamic(ScheduleKind)) { 490 ErrorUnsupported(&S, "OpenMP loop with dynamic schedule"); 491 return; 492 } 493 494 // Emit outer loop. 495 // 496 // OpenMP [2.7.1, Loop Construct, Description, table 2-1] 497 // When schedule(static, chunk_size) is specified, iterations are divided into 498 // chunks of size chunk_size, and the chunks are assigned to the threads in 499 // the team in a round-robin fashion in the order of the thread number. 500 // 501 // while(UB = min(UB, GlobalUB), idx = LB, idx < UB) { 502 // while (idx <= UB) { BODY; ++idx; } // inner loop 503 // LB = LB + ST; 504 // UB = UB + ST; 505 // } 506 // 507 const Expr *IVExpr = S.getIterationVariable(); 508 const unsigned IVSize = getContext().getTypeSize(IVExpr->getType()); 509 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation(); 510 511 RT.emitForInit(*this, S.getLocStart(), ScheduleKind, IVSize, IVSigned, IL, LB, 512 UB, ST, Chunk); 513 auto LoopExit = getJumpDestInCurrentScope("omp.dispatch.end"); 514 515 // Start the loop with a block that tests the condition. 516 auto CondBlock = createBasicBlock("omp.dispatch.cond"); 517 EmitBlock(CondBlock); 518 LoopStack.push(CondBlock); 519 520 llvm::Value *BoolCondVal = nullptr; 521 // UB = min(UB, GlobalUB) 522 EmitIgnoredExpr(S.getEnsureUpperBound()); 523 // IV = LB 524 EmitIgnoredExpr(S.getInit()); 525 // IV < UB 526 BoolCondVal = EvaluateExprAsBool(S.getCond(false)); 527 528 // If there are any cleanups between here and the loop-exit scope, 529 // create a block to stage a loop exit along. 530 auto ExitBlock = LoopExit.getBlock(); 531 if (LoopScope.requiresCleanups()) 532 ExitBlock = createBasicBlock("omp.dispatch.cleanup"); 533 534 auto LoopBody = createBasicBlock("omp.dispatch.body"); 535 Builder.CreateCondBr(BoolCondVal, LoopBody, ExitBlock); 536 if (ExitBlock != LoopExit.getBlock()) { 537 EmitBlock(ExitBlock); 538 EmitBranchThroughCleanup(LoopExit); 539 } 540 EmitBlock(LoopBody); 541 542 // Create a block for the increment. 543 auto Continue = getJumpDestInCurrentScope("omp.dispatch.inc"); 544 BreakContinueStack.push_back(BreakContinue(LoopExit, Continue)); 545 546 EmitOMPInnerLoop(S, LoopScope); 547 548 EmitBlock(Continue.getBlock()); 549 BreakContinueStack.pop_back(); 550 // Emit "LB = LB + Stride", "UB = UB + Stride". 551 EmitIgnoredExpr(S.getNextLowerBound()); 552 EmitIgnoredExpr(S.getNextUpperBound()); 553 554 EmitBranch(CondBlock); 555 LoopStack.pop(); 556 // Emit the fall-through block. 557 EmitBlock(LoopExit.getBlock()); 558 559 // Tell the runtime we are done. 560 RT.emitForFinish(*this, S.getLocStart(), ScheduleKind); 561 } 562 563 /// \brief Emit a helper variable and return corresponding lvalue. 564 static LValue EmitOMPHelperVar(CodeGenFunction &CGF, 565 const DeclRefExpr *Helper) { 566 auto VDecl = cast<VarDecl>(Helper->getDecl()); 567 CGF.EmitVarDecl(*VDecl); 568 return CGF.EmitLValue(Helper); 569 } 570 571 void CodeGenFunction::EmitOMPWorksharingLoop(const OMPLoopDirective &S) { 572 // Emit the loop iteration variable. 573 auto IVExpr = cast<DeclRefExpr>(S.getIterationVariable()); 574 auto IVDecl = cast<VarDecl>(IVExpr->getDecl()); 575 EmitVarDecl(*IVDecl); 576 577 // Emit the iterations count variable. 578 // If it is not a variable, Sema decided to calculate iterations count on each 579 // iteration (e.g., it is foldable into a constant). 580 if (auto LIExpr = dyn_cast<DeclRefExpr>(S.getLastIteration())) { 581 EmitVarDecl(*cast<VarDecl>(LIExpr->getDecl())); 582 // Emit calculation of the iterations count. 583 EmitIgnoredExpr(S.getCalcLastIteration()); 584 } 585 586 auto &RT = CGM.getOpenMPRuntime(); 587 588 // Check pre-condition. 589 { 590 // Skip the entire loop if we don't meet the precondition. 591 RegionCounter Cnt = getPGORegionCounter(&S); 592 auto ThenBlock = createBasicBlock("omp.precond.then"); 593 auto ContBlock = createBasicBlock("omp.precond.end"); 594 EmitBranchOnBoolExpr(S.getPreCond(), ThenBlock, ContBlock, Cnt.getCount()); 595 EmitBlock(ThenBlock); 596 Cnt.beginRegion(Builder); 597 // Emit 'then' code. 598 { 599 // Emit helper vars inits. 600 LValue LB = 601 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getLowerBoundVariable())); 602 LValue UB = 603 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getUpperBoundVariable())); 604 LValue ST = 605 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getStrideVariable())); 606 LValue IL = 607 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getIsLastIterVariable())); 608 609 OMPPrivateScope LoopScope(*this); 610 EmitPrivateLoopCounters(*this, LoopScope, S.counters()); 611 612 // Detect the loop schedule kind and chunk. 613 auto ScheduleKind = OMPC_SCHEDULE_unknown; 614 llvm::Value *Chunk = nullptr; 615 if (auto C = cast_or_null<OMPScheduleClause>( 616 S.getSingleClause(OMPC_schedule))) { 617 ScheduleKind = C->getScheduleKind(); 618 if (auto Ch = C->getChunkSize()) { 619 Chunk = EmitScalarExpr(Ch); 620 Chunk = EmitScalarConversion(Chunk, Ch->getType(), 621 S.getIterationVariable()->getType()); 622 } 623 } 624 const unsigned IVSize = getContext().getTypeSize(IVExpr->getType()); 625 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation(); 626 if (RT.isStaticNonchunked(ScheduleKind, 627 /* Chunked */ Chunk != nullptr)) { 628 // OpenMP [2.7.1, Loop Construct, Description, table 2-1] 629 // When no chunk_size is specified, the iteration space is divided into 630 // chunks that are approximately equal in size, and at most one chunk is 631 // distributed to each thread. Note that the size of the chunks is 632 // unspecified in this case. 633 RT.emitForInit(*this, S.getLocStart(), ScheduleKind, IVSize, IVSigned, 634 IL.getAddress(), LB.getAddress(), UB.getAddress(), 635 ST.getAddress()); 636 // UB = min(UB, GlobalUB); 637 EmitIgnoredExpr(S.getEnsureUpperBound()); 638 // IV = LB; 639 EmitIgnoredExpr(S.getInit()); 640 // while (idx <= UB) { BODY; ++idx; } 641 EmitOMPInnerLoop(S, LoopScope); 642 // Tell the runtime we are done. 643 RT.emitForFinish(*this, S.getLocStart(), ScheduleKind); 644 } else { 645 // Emit the outer loop, which requests its work chunk [LB..UB] from 646 // runtime and runs the inner loop to process it. 647 EmitOMPForOuterLoop(ScheduleKind, S, LoopScope, LB.getAddress(), 648 UB.getAddress(), ST.getAddress(), IL.getAddress(), 649 Chunk); 650 } 651 } 652 // We're now done with the loop, so jump to the continuation block. 653 EmitBranch(ContBlock); 654 EmitBlock(ContBlock, true); 655 } 656 } 657 658 void CodeGenFunction::EmitOMPForDirective(const OMPForDirective &S) { 659 InlinedOpenMPRegionScopeRAII Region(*this, S); 660 661 EmitOMPWorksharingLoop(S); 662 663 // Emit an implicit barrier at the end. 664 CGM.getOpenMPRuntime().emitBarrierCall(*this, S.getLocStart(), 665 /*IsExplicit*/ false); 666 } 667 668 void CodeGenFunction::EmitOMPForSimdDirective(const OMPForSimdDirective &) { 669 llvm_unreachable("CodeGen for 'omp for simd' is not supported yet."); 670 } 671 672 void CodeGenFunction::EmitOMPSectionsDirective(const OMPSectionsDirective &) { 673 llvm_unreachable("CodeGen for 'omp sections' is not supported yet."); 674 } 675 676 void CodeGenFunction::EmitOMPSectionDirective(const OMPSectionDirective &) { 677 llvm_unreachable("CodeGen for 'omp section' is not supported yet."); 678 } 679 680 void CodeGenFunction::EmitOMPSingleDirective(const OMPSingleDirective &S) { 681 CGM.getOpenMPRuntime().emitSingleRegion(*this, [&]() -> void { 682 InlinedOpenMPRegionScopeRAII Region(*this, S); 683 EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt()); 684 EnsureInsertPoint(); 685 }, S.getLocStart()); 686 } 687 688 void CodeGenFunction::EmitOMPMasterDirective(const OMPMasterDirective &S) { 689 CGM.getOpenMPRuntime().emitMasterRegion(*this, [&]() -> void { 690 InlinedOpenMPRegionScopeRAII Region(*this, S); 691 EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt()); 692 EnsureInsertPoint(); 693 }, S.getLocStart()); 694 } 695 696 void CodeGenFunction::EmitOMPCriticalDirective(const OMPCriticalDirective &S) { 697 CGM.getOpenMPRuntime().emitCriticalRegion( 698 *this, S.getDirectiveName().getAsString(), [&]() -> void { 699 InlinedOpenMPRegionScopeRAII Region(*this, S); 700 EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt()); 701 EnsureInsertPoint(); 702 }, S.getLocStart()); 703 } 704 705 void 706 CodeGenFunction::EmitOMPParallelForDirective(const OMPParallelForDirective &) { 707 llvm_unreachable("CodeGen for 'omp parallel for' is not supported yet."); 708 } 709 710 void CodeGenFunction::EmitOMPParallelForSimdDirective( 711 const OMPParallelForSimdDirective &) { 712 llvm_unreachable("CodeGen for 'omp parallel for simd' is not supported yet."); 713 } 714 715 void CodeGenFunction::EmitOMPParallelSectionsDirective( 716 const OMPParallelSectionsDirective &) { 717 llvm_unreachable("CodeGen for 'omp parallel sections' is not supported yet."); 718 } 719 720 void CodeGenFunction::EmitOMPTaskDirective(const OMPTaskDirective &S) { 721 // Emit outlined function for task construct. 722 auto CS = cast<CapturedStmt>(S.getAssociatedStmt()); 723 auto CapturedStruct = GenerateCapturedStmtArgument(*CS); 724 auto *I = CS->getCapturedDecl()->param_begin(); 725 // The first function argument for tasks is a thread id, the second one is a 726 // part id (0 for tied tasks, >=0 for untied task). 727 auto OutlinedFn = 728 CGM.getOpenMPRuntime().emitTaskOutlinedFunction(S, *I, *std::next(I)); 729 // Check if we should emit tied or untied task. 730 bool Tied = !S.getSingleClause(OMPC_untied); 731 // Check if the task is final 732 llvm::PointerIntPair<llvm::Value *, 1, bool> Final; 733 if (auto *Clause = S.getSingleClause(OMPC_final)) { 734 // If the condition constant folds and can be elided, try to avoid emitting 735 // the condition and the dead arm of the if/else. 736 auto *Cond = cast<OMPFinalClause>(Clause)->getCondition(); 737 bool CondConstant; 738 if (ConstantFoldsToSimpleInteger(Cond, CondConstant)) 739 Final.setInt(CondConstant); 740 else 741 Final.setPointer(EvaluateExprAsBool(Cond)); 742 } else { 743 // By default the task is not final. 744 Final.setInt(/*IntVal=*/false); 745 } 746 auto SharedsTy = getContext().getRecordType(CS->getCapturedRecordDecl()); 747 CGM.getOpenMPRuntime().emitTaskCall(*this, S.getLocStart(), Tied, Final, 748 OutlinedFn, SharedsTy, CapturedStruct); 749 } 750 751 void CodeGenFunction::EmitOMPTaskyieldDirective( 752 const OMPTaskyieldDirective &S) { 753 CGM.getOpenMPRuntime().emitTaskyieldCall(*this, S.getLocStart()); 754 } 755 756 void CodeGenFunction::EmitOMPBarrierDirective(const OMPBarrierDirective &S) { 757 CGM.getOpenMPRuntime().emitBarrierCall(*this, S.getLocStart()); 758 } 759 760 void CodeGenFunction::EmitOMPTaskwaitDirective(const OMPTaskwaitDirective &) { 761 llvm_unreachable("CodeGen for 'omp taskwait' is not supported yet."); 762 } 763 764 void CodeGenFunction::EmitOMPFlushDirective(const OMPFlushDirective &S) { 765 CGM.getOpenMPRuntime().emitFlush(*this, [&]() -> ArrayRef<const Expr *> { 766 if (auto C = S.getSingleClause(/*K*/ OMPC_flush)) { 767 auto FlushClause = cast<OMPFlushClause>(C); 768 return llvm::makeArrayRef(FlushClause->varlist_begin(), 769 FlushClause->varlist_end()); 770 } 771 return llvm::None; 772 }(), S.getLocStart()); 773 } 774 775 void CodeGenFunction::EmitOMPOrderedDirective(const OMPOrderedDirective &) { 776 llvm_unreachable("CodeGen for 'omp ordered' is not supported yet."); 777 } 778 779 static llvm::Value *convertToScalarValue(CodeGenFunction &CGF, RValue Val, 780 QualType SrcType, QualType DestType) { 781 assert(CGF.hasScalarEvaluationKind(DestType) && 782 "DestType must have scalar evaluation kind."); 783 assert(!Val.isAggregate() && "Must be a scalar or complex."); 784 return Val.isScalar() 785 ? CGF.EmitScalarConversion(Val.getScalarVal(), SrcType, DestType) 786 : CGF.EmitComplexToScalarConversion(Val.getComplexVal(), SrcType, 787 DestType); 788 } 789 790 static CodeGenFunction::ComplexPairTy 791 convertToComplexValue(CodeGenFunction &CGF, RValue Val, QualType SrcType, 792 QualType DestType) { 793 assert(CGF.getEvaluationKind(DestType) == TEK_Complex && 794 "DestType must have complex evaluation kind."); 795 CodeGenFunction::ComplexPairTy ComplexVal; 796 if (Val.isScalar()) { 797 // Convert the input element to the element type of the complex. 798 auto DestElementType = DestType->castAs<ComplexType>()->getElementType(); 799 auto ScalarVal = 800 CGF.EmitScalarConversion(Val.getScalarVal(), SrcType, DestElementType); 801 ComplexVal = CodeGenFunction::ComplexPairTy( 802 ScalarVal, llvm::Constant::getNullValue(ScalarVal->getType())); 803 } else { 804 assert(Val.isComplex() && "Must be a scalar or complex."); 805 auto SrcElementType = SrcType->castAs<ComplexType>()->getElementType(); 806 auto DestElementType = DestType->castAs<ComplexType>()->getElementType(); 807 ComplexVal.first = CGF.EmitScalarConversion( 808 Val.getComplexVal().first, SrcElementType, DestElementType); 809 ComplexVal.second = CGF.EmitScalarConversion( 810 Val.getComplexVal().second, SrcElementType, DestElementType); 811 } 812 return ComplexVal; 813 } 814 815 static void EmitOMPAtomicReadExpr(CodeGenFunction &CGF, bool IsSeqCst, 816 const Expr *X, const Expr *V, 817 SourceLocation Loc) { 818 // v = x; 819 assert(V->isLValue() && "V of 'omp atomic read' is not lvalue"); 820 assert(X->isLValue() && "X of 'omp atomic read' is not lvalue"); 821 LValue XLValue = CGF.EmitLValue(X); 822 LValue VLValue = CGF.EmitLValue(V); 823 RValue Res = XLValue.isGlobalReg() 824 ? CGF.EmitLoadOfLValue(XLValue, Loc) 825 : CGF.EmitAtomicLoad(XLValue, Loc, 826 IsSeqCst ? llvm::SequentiallyConsistent 827 : llvm::Monotonic, 828 XLValue.isVolatile()); 829 // OpenMP, 2.12.6, atomic Construct 830 // Any atomic construct with a seq_cst clause forces the atomically 831 // performed operation to include an implicit flush operation without a 832 // list. 833 if (IsSeqCst) 834 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, llvm::None, Loc); 835 switch (CGF.getEvaluationKind(V->getType())) { 836 case TEK_Scalar: 837 CGF.EmitStoreOfScalar( 838 convertToScalarValue(CGF, Res, X->getType(), V->getType()), VLValue); 839 break; 840 case TEK_Complex: 841 CGF.EmitStoreOfComplex( 842 convertToComplexValue(CGF, Res, X->getType(), V->getType()), VLValue, 843 /*isInit=*/false); 844 break; 845 case TEK_Aggregate: 846 llvm_unreachable("Must be a scalar or complex."); 847 } 848 } 849 850 static void EmitOMPAtomicWriteExpr(CodeGenFunction &CGF, bool IsSeqCst, 851 const Expr *X, const Expr *E, 852 SourceLocation Loc) { 853 // x = expr; 854 assert(X->isLValue() && "X of 'omp atomic write' is not lvalue"); 855 LValue XLValue = CGF.EmitLValue(X); 856 RValue ExprRValue = CGF.EmitAnyExpr(E); 857 if (XLValue.isGlobalReg()) 858 CGF.EmitStoreThroughGlobalRegLValue(ExprRValue, XLValue); 859 else 860 CGF.EmitAtomicStore(ExprRValue, XLValue, 861 IsSeqCst ? llvm::SequentiallyConsistent 862 : llvm::Monotonic, 863 XLValue.isVolatile(), /*IsInit=*/false); 864 // OpenMP, 2.12.6, atomic Construct 865 // Any atomic construct with a seq_cst clause forces the atomically 866 // performed operation to include an implicit flush operation without a 867 // list. 868 if (IsSeqCst) 869 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, llvm::None, Loc); 870 } 871 872 static void EmitOMPAtomicExpr(CodeGenFunction &CGF, OpenMPClauseKind Kind, 873 bool IsSeqCst, const Expr *X, const Expr *V, 874 const Expr *E, SourceLocation Loc) { 875 switch (Kind) { 876 case OMPC_read: 877 EmitOMPAtomicReadExpr(CGF, IsSeqCst, X, V, Loc); 878 break; 879 case OMPC_write: 880 EmitOMPAtomicWriteExpr(CGF, IsSeqCst, X, E, Loc); 881 break; 882 case OMPC_update: 883 case OMPC_capture: 884 llvm_unreachable("CodeGen for 'omp atomic clause' is not supported yet."); 885 case OMPC_if: 886 case OMPC_final: 887 case OMPC_num_threads: 888 case OMPC_private: 889 case OMPC_firstprivate: 890 case OMPC_lastprivate: 891 case OMPC_reduction: 892 case OMPC_safelen: 893 case OMPC_collapse: 894 case OMPC_default: 895 case OMPC_seq_cst: 896 case OMPC_shared: 897 case OMPC_linear: 898 case OMPC_aligned: 899 case OMPC_copyin: 900 case OMPC_copyprivate: 901 case OMPC_flush: 902 case OMPC_proc_bind: 903 case OMPC_schedule: 904 case OMPC_ordered: 905 case OMPC_nowait: 906 case OMPC_untied: 907 case OMPC_threadprivate: 908 case OMPC_mergeable: 909 case OMPC_unknown: 910 llvm_unreachable("Clause is not allowed in 'omp atomic'."); 911 } 912 } 913 914 void CodeGenFunction::EmitOMPAtomicDirective(const OMPAtomicDirective &S) { 915 bool IsSeqCst = S.getSingleClause(/*K=*/OMPC_seq_cst); 916 OpenMPClauseKind Kind = OMPC_unknown; 917 for (auto *C : S.clauses()) { 918 // Find first clause (skip seq_cst clause, if it is first). 919 if (C->getClauseKind() != OMPC_seq_cst) { 920 Kind = C->getClauseKind(); 921 break; 922 } 923 } 924 InlinedOpenMPRegionScopeRAII Region(*this, S); 925 EmitOMPAtomicExpr(*this, Kind, IsSeqCst, S.getX(), S.getV(), S.getExpr(), 926 S.getLocStart()); 927 } 928 929 void CodeGenFunction::EmitOMPTargetDirective(const OMPTargetDirective &) { 930 llvm_unreachable("CodeGen for 'omp target' is not supported yet."); 931 } 932 933 void CodeGenFunction::EmitOMPTeamsDirective(const OMPTeamsDirective &) { 934 llvm_unreachable("CodeGen for 'omp teams' is not supported yet."); 935 } 936 937