1 //===--- CGStmtOpenMP.cpp - Emit LLVM Code from Statements ----------------===// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //===----------------------------------------------------------------------===// 9 // 10 // This contains code to emit OpenMP nodes as LLVM code. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "CGOpenMPRuntime.h" 15 #include "CodeGenFunction.h" 16 #include "CodeGenModule.h" 17 #include "TargetInfo.h" 18 #include "clang/AST/Stmt.h" 19 #include "clang/AST/StmtOpenMP.h" 20 using namespace clang; 21 using namespace CodeGen; 22 23 //===----------------------------------------------------------------------===// 24 // OpenMP Directive Emission 25 //===----------------------------------------------------------------------===// 26 namespace { 27 /// \brief RAII for inlined OpenMP regions (like 'omp for', 'omp simd', 'omp 28 /// critical' etc.). Helps to generate proper debug info and provides correct 29 /// code generation for such constructs. 30 class InlinedOpenMPRegionScopeRAII { 31 InlinedOpenMPRegionRAII Region; 32 CodeGenFunction::LexicalScope DirectiveScope; 33 34 public: 35 InlinedOpenMPRegionScopeRAII(CodeGenFunction &CGF, 36 const OMPExecutableDirective &D) 37 : Region(CGF, D), DirectiveScope(CGF, D.getSourceRange()) {} 38 }; 39 } // namespace 40 41 /// \brief Emits code for OpenMP 'if' clause using specified \a CodeGen 42 /// function. Here is the logic: 43 /// if (Cond) { 44 /// CodeGen(true); 45 /// } else { 46 /// CodeGen(false); 47 /// } 48 static void EmitOMPIfClause(CodeGenFunction &CGF, const Expr *Cond, 49 const std::function<void(bool)> &CodeGen) { 50 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 51 52 // If the condition constant folds and can be elided, try to avoid emitting 53 // the condition and the dead arm of the if/else. 54 bool CondConstant; 55 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 56 CodeGen(CondConstant); 57 return; 58 } 59 60 // Otherwise, the condition did not fold, or we couldn't elide it. Just 61 // emit the conditional branch. 62 auto ThenBlock = CGF.createBasicBlock(/*name*/ "omp_if.then"); 63 auto ElseBlock = CGF.createBasicBlock(/*name*/ "omp_if.else"); 64 auto ContBlock = CGF.createBasicBlock(/*name*/ "omp_if.end"); 65 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount*/ 0); 66 67 // Emit the 'then' code. 68 CGF.EmitBlock(ThenBlock); 69 CodeGen(/*ThenBlock*/ true); 70 CGF.EmitBranch(ContBlock); 71 // Emit the 'else' code if present. 72 { 73 // There is no need to emit line number for unconditional branch. 74 auto NL = ApplyDebugLocation::CreateEmpty(CGF); 75 CGF.EmitBlock(ElseBlock); 76 } 77 CodeGen(/*ThenBlock*/ false); 78 { 79 // There is no need to emit line number for unconditional branch. 80 auto NL = ApplyDebugLocation::CreateEmpty(CGF); 81 CGF.EmitBranch(ContBlock); 82 } 83 // Emit the continuation block for code after the if. 84 CGF.EmitBlock(ContBlock, /*IsFinished*/ true); 85 } 86 87 void CodeGenFunction::EmitOMPAggregateAssign(LValue OriginalAddr, 88 llvm::Value *PrivateAddr, 89 const Expr *AssignExpr, 90 QualType OriginalType, 91 const VarDecl *VDInit) { 92 EmitBlock(createBasicBlock(".omp.assign.begin.")); 93 if (!isa<CXXConstructExpr>(AssignExpr) || isTrivialInitializer(AssignExpr)) { 94 // Perform simple memcpy. 95 EmitAggregateAssign(PrivateAddr, OriginalAddr.getAddress(), 96 AssignExpr->getType()); 97 } else { 98 // Perform element-by-element initialization. 99 QualType ElementTy; 100 auto SrcBegin = OriginalAddr.getAddress(); 101 auto DestBegin = PrivateAddr; 102 auto ArrayTy = OriginalType->getAsArrayTypeUnsafe(); 103 auto SrcNumElements = emitArrayLength(ArrayTy, ElementTy, SrcBegin); 104 auto DestNumElements = emitArrayLength(ArrayTy, ElementTy, DestBegin); 105 auto SrcEnd = Builder.CreateGEP(SrcBegin, SrcNumElements); 106 auto DestEnd = Builder.CreateGEP(DestBegin, DestNumElements); 107 // The basic structure here is a do-while loop, because we don't 108 // need to check for the zero-element case. 109 auto BodyBB = createBasicBlock("omp.arraycpy.body"); 110 auto DoneBB = createBasicBlock("omp.arraycpy.done"); 111 auto IsEmpty = 112 Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arraycpy.isempty"); 113 Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 114 115 // Enter the loop body, making that address the current address. 116 auto EntryBB = Builder.GetInsertBlock(); 117 EmitBlock(BodyBB); 118 auto SrcElementPast = Builder.CreatePHI(SrcBegin->getType(), 2, 119 "omp.arraycpy.srcElementPast"); 120 SrcElementPast->addIncoming(SrcEnd, EntryBB); 121 auto DestElementPast = Builder.CreatePHI(DestBegin->getType(), 2, 122 "omp.arraycpy.destElementPast"); 123 DestElementPast->addIncoming(DestEnd, EntryBB); 124 125 // Shift the address back by one element. 126 auto NegativeOne = llvm::ConstantInt::get(SizeTy, -1, true); 127 auto DestElement = Builder.CreateGEP(DestElementPast, NegativeOne, 128 "omp.arraycpy.dest.element"); 129 auto SrcElement = Builder.CreateGEP(SrcElementPast, NegativeOne, 130 "omp.arraycpy.src.element"); 131 { 132 // Create RunCleanScope to cleanup possible temps. 133 CodeGenFunction::RunCleanupsScope Init(*this); 134 // Emit initialization for single element. 135 LocalDeclMap[VDInit] = SrcElement; 136 EmitAnyExprToMem(AssignExpr, DestElement, 137 AssignExpr->getType().getQualifiers(), 138 /*IsInitializer*/ false); 139 LocalDeclMap.erase(VDInit); 140 } 141 142 // Check whether we've reached the end. 143 auto Done = 144 Builder.CreateICmpEQ(DestElement, DestBegin, "omp.arraycpy.done"); 145 Builder.CreateCondBr(Done, DoneBB, BodyBB); 146 DestElementPast->addIncoming(DestElement, Builder.GetInsertBlock()); 147 SrcElementPast->addIncoming(SrcElement, Builder.GetInsertBlock()); 148 149 // Done. 150 EmitBlock(DoneBB, true); 151 } 152 EmitBlock(createBasicBlock(".omp.assign.end.")); 153 } 154 155 void CodeGenFunction::EmitOMPFirstprivateClause( 156 const OMPExecutableDirective &D, 157 CodeGenFunction::OMPPrivateScope &PrivateScope) { 158 auto PrivateFilter = [](const OMPClause *C) -> bool { 159 return C->getClauseKind() == OMPC_firstprivate; 160 }; 161 for (OMPExecutableDirective::filtered_clause_iterator<decltype(PrivateFilter)> 162 I(D.clauses(), PrivateFilter); I; ++I) { 163 auto *C = cast<OMPFirstprivateClause>(*I); 164 auto IRef = C->varlist_begin(); 165 auto InitsRef = C->inits().begin(); 166 for (auto IInit : C->private_copies()) { 167 auto *OrigVD = cast<VarDecl>(cast<DeclRefExpr>(*IRef)->getDecl()); 168 auto *VD = cast<VarDecl>(cast<DeclRefExpr>(IInit)->getDecl()); 169 bool IsRegistered; 170 if (*InitsRef != nullptr) { 171 // Emit VarDecl with copy init for arrays. 172 auto *FD = CapturedStmtInfo->lookup(OrigVD); 173 LValue Base = MakeNaturalAlignAddrLValue( 174 CapturedStmtInfo->getContextValue(), 175 getContext().getTagDeclType(FD->getParent())); 176 auto OriginalAddr = EmitLValueForField(Base, FD); 177 auto VDInit = cast<VarDecl>(cast<DeclRefExpr>(*InitsRef)->getDecl()); 178 IsRegistered = PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * { 179 auto Emission = EmitAutoVarAlloca(*VD); 180 // Emit initialization of aggregate firstprivate vars. 181 EmitOMPAggregateAssign(OriginalAddr, Emission.getAllocatedAddress(), 182 VD->getInit(), (*IRef)->getType(), VDInit); 183 EmitAutoVarCleanups(Emission); 184 return Emission.getAllocatedAddress(); 185 }); 186 } else 187 IsRegistered = PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * { 188 // Emit private VarDecl with copy init. 189 EmitDecl(*VD); 190 return GetAddrOfLocalVar(VD); 191 }); 192 assert(IsRegistered && "firstprivate var already registered as private"); 193 // Silence the warning about unused variable. 194 (void)IsRegistered; 195 ++IRef, ++InitsRef; 196 } 197 } 198 } 199 200 void CodeGenFunction::EmitOMPPrivateClause( 201 const OMPExecutableDirective &D, 202 CodeGenFunction::OMPPrivateScope &PrivateScope) { 203 auto PrivateFilter = [](const OMPClause *C) -> bool { 204 return C->getClauseKind() == OMPC_private; 205 }; 206 for (OMPExecutableDirective::filtered_clause_iterator<decltype(PrivateFilter)> 207 I(D.clauses(), PrivateFilter); I; ++I) { 208 auto *C = cast<OMPPrivateClause>(*I); 209 auto IRef = C->varlist_begin(); 210 for (auto IInit : C->private_copies()) { 211 auto *OrigVD = cast<VarDecl>(cast<DeclRefExpr>(*IRef)->getDecl()); 212 auto VD = cast<VarDecl>(cast<DeclRefExpr>(IInit)->getDecl()); 213 bool IsRegistered = 214 PrivateScope.addPrivate(OrigVD, [&]() -> llvm::Value * { 215 // Emit private VarDecl with copy init. 216 EmitDecl(*VD); 217 return GetAddrOfLocalVar(VD); 218 }); 219 assert(IsRegistered && "private var already registered as private"); 220 // Silence the warning about unused variable. 221 (void)IsRegistered; 222 ++IRef; 223 } 224 } 225 } 226 227 /// \brief Emits code for OpenMP parallel directive in the parallel region. 228 static void EmitOMPParallelCall(CodeGenFunction &CGF, 229 const OMPParallelDirective &S, 230 llvm::Value *OutlinedFn, 231 llvm::Value *CapturedStruct) { 232 if (auto C = S.getSingleClause(/*K*/ OMPC_num_threads)) { 233 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 234 auto NumThreadsClause = cast<OMPNumThreadsClause>(C); 235 auto NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads(), 236 /*IgnoreResultAssign*/ true); 237 CGF.CGM.getOpenMPRuntime().emitNumThreadsClause( 238 CGF, NumThreads, NumThreadsClause->getLocStart()); 239 } 240 CGF.CGM.getOpenMPRuntime().emitParallelCall(CGF, S.getLocStart(), OutlinedFn, 241 CapturedStruct); 242 } 243 244 void CodeGenFunction::EmitOMPParallelDirective(const OMPParallelDirective &S) { 245 auto CS = cast<CapturedStmt>(S.getAssociatedStmt()); 246 auto CapturedStruct = GenerateCapturedStmtArgument(*CS); 247 auto OutlinedFn = CGM.getOpenMPRuntime().emitOutlinedFunction( 248 S, *CS->getCapturedDecl()->param_begin()); 249 if (auto C = S.getSingleClause(/*K*/ OMPC_if)) { 250 auto Cond = cast<OMPIfClause>(C)->getCondition(); 251 EmitOMPIfClause(*this, Cond, [&](bool ThenBlock) { 252 if (ThenBlock) 253 EmitOMPParallelCall(*this, S, OutlinedFn, CapturedStruct); 254 else 255 CGM.getOpenMPRuntime().emitSerialCall(*this, S.getLocStart(), 256 OutlinedFn, CapturedStruct); 257 }); 258 } else 259 EmitOMPParallelCall(*this, S, OutlinedFn, CapturedStruct); 260 } 261 262 void CodeGenFunction::EmitOMPLoopBody(const OMPLoopDirective &S, 263 bool SeparateIter) { 264 RunCleanupsScope BodyScope(*this); 265 // Update counters values on current iteration. 266 for (auto I : S.updates()) { 267 EmitIgnoredExpr(I); 268 } 269 // On a continue in the body, jump to the end. 270 auto Continue = getJumpDestInCurrentScope("omp.body.continue"); 271 BreakContinueStack.push_back(BreakContinue(JumpDest(), Continue)); 272 // Emit loop body. 273 EmitStmt(S.getBody()); 274 // The end (updates/cleanups). 275 EmitBlock(Continue.getBlock()); 276 BreakContinueStack.pop_back(); 277 if (SeparateIter) { 278 // TODO: Update lastprivates if the SeparateIter flag is true. 279 // This will be implemented in a follow-up OMPLastprivateClause patch, but 280 // result should be still correct without it, as we do not make these 281 // variables private yet. 282 } 283 } 284 285 void CodeGenFunction::EmitOMPInnerLoop(const Stmt &S, bool RequiresCleanup, 286 const Expr *LoopCond, 287 const Expr *IncExpr, 288 const std::function<void()> &BodyGen) { 289 auto LoopExit = getJumpDestInCurrentScope("omp.inner.for.end"); 290 auto Cnt = getPGORegionCounter(&S); 291 292 // Start the loop with a block that tests the condition. 293 auto CondBlock = createBasicBlock("omp.inner.for.cond"); 294 EmitBlock(CondBlock); 295 LoopStack.push(CondBlock); 296 297 // If there are any cleanups between here and the loop-exit scope, 298 // create a block to stage a loop exit along. 299 auto ExitBlock = LoopExit.getBlock(); 300 if (RequiresCleanup) 301 ExitBlock = createBasicBlock("omp.inner.for.cond.cleanup"); 302 303 auto LoopBody = createBasicBlock("omp.inner.for.body"); 304 305 // Emit condition. 306 EmitBranchOnBoolExpr(LoopCond, LoopBody, ExitBlock, Cnt.getCount()); 307 if (ExitBlock != LoopExit.getBlock()) { 308 EmitBlock(ExitBlock); 309 EmitBranchThroughCleanup(LoopExit); 310 } 311 312 EmitBlock(LoopBody); 313 Cnt.beginRegion(Builder); 314 315 // Create a block for the increment. 316 auto Continue = getJumpDestInCurrentScope("omp.inner.for.inc"); 317 BreakContinueStack.push_back(BreakContinue(LoopExit, Continue)); 318 319 BodyGen(); 320 321 // Emit "IV = IV + 1" and a back-edge to the condition block. 322 EmitBlock(Continue.getBlock()); 323 EmitIgnoredExpr(IncExpr); 324 BreakContinueStack.pop_back(); 325 EmitBranch(CondBlock); 326 LoopStack.pop(); 327 // Emit the fall-through block. 328 EmitBlock(LoopExit.getBlock()); 329 } 330 331 void CodeGenFunction::EmitOMPSimdFinal(const OMPLoopDirective &S) { 332 auto IC = S.counters().begin(); 333 for (auto F : S.finals()) { 334 if (LocalDeclMap.lookup(cast<DeclRefExpr>((*IC))->getDecl())) { 335 EmitIgnoredExpr(F); 336 } 337 ++IC; 338 } 339 } 340 341 static void EmitOMPAlignedClause(CodeGenFunction &CGF, CodeGenModule &CGM, 342 const OMPAlignedClause &Clause) { 343 unsigned ClauseAlignment = 0; 344 if (auto AlignmentExpr = Clause.getAlignment()) { 345 auto AlignmentCI = 346 cast<llvm::ConstantInt>(CGF.EmitScalarExpr(AlignmentExpr)); 347 ClauseAlignment = static_cast<unsigned>(AlignmentCI->getZExtValue()); 348 } 349 for (auto E : Clause.varlists()) { 350 unsigned Alignment = ClauseAlignment; 351 if (Alignment == 0) { 352 // OpenMP [2.8.1, Description] 353 // If no optional parameter is specified, implementation-defined default 354 // alignments for SIMD instructions on the target platforms are assumed. 355 Alignment = CGM.getTargetCodeGenInfo().getOpenMPSimdDefaultAlignment( 356 E->getType()); 357 } 358 assert((Alignment == 0 || llvm::isPowerOf2_32(Alignment)) && 359 "alignment is not power of 2"); 360 if (Alignment != 0) { 361 llvm::Value *PtrValue = CGF.EmitScalarExpr(E); 362 CGF.EmitAlignmentAssumption(PtrValue, Alignment); 363 } 364 } 365 } 366 367 static void EmitPrivateLoopCounters(CodeGenFunction &CGF, 368 CodeGenFunction::OMPPrivateScope &LoopScope, 369 ArrayRef<Expr *> Counters) { 370 for (auto *E : Counters) { 371 auto VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 372 bool IsRegistered = LoopScope.addPrivate(VD, [&]() -> llvm::Value * { 373 // Emit var without initialization. 374 auto VarEmission = CGF.EmitAutoVarAlloca(*VD); 375 CGF.EmitAutoVarCleanups(VarEmission); 376 return VarEmission.getAllocatedAddress(); 377 }); 378 assert(IsRegistered && "counter already registered as private"); 379 // Silence the warning about unused variable. 380 (void)IsRegistered; 381 } 382 } 383 384 void CodeGenFunction::EmitOMPSimdDirective(const OMPSimdDirective &S) { 385 // Pragma 'simd' code depends on presence of 'lastprivate'. 386 // If present, we have to separate last iteration of the loop: 387 // 388 // if (LastIteration != 0) { 389 // for (IV in 0..LastIteration-1) BODY; 390 // BODY with updates of lastprivate vars; 391 // <Final counter/linear vars updates>; 392 // } 393 // 394 // otherwise (when there's no lastprivate): 395 // 396 // for (IV in 0..LastIteration) BODY; 397 // <Final counter/linear vars updates>; 398 // 399 400 // Walk clauses and process safelen/lastprivate. 401 bool SeparateIter = false; 402 LoopStack.setParallel(); 403 LoopStack.setVectorizerEnable(true); 404 for (auto C : S.clauses()) { 405 switch (C->getClauseKind()) { 406 case OMPC_safelen: { 407 RValue Len = EmitAnyExpr(cast<OMPSafelenClause>(C)->getSafelen(), 408 AggValueSlot::ignored(), true); 409 llvm::ConstantInt *Val = cast<llvm::ConstantInt>(Len.getScalarVal()); 410 LoopStack.setVectorizerWidth(Val->getZExtValue()); 411 // In presence of finite 'safelen', it may be unsafe to mark all 412 // the memory instructions parallel, because loop-carried 413 // dependences of 'safelen' iterations are possible. 414 LoopStack.setParallel(false); 415 break; 416 } 417 case OMPC_aligned: 418 EmitOMPAlignedClause(*this, CGM, cast<OMPAlignedClause>(*C)); 419 break; 420 case OMPC_lastprivate: 421 SeparateIter = true; 422 break; 423 default: 424 // Not handled yet 425 ; 426 } 427 } 428 429 InlinedOpenMPRegionScopeRAII Region(*this, S); 430 431 // Emit the loop iteration variable. 432 const Expr *IVExpr = S.getIterationVariable(); 433 const VarDecl *IVDecl = cast<VarDecl>(cast<DeclRefExpr>(IVExpr)->getDecl()); 434 EmitVarDecl(*IVDecl); 435 EmitIgnoredExpr(S.getInit()); 436 437 // Emit the iterations count variable. 438 // If it is not a variable, Sema decided to calculate iterations count on each 439 // iteration (e.g., it is foldable into a constant). 440 if (auto LIExpr = dyn_cast<DeclRefExpr>(S.getLastIteration())) { 441 EmitVarDecl(*cast<VarDecl>(LIExpr->getDecl())); 442 // Emit calculation of the iterations count. 443 EmitIgnoredExpr(S.getCalcLastIteration()); 444 } 445 446 if (SeparateIter) { 447 // Emit: if (LastIteration > 0) - begin. 448 RegionCounter Cnt = getPGORegionCounter(&S); 449 auto ThenBlock = createBasicBlock("simd.if.then"); 450 auto ContBlock = createBasicBlock("simd.if.end"); 451 EmitBranchOnBoolExpr(S.getPreCond(), ThenBlock, ContBlock, Cnt.getCount()); 452 EmitBlock(ThenBlock); 453 Cnt.beginRegion(Builder); 454 // Emit 'then' code. 455 { 456 OMPPrivateScope LoopScope(*this); 457 EmitPrivateLoopCounters(*this, LoopScope, S.counters()); 458 EmitOMPPrivateClause(S, LoopScope); 459 (void)LoopScope.Privatize(); 460 EmitOMPInnerLoop(S, LoopScope.requiresCleanups(), 461 S.getCond(/*SeparateIter=*/true), S.getInc(), 462 [&S, this]() { 463 EmitOMPLoopBody(S); 464 EmitStopPoint(&S); 465 }); 466 EmitOMPLoopBody(S, /* SeparateIter */ true); 467 } 468 EmitOMPSimdFinal(S); 469 // Emit: if (LastIteration != 0) - end. 470 EmitBranch(ContBlock); 471 EmitBlock(ContBlock, true); 472 } else { 473 { 474 OMPPrivateScope LoopScope(*this); 475 EmitPrivateLoopCounters(*this, LoopScope, S.counters()); 476 EmitOMPPrivateClause(S, LoopScope); 477 (void)LoopScope.Privatize(); 478 EmitOMPInnerLoop(S, LoopScope.requiresCleanups(), 479 S.getCond(/*SeparateIter=*/false), S.getInc(), 480 [&S, this]() { 481 EmitOMPLoopBody(S); 482 EmitStopPoint(&S); 483 }); 484 } 485 EmitOMPSimdFinal(S); 486 } 487 } 488 489 void CodeGenFunction::EmitOMPForOuterLoop(OpenMPScheduleClauseKind ScheduleKind, 490 const OMPLoopDirective &S, 491 OMPPrivateScope &LoopScope, 492 llvm::Value *LB, llvm::Value *UB, 493 llvm::Value *ST, llvm::Value *IL, 494 llvm::Value *Chunk) { 495 auto &RT = CGM.getOpenMPRuntime(); 496 497 // Dynamic scheduling of the outer loop (dynamic, guided, auto, runtime). 498 const bool Dynamic = RT.isDynamic(ScheduleKind); 499 500 assert(!RT.isStaticNonchunked(ScheduleKind, /* Chunked */ Chunk != nullptr) && 501 "static non-chunked schedule does not need outer loop"); 502 503 // Emit outer loop. 504 // 505 // OpenMP [2.7.1, Loop Construct, Description, table 2-1] 506 // When schedule(dynamic,chunk_size) is specified, the iterations are 507 // distributed to threads in the team in chunks as the threads request them. 508 // Each thread executes a chunk of iterations, then requests another chunk, 509 // until no chunks remain to be distributed. Each chunk contains chunk_size 510 // iterations, except for the last chunk to be distributed, which may have 511 // fewer iterations. When no chunk_size is specified, it defaults to 1. 512 // 513 // When schedule(guided,chunk_size) is specified, the iterations are assigned 514 // to threads in the team in chunks as the executing threads request them. 515 // Each thread executes a chunk of iterations, then requests another chunk, 516 // until no chunks remain to be assigned. For a chunk_size of 1, the size of 517 // each chunk is proportional to the number of unassigned iterations divided 518 // by the number of threads in the team, decreasing to 1. For a chunk_size 519 // with value k (greater than 1), the size of each chunk is determined in the 520 // same way, with the restriction that the chunks do not contain fewer than k 521 // iterations (except for the last chunk to be assigned, which may have fewer 522 // than k iterations). 523 // 524 // When schedule(auto) is specified, the decision regarding scheduling is 525 // delegated to the compiler and/or runtime system. The programmer gives the 526 // implementation the freedom to choose any possible mapping of iterations to 527 // threads in the team. 528 // 529 // When schedule(runtime) is specified, the decision regarding scheduling is 530 // deferred until run time, and the schedule and chunk size are taken from the 531 // run-sched-var ICV. If the ICV is set to auto, the schedule is 532 // implementation defined 533 // 534 // while(__kmpc_dispatch_next(&LB, &UB)) { 535 // idx = LB; 536 // while (idx <= UB) { BODY; ++idx; } // inner loop 537 // } 538 // 539 // OpenMP [2.7.1, Loop Construct, Description, table 2-1] 540 // When schedule(static, chunk_size) is specified, iterations are divided into 541 // chunks of size chunk_size, and the chunks are assigned to the threads in 542 // the team in a round-robin fashion in the order of the thread number. 543 // 544 // while(UB = min(UB, GlobalUB), idx = LB, idx < UB) { 545 // while (idx <= UB) { BODY; ++idx; } // inner loop 546 // LB = LB + ST; 547 // UB = UB + ST; 548 // } 549 // 550 551 const Expr *IVExpr = S.getIterationVariable(); 552 const unsigned IVSize = getContext().getTypeSize(IVExpr->getType()); 553 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation(); 554 555 RT.emitForInit( 556 *this, S.getLocStart(), ScheduleKind, IVSize, IVSigned, IL, LB, 557 (Dynamic ? EmitAnyExpr(S.getLastIteration()).getScalarVal() : UB), ST, 558 Chunk); 559 560 auto LoopExit = getJumpDestInCurrentScope("omp.dispatch.end"); 561 562 // Start the loop with a block that tests the condition. 563 auto CondBlock = createBasicBlock("omp.dispatch.cond"); 564 EmitBlock(CondBlock); 565 LoopStack.push(CondBlock); 566 567 llvm::Value *BoolCondVal = nullptr; 568 if (!Dynamic) { 569 // UB = min(UB, GlobalUB) 570 EmitIgnoredExpr(S.getEnsureUpperBound()); 571 // IV = LB 572 EmitIgnoredExpr(S.getInit()); 573 // IV < UB 574 BoolCondVal = EvaluateExprAsBool(S.getCond(false)); 575 } else { 576 BoolCondVal = RT.emitForNext(*this, S.getLocStart(), IVSize, IVSigned, 577 IL, LB, UB, ST); 578 } 579 580 // If there are any cleanups between here and the loop-exit scope, 581 // create a block to stage a loop exit along. 582 auto ExitBlock = LoopExit.getBlock(); 583 if (LoopScope.requiresCleanups()) 584 ExitBlock = createBasicBlock("omp.dispatch.cleanup"); 585 586 auto LoopBody = createBasicBlock("omp.dispatch.body"); 587 Builder.CreateCondBr(BoolCondVal, LoopBody, ExitBlock); 588 if (ExitBlock != LoopExit.getBlock()) { 589 EmitBlock(ExitBlock); 590 EmitBranchThroughCleanup(LoopExit); 591 } 592 EmitBlock(LoopBody); 593 594 // Emit "IV = LB" (in case of static schedule, we have already calculated new 595 // LB for loop condition and emitted it above). 596 if (Dynamic) 597 EmitIgnoredExpr(S.getInit()); 598 599 // Create a block for the increment. 600 auto Continue = getJumpDestInCurrentScope("omp.dispatch.inc"); 601 BreakContinueStack.push_back(BreakContinue(LoopExit, Continue)); 602 603 EmitOMPInnerLoop(S, LoopScope.requiresCleanups(), 604 S.getCond(/*SeparateIter=*/false), S.getInc(), [&S, this]() { 605 EmitOMPLoopBody(S); 606 EmitStopPoint(&S); 607 }); 608 609 EmitBlock(Continue.getBlock()); 610 BreakContinueStack.pop_back(); 611 if (!Dynamic) { 612 // Emit "LB = LB + Stride", "UB = UB + Stride". 613 EmitIgnoredExpr(S.getNextLowerBound()); 614 EmitIgnoredExpr(S.getNextUpperBound()); 615 } 616 617 EmitBranch(CondBlock); 618 LoopStack.pop(); 619 // Emit the fall-through block. 620 EmitBlock(LoopExit.getBlock()); 621 622 // Tell the runtime we are done. 623 // FIXME: Also call fini for ordered loops with dynamic scheduling. 624 if (!Dynamic) 625 RT.emitForFinish(*this, S.getLocStart(), ScheduleKind); 626 } 627 628 /// \brief Emit a helper variable and return corresponding lvalue. 629 static LValue EmitOMPHelperVar(CodeGenFunction &CGF, 630 const DeclRefExpr *Helper) { 631 auto VDecl = cast<VarDecl>(Helper->getDecl()); 632 CGF.EmitVarDecl(*VDecl); 633 return CGF.EmitLValue(Helper); 634 } 635 636 void CodeGenFunction::EmitOMPWorksharingLoop(const OMPLoopDirective &S) { 637 // Emit the loop iteration variable. 638 auto IVExpr = cast<DeclRefExpr>(S.getIterationVariable()); 639 auto IVDecl = cast<VarDecl>(IVExpr->getDecl()); 640 EmitVarDecl(*IVDecl); 641 642 // Emit the iterations count variable. 643 // If it is not a variable, Sema decided to calculate iterations count on each 644 // iteration (e.g., it is foldable into a constant). 645 if (auto LIExpr = dyn_cast<DeclRefExpr>(S.getLastIteration())) { 646 EmitVarDecl(*cast<VarDecl>(LIExpr->getDecl())); 647 // Emit calculation of the iterations count. 648 EmitIgnoredExpr(S.getCalcLastIteration()); 649 } 650 651 auto &RT = CGM.getOpenMPRuntime(); 652 653 // Check pre-condition. 654 { 655 // Skip the entire loop if we don't meet the precondition. 656 RegionCounter Cnt = getPGORegionCounter(&S); 657 auto ThenBlock = createBasicBlock("omp.precond.then"); 658 auto ContBlock = createBasicBlock("omp.precond.end"); 659 EmitBranchOnBoolExpr(S.getPreCond(), ThenBlock, ContBlock, Cnt.getCount()); 660 EmitBlock(ThenBlock); 661 Cnt.beginRegion(Builder); 662 // Emit 'then' code. 663 { 664 // Emit helper vars inits. 665 LValue LB = 666 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getLowerBoundVariable())); 667 LValue UB = 668 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getUpperBoundVariable())); 669 LValue ST = 670 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getStrideVariable())); 671 LValue IL = 672 EmitOMPHelperVar(*this, cast<DeclRefExpr>(S.getIsLastIterVariable())); 673 674 OMPPrivateScope LoopScope(*this); 675 EmitPrivateLoopCounters(*this, LoopScope, S.counters()); 676 (void)LoopScope.Privatize(); 677 678 // Detect the loop schedule kind and chunk. 679 auto ScheduleKind = OMPC_SCHEDULE_unknown; 680 llvm::Value *Chunk = nullptr; 681 if (auto C = cast_or_null<OMPScheduleClause>( 682 S.getSingleClause(OMPC_schedule))) { 683 ScheduleKind = C->getScheduleKind(); 684 if (auto Ch = C->getChunkSize()) { 685 Chunk = EmitScalarExpr(Ch); 686 Chunk = EmitScalarConversion(Chunk, Ch->getType(), 687 S.getIterationVariable()->getType()); 688 } 689 } 690 const unsigned IVSize = getContext().getTypeSize(IVExpr->getType()); 691 const bool IVSigned = IVExpr->getType()->hasSignedIntegerRepresentation(); 692 if (RT.isStaticNonchunked(ScheduleKind, 693 /* Chunked */ Chunk != nullptr)) { 694 // OpenMP [2.7.1, Loop Construct, Description, table 2-1] 695 // When no chunk_size is specified, the iteration space is divided into 696 // chunks that are approximately equal in size, and at most one chunk is 697 // distributed to each thread. Note that the size of the chunks is 698 // unspecified in this case. 699 RT.emitForInit(*this, S.getLocStart(), ScheduleKind, IVSize, IVSigned, 700 IL.getAddress(), LB.getAddress(), UB.getAddress(), 701 ST.getAddress()); 702 // UB = min(UB, GlobalUB); 703 EmitIgnoredExpr(S.getEnsureUpperBound()); 704 // IV = LB; 705 EmitIgnoredExpr(S.getInit()); 706 // while (idx <= UB) { BODY; ++idx; } 707 EmitOMPInnerLoop(S, LoopScope.requiresCleanups(), 708 S.getCond(/*SeparateIter=*/false), S.getInc(), 709 [&S, this]() { 710 EmitOMPLoopBody(S); 711 EmitStopPoint(&S); 712 }); 713 // Tell the runtime we are done. 714 RT.emitForFinish(*this, S.getLocStart(), ScheduleKind); 715 } else { 716 // Emit the outer loop, which requests its work chunk [LB..UB] from 717 // runtime and runs the inner loop to process it. 718 EmitOMPForOuterLoop(ScheduleKind, S, LoopScope, LB.getAddress(), 719 UB.getAddress(), ST.getAddress(), IL.getAddress(), 720 Chunk); 721 } 722 } 723 // We're now done with the loop, so jump to the continuation block. 724 EmitBranch(ContBlock); 725 EmitBlock(ContBlock, true); 726 } 727 } 728 729 void CodeGenFunction::EmitOMPForDirective(const OMPForDirective &S) { 730 InlinedOpenMPRegionScopeRAII Region(*this, S); 731 732 EmitOMPWorksharingLoop(S); 733 734 // Emit an implicit barrier at the end. 735 CGM.getOpenMPRuntime().emitBarrierCall(*this, S.getLocStart(), 736 /*IsExplicit*/ false); 737 } 738 739 void CodeGenFunction::EmitOMPForSimdDirective(const OMPForSimdDirective &) { 740 llvm_unreachable("CodeGen for 'omp for simd' is not supported yet."); 741 } 742 743 static LValue createSectionLVal(CodeGenFunction &CGF, QualType Ty, 744 const Twine &Name, 745 llvm::Value *Init = nullptr) { 746 auto LVal = CGF.MakeNaturalAlignAddrLValue(CGF.CreateMemTemp(Ty, Name), Ty); 747 if (Init) 748 CGF.EmitScalarInit(Init, LVal); 749 return LVal; 750 } 751 752 void CodeGenFunction::EmitOMPSectionsDirective(const OMPSectionsDirective &S) { 753 InlinedOpenMPRegionScopeRAII Region(*this, S); 754 755 auto *Stmt = cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt(); 756 auto *CS = dyn_cast<CompoundStmt>(Stmt); 757 if (CS && CS->size() > 1) { 758 auto &C = CGM.getContext(); 759 auto KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 760 // Emit helper vars inits. 761 LValue LB = createSectionLVal(*this, KmpInt32Ty, ".omp.sections.lb.", 762 Builder.getInt32(0)); 763 auto *GlobalUBVal = Builder.getInt32(CS->size() - 1); 764 LValue UB = 765 createSectionLVal(*this, KmpInt32Ty, ".omp.sections.ub.", GlobalUBVal); 766 LValue ST = createSectionLVal(*this, KmpInt32Ty, ".omp.sections.st.", 767 Builder.getInt32(1)); 768 LValue IL = createSectionLVal(*this, KmpInt32Ty, ".omp.sections.il.", 769 Builder.getInt32(0)); 770 // Loop counter. 771 LValue IV = createSectionLVal(*this, KmpInt32Ty, ".omp.sections.iv."); 772 OpaqueValueExpr IVRefExpr(S.getLocStart(), KmpInt32Ty, VK_LValue); 773 OpaqueValueMapping OpaqueIV(*this, &IVRefExpr, IV); 774 OpaqueValueExpr UBRefExpr(S.getLocStart(), KmpInt32Ty, VK_LValue); 775 OpaqueValueMapping OpaqueUB(*this, &UBRefExpr, UB); 776 // Generate condition for loop. 777 BinaryOperator Cond(&IVRefExpr, &UBRefExpr, BO_LE, C.BoolTy, VK_RValue, 778 OK_Ordinary, S.getLocStart(), /*fpContractable=*/false); 779 // Increment for loop counter. 780 UnaryOperator Inc(&IVRefExpr, UO_PreInc, KmpInt32Ty, VK_RValue, OK_Ordinary, 781 S.getLocStart()); 782 auto BodyGen = [this, CS, &S, &IV]() { 783 // Iterate through all sections and emit a switch construct: 784 // switch (IV) { 785 // case 0: 786 // <SectionStmt[0]>; 787 // break; 788 // ... 789 // case <NumSection> - 1: 790 // <SectionStmt[<NumSection> - 1]>; 791 // break; 792 // } 793 // .omp.sections.exit: 794 auto *ExitBB = createBasicBlock(".omp.sections.exit"); 795 auto *SwitchStmt = Builder.CreateSwitch( 796 EmitLoadOfLValue(IV, S.getLocStart()).getScalarVal(), ExitBB, 797 CS->size()); 798 unsigned CaseNumber = 0; 799 for (auto C = CS->children(); C; ++C, ++CaseNumber) { 800 auto CaseBB = createBasicBlock(".omp.sections.case"); 801 EmitBlock(CaseBB); 802 SwitchStmt->addCase(Builder.getInt32(CaseNumber), CaseBB); 803 EmitStmt(*C); 804 EmitBranch(ExitBB); 805 } 806 EmitBlock(ExitBB, /*IsFinished=*/true); 807 }; 808 // Emit static non-chunked loop. 809 CGM.getOpenMPRuntime().emitForInit( 810 *this, S.getLocStart(), OMPC_SCHEDULE_static, /*IVSize=*/32, 811 /*IVSigned=*/true, IL.getAddress(), LB.getAddress(), UB.getAddress(), 812 ST.getAddress()); 813 // UB = min(UB, GlobalUB); 814 auto *UBVal = EmitLoadOfScalar(UB, S.getLocStart()); 815 auto *MinUBGlobalUB = Builder.CreateSelect( 816 Builder.CreateICmpSLT(UBVal, GlobalUBVal), UBVal, GlobalUBVal); 817 EmitStoreOfScalar(MinUBGlobalUB, UB); 818 // IV = LB; 819 EmitStoreOfScalar(EmitLoadOfScalar(LB, S.getLocStart()), IV); 820 // while (idx <= UB) { BODY; ++idx; } 821 EmitOMPInnerLoop(S, /*RequiresCleanup=*/false, &Cond, &Inc, BodyGen); 822 // Tell the runtime we are done. 823 CGM.getOpenMPRuntime().emitForFinish(*this, S.getLocStart(), 824 OMPC_SCHEDULE_static); 825 } else { 826 // If only one section is found - no need to generate loop, emit as a single 827 // region. 828 CGM.getOpenMPRuntime().emitSingleRegion(*this, [&]() -> void { 829 InlinedOpenMPRegionScopeRAII Region(*this, S); 830 EmitStmt(Stmt); 831 EnsureInsertPoint(); 832 }, S.getLocStart()); 833 } 834 835 // Emit an implicit barrier at the end. 836 if (!S.getSingleClause(OMPC_nowait)) 837 CGM.getOpenMPRuntime().emitBarrierCall(*this, S.getLocStart(), 838 /*IsExplicit=*/false); 839 } 840 841 void CodeGenFunction::EmitOMPSectionDirective(const OMPSectionDirective &S) { 842 InlinedOpenMPRegionScopeRAII Region(*this, S); 843 EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt()); 844 EnsureInsertPoint(); 845 } 846 847 void CodeGenFunction::EmitOMPSingleDirective(const OMPSingleDirective &S) { 848 CGM.getOpenMPRuntime().emitSingleRegion(*this, [&]() -> void { 849 InlinedOpenMPRegionScopeRAII Region(*this, S); 850 EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt()); 851 EnsureInsertPoint(); 852 }, S.getLocStart()); 853 } 854 855 void CodeGenFunction::EmitOMPMasterDirective(const OMPMasterDirective &S) { 856 CGM.getOpenMPRuntime().emitMasterRegion(*this, [&]() -> void { 857 InlinedOpenMPRegionScopeRAII Region(*this, S); 858 EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt()); 859 EnsureInsertPoint(); 860 }, S.getLocStart()); 861 } 862 863 void CodeGenFunction::EmitOMPCriticalDirective(const OMPCriticalDirective &S) { 864 CGM.getOpenMPRuntime().emitCriticalRegion( 865 *this, S.getDirectiveName().getAsString(), [&]() -> void { 866 InlinedOpenMPRegionScopeRAII Region(*this, S); 867 EmitStmt(cast<CapturedStmt>(S.getAssociatedStmt())->getCapturedStmt()); 868 EnsureInsertPoint(); 869 }, S.getLocStart()); 870 } 871 872 void 873 CodeGenFunction::EmitOMPParallelForDirective(const OMPParallelForDirective &) { 874 llvm_unreachable("CodeGen for 'omp parallel for' is not supported yet."); 875 } 876 877 void CodeGenFunction::EmitOMPParallelForSimdDirective( 878 const OMPParallelForSimdDirective &) { 879 llvm_unreachable("CodeGen for 'omp parallel for simd' is not supported yet."); 880 } 881 882 void CodeGenFunction::EmitOMPParallelSectionsDirective( 883 const OMPParallelSectionsDirective &) { 884 llvm_unreachable("CodeGen for 'omp parallel sections' is not supported yet."); 885 } 886 887 void CodeGenFunction::EmitOMPTaskDirective(const OMPTaskDirective &S) { 888 // Emit outlined function for task construct. 889 auto CS = cast<CapturedStmt>(S.getAssociatedStmt()); 890 auto CapturedStruct = GenerateCapturedStmtArgument(*CS); 891 auto *I = CS->getCapturedDecl()->param_begin(); 892 // The first function argument for tasks is a thread id, the second one is a 893 // part id (0 for tied tasks, >=0 for untied task). 894 auto OutlinedFn = 895 CGM.getOpenMPRuntime().emitTaskOutlinedFunction(S, *I, *std::next(I)); 896 // Check if we should emit tied or untied task. 897 bool Tied = !S.getSingleClause(OMPC_untied); 898 // Check if the task is final 899 llvm::PointerIntPair<llvm::Value *, 1, bool> Final; 900 if (auto *Clause = S.getSingleClause(OMPC_final)) { 901 // If the condition constant folds and can be elided, try to avoid emitting 902 // the condition and the dead arm of the if/else. 903 auto *Cond = cast<OMPFinalClause>(Clause)->getCondition(); 904 bool CondConstant; 905 if (ConstantFoldsToSimpleInteger(Cond, CondConstant)) 906 Final.setInt(CondConstant); 907 else 908 Final.setPointer(EvaluateExprAsBool(Cond)); 909 } else { 910 // By default the task is not final. 911 Final.setInt(/*IntVal=*/false); 912 } 913 auto SharedsTy = getContext().getRecordType(CS->getCapturedRecordDecl()); 914 CGM.getOpenMPRuntime().emitTaskCall(*this, S.getLocStart(), Tied, Final, 915 OutlinedFn, SharedsTy, CapturedStruct); 916 } 917 918 void CodeGenFunction::EmitOMPTaskyieldDirective( 919 const OMPTaskyieldDirective &S) { 920 CGM.getOpenMPRuntime().emitTaskyieldCall(*this, S.getLocStart()); 921 } 922 923 void CodeGenFunction::EmitOMPBarrierDirective(const OMPBarrierDirective &S) { 924 CGM.getOpenMPRuntime().emitBarrierCall(*this, S.getLocStart()); 925 } 926 927 void CodeGenFunction::EmitOMPTaskwaitDirective(const OMPTaskwaitDirective &) { 928 llvm_unreachable("CodeGen for 'omp taskwait' is not supported yet."); 929 } 930 931 void CodeGenFunction::EmitOMPFlushDirective(const OMPFlushDirective &S) { 932 CGM.getOpenMPRuntime().emitFlush(*this, [&]() -> ArrayRef<const Expr *> { 933 if (auto C = S.getSingleClause(/*K*/ OMPC_flush)) { 934 auto FlushClause = cast<OMPFlushClause>(C); 935 return llvm::makeArrayRef(FlushClause->varlist_begin(), 936 FlushClause->varlist_end()); 937 } 938 return llvm::None; 939 }(), S.getLocStart()); 940 } 941 942 void CodeGenFunction::EmitOMPOrderedDirective(const OMPOrderedDirective &) { 943 llvm_unreachable("CodeGen for 'omp ordered' is not supported yet."); 944 } 945 946 static llvm::Value *convertToScalarValue(CodeGenFunction &CGF, RValue Val, 947 QualType SrcType, QualType DestType) { 948 assert(CGF.hasScalarEvaluationKind(DestType) && 949 "DestType must have scalar evaluation kind."); 950 assert(!Val.isAggregate() && "Must be a scalar or complex."); 951 return Val.isScalar() 952 ? CGF.EmitScalarConversion(Val.getScalarVal(), SrcType, DestType) 953 : CGF.EmitComplexToScalarConversion(Val.getComplexVal(), SrcType, 954 DestType); 955 } 956 957 static CodeGenFunction::ComplexPairTy 958 convertToComplexValue(CodeGenFunction &CGF, RValue Val, QualType SrcType, 959 QualType DestType) { 960 assert(CGF.getEvaluationKind(DestType) == TEK_Complex && 961 "DestType must have complex evaluation kind."); 962 CodeGenFunction::ComplexPairTy ComplexVal; 963 if (Val.isScalar()) { 964 // Convert the input element to the element type of the complex. 965 auto DestElementType = DestType->castAs<ComplexType>()->getElementType(); 966 auto ScalarVal = 967 CGF.EmitScalarConversion(Val.getScalarVal(), SrcType, DestElementType); 968 ComplexVal = CodeGenFunction::ComplexPairTy( 969 ScalarVal, llvm::Constant::getNullValue(ScalarVal->getType())); 970 } else { 971 assert(Val.isComplex() && "Must be a scalar or complex."); 972 auto SrcElementType = SrcType->castAs<ComplexType>()->getElementType(); 973 auto DestElementType = DestType->castAs<ComplexType>()->getElementType(); 974 ComplexVal.first = CGF.EmitScalarConversion( 975 Val.getComplexVal().first, SrcElementType, DestElementType); 976 ComplexVal.second = CGF.EmitScalarConversion( 977 Val.getComplexVal().second, SrcElementType, DestElementType); 978 } 979 return ComplexVal; 980 } 981 982 static void EmitOMPAtomicReadExpr(CodeGenFunction &CGF, bool IsSeqCst, 983 const Expr *X, const Expr *V, 984 SourceLocation Loc) { 985 // v = x; 986 assert(V->isLValue() && "V of 'omp atomic read' is not lvalue"); 987 assert(X->isLValue() && "X of 'omp atomic read' is not lvalue"); 988 LValue XLValue = CGF.EmitLValue(X); 989 LValue VLValue = CGF.EmitLValue(V); 990 RValue Res = XLValue.isGlobalReg() 991 ? CGF.EmitLoadOfLValue(XLValue, Loc) 992 : CGF.EmitAtomicLoad(XLValue, Loc, 993 IsSeqCst ? llvm::SequentiallyConsistent 994 : llvm::Monotonic, 995 XLValue.isVolatile()); 996 // OpenMP, 2.12.6, atomic Construct 997 // Any atomic construct with a seq_cst clause forces the atomically 998 // performed operation to include an implicit flush operation without a 999 // list. 1000 if (IsSeqCst) 1001 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, llvm::None, Loc); 1002 switch (CGF.getEvaluationKind(V->getType())) { 1003 case TEK_Scalar: 1004 CGF.EmitStoreOfScalar( 1005 convertToScalarValue(CGF, Res, X->getType(), V->getType()), VLValue); 1006 break; 1007 case TEK_Complex: 1008 CGF.EmitStoreOfComplex( 1009 convertToComplexValue(CGF, Res, X->getType(), V->getType()), VLValue, 1010 /*isInit=*/false); 1011 break; 1012 case TEK_Aggregate: 1013 llvm_unreachable("Must be a scalar or complex."); 1014 } 1015 } 1016 1017 static void EmitOMPAtomicWriteExpr(CodeGenFunction &CGF, bool IsSeqCst, 1018 const Expr *X, const Expr *E, 1019 SourceLocation Loc) { 1020 // x = expr; 1021 assert(X->isLValue() && "X of 'omp atomic write' is not lvalue"); 1022 LValue XLValue = CGF.EmitLValue(X); 1023 RValue ExprRValue = CGF.EmitAnyExpr(E); 1024 if (XLValue.isGlobalReg()) 1025 CGF.EmitStoreThroughGlobalRegLValue(ExprRValue, XLValue); 1026 else 1027 CGF.EmitAtomicStore(ExprRValue, XLValue, 1028 IsSeqCst ? llvm::SequentiallyConsistent 1029 : llvm::Monotonic, 1030 XLValue.isVolatile(), /*IsInit=*/false); 1031 // OpenMP, 2.12.6, atomic Construct 1032 // Any atomic construct with a seq_cst clause forces the atomically 1033 // performed operation to include an implicit flush operation without a 1034 // list. 1035 if (IsSeqCst) 1036 CGF.CGM.getOpenMPRuntime().emitFlush(CGF, llvm::None, Loc); 1037 } 1038 1039 static void EmitOMPAtomicExpr(CodeGenFunction &CGF, OpenMPClauseKind Kind, 1040 bool IsSeqCst, const Expr *X, const Expr *V, 1041 const Expr *E, SourceLocation Loc) { 1042 switch (Kind) { 1043 case OMPC_read: 1044 EmitOMPAtomicReadExpr(CGF, IsSeqCst, X, V, Loc); 1045 break; 1046 case OMPC_write: 1047 EmitOMPAtomicWriteExpr(CGF, IsSeqCst, X, E, Loc); 1048 break; 1049 case OMPC_update: 1050 case OMPC_capture: 1051 llvm_unreachable("CodeGen for 'omp atomic clause' is not supported yet."); 1052 case OMPC_if: 1053 case OMPC_final: 1054 case OMPC_num_threads: 1055 case OMPC_private: 1056 case OMPC_firstprivate: 1057 case OMPC_lastprivate: 1058 case OMPC_reduction: 1059 case OMPC_safelen: 1060 case OMPC_collapse: 1061 case OMPC_default: 1062 case OMPC_seq_cst: 1063 case OMPC_shared: 1064 case OMPC_linear: 1065 case OMPC_aligned: 1066 case OMPC_copyin: 1067 case OMPC_copyprivate: 1068 case OMPC_flush: 1069 case OMPC_proc_bind: 1070 case OMPC_schedule: 1071 case OMPC_ordered: 1072 case OMPC_nowait: 1073 case OMPC_untied: 1074 case OMPC_threadprivate: 1075 case OMPC_mergeable: 1076 case OMPC_unknown: 1077 llvm_unreachable("Clause is not allowed in 'omp atomic'."); 1078 } 1079 } 1080 1081 void CodeGenFunction::EmitOMPAtomicDirective(const OMPAtomicDirective &S) { 1082 bool IsSeqCst = S.getSingleClause(/*K=*/OMPC_seq_cst); 1083 OpenMPClauseKind Kind = OMPC_unknown; 1084 for (auto *C : S.clauses()) { 1085 // Find first clause (skip seq_cst clause, if it is first). 1086 if (C->getClauseKind() != OMPC_seq_cst) { 1087 Kind = C->getClauseKind(); 1088 break; 1089 } 1090 } 1091 1092 const auto *CS = 1093 S.getAssociatedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 1094 if (const auto *EWC = dyn_cast<ExprWithCleanups>(CS)) 1095 enterFullExpression(EWC); 1096 InlinedOpenMPRegionScopeRAII Region(*this, S); 1097 1098 EmitOMPAtomicExpr(*this, Kind, IsSeqCst, S.getX(), S.getV(), S.getExpr(), 1099 S.getLocStart()); 1100 } 1101 1102 void CodeGenFunction::EmitOMPTargetDirective(const OMPTargetDirective &) { 1103 llvm_unreachable("CodeGen for 'omp target' is not supported yet."); 1104 } 1105 1106 void CodeGenFunction::EmitOMPTeamsDirective(const OMPTeamsDirective &) { 1107 llvm_unreachable("CodeGen for 'omp teams' is not supported yet."); 1108 } 1109 1110