1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This provides a class for OpenMP runtime code generation. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "CGOpenMPRuntime.h" 14 #include "CGCXXABI.h" 15 #include "CGCleanup.h" 16 #include "CGRecordLayout.h" 17 #include "CodeGenFunction.h" 18 #include "clang/AST/Attr.h" 19 #include "clang/AST/Decl.h" 20 #include "clang/AST/OpenMPClause.h" 21 #include "clang/AST/StmtOpenMP.h" 22 #include "clang/AST/StmtVisitor.h" 23 #include "clang/Basic/BitmaskEnum.h" 24 #include "clang/Basic/FileManager.h" 25 #include "clang/Basic/OpenMPKinds.h" 26 #include "clang/Basic/SourceManager.h" 27 #include "clang/CodeGen/ConstantInitBuilder.h" 28 #include "llvm/ADT/ArrayRef.h" 29 #include "llvm/ADT/SetOperations.h" 30 #include "llvm/ADT/StringExtras.h" 31 #include "llvm/Bitcode/BitcodeReader.h" 32 #include "llvm/IR/Constants.h" 33 #include "llvm/IR/DerivedTypes.h" 34 #include "llvm/IR/GlobalValue.h" 35 #include "llvm/IR/Value.h" 36 #include "llvm/Support/AtomicOrdering.h" 37 #include "llvm/Support/Format.h" 38 #include "llvm/Support/raw_ostream.h" 39 #include <cassert> 40 #include <numeric> 41 42 using namespace clang; 43 using namespace CodeGen; 44 using namespace llvm::omp; 45 46 namespace { 47 /// Base class for handling code generation inside OpenMP regions. 48 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo { 49 public: 50 /// Kinds of OpenMP regions used in codegen. 51 enum CGOpenMPRegionKind { 52 /// Region with outlined function for standalone 'parallel' 53 /// directive. 54 ParallelOutlinedRegion, 55 /// Region with outlined function for standalone 'task' directive. 56 TaskOutlinedRegion, 57 /// Region for constructs that do not require function outlining, 58 /// like 'for', 'sections', 'atomic' etc. directives. 59 InlinedRegion, 60 /// Region with outlined function for standalone 'target' directive. 61 TargetRegion, 62 }; 63 64 CGOpenMPRegionInfo(const CapturedStmt &CS, 65 const CGOpenMPRegionKind RegionKind, 66 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 67 bool HasCancel) 68 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind), 69 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {} 70 71 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind, 72 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 73 bool HasCancel) 74 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen), 75 Kind(Kind), HasCancel(HasCancel) {} 76 77 /// Get a variable or parameter for storing global thread id 78 /// inside OpenMP construct. 79 virtual const VarDecl *getThreadIDVariable() const = 0; 80 81 /// Emit the captured statement body. 82 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override; 83 84 /// Get an LValue for the current ThreadID variable. 85 /// \return LValue for thread id variable. This LValue always has type int32*. 86 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF); 87 88 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {} 89 90 CGOpenMPRegionKind getRegionKind() const { return RegionKind; } 91 92 OpenMPDirectiveKind getDirectiveKind() const { return Kind; } 93 94 bool hasCancel() const { return HasCancel; } 95 96 static bool classof(const CGCapturedStmtInfo *Info) { 97 return Info->getKind() == CR_OpenMP; 98 } 99 100 ~CGOpenMPRegionInfo() override = default; 101 102 protected: 103 CGOpenMPRegionKind RegionKind; 104 RegionCodeGenTy CodeGen; 105 OpenMPDirectiveKind Kind; 106 bool HasCancel; 107 }; 108 109 /// API for captured statement code generation in OpenMP constructs. 110 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo { 111 public: 112 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar, 113 const RegionCodeGenTy &CodeGen, 114 OpenMPDirectiveKind Kind, bool HasCancel, 115 StringRef HelperName) 116 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind, 117 HasCancel), 118 ThreadIDVar(ThreadIDVar), HelperName(HelperName) { 119 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 120 } 121 122 /// Get a variable or parameter for storing global thread id 123 /// inside OpenMP construct. 124 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 125 126 /// Get the name of the capture helper. 127 StringRef getHelperName() const override { return HelperName; } 128 129 static bool classof(const CGCapturedStmtInfo *Info) { 130 return CGOpenMPRegionInfo::classof(Info) && 131 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 132 ParallelOutlinedRegion; 133 } 134 135 private: 136 /// A variable or parameter storing global thread id for OpenMP 137 /// constructs. 138 const VarDecl *ThreadIDVar; 139 StringRef HelperName; 140 }; 141 142 /// API for captured statement code generation in OpenMP constructs. 143 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo { 144 public: 145 class UntiedTaskActionTy final : public PrePostActionTy { 146 bool Untied; 147 const VarDecl *PartIDVar; 148 const RegionCodeGenTy UntiedCodeGen; 149 llvm::SwitchInst *UntiedSwitch = nullptr; 150 151 public: 152 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar, 153 const RegionCodeGenTy &UntiedCodeGen) 154 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {} 155 void Enter(CodeGenFunction &CGF) override { 156 if (Untied) { 157 // Emit task switching point. 158 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 159 CGF.GetAddrOfLocalVar(PartIDVar), 160 PartIDVar->getType()->castAs<PointerType>()); 161 llvm::Value *Res = 162 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation()); 163 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done."); 164 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB); 165 CGF.EmitBlock(DoneBB); 166 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 167 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 168 UntiedSwitch->addCase(CGF.Builder.getInt32(0), 169 CGF.Builder.GetInsertBlock()); 170 emitUntiedSwitch(CGF); 171 } 172 } 173 void emitUntiedSwitch(CodeGenFunction &CGF) const { 174 if (Untied) { 175 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 176 CGF.GetAddrOfLocalVar(PartIDVar), 177 PartIDVar->getType()->castAs<PointerType>()); 178 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 179 PartIdLVal); 180 UntiedCodeGen(CGF); 181 CodeGenFunction::JumpDest CurPoint = 182 CGF.getJumpDestInCurrentScope(".untied.next."); 183 CGF.EmitBranch(CGF.ReturnBlock.getBlock()); 184 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 185 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 186 CGF.Builder.GetInsertBlock()); 187 CGF.EmitBranchThroughCleanup(CurPoint); 188 CGF.EmitBlock(CurPoint.getBlock()); 189 } 190 } 191 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); } 192 }; 193 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS, 194 const VarDecl *ThreadIDVar, 195 const RegionCodeGenTy &CodeGen, 196 OpenMPDirectiveKind Kind, bool HasCancel, 197 const UntiedTaskActionTy &Action) 198 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel), 199 ThreadIDVar(ThreadIDVar), Action(Action) { 200 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 201 } 202 203 /// Get a variable or parameter for storing global thread id 204 /// inside OpenMP construct. 205 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 206 207 /// Get an LValue for the current ThreadID variable. 208 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override; 209 210 /// Get the name of the capture helper. 211 StringRef getHelperName() const override { return ".omp_outlined."; } 212 213 void emitUntiedSwitch(CodeGenFunction &CGF) override { 214 Action.emitUntiedSwitch(CGF); 215 } 216 217 static bool classof(const CGCapturedStmtInfo *Info) { 218 return CGOpenMPRegionInfo::classof(Info) && 219 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 220 TaskOutlinedRegion; 221 } 222 223 private: 224 /// A variable or parameter storing global thread id for OpenMP 225 /// constructs. 226 const VarDecl *ThreadIDVar; 227 /// Action for emitting code for untied tasks. 228 const UntiedTaskActionTy &Action; 229 }; 230 231 /// API for inlined captured statement code generation in OpenMP 232 /// constructs. 233 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo { 234 public: 235 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI, 236 const RegionCodeGenTy &CodeGen, 237 OpenMPDirectiveKind Kind, bool HasCancel) 238 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel), 239 OldCSI(OldCSI), 240 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {} 241 242 // Retrieve the value of the context parameter. 243 llvm::Value *getContextValue() const override { 244 if (OuterRegionInfo) 245 return OuterRegionInfo->getContextValue(); 246 llvm_unreachable("No context value for inlined OpenMP region"); 247 } 248 249 void setContextValue(llvm::Value *V) override { 250 if (OuterRegionInfo) { 251 OuterRegionInfo->setContextValue(V); 252 return; 253 } 254 llvm_unreachable("No context value for inlined OpenMP region"); 255 } 256 257 /// Lookup the captured field decl for a variable. 258 const FieldDecl *lookup(const VarDecl *VD) const override { 259 if (OuterRegionInfo) 260 return OuterRegionInfo->lookup(VD); 261 // If there is no outer outlined region,no need to lookup in a list of 262 // captured variables, we can use the original one. 263 return nullptr; 264 } 265 266 FieldDecl *getThisFieldDecl() const override { 267 if (OuterRegionInfo) 268 return OuterRegionInfo->getThisFieldDecl(); 269 return nullptr; 270 } 271 272 /// Get a variable or parameter for storing global thread id 273 /// inside OpenMP construct. 274 const VarDecl *getThreadIDVariable() const override { 275 if (OuterRegionInfo) 276 return OuterRegionInfo->getThreadIDVariable(); 277 return nullptr; 278 } 279 280 /// Get an LValue for the current ThreadID variable. 281 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override { 282 if (OuterRegionInfo) 283 return OuterRegionInfo->getThreadIDVariableLValue(CGF); 284 llvm_unreachable("No LValue for inlined OpenMP construct"); 285 } 286 287 /// Get the name of the capture helper. 288 StringRef getHelperName() const override { 289 if (auto *OuterRegionInfo = getOldCSI()) 290 return OuterRegionInfo->getHelperName(); 291 llvm_unreachable("No helper name for inlined OpenMP construct"); 292 } 293 294 void emitUntiedSwitch(CodeGenFunction &CGF) override { 295 if (OuterRegionInfo) 296 OuterRegionInfo->emitUntiedSwitch(CGF); 297 } 298 299 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; } 300 301 static bool classof(const CGCapturedStmtInfo *Info) { 302 return CGOpenMPRegionInfo::classof(Info) && 303 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion; 304 } 305 306 ~CGOpenMPInlinedRegionInfo() override = default; 307 308 private: 309 /// CodeGen info about outer OpenMP region. 310 CodeGenFunction::CGCapturedStmtInfo *OldCSI; 311 CGOpenMPRegionInfo *OuterRegionInfo; 312 }; 313 314 /// API for captured statement code generation in OpenMP target 315 /// constructs. For this captures, implicit parameters are used instead of the 316 /// captured fields. The name of the target region has to be unique in a given 317 /// application so it is provided by the client, because only the client has 318 /// the information to generate that. 319 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo { 320 public: 321 CGOpenMPTargetRegionInfo(const CapturedStmt &CS, 322 const RegionCodeGenTy &CodeGen, StringRef HelperName) 323 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target, 324 /*HasCancel=*/false), 325 HelperName(HelperName) {} 326 327 /// This is unused for target regions because each starts executing 328 /// with a single thread. 329 const VarDecl *getThreadIDVariable() const override { return nullptr; } 330 331 /// Get the name of the capture helper. 332 StringRef getHelperName() const override { return HelperName; } 333 334 static bool classof(const CGCapturedStmtInfo *Info) { 335 return CGOpenMPRegionInfo::classof(Info) && 336 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion; 337 } 338 339 private: 340 StringRef HelperName; 341 }; 342 343 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) { 344 llvm_unreachable("No codegen for expressions"); 345 } 346 /// API for generation of expressions captured in a innermost OpenMP 347 /// region. 348 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo { 349 public: 350 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS) 351 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen, 352 OMPD_unknown, 353 /*HasCancel=*/false), 354 PrivScope(CGF) { 355 // Make sure the globals captured in the provided statement are local by 356 // using the privatization logic. We assume the same variable is not 357 // captured more than once. 358 for (const auto &C : CS.captures()) { 359 if (!C.capturesVariable() && !C.capturesVariableByCopy()) 360 continue; 361 362 const VarDecl *VD = C.getCapturedVar(); 363 if (VD->isLocalVarDeclOrParm()) 364 continue; 365 366 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD), 367 /*RefersToEnclosingVariableOrCapture=*/false, 368 VD->getType().getNonReferenceType(), VK_LValue, 369 C.getLocation()); 370 PrivScope.addPrivate( 371 VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(CGF); }); 372 } 373 (void)PrivScope.Privatize(); 374 } 375 376 /// Lookup the captured field decl for a variable. 377 const FieldDecl *lookup(const VarDecl *VD) const override { 378 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD)) 379 return FD; 380 return nullptr; 381 } 382 383 /// Emit the captured statement body. 384 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override { 385 llvm_unreachable("No body for expressions"); 386 } 387 388 /// Get a variable or parameter for storing global thread id 389 /// inside OpenMP construct. 390 const VarDecl *getThreadIDVariable() const override { 391 llvm_unreachable("No thread id for expressions"); 392 } 393 394 /// Get the name of the capture helper. 395 StringRef getHelperName() const override { 396 llvm_unreachable("No helper name for expressions"); 397 } 398 399 static bool classof(const CGCapturedStmtInfo *Info) { return false; } 400 401 private: 402 /// Private scope to capture global variables. 403 CodeGenFunction::OMPPrivateScope PrivScope; 404 }; 405 406 /// RAII for emitting code of OpenMP constructs. 407 class InlinedOpenMPRegionRAII { 408 CodeGenFunction &CGF; 409 llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields; 410 FieldDecl *LambdaThisCaptureField = nullptr; 411 const CodeGen::CGBlockInfo *BlockInfo = nullptr; 412 413 public: 414 /// Constructs region for combined constructs. 415 /// \param CodeGen Code generation sequence for combined directives. Includes 416 /// a list of functions used for code generation of implicitly inlined 417 /// regions. 418 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen, 419 OpenMPDirectiveKind Kind, bool HasCancel) 420 : CGF(CGF) { 421 // Start emission for the construct. 422 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo( 423 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel); 424 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 425 LambdaThisCaptureField = CGF.LambdaThisCaptureField; 426 CGF.LambdaThisCaptureField = nullptr; 427 BlockInfo = CGF.BlockInfo; 428 CGF.BlockInfo = nullptr; 429 } 430 431 ~InlinedOpenMPRegionRAII() { 432 // Restore original CapturedStmtInfo only if we're done with code emission. 433 auto *OldCSI = 434 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI(); 435 delete CGF.CapturedStmtInfo; 436 CGF.CapturedStmtInfo = OldCSI; 437 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 438 CGF.LambdaThisCaptureField = LambdaThisCaptureField; 439 CGF.BlockInfo = BlockInfo; 440 } 441 }; 442 443 /// Values for bit flags used in the ident_t to describe the fields. 444 /// All enumeric elements are named and described in accordance with the code 445 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 446 enum OpenMPLocationFlags : unsigned { 447 /// Use trampoline for internal microtask. 448 OMP_IDENT_IMD = 0x01, 449 /// Use c-style ident structure. 450 OMP_IDENT_KMPC = 0x02, 451 /// Atomic reduction option for kmpc_reduce. 452 OMP_ATOMIC_REDUCE = 0x10, 453 /// Explicit 'barrier' directive. 454 OMP_IDENT_BARRIER_EXPL = 0x20, 455 /// Implicit barrier in code. 456 OMP_IDENT_BARRIER_IMPL = 0x40, 457 /// Implicit barrier in 'for' directive. 458 OMP_IDENT_BARRIER_IMPL_FOR = 0x40, 459 /// Implicit barrier in 'sections' directive. 460 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0, 461 /// Implicit barrier in 'single' directive. 462 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140, 463 /// Call of __kmp_for_static_init for static loop. 464 OMP_IDENT_WORK_LOOP = 0x200, 465 /// Call of __kmp_for_static_init for sections. 466 OMP_IDENT_WORK_SECTIONS = 0x400, 467 /// Call of __kmp_for_static_init for distribute. 468 OMP_IDENT_WORK_DISTRIBUTE = 0x800, 469 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE) 470 }; 471 472 namespace { 473 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 474 /// Values for bit flags for marking which requires clauses have been used. 475 enum OpenMPOffloadingRequiresDirFlags : int64_t { 476 /// flag undefined. 477 OMP_REQ_UNDEFINED = 0x000, 478 /// no requires clause present. 479 OMP_REQ_NONE = 0x001, 480 /// reverse_offload clause. 481 OMP_REQ_REVERSE_OFFLOAD = 0x002, 482 /// unified_address clause. 483 OMP_REQ_UNIFIED_ADDRESS = 0x004, 484 /// unified_shared_memory clause. 485 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008, 486 /// dynamic_allocators clause. 487 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010, 488 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS) 489 }; 490 491 enum OpenMPOffloadingReservedDeviceIDs { 492 /// Device ID if the device was not defined, runtime should get it 493 /// from environment variables in the spec. 494 OMP_DEVICEID_UNDEF = -1, 495 }; 496 } // anonymous namespace 497 498 /// Describes ident structure that describes a source location. 499 /// All descriptions are taken from 500 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 501 /// Original structure: 502 /// typedef struct ident { 503 /// kmp_int32 reserved_1; /**< might be used in Fortran; 504 /// see above */ 505 /// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags; 506 /// KMP_IDENT_KMPC identifies this union 507 /// member */ 508 /// kmp_int32 reserved_2; /**< not really used in Fortran any more; 509 /// see above */ 510 ///#if USE_ITT_BUILD 511 /// /* but currently used for storing 512 /// region-specific ITT */ 513 /// /* contextual information. */ 514 ///#endif /* USE_ITT_BUILD */ 515 /// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for 516 /// C++ */ 517 /// char const *psource; /**< String describing the source location. 518 /// The string is composed of semi-colon separated 519 // fields which describe the source file, 520 /// the function and a pair of line numbers that 521 /// delimit the construct. 522 /// */ 523 /// } ident_t; 524 enum IdentFieldIndex { 525 /// might be used in Fortran 526 IdentField_Reserved_1, 527 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member. 528 IdentField_Flags, 529 /// Not really used in Fortran any more 530 IdentField_Reserved_2, 531 /// Source[4] in Fortran, do not use for C++ 532 IdentField_Reserved_3, 533 /// String describing the source location. The string is composed of 534 /// semi-colon separated fields which describe the source file, the function 535 /// and a pair of line numbers that delimit the construct. 536 IdentField_PSource 537 }; 538 539 /// Schedule types for 'omp for' loops (these enumerators are taken from 540 /// the enum sched_type in kmp.h). 541 enum OpenMPSchedType { 542 /// Lower bound for default (unordered) versions. 543 OMP_sch_lower = 32, 544 OMP_sch_static_chunked = 33, 545 OMP_sch_static = 34, 546 OMP_sch_dynamic_chunked = 35, 547 OMP_sch_guided_chunked = 36, 548 OMP_sch_runtime = 37, 549 OMP_sch_auto = 38, 550 /// static with chunk adjustment (e.g., simd) 551 OMP_sch_static_balanced_chunked = 45, 552 /// Lower bound for 'ordered' versions. 553 OMP_ord_lower = 64, 554 OMP_ord_static_chunked = 65, 555 OMP_ord_static = 66, 556 OMP_ord_dynamic_chunked = 67, 557 OMP_ord_guided_chunked = 68, 558 OMP_ord_runtime = 69, 559 OMP_ord_auto = 70, 560 OMP_sch_default = OMP_sch_static, 561 /// dist_schedule types 562 OMP_dist_sch_static_chunked = 91, 563 OMP_dist_sch_static = 92, 564 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers. 565 /// Set if the monotonic schedule modifier was present. 566 OMP_sch_modifier_monotonic = (1 << 29), 567 /// Set if the nonmonotonic schedule modifier was present. 568 OMP_sch_modifier_nonmonotonic = (1 << 30), 569 }; 570 571 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP 572 /// region. 573 class CleanupTy final : public EHScopeStack::Cleanup { 574 PrePostActionTy *Action; 575 576 public: 577 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {} 578 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 579 if (!CGF.HaveInsertPoint()) 580 return; 581 Action->Exit(CGF); 582 } 583 }; 584 585 } // anonymous namespace 586 587 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const { 588 CodeGenFunction::RunCleanupsScope Scope(CGF); 589 if (PrePostAction) { 590 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction); 591 Callback(CodeGen, CGF, *PrePostAction); 592 } else { 593 PrePostActionTy Action; 594 Callback(CodeGen, CGF, Action); 595 } 596 } 597 598 /// Check if the combiner is a call to UDR combiner and if it is so return the 599 /// UDR decl used for reduction. 600 static const OMPDeclareReductionDecl * 601 getReductionInit(const Expr *ReductionOp) { 602 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 603 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 604 if (const auto *DRE = 605 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 606 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) 607 return DRD; 608 return nullptr; 609 } 610 611 static void emitInitWithReductionInitializer(CodeGenFunction &CGF, 612 const OMPDeclareReductionDecl *DRD, 613 const Expr *InitOp, 614 Address Private, Address Original, 615 QualType Ty) { 616 if (DRD->getInitializer()) { 617 std::pair<llvm::Function *, llvm::Function *> Reduction = 618 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 619 const auto *CE = cast<CallExpr>(InitOp); 620 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee()); 621 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts(); 622 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts(); 623 const auto *LHSDRE = 624 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr()); 625 const auto *RHSDRE = 626 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr()); 627 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 628 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), 629 [=]() { return Private; }); 630 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), 631 [=]() { return Original; }); 632 (void)PrivateScope.Privatize(); 633 RValue Func = RValue::get(Reduction.second); 634 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 635 CGF.EmitIgnoredExpr(InitOp); 636 } else { 637 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty); 638 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"}); 639 auto *GV = new llvm::GlobalVariable( 640 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true, 641 llvm::GlobalValue::PrivateLinkage, Init, Name); 642 LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty); 643 RValue InitRVal; 644 switch (CGF.getEvaluationKind(Ty)) { 645 case TEK_Scalar: 646 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation()); 647 break; 648 case TEK_Complex: 649 InitRVal = 650 RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation())); 651 break; 652 case TEK_Aggregate: 653 InitRVal = RValue::getAggregate(LV.getAddress(CGF)); 654 break; 655 } 656 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue); 657 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal); 658 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 659 /*IsInitializer=*/false); 660 } 661 } 662 663 /// Emit initialization of arrays of complex types. 664 /// \param DestAddr Address of the array. 665 /// \param Type Type of array. 666 /// \param Init Initial expression of array. 667 /// \param SrcAddr Address of the original array. 668 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, 669 QualType Type, bool EmitDeclareReductionInit, 670 const Expr *Init, 671 const OMPDeclareReductionDecl *DRD, 672 Address SrcAddr = Address::invalid()) { 673 // Perform element-by-element initialization. 674 QualType ElementTy; 675 676 // Drill down to the base element type on both arrays. 677 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 678 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr); 679 DestAddr = 680 CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType()); 681 if (DRD) 682 SrcAddr = 683 CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType()); 684 685 llvm::Value *SrcBegin = nullptr; 686 if (DRD) 687 SrcBegin = SrcAddr.getPointer(); 688 llvm::Value *DestBegin = DestAddr.getPointer(); 689 // Cast from pointer to array type to pointer to single element. 690 llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements); 691 // The basic structure here is a while-do loop. 692 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body"); 693 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done"); 694 llvm::Value *IsEmpty = 695 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty"); 696 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 697 698 // Enter the loop body, making that address the current address. 699 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 700 CGF.EmitBlock(BodyBB); 701 702 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 703 704 llvm::PHINode *SrcElementPHI = nullptr; 705 Address SrcElementCurrent = Address::invalid(); 706 if (DRD) { 707 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2, 708 "omp.arraycpy.srcElementPast"); 709 SrcElementPHI->addIncoming(SrcBegin, EntryBB); 710 SrcElementCurrent = 711 Address(SrcElementPHI, 712 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 713 } 714 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI( 715 DestBegin->getType(), 2, "omp.arraycpy.destElementPast"); 716 DestElementPHI->addIncoming(DestBegin, EntryBB); 717 Address DestElementCurrent = 718 Address(DestElementPHI, 719 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 720 721 // Emit copy. 722 { 723 CodeGenFunction::RunCleanupsScope InitScope(CGF); 724 if (EmitDeclareReductionInit) { 725 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent, 726 SrcElementCurrent, ElementTy); 727 } else 728 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(), 729 /*IsInitializer=*/false); 730 } 731 732 if (DRD) { 733 // Shift the address forward by one element. 734 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32( 735 SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 736 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock()); 737 } 738 739 // Shift the address forward by one element. 740 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32( 741 DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 742 // Check whether we've reached the end. 743 llvm::Value *Done = 744 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done"); 745 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 746 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock()); 747 748 // Done. 749 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 750 } 751 752 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) { 753 return CGF.EmitOMPSharedLValue(E); 754 } 755 756 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF, 757 const Expr *E) { 758 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E)) 759 return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false); 760 return LValue(); 761 } 762 763 void ReductionCodeGen::emitAggregateInitialization( 764 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 765 const OMPDeclareReductionDecl *DRD) { 766 // Emit VarDecl with copy init for arrays. 767 // Get the address of the original variable captured in current 768 // captured region. 769 const auto *PrivateVD = 770 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 771 bool EmitDeclareReductionInit = 772 DRD && (DRD->getInitializer() || !PrivateVD->hasInit()); 773 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(), 774 EmitDeclareReductionInit, 775 EmitDeclareReductionInit ? ClausesData[N].ReductionOp 776 : PrivateVD->getInit(), 777 DRD, SharedLVal.getAddress(CGF)); 778 } 779 780 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds, 781 ArrayRef<const Expr *> Origs, 782 ArrayRef<const Expr *> Privates, 783 ArrayRef<const Expr *> ReductionOps) { 784 ClausesData.reserve(Shareds.size()); 785 SharedAddresses.reserve(Shareds.size()); 786 Sizes.reserve(Shareds.size()); 787 BaseDecls.reserve(Shareds.size()); 788 const auto *IOrig = Origs.begin(); 789 const auto *IPriv = Privates.begin(); 790 const auto *IRed = ReductionOps.begin(); 791 for (const Expr *Ref : Shareds) { 792 ClausesData.emplace_back(Ref, *IOrig, *IPriv, *IRed); 793 std::advance(IOrig, 1); 794 std::advance(IPriv, 1); 795 std::advance(IRed, 1); 796 } 797 } 798 799 void ReductionCodeGen::emitSharedOrigLValue(CodeGenFunction &CGF, unsigned N) { 800 assert(SharedAddresses.size() == N && OrigAddresses.size() == N && 801 "Number of generated lvalues must be exactly N."); 802 LValue First = emitSharedLValue(CGF, ClausesData[N].Shared); 803 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Shared); 804 SharedAddresses.emplace_back(First, Second); 805 if (ClausesData[N].Shared == ClausesData[N].Ref) { 806 OrigAddresses.emplace_back(First, Second); 807 } else { 808 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref); 809 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref); 810 OrigAddresses.emplace_back(First, Second); 811 } 812 } 813 814 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) { 815 const auto *PrivateVD = 816 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 817 QualType PrivateType = PrivateVD->getType(); 818 bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref); 819 if (!PrivateType->isVariablyModifiedType()) { 820 Sizes.emplace_back( 821 CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType()), 822 nullptr); 823 return; 824 } 825 llvm::Value *Size; 826 llvm::Value *SizeInChars; 827 auto *ElemType = 828 cast<llvm::PointerType>(OrigAddresses[N].first.getPointer(CGF)->getType()) 829 ->getElementType(); 830 auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType); 831 if (AsArraySection) { 832 Size = CGF.Builder.CreatePtrDiff(OrigAddresses[N].second.getPointer(CGF), 833 OrigAddresses[N].first.getPointer(CGF)); 834 Size = CGF.Builder.CreateNUWAdd( 835 Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1)); 836 SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf); 837 } else { 838 SizeInChars = 839 CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType()); 840 Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf); 841 } 842 Sizes.emplace_back(SizeInChars, Size); 843 CodeGenFunction::OpaqueValueMapping OpaqueMap( 844 CGF, 845 cast<OpaqueValueExpr>( 846 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 847 RValue::get(Size)); 848 CGF.EmitVariablyModifiedType(PrivateType); 849 } 850 851 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N, 852 llvm::Value *Size) { 853 const auto *PrivateVD = 854 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 855 QualType PrivateType = PrivateVD->getType(); 856 if (!PrivateType->isVariablyModifiedType()) { 857 assert(!Size && !Sizes[N].second && 858 "Size should be nullptr for non-variably modified reduction " 859 "items."); 860 return; 861 } 862 CodeGenFunction::OpaqueValueMapping OpaqueMap( 863 CGF, 864 cast<OpaqueValueExpr>( 865 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 866 RValue::get(Size)); 867 CGF.EmitVariablyModifiedType(PrivateType); 868 } 869 870 void ReductionCodeGen::emitInitialization( 871 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 872 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) { 873 assert(SharedAddresses.size() > N && "No variable was generated"); 874 const auto *PrivateVD = 875 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 876 const OMPDeclareReductionDecl *DRD = 877 getReductionInit(ClausesData[N].ReductionOp); 878 QualType PrivateType = PrivateVD->getType(); 879 PrivateAddr = CGF.Builder.CreateElementBitCast( 880 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 881 QualType SharedType = SharedAddresses[N].first.getType(); 882 SharedLVal = CGF.MakeAddrLValue( 883 CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(CGF), 884 CGF.ConvertTypeForMem(SharedType)), 885 SharedType, SharedAddresses[N].first.getBaseInfo(), 886 CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType)); 887 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) { 888 if (DRD && DRD->getInitializer()) 889 (void)DefaultInit(CGF); 890 emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD); 891 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) { 892 (void)DefaultInit(CGF); 893 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp, 894 PrivateAddr, SharedLVal.getAddress(CGF), 895 SharedLVal.getType()); 896 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() && 897 !CGF.isTrivialInitializer(PrivateVD->getInit())) { 898 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr, 899 PrivateVD->getType().getQualifiers(), 900 /*IsInitializer=*/false); 901 } 902 } 903 904 bool ReductionCodeGen::needCleanups(unsigned N) { 905 const auto *PrivateVD = 906 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 907 QualType PrivateType = PrivateVD->getType(); 908 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 909 return DTorKind != QualType::DK_none; 910 } 911 912 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N, 913 Address PrivateAddr) { 914 const auto *PrivateVD = 915 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 916 QualType PrivateType = PrivateVD->getType(); 917 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 918 if (needCleanups(N)) { 919 PrivateAddr = CGF.Builder.CreateElementBitCast( 920 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 921 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType); 922 } 923 } 924 925 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 926 LValue BaseLV) { 927 BaseTy = BaseTy.getNonReferenceType(); 928 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 929 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 930 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) { 931 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy); 932 } else { 933 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy); 934 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal); 935 } 936 BaseTy = BaseTy->getPointeeType(); 937 } 938 return CGF.MakeAddrLValue( 939 CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF), 940 CGF.ConvertTypeForMem(ElTy)), 941 BaseLV.getType(), BaseLV.getBaseInfo(), 942 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType())); 943 } 944 945 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 946 llvm::Type *BaseLVType, CharUnits BaseLVAlignment, 947 llvm::Value *Addr) { 948 Address Tmp = Address::invalid(); 949 Address TopTmp = Address::invalid(); 950 Address MostTopTmp = Address::invalid(); 951 BaseTy = BaseTy.getNonReferenceType(); 952 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 953 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 954 Tmp = CGF.CreateMemTemp(BaseTy); 955 if (TopTmp.isValid()) 956 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp); 957 else 958 MostTopTmp = Tmp; 959 TopTmp = Tmp; 960 BaseTy = BaseTy->getPointeeType(); 961 } 962 llvm::Type *Ty = BaseLVType; 963 if (Tmp.isValid()) 964 Ty = Tmp.getElementType(); 965 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty); 966 if (Tmp.isValid()) { 967 CGF.Builder.CreateStore(Addr, Tmp); 968 return MostTopTmp; 969 } 970 return Address(Addr, BaseLVAlignment); 971 } 972 973 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) { 974 const VarDecl *OrigVD = nullptr; 975 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) { 976 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 977 while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base)) 978 Base = TempOASE->getBase()->IgnoreParenImpCasts(); 979 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 980 Base = TempASE->getBase()->IgnoreParenImpCasts(); 981 DE = cast<DeclRefExpr>(Base); 982 OrigVD = cast<VarDecl>(DE->getDecl()); 983 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) { 984 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts(); 985 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 986 Base = TempASE->getBase()->IgnoreParenImpCasts(); 987 DE = cast<DeclRefExpr>(Base); 988 OrigVD = cast<VarDecl>(DE->getDecl()); 989 } 990 return OrigVD; 991 } 992 993 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, 994 Address PrivateAddr) { 995 const DeclRefExpr *DE; 996 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) { 997 BaseDecls.emplace_back(OrigVD); 998 LValue OriginalBaseLValue = CGF.EmitLValue(DE); 999 LValue BaseLValue = 1000 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(), 1001 OriginalBaseLValue); 1002 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff( 1003 BaseLValue.getPointer(CGF), SharedAddresses[N].first.getPointer(CGF)); 1004 llvm::Value *PrivatePointer = 1005 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 1006 PrivateAddr.getPointer(), 1007 SharedAddresses[N].first.getAddress(CGF).getType()); 1008 llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment); 1009 return castToBase(CGF, OrigVD->getType(), 1010 SharedAddresses[N].first.getType(), 1011 OriginalBaseLValue.getAddress(CGF).getType(), 1012 OriginalBaseLValue.getAlignment(), Ptr); 1013 } 1014 BaseDecls.emplace_back( 1015 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl())); 1016 return PrivateAddr; 1017 } 1018 1019 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const { 1020 const OMPDeclareReductionDecl *DRD = 1021 getReductionInit(ClausesData[N].ReductionOp); 1022 return DRD && DRD->getInitializer(); 1023 } 1024 1025 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) { 1026 return CGF.EmitLoadOfPointerLValue( 1027 CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1028 getThreadIDVariable()->getType()->castAs<PointerType>()); 1029 } 1030 1031 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) { 1032 if (!CGF.HaveInsertPoint()) 1033 return; 1034 // 1.2.2 OpenMP Language Terminology 1035 // Structured block - An executable statement with a single entry at the 1036 // top and a single exit at the bottom. 1037 // The point of exit cannot be a branch out of the structured block. 1038 // longjmp() and throw() must not violate the entry/exit criteria. 1039 CGF.EHStack.pushTerminate(); 1040 CodeGen(CGF); 1041 CGF.EHStack.popTerminate(); 1042 } 1043 1044 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue( 1045 CodeGenFunction &CGF) { 1046 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1047 getThreadIDVariable()->getType(), 1048 AlignmentSource::Decl); 1049 } 1050 1051 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC, 1052 QualType FieldTy) { 1053 auto *Field = FieldDecl::Create( 1054 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy, 1055 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()), 1056 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit); 1057 Field->setAccess(AS_public); 1058 DC->addDecl(Field); 1059 return Field; 1060 } 1061 1062 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator, 1063 StringRef Separator) 1064 : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator), 1065 OMPBuilder(CGM.getModule()), OffloadEntriesInfoManager(CGM) { 1066 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8); 1067 1068 // Initialize Types used in OpenMPIRBuilder from OMPKinds.def 1069 OMPBuilder.initialize(); 1070 loadOffloadInfoMetadata(); 1071 } 1072 1073 void CGOpenMPRuntime::clear() { 1074 InternalVars.clear(); 1075 // Clean non-target variable declarations possibly used only in debug info. 1076 for (const auto &Data : EmittedNonTargetVariables) { 1077 if (!Data.getValue().pointsToAliveValue()) 1078 continue; 1079 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue()); 1080 if (!GV) 1081 continue; 1082 if (!GV->isDeclaration() || GV->getNumUses() > 0) 1083 continue; 1084 GV->eraseFromParent(); 1085 } 1086 } 1087 1088 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const { 1089 SmallString<128> Buffer; 1090 llvm::raw_svector_ostream OS(Buffer); 1091 StringRef Sep = FirstSeparator; 1092 for (StringRef Part : Parts) { 1093 OS << Sep << Part; 1094 Sep = Separator; 1095 } 1096 return std::string(OS.str()); 1097 } 1098 1099 static llvm::Function * 1100 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, 1101 const Expr *CombinerInitializer, const VarDecl *In, 1102 const VarDecl *Out, bool IsCombiner) { 1103 // void .omp_combiner.(Ty *in, Ty *out); 1104 ASTContext &C = CGM.getContext(); 1105 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 1106 FunctionArgList Args; 1107 ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(), 1108 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1109 ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(), 1110 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1111 Args.push_back(&OmpOutParm); 1112 Args.push_back(&OmpInParm); 1113 const CGFunctionInfo &FnInfo = 1114 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 1115 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 1116 std::string Name = CGM.getOpenMPRuntime().getName( 1117 {IsCombiner ? "omp_combiner" : "omp_initializer", ""}); 1118 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 1119 Name, &CGM.getModule()); 1120 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 1121 if (CGM.getLangOpts().Optimize) { 1122 Fn->removeFnAttr(llvm::Attribute::NoInline); 1123 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 1124 Fn->addFnAttr(llvm::Attribute::AlwaysInline); 1125 } 1126 CodeGenFunction CGF(CGM); 1127 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions. 1128 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions. 1129 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(), 1130 Out->getLocation()); 1131 CodeGenFunction::OMPPrivateScope Scope(CGF); 1132 Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm); 1133 Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() { 1134 return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>()) 1135 .getAddress(CGF); 1136 }); 1137 Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm); 1138 Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() { 1139 return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>()) 1140 .getAddress(CGF); 1141 }); 1142 (void)Scope.Privatize(); 1143 if (!IsCombiner && Out->hasInit() && 1144 !CGF.isTrivialInitializer(Out->getInit())) { 1145 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out), 1146 Out->getType().getQualifiers(), 1147 /*IsInitializer=*/true); 1148 } 1149 if (CombinerInitializer) 1150 CGF.EmitIgnoredExpr(CombinerInitializer); 1151 Scope.ForceCleanup(); 1152 CGF.FinishFunction(); 1153 return Fn; 1154 } 1155 1156 void CGOpenMPRuntime::emitUserDefinedReduction( 1157 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) { 1158 if (UDRMap.count(D) > 0) 1159 return; 1160 llvm::Function *Combiner = emitCombinerOrInitializer( 1161 CGM, D->getType(), D->getCombiner(), 1162 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()), 1163 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()), 1164 /*IsCombiner=*/true); 1165 llvm::Function *Initializer = nullptr; 1166 if (const Expr *Init = D->getInitializer()) { 1167 Initializer = emitCombinerOrInitializer( 1168 CGM, D->getType(), 1169 D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init 1170 : nullptr, 1171 cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()), 1172 cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()), 1173 /*IsCombiner=*/false); 1174 } 1175 UDRMap.try_emplace(D, Combiner, Initializer); 1176 if (CGF) { 1177 auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn); 1178 Decls.second.push_back(D); 1179 } 1180 } 1181 1182 std::pair<llvm::Function *, llvm::Function *> 1183 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) { 1184 auto I = UDRMap.find(D); 1185 if (I != UDRMap.end()) 1186 return I->second; 1187 emitUserDefinedReduction(/*CGF=*/nullptr, D); 1188 return UDRMap.lookup(D); 1189 } 1190 1191 namespace { 1192 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR 1193 // Builder if one is present. 1194 struct PushAndPopStackRAII { 1195 PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF, 1196 bool HasCancel) 1197 : OMPBuilder(OMPBuilder) { 1198 if (!OMPBuilder) 1199 return; 1200 1201 // The following callback is the crucial part of clangs cleanup process. 1202 // 1203 // NOTE: 1204 // Once the OpenMPIRBuilder is used to create parallel regions (and 1205 // similar), the cancellation destination (Dest below) is determined via 1206 // IP. That means if we have variables to finalize we split the block at IP, 1207 // use the new block (=BB) as destination to build a JumpDest (via 1208 // getJumpDestInCurrentScope(BB)) which then is fed to 1209 // EmitBranchThroughCleanup. Furthermore, there will not be the need 1210 // to push & pop an FinalizationInfo object. 1211 // The FiniCB will still be needed but at the point where the 1212 // OpenMPIRBuilder is asked to construct a parallel (or similar) construct. 1213 auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) { 1214 assert(IP.getBlock()->end() == IP.getPoint() && 1215 "Clang CG should cause non-terminated block!"); 1216 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1217 CGF.Builder.restoreIP(IP); 1218 CodeGenFunction::JumpDest Dest = 1219 CGF.getOMPCancelDestination(OMPD_parallel); 1220 CGF.EmitBranchThroughCleanup(Dest); 1221 }; 1222 1223 // TODO: Remove this once we emit parallel regions through the 1224 // OpenMPIRBuilder as it can do this setup internally. 1225 llvm::OpenMPIRBuilder::FinalizationInfo FI( 1226 {FiniCB, OMPD_parallel, HasCancel}); 1227 OMPBuilder->pushFinalizationCB(std::move(FI)); 1228 } 1229 ~PushAndPopStackRAII() { 1230 if (OMPBuilder) 1231 OMPBuilder->popFinalizationCB(); 1232 } 1233 llvm::OpenMPIRBuilder *OMPBuilder; 1234 }; 1235 } // namespace 1236 1237 static llvm::Function *emitParallelOrTeamsOutlinedFunction( 1238 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, 1239 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, 1240 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) { 1241 assert(ThreadIDVar->getType()->isPointerType() && 1242 "thread id variable must be of type kmp_int32 *"); 1243 CodeGenFunction CGF(CGM, true); 1244 bool HasCancel = false; 1245 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D)) 1246 HasCancel = OPD->hasCancel(); 1247 else if (const auto *OPD = dyn_cast<OMPTargetParallelDirective>(&D)) 1248 HasCancel = OPD->hasCancel(); 1249 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D)) 1250 HasCancel = OPSD->hasCancel(); 1251 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D)) 1252 HasCancel = OPFD->hasCancel(); 1253 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D)) 1254 HasCancel = OPFD->hasCancel(); 1255 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D)) 1256 HasCancel = OPFD->hasCancel(); 1257 else if (const auto *OPFD = 1258 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D)) 1259 HasCancel = OPFD->hasCancel(); 1260 else if (const auto *OPFD = 1261 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D)) 1262 HasCancel = OPFD->hasCancel(); 1263 1264 // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new 1265 // parallel region to make cancellation barriers work properly. 1266 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder(); 1267 PushAndPopStackRAII PSR(&OMPBuilder, CGF, HasCancel); 1268 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind, 1269 HasCancel, OutlinedHelperName); 1270 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1271 return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc()); 1272 } 1273 1274 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction( 1275 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1276 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1277 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel); 1278 return emitParallelOrTeamsOutlinedFunction( 1279 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1280 } 1281 1282 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction( 1283 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1284 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1285 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams); 1286 return emitParallelOrTeamsOutlinedFunction( 1287 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1288 } 1289 1290 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction( 1291 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1292 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 1293 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 1294 bool Tied, unsigned &NumberOfParts) { 1295 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF, 1296 PrePostActionTy &) { 1297 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc()); 1298 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 1299 llvm::Value *TaskArgs[] = { 1300 UpLoc, ThreadID, 1301 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar), 1302 TaskTVar->getType()->castAs<PointerType>()) 1303 .getPointer(CGF)}; 1304 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 1305 CGM.getModule(), OMPRTL___kmpc_omp_task), 1306 TaskArgs); 1307 }; 1308 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar, 1309 UntiedCodeGen); 1310 CodeGen.setAction(Action); 1311 assert(!ThreadIDVar->getType()->isPointerType() && 1312 "thread id variable must be of type kmp_int32 for tasks"); 1313 const OpenMPDirectiveKind Region = 1314 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop 1315 : OMPD_task; 1316 const CapturedStmt *CS = D.getCapturedStmt(Region); 1317 bool HasCancel = false; 1318 if (const auto *TD = dyn_cast<OMPTaskDirective>(&D)) 1319 HasCancel = TD->hasCancel(); 1320 else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D)) 1321 HasCancel = TD->hasCancel(); 1322 else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D)) 1323 HasCancel = TD->hasCancel(); 1324 else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D)) 1325 HasCancel = TD->hasCancel(); 1326 1327 CodeGenFunction CGF(CGM, true); 1328 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, 1329 InnermostKind, HasCancel, Action); 1330 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1331 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS); 1332 if (!Tied) 1333 NumberOfParts = Action.getNumberOfParts(); 1334 return Res; 1335 } 1336 1337 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM, 1338 const RecordDecl *RD, const CGRecordLayout &RL, 1339 ArrayRef<llvm::Constant *> Data) { 1340 llvm::StructType *StructTy = RL.getLLVMType(); 1341 unsigned PrevIdx = 0; 1342 ConstantInitBuilder CIBuilder(CGM); 1343 auto DI = Data.begin(); 1344 for (const FieldDecl *FD : RD->fields()) { 1345 unsigned Idx = RL.getLLVMFieldNo(FD); 1346 // Fill the alignment. 1347 for (unsigned I = PrevIdx; I < Idx; ++I) 1348 Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I))); 1349 PrevIdx = Idx + 1; 1350 Fields.add(*DI); 1351 ++DI; 1352 } 1353 } 1354 1355 template <class... As> 1356 static llvm::GlobalVariable * 1357 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant, 1358 ArrayRef<llvm::Constant *> Data, const Twine &Name, 1359 As &&... Args) { 1360 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1361 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1362 ConstantInitBuilder CIBuilder(CGM); 1363 ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType()); 1364 buildStructValue(Fields, CGM, RD, RL, Data); 1365 return Fields.finishAndCreateGlobal( 1366 Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant, 1367 std::forward<As>(Args)...); 1368 } 1369 1370 template <typename T> 1371 static void 1372 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty, 1373 ArrayRef<llvm::Constant *> Data, 1374 T &Parent) { 1375 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1376 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1377 ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType()); 1378 buildStructValue(Fields, CGM, RD, RL, Data); 1379 Fields.finishAndAddTo(Parent); 1380 } 1381 1382 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF, 1383 bool AtCurrentPoint) { 1384 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1385 assert(!Elem.second.ServiceInsertPt && "Insert point is set already."); 1386 1387 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty); 1388 if (AtCurrentPoint) { 1389 Elem.second.ServiceInsertPt = new llvm::BitCastInst( 1390 Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock()); 1391 } else { 1392 Elem.second.ServiceInsertPt = 1393 new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt"); 1394 Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt); 1395 } 1396 } 1397 1398 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) { 1399 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1400 if (Elem.second.ServiceInsertPt) { 1401 llvm::Instruction *Ptr = Elem.second.ServiceInsertPt; 1402 Elem.second.ServiceInsertPt = nullptr; 1403 Ptr->eraseFromParent(); 1404 } 1405 } 1406 1407 static StringRef getIdentStringFromSourceLocation(CodeGenFunction &CGF, 1408 SourceLocation Loc, 1409 SmallString<128> &Buffer) { 1410 llvm::raw_svector_ostream OS(Buffer); 1411 // Build debug location 1412 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1413 OS << ";" << PLoc.getFilename() << ";"; 1414 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1415 OS << FD->getQualifiedNameAsString(); 1416 OS << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;"; 1417 return OS.str(); 1418 } 1419 1420 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF, 1421 SourceLocation Loc, 1422 unsigned Flags) { 1423 llvm::Constant *SrcLocStr; 1424 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo || 1425 Loc.isInvalid()) { 1426 SrcLocStr = OMPBuilder.getOrCreateDefaultSrcLocStr(); 1427 } else { 1428 std::string FunctionName = ""; 1429 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1430 FunctionName = FD->getQualifiedNameAsString(); 1431 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1432 const char *FileName = PLoc.getFilename(); 1433 unsigned Line = PLoc.getLine(); 1434 unsigned Column = PLoc.getColumn(); 1435 SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(FunctionName.c_str(), FileName, 1436 Line, Column); 1437 } 1438 unsigned Reserved2Flags = getDefaultLocationReserved2Flags(); 1439 return OMPBuilder.getOrCreateIdent(SrcLocStr, llvm::omp::IdentFlag(Flags), 1440 Reserved2Flags); 1441 } 1442 1443 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF, 1444 SourceLocation Loc) { 1445 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1446 // If the OpenMPIRBuilder is used we need to use it for all thread id calls as 1447 // the clang invariants used below might be broken. 1448 if (CGM.getLangOpts().OpenMPIRBuilder) { 1449 SmallString<128> Buffer; 1450 OMPBuilder.updateToLocation(CGF.Builder.saveIP()); 1451 auto *SrcLocStr = OMPBuilder.getOrCreateSrcLocStr( 1452 getIdentStringFromSourceLocation(CGF, Loc, Buffer)); 1453 return OMPBuilder.getOrCreateThreadID( 1454 OMPBuilder.getOrCreateIdent(SrcLocStr)); 1455 } 1456 1457 llvm::Value *ThreadID = nullptr; 1458 // Check whether we've already cached a load of the thread id in this 1459 // function. 1460 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1461 if (I != OpenMPLocThreadIDMap.end()) { 1462 ThreadID = I->second.ThreadID; 1463 if (ThreadID != nullptr) 1464 return ThreadID; 1465 } 1466 // If exceptions are enabled, do not use parameter to avoid possible crash. 1467 if (auto *OMPRegionInfo = 1468 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 1469 if (OMPRegionInfo->getThreadIDVariable()) { 1470 // Check if this an outlined function with thread id passed as argument. 1471 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF); 1472 llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent(); 1473 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions || 1474 !CGF.getLangOpts().CXXExceptions || 1475 CGF.Builder.GetInsertBlock() == TopBlock || 1476 !isa<llvm::Instruction>(LVal.getPointer(CGF)) || 1477 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1478 TopBlock || 1479 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1480 CGF.Builder.GetInsertBlock()) { 1481 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc); 1482 // If value loaded in entry block, cache it and use it everywhere in 1483 // function. 1484 if (CGF.Builder.GetInsertBlock() == TopBlock) { 1485 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1486 Elem.second.ThreadID = ThreadID; 1487 } 1488 return ThreadID; 1489 } 1490 } 1491 } 1492 1493 // This is not an outlined function region - need to call __kmpc_int32 1494 // kmpc_global_thread_num(ident_t *loc). 1495 // Generate thread id value and cache this value for use across the 1496 // function. 1497 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1498 if (!Elem.second.ServiceInsertPt) 1499 setLocThreadIdInsertPt(CGF); 1500 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1501 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1502 llvm::CallInst *Call = CGF.Builder.CreateCall( 1503 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 1504 OMPRTL___kmpc_global_thread_num), 1505 emitUpdateLocation(CGF, Loc)); 1506 Call->setCallingConv(CGF.getRuntimeCC()); 1507 Elem.second.ThreadID = Call; 1508 return Call; 1509 } 1510 1511 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) { 1512 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1513 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) { 1514 clearLocThreadIdInsertPt(CGF); 1515 OpenMPLocThreadIDMap.erase(CGF.CurFn); 1516 } 1517 if (FunctionUDRMap.count(CGF.CurFn) > 0) { 1518 for(const auto *D : FunctionUDRMap[CGF.CurFn]) 1519 UDRMap.erase(D); 1520 FunctionUDRMap.erase(CGF.CurFn); 1521 } 1522 auto I = FunctionUDMMap.find(CGF.CurFn); 1523 if (I != FunctionUDMMap.end()) { 1524 for(const auto *D : I->second) 1525 UDMMap.erase(D); 1526 FunctionUDMMap.erase(I); 1527 } 1528 LastprivateConditionalToTypes.erase(CGF.CurFn); 1529 FunctionToUntiedTaskStackMap.erase(CGF.CurFn); 1530 } 1531 1532 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() { 1533 return OMPBuilder.IdentPtr; 1534 } 1535 1536 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() { 1537 if (!Kmpc_MicroTy) { 1538 // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...) 1539 llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty), 1540 llvm::PointerType::getUnqual(CGM.Int32Ty)}; 1541 Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true); 1542 } 1543 return llvm::PointerType::getUnqual(Kmpc_MicroTy); 1544 } 1545 1546 llvm::FunctionCallee 1547 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) { 1548 assert((IVSize == 32 || IVSize == 64) && 1549 "IV size is not compatible with the omp runtime"); 1550 StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4" 1551 : "__kmpc_for_static_init_4u") 1552 : (IVSigned ? "__kmpc_for_static_init_8" 1553 : "__kmpc_for_static_init_8u"); 1554 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 1555 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 1556 llvm::Type *TypeParams[] = { 1557 getIdentTyPointerTy(), // loc 1558 CGM.Int32Ty, // tid 1559 CGM.Int32Ty, // schedtype 1560 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 1561 PtrTy, // p_lower 1562 PtrTy, // p_upper 1563 PtrTy, // p_stride 1564 ITy, // incr 1565 ITy // chunk 1566 }; 1567 auto *FnTy = 1568 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1569 return CGM.CreateRuntimeFunction(FnTy, Name); 1570 } 1571 1572 llvm::FunctionCallee 1573 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) { 1574 assert((IVSize == 32 || IVSize == 64) && 1575 "IV size is not compatible with the omp runtime"); 1576 StringRef Name = 1577 IVSize == 32 1578 ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u") 1579 : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u"); 1580 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 1581 llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc 1582 CGM.Int32Ty, // tid 1583 CGM.Int32Ty, // schedtype 1584 ITy, // lower 1585 ITy, // upper 1586 ITy, // stride 1587 ITy // chunk 1588 }; 1589 auto *FnTy = 1590 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1591 return CGM.CreateRuntimeFunction(FnTy, Name); 1592 } 1593 1594 llvm::FunctionCallee 1595 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) { 1596 assert((IVSize == 32 || IVSize == 64) && 1597 "IV size is not compatible with the omp runtime"); 1598 StringRef Name = 1599 IVSize == 32 1600 ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u") 1601 : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u"); 1602 llvm::Type *TypeParams[] = { 1603 getIdentTyPointerTy(), // loc 1604 CGM.Int32Ty, // tid 1605 }; 1606 auto *FnTy = 1607 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1608 return CGM.CreateRuntimeFunction(FnTy, Name); 1609 } 1610 1611 llvm::FunctionCallee 1612 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) { 1613 assert((IVSize == 32 || IVSize == 64) && 1614 "IV size is not compatible with the omp runtime"); 1615 StringRef Name = 1616 IVSize == 32 1617 ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u") 1618 : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u"); 1619 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 1620 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 1621 llvm::Type *TypeParams[] = { 1622 getIdentTyPointerTy(), // loc 1623 CGM.Int32Ty, // tid 1624 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 1625 PtrTy, // p_lower 1626 PtrTy, // p_upper 1627 PtrTy // p_stride 1628 }; 1629 auto *FnTy = 1630 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1631 return CGM.CreateRuntimeFunction(FnTy, Name); 1632 } 1633 1634 /// Obtain information that uniquely identifies a target entry. This 1635 /// consists of the file and device IDs as well as line number associated with 1636 /// the relevant entry source location. 1637 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc, 1638 unsigned &DeviceID, unsigned &FileID, 1639 unsigned &LineNum) { 1640 SourceManager &SM = C.getSourceManager(); 1641 1642 // The loc should be always valid and have a file ID (the user cannot use 1643 // #pragma directives in macros) 1644 1645 assert(Loc.isValid() && "Source location is expected to be always valid."); 1646 1647 PresumedLoc PLoc = SM.getPresumedLoc(Loc); 1648 assert(PLoc.isValid() && "Source location is expected to be always valid."); 1649 1650 llvm::sys::fs::UniqueID ID; 1651 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) 1652 SM.getDiagnostics().Report(diag::err_cannot_open_file) 1653 << PLoc.getFilename() << EC.message(); 1654 1655 DeviceID = ID.getDevice(); 1656 FileID = ID.getFile(); 1657 LineNum = PLoc.getLine(); 1658 } 1659 1660 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) { 1661 if (CGM.getLangOpts().OpenMPSimd) 1662 return Address::invalid(); 1663 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 1664 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 1665 if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link || 1666 (*Res == OMPDeclareTargetDeclAttr::MT_To && 1667 HasRequiresUnifiedSharedMemory))) { 1668 SmallString<64> PtrName; 1669 { 1670 llvm::raw_svector_ostream OS(PtrName); 1671 OS << CGM.getMangledName(GlobalDecl(VD)); 1672 if (!VD->isExternallyVisible()) { 1673 unsigned DeviceID, FileID, Line; 1674 getTargetEntryUniqueInfo(CGM.getContext(), 1675 VD->getCanonicalDecl()->getBeginLoc(), 1676 DeviceID, FileID, Line); 1677 OS << llvm::format("_%x", FileID); 1678 } 1679 OS << "_decl_tgt_ref_ptr"; 1680 } 1681 llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName); 1682 if (!Ptr) { 1683 QualType PtrTy = CGM.getContext().getPointerType(VD->getType()); 1684 Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy), 1685 PtrName); 1686 1687 auto *GV = cast<llvm::GlobalVariable>(Ptr); 1688 GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 1689 1690 if (!CGM.getLangOpts().OpenMPIsDevice) 1691 GV->setInitializer(CGM.GetAddrOfGlobal(VD)); 1692 registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr)); 1693 } 1694 return Address(Ptr, CGM.getContext().getDeclAlign(VD)); 1695 } 1696 return Address::invalid(); 1697 } 1698 1699 llvm::Constant * 1700 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) { 1701 assert(!CGM.getLangOpts().OpenMPUseTLS || 1702 !CGM.getContext().getTargetInfo().isTLSSupported()); 1703 // Lookup the entry, lazily creating it if necessary. 1704 std::string Suffix = getName({"cache", ""}); 1705 return getOrCreateInternalVariable( 1706 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix)); 1707 } 1708 1709 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 1710 const VarDecl *VD, 1711 Address VDAddr, 1712 SourceLocation Loc) { 1713 if (CGM.getLangOpts().OpenMPUseTLS && 1714 CGM.getContext().getTargetInfo().isTLSSupported()) 1715 return VDAddr; 1716 1717 llvm::Type *VarTy = VDAddr.getElementType(); 1718 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 1719 CGF.Builder.CreatePointerCast(VDAddr.getPointer(), 1720 CGM.Int8PtrTy), 1721 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)), 1722 getOrCreateThreadPrivateCache(VD)}; 1723 return Address(CGF.EmitRuntimeCall( 1724 OMPBuilder.getOrCreateRuntimeFunction( 1725 CGM.getModule(), OMPRTL___kmpc_threadprivate_cached), 1726 Args), 1727 VDAddr.getAlignment()); 1728 } 1729 1730 void CGOpenMPRuntime::emitThreadPrivateVarInit( 1731 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, 1732 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) { 1733 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime 1734 // library. 1735 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc); 1736 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 1737 CGM.getModule(), OMPRTL___kmpc_global_thread_num), 1738 OMPLoc); 1739 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor) 1740 // to register constructor/destructor for variable. 1741 llvm::Value *Args[] = { 1742 OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy), 1743 Ctor, CopyCtor, Dtor}; 1744 CGF.EmitRuntimeCall( 1745 OMPBuilder.getOrCreateRuntimeFunction( 1746 CGM.getModule(), OMPRTL___kmpc_threadprivate_register), 1747 Args); 1748 } 1749 1750 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition( 1751 const VarDecl *VD, Address VDAddr, SourceLocation Loc, 1752 bool PerformInit, CodeGenFunction *CGF) { 1753 if (CGM.getLangOpts().OpenMPUseTLS && 1754 CGM.getContext().getTargetInfo().isTLSSupported()) 1755 return nullptr; 1756 1757 VD = VD->getDefinition(CGM.getContext()); 1758 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) { 1759 QualType ASTTy = VD->getType(); 1760 1761 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr; 1762 const Expr *Init = VD->getAnyInitializer(); 1763 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 1764 // Generate function that re-emits the declaration's initializer into the 1765 // threadprivate copy of the variable VD 1766 CodeGenFunction CtorCGF(CGM); 1767 FunctionArgList Args; 1768 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 1769 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 1770 ImplicitParamDecl::Other); 1771 Args.push_back(&Dst); 1772 1773 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 1774 CGM.getContext().VoidPtrTy, Args); 1775 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1776 std::string Name = getName({"__kmpc_global_ctor_", ""}); 1777 llvm::Function *Fn = 1778 CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc); 1779 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI, 1780 Args, Loc, Loc); 1781 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar( 1782 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 1783 CGM.getContext().VoidPtrTy, Dst.getLocation()); 1784 Address Arg = Address(ArgVal, VDAddr.getAlignment()); 1785 Arg = CtorCGF.Builder.CreateElementBitCast( 1786 Arg, CtorCGF.ConvertTypeForMem(ASTTy)); 1787 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(), 1788 /*IsInitializer=*/true); 1789 ArgVal = CtorCGF.EmitLoadOfScalar( 1790 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 1791 CGM.getContext().VoidPtrTy, Dst.getLocation()); 1792 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue); 1793 CtorCGF.FinishFunction(); 1794 Ctor = Fn; 1795 } 1796 if (VD->getType().isDestructedType() != QualType::DK_none) { 1797 // Generate function that emits destructor call for the threadprivate copy 1798 // of the variable VD 1799 CodeGenFunction DtorCGF(CGM); 1800 FunctionArgList Args; 1801 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 1802 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 1803 ImplicitParamDecl::Other); 1804 Args.push_back(&Dst); 1805 1806 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 1807 CGM.getContext().VoidTy, Args); 1808 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1809 std::string Name = getName({"__kmpc_global_dtor_", ""}); 1810 llvm::Function *Fn = 1811 CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc); 1812 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 1813 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args, 1814 Loc, Loc); 1815 // Create a scope with an artificial location for the body of this function. 1816 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 1817 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar( 1818 DtorCGF.GetAddrOfLocalVar(&Dst), 1819 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation()); 1820 DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy, 1821 DtorCGF.getDestroyer(ASTTy.isDestructedType()), 1822 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 1823 DtorCGF.FinishFunction(); 1824 Dtor = Fn; 1825 } 1826 // Do not emit init function if it is not required. 1827 if (!Ctor && !Dtor) 1828 return nullptr; 1829 1830 llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1831 auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs, 1832 /*isVarArg=*/false) 1833 ->getPointerTo(); 1834 // Copying constructor for the threadprivate variable. 1835 // Must be NULL - reserved by runtime, but currently it requires that this 1836 // parameter is always NULL. Otherwise it fires assertion. 1837 CopyCtor = llvm::Constant::getNullValue(CopyCtorTy); 1838 if (Ctor == nullptr) { 1839 auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 1840 /*isVarArg=*/false) 1841 ->getPointerTo(); 1842 Ctor = llvm::Constant::getNullValue(CtorTy); 1843 } 1844 if (Dtor == nullptr) { 1845 auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, 1846 /*isVarArg=*/false) 1847 ->getPointerTo(); 1848 Dtor = llvm::Constant::getNullValue(DtorTy); 1849 } 1850 if (!CGF) { 1851 auto *InitFunctionTy = 1852 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false); 1853 std::string Name = getName({"__omp_threadprivate_init_", ""}); 1854 llvm::Function *InitFunction = CGM.CreateGlobalInitOrCleanUpFunction( 1855 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction()); 1856 CodeGenFunction InitCGF(CGM); 1857 FunctionArgList ArgList; 1858 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction, 1859 CGM.getTypes().arrangeNullaryFunction(), ArgList, 1860 Loc, Loc); 1861 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 1862 InitCGF.FinishFunction(); 1863 return InitFunction; 1864 } 1865 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 1866 } 1867 return nullptr; 1868 } 1869 1870 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD, 1871 llvm::GlobalVariable *Addr, 1872 bool PerformInit) { 1873 if (CGM.getLangOpts().OMPTargetTriples.empty() && 1874 !CGM.getLangOpts().OpenMPIsDevice) 1875 return false; 1876 Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 1877 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 1878 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 1879 (*Res == OMPDeclareTargetDeclAttr::MT_To && 1880 HasRequiresUnifiedSharedMemory)) 1881 return CGM.getLangOpts().OpenMPIsDevice; 1882 VD = VD->getDefinition(CGM.getContext()); 1883 assert(VD && "Unknown VarDecl"); 1884 1885 if (!DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second) 1886 return CGM.getLangOpts().OpenMPIsDevice; 1887 1888 QualType ASTTy = VD->getType(); 1889 SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc(); 1890 1891 // Produce the unique prefix to identify the new target regions. We use 1892 // the source location of the variable declaration which we know to not 1893 // conflict with any target region. 1894 unsigned DeviceID; 1895 unsigned FileID; 1896 unsigned Line; 1897 getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line); 1898 SmallString<128> Buffer, Out; 1899 { 1900 llvm::raw_svector_ostream OS(Buffer); 1901 OS << "__omp_offloading_" << llvm::format("_%x", DeviceID) 1902 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 1903 } 1904 1905 const Expr *Init = VD->getAnyInitializer(); 1906 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 1907 llvm::Constant *Ctor; 1908 llvm::Constant *ID; 1909 if (CGM.getLangOpts().OpenMPIsDevice) { 1910 // Generate function that re-emits the declaration's initializer into 1911 // the threadprivate copy of the variable VD 1912 CodeGenFunction CtorCGF(CGM); 1913 1914 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 1915 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1916 llvm::Function *Fn = CGM.CreateGlobalInitOrCleanUpFunction( 1917 FTy, Twine(Buffer, "_ctor"), FI, Loc); 1918 auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF); 1919 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 1920 FunctionArgList(), Loc, Loc); 1921 auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF); 1922 CtorCGF.EmitAnyExprToMem(Init, 1923 Address(Addr, CGM.getContext().getDeclAlign(VD)), 1924 Init->getType().getQualifiers(), 1925 /*IsInitializer=*/true); 1926 CtorCGF.FinishFunction(); 1927 Ctor = Fn; 1928 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 1929 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor)); 1930 } else { 1931 Ctor = new llvm::GlobalVariable( 1932 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 1933 llvm::GlobalValue::PrivateLinkage, 1934 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor")); 1935 ID = Ctor; 1936 } 1937 1938 // Register the information for the entry associated with the constructor. 1939 Out.clear(); 1940 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 1941 DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor, 1942 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor); 1943 } 1944 if (VD->getType().isDestructedType() != QualType::DK_none) { 1945 llvm::Constant *Dtor; 1946 llvm::Constant *ID; 1947 if (CGM.getLangOpts().OpenMPIsDevice) { 1948 // Generate function that emits destructor call for the threadprivate 1949 // copy of the variable VD 1950 CodeGenFunction DtorCGF(CGM); 1951 1952 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 1953 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1954 llvm::Function *Fn = CGM.CreateGlobalInitOrCleanUpFunction( 1955 FTy, Twine(Buffer, "_dtor"), FI, Loc); 1956 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 1957 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 1958 FunctionArgList(), Loc, Loc); 1959 // Create a scope with an artificial location for the body of this 1960 // function. 1961 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 1962 DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)), 1963 ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()), 1964 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 1965 DtorCGF.FinishFunction(); 1966 Dtor = Fn; 1967 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 1968 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor)); 1969 } else { 1970 Dtor = new llvm::GlobalVariable( 1971 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 1972 llvm::GlobalValue::PrivateLinkage, 1973 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor")); 1974 ID = Dtor; 1975 } 1976 // Register the information for the entry associated with the destructor. 1977 Out.clear(); 1978 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 1979 DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor, 1980 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor); 1981 } 1982 return CGM.getLangOpts().OpenMPIsDevice; 1983 } 1984 1985 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, 1986 QualType VarType, 1987 StringRef Name) { 1988 std::string Suffix = getName({"artificial", ""}); 1989 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType); 1990 llvm::Value *GAddr = 1991 getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix)); 1992 if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS && 1993 CGM.getTarget().isTLSSupported()) { 1994 cast<llvm::GlobalVariable>(GAddr)->setThreadLocal(/*Val=*/true); 1995 return Address(GAddr, CGM.getContext().getTypeAlignInChars(VarType)); 1996 } 1997 std::string CacheSuffix = getName({"cache", ""}); 1998 llvm::Value *Args[] = { 1999 emitUpdateLocation(CGF, SourceLocation()), 2000 getThreadID(CGF, SourceLocation()), 2001 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy), 2002 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy, 2003 /*isSigned=*/false), 2004 getOrCreateInternalVariable( 2005 CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))}; 2006 return Address( 2007 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2008 CGF.EmitRuntimeCall( 2009 OMPBuilder.getOrCreateRuntimeFunction( 2010 CGM.getModule(), OMPRTL___kmpc_threadprivate_cached), 2011 Args), 2012 VarLVType->getPointerTo(/*AddrSpace=*/0)), 2013 CGM.getContext().getTypeAlignInChars(VarType)); 2014 } 2015 2016 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond, 2017 const RegionCodeGenTy &ThenGen, 2018 const RegionCodeGenTy &ElseGen) { 2019 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 2020 2021 // If the condition constant folds and can be elided, try to avoid emitting 2022 // the condition and the dead arm of the if/else. 2023 bool CondConstant; 2024 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 2025 if (CondConstant) 2026 ThenGen(CGF); 2027 else 2028 ElseGen(CGF); 2029 return; 2030 } 2031 2032 // Otherwise, the condition did not fold, or we couldn't elide it. Just 2033 // emit the conditional branch. 2034 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then"); 2035 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else"); 2036 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end"); 2037 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0); 2038 2039 // Emit the 'then' code. 2040 CGF.EmitBlock(ThenBlock); 2041 ThenGen(CGF); 2042 CGF.EmitBranch(ContBlock); 2043 // Emit the 'else' code if present. 2044 // There is no need to emit line number for unconditional branch. 2045 (void)ApplyDebugLocation::CreateEmpty(CGF); 2046 CGF.EmitBlock(ElseBlock); 2047 ElseGen(CGF); 2048 // There is no need to emit line number for unconditional branch. 2049 (void)ApplyDebugLocation::CreateEmpty(CGF); 2050 CGF.EmitBranch(ContBlock); 2051 // Emit the continuation block for code after the if. 2052 CGF.EmitBlock(ContBlock, /*IsFinished=*/true); 2053 } 2054 2055 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, 2056 llvm::Function *OutlinedFn, 2057 ArrayRef<llvm::Value *> CapturedVars, 2058 const Expr *IfCond) { 2059 if (!CGF.HaveInsertPoint()) 2060 return; 2061 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 2062 auto &M = CGM.getModule(); 2063 auto &&ThenGen = [&M, OutlinedFn, CapturedVars, RTLoc, 2064 this](CodeGenFunction &CGF, PrePostActionTy &) { 2065 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn); 2066 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 2067 llvm::Value *Args[] = { 2068 RTLoc, 2069 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 2070 CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())}; 2071 llvm::SmallVector<llvm::Value *, 16> RealArgs; 2072 RealArgs.append(std::begin(Args), std::end(Args)); 2073 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 2074 2075 llvm::FunctionCallee RTLFn = 2076 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_fork_call); 2077 CGF.EmitRuntimeCall(RTLFn, RealArgs); 2078 }; 2079 auto &&ElseGen = [&M, OutlinedFn, CapturedVars, RTLoc, Loc, 2080 this](CodeGenFunction &CGF, PrePostActionTy &) { 2081 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 2082 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc); 2083 // Build calls: 2084 // __kmpc_serialized_parallel(&Loc, GTid); 2085 llvm::Value *Args[] = {RTLoc, ThreadID}; 2086 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2087 M, OMPRTL___kmpc_serialized_parallel), 2088 Args); 2089 2090 // OutlinedFn(>id, &zero_bound, CapturedStruct); 2091 Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc); 2092 Address ZeroAddrBound = 2093 CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty, 2094 /*Name=*/".bound.zero.addr"); 2095 CGF.InitTempAlloca(ZeroAddrBound, CGF.Builder.getInt32(/*C*/ 0)); 2096 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs; 2097 // ThreadId for serialized parallels is 0. 2098 OutlinedFnArgs.push_back(ThreadIDAddr.getPointer()); 2099 OutlinedFnArgs.push_back(ZeroAddrBound.getPointer()); 2100 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end()); 2101 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs); 2102 2103 // __kmpc_end_serialized_parallel(&Loc, GTid); 2104 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID}; 2105 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2106 M, OMPRTL___kmpc_end_serialized_parallel), 2107 EndArgs); 2108 }; 2109 if (IfCond) { 2110 emitIfClause(CGF, IfCond, ThenGen, ElseGen); 2111 } else { 2112 RegionCodeGenTy ThenRCG(ThenGen); 2113 ThenRCG(CGF); 2114 } 2115 } 2116 2117 // If we're inside an (outlined) parallel region, use the region info's 2118 // thread-ID variable (it is passed in a first argument of the outlined function 2119 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in 2120 // regular serial code region, get thread ID by calling kmp_int32 2121 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and 2122 // return the address of that temp. 2123 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF, 2124 SourceLocation Loc) { 2125 if (auto *OMPRegionInfo = 2126 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 2127 if (OMPRegionInfo->getThreadIDVariable()) 2128 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF); 2129 2130 llvm::Value *ThreadID = getThreadID(CGF, Loc); 2131 QualType Int32Ty = 2132 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true); 2133 Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp."); 2134 CGF.EmitStoreOfScalar(ThreadID, 2135 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty)); 2136 2137 return ThreadIDTemp; 2138 } 2139 2140 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable( 2141 llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) { 2142 SmallString<256> Buffer; 2143 llvm::raw_svector_ostream Out(Buffer); 2144 Out << Name; 2145 StringRef RuntimeName = Out.str(); 2146 auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first; 2147 if (Elem.second) { 2148 assert(Elem.second->getType()->getPointerElementType() == Ty && 2149 "OMP internal variable has different type than requested"); 2150 return &*Elem.second; 2151 } 2152 2153 return Elem.second = new llvm::GlobalVariable( 2154 CGM.getModule(), Ty, /*IsConstant*/ false, 2155 llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty), 2156 Elem.first(), /*InsertBefore=*/nullptr, 2157 llvm::GlobalValue::NotThreadLocal, AddressSpace); 2158 } 2159 2160 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) { 2161 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str(); 2162 std::string Name = getName({Prefix, "var"}); 2163 return getOrCreateInternalVariable(KmpCriticalNameTy, Name); 2164 } 2165 2166 namespace { 2167 /// Common pre(post)-action for different OpenMP constructs. 2168 class CommonActionTy final : public PrePostActionTy { 2169 llvm::FunctionCallee EnterCallee; 2170 ArrayRef<llvm::Value *> EnterArgs; 2171 llvm::FunctionCallee ExitCallee; 2172 ArrayRef<llvm::Value *> ExitArgs; 2173 bool Conditional; 2174 llvm::BasicBlock *ContBlock = nullptr; 2175 2176 public: 2177 CommonActionTy(llvm::FunctionCallee EnterCallee, 2178 ArrayRef<llvm::Value *> EnterArgs, 2179 llvm::FunctionCallee ExitCallee, 2180 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false) 2181 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee), 2182 ExitArgs(ExitArgs), Conditional(Conditional) {} 2183 void Enter(CodeGenFunction &CGF) override { 2184 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs); 2185 if (Conditional) { 2186 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes); 2187 auto *ThenBlock = CGF.createBasicBlock("omp_if.then"); 2188 ContBlock = CGF.createBasicBlock("omp_if.end"); 2189 // Generate the branch (If-stmt) 2190 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock); 2191 CGF.EmitBlock(ThenBlock); 2192 } 2193 } 2194 void Done(CodeGenFunction &CGF) { 2195 // Emit the rest of blocks/branches 2196 CGF.EmitBranch(ContBlock); 2197 CGF.EmitBlock(ContBlock, true); 2198 } 2199 void Exit(CodeGenFunction &CGF) override { 2200 CGF.EmitRuntimeCall(ExitCallee, ExitArgs); 2201 } 2202 }; 2203 } // anonymous namespace 2204 2205 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF, 2206 StringRef CriticalName, 2207 const RegionCodeGenTy &CriticalOpGen, 2208 SourceLocation Loc, const Expr *Hint) { 2209 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]); 2210 // CriticalOpGen(); 2211 // __kmpc_end_critical(ident_t *, gtid, Lock); 2212 // Prepare arguments and build a call to __kmpc_critical 2213 if (!CGF.HaveInsertPoint()) 2214 return; 2215 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2216 getCriticalRegionLock(CriticalName)}; 2217 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args), 2218 std::end(Args)); 2219 if (Hint) { 2220 EnterArgs.push_back(CGF.Builder.CreateIntCast( 2221 CGF.EmitScalarExpr(Hint), CGM.Int32Ty, /*isSigned=*/false)); 2222 } 2223 CommonActionTy Action( 2224 OMPBuilder.getOrCreateRuntimeFunction( 2225 CGM.getModule(), 2226 Hint ? OMPRTL___kmpc_critical_with_hint : OMPRTL___kmpc_critical), 2227 EnterArgs, 2228 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 2229 OMPRTL___kmpc_end_critical), 2230 Args); 2231 CriticalOpGen.setAction(Action); 2232 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen); 2233 } 2234 2235 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF, 2236 const RegionCodeGenTy &MasterOpGen, 2237 SourceLocation Loc) { 2238 if (!CGF.HaveInsertPoint()) 2239 return; 2240 // if(__kmpc_master(ident_t *, gtid)) { 2241 // MasterOpGen(); 2242 // __kmpc_end_master(ident_t *, gtid); 2243 // } 2244 // Prepare arguments and build a call to __kmpc_master 2245 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2246 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2247 CGM.getModule(), OMPRTL___kmpc_master), 2248 Args, 2249 OMPBuilder.getOrCreateRuntimeFunction( 2250 CGM.getModule(), OMPRTL___kmpc_end_master), 2251 Args, 2252 /*Conditional=*/true); 2253 MasterOpGen.setAction(Action); 2254 emitInlinedDirective(CGF, OMPD_master, MasterOpGen); 2255 Action.Done(CGF); 2256 } 2257 2258 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 2259 SourceLocation Loc) { 2260 if (!CGF.HaveInsertPoint()) 2261 return; 2262 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) { 2263 OMPBuilder.createTaskyield(CGF.Builder); 2264 } else { 2265 // Build call __kmpc_omp_taskyield(loc, thread_id, 0); 2266 llvm::Value *Args[] = { 2267 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2268 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)}; 2269 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2270 CGM.getModule(), OMPRTL___kmpc_omp_taskyield), 2271 Args); 2272 } 2273 2274 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 2275 Region->emitUntiedSwitch(CGF); 2276 } 2277 2278 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF, 2279 const RegionCodeGenTy &TaskgroupOpGen, 2280 SourceLocation Loc) { 2281 if (!CGF.HaveInsertPoint()) 2282 return; 2283 // __kmpc_taskgroup(ident_t *, gtid); 2284 // TaskgroupOpGen(); 2285 // __kmpc_end_taskgroup(ident_t *, gtid); 2286 // Prepare arguments and build a call to __kmpc_taskgroup 2287 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2288 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2289 CGM.getModule(), OMPRTL___kmpc_taskgroup), 2290 Args, 2291 OMPBuilder.getOrCreateRuntimeFunction( 2292 CGM.getModule(), OMPRTL___kmpc_end_taskgroup), 2293 Args); 2294 TaskgroupOpGen.setAction(Action); 2295 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen); 2296 } 2297 2298 /// Given an array of pointers to variables, project the address of a 2299 /// given variable. 2300 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, 2301 unsigned Index, const VarDecl *Var) { 2302 // Pull out the pointer to the variable. 2303 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index); 2304 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr); 2305 2306 Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var)); 2307 Addr = CGF.Builder.CreateElementBitCast( 2308 Addr, CGF.ConvertTypeForMem(Var->getType())); 2309 return Addr; 2310 } 2311 2312 static llvm::Value *emitCopyprivateCopyFunction( 2313 CodeGenModule &CGM, llvm::Type *ArgsType, 2314 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs, 2315 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps, 2316 SourceLocation Loc) { 2317 ASTContext &C = CGM.getContext(); 2318 // void copy_func(void *LHSArg, void *RHSArg); 2319 FunctionArgList Args; 2320 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 2321 ImplicitParamDecl::Other); 2322 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 2323 ImplicitParamDecl::Other); 2324 Args.push_back(&LHSArg); 2325 Args.push_back(&RHSArg); 2326 const auto &CGFI = 2327 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 2328 std::string Name = 2329 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"}); 2330 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 2331 llvm::GlobalValue::InternalLinkage, Name, 2332 &CGM.getModule()); 2333 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 2334 Fn->setDoesNotRecurse(); 2335 CodeGenFunction CGF(CGM); 2336 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 2337 // Dest = (void*[n])(LHSArg); 2338 // Src = (void*[n])(RHSArg); 2339 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2340 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 2341 ArgsType), CGF.getPointerAlign()); 2342 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2343 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 2344 ArgsType), CGF.getPointerAlign()); 2345 // *(Type0*)Dst[0] = *(Type0*)Src[0]; 2346 // *(Type1*)Dst[1] = *(Type1*)Src[1]; 2347 // ... 2348 // *(Typen*)Dst[n] = *(Typen*)Src[n]; 2349 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) { 2350 const auto *DestVar = 2351 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()); 2352 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar); 2353 2354 const auto *SrcVar = 2355 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()); 2356 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar); 2357 2358 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl(); 2359 QualType Type = VD->getType(); 2360 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]); 2361 } 2362 CGF.FinishFunction(); 2363 return Fn; 2364 } 2365 2366 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF, 2367 const RegionCodeGenTy &SingleOpGen, 2368 SourceLocation Loc, 2369 ArrayRef<const Expr *> CopyprivateVars, 2370 ArrayRef<const Expr *> SrcExprs, 2371 ArrayRef<const Expr *> DstExprs, 2372 ArrayRef<const Expr *> AssignmentOps) { 2373 if (!CGF.HaveInsertPoint()) 2374 return; 2375 assert(CopyprivateVars.size() == SrcExprs.size() && 2376 CopyprivateVars.size() == DstExprs.size() && 2377 CopyprivateVars.size() == AssignmentOps.size()); 2378 ASTContext &C = CGM.getContext(); 2379 // int32 did_it = 0; 2380 // if(__kmpc_single(ident_t *, gtid)) { 2381 // SingleOpGen(); 2382 // __kmpc_end_single(ident_t *, gtid); 2383 // did_it = 1; 2384 // } 2385 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 2386 // <copy_func>, did_it); 2387 2388 Address DidIt = Address::invalid(); 2389 if (!CopyprivateVars.empty()) { 2390 // int32 did_it = 0; 2391 QualType KmpInt32Ty = 2392 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 2393 DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it"); 2394 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt); 2395 } 2396 // Prepare arguments and build a call to __kmpc_single 2397 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2398 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2399 CGM.getModule(), OMPRTL___kmpc_single), 2400 Args, 2401 OMPBuilder.getOrCreateRuntimeFunction( 2402 CGM.getModule(), OMPRTL___kmpc_end_single), 2403 Args, 2404 /*Conditional=*/true); 2405 SingleOpGen.setAction(Action); 2406 emitInlinedDirective(CGF, OMPD_single, SingleOpGen); 2407 if (DidIt.isValid()) { 2408 // did_it = 1; 2409 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt); 2410 } 2411 Action.Done(CGF); 2412 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 2413 // <copy_func>, did_it); 2414 if (DidIt.isValid()) { 2415 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size()); 2416 QualType CopyprivateArrayTy = C.getConstantArrayType( 2417 C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 2418 /*IndexTypeQuals=*/0); 2419 // Create a list of all private variables for copyprivate. 2420 Address CopyprivateList = 2421 CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list"); 2422 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) { 2423 Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I); 2424 CGF.Builder.CreateStore( 2425 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2426 CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF), 2427 CGF.VoidPtrTy), 2428 Elem); 2429 } 2430 // Build function that copies private values from single region to all other 2431 // threads in the corresponding parallel region. 2432 llvm::Value *CpyFn = emitCopyprivateCopyFunction( 2433 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(), 2434 CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc); 2435 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy); 2436 Address CL = 2437 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList, 2438 CGF.VoidPtrTy); 2439 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt); 2440 llvm::Value *Args[] = { 2441 emitUpdateLocation(CGF, Loc), // ident_t *<loc> 2442 getThreadID(CGF, Loc), // i32 <gtid> 2443 BufSize, // size_t <buf_size> 2444 CL.getPointer(), // void *<copyprivate list> 2445 CpyFn, // void (*) (void *, void *) <copy_func> 2446 DidItVal // i32 did_it 2447 }; 2448 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2449 CGM.getModule(), OMPRTL___kmpc_copyprivate), 2450 Args); 2451 } 2452 } 2453 2454 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF, 2455 const RegionCodeGenTy &OrderedOpGen, 2456 SourceLocation Loc, bool IsThreads) { 2457 if (!CGF.HaveInsertPoint()) 2458 return; 2459 // __kmpc_ordered(ident_t *, gtid); 2460 // OrderedOpGen(); 2461 // __kmpc_end_ordered(ident_t *, gtid); 2462 // Prepare arguments and build a call to __kmpc_ordered 2463 if (IsThreads) { 2464 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2465 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2466 CGM.getModule(), OMPRTL___kmpc_ordered), 2467 Args, 2468 OMPBuilder.getOrCreateRuntimeFunction( 2469 CGM.getModule(), OMPRTL___kmpc_end_ordered), 2470 Args); 2471 OrderedOpGen.setAction(Action); 2472 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 2473 return; 2474 } 2475 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 2476 } 2477 2478 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) { 2479 unsigned Flags; 2480 if (Kind == OMPD_for) 2481 Flags = OMP_IDENT_BARRIER_IMPL_FOR; 2482 else if (Kind == OMPD_sections) 2483 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS; 2484 else if (Kind == OMPD_single) 2485 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE; 2486 else if (Kind == OMPD_barrier) 2487 Flags = OMP_IDENT_BARRIER_EXPL; 2488 else 2489 Flags = OMP_IDENT_BARRIER_IMPL; 2490 return Flags; 2491 } 2492 2493 void CGOpenMPRuntime::getDefaultScheduleAndChunk( 2494 CodeGenFunction &CGF, const OMPLoopDirective &S, 2495 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const { 2496 // Check if the loop directive is actually a doacross loop directive. In this 2497 // case choose static, 1 schedule. 2498 if (llvm::any_of( 2499 S.getClausesOfKind<OMPOrderedClause>(), 2500 [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) { 2501 ScheduleKind = OMPC_SCHEDULE_static; 2502 // Chunk size is 1 in this case. 2503 llvm::APInt ChunkSize(32, 1); 2504 ChunkExpr = IntegerLiteral::Create( 2505 CGF.getContext(), ChunkSize, 2506 CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0), 2507 SourceLocation()); 2508 } 2509 } 2510 2511 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, 2512 OpenMPDirectiveKind Kind, bool EmitChecks, 2513 bool ForceSimpleCall) { 2514 // Check if we should use the OMPBuilder 2515 auto *OMPRegionInfo = 2516 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo); 2517 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) { 2518 CGF.Builder.restoreIP(OMPBuilder.createBarrier( 2519 CGF.Builder, Kind, ForceSimpleCall, EmitChecks)); 2520 return; 2521 } 2522 2523 if (!CGF.HaveInsertPoint()) 2524 return; 2525 // Build call __kmpc_cancel_barrier(loc, thread_id); 2526 // Build call __kmpc_barrier(loc, thread_id); 2527 unsigned Flags = getDefaultFlagsForBarriers(Kind); 2528 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc, 2529 // thread_id); 2530 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags), 2531 getThreadID(CGF, Loc)}; 2532 if (OMPRegionInfo) { 2533 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) { 2534 llvm::Value *Result = CGF.EmitRuntimeCall( 2535 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 2536 OMPRTL___kmpc_cancel_barrier), 2537 Args); 2538 if (EmitChecks) { 2539 // if (__kmpc_cancel_barrier()) { 2540 // exit from construct; 2541 // } 2542 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 2543 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 2544 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 2545 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 2546 CGF.EmitBlock(ExitBB); 2547 // exit from construct; 2548 CodeGenFunction::JumpDest CancelDestination = 2549 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 2550 CGF.EmitBranchThroughCleanup(CancelDestination); 2551 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 2552 } 2553 return; 2554 } 2555 } 2556 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2557 CGM.getModule(), OMPRTL___kmpc_barrier), 2558 Args); 2559 } 2560 2561 /// Map the OpenMP loop schedule to the runtime enumeration. 2562 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, 2563 bool Chunked, bool Ordered) { 2564 switch (ScheduleKind) { 2565 case OMPC_SCHEDULE_static: 2566 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked) 2567 : (Ordered ? OMP_ord_static : OMP_sch_static); 2568 case OMPC_SCHEDULE_dynamic: 2569 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked; 2570 case OMPC_SCHEDULE_guided: 2571 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked; 2572 case OMPC_SCHEDULE_runtime: 2573 return Ordered ? OMP_ord_runtime : OMP_sch_runtime; 2574 case OMPC_SCHEDULE_auto: 2575 return Ordered ? OMP_ord_auto : OMP_sch_auto; 2576 case OMPC_SCHEDULE_unknown: 2577 assert(!Chunked && "chunk was specified but schedule kind not known"); 2578 return Ordered ? OMP_ord_static : OMP_sch_static; 2579 } 2580 llvm_unreachable("Unexpected runtime schedule"); 2581 } 2582 2583 /// Map the OpenMP distribute schedule to the runtime enumeration. 2584 static OpenMPSchedType 2585 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) { 2586 // only static is allowed for dist_schedule 2587 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static; 2588 } 2589 2590 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, 2591 bool Chunked) const { 2592 OpenMPSchedType Schedule = 2593 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 2594 return Schedule == OMP_sch_static; 2595 } 2596 2597 bool CGOpenMPRuntime::isStaticNonchunked( 2598 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 2599 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 2600 return Schedule == OMP_dist_sch_static; 2601 } 2602 2603 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, 2604 bool Chunked) const { 2605 OpenMPSchedType Schedule = 2606 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 2607 return Schedule == OMP_sch_static_chunked; 2608 } 2609 2610 bool CGOpenMPRuntime::isStaticChunked( 2611 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 2612 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 2613 return Schedule == OMP_dist_sch_static_chunked; 2614 } 2615 2616 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const { 2617 OpenMPSchedType Schedule = 2618 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false); 2619 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here"); 2620 return Schedule != OMP_sch_static; 2621 } 2622 2623 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule, 2624 OpenMPScheduleClauseModifier M1, 2625 OpenMPScheduleClauseModifier M2) { 2626 int Modifier = 0; 2627 switch (M1) { 2628 case OMPC_SCHEDULE_MODIFIER_monotonic: 2629 Modifier = OMP_sch_modifier_monotonic; 2630 break; 2631 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 2632 Modifier = OMP_sch_modifier_nonmonotonic; 2633 break; 2634 case OMPC_SCHEDULE_MODIFIER_simd: 2635 if (Schedule == OMP_sch_static_chunked) 2636 Schedule = OMP_sch_static_balanced_chunked; 2637 break; 2638 case OMPC_SCHEDULE_MODIFIER_last: 2639 case OMPC_SCHEDULE_MODIFIER_unknown: 2640 break; 2641 } 2642 switch (M2) { 2643 case OMPC_SCHEDULE_MODIFIER_monotonic: 2644 Modifier = OMP_sch_modifier_monotonic; 2645 break; 2646 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 2647 Modifier = OMP_sch_modifier_nonmonotonic; 2648 break; 2649 case OMPC_SCHEDULE_MODIFIER_simd: 2650 if (Schedule == OMP_sch_static_chunked) 2651 Schedule = OMP_sch_static_balanced_chunked; 2652 break; 2653 case OMPC_SCHEDULE_MODIFIER_last: 2654 case OMPC_SCHEDULE_MODIFIER_unknown: 2655 break; 2656 } 2657 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription. 2658 // If the static schedule kind is specified or if the ordered clause is 2659 // specified, and if the nonmonotonic modifier is not specified, the effect is 2660 // as if the monotonic modifier is specified. Otherwise, unless the monotonic 2661 // modifier is specified, the effect is as if the nonmonotonic modifier is 2662 // specified. 2663 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) { 2664 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static || 2665 Schedule == OMP_sch_static_balanced_chunked || 2666 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static || 2667 Schedule == OMP_dist_sch_static_chunked || 2668 Schedule == OMP_dist_sch_static)) 2669 Modifier = OMP_sch_modifier_nonmonotonic; 2670 } 2671 return Schedule | Modifier; 2672 } 2673 2674 void CGOpenMPRuntime::emitForDispatchInit( 2675 CodeGenFunction &CGF, SourceLocation Loc, 2676 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 2677 bool Ordered, const DispatchRTInput &DispatchValues) { 2678 if (!CGF.HaveInsertPoint()) 2679 return; 2680 OpenMPSchedType Schedule = getRuntimeSchedule( 2681 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered); 2682 assert(Ordered || 2683 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked && 2684 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked && 2685 Schedule != OMP_sch_static_balanced_chunked)); 2686 // Call __kmpc_dispatch_init( 2687 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule, 2688 // kmp_int[32|64] lower, kmp_int[32|64] upper, 2689 // kmp_int[32|64] stride, kmp_int[32|64] chunk); 2690 2691 // If the Chunk was not specified in the clause - use default value 1. 2692 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk 2693 : CGF.Builder.getIntN(IVSize, 1); 2694 llvm::Value *Args[] = { 2695 emitUpdateLocation(CGF, Loc), 2696 getThreadID(CGF, Loc), 2697 CGF.Builder.getInt32(addMonoNonMonoModifier( 2698 CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type 2699 DispatchValues.LB, // Lower 2700 DispatchValues.UB, // Upper 2701 CGF.Builder.getIntN(IVSize, 1), // Stride 2702 Chunk // Chunk 2703 }; 2704 CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args); 2705 } 2706 2707 static void emitForStaticInitCall( 2708 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, 2709 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, 2710 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, 2711 const CGOpenMPRuntime::StaticRTInput &Values) { 2712 if (!CGF.HaveInsertPoint()) 2713 return; 2714 2715 assert(!Values.Ordered); 2716 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked || 2717 Schedule == OMP_sch_static_balanced_chunked || 2718 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked || 2719 Schedule == OMP_dist_sch_static || 2720 Schedule == OMP_dist_sch_static_chunked); 2721 2722 // Call __kmpc_for_static_init( 2723 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype, 2724 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower, 2725 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride, 2726 // kmp_int[32|64] incr, kmp_int[32|64] chunk); 2727 llvm::Value *Chunk = Values.Chunk; 2728 if (Chunk == nullptr) { 2729 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static || 2730 Schedule == OMP_dist_sch_static) && 2731 "expected static non-chunked schedule"); 2732 // If the Chunk was not specified in the clause - use default value 1. 2733 Chunk = CGF.Builder.getIntN(Values.IVSize, 1); 2734 } else { 2735 assert((Schedule == OMP_sch_static_chunked || 2736 Schedule == OMP_sch_static_balanced_chunked || 2737 Schedule == OMP_ord_static_chunked || 2738 Schedule == OMP_dist_sch_static_chunked) && 2739 "expected static chunked schedule"); 2740 } 2741 llvm::Value *Args[] = { 2742 UpdateLocation, 2743 ThreadId, 2744 CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1, 2745 M2)), // Schedule type 2746 Values.IL.getPointer(), // &isLastIter 2747 Values.LB.getPointer(), // &LB 2748 Values.UB.getPointer(), // &UB 2749 Values.ST.getPointer(), // &Stride 2750 CGF.Builder.getIntN(Values.IVSize, 1), // Incr 2751 Chunk // Chunk 2752 }; 2753 CGF.EmitRuntimeCall(ForStaticInitFunction, Args); 2754 } 2755 2756 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF, 2757 SourceLocation Loc, 2758 OpenMPDirectiveKind DKind, 2759 const OpenMPScheduleTy &ScheduleKind, 2760 const StaticRTInput &Values) { 2761 OpenMPSchedType ScheduleNum = getRuntimeSchedule( 2762 ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered); 2763 assert(isOpenMPWorksharingDirective(DKind) && 2764 "Expected loop-based or sections-based directive."); 2765 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc, 2766 isOpenMPLoopDirective(DKind) 2767 ? OMP_IDENT_WORK_LOOP 2768 : OMP_IDENT_WORK_SECTIONS); 2769 llvm::Value *ThreadId = getThreadID(CGF, Loc); 2770 llvm::FunctionCallee StaticInitFunction = 2771 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 2772 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 2773 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 2774 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values); 2775 } 2776 2777 void CGOpenMPRuntime::emitDistributeStaticInit( 2778 CodeGenFunction &CGF, SourceLocation Loc, 2779 OpenMPDistScheduleClauseKind SchedKind, 2780 const CGOpenMPRuntime::StaticRTInput &Values) { 2781 OpenMPSchedType ScheduleNum = 2782 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr); 2783 llvm::Value *UpdatedLocation = 2784 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE); 2785 llvm::Value *ThreadId = getThreadID(CGF, Loc); 2786 llvm::FunctionCallee StaticInitFunction = 2787 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 2788 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 2789 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown, 2790 OMPC_SCHEDULE_MODIFIER_unknown, Values); 2791 } 2792 2793 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF, 2794 SourceLocation Loc, 2795 OpenMPDirectiveKind DKind) { 2796 if (!CGF.HaveInsertPoint()) 2797 return; 2798 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid); 2799 llvm::Value *Args[] = { 2800 emitUpdateLocation(CGF, Loc, 2801 isOpenMPDistributeDirective(DKind) 2802 ? OMP_IDENT_WORK_DISTRIBUTE 2803 : isOpenMPLoopDirective(DKind) 2804 ? OMP_IDENT_WORK_LOOP 2805 : OMP_IDENT_WORK_SECTIONS), 2806 getThreadID(CGF, Loc)}; 2807 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 2808 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2809 CGM.getModule(), OMPRTL___kmpc_for_static_fini), 2810 Args); 2811 } 2812 2813 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 2814 SourceLocation Loc, 2815 unsigned IVSize, 2816 bool IVSigned) { 2817 if (!CGF.HaveInsertPoint()) 2818 return; 2819 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid); 2820 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2821 CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args); 2822 } 2823 2824 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF, 2825 SourceLocation Loc, unsigned IVSize, 2826 bool IVSigned, Address IL, 2827 Address LB, Address UB, 2828 Address ST) { 2829 // Call __kmpc_dispatch_next( 2830 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, 2831 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper, 2832 // kmp_int[32|64] *p_stride); 2833 llvm::Value *Args[] = { 2834 emitUpdateLocation(CGF, Loc), 2835 getThreadID(CGF, Loc), 2836 IL.getPointer(), // &isLastIter 2837 LB.getPointer(), // &Lower 2838 UB.getPointer(), // &Upper 2839 ST.getPointer() // &Stride 2840 }; 2841 llvm::Value *Call = 2842 CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args); 2843 return CGF.EmitScalarConversion( 2844 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1), 2845 CGF.getContext().BoolTy, Loc); 2846 } 2847 2848 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 2849 llvm::Value *NumThreads, 2850 SourceLocation Loc) { 2851 if (!CGF.HaveInsertPoint()) 2852 return; 2853 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads) 2854 llvm::Value *Args[] = { 2855 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2856 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)}; 2857 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2858 CGM.getModule(), OMPRTL___kmpc_push_num_threads), 2859 Args); 2860 } 2861 2862 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF, 2863 ProcBindKind ProcBind, 2864 SourceLocation Loc) { 2865 if (!CGF.HaveInsertPoint()) 2866 return; 2867 assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value."); 2868 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind) 2869 llvm::Value *Args[] = { 2870 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2871 llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)}; 2872 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2873 CGM.getModule(), OMPRTL___kmpc_push_proc_bind), 2874 Args); 2875 } 2876 2877 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>, 2878 SourceLocation Loc, llvm::AtomicOrdering AO) { 2879 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) { 2880 OMPBuilder.createFlush(CGF.Builder); 2881 } else { 2882 if (!CGF.HaveInsertPoint()) 2883 return; 2884 // Build call void __kmpc_flush(ident_t *loc) 2885 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2886 CGM.getModule(), OMPRTL___kmpc_flush), 2887 emitUpdateLocation(CGF, Loc)); 2888 } 2889 } 2890 2891 namespace { 2892 /// Indexes of fields for type kmp_task_t. 2893 enum KmpTaskTFields { 2894 /// List of shared variables. 2895 KmpTaskTShareds, 2896 /// Task routine. 2897 KmpTaskTRoutine, 2898 /// Partition id for the untied tasks. 2899 KmpTaskTPartId, 2900 /// Function with call of destructors for private variables. 2901 Data1, 2902 /// Task priority. 2903 Data2, 2904 /// (Taskloops only) Lower bound. 2905 KmpTaskTLowerBound, 2906 /// (Taskloops only) Upper bound. 2907 KmpTaskTUpperBound, 2908 /// (Taskloops only) Stride. 2909 KmpTaskTStride, 2910 /// (Taskloops only) Is last iteration flag. 2911 KmpTaskTLastIter, 2912 /// (Taskloops only) Reduction data. 2913 KmpTaskTReductions, 2914 }; 2915 } // anonymous namespace 2916 2917 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const { 2918 return OffloadEntriesTargetRegion.empty() && 2919 OffloadEntriesDeviceGlobalVar.empty(); 2920 } 2921 2922 /// Initialize target region entry. 2923 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 2924 initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 2925 StringRef ParentName, unsigned LineNum, 2926 unsigned Order) { 2927 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 2928 "only required for the device " 2929 "code generation."); 2930 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = 2931 OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr, 2932 OMPTargetRegionEntryTargetRegion); 2933 ++OffloadingEntriesNum; 2934 } 2935 2936 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 2937 registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 2938 StringRef ParentName, unsigned LineNum, 2939 llvm::Constant *Addr, llvm::Constant *ID, 2940 OMPTargetRegionEntryKind Flags) { 2941 // If we are emitting code for a target, the entry is already initialized, 2942 // only has to be registered. 2943 if (CGM.getLangOpts().OpenMPIsDevice) { 2944 if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) { 2945 unsigned DiagID = CGM.getDiags().getCustomDiagID( 2946 DiagnosticsEngine::Error, 2947 "Unable to find target region on line '%0' in the device code."); 2948 CGM.getDiags().Report(DiagID) << LineNum; 2949 return; 2950 } 2951 auto &Entry = 2952 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum]; 2953 assert(Entry.isValid() && "Entry not initialized!"); 2954 Entry.setAddress(Addr); 2955 Entry.setID(ID); 2956 Entry.setFlags(Flags); 2957 } else { 2958 if (Flags == 2959 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion && 2960 hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum, 2961 /*IgnoreAddressId*/ true)) 2962 return; 2963 assert(!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum) && 2964 "Target region entry already registered!"); 2965 OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags); 2966 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry; 2967 ++OffloadingEntriesNum; 2968 } 2969 } 2970 2971 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo( 2972 unsigned DeviceID, unsigned FileID, StringRef ParentName, unsigned LineNum, 2973 bool IgnoreAddressId) const { 2974 auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID); 2975 if (PerDevice == OffloadEntriesTargetRegion.end()) 2976 return false; 2977 auto PerFile = PerDevice->second.find(FileID); 2978 if (PerFile == PerDevice->second.end()) 2979 return false; 2980 auto PerParentName = PerFile->second.find(ParentName); 2981 if (PerParentName == PerFile->second.end()) 2982 return false; 2983 auto PerLine = PerParentName->second.find(LineNum); 2984 if (PerLine == PerParentName->second.end()) 2985 return false; 2986 // Fail if this entry is already registered. 2987 if (!IgnoreAddressId && 2988 (PerLine->second.getAddress() || PerLine->second.getID())) 2989 return false; 2990 return true; 2991 } 2992 2993 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo( 2994 const OffloadTargetRegionEntryInfoActTy &Action) { 2995 // Scan all target region entries and perform the provided action. 2996 for (const auto &D : OffloadEntriesTargetRegion) 2997 for (const auto &F : D.second) 2998 for (const auto &P : F.second) 2999 for (const auto &L : P.second) 3000 Action(D.first, F.first, P.first(), L.first, L.second); 3001 } 3002 3003 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3004 initializeDeviceGlobalVarEntryInfo(StringRef Name, 3005 OMPTargetGlobalVarEntryKind Flags, 3006 unsigned Order) { 3007 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3008 "only required for the device " 3009 "code generation."); 3010 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags); 3011 ++OffloadingEntriesNum; 3012 } 3013 3014 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3015 registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr, 3016 CharUnits VarSize, 3017 OMPTargetGlobalVarEntryKind Flags, 3018 llvm::GlobalValue::LinkageTypes Linkage) { 3019 if (CGM.getLangOpts().OpenMPIsDevice) { 3020 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3021 assert(Entry.isValid() && Entry.getFlags() == Flags && 3022 "Entry not initialized!"); 3023 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 3024 "Resetting with the new address."); 3025 if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) { 3026 if (Entry.getVarSize().isZero()) { 3027 Entry.setVarSize(VarSize); 3028 Entry.setLinkage(Linkage); 3029 } 3030 return; 3031 } 3032 Entry.setVarSize(VarSize); 3033 Entry.setLinkage(Linkage); 3034 Entry.setAddress(Addr); 3035 } else { 3036 if (hasDeviceGlobalVarEntryInfo(VarName)) { 3037 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3038 assert(Entry.isValid() && Entry.getFlags() == Flags && 3039 "Entry not initialized!"); 3040 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 3041 "Resetting with the new address."); 3042 if (Entry.getVarSize().isZero()) { 3043 Entry.setVarSize(VarSize); 3044 Entry.setLinkage(Linkage); 3045 } 3046 return; 3047 } 3048 OffloadEntriesDeviceGlobalVar.try_emplace( 3049 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage); 3050 ++OffloadingEntriesNum; 3051 } 3052 } 3053 3054 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3055 actOnDeviceGlobalVarEntriesInfo( 3056 const OffloadDeviceGlobalVarEntryInfoActTy &Action) { 3057 // Scan all target region entries and perform the provided action. 3058 for (const auto &E : OffloadEntriesDeviceGlobalVar) 3059 Action(E.getKey(), E.getValue()); 3060 } 3061 3062 void CGOpenMPRuntime::createOffloadEntry( 3063 llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags, 3064 llvm::GlobalValue::LinkageTypes Linkage) { 3065 StringRef Name = Addr->getName(); 3066 llvm::Module &M = CGM.getModule(); 3067 llvm::LLVMContext &C = M.getContext(); 3068 3069 // Create constant string with the name. 3070 llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name); 3071 3072 std::string StringName = getName({"omp_offloading", "entry_name"}); 3073 auto *Str = new llvm::GlobalVariable( 3074 M, StrPtrInit->getType(), /*isConstant=*/true, 3075 llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName); 3076 Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 3077 3078 llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy), 3079 llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy), 3080 llvm::ConstantInt::get(CGM.SizeTy, Size), 3081 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 3082 llvm::ConstantInt::get(CGM.Int32Ty, 0)}; 3083 std::string EntryName = getName({"omp_offloading", "entry", ""}); 3084 llvm::GlobalVariable *Entry = createGlobalStruct( 3085 CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data, 3086 Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage); 3087 3088 // The entry has to be created in the section the linker expects it to be. 3089 Entry->setSection("omp_offloading_entries"); 3090 } 3091 3092 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() { 3093 // Emit the offloading entries and metadata so that the device codegen side 3094 // can easily figure out what to emit. The produced metadata looks like 3095 // this: 3096 // 3097 // !omp_offload.info = !{!1, ...} 3098 // 3099 // Right now we only generate metadata for function that contain target 3100 // regions. 3101 3102 // If we are in simd mode or there are no entries, we don't need to do 3103 // anything. 3104 if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty()) 3105 return; 3106 3107 llvm::Module &M = CGM.getModule(); 3108 llvm::LLVMContext &C = M.getContext(); 3109 SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *, 3110 SourceLocation, StringRef>, 3111 16> 3112 OrderedEntries(OffloadEntriesInfoManager.size()); 3113 llvm::SmallVector<StringRef, 16> ParentFunctions( 3114 OffloadEntriesInfoManager.size()); 3115 3116 // Auxiliary methods to create metadata values and strings. 3117 auto &&GetMDInt = [this](unsigned V) { 3118 return llvm::ConstantAsMetadata::get( 3119 llvm::ConstantInt::get(CGM.Int32Ty, V)); 3120 }; 3121 3122 auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); }; 3123 3124 // Create the offloading info metadata node. 3125 llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info"); 3126 3127 // Create function that emits metadata for each target region entry; 3128 auto &&TargetRegionMetadataEmitter = 3129 [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt, 3130 &GetMDString]( 3131 unsigned DeviceID, unsigned FileID, StringRef ParentName, 3132 unsigned Line, 3133 const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) { 3134 // Generate metadata for target regions. Each entry of this metadata 3135 // contains: 3136 // - Entry 0 -> Kind of this type of metadata (0). 3137 // - Entry 1 -> Device ID of the file where the entry was identified. 3138 // - Entry 2 -> File ID of the file where the entry was identified. 3139 // - Entry 3 -> Mangled name of the function where the entry was 3140 // identified. 3141 // - Entry 4 -> Line in the file where the entry was identified. 3142 // - Entry 5 -> Order the entry was created. 3143 // The first element of the metadata node is the kind. 3144 llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID), 3145 GetMDInt(FileID), GetMDString(ParentName), 3146 GetMDInt(Line), GetMDInt(E.getOrder())}; 3147 3148 SourceLocation Loc; 3149 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(), 3150 E = CGM.getContext().getSourceManager().fileinfo_end(); 3151 I != E; ++I) { 3152 if (I->getFirst()->getUniqueID().getDevice() == DeviceID && 3153 I->getFirst()->getUniqueID().getFile() == FileID) { 3154 Loc = CGM.getContext().getSourceManager().translateFileLineCol( 3155 I->getFirst(), Line, 1); 3156 break; 3157 } 3158 } 3159 // Save this entry in the right position of the ordered entries array. 3160 OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName); 3161 ParentFunctions[E.getOrder()] = ParentName; 3162 3163 // Add metadata to the named metadata node. 3164 MD->addOperand(llvm::MDNode::get(C, Ops)); 3165 }; 3166 3167 OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo( 3168 TargetRegionMetadataEmitter); 3169 3170 // Create function that emits metadata for each device global variable entry; 3171 auto &&DeviceGlobalVarMetadataEmitter = 3172 [&C, &OrderedEntries, &GetMDInt, &GetMDString, 3173 MD](StringRef MangledName, 3174 const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar 3175 &E) { 3176 // Generate metadata for global variables. Each entry of this metadata 3177 // contains: 3178 // - Entry 0 -> Kind of this type of metadata (1). 3179 // - Entry 1 -> Mangled name of the variable. 3180 // - Entry 2 -> Declare target kind. 3181 // - Entry 3 -> Order the entry was created. 3182 // The first element of the metadata node is the kind. 3183 llvm::Metadata *Ops[] = { 3184 GetMDInt(E.getKind()), GetMDString(MangledName), 3185 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())}; 3186 3187 // Save this entry in the right position of the ordered entries array. 3188 OrderedEntries[E.getOrder()] = 3189 std::make_tuple(&E, SourceLocation(), MangledName); 3190 3191 // Add metadata to the named metadata node. 3192 MD->addOperand(llvm::MDNode::get(C, Ops)); 3193 }; 3194 3195 OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo( 3196 DeviceGlobalVarMetadataEmitter); 3197 3198 for (const auto &E : OrderedEntries) { 3199 assert(std::get<0>(E) && "All ordered entries must exist!"); 3200 if (const auto *CE = 3201 dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>( 3202 std::get<0>(E))) { 3203 if (!CE->getID() || !CE->getAddress()) { 3204 // Do not blame the entry if the parent funtion is not emitted. 3205 StringRef FnName = ParentFunctions[CE->getOrder()]; 3206 if (!CGM.GetGlobalValue(FnName)) 3207 continue; 3208 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3209 DiagnosticsEngine::Error, 3210 "Offloading entry for target region in %0 is incorrect: either the " 3211 "address or the ID is invalid."); 3212 CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName; 3213 continue; 3214 } 3215 createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0, 3216 CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage); 3217 } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy:: 3218 OffloadEntryInfoDeviceGlobalVar>( 3219 std::get<0>(E))) { 3220 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags = 3221 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 3222 CE->getFlags()); 3223 switch (Flags) { 3224 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: { 3225 if (CGM.getLangOpts().OpenMPIsDevice && 3226 CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory()) 3227 continue; 3228 if (!CE->getAddress()) { 3229 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3230 DiagnosticsEngine::Error, "Offloading entry for declare target " 3231 "variable %0 is incorrect: the " 3232 "address is invalid."); 3233 CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E); 3234 continue; 3235 } 3236 // The vaiable has no definition - no need to add the entry. 3237 if (CE->getVarSize().isZero()) 3238 continue; 3239 break; 3240 } 3241 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink: 3242 assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) || 3243 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) && 3244 "Declaret target link address is set."); 3245 if (CGM.getLangOpts().OpenMPIsDevice) 3246 continue; 3247 if (!CE->getAddress()) { 3248 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3249 DiagnosticsEngine::Error, 3250 "Offloading entry for declare target variable is incorrect: the " 3251 "address is invalid."); 3252 CGM.getDiags().Report(DiagID); 3253 continue; 3254 } 3255 break; 3256 } 3257 createOffloadEntry(CE->getAddress(), CE->getAddress(), 3258 CE->getVarSize().getQuantity(), Flags, 3259 CE->getLinkage()); 3260 } else { 3261 llvm_unreachable("Unsupported entry kind."); 3262 } 3263 } 3264 } 3265 3266 /// Loads all the offload entries information from the host IR 3267 /// metadata. 3268 void CGOpenMPRuntime::loadOffloadInfoMetadata() { 3269 // If we are in target mode, load the metadata from the host IR. This code has 3270 // to match the metadaata creation in createOffloadEntriesAndInfoMetadata(). 3271 3272 if (!CGM.getLangOpts().OpenMPIsDevice) 3273 return; 3274 3275 if (CGM.getLangOpts().OMPHostIRFile.empty()) 3276 return; 3277 3278 auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile); 3279 if (auto EC = Buf.getError()) { 3280 CGM.getDiags().Report(diag::err_cannot_open_file) 3281 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 3282 return; 3283 } 3284 3285 llvm::LLVMContext C; 3286 auto ME = expectedToErrorOrAndEmitErrors( 3287 C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C)); 3288 3289 if (auto EC = ME.getError()) { 3290 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3291 DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'"); 3292 CGM.getDiags().Report(DiagID) 3293 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 3294 return; 3295 } 3296 3297 llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info"); 3298 if (!MD) 3299 return; 3300 3301 for (llvm::MDNode *MN : MD->operands()) { 3302 auto &&GetMDInt = [MN](unsigned Idx) { 3303 auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx)); 3304 return cast<llvm::ConstantInt>(V->getValue())->getZExtValue(); 3305 }; 3306 3307 auto &&GetMDString = [MN](unsigned Idx) { 3308 auto *V = cast<llvm::MDString>(MN->getOperand(Idx)); 3309 return V->getString(); 3310 }; 3311 3312 switch (GetMDInt(0)) { 3313 default: 3314 llvm_unreachable("Unexpected metadata!"); 3315 break; 3316 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 3317 OffloadingEntryInfoTargetRegion: 3318 OffloadEntriesInfoManager.initializeTargetRegionEntryInfo( 3319 /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2), 3320 /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4), 3321 /*Order=*/GetMDInt(5)); 3322 break; 3323 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 3324 OffloadingEntryInfoDeviceGlobalVar: 3325 OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo( 3326 /*MangledName=*/GetMDString(1), 3327 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 3328 /*Flags=*/GetMDInt(2)), 3329 /*Order=*/GetMDInt(3)); 3330 break; 3331 } 3332 } 3333 } 3334 3335 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) { 3336 if (!KmpRoutineEntryPtrTy) { 3337 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type. 3338 ASTContext &C = CGM.getContext(); 3339 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy}; 3340 FunctionProtoType::ExtProtoInfo EPI; 3341 KmpRoutineEntryPtrQTy = C.getPointerType( 3342 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI)); 3343 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy); 3344 } 3345 } 3346 3347 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() { 3348 // Make sure the type of the entry is already created. This is the type we 3349 // have to create: 3350 // struct __tgt_offload_entry{ 3351 // void *addr; // Pointer to the offload entry info. 3352 // // (function or global) 3353 // char *name; // Name of the function or global. 3354 // size_t size; // Size of the entry info (0 if it a function). 3355 // int32_t flags; // Flags associated with the entry, e.g. 'link'. 3356 // int32_t reserved; // Reserved, to use by the runtime library. 3357 // }; 3358 if (TgtOffloadEntryQTy.isNull()) { 3359 ASTContext &C = CGM.getContext(); 3360 RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry"); 3361 RD->startDefinition(); 3362 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 3363 addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy)); 3364 addFieldToRecordDecl(C, RD, C.getSizeType()); 3365 addFieldToRecordDecl( 3366 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 3367 addFieldToRecordDecl( 3368 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 3369 RD->completeDefinition(); 3370 RD->addAttr(PackedAttr::CreateImplicit(C)); 3371 TgtOffloadEntryQTy = C.getRecordType(RD); 3372 } 3373 return TgtOffloadEntryQTy; 3374 } 3375 3376 namespace { 3377 struct PrivateHelpersTy { 3378 PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original, 3379 const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit) 3380 : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy), 3381 PrivateElemInit(PrivateElemInit) {} 3382 PrivateHelpersTy(const VarDecl *Original) : Original(Original) {} 3383 const Expr *OriginalRef = nullptr; 3384 const VarDecl *Original = nullptr; 3385 const VarDecl *PrivateCopy = nullptr; 3386 const VarDecl *PrivateElemInit = nullptr; 3387 bool isLocalPrivate() const { 3388 return !OriginalRef && !PrivateCopy && !PrivateElemInit; 3389 } 3390 }; 3391 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy; 3392 } // anonymous namespace 3393 3394 static bool isAllocatableDecl(const VarDecl *VD) { 3395 const VarDecl *CVD = VD->getCanonicalDecl(); 3396 if (!CVD->hasAttr<OMPAllocateDeclAttr>()) 3397 return false; 3398 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 3399 // Use the default allocation. 3400 return !((AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc || 3401 AA->getAllocatorType() == OMPAllocateDeclAttr::OMPNullMemAlloc) && 3402 !AA->getAllocator()); 3403 } 3404 3405 static RecordDecl * 3406 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) { 3407 if (!Privates.empty()) { 3408 ASTContext &C = CGM.getContext(); 3409 // Build struct .kmp_privates_t. { 3410 // /* private vars */ 3411 // }; 3412 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t"); 3413 RD->startDefinition(); 3414 for (const auto &Pair : Privates) { 3415 const VarDecl *VD = Pair.second.Original; 3416 QualType Type = VD->getType().getNonReferenceType(); 3417 // If the private variable is a local variable with lvalue ref type, 3418 // allocate the pointer instead of the pointee type. 3419 if (Pair.second.isLocalPrivate()) { 3420 if (VD->getType()->isLValueReferenceType()) 3421 Type = C.getPointerType(Type); 3422 if (isAllocatableDecl(VD)) 3423 Type = C.getPointerType(Type); 3424 } 3425 FieldDecl *FD = addFieldToRecordDecl(C, RD, Type); 3426 if (VD->hasAttrs()) { 3427 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()), 3428 E(VD->getAttrs().end()); 3429 I != E; ++I) 3430 FD->addAttr(*I); 3431 } 3432 } 3433 RD->completeDefinition(); 3434 return RD; 3435 } 3436 return nullptr; 3437 } 3438 3439 static RecordDecl * 3440 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, 3441 QualType KmpInt32Ty, 3442 QualType KmpRoutineEntryPointerQTy) { 3443 ASTContext &C = CGM.getContext(); 3444 // Build struct kmp_task_t { 3445 // void * shareds; 3446 // kmp_routine_entry_t routine; 3447 // kmp_int32 part_id; 3448 // kmp_cmplrdata_t data1; 3449 // kmp_cmplrdata_t data2; 3450 // For taskloops additional fields: 3451 // kmp_uint64 lb; 3452 // kmp_uint64 ub; 3453 // kmp_int64 st; 3454 // kmp_int32 liter; 3455 // void * reductions; 3456 // }; 3457 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union); 3458 UD->startDefinition(); 3459 addFieldToRecordDecl(C, UD, KmpInt32Ty); 3460 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy); 3461 UD->completeDefinition(); 3462 QualType KmpCmplrdataTy = C.getRecordType(UD); 3463 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t"); 3464 RD->startDefinition(); 3465 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 3466 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy); 3467 addFieldToRecordDecl(C, RD, KmpInt32Ty); 3468 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 3469 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 3470 if (isOpenMPTaskLoopDirective(Kind)) { 3471 QualType KmpUInt64Ty = 3472 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 3473 QualType KmpInt64Ty = 3474 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 3475 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 3476 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 3477 addFieldToRecordDecl(C, RD, KmpInt64Ty); 3478 addFieldToRecordDecl(C, RD, KmpInt32Ty); 3479 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 3480 } 3481 RD->completeDefinition(); 3482 return RD; 3483 } 3484 3485 static RecordDecl * 3486 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, 3487 ArrayRef<PrivateDataTy> Privates) { 3488 ASTContext &C = CGM.getContext(); 3489 // Build struct kmp_task_t_with_privates { 3490 // kmp_task_t task_data; 3491 // .kmp_privates_t. privates; 3492 // }; 3493 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates"); 3494 RD->startDefinition(); 3495 addFieldToRecordDecl(C, RD, KmpTaskTQTy); 3496 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) 3497 addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD)); 3498 RD->completeDefinition(); 3499 return RD; 3500 } 3501 3502 /// Emit a proxy function which accepts kmp_task_t as the second 3503 /// argument. 3504 /// \code 3505 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) { 3506 /// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt, 3507 /// For taskloops: 3508 /// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 3509 /// tt->reductions, tt->shareds); 3510 /// return 0; 3511 /// } 3512 /// \endcode 3513 static llvm::Function * 3514 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, 3515 OpenMPDirectiveKind Kind, QualType KmpInt32Ty, 3516 QualType KmpTaskTWithPrivatesPtrQTy, 3517 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, 3518 QualType SharedsPtrTy, llvm::Function *TaskFunction, 3519 llvm::Value *TaskPrivatesMap) { 3520 ASTContext &C = CGM.getContext(); 3521 FunctionArgList Args; 3522 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 3523 ImplicitParamDecl::Other); 3524 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3525 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 3526 ImplicitParamDecl::Other); 3527 Args.push_back(&GtidArg); 3528 Args.push_back(&TaskTypeArg); 3529 const auto &TaskEntryFnInfo = 3530 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 3531 llvm::FunctionType *TaskEntryTy = 3532 CGM.getTypes().GetFunctionType(TaskEntryFnInfo); 3533 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""}); 3534 auto *TaskEntry = llvm::Function::Create( 3535 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 3536 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo); 3537 TaskEntry->setDoesNotRecurse(); 3538 CodeGenFunction CGF(CGM); 3539 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args, 3540 Loc, Loc); 3541 3542 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map, 3543 // tt, 3544 // For taskloops: 3545 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 3546 // tt->task_data.shareds); 3547 llvm::Value *GtidParam = CGF.EmitLoadOfScalar( 3548 CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc); 3549 LValue TDBase = CGF.EmitLoadOfPointerLValue( 3550 CGF.GetAddrOfLocalVar(&TaskTypeArg), 3551 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 3552 const auto *KmpTaskTWithPrivatesQTyRD = 3553 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 3554 LValue Base = 3555 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 3556 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 3557 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 3558 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI); 3559 llvm::Value *PartidParam = PartIdLVal.getPointer(CGF); 3560 3561 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds); 3562 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI); 3563 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3564 CGF.EmitLoadOfScalar(SharedsLVal, Loc), 3565 CGF.ConvertTypeForMem(SharedsPtrTy)); 3566 3567 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 3568 llvm::Value *PrivatesParam; 3569 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) { 3570 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI); 3571 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3572 PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy); 3573 } else { 3574 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 3575 } 3576 3577 llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam, 3578 TaskPrivatesMap, 3579 CGF.Builder 3580 .CreatePointerBitCastOrAddrSpaceCast( 3581 TDBase.getAddress(CGF), CGF.VoidPtrTy) 3582 .getPointer()}; 3583 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs), 3584 std::end(CommonArgs)); 3585 if (isOpenMPTaskLoopDirective(Kind)) { 3586 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound); 3587 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI); 3588 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc); 3589 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound); 3590 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI); 3591 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc); 3592 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride); 3593 LValue StLVal = CGF.EmitLValueForField(Base, *StFI); 3594 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc); 3595 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 3596 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 3597 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc); 3598 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions); 3599 LValue RLVal = CGF.EmitLValueForField(Base, *RFI); 3600 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc); 3601 CallArgs.push_back(LBParam); 3602 CallArgs.push_back(UBParam); 3603 CallArgs.push_back(StParam); 3604 CallArgs.push_back(LIParam); 3605 CallArgs.push_back(RParam); 3606 } 3607 CallArgs.push_back(SharedsParam); 3608 3609 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction, 3610 CallArgs); 3611 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)), 3612 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty)); 3613 CGF.FinishFunction(); 3614 return TaskEntry; 3615 } 3616 3617 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM, 3618 SourceLocation Loc, 3619 QualType KmpInt32Ty, 3620 QualType KmpTaskTWithPrivatesPtrQTy, 3621 QualType KmpTaskTWithPrivatesQTy) { 3622 ASTContext &C = CGM.getContext(); 3623 FunctionArgList Args; 3624 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 3625 ImplicitParamDecl::Other); 3626 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3627 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 3628 ImplicitParamDecl::Other); 3629 Args.push_back(&GtidArg); 3630 Args.push_back(&TaskTypeArg); 3631 const auto &DestructorFnInfo = 3632 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 3633 llvm::FunctionType *DestructorFnTy = 3634 CGM.getTypes().GetFunctionType(DestructorFnInfo); 3635 std::string Name = 3636 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""}); 3637 auto *DestructorFn = 3638 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage, 3639 Name, &CGM.getModule()); 3640 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn, 3641 DestructorFnInfo); 3642 DestructorFn->setDoesNotRecurse(); 3643 CodeGenFunction CGF(CGM); 3644 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo, 3645 Args, Loc, Loc); 3646 3647 LValue Base = CGF.EmitLoadOfPointerLValue( 3648 CGF.GetAddrOfLocalVar(&TaskTypeArg), 3649 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 3650 const auto *KmpTaskTWithPrivatesQTyRD = 3651 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 3652 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 3653 Base = CGF.EmitLValueForField(Base, *FI); 3654 for (const auto *Field : 3655 cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) { 3656 if (QualType::DestructionKind DtorKind = 3657 Field->getType().isDestructedType()) { 3658 LValue FieldLValue = CGF.EmitLValueForField(Base, Field); 3659 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType()); 3660 } 3661 } 3662 CGF.FinishFunction(); 3663 return DestructorFn; 3664 } 3665 3666 /// Emit a privates mapping function for correct handling of private and 3667 /// firstprivate variables. 3668 /// \code 3669 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1> 3670 /// **noalias priv1,..., <tyn> **noalias privn) { 3671 /// *priv1 = &.privates.priv1; 3672 /// ...; 3673 /// *privn = &.privates.privn; 3674 /// } 3675 /// \endcode 3676 static llvm::Value * 3677 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, 3678 const OMPTaskDataTy &Data, QualType PrivatesQTy, 3679 ArrayRef<PrivateDataTy> Privates) { 3680 ASTContext &C = CGM.getContext(); 3681 FunctionArgList Args; 3682 ImplicitParamDecl TaskPrivatesArg( 3683 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3684 C.getPointerType(PrivatesQTy).withConst().withRestrict(), 3685 ImplicitParamDecl::Other); 3686 Args.push_back(&TaskPrivatesArg); 3687 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, unsigned> PrivateVarsPos; 3688 unsigned Counter = 1; 3689 for (const Expr *E : Data.PrivateVars) { 3690 Args.push_back(ImplicitParamDecl::Create( 3691 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3692 C.getPointerType(C.getPointerType(E->getType())) 3693 .withConst() 3694 .withRestrict(), 3695 ImplicitParamDecl::Other)); 3696 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 3697 PrivateVarsPos[VD] = Counter; 3698 ++Counter; 3699 } 3700 for (const Expr *E : Data.FirstprivateVars) { 3701 Args.push_back(ImplicitParamDecl::Create( 3702 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3703 C.getPointerType(C.getPointerType(E->getType())) 3704 .withConst() 3705 .withRestrict(), 3706 ImplicitParamDecl::Other)); 3707 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 3708 PrivateVarsPos[VD] = Counter; 3709 ++Counter; 3710 } 3711 for (const Expr *E : Data.LastprivateVars) { 3712 Args.push_back(ImplicitParamDecl::Create( 3713 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3714 C.getPointerType(C.getPointerType(E->getType())) 3715 .withConst() 3716 .withRestrict(), 3717 ImplicitParamDecl::Other)); 3718 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 3719 PrivateVarsPos[VD] = Counter; 3720 ++Counter; 3721 } 3722 for (const VarDecl *VD : Data.PrivateLocals) { 3723 QualType Ty = VD->getType().getNonReferenceType(); 3724 if (VD->getType()->isLValueReferenceType()) 3725 Ty = C.getPointerType(Ty); 3726 if (isAllocatableDecl(VD)) 3727 Ty = C.getPointerType(Ty); 3728 Args.push_back(ImplicitParamDecl::Create( 3729 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3730 C.getPointerType(C.getPointerType(Ty)).withConst().withRestrict(), 3731 ImplicitParamDecl::Other)); 3732 PrivateVarsPos[VD] = Counter; 3733 ++Counter; 3734 } 3735 const auto &TaskPrivatesMapFnInfo = 3736 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3737 llvm::FunctionType *TaskPrivatesMapTy = 3738 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo); 3739 std::string Name = 3740 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""}); 3741 auto *TaskPrivatesMap = llvm::Function::Create( 3742 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name, 3743 &CGM.getModule()); 3744 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap, 3745 TaskPrivatesMapFnInfo); 3746 if (CGM.getLangOpts().Optimize) { 3747 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline); 3748 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone); 3749 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline); 3750 } 3751 CodeGenFunction CGF(CGM); 3752 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap, 3753 TaskPrivatesMapFnInfo, Args, Loc, Loc); 3754 3755 // *privi = &.privates.privi; 3756 LValue Base = CGF.EmitLoadOfPointerLValue( 3757 CGF.GetAddrOfLocalVar(&TaskPrivatesArg), 3758 TaskPrivatesArg.getType()->castAs<PointerType>()); 3759 const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl()); 3760 Counter = 0; 3761 for (const FieldDecl *Field : PrivatesQTyRD->fields()) { 3762 LValue FieldLVal = CGF.EmitLValueForField(Base, Field); 3763 const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]]; 3764 LValue RefLVal = 3765 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType()); 3766 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue( 3767 RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>()); 3768 CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal); 3769 ++Counter; 3770 } 3771 CGF.FinishFunction(); 3772 return TaskPrivatesMap; 3773 } 3774 3775 /// Emit initialization for private variables in task-based directives. 3776 static void emitPrivatesInit(CodeGenFunction &CGF, 3777 const OMPExecutableDirective &D, 3778 Address KmpTaskSharedsPtr, LValue TDBase, 3779 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 3780 QualType SharedsTy, QualType SharedsPtrTy, 3781 const OMPTaskDataTy &Data, 3782 ArrayRef<PrivateDataTy> Privates, bool ForDup) { 3783 ASTContext &C = CGF.getContext(); 3784 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 3785 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI); 3786 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind()) 3787 ? OMPD_taskloop 3788 : OMPD_task; 3789 const CapturedStmt &CS = *D.getCapturedStmt(Kind); 3790 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS); 3791 LValue SrcBase; 3792 bool IsTargetTask = 3793 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) || 3794 isOpenMPTargetExecutionDirective(D.getDirectiveKind()); 3795 // For target-based directives skip 4 firstprivate arrays BasePointersArray, 3796 // PointersArray, SizesArray, and MappersArray. The original variables for 3797 // these arrays are not captured and we get their addresses explicitly. 3798 if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) || 3799 (IsTargetTask && KmpTaskSharedsPtr.isValid())) { 3800 SrcBase = CGF.MakeAddrLValue( 3801 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3802 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)), 3803 SharedsTy); 3804 } 3805 FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin(); 3806 for (const PrivateDataTy &Pair : Privates) { 3807 // Do not initialize private locals. 3808 if (Pair.second.isLocalPrivate()) { 3809 ++FI; 3810 continue; 3811 } 3812 const VarDecl *VD = Pair.second.PrivateCopy; 3813 const Expr *Init = VD->getAnyInitializer(); 3814 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) && 3815 !CGF.isTrivialInitializer(Init)))) { 3816 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI); 3817 if (const VarDecl *Elem = Pair.second.PrivateElemInit) { 3818 const VarDecl *OriginalVD = Pair.second.Original; 3819 // Check if the variable is the target-based BasePointersArray, 3820 // PointersArray, SizesArray, or MappersArray. 3821 LValue SharedRefLValue; 3822 QualType Type = PrivateLValue.getType(); 3823 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD); 3824 if (IsTargetTask && !SharedField) { 3825 assert(isa<ImplicitParamDecl>(OriginalVD) && 3826 isa<CapturedDecl>(OriginalVD->getDeclContext()) && 3827 cast<CapturedDecl>(OriginalVD->getDeclContext()) 3828 ->getNumParams() == 0 && 3829 isa<TranslationUnitDecl>( 3830 cast<CapturedDecl>(OriginalVD->getDeclContext()) 3831 ->getDeclContext()) && 3832 "Expected artificial target data variable."); 3833 SharedRefLValue = 3834 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type); 3835 } else if (ForDup) { 3836 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField); 3837 SharedRefLValue = CGF.MakeAddrLValue( 3838 Address(SharedRefLValue.getPointer(CGF), 3839 C.getDeclAlign(OriginalVD)), 3840 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl), 3841 SharedRefLValue.getTBAAInfo()); 3842 } else if (CGF.LambdaCaptureFields.count( 3843 Pair.second.Original->getCanonicalDecl()) > 0 || 3844 dyn_cast_or_null<BlockDecl>(CGF.CurCodeDecl)) { 3845 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef); 3846 } else { 3847 // Processing for implicitly captured variables. 3848 InlinedOpenMPRegionRAII Region( 3849 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown, 3850 /*HasCancel=*/false); 3851 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef); 3852 } 3853 if (Type->isArrayType()) { 3854 // Initialize firstprivate array. 3855 if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) { 3856 // Perform simple memcpy. 3857 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type); 3858 } else { 3859 // Initialize firstprivate array using element-by-element 3860 // initialization. 3861 CGF.EmitOMPAggregateAssign( 3862 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF), 3863 Type, 3864 [&CGF, Elem, Init, &CapturesInfo](Address DestElement, 3865 Address SrcElement) { 3866 // Clean up any temporaries needed by the initialization. 3867 CodeGenFunction::OMPPrivateScope InitScope(CGF); 3868 InitScope.addPrivate( 3869 Elem, [SrcElement]() -> Address { return SrcElement; }); 3870 (void)InitScope.Privatize(); 3871 // Emit initialization for single element. 3872 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII( 3873 CGF, &CapturesInfo); 3874 CGF.EmitAnyExprToMem(Init, DestElement, 3875 Init->getType().getQualifiers(), 3876 /*IsInitializer=*/false); 3877 }); 3878 } 3879 } else { 3880 CodeGenFunction::OMPPrivateScope InitScope(CGF); 3881 InitScope.addPrivate(Elem, [SharedRefLValue, &CGF]() -> Address { 3882 return SharedRefLValue.getAddress(CGF); 3883 }); 3884 (void)InitScope.Privatize(); 3885 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo); 3886 CGF.EmitExprAsInit(Init, VD, PrivateLValue, 3887 /*capturedByInit=*/false); 3888 } 3889 } else { 3890 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false); 3891 } 3892 } 3893 ++FI; 3894 } 3895 } 3896 3897 /// Check if duplication function is required for taskloops. 3898 static bool checkInitIsRequired(CodeGenFunction &CGF, 3899 ArrayRef<PrivateDataTy> Privates) { 3900 bool InitRequired = false; 3901 for (const PrivateDataTy &Pair : Privates) { 3902 if (Pair.second.isLocalPrivate()) 3903 continue; 3904 const VarDecl *VD = Pair.second.PrivateCopy; 3905 const Expr *Init = VD->getAnyInitializer(); 3906 InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) && 3907 !CGF.isTrivialInitializer(Init)); 3908 if (InitRequired) 3909 break; 3910 } 3911 return InitRequired; 3912 } 3913 3914 3915 /// Emit task_dup function (for initialization of 3916 /// private/firstprivate/lastprivate vars and last_iter flag) 3917 /// \code 3918 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int 3919 /// lastpriv) { 3920 /// // setup lastprivate flag 3921 /// task_dst->last = lastpriv; 3922 /// // could be constructor calls here... 3923 /// } 3924 /// \endcode 3925 static llvm::Value * 3926 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, 3927 const OMPExecutableDirective &D, 3928 QualType KmpTaskTWithPrivatesPtrQTy, 3929 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 3930 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, 3931 QualType SharedsPtrTy, const OMPTaskDataTy &Data, 3932 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) { 3933 ASTContext &C = CGM.getContext(); 3934 FunctionArgList Args; 3935 ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3936 KmpTaskTWithPrivatesPtrQTy, 3937 ImplicitParamDecl::Other); 3938 ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3939 KmpTaskTWithPrivatesPtrQTy, 3940 ImplicitParamDecl::Other); 3941 ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy, 3942 ImplicitParamDecl::Other); 3943 Args.push_back(&DstArg); 3944 Args.push_back(&SrcArg); 3945 Args.push_back(&LastprivArg); 3946 const auto &TaskDupFnInfo = 3947 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3948 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo); 3949 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""}); 3950 auto *TaskDup = llvm::Function::Create( 3951 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 3952 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo); 3953 TaskDup->setDoesNotRecurse(); 3954 CodeGenFunction CGF(CGM); 3955 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc, 3956 Loc); 3957 3958 LValue TDBase = CGF.EmitLoadOfPointerLValue( 3959 CGF.GetAddrOfLocalVar(&DstArg), 3960 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 3961 // task_dst->liter = lastpriv; 3962 if (WithLastIter) { 3963 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 3964 LValue Base = CGF.EmitLValueForField( 3965 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 3966 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 3967 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar( 3968 CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc); 3969 CGF.EmitStoreOfScalar(Lastpriv, LILVal); 3970 } 3971 3972 // Emit initial values for private copies (if any). 3973 assert(!Privates.empty()); 3974 Address KmpTaskSharedsPtr = Address::invalid(); 3975 if (!Data.FirstprivateVars.empty()) { 3976 LValue TDBase = CGF.EmitLoadOfPointerLValue( 3977 CGF.GetAddrOfLocalVar(&SrcArg), 3978 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 3979 LValue Base = CGF.EmitLValueForField( 3980 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 3981 KmpTaskSharedsPtr = Address( 3982 CGF.EmitLoadOfScalar(CGF.EmitLValueForField( 3983 Base, *std::next(KmpTaskTQTyRD->field_begin(), 3984 KmpTaskTShareds)), 3985 Loc), 3986 CGM.getNaturalTypeAlignment(SharedsTy)); 3987 } 3988 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD, 3989 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true); 3990 CGF.FinishFunction(); 3991 return TaskDup; 3992 } 3993 3994 /// Checks if destructor function is required to be generated. 3995 /// \return true if cleanups are required, false otherwise. 3996 static bool 3997 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD, 3998 ArrayRef<PrivateDataTy> Privates) { 3999 for (const PrivateDataTy &P : Privates) { 4000 if (P.second.isLocalPrivate()) 4001 continue; 4002 QualType Ty = P.second.Original->getType().getNonReferenceType(); 4003 if (Ty.isDestructedType()) 4004 return true; 4005 } 4006 return false; 4007 } 4008 4009 namespace { 4010 /// Loop generator for OpenMP iterator expression. 4011 class OMPIteratorGeneratorScope final 4012 : public CodeGenFunction::OMPPrivateScope { 4013 CodeGenFunction &CGF; 4014 const OMPIteratorExpr *E = nullptr; 4015 SmallVector<CodeGenFunction::JumpDest, 4> ContDests; 4016 SmallVector<CodeGenFunction::JumpDest, 4> ExitDests; 4017 OMPIteratorGeneratorScope() = delete; 4018 OMPIteratorGeneratorScope(OMPIteratorGeneratorScope &) = delete; 4019 4020 public: 4021 OMPIteratorGeneratorScope(CodeGenFunction &CGF, const OMPIteratorExpr *E) 4022 : CodeGenFunction::OMPPrivateScope(CGF), CGF(CGF), E(E) { 4023 if (!E) 4024 return; 4025 SmallVector<llvm::Value *, 4> Uppers; 4026 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) { 4027 Uppers.push_back(CGF.EmitScalarExpr(E->getHelper(I).Upper)); 4028 const auto *VD = cast<VarDecl>(E->getIteratorDecl(I)); 4029 addPrivate(VD, [&CGF, VD]() { 4030 return CGF.CreateMemTemp(VD->getType(), VD->getName()); 4031 }); 4032 const OMPIteratorHelperData &HelperData = E->getHelper(I); 4033 addPrivate(HelperData.CounterVD, [&CGF, &HelperData]() { 4034 return CGF.CreateMemTemp(HelperData.CounterVD->getType(), 4035 "counter.addr"); 4036 }); 4037 } 4038 Privatize(); 4039 4040 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) { 4041 const OMPIteratorHelperData &HelperData = E->getHelper(I); 4042 LValue CLVal = 4043 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(HelperData.CounterVD), 4044 HelperData.CounterVD->getType()); 4045 // Counter = 0; 4046 CGF.EmitStoreOfScalar( 4047 llvm::ConstantInt::get(CLVal.getAddress(CGF).getElementType(), 0), 4048 CLVal); 4049 CodeGenFunction::JumpDest &ContDest = 4050 ContDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.cont")); 4051 CodeGenFunction::JumpDest &ExitDest = 4052 ExitDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.exit")); 4053 // N = <number-of_iterations>; 4054 llvm::Value *N = Uppers[I]; 4055 // cont: 4056 // if (Counter < N) goto body; else goto exit; 4057 CGF.EmitBlock(ContDest.getBlock()); 4058 auto *CVal = 4059 CGF.EmitLoadOfScalar(CLVal, HelperData.CounterVD->getLocation()); 4060 llvm::Value *Cmp = 4061 HelperData.CounterVD->getType()->isSignedIntegerOrEnumerationType() 4062 ? CGF.Builder.CreateICmpSLT(CVal, N) 4063 : CGF.Builder.CreateICmpULT(CVal, N); 4064 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("iter.body"); 4065 CGF.Builder.CreateCondBr(Cmp, BodyBB, ExitDest.getBlock()); 4066 // body: 4067 CGF.EmitBlock(BodyBB); 4068 // Iteri = Begini + Counter * Stepi; 4069 CGF.EmitIgnoredExpr(HelperData.Update); 4070 } 4071 } 4072 ~OMPIteratorGeneratorScope() { 4073 if (!E) 4074 return; 4075 for (unsigned I = E->numOfIterators(); I > 0; --I) { 4076 // Counter = Counter + 1; 4077 const OMPIteratorHelperData &HelperData = E->getHelper(I - 1); 4078 CGF.EmitIgnoredExpr(HelperData.CounterUpdate); 4079 // goto cont; 4080 CGF.EmitBranchThroughCleanup(ContDests[I - 1]); 4081 // exit: 4082 CGF.EmitBlock(ExitDests[I - 1].getBlock(), /*IsFinished=*/I == 1); 4083 } 4084 } 4085 }; 4086 } // namespace 4087 4088 static std::pair<llvm::Value *, llvm::Value *> 4089 getPointerAndSize(CodeGenFunction &CGF, const Expr *E) { 4090 const auto *OASE = dyn_cast<OMPArrayShapingExpr>(E); 4091 llvm::Value *Addr; 4092 if (OASE) { 4093 const Expr *Base = OASE->getBase(); 4094 Addr = CGF.EmitScalarExpr(Base); 4095 } else { 4096 Addr = CGF.EmitLValue(E).getPointer(CGF); 4097 } 4098 llvm::Value *SizeVal; 4099 QualType Ty = E->getType(); 4100 if (OASE) { 4101 SizeVal = CGF.getTypeSize(OASE->getBase()->getType()->getPointeeType()); 4102 for (const Expr *SE : OASE->getDimensions()) { 4103 llvm::Value *Sz = CGF.EmitScalarExpr(SE); 4104 Sz = CGF.EmitScalarConversion( 4105 Sz, SE->getType(), CGF.getContext().getSizeType(), SE->getExprLoc()); 4106 SizeVal = CGF.Builder.CreateNUWMul(SizeVal, Sz); 4107 } 4108 } else if (const auto *ASE = 4109 dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) { 4110 LValue UpAddrLVal = 4111 CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false); 4112 llvm::Value *UpAddr = 4113 CGF.Builder.CreateConstGEP1_32(UpAddrLVal.getPointer(CGF), /*Idx0=*/1); 4114 llvm::Value *LowIntPtr = CGF.Builder.CreatePtrToInt(Addr, CGF.SizeTy); 4115 llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGF.SizeTy); 4116 SizeVal = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr); 4117 } else { 4118 SizeVal = CGF.getTypeSize(Ty); 4119 } 4120 return std::make_pair(Addr, SizeVal); 4121 } 4122 4123 /// Builds kmp_depend_info, if it is not built yet, and builds flags type. 4124 static void getKmpAffinityType(ASTContext &C, QualType &KmpTaskAffinityInfoTy) { 4125 QualType FlagsTy = C.getIntTypeForBitwidth(32, /*Signed=*/false); 4126 if (KmpTaskAffinityInfoTy.isNull()) { 4127 RecordDecl *KmpAffinityInfoRD = 4128 C.buildImplicitRecord("kmp_task_affinity_info_t"); 4129 KmpAffinityInfoRD->startDefinition(); 4130 addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getIntPtrType()); 4131 addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getSizeType()); 4132 addFieldToRecordDecl(C, KmpAffinityInfoRD, FlagsTy); 4133 KmpAffinityInfoRD->completeDefinition(); 4134 KmpTaskAffinityInfoTy = C.getRecordType(KmpAffinityInfoRD); 4135 } 4136 } 4137 4138 CGOpenMPRuntime::TaskResultTy 4139 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, 4140 const OMPExecutableDirective &D, 4141 llvm::Function *TaskFunction, QualType SharedsTy, 4142 Address Shareds, const OMPTaskDataTy &Data) { 4143 ASTContext &C = CGM.getContext(); 4144 llvm::SmallVector<PrivateDataTy, 4> Privates; 4145 // Aggregate privates and sort them by the alignment. 4146 const auto *I = Data.PrivateCopies.begin(); 4147 for (const Expr *E : Data.PrivateVars) { 4148 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4149 Privates.emplace_back( 4150 C.getDeclAlign(VD), 4151 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4152 /*PrivateElemInit=*/nullptr)); 4153 ++I; 4154 } 4155 I = Data.FirstprivateCopies.begin(); 4156 const auto *IElemInitRef = Data.FirstprivateInits.begin(); 4157 for (const Expr *E : Data.FirstprivateVars) { 4158 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4159 Privates.emplace_back( 4160 C.getDeclAlign(VD), 4161 PrivateHelpersTy( 4162 E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4163 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))); 4164 ++I; 4165 ++IElemInitRef; 4166 } 4167 I = Data.LastprivateCopies.begin(); 4168 for (const Expr *E : Data.LastprivateVars) { 4169 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4170 Privates.emplace_back( 4171 C.getDeclAlign(VD), 4172 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4173 /*PrivateElemInit=*/nullptr)); 4174 ++I; 4175 } 4176 for (const VarDecl *VD : Data.PrivateLocals) { 4177 if (isAllocatableDecl(VD)) 4178 Privates.emplace_back(CGM.getPointerAlign(), PrivateHelpersTy(VD)); 4179 else 4180 Privates.emplace_back(C.getDeclAlign(VD), PrivateHelpersTy(VD)); 4181 } 4182 llvm::stable_sort(Privates, 4183 [](const PrivateDataTy &L, const PrivateDataTy &R) { 4184 return L.first > R.first; 4185 }); 4186 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 4187 // Build type kmp_routine_entry_t (if not built yet). 4188 emitKmpRoutineEntryT(KmpInt32Ty); 4189 // Build type kmp_task_t (if not built yet). 4190 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) { 4191 if (SavedKmpTaskloopTQTy.isNull()) { 4192 SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4193 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4194 } 4195 KmpTaskTQTy = SavedKmpTaskloopTQTy; 4196 } else { 4197 assert((D.getDirectiveKind() == OMPD_task || 4198 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) || 4199 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) && 4200 "Expected taskloop, task or target directive"); 4201 if (SavedKmpTaskTQTy.isNull()) { 4202 SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4203 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4204 } 4205 KmpTaskTQTy = SavedKmpTaskTQTy; 4206 } 4207 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4208 // Build particular struct kmp_task_t for the given task. 4209 const RecordDecl *KmpTaskTWithPrivatesQTyRD = 4210 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates); 4211 QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD); 4212 QualType KmpTaskTWithPrivatesPtrQTy = 4213 C.getPointerType(KmpTaskTWithPrivatesQTy); 4214 llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy); 4215 llvm::Type *KmpTaskTWithPrivatesPtrTy = 4216 KmpTaskTWithPrivatesTy->getPointerTo(); 4217 llvm::Value *KmpTaskTWithPrivatesTySize = 4218 CGF.getTypeSize(KmpTaskTWithPrivatesQTy); 4219 QualType SharedsPtrTy = C.getPointerType(SharedsTy); 4220 4221 // Emit initial values for private copies (if any). 4222 llvm::Value *TaskPrivatesMap = nullptr; 4223 llvm::Type *TaskPrivatesMapTy = 4224 std::next(TaskFunction->arg_begin(), 3)->getType(); 4225 if (!Privates.empty()) { 4226 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4227 TaskPrivatesMap = 4228 emitTaskPrivateMappingFunction(CGM, Loc, Data, FI->getType(), Privates); 4229 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4230 TaskPrivatesMap, TaskPrivatesMapTy); 4231 } else { 4232 TaskPrivatesMap = llvm::ConstantPointerNull::get( 4233 cast<llvm::PointerType>(TaskPrivatesMapTy)); 4234 } 4235 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid, 4236 // kmp_task_t *tt); 4237 llvm::Function *TaskEntry = emitProxyTaskFunction( 4238 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 4239 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction, 4240 TaskPrivatesMap); 4241 4242 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 4243 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 4244 // kmp_routine_entry_t *task_entry); 4245 // Task flags. Format is taken from 4246 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h, 4247 // description of kmp_tasking_flags struct. 4248 enum { 4249 TiedFlag = 0x1, 4250 FinalFlag = 0x2, 4251 DestructorsFlag = 0x8, 4252 PriorityFlag = 0x20, 4253 DetachableFlag = 0x40, 4254 }; 4255 unsigned Flags = Data.Tied ? TiedFlag : 0; 4256 bool NeedsCleanup = false; 4257 if (!Privates.empty()) { 4258 NeedsCleanup = 4259 checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD, Privates); 4260 if (NeedsCleanup) 4261 Flags = Flags | DestructorsFlag; 4262 } 4263 if (Data.Priority.getInt()) 4264 Flags = Flags | PriorityFlag; 4265 if (D.hasClausesOfKind<OMPDetachClause>()) 4266 Flags = Flags | DetachableFlag; 4267 llvm::Value *TaskFlags = 4268 Data.Final.getPointer() 4269 ? CGF.Builder.CreateSelect(Data.Final.getPointer(), 4270 CGF.Builder.getInt32(FinalFlag), 4271 CGF.Builder.getInt32(/*C=*/0)) 4272 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0); 4273 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags)); 4274 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy)); 4275 SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc), 4276 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize, 4277 SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4278 TaskEntry, KmpRoutineEntryPtrTy)}; 4279 llvm::Value *NewTask; 4280 if (D.hasClausesOfKind<OMPNowaitClause>()) { 4281 // Check if we have any device clause associated with the directive. 4282 const Expr *Device = nullptr; 4283 if (auto *C = D.getSingleClause<OMPDeviceClause>()) 4284 Device = C->getDevice(); 4285 // Emit device ID if any otherwise use default value. 4286 llvm::Value *DeviceID; 4287 if (Device) 4288 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 4289 CGF.Int64Ty, /*isSigned=*/true); 4290 else 4291 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 4292 AllocArgs.push_back(DeviceID); 4293 NewTask = CGF.EmitRuntimeCall( 4294 OMPBuilder.getOrCreateRuntimeFunction( 4295 CGM.getModule(), OMPRTL___kmpc_omp_target_task_alloc), 4296 AllocArgs); 4297 } else { 4298 NewTask = 4299 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 4300 CGM.getModule(), OMPRTL___kmpc_omp_task_alloc), 4301 AllocArgs); 4302 } 4303 // Emit detach clause initialization. 4304 // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid, 4305 // task_descriptor); 4306 if (const auto *DC = D.getSingleClause<OMPDetachClause>()) { 4307 const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts(); 4308 LValue EvtLVal = CGF.EmitLValue(Evt); 4309 4310 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 4311 // int gtid, kmp_task_t *task); 4312 llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc()); 4313 llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc()); 4314 Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false); 4315 llvm::Value *EvtVal = CGF.EmitRuntimeCall( 4316 OMPBuilder.getOrCreateRuntimeFunction( 4317 CGM.getModule(), OMPRTL___kmpc_task_allow_completion_event), 4318 {Loc, Tid, NewTask}); 4319 EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(), 4320 Evt->getExprLoc()); 4321 CGF.EmitStoreOfScalar(EvtVal, EvtLVal); 4322 } 4323 // Process affinity clauses. 4324 if (D.hasClausesOfKind<OMPAffinityClause>()) { 4325 // Process list of affinity data. 4326 ASTContext &C = CGM.getContext(); 4327 Address AffinitiesArray = Address::invalid(); 4328 // Calculate number of elements to form the array of affinity data. 4329 llvm::Value *NumOfElements = nullptr; 4330 unsigned NumAffinities = 0; 4331 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) { 4332 if (const Expr *Modifier = C->getModifier()) { 4333 const auto *IE = cast<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts()); 4334 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) { 4335 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper); 4336 Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false); 4337 NumOfElements = 4338 NumOfElements ? CGF.Builder.CreateNUWMul(NumOfElements, Sz) : Sz; 4339 } 4340 } else { 4341 NumAffinities += C->varlist_size(); 4342 } 4343 } 4344 getKmpAffinityType(CGM.getContext(), KmpTaskAffinityInfoTy); 4345 // Fields ids in kmp_task_affinity_info record. 4346 enum RTLAffinityInfoFieldsTy { BaseAddr, Len, Flags }; 4347 4348 QualType KmpTaskAffinityInfoArrayTy; 4349 if (NumOfElements) { 4350 NumOfElements = CGF.Builder.CreateNUWAdd( 4351 llvm::ConstantInt::get(CGF.SizeTy, NumAffinities), NumOfElements); 4352 OpaqueValueExpr OVE( 4353 Loc, 4354 C.getIntTypeForBitwidth(C.getTypeSize(C.getSizeType()), /*Signed=*/0), 4355 VK_RValue); 4356 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, 4357 RValue::get(NumOfElements)); 4358 KmpTaskAffinityInfoArrayTy = 4359 C.getVariableArrayType(KmpTaskAffinityInfoTy, &OVE, ArrayType::Normal, 4360 /*IndexTypeQuals=*/0, SourceRange(Loc, Loc)); 4361 // Properly emit variable-sized array. 4362 auto *PD = ImplicitParamDecl::Create(C, KmpTaskAffinityInfoArrayTy, 4363 ImplicitParamDecl::Other); 4364 CGF.EmitVarDecl(*PD); 4365 AffinitiesArray = CGF.GetAddrOfLocalVar(PD); 4366 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty, 4367 /*isSigned=*/false); 4368 } else { 4369 KmpTaskAffinityInfoArrayTy = C.getConstantArrayType( 4370 KmpTaskAffinityInfoTy, 4371 llvm::APInt(C.getTypeSize(C.getSizeType()), NumAffinities), nullptr, 4372 ArrayType::Normal, /*IndexTypeQuals=*/0); 4373 AffinitiesArray = 4374 CGF.CreateMemTemp(KmpTaskAffinityInfoArrayTy, ".affs.arr.addr"); 4375 AffinitiesArray = CGF.Builder.CreateConstArrayGEP(AffinitiesArray, 0); 4376 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumAffinities, 4377 /*isSigned=*/false); 4378 } 4379 4380 const auto *KmpAffinityInfoRD = KmpTaskAffinityInfoTy->getAsRecordDecl(); 4381 // Fill array by elements without iterators. 4382 unsigned Pos = 0; 4383 bool HasIterator = false; 4384 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) { 4385 if (C->getModifier()) { 4386 HasIterator = true; 4387 continue; 4388 } 4389 for (const Expr *E : C->varlists()) { 4390 llvm::Value *Addr; 4391 llvm::Value *Size; 4392 std::tie(Addr, Size) = getPointerAndSize(CGF, E); 4393 LValue Base = 4394 CGF.MakeAddrLValue(CGF.Builder.CreateConstGEP(AffinitiesArray, Pos), 4395 KmpTaskAffinityInfoTy); 4396 // affs[i].base_addr = &<Affinities[i].second>; 4397 LValue BaseAddrLVal = CGF.EmitLValueForField( 4398 Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr)); 4399 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy), 4400 BaseAddrLVal); 4401 // affs[i].len = sizeof(<Affinities[i].second>); 4402 LValue LenLVal = CGF.EmitLValueForField( 4403 Base, *std::next(KmpAffinityInfoRD->field_begin(), Len)); 4404 CGF.EmitStoreOfScalar(Size, LenLVal); 4405 ++Pos; 4406 } 4407 } 4408 LValue PosLVal; 4409 if (HasIterator) { 4410 PosLVal = CGF.MakeAddrLValue( 4411 CGF.CreateMemTemp(C.getSizeType(), "affs.counter.addr"), 4412 C.getSizeType()); 4413 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal); 4414 } 4415 // Process elements with iterators. 4416 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) { 4417 const Expr *Modifier = C->getModifier(); 4418 if (!Modifier) 4419 continue; 4420 OMPIteratorGeneratorScope IteratorScope( 4421 CGF, cast_or_null<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts())); 4422 for (const Expr *E : C->varlists()) { 4423 llvm::Value *Addr; 4424 llvm::Value *Size; 4425 std::tie(Addr, Size) = getPointerAndSize(CGF, E); 4426 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4427 LValue Base = CGF.MakeAddrLValue( 4428 Address(CGF.Builder.CreateGEP(AffinitiesArray.getPointer(), Idx), 4429 AffinitiesArray.getAlignment()), 4430 KmpTaskAffinityInfoTy); 4431 // affs[i].base_addr = &<Affinities[i].second>; 4432 LValue BaseAddrLVal = CGF.EmitLValueForField( 4433 Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr)); 4434 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy), 4435 BaseAddrLVal); 4436 // affs[i].len = sizeof(<Affinities[i].second>); 4437 LValue LenLVal = CGF.EmitLValueForField( 4438 Base, *std::next(KmpAffinityInfoRD->field_begin(), Len)); 4439 CGF.EmitStoreOfScalar(Size, LenLVal); 4440 Idx = CGF.Builder.CreateNUWAdd( 4441 Idx, llvm::ConstantInt::get(Idx->getType(), 1)); 4442 CGF.EmitStoreOfScalar(Idx, PosLVal); 4443 } 4444 } 4445 // Call to kmp_int32 __kmpc_omp_reg_task_with_affinity(ident_t *loc_ref, 4446 // kmp_int32 gtid, kmp_task_t *new_task, kmp_int32 4447 // naffins, kmp_task_affinity_info_t *affin_list); 4448 llvm::Value *LocRef = emitUpdateLocation(CGF, Loc); 4449 llvm::Value *GTid = getThreadID(CGF, Loc); 4450 llvm::Value *AffinListPtr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4451 AffinitiesArray.getPointer(), CGM.VoidPtrTy); 4452 // FIXME: Emit the function and ignore its result for now unless the 4453 // runtime function is properly implemented. 4454 (void)CGF.EmitRuntimeCall( 4455 OMPBuilder.getOrCreateRuntimeFunction( 4456 CGM.getModule(), OMPRTL___kmpc_omp_reg_task_with_affinity), 4457 {LocRef, GTid, NewTask, NumOfElements, AffinListPtr}); 4458 } 4459 llvm::Value *NewTaskNewTaskTTy = 4460 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4461 NewTask, KmpTaskTWithPrivatesPtrTy); 4462 LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy, 4463 KmpTaskTWithPrivatesQTy); 4464 LValue TDBase = 4465 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4466 // Fill the data in the resulting kmp_task_t record. 4467 // Copy shareds if there are any. 4468 Address KmpTaskSharedsPtr = Address::invalid(); 4469 if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) { 4470 KmpTaskSharedsPtr = 4471 Address(CGF.EmitLoadOfScalar( 4472 CGF.EmitLValueForField( 4473 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), 4474 KmpTaskTShareds)), 4475 Loc), 4476 CGM.getNaturalTypeAlignment(SharedsTy)); 4477 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy); 4478 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy); 4479 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap); 4480 } 4481 // Emit initial values for private copies (if any). 4482 TaskResultTy Result; 4483 if (!Privates.empty()) { 4484 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD, 4485 SharedsTy, SharedsPtrTy, Data, Privates, 4486 /*ForDup=*/false); 4487 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) && 4488 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) { 4489 Result.TaskDupFn = emitTaskDupFunction( 4490 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD, 4491 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates, 4492 /*WithLastIter=*/!Data.LastprivateVars.empty()); 4493 } 4494 } 4495 // Fields of union "kmp_cmplrdata_t" for destructors and priority. 4496 enum { Priority = 0, Destructors = 1 }; 4497 // Provide pointer to function with destructors for privates. 4498 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1); 4499 const RecordDecl *KmpCmplrdataUD = 4500 (*FI)->getType()->getAsUnionType()->getDecl(); 4501 if (NeedsCleanup) { 4502 llvm::Value *DestructorFn = emitDestructorsFunction( 4503 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 4504 KmpTaskTWithPrivatesQTy); 4505 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI); 4506 LValue DestructorsLV = CGF.EmitLValueForField( 4507 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors)); 4508 CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4509 DestructorFn, KmpRoutineEntryPtrTy), 4510 DestructorsLV); 4511 } 4512 // Set priority. 4513 if (Data.Priority.getInt()) { 4514 LValue Data2LV = CGF.EmitLValueForField( 4515 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2)); 4516 LValue PriorityLV = CGF.EmitLValueForField( 4517 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority)); 4518 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV); 4519 } 4520 Result.NewTask = NewTask; 4521 Result.TaskEntry = TaskEntry; 4522 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy; 4523 Result.TDBase = TDBase; 4524 Result.KmpTaskTQTyRD = KmpTaskTQTyRD; 4525 return Result; 4526 } 4527 4528 namespace { 4529 /// Dependence kind for RTL. 4530 enum RTLDependenceKindTy { 4531 DepIn = 0x01, 4532 DepInOut = 0x3, 4533 DepMutexInOutSet = 0x4 4534 }; 4535 /// Fields ids in kmp_depend_info record. 4536 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags }; 4537 } // namespace 4538 4539 /// Translates internal dependency kind into the runtime kind. 4540 static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) { 4541 RTLDependenceKindTy DepKind; 4542 switch (K) { 4543 case OMPC_DEPEND_in: 4544 DepKind = DepIn; 4545 break; 4546 // Out and InOut dependencies must use the same code. 4547 case OMPC_DEPEND_out: 4548 case OMPC_DEPEND_inout: 4549 DepKind = DepInOut; 4550 break; 4551 case OMPC_DEPEND_mutexinoutset: 4552 DepKind = DepMutexInOutSet; 4553 break; 4554 case OMPC_DEPEND_source: 4555 case OMPC_DEPEND_sink: 4556 case OMPC_DEPEND_depobj: 4557 case OMPC_DEPEND_unknown: 4558 llvm_unreachable("Unknown task dependence type"); 4559 } 4560 return DepKind; 4561 } 4562 4563 /// Builds kmp_depend_info, if it is not built yet, and builds flags type. 4564 static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy, 4565 QualType &FlagsTy) { 4566 FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false); 4567 if (KmpDependInfoTy.isNull()) { 4568 RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info"); 4569 KmpDependInfoRD->startDefinition(); 4570 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType()); 4571 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType()); 4572 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy); 4573 KmpDependInfoRD->completeDefinition(); 4574 KmpDependInfoTy = C.getRecordType(KmpDependInfoRD); 4575 } 4576 } 4577 4578 std::pair<llvm::Value *, LValue> 4579 CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal, 4580 SourceLocation Loc) { 4581 ASTContext &C = CGM.getContext(); 4582 QualType FlagsTy; 4583 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4584 RecordDecl *KmpDependInfoRD = 4585 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4586 LValue Base = CGF.EmitLoadOfPointerLValue( 4587 DepobjLVal.getAddress(CGF), 4588 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 4589 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 4590 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4591 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy)); 4592 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 4593 Base.getTBAAInfo()); 4594 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 4595 Addr.getPointer(), 4596 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 4597 LValue NumDepsBase = CGF.MakeAddrLValue( 4598 Address(DepObjAddr, Addr.getAlignment()), KmpDependInfoTy, 4599 Base.getBaseInfo(), Base.getTBAAInfo()); 4600 // NumDeps = deps[i].base_addr; 4601 LValue BaseAddrLVal = CGF.EmitLValueForField( 4602 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4603 llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc); 4604 return std::make_pair(NumDeps, Base); 4605 } 4606 4607 static void emitDependData(CodeGenFunction &CGF, QualType &KmpDependInfoTy, 4608 llvm::PointerUnion<unsigned *, LValue *> Pos, 4609 const OMPTaskDataTy::DependData &Data, 4610 Address DependenciesArray) { 4611 CodeGenModule &CGM = CGF.CGM; 4612 ASTContext &C = CGM.getContext(); 4613 QualType FlagsTy; 4614 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4615 RecordDecl *KmpDependInfoRD = 4616 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4617 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 4618 4619 OMPIteratorGeneratorScope IteratorScope( 4620 CGF, cast_or_null<OMPIteratorExpr>( 4621 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts() 4622 : nullptr)); 4623 for (const Expr *E : Data.DepExprs) { 4624 llvm::Value *Addr; 4625 llvm::Value *Size; 4626 std::tie(Addr, Size) = getPointerAndSize(CGF, E); 4627 LValue Base; 4628 if (unsigned *P = Pos.dyn_cast<unsigned *>()) { 4629 Base = CGF.MakeAddrLValue( 4630 CGF.Builder.CreateConstGEP(DependenciesArray, *P), KmpDependInfoTy); 4631 } else { 4632 LValue &PosLVal = *Pos.get<LValue *>(); 4633 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4634 Base = CGF.MakeAddrLValue( 4635 Address(CGF.Builder.CreateGEP(DependenciesArray.getPointer(), Idx), 4636 DependenciesArray.getAlignment()), 4637 KmpDependInfoTy); 4638 } 4639 // deps[i].base_addr = &<Dependencies[i].second>; 4640 LValue BaseAddrLVal = CGF.EmitLValueForField( 4641 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4642 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy), 4643 BaseAddrLVal); 4644 // deps[i].len = sizeof(<Dependencies[i].second>); 4645 LValue LenLVal = CGF.EmitLValueForField( 4646 Base, *std::next(KmpDependInfoRD->field_begin(), Len)); 4647 CGF.EmitStoreOfScalar(Size, LenLVal); 4648 // deps[i].flags = <Dependencies[i].first>; 4649 RTLDependenceKindTy DepKind = translateDependencyKind(Data.DepKind); 4650 LValue FlagsLVal = CGF.EmitLValueForField( 4651 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 4652 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 4653 FlagsLVal); 4654 if (unsigned *P = Pos.dyn_cast<unsigned *>()) { 4655 ++(*P); 4656 } else { 4657 LValue &PosLVal = *Pos.get<LValue *>(); 4658 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4659 Idx = CGF.Builder.CreateNUWAdd(Idx, 4660 llvm::ConstantInt::get(Idx->getType(), 1)); 4661 CGF.EmitStoreOfScalar(Idx, PosLVal); 4662 } 4663 } 4664 } 4665 4666 static SmallVector<llvm::Value *, 4> 4667 emitDepobjElementsSizes(CodeGenFunction &CGF, QualType &KmpDependInfoTy, 4668 const OMPTaskDataTy::DependData &Data) { 4669 assert(Data.DepKind == OMPC_DEPEND_depobj && 4670 "Expected depobj dependecy kind."); 4671 SmallVector<llvm::Value *, 4> Sizes; 4672 SmallVector<LValue, 4> SizeLVals; 4673 ASTContext &C = CGF.getContext(); 4674 QualType FlagsTy; 4675 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4676 RecordDecl *KmpDependInfoRD = 4677 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4678 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 4679 llvm::Type *KmpDependInfoPtrT = CGF.ConvertTypeForMem(KmpDependInfoPtrTy); 4680 { 4681 OMPIteratorGeneratorScope IteratorScope( 4682 CGF, cast_or_null<OMPIteratorExpr>( 4683 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts() 4684 : nullptr)); 4685 for (const Expr *E : Data.DepExprs) { 4686 LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts()); 4687 LValue Base = CGF.EmitLoadOfPointerLValue( 4688 DepobjLVal.getAddress(CGF), 4689 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 4690 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4691 Base.getAddress(CGF), KmpDependInfoPtrT); 4692 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 4693 Base.getTBAAInfo()); 4694 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 4695 Addr.getPointer(), 4696 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 4697 LValue NumDepsBase = CGF.MakeAddrLValue( 4698 Address(DepObjAddr, Addr.getAlignment()), KmpDependInfoTy, 4699 Base.getBaseInfo(), Base.getTBAAInfo()); 4700 // NumDeps = deps[i].base_addr; 4701 LValue BaseAddrLVal = CGF.EmitLValueForField( 4702 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4703 llvm::Value *NumDeps = 4704 CGF.EmitLoadOfScalar(BaseAddrLVal, E->getExprLoc()); 4705 LValue NumLVal = CGF.MakeAddrLValue( 4706 CGF.CreateMemTemp(C.getUIntPtrType(), "depobj.size.addr"), 4707 C.getUIntPtrType()); 4708 CGF.InitTempAlloca(NumLVal.getAddress(CGF), 4709 llvm::ConstantInt::get(CGF.IntPtrTy, 0)); 4710 llvm::Value *PrevVal = CGF.EmitLoadOfScalar(NumLVal, E->getExprLoc()); 4711 llvm::Value *Add = CGF.Builder.CreateNUWAdd(PrevVal, NumDeps); 4712 CGF.EmitStoreOfScalar(Add, NumLVal); 4713 SizeLVals.push_back(NumLVal); 4714 } 4715 } 4716 for (unsigned I = 0, E = SizeLVals.size(); I < E; ++I) { 4717 llvm::Value *Size = 4718 CGF.EmitLoadOfScalar(SizeLVals[I], Data.DepExprs[I]->getExprLoc()); 4719 Sizes.push_back(Size); 4720 } 4721 return Sizes; 4722 } 4723 4724 static void emitDepobjElements(CodeGenFunction &CGF, QualType &KmpDependInfoTy, 4725 LValue PosLVal, 4726 const OMPTaskDataTy::DependData &Data, 4727 Address DependenciesArray) { 4728 assert(Data.DepKind == OMPC_DEPEND_depobj && 4729 "Expected depobj dependecy kind."); 4730 ASTContext &C = CGF.getContext(); 4731 QualType FlagsTy; 4732 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4733 RecordDecl *KmpDependInfoRD = 4734 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4735 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 4736 llvm::Type *KmpDependInfoPtrT = CGF.ConvertTypeForMem(KmpDependInfoPtrTy); 4737 llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy); 4738 { 4739 OMPIteratorGeneratorScope IteratorScope( 4740 CGF, cast_or_null<OMPIteratorExpr>( 4741 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts() 4742 : nullptr)); 4743 for (unsigned I = 0, End = Data.DepExprs.size(); I < End; ++I) { 4744 const Expr *E = Data.DepExprs[I]; 4745 LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts()); 4746 LValue Base = CGF.EmitLoadOfPointerLValue( 4747 DepobjLVal.getAddress(CGF), 4748 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 4749 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4750 Base.getAddress(CGF), KmpDependInfoPtrT); 4751 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 4752 Base.getTBAAInfo()); 4753 4754 // Get number of elements in a single depobj. 4755 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 4756 Addr.getPointer(), 4757 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 4758 LValue NumDepsBase = CGF.MakeAddrLValue( 4759 Address(DepObjAddr, Addr.getAlignment()), KmpDependInfoTy, 4760 Base.getBaseInfo(), Base.getTBAAInfo()); 4761 // NumDeps = deps[i].base_addr; 4762 LValue BaseAddrLVal = CGF.EmitLValueForField( 4763 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4764 llvm::Value *NumDeps = 4765 CGF.EmitLoadOfScalar(BaseAddrLVal, E->getExprLoc()); 4766 4767 // memcopy dependency data. 4768 llvm::Value *Size = CGF.Builder.CreateNUWMul( 4769 ElSize, 4770 CGF.Builder.CreateIntCast(NumDeps, CGF.SizeTy, /*isSigned=*/false)); 4771 llvm::Value *Pos = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4772 Address DepAddr = 4773 Address(CGF.Builder.CreateGEP(DependenciesArray.getPointer(), Pos), 4774 DependenciesArray.getAlignment()); 4775 CGF.Builder.CreateMemCpy(DepAddr, Base.getAddress(CGF), Size); 4776 4777 // Increase pos. 4778 // pos += size; 4779 llvm::Value *Add = CGF.Builder.CreateNUWAdd(Pos, NumDeps); 4780 CGF.EmitStoreOfScalar(Add, PosLVal); 4781 } 4782 } 4783 } 4784 4785 std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause( 4786 CodeGenFunction &CGF, ArrayRef<OMPTaskDataTy::DependData> Dependencies, 4787 SourceLocation Loc) { 4788 if (llvm::all_of(Dependencies, [](const OMPTaskDataTy::DependData &D) { 4789 return D.DepExprs.empty(); 4790 })) 4791 return std::make_pair(nullptr, Address::invalid()); 4792 // Process list of dependencies. 4793 ASTContext &C = CGM.getContext(); 4794 Address DependenciesArray = Address::invalid(); 4795 llvm::Value *NumOfElements = nullptr; 4796 unsigned NumDependencies = std::accumulate( 4797 Dependencies.begin(), Dependencies.end(), 0, 4798 [](unsigned V, const OMPTaskDataTy::DependData &D) { 4799 return D.DepKind == OMPC_DEPEND_depobj 4800 ? V 4801 : (V + (D.IteratorExpr ? 0 : D.DepExprs.size())); 4802 }); 4803 QualType FlagsTy; 4804 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4805 bool HasDepobjDeps = false; 4806 bool HasRegularWithIterators = false; 4807 llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0); 4808 llvm::Value *NumOfRegularWithIterators = 4809 llvm::ConstantInt::get(CGF.IntPtrTy, 1); 4810 // Calculate number of depobj dependecies and regular deps with the iterators. 4811 for (const OMPTaskDataTy::DependData &D : Dependencies) { 4812 if (D.DepKind == OMPC_DEPEND_depobj) { 4813 SmallVector<llvm::Value *, 4> Sizes = 4814 emitDepobjElementsSizes(CGF, KmpDependInfoTy, D); 4815 for (llvm::Value *Size : Sizes) { 4816 NumOfDepobjElements = 4817 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, Size); 4818 } 4819 HasDepobjDeps = true; 4820 continue; 4821 } 4822 // Include number of iterations, if any. 4823 if (const auto *IE = cast_or_null<OMPIteratorExpr>(D.IteratorExpr)) { 4824 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) { 4825 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper); 4826 Sz = CGF.Builder.CreateIntCast(Sz, CGF.IntPtrTy, /*isSigned=*/false); 4827 NumOfRegularWithIterators = 4828 CGF.Builder.CreateNUWMul(NumOfRegularWithIterators, Sz); 4829 } 4830 HasRegularWithIterators = true; 4831 continue; 4832 } 4833 } 4834 4835 QualType KmpDependInfoArrayTy; 4836 if (HasDepobjDeps || HasRegularWithIterators) { 4837 NumOfElements = llvm::ConstantInt::get(CGM.IntPtrTy, NumDependencies, 4838 /*isSigned=*/false); 4839 if (HasDepobjDeps) { 4840 NumOfElements = 4841 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumOfElements); 4842 } 4843 if (HasRegularWithIterators) { 4844 NumOfElements = 4845 CGF.Builder.CreateNUWAdd(NumOfRegularWithIterators, NumOfElements); 4846 } 4847 OpaqueValueExpr OVE(Loc, 4848 C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0), 4849 VK_RValue); 4850 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, 4851 RValue::get(NumOfElements)); 4852 KmpDependInfoArrayTy = 4853 C.getVariableArrayType(KmpDependInfoTy, &OVE, ArrayType::Normal, 4854 /*IndexTypeQuals=*/0, SourceRange(Loc, Loc)); 4855 // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy); 4856 // Properly emit variable-sized array. 4857 auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy, 4858 ImplicitParamDecl::Other); 4859 CGF.EmitVarDecl(*PD); 4860 DependenciesArray = CGF.GetAddrOfLocalVar(PD); 4861 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty, 4862 /*isSigned=*/false); 4863 } else { 4864 KmpDependInfoArrayTy = C.getConstantArrayType( 4865 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), nullptr, 4866 ArrayType::Normal, /*IndexTypeQuals=*/0); 4867 DependenciesArray = 4868 CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr"); 4869 DependenciesArray = CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0); 4870 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies, 4871 /*isSigned=*/false); 4872 } 4873 unsigned Pos = 0; 4874 for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) { 4875 if (Dependencies[I].DepKind == OMPC_DEPEND_depobj || 4876 Dependencies[I].IteratorExpr) 4877 continue; 4878 emitDependData(CGF, KmpDependInfoTy, &Pos, Dependencies[I], 4879 DependenciesArray); 4880 } 4881 // Copy regular dependecies with iterators. 4882 LValue PosLVal = CGF.MakeAddrLValue( 4883 CGF.CreateMemTemp(C.getSizeType(), "dep.counter.addr"), C.getSizeType()); 4884 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal); 4885 for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) { 4886 if (Dependencies[I].DepKind == OMPC_DEPEND_depobj || 4887 !Dependencies[I].IteratorExpr) 4888 continue; 4889 emitDependData(CGF, KmpDependInfoTy, &PosLVal, Dependencies[I], 4890 DependenciesArray); 4891 } 4892 // Copy final depobj arrays without iterators. 4893 if (HasDepobjDeps) { 4894 for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) { 4895 if (Dependencies[I].DepKind != OMPC_DEPEND_depobj) 4896 continue; 4897 emitDepobjElements(CGF, KmpDependInfoTy, PosLVal, Dependencies[I], 4898 DependenciesArray); 4899 } 4900 } 4901 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4902 DependenciesArray, CGF.VoidPtrTy); 4903 return std::make_pair(NumOfElements, DependenciesArray); 4904 } 4905 4906 Address CGOpenMPRuntime::emitDepobjDependClause( 4907 CodeGenFunction &CGF, const OMPTaskDataTy::DependData &Dependencies, 4908 SourceLocation Loc) { 4909 if (Dependencies.DepExprs.empty()) 4910 return Address::invalid(); 4911 // Process list of dependencies. 4912 ASTContext &C = CGM.getContext(); 4913 Address DependenciesArray = Address::invalid(); 4914 unsigned NumDependencies = Dependencies.DepExprs.size(); 4915 QualType FlagsTy; 4916 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4917 RecordDecl *KmpDependInfoRD = 4918 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4919 4920 llvm::Value *Size; 4921 // Define type kmp_depend_info[<Dependencies.size()>]; 4922 // For depobj reserve one extra element to store the number of elements. 4923 // It is required to handle depobj(x) update(in) construct. 4924 // kmp_depend_info[<Dependencies.size()>] deps; 4925 llvm::Value *NumDepsVal; 4926 CharUnits Align = C.getTypeAlignInChars(KmpDependInfoTy); 4927 if (const auto *IE = 4928 cast_or_null<OMPIteratorExpr>(Dependencies.IteratorExpr)) { 4929 NumDepsVal = llvm::ConstantInt::get(CGF.SizeTy, 1); 4930 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) { 4931 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper); 4932 Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false); 4933 NumDepsVal = CGF.Builder.CreateNUWMul(NumDepsVal, Sz); 4934 } 4935 Size = CGF.Builder.CreateNUWAdd(llvm::ConstantInt::get(CGF.SizeTy, 1), 4936 NumDepsVal); 4937 CharUnits SizeInBytes = 4938 C.getTypeSizeInChars(KmpDependInfoTy).alignTo(Align); 4939 llvm::Value *RecSize = CGM.getSize(SizeInBytes); 4940 Size = CGF.Builder.CreateNUWMul(Size, RecSize); 4941 NumDepsVal = 4942 CGF.Builder.CreateIntCast(NumDepsVal, CGF.IntPtrTy, /*isSigned=*/false); 4943 } else { 4944 QualType KmpDependInfoArrayTy = C.getConstantArrayType( 4945 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1), 4946 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 4947 CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy); 4948 Size = CGM.getSize(Sz.alignTo(Align)); 4949 NumDepsVal = llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies); 4950 } 4951 // Need to allocate on the dynamic memory. 4952 llvm::Value *ThreadID = getThreadID(CGF, Loc); 4953 // Use default allocator. 4954 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 4955 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 4956 4957 llvm::Value *Addr = 4958 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 4959 CGM.getModule(), OMPRTL___kmpc_alloc), 4960 Args, ".dep.arr.addr"); 4961 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4962 Addr, CGF.ConvertTypeForMem(KmpDependInfoTy)->getPointerTo()); 4963 DependenciesArray = Address(Addr, Align); 4964 // Write number of elements in the first element of array for depobj. 4965 LValue Base = CGF.MakeAddrLValue(DependenciesArray, KmpDependInfoTy); 4966 // deps[i].base_addr = NumDependencies; 4967 LValue BaseAddrLVal = CGF.EmitLValueForField( 4968 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4969 CGF.EmitStoreOfScalar(NumDepsVal, BaseAddrLVal); 4970 llvm::PointerUnion<unsigned *, LValue *> Pos; 4971 unsigned Idx = 1; 4972 LValue PosLVal; 4973 if (Dependencies.IteratorExpr) { 4974 PosLVal = CGF.MakeAddrLValue( 4975 CGF.CreateMemTemp(C.getSizeType(), "iterator.counter.addr"), 4976 C.getSizeType()); 4977 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Idx), PosLVal, 4978 /*IsInit=*/true); 4979 Pos = &PosLVal; 4980 } else { 4981 Pos = &Idx; 4982 } 4983 emitDependData(CGF, KmpDependInfoTy, Pos, Dependencies, DependenciesArray); 4984 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4985 CGF.Builder.CreateConstGEP(DependenciesArray, 1), CGF.VoidPtrTy); 4986 return DependenciesArray; 4987 } 4988 4989 void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal, 4990 SourceLocation Loc) { 4991 ASTContext &C = CGM.getContext(); 4992 QualType FlagsTy; 4993 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4994 LValue Base = CGF.EmitLoadOfPointerLValue( 4995 DepobjLVal.getAddress(CGF), 4996 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 4997 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 4998 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4999 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy)); 5000 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 5001 Addr.getPointer(), 5002 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 5003 DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr, 5004 CGF.VoidPtrTy); 5005 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5006 // Use default allocator. 5007 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5008 llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator}; 5009 5010 // _kmpc_free(gtid, addr, nullptr); 5011 (void)CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 5012 CGM.getModule(), OMPRTL___kmpc_free), 5013 Args); 5014 } 5015 5016 void CGOpenMPRuntime::emitUpdateClause(CodeGenFunction &CGF, LValue DepobjLVal, 5017 OpenMPDependClauseKind NewDepKind, 5018 SourceLocation Loc) { 5019 ASTContext &C = CGM.getContext(); 5020 QualType FlagsTy; 5021 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5022 RecordDecl *KmpDependInfoRD = 5023 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5024 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5025 llvm::Value *NumDeps; 5026 LValue Base; 5027 std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc); 5028 5029 Address Begin = Base.getAddress(CGF); 5030 // Cast from pointer to array type to pointer to single element. 5031 llvm::Value *End = CGF.Builder.CreateGEP(Begin.getPointer(), NumDeps); 5032 // The basic structure here is a while-do loop. 5033 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body"); 5034 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done"); 5035 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5036 CGF.EmitBlock(BodyBB); 5037 llvm::PHINode *ElementPHI = 5038 CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast"); 5039 ElementPHI->addIncoming(Begin.getPointer(), EntryBB); 5040 Begin = Address(ElementPHI, Begin.getAlignment()); 5041 Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(), 5042 Base.getTBAAInfo()); 5043 // deps[i].flags = NewDepKind; 5044 RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind); 5045 LValue FlagsLVal = CGF.EmitLValueForField( 5046 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5047 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5048 FlagsLVal); 5049 5050 // Shift the address forward by one element. 5051 Address ElementNext = 5052 CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext"); 5053 ElementPHI->addIncoming(ElementNext.getPointer(), 5054 CGF.Builder.GetInsertBlock()); 5055 llvm::Value *IsEmpty = 5056 CGF.Builder.CreateICmpEQ(ElementNext.getPointer(), End, "omp.isempty"); 5057 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5058 // Done. 5059 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5060 } 5061 5062 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 5063 const OMPExecutableDirective &D, 5064 llvm::Function *TaskFunction, 5065 QualType SharedsTy, Address Shareds, 5066 const Expr *IfCond, 5067 const OMPTaskDataTy &Data) { 5068 if (!CGF.HaveInsertPoint()) 5069 return; 5070 5071 TaskResultTy Result = 5072 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5073 llvm::Value *NewTask = Result.NewTask; 5074 llvm::Function *TaskEntry = Result.TaskEntry; 5075 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy; 5076 LValue TDBase = Result.TDBase; 5077 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD; 5078 // Process list of dependences. 5079 Address DependenciesArray = Address::invalid(); 5080 llvm::Value *NumOfElements; 5081 std::tie(NumOfElements, DependenciesArray) = 5082 emitDependClause(CGF, Data.Dependences, Loc); 5083 5084 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5085 // libcall. 5086 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 5087 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 5088 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence 5089 // list is not empty 5090 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5091 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5092 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask }; 5093 llvm::Value *DepTaskArgs[7]; 5094 if (!Data.Dependences.empty()) { 5095 DepTaskArgs[0] = UpLoc; 5096 DepTaskArgs[1] = ThreadID; 5097 DepTaskArgs[2] = NewTask; 5098 DepTaskArgs[3] = NumOfElements; 5099 DepTaskArgs[4] = DependenciesArray.getPointer(); 5100 DepTaskArgs[5] = CGF.Builder.getInt32(0); 5101 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5102 } 5103 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs, 5104 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) { 5105 if (!Data.Tied) { 5106 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 5107 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI); 5108 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal); 5109 } 5110 if (!Data.Dependences.empty()) { 5111 CGF.EmitRuntimeCall( 5112 OMPBuilder.getOrCreateRuntimeFunction( 5113 CGM.getModule(), OMPRTL___kmpc_omp_task_with_deps), 5114 DepTaskArgs); 5115 } else { 5116 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 5117 CGM.getModule(), OMPRTL___kmpc_omp_task), 5118 TaskArgs); 5119 } 5120 // Check if parent region is untied and build return for untied task; 5121 if (auto *Region = 5122 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 5123 Region->emitUntiedSwitch(CGF); 5124 }; 5125 5126 llvm::Value *DepWaitTaskArgs[6]; 5127 if (!Data.Dependences.empty()) { 5128 DepWaitTaskArgs[0] = UpLoc; 5129 DepWaitTaskArgs[1] = ThreadID; 5130 DepWaitTaskArgs[2] = NumOfElements; 5131 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 5132 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 5133 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5134 } 5135 auto &M = CGM.getModule(); 5136 auto &&ElseCodeGen = [this, &M, &TaskArgs, ThreadID, NewTaskNewTaskTTy, 5137 TaskEntry, &Data, &DepWaitTaskArgs, 5138 Loc](CodeGenFunction &CGF, PrePostActionTy &) { 5139 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 5140 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 5141 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 5142 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 5143 // is specified. 5144 if (!Data.Dependences.empty()) 5145 CGF.EmitRuntimeCall( 5146 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_wait_deps), 5147 DepWaitTaskArgs); 5148 // Call proxy_task_entry(gtid, new_task); 5149 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy, 5150 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 5151 Action.Enter(CGF); 5152 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy}; 5153 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry, 5154 OutlinedFnArgs); 5155 }; 5156 5157 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 5158 // kmp_task_t *new_task); 5159 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 5160 // kmp_task_t *new_task); 5161 RegionCodeGenTy RCG(CodeGen); 5162 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 5163 M, OMPRTL___kmpc_omp_task_begin_if0), 5164 TaskArgs, 5165 OMPBuilder.getOrCreateRuntimeFunction( 5166 M, OMPRTL___kmpc_omp_task_complete_if0), 5167 TaskArgs); 5168 RCG.setAction(Action); 5169 RCG(CGF); 5170 }; 5171 5172 if (IfCond) { 5173 emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen); 5174 } else { 5175 RegionCodeGenTy ThenRCG(ThenCodeGen); 5176 ThenRCG(CGF); 5177 } 5178 } 5179 5180 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, 5181 const OMPLoopDirective &D, 5182 llvm::Function *TaskFunction, 5183 QualType SharedsTy, Address Shareds, 5184 const Expr *IfCond, 5185 const OMPTaskDataTy &Data) { 5186 if (!CGF.HaveInsertPoint()) 5187 return; 5188 TaskResultTy Result = 5189 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5190 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5191 // libcall. 5192 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 5193 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 5194 // sched, kmp_uint64 grainsize, void *task_dup); 5195 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5196 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5197 llvm::Value *IfVal; 5198 if (IfCond) { 5199 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy, 5200 /*isSigned=*/true); 5201 } else { 5202 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1); 5203 } 5204 5205 LValue LBLVal = CGF.EmitLValueForField( 5206 Result.TDBase, 5207 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound)); 5208 const auto *LBVar = 5209 cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl()); 5210 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF), 5211 LBLVal.getQuals(), 5212 /*IsInitializer=*/true); 5213 LValue UBLVal = CGF.EmitLValueForField( 5214 Result.TDBase, 5215 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound)); 5216 const auto *UBVar = 5217 cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl()); 5218 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF), 5219 UBLVal.getQuals(), 5220 /*IsInitializer=*/true); 5221 LValue StLVal = CGF.EmitLValueForField( 5222 Result.TDBase, 5223 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride)); 5224 const auto *StVar = 5225 cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl()); 5226 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF), 5227 StLVal.getQuals(), 5228 /*IsInitializer=*/true); 5229 // Store reductions address. 5230 LValue RedLVal = CGF.EmitLValueForField( 5231 Result.TDBase, 5232 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions)); 5233 if (Data.Reductions) { 5234 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal); 5235 } else { 5236 CGF.EmitNullInitialization(RedLVal.getAddress(CGF), 5237 CGF.getContext().VoidPtrTy); 5238 } 5239 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 }; 5240 llvm::Value *TaskArgs[] = { 5241 UpLoc, 5242 ThreadID, 5243 Result.NewTask, 5244 IfVal, 5245 LBLVal.getPointer(CGF), 5246 UBLVal.getPointer(CGF), 5247 CGF.EmitLoadOfScalar(StLVal, Loc), 5248 llvm::ConstantInt::getSigned( 5249 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler 5250 llvm::ConstantInt::getSigned( 5251 CGF.IntTy, Data.Schedule.getPointer() 5252 ? Data.Schedule.getInt() ? NumTasks : Grainsize 5253 : NoSchedule), 5254 Data.Schedule.getPointer() 5255 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty, 5256 /*isSigned=*/false) 5257 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0), 5258 Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5259 Result.TaskDupFn, CGF.VoidPtrTy) 5260 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)}; 5261 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 5262 CGM.getModule(), OMPRTL___kmpc_taskloop), 5263 TaskArgs); 5264 } 5265 5266 /// Emit reduction operation for each element of array (required for 5267 /// array sections) LHS op = RHS. 5268 /// \param Type Type of array. 5269 /// \param LHSVar Variable on the left side of the reduction operation 5270 /// (references element of array in original variable). 5271 /// \param RHSVar Variable on the right side of the reduction operation 5272 /// (references element of array in original variable). 5273 /// \param RedOpGen Generator of reduction operation with use of LHSVar and 5274 /// RHSVar. 5275 static void EmitOMPAggregateReduction( 5276 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, 5277 const VarDecl *RHSVar, 5278 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *, 5279 const Expr *, const Expr *)> &RedOpGen, 5280 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr, 5281 const Expr *UpExpr = nullptr) { 5282 // Perform element-by-element initialization. 5283 QualType ElementTy; 5284 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar); 5285 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar); 5286 5287 // Drill down to the base element type on both arrays. 5288 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 5289 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr); 5290 5291 llvm::Value *RHSBegin = RHSAddr.getPointer(); 5292 llvm::Value *LHSBegin = LHSAddr.getPointer(); 5293 // Cast from pointer to array type to pointer to single element. 5294 llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements); 5295 // The basic structure here is a while-do loop. 5296 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body"); 5297 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done"); 5298 llvm::Value *IsEmpty = 5299 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty"); 5300 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5301 5302 // Enter the loop body, making that address the current address. 5303 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5304 CGF.EmitBlock(BodyBB); 5305 5306 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 5307 5308 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI( 5309 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast"); 5310 RHSElementPHI->addIncoming(RHSBegin, EntryBB); 5311 Address RHSElementCurrent = 5312 Address(RHSElementPHI, 5313 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5314 5315 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI( 5316 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast"); 5317 LHSElementPHI->addIncoming(LHSBegin, EntryBB); 5318 Address LHSElementCurrent = 5319 Address(LHSElementPHI, 5320 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5321 5322 // Emit copy. 5323 CodeGenFunction::OMPPrivateScope Scope(CGF); 5324 Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; }); 5325 Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; }); 5326 Scope.Privatize(); 5327 RedOpGen(CGF, XExpr, EExpr, UpExpr); 5328 Scope.ForceCleanup(); 5329 5330 // Shift the address forward by one element. 5331 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32( 5332 LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 5333 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32( 5334 RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element"); 5335 // Check whether we've reached the end. 5336 llvm::Value *Done = 5337 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done"); 5338 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 5339 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock()); 5340 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock()); 5341 5342 // Done. 5343 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5344 } 5345 5346 /// Emit reduction combiner. If the combiner is a simple expression emit it as 5347 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of 5348 /// UDR combiner function. 5349 static void emitReductionCombiner(CodeGenFunction &CGF, 5350 const Expr *ReductionOp) { 5351 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 5352 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 5353 if (const auto *DRE = 5354 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 5355 if (const auto *DRD = 5356 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) { 5357 std::pair<llvm::Function *, llvm::Function *> Reduction = 5358 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 5359 RValue Func = RValue::get(Reduction.first); 5360 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 5361 CGF.EmitIgnoredExpr(ReductionOp); 5362 return; 5363 } 5364 CGF.EmitIgnoredExpr(ReductionOp); 5365 } 5366 5367 llvm::Function *CGOpenMPRuntime::emitReductionFunction( 5368 SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates, 5369 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 5370 ArrayRef<const Expr *> ReductionOps) { 5371 ASTContext &C = CGM.getContext(); 5372 5373 // void reduction_func(void *LHSArg, void *RHSArg); 5374 FunctionArgList Args; 5375 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5376 ImplicitParamDecl::Other); 5377 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5378 ImplicitParamDecl::Other); 5379 Args.push_back(&LHSArg); 5380 Args.push_back(&RHSArg); 5381 const auto &CGFI = 5382 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5383 std::string Name = getName({"omp", "reduction", "reduction_func"}); 5384 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 5385 llvm::GlobalValue::InternalLinkage, Name, 5386 &CGM.getModule()); 5387 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 5388 Fn->setDoesNotRecurse(); 5389 CodeGenFunction CGF(CGM); 5390 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 5391 5392 // Dst = (void*[n])(LHSArg); 5393 // Src = (void*[n])(RHSArg); 5394 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5395 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 5396 ArgsType), CGF.getPointerAlign()); 5397 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5398 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 5399 ArgsType), CGF.getPointerAlign()); 5400 5401 // ... 5402 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]); 5403 // ... 5404 CodeGenFunction::OMPPrivateScope Scope(CGF); 5405 auto IPriv = Privates.begin(); 5406 unsigned Idx = 0; 5407 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) { 5408 const auto *RHSVar = 5409 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()); 5410 Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() { 5411 return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar); 5412 }); 5413 const auto *LHSVar = 5414 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()); 5415 Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() { 5416 return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar); 5417 }); 5418 QualType PrivTy = (*IPriv)->getType(); 5419 if (PrivTy->isVariablyModifiedType()) { 5420 // Get array size and emit VLA type. 5421 ++Idx; 5422 Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx); 5423 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem); 5424 const VariableArrayType *VLA = 5425 CGF.getContext().getAsVariableArrayType(PrivTy); 5426 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr()); 5427 CodeGenFunction::OpaqueValueMapping OpaqueMap( 5428 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy))); 5429 CGF.EmitVariablyModifiedType(PrivTy); 5430 } 5431 } 5432 Scope.Privatize(); 5433 IPriv = Privates.begin(); 5434 auto ILHS = LHSExprs.begin(); 5435 auto IRHS = RHSExprs.begin(); 5436 for (const Expr *E : ReductionOps) { 5437 if ((*IPriv)->getType()->isArrayType()) { 5438 // Emit reduction for array section. 5439 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5440 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5441 EmitOMPAggregateReduction( 5442 CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5443 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5444 emitReductionCombiner(CGF, E); 5445 }); 5446 } else { 5447 // Emit reduction for array subscript or single variable. 5448 emitReductionCombiner(CGF, E); 5449 } 5450 ++IPriv; 5451 ++ILHS; 5452 ++IRHS; 5453 } 5454 Scope.ForceCleanup(); 5455 CGF.FinishFunction(); 5456 return Fn; 5457 } 5458 5459 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF, 5460 const Expr *ReductionOp, 5461 const Expr *PrivateRef, 5462 const DeclRefExpr *LHS, 5463 const DeclRefExpr *RHS) { 5464 if (PrivateRef->getType()->isArrayType()) { 5465 // Emit reduction for array section. 5466 const auto *LHSVar = cast<VarDecl>(LHS->getDecl()); 5467 const auto *RHSVar = cast<VarDecl>(RHS->getDecl()); 5468 EmitOMPAggregateReduction( 5469 CGF, PrivateRef->getType(), LHSVar, RHSVar, 5470 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5471 emitReductionCombiner(CGF, ReductionOp); 5472 }); 5473 } else { 5474 // Emit reduction for array subscript or single variable. 5475 emitReductionCombiner(CGF, ReductionOp); 5476 } 5477 } 5478 5479 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc, 5480 ArrayRef<const Expr *> Privates, 5481 ArrayRef<const Expr *> LHSExprs, 5482 ArrayRef<const Expr *> RHSExprs, 5483 ArrayRef<const Expr *> ReductionOps, 5484 ReductionOptionsTy Options) { 5485 if (!CGF.HaveInsertPoint()) 5486 return; 5487 5488 bool WithNowait = Options.WithNowait; 5489 bool SimpleReduction = Options.SimpleReduction; 5490 5491 // Next code should be emitted for reduction: 5492 // 5493 // static kmp_critical_name lock = { 0 }; 5494 // 5495 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) { 5496 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]); 5497 // ... 5498 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1], 5499 // *(Type<n>-1*)rhs[<n>-1]); 5500 // } 5501 // 5502 // ... 5503 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]}; 5504 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5505 // RedList, reduce_func, &<lock>)) { 5506 // case 1: 5507 // ... 5508 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5509 // ... 5510 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5511 // break; 5512 // case 2: 5513 // ... 5514 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5515 // ... 5516 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);] 5517 // break; 5518 // default:; 5519 // } 5520 // 5521 // if SimpleReduction is true, only the next code is generated: 5522 // ... 5523 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5524 // ... 5525 5526 ASTContext &C = CGM.getContext(); 5527 5528 if (SimpleReduction) { 5529 CodeGenFunction::RunCleanupsScope Scope(CGF); 5530 auto IPriv = Privates.begin(); 5531 auto ILHS = LHSExprs.begin(); 5532 auto IRHS = RHSExprs.begin(); 5533 for (const Expr *E : ReductionOps) { 5534 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5535 cast<DeclRefExpr>(*IRHS)); 5536 ++IPriv; 5537 ++ILHS; 5538 ++IRHS; 5539 } 5540 return; 5541 } 5542 5543 // 1. Build a list of reduction variables. 5544 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]}; 5545 auto Size = RHSExprs.size(); 5546 for (const Expr *E : Privates) { 5547 if (E->getType()->isVariablyModifiedType()) 5548 // Reserve place for array size. 5549 ++Size; 5550 } 5551 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size); 5552 QualType ReductionArrayTy = 5553 C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 5554 /*IndexTypeQuals=*/0); 5555 Address ReductionList = 5556 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list"); 5557 auto IPriv = Privates.begin(); 5558 unsigned Idx = 0; 5559 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) { 5560 Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5561 CGF.Builder.CreateStore( 5562 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5563 CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy), 5564 Elem); 5565 if ((*IPriv)->getType()->isVariablyModifiedType()) { 5566 // Store array size. 5567 ++Idx; 5568 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5569 llvm::Value *Size = CGF.Builder.CreateIntCast( 5570 CGF.getVLASize( 5571 CGF.getContext().getAsVariableArrayType((*IPriv)->getType())) 5572 .NumElts, 5573 CGF.SizeTy, /*isSigned=*/false); 5574 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy), 5575 Elem); 5576 } 5577 } 5578 5579 // 2. Emit reduce_func(). 5580 llvm::Function *ReductionFn = emitReductionFunction( 5581 Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates, 5582 LHSExprs, RHSExprs, ReductionOps); 5583 5584 // 3. Create static kmp_critical_name lock = { 0 }; 5585 std::string Name = getName({"reduction"}); 5586 llvm::Value *Lock = getCriticalRegionLock(Name); 5587 5588 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5589 // RedList, reduce_func, &<lock>); 5590 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE); 5591 llvm::Value *ThreadId = getThreadID(CGF, Loc); 5592 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy); 5593 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5594 ReductionList.getPointer(), CGF.VoidPtrTy); 5595 llvm::Value *Args[] = { 5596 IdentTLoc, // ident_t *<loc> 5597 ThreadId, // i32 <gtid> 5598 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n> 5599 ReductionArrayTySize, // size_type sizeof(RedList) 5600 RL, // void *RedList 5601 ReductionFn, // void (*) (void *, void *) <reduce_func> 5602 Lock // kmp_critical_name *&<lock> 5603 }; 5604 llvm::Value *Res = CGF.EmitRuntimeCall( 5605 OMPBuilder.getOrCreateRuntimeFunction( 5606 CGM.getModule(), 5607 WithNowait ? OMPRTL___kmpc_reduce_nowait : OMPRTL___kmpc_reduce), 5608 Args); 5609 5610 // 5. Build switch(res) 5611 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default"); 5612 llvm::SwitchInst *SwInst = 5613 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2); 5614 5615 // 6. Build case 1: 5616 // ... 5617 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5618 // ... 5619 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5620 // break; 5621 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1"); 5622 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB); 5623 CGF.EmitBlock(Case1BB); 5624 5625 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5626 llvm::Value *EndArgs[] = { 5627 IdentTLoc, // ident_t *<loc> 5628 ThreadId, // i32 <gtid> 5629 Lock // kmp_critical_name *&<lock> 5630 }; 5631 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps]( 5632 CodeGenFunction &CGF, PrePostActionTy &Action) { 5633 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5634 auto IPriv = Privates.begin(); 5635 auto ILHS = LHSExprs.begin(); 5636 auto IRHS = RHSExprs.begin(); 5637 for (const Expr *E : ReductionOps) { 5638 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5639 cast<DeclRefExpr>(*IRHS)); 5640 ++IPriv; 5641 ++ILHS; 5642 ++IRHS; 5643 } 5644 }; 5645 RegionCodeGenTy RCG(CodeGen); 5646 CommonActionTy Action( 5647 nullptr, llvm::None, 5648 OMPBuilder.getOrCreateRuntimeFunction( 5649 CGM.getModule(), WithNowait ? OMPRTL___kmpc_end_reduce_nowait 5650 : OMPRTL___kmpc_end_reduce), 5651 EndArgs); 5652 RCG.setAction(Action); 5653 RCG(CGF); 5654 5655 CGF.EmitBranch(DefaultBB); 5656 5657 // 7. Build case 2: 5658 // ... 5659 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5660 // ... 5661 // break; 5662 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2"); 5663 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB); 5664 CGF.EmitBlock(Case2BB); 5665 5666 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps]( 5667 CodeGenFunction &CGF, PrePostActionTy &Action) { 5668 auto ILHS = LHSExprs.begin(); 5669 auto IRHS = RHSExprs.begin(); 5670 auto IPriv = Privates.begin(); 5671 for (const Expr *E : ReductionOps) { 5672 const Expr *XExpr = nullptr; 5673 const Expr *EExpr = nullptr; 5674 const Expr *UpExpr = nullptr; 5675 BinaryOperatorKind BO = BO_Comma; 5676 if (const auto *BO = dyn_cast<BinaryOperator>(E)) { 5677 if (BO->getOpcode() == BO_Assign) { 5678 XExpr = BO->getLHS(); 5679 UpExpr = BO->getRHS(); 5680 } 5681 } 5682 // Try to emit update expression as a simple atomic. 5683 const Expr *RHSExpr = UpExpr; 5684 if (RHSExpr) { 5685 // Analyze RHS part of the whole expression. 5686 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>( 5687 RHSExpr->IgnoreParenImpCasts())) { 5688 // If this is a conditional operator, analyze its condition for 5689 // min/max reduction operator. 5690 RHSExpr = ACO->getCond(); 5691 } 5692 if (const auto *BORHS = 5693 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) { 5694 EExpr = BORHS->getRHS(); 5695 BO = BORHS->getOpcode(); 5696 } 5697 } 5698 if (XExpr) { 5699 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5700 auto &&AtomicRedGen = [BO, VD, 5701 Loc](CodeGenFunction &CGF, const Expr *XExpr, 5702 const Expr *EExpr, const Expr *UpExpr) { 5703 LValue X = CGF.EmitLValue(XExpr); 5704 RValue E; 5705 if (EExpr) 5706 E = CGF.EmitAnyExpr(EExpr); 5707 CGF.EmitOMPAtomicSimpleUpdateExpr( 5708 X, E, BO, /*IsXLHSInRHSPart=*/true, 5709 llvm::AtomicOrdering::Monotonic, Loc, 5710 [&CGF, UpExpr, VD, Loc](RValue XRValue) { 5711 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 5712 PrivateScope.addPrivate( 5713 VD, [&CGF, VD, XRValue, Loc]() { 5714 Address LHSTemp = CGF.CreateMemTemp(VD->getType()); 5715 CGF.emitOMPSimpleStore( 5716 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue, 5717 VD->getType().getNonReferenceType(), Loc); 5718 return LHSTemp; 5719 }); 5720 (void)PrivateScope.Privatize(); 5721 return CGF.EmitAnyExpr(UpExpr); 5722 }); 5723 }; 5724 if ((*IPriv)->getType()->isArrayType()) { 5725 // Emit atomic reduction for array section. 5726 const auto *RHSVar = 5727 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5728 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar, 5729 AtomicRedGen, XExpr, EExpr, UpExpr); 5730 } else { 5731 // Emit atomic reduction for array subscript or single variable. 5732 AtomicRedGen(CGF, XExpr, EExpr, UpExpr); 5733 } 5734 } else { 5735 // Emit as a critical region. 5736 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *, 5737 const Expr *, const Expr *) { 5738 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5739 std::string Name = RT.getName({"atomic_reduction"}); 5740 RT.emitCriticalRegion( 5741 CGF, Name, 5742 [=](CodeGenFunction &CGF, PrePostActionTy &Action) { 5743 Action.Enter(CGF); 5744 emitReductionCombiner(CGF, E); 5745 }, 5746 Loc); 5747 }; 5748 if ((*IPriv)->getType()->isArrayType()) { 5749 const auto *LHSVar = 5750 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5751 const auto *RHSVar = 5752 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5753 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5754 CritRedGen); 5755 } else { 5756 CritRedGen(CGF, nullptr, nullptr, nullptr); 5757 } 5758 } 5759 ++ILHS; 5760 ++IRHS; 5761 ++IPriv; 5762 } 5763 }; 5764 RegionCodeGenTy AtomicRCG(AtomicCodeGen); 5765 if (!WithNowait) { 5766 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>); 5767 llvm::Value *EndArgs[] = { 5768 IdentTLoc, // ident_t *<loc> 5769 ThreadId, // i32 <gtid> 5770 Lock // kmp_critical_name *&<lock> 5771 }; 5772 CommonActionTy Action(nullptr, llvm::None, 5773 OMPBuilder.getOrCreateRuntimeFunction( 5774 CGM.getModule(), OMPRTL___kmpc_end_reduce), 5775 EndArgs); 5776 AtomicRCG.setAction(Action); 5777 AtomicRCG(CGF); 5778 } else { 5779 AtomicRCG(CGF); 5780 } 5781 5782 CGF.EmitBranch(DefaultBB); 5783 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true); 5784 } 5785 5786 /// Generates unique name for artificial threadprivate variables. 5787 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>" 5788 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix, 5789 const Expr *Ref) { 5790 SmallString<256> Buffer; 5791 llvm::raw_svector_ostream Out(Buffer); 5792 const clang::DeclRefExpr *DE; 5793 const VarDecl *D = ::getBaseDecl(Ref, DE); 5794 if (!D) 5795 D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl()); 5796 D = D->getCanonicalDecl(); 5797 std::string Name = CGM.getOpenMPRuntime().getName( 5798 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)}); 5799 Out << Prefix << Name << "_" 5800 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding(); 5801 return std::string(Out.str()); 5802 } 5803 5804 /// Emits reduction initializer function: 5805 /// \code 5806 /// void @.red_init(void* %arg, void* %orig) { 5807 /// %0 = bitcast void* %arg to <type>* 5808 /// store <type> <init>, <type>* %0 5809 /// ret void 5810 /// } 5811 /// \endcode 5812 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM, 5813 SourceLocation Loc, 5814 ReductionCodeGen &RCG, unsigned N) { 5815 ASTContext &C = CGM.getContext(); 5816 QualType VoidPtrTy = C.VoidPtrTy; 5817 VoidPtrTy.addRestrict(); 5818 FunctionArgList Args; 5819 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, VoidPtrTy, 5820 ImplicitParamDecl::Other); 5821 ImplicitParamDecl ParamOrig(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, VoidPtrTy, 5822 ImplicitParamDecl::Other); 5823 Args.emplace_back(&Param); 5824 Args.emplace_back(&ParamOrig); 5825 const auto &FnInfo = 5826 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5827 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 5828 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""}); 5829 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 5830 Name, &CGM.getModule()); 5831 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 5832 Fn->setDoesNotRecurse(); 5833 CodeGenFunction CGF(CGM); 5834 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 5835 Address PrivateAddr = CGF.EmitLoadOfPointer( 5836 CGF.GetAddrOfLocalVar(&Param), 5837 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5838 llvm::Value *Size = nullptr; 5839 // If the size of the reduction item is non-constant, load it from global 5840 // threadprivate variable. 5841 if (RCG.getSizes(N).second) { 5842 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5843 CGF, CGM.getContext().getSizeType(), 5844 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 5845 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 5846 CGM.getContext().getSizeType(), Loc); 5847 } 5848 RCG.emitAggregateType(CGF, N, Size); 5849 LValue OrigLVal; 5850 // If initializer uses initializer from declare reduction construct, emit a 5851 // pointer to the address of the original reduction item (reuired by reduction 5852 // initializer) 5853 if (RCG.usesReductionInitializer(N)) { 5854 Address SharedAddr = CGF.GetAddrOfLocalVar(&ParamOrig); 5855 SharedAddr = CGF.EmitLoadOfPointer( 5856 SharedAddr, 5857 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr()); 5858 OrigLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy); 5859 } else { 5860 OrigLVal = CGF.MakeNaturalAlignAddrLValue( 5861 llvm::ConstantPointerNull::get(CGM.VoidPtrTy), 5862 CGM.getContext().VoidPtrTy); 5863 } 5864 // Emit the initializer: 5865 // %0 = bitcast void* %arg to <type>* 5866 // store <type> <init>, <type>* %0 5867 RCG.emitInitialization(CGF, N, PrivateAddr, OrigLVal, 5868 [](CodeGenFunction &) { return false; }); 5869 CGF.FinishFunction(); 5870 return Fn; 5871 } 5872 5873 /// Emits reduction combiner function: 5874 /// \code 5875 /// void @.red_comb(void* %arg0, void* %arg1) { 5876 /// %lhs = bitcast void* %arg0 to <type>* 5877 /// %rhs = bitcast void* %arg1 to <type>* 5878 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs) 5879 /// store <type> %2, <type>* %lhs 5880 /// ret void 5881 /// } 5882 /// \endcode 5883 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM, 5884 SourceLocation Loc, 5885 ReductionCodeGen &RCG, unsigned N, 5886 const Expr *ReductionOp, 5887 const Expr *LHS, const Expr *RHS, 5888 const Expr *PrivateRef) { 5889 ASTContext &C = CGM.getContext(); 5890 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl()); 5891 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl()); 5892 FunctionArgList Args; 5893 ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 5894 C.VoidPtrTy, ImplicitParamDecl::Other); 5895 ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5896 ImplicitParamDecl::Other); 5897 Args.emplace_back(&ParamInOut); 5898 Args.emplace_back(&ParamIn); 5899 const auto &FnInfo = 5900 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5901 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 5902 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""}); 5903 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 5904 Name, &CGM.getModule()); 5905 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 5906 Fn->setDoesNotRecurse(); 5907 CodeGenFunction CGF(CGM); 5908 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 5909 llvm::Value *Size = nullptr; 5910 // If the size of the reduction item is non-constant, load it from global 5911 // threadprivate variable. 5912 if (RCG.getSizes(N).second) { 5913 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5914 CGF, CGM.getContext().getSizeType(), 5915 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 5916 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 5917 CGM.getContext().getSizeType(), Loc); 5918 } 5919 RCG.emitAggregateType(CGF, N, Size); 5920 // Remap lhs and rhs variables to the addresses of the function arguments. 5921 // %lhs = bitcast void* %arg0 to <type>* 5922 // %rhs = bitcast void* %arg1 to <type>* 5923 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 5924 PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() { 5925 // Pull out the pointer to the variable. 5926 Address PtrAddr = CGF.EmitLoadOfPointer( 5927 CGF.GetAddrOfLocalVar(&ParamInOut), 5928 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5929 return CGF.Builder.CreateElementBitCast( 5930 PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType())); 5931 }); 5932 PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() { 5933 // Pull out the pointer to the variable. 5934 Address PtrAddr = CGF.EmitLoadOfPointer( 5935 CGF.GetAddrOfLocalVar(&ParamIn), 5936 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5937 return CGF.Builder.CreateElementBitCast( 5938 PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType())); 5939 }); 5940 PrivateScope.Privatize(); 5941 // Emit the combiner body: 5942 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs) 5943 // store <type> %2, <type>* %lhs 5944 CGM.getOpenMPRuntime().emitSingleReductionCombiner( 5945 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS), 5946 cast<DeclRefExpr>(RHS)); 5947 CGF.FinishFunction(); 5948 return Fn; 5949 } 5950 5951 /// Emits reduction finalizer function: 5952 /// \code 5953 /// void @.red_fini(void* %arg) { 5954 /// %0 = bitcast void* %arg to <type>* 5955 /// <destroy>(<type>* %0) 5956 /// ret void 5957 /// } 5958 /// \endcode 5959 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM, 5960 SourceLocation Loc, 5961 ReductionCodeGen &RCG, unsigned N) { 5962 if (!RCG.needCleanups(N)) 5963 return nullptr; 5964 ASTContext &C = CGM.getContext(); 5965 FunctionArgList Args; 5966 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5967 ImplicitParamDecl::Other); 5968 Args.emplace_back(&Param); 5969 const auto &FnInfo = 5970 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5971 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 5972 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""}); 5973 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 5974 Name, &CGM.getModule()); 5975 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 5976 Fn->setDoesNotRecurse(); 5977 CodeGenFunction CGF(CGM); 5978 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 5979 Address PrivateAddr = CGF.EmitLoadOfPointer( 5980 CGF.GetAddrOfLocalVar(&Param), 5981 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5982 llvm::Value *Size = nullptr; 5983 // If the size of the reduction item is non-constant, load it from global 5984 // threadprivate variable. 5985 if (RCG.getSizes(N).second) { 5986 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5987 CGF, CGM.getContext().getSizeType(), 5988 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 5989 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 5990 CGM.getContext().getSizeType(), Loc); 5991 } 5992 RCG.emitAggregateType(CGF, N, Size); 5993 // Emit the finalizer body: 5994 // <destroy>(<type>* %0) 5995 RCG.emitCleanups(CGF, N, PrivateAddr); 5996 CGF.FinishFunction(Loc); 5997 return Fn; 5998 } 5999 6000 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit( 6001 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 6002 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 6003 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty()) 6004 return nullptr; 6005 6006 // Build typedef struct: 6007 // kmp_taskred_input { 6008 // void *reduce_shar; // shared reduction item 6009 // void *reduce_orig; // original reduction item used for initialization 6010 // size_t reduce_size; // size of data item 6011 // void *reduce_init; // data initialization routine 6012 // void *reduce_fini; // data finalization routine 6013 // void *reduce_comb; // data combiner routine 6014 // kmp_task_red_flags_t flags; // flags for additional info from compiler 6015 // } kmp_taskred_input_t; 6016 ASTContext &C = CGM.getContext(); 6017 RecordDecl *RD = C.buildImplicitRecord("kmp_taskred_input_t"); 6018 RD->startDefinition(); 6019 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6020 const FieldDecl *OrigFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6021 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType()); 6022 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6023 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6024 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6025 const FieldDecl *FlagsFD = addFieldToRecordDecl( 6026 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false)); 6027 RD->completeDefinition(); 6028 QualType RDType = C.getRecordType(RD); 6029 unsigned Size = Data.ReductionVars.size(); 6030 llvm::APInt ArraySize(/*numBits=*/64, Size); 6031 QualType ArrayRDType = C.getConstantArrayType( 6032 RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 6033 // kmp_task_red_input_t .rd_input.[Size]; 6034 Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input."); 6035 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionOrigs, 6036 Data.ReductionCopies, Data.ReductionOps); 6037 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) { 6038 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt]; 6039 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0), 6040 llvm::ConstantInt::get(CGM.SizeTy, Cnt)}; 6041 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP( 6042 TaskRedInput.getPointer(), Idxs, 6043 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc, 6044 ".rd_input.gep."); 6045 LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType); 6046 // ElemLVal.reduce_shar = &Shareds[Cnt]; 6047 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD); 6048 RCG.emitSharedOrigLValue(CGF, Cnt); 6049 llvm::Value *CastedShared = 6050 CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF)); 6051 CGF.EmitStoreOfScalar(CastedShared, SharedLVal); 6052 // ElemLVal.reduce_orig = &Origs[Cnt]; 6053 LValue OrigLVal = CGF.EmitLValueForField(ElemLVal, OrigFD); 6054 llvm::Value *CastedOrig = 6055 CGF.EmitCastToVoidPtr(RCG.getOrigLValue(Cnt).getPointer(CGF)); 6056 CGF.EmitStoreOfScalar(CastedOrig, OrigLVal); 6057 RCG.emitAggregateType(CGF, Cnt); 6058 llvm::Value *SizeValInChars; 6059 llvm::Value *SizeVal; 6060 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt); 6061 // We use delayed creation/initialization for VLAs and array sections. It is 6062 // required because runtime does not provide the way to pass the sizes of 6063 // VLAs/array sections to initializer/combiner/finalizer functions. Instead 6064 // threadprivate global variables are used to store these values and use 6065 // them in the functions. 6066 bool DelayedCreation = !!SizeVal; 6067 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy, 6068 /*isSigned=*/false); 6069 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD); 6070 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal); 6071 // ElemLVal.reduce_init = init; 6072 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD); 6073 llvm::Value *InitAddr = 6074 CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt)); 6075 CGF.EmitStoreOfScalar(InitAddr, InitLVal); 6076 // ElemLVal.reduce_fini = fini; 6077 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD); 6078 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt); 6079 llvm::Value *FiniAddr = Fini 6080 ? CGF.EmitCastToVoidPtr(Fini) 6081 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 6082 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal); 6083 // ElemLVal.reduce_comb = comb; 6084 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD); 6085 llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction( 6086 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt], 6087 RHSExprs[Cnt], Data.ReductionCopies[Cnt])); 6088 CGF.EmitStoreOfScalar(CombAddr, CombLVal); 6089 // ElemLVal.flags = 0; 6090 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD); 6091 if (DelayedCreation) { 6092 CGF.EmitStoreOfScalar( 6093 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true), 6094 FlagsLVal); 6095 } else 6096 CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF), 6097 FlagsLVal.getType()); 6098 } 6099 if (Data.IsReductionWithTaskMod) { 6100 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int 6101 // is_ws, int num, void *data); 6102 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc); 6103 llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6104 CGM.IntTy, /*isSigned=*/true); 6105 llvm::Value *Args[] = { 6106 IdentTLoc, GTid, 6107 llvm::ConstantInt::get(CGM.IntTy, Data.IsWorksharingReduction ? 1 : 0, 6108 /*isSigned=*/true), 6109 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6110 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6111 TaskRedInput.getPointer(), CGM.VoidPtrTy)}; 6112 return CGF.EmitRuntimeCall( 6113 OMPBuilder.getOrCreateRuntimeFunction( 6114 CGM.getModule(), OMPRTL___kmpc_taskred_modifier_init), 6115 Args); 6116 } 6117 // Build call void *__kmpc_taskred_init(int gtid, int num_data, void *data); 6118 llvm::Value *Args[] = { 6119 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6120 /*isSigned=*/true), 6121 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6122 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(), 6123 CGM.VoidPtrTy)}; 6124 return CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 6125 CGM.getModule(), OMPRTL___kmpc_taskred_init), 6126 Args); 6127 } 6128 6129 void CGOpenMPRuntime::emitTaskReductionFini(CodeGenFunction &CGF, 6130 SourceLocation Loc, 6131 bool IsWorksharingReduction) { 6132 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int 6133 // is_ws, int num, void *data); 6134 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc); 6135 llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6136 CGM.IntTy, /*isSigned=*/true); 6137 llvm::Value *Args[] = {IdentTLoc, GTid, 6138 llvm::ConstantInt::get(CGM.IntTy, 6139 IsWorksharingReduction ? 1 : 0, 6140 /*isSigned=*/true)}; 6141 (void)CGF.EmitRuntimeCall( 6142 OMPBuilder.getOrCreateRuntimeFunction( 6143 CGM.getModule(), OMPRTL___kmpc_task_reduction_modifier_fini), 6144 Args); 6145 } 6146 6147 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 6148 SourceLocation Loc, 6149 ReductionCodeGen &RCG, 6150 unsigned N) { 6151 auto Sizes = RCG.getSizes(N); 6152 // Emit threadprivate global variable if the type is non-constant 6153 // (Sizes.second = nullptr). 6154 if (Sizes.second) { 6155 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy, 6156 /*isSigned=*/false); 6157 Address SizeAddr = getAddrOfArtificialThreadPrivate( 6158 CGF, CGM.getContext().getSizeType(), 6159 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6160 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false); 6161 } 6162 } 6163 6164 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF, 6165 SourceLocation Loc, 6166 llvm::Value *ReductionsPtr, 6167 LValue SharedLVal) { 6168 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 6169 // *d); 6170 llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6171 CGM.IntTy, 6172 /*isSigned=*/true), 6173 ReductionsPtr, 6174 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6175 SharedLVal.getPointer(CGF), CGM.VoidPtrTy)}; 6176 return Address( 6177 CGF.EmitRuntimeCall( 6178 OMPBuilder.getOrCreateRuntimeFunction( 6179 CGM.getModule(), OMPRTL___kmpc_task_reduction_get_th_data), 6180 Args), 6181 SharedLVal.getAlignment()); 6182 } 6183 6184 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 6185 SourceLocation Loc) { 6186 if (!CGF.HaveInsertPoint()) 6187 return; 6188 6189 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) { 6190 OMPBuilder.createTaskwait(CGF.Builder); 6191 } else { 6192 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 6193 // global_tid); 6194 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 6195 // Ignore return result until untied tasks are supported. 6196 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 6197 CGM.getModule(), OMPRTL___kmpc_omp_taskwait), 6198 Args); 6199 } 6200 6201 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 6202 Region->emitUntiedSwitch(CGF); 6203 } 6204 6205 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF, 6206 OpenMPDirectiveKind InnerKind, 6207 const RegionCodeGenTy &CodeGen, 6208 bool HasCancel) { 6209 if (!CGF.HaveInsertPoint()) 6210 return; 6211 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel); 6212 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr); 6213 } 6214 6215 namespace { 6216 enum RTCancelKind { 6217 CancelNoreq = 0, 6218 CancelParallel = 1, 6219 CancelLoop = 2, 6220 CancelSections = 3, 6221 CancelTaskgroup = 4 6222 }; 6223 } // anonymous namespace 6224 6225 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) { 6226 RTCancelKind CancelKind = CancelNoreq; 6227 if (CancelRegion == OMPD_parallel) 6228 CancelKind = CancelParallel; 6229 else if (CancelRegion == OMPD_for) 6230 CancelKind = CancelLoop; 6231 else if (CancelRegion == OMPD_sections) 6232 CancelKind = CancelSections; 6233 else { 6234 assert(CancelRegion == OMPD_taskgroup); 6235 CancelKind = CancelTaskgroup; 6236 } 6237 return CancelKind; 6238 } 6239 6240 void CGOpenMPRuntime::emitCancellationPointCall( 6241 CodeGenFunction &CGF, SourceLocation Loc, 6242 OpenMPDirectiveKind CancelRegion) { 6243 if (!CGF.HaveInsertPoint()) 6244 return; 6245 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 6246 // global_tid, kmp_int32 cncl_kind); 6247 if (auto *OMPRegionInfo = 6248 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6249 // For 'cancellation point taskgroup', the task region info may not have a 6250 // cancel. This may instead happen in another adjacent task. 6251 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) { 6252 llvm::Value *Args[] = { 6253 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 6254 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6255 // Ignore return result until untied tasks are supported. 6256 llvm::Value *Result = CGF.EmitRuntimeCall( 6257 OMPBuilder.getOrCreateRuntimeFunction( 6258 CGM.getModule(), OMPRTL___kmpc_cancellationpoint), 6259 Args); 6260 // if (__kmpc_cancellationpoint()) { 6261 // exit from construct; 6262 // } 6263 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6264 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6265 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6266 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6267 CGF.EmitBlock(ExitBB); 6268 // exit from construct; 6269 CodeGenFunction::JumpDest CancelDest = 6270 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6271 CGF.EmitBranchThroughCleanup(CancelDest); 6272 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6273 } 6274 } 6275 } 6276 6277 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, 6278 const Expr *IfCond, 6279 OpenMPDirectiveKind CancelRegion) { 6280 if (!CGF.HaveInsertPoint()) 6281 return; 6282 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 6283 // kmp_int32 cncl_kind); 6284 auto &M = CGM.getModule(); 6285 if (auto *OMPRegionInfo = 6286 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6287 auto &&ThenGen = [this, &M, Loc, CancelRegion, 6288 OMPRegionInfo](CodeGenFunction &CGF, PrePostActionTy &) { 6289 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6290 llvm::Value *Args[] = { 6291 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc), 6292 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6293 // Ignore return result until untied tasks are supported. 6294 llvm::Value *Result = CGF.EmitRuntimeCall( 6295 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_cancel), Args); 6296 // if (__kmpc_cancel()) { 6297 // exit from construct; 6298 // } 6299 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6300 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6301 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6302 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6303 CGF.EmitBlock(ExitBB); 6304 // exit from construct; 6305 CodeGenFunction::JumpDest CancelDest = 6306 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6307 CGF.EmitBranchThroughCleanup(CancelDest); 6308 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6309 }; 6310 if (IfCond) { 6311 emitIfClause(CGF, IfCond, ThenGen, 6312 [](CodeGenFunction &, PrePostActionTy &) {}); 6313 } else { 6314 RegionCodeGenTy ThenRCG(ThenGen); 6315 ThenRCG(CGF); 6316 } 6317 } 6318 } 6319 6320 namespace { 6321 /// Cleanup action for uses_allocators support. 6322 class OMPUsesAllocatorsActionTy final : public PrePostActionTy { 6323 ArrayRef<std::pair<const Expr *, const Expr *>> Allocators; 6324 6325 public: 6326 OMPUsesAllocatorsActionTy( 6327 ArrayRef<std::pair<const Expr *, const Expr *>> Allocators) 6328 : Allocators(Allocators) {} 6329 void Enter(CodeGenFunction &CGF) override { 6330 if (!CGF.HaveInsertPoint()) 6331 return; 6332 for (const auto &AllocatorData : Allocators) { 6333 CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsInit( 6334 CGF, AllocatorData.first, AllocatorData.second); 6335 } 6336 } 6337 void Exit(CodeGenFunction &CGF) override { 6338 if (!CGF.HaveInsertPoint()) 6339 return; 6340 for (const auto &AllocatorData : Allocators) { 6341 CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsFini(CGF, 6342 AllocatorData.first); 6343 } 6344 } 6345 }; 6346 } // namespace 6347 6348 void CGOpenMPRuntime::emitTargetOutlinedFunction( 6349 const OMPExecutableDirective &D, StringRef ParentName, 6350 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6351 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6352 assert(!ParentName.empty() && "Invalid target region parent name!"); 6353 HasEmittedTargetRegion = true; 6354 SmallVector<std::pair<const Expr *, const Expr *>, 4> Allocators; 6355 for (const auto *C : D.getClausesOfKind<OMPUsesAllocatorsClause>()) { 6356 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) { 6357 const OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I); 6358 if (!D.AllocatorTraits) 6359 continue; 6360 Allocators.emplace_back(D.Allocator, D.AllocatorTraits); 6361 } 6362 } 6363 OMPUsesAllocatorsActionTy UsesAllocatorAction(Allocators); 6364 CodeGen.setAction(UsesAllocatorAction); 6365 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID, 6366 IsOffloadEntry, CodeGen); 6367 } 6368 6369 void CGOpenMPRuntime::emitUsesAllocatorsInit(CodeGenFunction &CGF, 6370 const Expr *Allocator, 6371 const Expr *AllocatorTraits) { 6372 llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc()); 6373 ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true); 6374 // Use default memspace handle. 6375 llvm::Value *MemSpaceHandle = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 6376 llvm::Value *NumTraits = llvm::ConstantInt::get( 6377 CGF.IntTy, cast<ConstantArrayType>( 6378 AllocatorTraits->getType()->getAsArrayTypeUnsafe()) 6379 ->getSize() 6380 .getLimitedValue()); 6381 LValue AllocatorTraitsLVal = CGF.EmitLValue(AllocatorTraits); 6382 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6383 AllocatorTraitsLVal.getAddress(CGF), CGF.VoidPtrPtrTy); 6384 AllocatorTraitsLVal = CGF.MakeAddrLValue(Addr, CGF.getContext().VoidPtrTy, 6385 AllocatorTraitsLVal.getBaseInfo(), 6386 AllocatorTraitsLVal.getTBAAInfo()); 6387 llvm::Value *Traits = 6388 CGF.EmitLoadOfScalar(AllocatorTraitsLVal, AllocatorTraits->getExprLoc()); 6389 6390 llvm::Value *AllocatorVal = 6391 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 6392 CGM.getModule(), OMPRTL___kmpc_init_allocator), 6393 {ThreadId, MemSpaceHandle, NumTraits, Traits}); 6394 // Store to allocator. 6395 CGF.EmitVarDecl(*cast<VarDecl>( 6396 cast<DeclRefExpr>(Allocator->IgnoreParenImpCasts())->getDecl())); 6397 LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts()); 6398 AllocatorVal = 6399 CGF.EmitScalarConversion(AllocatorVal, CGF.getContext().VoidPtrTy, 6400 Allocator->getType(), Allocator->getExprLoc()); 6401 CGF.EmitStoreOfScalar(AllocatorVal, AllocatorLVal); 6402 } 6403 6404 void CGOpenMPRuntime::emitUsesAllocatorsFini(CodeGenFunction &CGF, 6405 const Expr *Allocator) { 6406 llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc()); 6407 ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true); 6408 LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts()); 6409 llvm::Value *AllocatorVal = 6410 CGF.EmitLoadOfScalar(AllocatorLVal, Allocator->getExprLoc()); 6411 AllocatorVal = CGF.EmitScalarConversion(AllocatorVal, Allocator->getType(), 6412 CGF.getContext().VoidPtrTy, 6413 Allocator->getExprLoc()); 6414 (void)CGF.EmitRuntimeCall( 6415 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 6416 OMPRTL___kmpc_destroy_allocator), 6417 {ThreadId, AllocatorVal}); 6418 } 6419 6420 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper( 6421 const OMPExecutableDirective &D, StringRef ParentName, 6422 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6423 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6424 // Create a unique name for the entry function using the source location 6425 // information of the current target region. The name will be something like: 6426 // 6427 // __omp_offloading_DD_FFFF_PP_lBB 6428 // 6429 // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the 6430 // mangled name of the function that encloses the target region and BB is the 6431 // line number of the target region. 6432 6433 unsigned DeviceID; 6434 unsigned FileID; 6435 unsigned Line; 6436 getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID, 6437 Line); 6438 SmallString<64> EntryFnName; 6439 { 6440 llvm::raw_svector_ostream OS(EntryFnName); 6441 OS << "__omp_offloading" << llvm::format("_%x", DeviceID) 6442 << llvm::format("_%x_", FileID) << ParentName << "_l" << Line; 6443 } 6444 6445 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6446 6447 CodeGenFunction CGF(CGM, true); 6448 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName); 6449 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6450 6451 OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc()); 6452 6453 // If this target outline function is not an offload entry, we don't need to 6454 // register it. 6455 if (!IsOffloadEntry) 6456 return; 6457 6458 // The target region ID is used by the runtime library to identify the current 6459 // target region, so it only has to be unique and not necessarily point to 6460 // anything. It could be the pointer to the outlined function that implements 6461 // the target region, but we aren't using that so that the compiler doesn't 6462 // need to keep that, and could therefore inline the host function if proven 6463 // worthwhile during optimization. In the other hand, if emitting code for the 6464 // device, the ID has to be the function address so that it can retrieved from 6465 // the offloading entry and launched by the runtime library. We also mark the 6466 // outlined function to have external linkage in case we are emitting code for 6467 // the device, because these functions will be entry points to the device. 6468 6469 if (CGM.getLangOpts().OpenMPIsDevice) { 6470 OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy); 6471 OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 6472 OutlinedFn->setDSOLocal(false); 6473 } else { 6474 std::string Name = getName({EntryFnName, "region_id"}); 6475 OutlinedFnID = new llvm::GlobalVariable( 6476 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6477 llvm::GlobalValue::WeakAnyLinkage, 6478 llvm::Constant::getNullValue(CGM.Int8Ty), Name); 6479 } 6480 6481 // Register the information for the entry associated with this target region. 6482 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 6483 DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID, 6484 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion); 6485 } 6486 6487 /// Checks if the expression is constant or does not have non-trivial function 6488 /// calls. 6489 static bool isTrivial(ASTContext &Ctx, const Expr * E) { 6490 // We can skip constant expressions. 6491 // We can skip expressions with trivial calls or simple expressions. 6492 return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) || 6493 !E->hasNonTrivialCall(Ctx)) && 6494 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true); 6495 } 6496 6497 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx, 6498 const Stmt *Body) { 6499 const Stmt *Child = Body->IgnoreContainers(); 6500 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) { 6501 Child = nullptr; 6502 for (const Stmt *S : C->body()) { 6503 if (const auto *E = dyn_cast<Expr>(S)) { 6504 if (isTrivial(Ctx, E)) 6505 continue; 6506 } 6507 // Some of the statements can be ignored. 6508 if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) || 6509 isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S)) 6510 continue; 6511 // Analyze declarations. 6512 if (const auto *DS = dyn_cast<DeclStmt>(S)) { 6513 if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) { 6514 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) || 6515 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) || 6516 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) || 6517 isa<UsingDirectiveDecl>(D) || 6518 isa<OMPDeclareReductionDecl>(D) || 6519 isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D)) 6520 return true; 6521 const auto *VD = dyn_cast<VarDecl>(D); 6522 if (!VD) 6523 return false; 6524 return VD->isConstexpr() || 6525 ((VD->getType().isTrivialType(Ctx) || 6526 VD->getType()->isReferenceType()) && 6527 (!VD->hasInit() || isTrivial(Ctx, VD->getInit()))); 6528 })) 6529 continue; 6530 } 6531 // Found multiple children - cannot get the one child only. 6532 if (Child) 6533 return nullptr; 6534 Child = S; 6535 } 6536 if (Child) 6537 Child = Child->IgnoreContainers(); 6538 } 6539 return Child; 6540 } 6541 6542 /// Emit the number of teams for a target directive. Inspect the num_teams 6543 /// clause associated with a teams construct combined or closely nested 6544 /// with the target directive. 6545 /// 6546 /// Emit a team of size one for directives such as 'target parallel' that 6547 /// have no associated teams construct. 6548 /// 6549 /// Otherwise, return nullptr. 6550 static llvm::Value * 6551 emitNumTeamsForTargetDirective(CodeGenFunction &CGF, 6552 const OMPExecutableDirective &D) { 6553 assert(!CGF.getLangOpts().OpenMPIsDevice && 6554 "Clauses associated with the teams directive expected to be emitted " 6555 "only for the host!"); 6556 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6557 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6558 "Expected target-based executable directive."); 6559 CGBuilderTy &Bld = CGF.Builder; 6560 switch (DirectiveKind) { 6561 case OMPD_target: { 6562 const auto *CS = D.getInnermostCapturedStmt(); 6563 const auto *Body = 6564 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 6565 const Stmt *ChildStmt = 6566 CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body); 6567 if (const auto *NestedDir = 6568 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 6569 if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) { 6570 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) { 6571 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6572 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6573 const Expr *NumTeams = 6574 NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6575 llvm::Value *NumTeamsVal = 6576 CGF.EmitScalarExpr(NumTeams, 6577 /*IgnoreResultAssign*/ true); 6578 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6579 /*isSigned=*/true); 6580 } 6581 return Bld.getInt32(0); 6582 } 6583 if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) || 6584 isOpenMPSimdDirective(NestedDir->getDirectiveKind())) 6585 return Bld.getInt32(1); 6586 return Bld.getInt32(0); 6587 } 6588 return nullptr; 6589 } 6590 case OMPD_target_teams: 6591 case OMPD_target_teams_distribute: 6592 case OMPD_target_teams_distribute_simd: 6593 case OMPD_target_teams_distribute_parallel_for: 6594 case OMPD_target_teams_distribute_parallel_for_simd: { 6595 if (D.hasClausesOfKind<OMPNumTeamsClause>()) { 6596 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF); 6597 const Expr *NumTeams = 6598 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6599 llvm::Value *NumTeamsVal = 6600 CGF.EmitScalarExpr(NumTeams, 6601 /*IgnoreResultAssign*/ true); 6602 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6603 /*isSigned=*/true); 6604 } 6605 return Bld.getInt32(0); 6606 } 6607 case OMPD_target_parallel: 6608 case OMPD_target_parallel_for: 6609 case OMPD_target_parallel_for_simd: 6610 case OMPD_target_simd: 6611 return Bld.getInt32(1); 6612 case OMPD_parallel: 6613 case OMPD_for: 6614 case OMPD_parallel_for: 6615 case OMPD_parallel_master: 6616 case OMPD_parallel_sections: 6617 case OMPD_for_simd: 6618 case OMPD_parallel_for_simd: 6619 case OMPD_cancel: 6620 case OMPD_cancellation_point: 6621 case OMPD_ordered: 6622 case OMPD_threadprivate: 6623 case OMPD_allocate: 6624 case OMPD_task: 6625 case OMPD_simd: 6626 case OMPD_sections: 6627 case OMPD_section: 6628 case OMPD_single: 6629 case OMPD_master: 6630 case OMPD_critical: 6631 case OMPD_taskyield: 6632 case OMPD_barrier: 6633 case OMPD_taskwait: 6634 case OMPD_taskgroup: 6635 case OMPD_atomic: 6636 case OMPD_flush: 6637 case OMPD_depobj: 6638 case OMPD_scan: 6639 case OMPD_teams: 6640 case OMPD_target_data: 6641 case OMPD_target_exit_data: 6642 case OMPD_target_enter_data: 6643 case OMPD_distribute: 6644 case OMPD_distribute_simd: 6645 case OMPD_distribute_parallel_for: 6646 case OMPD_distribute_parallel_for_simd: 6647 case OMPD_teams_distribute: 6648 case OMPD_teams_distribute_simd: 6649 case OMPD_teams_distribute_parallel_for: 6650 case OMPD_teams_distribute_parallel_for_simd: 6651 case OMPD_target_update: 6652 case OMPD_declare_simd: 6653 case OMPD_declare_variant: 6654 case OMPD_begin_declare_variant: 6655 case OMPD_end_declare_variant: 6656 case OMPD_declare_target: 6657 case OMPD_end_declare_target: 6658 case OMPD_declare_reduction: 6659 case OMPD_declare_mapper: 6660 case OMPD_taskloop: 6661 case OMPD_taskloop_simd: 6662 case OMPD_master_taskloop: 6663 case OMPD_master_taskloop_simd: 6664 case OMPD_parallel_master_taskloop: 6665 case OMPD_parallel_master_taskloop_simd: 6666 case OMPD_requires: 6667 case OMPD_unknown: 6668 break; 6669 default: 6670 break; 6671 } 6672 llvm_unreachable("Unexpected directive kind."); 6673 } 6674 6675 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS, 6676 llvm::Value *DefaultThreadLimitVal) { 6677 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6678 CGF.getContext(), CS->getCapturedStmt()); 6679 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6680 if (isOpenMPParallelDirective(Dir->getDirectiveKind())) { 6681 llvm::Value *NumThreads = nullptr; 6682 llvm::Value *CondVal = nullptr; 6683 // Handle if clause. If if clause present, the number of threads is 6684 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 6685 if (Dir->hasClausesOfKind<OMPIfClause>()) { 6686 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6687 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6688 const OMPIfClause *IfClause = nullptr; 6689 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) { 6690 if (C->getNameModifier() == OMPD_unknown || 6691 C->getNameModifier() == OMPD_parallel) { 6692 IfClause = C; 6693 break; 6694 } 6695 } 6696 if (IfClause) { 6697 const Expr *Cond = IfClause->getCondition(); 6698 bool Result; 6699 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 6700 if (!Result) 6701 return CGF.Builder.getInt32(1); 6702 } else { 6703 CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange()); 6704 if (const auto *PreInit = 6705 cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) { 6706 for (const auto *I : PreInit->decls()) { 6707 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6708 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6709 } else { 6710 CodeGenFunction::AutoVarEmission Emission = 6711 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6712 CGF.EmitAutoVarCleanups(Emission); 6713 } 6714 } 6715 } 6716 CondVal = CGF.EvaluateExprAsBool(Cond); 6717 } 6718 } 6719 } 6720 // Check the value of num_threads clause iff if clause was not specified 6721 // or is not evaluated to false. 6722 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) { 6723 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6724 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6725 const auto *NumThreadsClause = 6726 Dir->getSingleClause<OMPNumThreadsClause>(); 6727 CodeGenFunction::LexicalScope Scope( 6728 CGF, NumThreadsClause->getNumThreads()->getSourceRange()); 6729 if (const auto *PreInit = 6730 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) { 6731 for (const auto *I : PreInit->decls()) { 6732 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6733 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6734 } else { 6735 CodeGenFunction::AutoVarEmission Emission = 6736 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6737 CGF.EmitAutoVarCleanups(Emission); 6738 } 6739 } 6740 } 6741 NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads()); 6742 NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, 6743 /*isSigned=*/false); 6744 if (DefaultThreadLimitVal) 6745 NumThreads = CGF.Builder.CreateSelect( 6746 CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads), 6747 DefaultThreadLimitVal, NumThreads); 6748 } else { 6749 NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal 6750 : CGF.Builder.getInt32(0); 6751 } 6752 // Process condition of the if clause. 6753 if (CondVal) { 6754 NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads, 6755 CGF.Builder.getInt32(1)); 6756 } 6757 return NumThreads; 6758 } 6759 if (isOpenMPSimdDirective(Dir->getDirectiveKind())) 6760 return CGF.Builder.getInt32(1); 6761 return DefaultThreadLimitVal; 6762 } 6763 return DefaultThreadLimitVal ? DefaultThreadLimitVal 6764 : CGF.Builder.getInt32(0); 6765 } 6766 6767 /// Emit the number of threads for a target directive. Inspect the 6768 /// thread_limit clause associated with a teams construct combined or closely 6769 /// nested with the target directive. 6770 /// 6771 /// Emit the num_threads clause for directives such as 'target parallel' that 6772 /// have no associated teams construct. 6773 /// 6774 /// Otherwise, return nullptr. 6775 static llvm::Value * 6776 emitNumThreadsForTargetDirective(CodeGenFunction &CGF, 6777 const OMPExecutableDirective &D) { 6778 assert(!CGF.getLangOpts().OpenMPIsDevice && 6779 "Clauses associated with the teams directive expected to be emitted " 6780 "only for the host!"); 6781 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6782 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6783 "Expected target-based executable directive."); 6784 CGBuilderTy &Bld = CGF.Builder; 6785 llvm::Value *ThreadLimitVal = nullptr; 6786 llvm::Value *NumThreadsVal = nullptr; 6787 switch (DirectiveKind) { 6788 case OMPD_target: { 6789 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 6790 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6791 return NumThreads; 6792 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6793 CGF.getContext(), CS->getCapturedStmt()); 6794 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6795 if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) { 6796 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6797 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6798 const auto *ThreadLimitClause = 6799 Dir->getSingleClause<OMPThreadLimitClause>(); 6800 CodeGenFunction::LexicalScope Scope( 6801 CGF, ThreadLimitClause->getThreadLimit()->getSourceRange()); 6802 if (const auto *PreInit = 6803 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) { 6804 for (const auto *I : PreInit->decls()) { 6805 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6806 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6807 } else { 6808 CodeGenFunction::AutoVarEmission Emission = 6809 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6810 CGF.EmitAutoVarCleanups(Emission); 6811 } 6812 } 6813 } 6814 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6815 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6816 ThreadLimitVal = 6817 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6818 } 6819 if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) && 6820 !isOpenMPDistributeDirective(Dir->getDirectiveKind())) { 6821 CS = Dir->getInnermostCapturedStmt(); 6822 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6823 CGF.getContext(), CS->getCapturedStmt()); 6824 Dir = dyn_cast_or_null<OMPExecutableDirective>(Child); 6825 } 6826 if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) && 6827 !isOpenMPSimdDirective(Dir->getDirectiveKind())) { 6828 CS = Dir->getInnermostCapturedStmt(); 6829 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6830 return NumThreads; 6831 } 6832 if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind())) 6833 return Bld.getInt32(1); 6834 } 6835 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 6836 } 6837 case OMPD_target_teams: { 6838 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6839 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 6840 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6841 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6842 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6843 ThreadLimitVal = 6844 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6845 } 6846 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 6847 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6848 return NumThreads; 6849 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6850 CGF.getContext(), CS->getCapturedStmt()); 6851 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6852 if (Dir->getDirectiveKind() == OMPD_distribute) { 6853 CS = Dir->getInnermostCapturedStmt(); 6854 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6855 return NumThreads; 6856 } 6857 } 6858 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 6859 } 6860 case OMPD_target_teams_distribute: 6861 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6862 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 6863 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6864 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6865 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6866 ThreadLimitVal = 6867 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6868 } 6869 return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal); 6870 case OMPD_target_parallel: 6871 case OMPD_target_parallel_for: 6872 case OMPD_target_parallel_for_simd: 6873 case OMPD_target_teams_distribute_parallel_for: 6874 case OMPD_target_teams_distribute_parallel_for_simd: { 6875 llvm::Value *CondVal = nullptr; 6876 // Handle if clause. If if clause present, the number of threads is 6877 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 6878 if (D.hasClausesOfKind<OMPIfClause>()) { 6879 const OMPIfClause *IfClause = nullptr; 6880 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) { 6881 if (C->getNameModifier() == OMPD_unknown || 6882 C->getNameModifier() == OMPD_parallel) { 6883 IfClause = C; 6884 break; 6885 } 6886 } 6887 if (IfClause) { 6888 const Expr *Cond = IfClause->getCondition(); 6889 bool Result; 6890 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 6891 if (!Result) 6892 return Bld.getInt32(1); 6893 } else { 6894 CodeGenFunction::RunCleanupsScope Scope(CGF); 6895 CondVal = CGF.EvaluateExprAsBool(Cond); 6896 } 6897 } 6898 } 6899 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6900 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 6901 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6902 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6903 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6904 ThreadLimitVal = 6905 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6906 } 6907 if (D.hasClausesOfKind<OMPNumThreadsClause>()) { 6908 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 6909 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>(); 6910 llvm::Value *NumThreads = CGF.EmitScalarExpr( 6911 NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true); 6912 NumThreadsVal = 6913 Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false); 6914 ThreadLimitVal = ThreadLimitVal 6915 ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal, 6916 ThreadLimitVal), 6917 NumThreadsVal, ThreadLimitVal) 6918 : NumThreadsVal; 6919 } 6920 if (!ThreadLimitVal) 6921 ThreadLimitVal = Bld.getInt32(0); 6922 if (CondVal) 6923 return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1)); 6924 return ThreadLimitVal; 6925 } 6926 case OMPD_target_teams_distribute_simd: 6927 case OMPD_target_simd: 6928 return Bld.getInt32(1); 6929 case OMPD_parallel: 6930 case OMPD_for: 6931 case OMPD_parallel_for: 6932 case OMPD_parallel_master: 6933 case OMPD_parallel_sections: 6934 case OMPD_for_simd: 6935 case OMPD_parallel_for_simd: 6936 case OMPD_cancel: 6937 case OMPD_cancellation_point: 6938 case OMPD_ordered: 6939 case OMPD_threadprivate: 6940 case OMPD_allocate: 6941 case OMPD_task: 6942 case OMPD_simd: 6943 case OMPD_sections: 6944 case OMPD_section: 6945 case OMPD_single: 6946 case OMPD_master: 6947 case OMPD_critical: 6948 case OMPD_taskyield: 6949 case OMPD_barrier: 6950 case OMPD_taskwait: 6951 case OMPD_taskgroup: 6952 case OMPD_atomic: 6953 case OMPD_flush: 6954 case OMPD_depobj: 6955 case OMPD_scan: 6956 case OMPD_teams: 6957 case OMPD_target_data: 6958 case OMPD_target_exit_data: 6959 case OMPD_target_enter_data: 6960 case OMPD_distribute: 6961 case OMPD_distribute_simd: 6962 case OMPD_distribute_parallel_for: 6963 case OMPD_distribute_parallel_for_simd: 6964 case OMPD_teams_distribute: 6965 case OMPD_teams_distribute_simd: 6966 case OMPD_teams_distribute_parallel_for: 6967 case OMPD_teams_distribute_parallel_for_simd: 6968 case OMPD_target_update: 6969 case OMPD_declare_simd: 6970 case OMPD_declare_variant: 6971 case OMPD_begin_declare_variant: 6972 case OMPD_end_declare_variant: 6973 case OMPD_declare_target: 6974 case OMPD_end_declare_target: 6975 case OMPD_declare_reduction: 6976 case OMPD_declare_mapper: 6977 case OMPD_taskloop: 6978 case OMPD_taskloop_simd: 6979 case OMPD_master_taskloop: 6980 case OMPD_master_taskloop_simd: 6981 case OMPD_parallel_master_taskloop: 6982 case OMPD_parallel_master_taskloop_simd: 6983 case OMPD_requires: 6984 case OMPD_unknown: 6985 break; 6986 default: 6987 break; 6988 } 6989 llvm_unreachable("Unsupported directive kind."); 6990 } 6991 6992 namespace { 6993 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 6994 6995 // Utility to handle information from clauses associated with a given 6996 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause). 6997 // It provides a convenient interface to obtain the information and generate 6998 // code for that information. 6999 class MappableExprsHandler { 7000 public: 7001 /// Values for bit flags used to specify the mapping type for 7002 /// offloading. 7003 enum OpenMPOffloadMappingFlags : uint64_t { 7004 /// No flags 7005 OMP_MAP_NONE = 0x0, 7006 /// Allocate memory on the device and move data from host to device. 7007 OMP_MAP_TO = 0x01, 7008 /// Allocate memory on the device and move data from device to host. 7009 OMP_MAP_FROM = 0x02, 7010 /// Always perform the requested mapping action on the element, even 7011 /// if it was already mapped before. 7012 OMP_MAP_ALWAYS = 0x04, 7013 /// Delete the element from the device environment, ignoring the 7014 /// current reference count associated with the element. 7015 OMP_MAP_DELETE = 0x08, 7016 /// The element being mapped is a pointer-pointee pair; both the 7017 /// pointer and the pointee should be mapped. 7018 OMP_MAP_PTR_AND_OBJ = 0x10, 7019 /// This flags signals that the base address of an entry should be 7020 /// passed to the target kernel as an argument. 7021 OMP_MAP_TARGET_PARAM = 0x20, 7022 /// Signal that the runtime library has to return the device pointer 7023 /// in the current position for the data being mapped. Used when we have the 7024 /// use_device_ptr or use_device_addr clause. 7025 OMP_MAP_RETURN_PARAM = 0x40, 7026 /// This flag signals that the reference being passed is a pointer to 7027 /// private data. 7028 OMP_MAP_PRIVATE = 0x80, 7029 /// Pass the element to the device by value. 7030 OMP_MAP_LITERAL = 0x100, 7031 /// Implicit map 7032 OMP_MAP_IMPLICIT = 0x200, 7033 /// Close is a hint to the runtime to allocate memory close to 7034 /// the target device. 7035 OMP_MAP_CLOSE = 0x400, 7036 /// 0x800 is reserved for compatibility with XLC. 7037 /// Produce a runtime error if the data is not already allocated. 7038 OMP_MAP_PRESENT = 0x1000, 7039 /// Signal that the runtime library should use args as an array of 7040 /// descriptor_dim pointers and use args_size as dims. Used when we have 7041 /// non-contiguous list items in target update directive 7042 OMP_MAP_NON_CONTIG = 0x100000000000, 7043 /// The 16 MSBs of the flags indicate whether the entry is member of some 7044 /// struct/class. 7045 OMP_MAP_MEMBER_OF = 0xffff000000000000, 7046 LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF), 7047 }; 7048 7049 /// Get the offset of the OMP_MAP_MEMBER_OF field. 7050 static unsigned getFlagMemberOffset() { 7051 unsigned Offset = 0; 7052 for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1); 7053 Remain = Remain >> 1) 7054 Offset++; 7055 return Offset; 7056 } 7057 7058 /// Class that associates information with a base pointer to be passed to the 7059 /// runtime library. 7060 class BasePointerInfo { 7061 /// The base pointer. 7062 llvm::Value *Ptr = nullptr; 7063 /// The base declaration that refers to this device pointer, or null if 7064 /// there is none. 7065 const ValueDecl *DevPtrDecl = nullptr; 7066 7067 public: 7068 BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr) 7069 : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {} 7070 llvm::Value *operator*() const { return Ptr; } 7071 const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; } 7072 void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; } 7073 }; 7074 7075 using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>; 7076 using MapValuesArrayTy = SmallVector<llvm::Value *, 4>; 7077 using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>; 7078 using MapMappersArrayTy = SmallVector<const ValueDecl *, 4>; 7079 using MapDimArrayTy = SmallVector<uint64_t, 4>; 7080 using MapNonContiguousArrayTy = SmallVector<MapValuesArrayTy, 4>; 7081 7082 /// This structure contains combined information generated for mappable 7083 /// clauses, including base pointers, pointers, sizes, map types, user-defined 7084 /// mappers, and non-contiguous information. 7085 struct MapCombinedInfoTy { 7086 struct StructNonContiguousInfo { 7087 bool IsNonContiguous = false; 7088 MapDimArrayTy Dims; 7089 MapNonContiguousArrayTy Offsets; 7090 MapNonContiguousArrayTy Counts; 7091 MapNonContiguousArrayTy Strides; 7092 }; 7093 MapBaseValuesArrayTy BasePointers; 7094 MapValuesArrayTy Pointers; 7095 MapValuesArrayTy Sizes; 7096 MapFlagsArrayTy Types; 7097 MapMappersArrayTy Mappers; 7098 StructNonContiguousInfo NonContigInfo; 7099 7100 /// Append arrays in \a CurInfo. 7101 void append(MapCombinedInfoTy &CurInfo) { 7102 BasePointers.append(CurInfo.BasePointers.begin(), 7103 CurInfo.BasePointers.end()); 7104 Pointers.append(CurInfo.Pointers.begin(), CurInfo.Pointers.end()); 7105 Sizes.append(CurInfo.Sizes.begin(), CurInfo.Sizes.end()); 7106 Types.append(CurInfo.Types.begin(), CurInfo.Types.end()); 7107 Mappers.append(CurInfo.Mappers.begin(), CurInfo.Mappers.end()); 7108 NonContigInfo.Dims.append(CurInfo.NonContigInfo.Dims.begin(), 7109 CurInfo.NonContigInfo.Dims.end()); 7110 NonContigInfo.Offsets.append(CurInfo.NonContigInfo.Offsets.begin(), 7111 CurInfo.NonContigInfo.Offsets.end()); 7112 NonContigInfo.Counts.append(CurInfo.NonContigInfo.Counts.begin(), 7113 CurInfo.NonContigInfo.Counts.end()); 7114 NonContigInfo.Strides.append(CurInfo.NonContigInfo.Strides.begin(), 7115 CurInfo.NonContigInfo.Strides.end()); 7116 } 7117 }; 7118 7119 /// Map between a struct and the its lowest & highest elements which have been 7120 /// mapped. 7121 /// [ValueDecl *] --> {LE(FieldIndex, Pointer), 7122 /// HE(FieldIndex, Pointer)} 7123 struct StructRangeInfoTy { 7124 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = { 7125 0, Address::invalid()}; 7126 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = { 7127 0, Address::invalid()}; 7128 Address Base = Address::invalid(); 7129 bool IsArraySection = false; 7130 }; 7131 7132 private: 7133 /// Kind that defines how a device pointer has to be returned. 7134 struct MapInfo { 7135 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7136 OpenMPMapClauseKind MapType = OMPC_MAP_unknown; 7137 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7138 ArrayRef<OpenMPMotionModifierKind> MotionModifiers; 7139 bool ReturnDevicePointer = false; 7140 bool IsImplicit = false; 7141 const ValueDecl *Mapper = nullptr; 7142 bool ForDeviceAddr = false; 7143 7144 MapInfo() = default; 7145 MapInfo( 7146 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7147 OpenMPMapClauseKind MapType, 7148 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7149 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, 7150 bool ReturnDevicePointer, bool IsImplicit, 7151 const ValueDecl *Mapper = nullptr, bool ForDeviceAddr = false) 7152 : Components(Components), MapType(MapType), MapModifiers(MapModifiers), 7153 MotionModifiers(MotionModifiers), 7154 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit), 7155 Mapper(Mapper), ForDeviceAddr(ForDeviceAddr) {} 7156 }; 7157 7158 /// If use_device_ptr or use_device_addr is used on a decl which is a struct 7159 /// member and there is no map information about it, then emission of that 7160 /// entry is deferred until the whole struct has been processed. 7161 struct DeferredDevicePtrEntryTy { 7162 const Expr *IE = nullptr; 7163 const ValueDecl *VD = nullptr; 7164 bool ForDeviceAddr = false; 7165 7166 DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD, 7167 bool ForDeviceAddr) 7168 : IE(IE), VD(VD), ForDeviceAddr(ForDeviceAddr) {} 7169 }; 7170 7171 /// The target directive from where the mappable clauses were extracted. It 7172 /// is either a executable directive or a user-defined mapper directive. 7173 llvm::PointerUnion<const OMPExecutableDirective *, 7174 const OMPDeclareMapperDecl *> 7175 CurDir; 7176 7177 /// Function the directive is being generated for. 7178 CodeGenFunction &CGF; 7179 7180 /// Set of all first private variables in the current directive. 7181 /// bool data is set to true if the variable is implicitly marked as 7182 /// firstprivate, false otherwise. 7183 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls; 7184 7185 /// Map between device pointer declarations and their expression components. 7186 /// The key value for declarations in 'this' is null. 7187 llvm::DenseMap< 7188 const ValueDecl *, 7189 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>> 7190 DevPointersMap; 7191 7192 llvm::Value *getExprTypeSize(const Expr *E) const { 7193 QualType ExprTy = E->getType().getCanonicalType(); 7194 7195 // Calculate the size for array shaping expression. 7196 if (const auto *OAE = dyn_cast<OMPArrayShapingExpr>(E)) { 7197 llvm::Value *Size = 7198 CGF.getTypeSize(OAE->getBase()->getType()->getPointeeType()); 7199 for (const Expr *SE : OAE->getDimensions()) { 7200 llvm::Value *Sz = CGF.EmitScalarExpr(SE); 7201 Sz = CGF.EmitScalarConversion(Sz, SE->getType(), 7202 CGF.getContext().getSizeType(), 7203 SE->getExprLoc()); 7204 Size = CGF.Builder.CreateNUWMul(Size, Sz); 7205 } 7206 return Size; 7207 } 7208 7209 // Reference types are ignored for mapping purposes. 7210 if (const auto *RefTy = ExprTy->getAs<ReferenceType>()) 7211 ExprTy = RefTy->getPointeeType().getCanonicalType(); 7212 7213 // Given that an array section is considered a built-in type, we need to 7214 // do the calculation based on the length of the section instead of relying 7215 // on CGF.getTypeSize(E->getType()). 7216 if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) { 7217 QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType( 7218 OAE->getBase()->IgnoreParenImpCasts()) 7219 .getCanonicalType(); 7220 7221 // If there is no length associated with the expression and lower bound is 7222 // not specified too, that means we are using the whole length of the 7223 // base. 7224 if (!OAE->getLength() && OAE->getColonLocFirst().isValid() && 7225 !OAE->getLowerBound()) 7226 return CGF.getTypeSize(BaseTy); 7227 7228 llvm::Value *ElemSize; 7229 if (const auto *PTy = BaseTy->getAs<PointerType>()) { 7230 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType()); 7231 } else { 7232 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr()); 7233 assert(ATy && "Expecting array type if not a pointer type."); 7234 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType()); 7235 } 7236 7237 // If we don't have a length at this point, that is because we have an 7238 // array section with a single element. 7239 if (!OAE->getLength() && OAE->getColonLocFirst().isInvalid()) 7240 return ElemSize; 7241 7242 if (const Expr *LenExpr = OAE->getLength()) { 7243 llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr); 7244 LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(), 7245 CGF.getContext().getSizeType(), 7246 LenExpr->getExprLoc()); 7247 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize); 7248 } 7249 assert(!OAE->getLength() && OAE->getColonLocFirst().isValid() && 7250 OAE->getLowerBound() && "expected array_section[lb:]."); 7251 // Size = sizetype - lb * elemtype; 7252 llvm::Value *LengthVal = CGF.getTypeSize(BaseTy); 7253 llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound()); 7254 LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(), 7255 CGF.getContext().getSizeType(), 7256 OAE->getLowerBound()->getExprLoc()); 7257 LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize); 7258 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal); 7259 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal); 7260 LengthVal = CGF.Builder.CreateSelect( 7261 Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0)); 7262 return LengthVal; 7263 } 7264 return CGF.getTypeSize(ExprTy); 7265 } 7266 7267 /// Return the corresponding bits for a given map clause modifier. Add 7268 /// a flag marking the map as a pointer if requested. Add a flag marking the 7269 /// map as the first one of a series of maps that relate to the same map 7270 /// expression. 7271 OpenMPOffloadMappingFlags getMapTypeBits( 7272 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7273 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, bool IsImplicit, 7274 bool AddPtrFlag, bool AddIsTargetParamFlag, bool IsNonContiguous) const { 7275 OpenMPOffloadMappingFlags Bits = 7276 IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE; 7277 switch (MapType) { 7278 case OMPC_MAP_alloc: 7279 case OMPC_MAP_release: 7280 // alloc and release is the default behavior in the runtime library, i.e. 7281 // if we don't pass any bits alloc/release that is what the runtime is 7282 // going to do. Therefore, we don't need to signal anything for these two 7283 // type modifiers. 7284 break; 7285 case OMPC_MAP_to: 7286 Bits |= OMP_MAP_TO; 7287 break; 7288 case OMPC_MAP_from: 7289 Bits |= OMP_MAP_FROM; 7290 break; 7291 case OMPC_MAP_tofrom: 7292 Bits |= OMP_MAP_TO | OMP_MAP_FROM; 7293 break; 7294 case OMPC_MAP_delete: 7295 Bits |= OMP_MAP_DELETE; 7296 break; 7297 case OMPC_MAP_unknown: 7298 llvm_unreachable("Unexpected map type!"); 7299 } 7300 if (AddPtrFlag) 7301 Bits |= OMP_MAP_PTR_AND_OBJ; 7302 if (AddIsTargetParamFlag) 7303 Bits |= OMP_MAP_TARGET_PARAM; 7304 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always) 7305 != MapModifiers.end()) 7306 Bits |= OMP_MAP_ALWAYS; 7307 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close) 7308 != MapModifiers.end()) 7309 Bits |= OMP_MAP_CLOSE; 7310 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_present) 7311 != MapModifiers.end()) 7312 Bits |= OMP_MAP_PRESENT; 7313 if (llvm::find(MotionModifiers, OMPC_MOTION_MODIFIER_present) 7314 != MotionModifiers.end()) 7315 Bits |= OMP_MAP_PRESENT; 7316 if (IsNonContiguous) 7317 Bits |= OMP_MAP_NON_CONTIG; 7318 return Bits; 7319 } 7320 7321 /// Return true if the provided expression is a final array section. A 7322 /// final array section, is one whose length can't be proved to be one. 7323 bool isFinalArraySectionExpression(const Expr *E) const { 7324 const auto *OASE = dyn_cast<OMPArraySectionExpr>(E); 7325 7326 // It is not an array section and therefore not a unity-size one. 7327 if (!OASE) 7328 return false; 7329 7330 // An array section with no colon always refer to a single element. 7331 if (OASE->getColonLocFirst().isInvalid()) 7332 return false; 7333 7334 const Expr *Length = OASE->getLength(); 7335 7336 // If we don't have a length we have to check if the array has size 1 7337 // for this dimension. Also, we should always expect a length if the 7338 // base type is pointer. 7339 if (!Length) { 7340 QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType( 7341 OASE->getBase()->IgnoreParenImpCasts()) 7342 .getCanonicalType(); 7343 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr())) 7344 return ATy->getSize().getSExtValue() != 1; 7345 // If we don't have a constant dimension length, we have to consider 7346 // the current section as having any size, so it is not necessarily 7347 // unitary. If it happen to be unity size, that's user fault. 7348 return true; 7349 } 7350 7351 // Check if the length evaluates to 1. 7352 Expr::EvalResult Result; 7353 if (!Length->EvaluateAsInt(Result, CGF.getContext())) 7354 return true; // Can have more that size 1. 7355 7356 llvm::APSInt ConstLength = Result.Val.getInt(); 7357 return ConstLength.getSExtValue() != 1; 7358 } 7359 7360 /// Generate the base pointers, section pointers, sizes, map type bits, and 7361 /// user-defined mappers (all included in \a CombinedInfo) for the provided 7362 /// map type, map or motion modifiers, and expression components. 7363 /// \a IsFirstComponent should be set to true if the provided set of 7364 /// components is the first associated with a capture. 7365 void generateInfoForComponentList( 7366 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7367 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, 7368 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7369 MapCombinedInfoTy &CombinedInfo, StructRangeInfoTy &PartialStruct, 7370 bool IsFirstComponentList, bool IsImplicit, 7371 const ValueDecl *Mapper = nullptr, bool ForDeviceAddr = false, 7372 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 7373 OverlappedElements = llvm::None) const { 7374 // The following summarizes what has to be generated for each map and the 7375 // types below. The generated information is expressed in this order: 7376 // base pointer, section pointer, size, flags 7377 // (to add to the ones that come from the map type and modifier). 7378 // 7379 // double d; 7380 // int i[100]; 7381 // float *p; 7382 // 7383 // struct S1 { 7384 // int i; 7385 // float f[50]; 7386 // } 7387 // struct S2 { 7388 // int i; 7389 // float f[50]; 7390 // S1 s; 7391 // double *p; 7392 // struct S2 *ps; 7393 // } 7394 // S2 s; 7395 // S2 *ps; 7396 // 7397 // map(d) 7398 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM 7399 // 7400 // map(i) 7401 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM 7402 // 7403 // map(i[1:23]) 7404 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM 7405 // 7406 // map(p) 7407 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM 7408 // 7409 // map(p[1:24]) 7410 // &p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM | PTR_AND_OBJ 7411 // in unified shared memory mode or for local pointers 7412 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM 7413 // 7414 // map(s) 7415 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM 7416 // 7417 // map(s.i) 7418 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM 7419 // 7420 // map(s.s.f) 7421 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7422 // 7423 // map(s.p) 7424 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM 7425 // 7426 // map(to: s.p[:22]) 7427 // &s, &(s.p), sizeof(double*), TARGET_PARAM (*) 7428 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**) 7429 // &(s.p), &(s.p[0]), 22*sizeof(double), 7430 // MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 7431 // (*) alloc space for struct members, only this is a target parameter 7432 // (**) map the pointer (nothing to be mapped in this example) (the compiler 7433 // optimizes this entry out, same in the examples below) 7434 // (***) map the pointee (map: to) 7435 // 7436 // map(s.ps) 7437 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7438 // 7439 // map(from: s.ps->s.i) 7440 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7441 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7442 // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7443 // 7444 // map(to: s.ps->ps) 7445 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7446 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7447 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | TO 7448 // 7449 // map(s.ps->ps->ps) 7450 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7451 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7452 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7453 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7454 // 7455 // map(to: s.ps->ps->s.f[:22]) 7456 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7457 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7458 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7459 // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7460 // 7461 // map(ps) 7462 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM 7463 // 7464 // map(ps->i) 7465 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM 7466 // 7467 // map(ps->s.f) 7468 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7469 // 7470 // map(from: ps->p) 7471 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM 7472 // 7473 // map(to: ps->p[:22]) 7474 // ps, &(ps->p), sizeof(double*), TARGET_PARAM 7475 // ps, &(ps->p), sizeof(double*), MEMBER_OF(1) 7476 // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO 7477 // 7478 // map(ps->ps) 7479 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7480 // 7481 // map(from: ps->ps->s.i) 7482 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7483 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7484 // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7485 // 7486 // map(from: ps->ps->ps) 7487 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7488 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7489 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7490 // 7491 // map(ps->ps->ps->ps) 7492 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7493 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7494 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7495 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7496 // 7497 // map(to: ps->ps->ps->s.f[:22]) 7498 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7499 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7500 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7501 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7502 // 7503 // map(to: s.f[:22]) map(from: s.p[:33]) 7504 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) + 7505 // sizeof(double*) (**), TARGET_PARAM 7506 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO 7507 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) 7508 // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7509 // (*) allocate contiguous space needed to fit all mapped members even if 7510 // we allocate space for members not mapped (in this example, 7511 // s.f[22..49] and s.s are not mapped, yet we must allocate space for 7512 // them as well because they fall between &s.f[0] and &s.p) 7513 // 7514 // map(from: s.f[:22]) map(to: ps->p[:33]) 7515 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM 7516 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7517 // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*) 7518 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO 7519 // (*) the struct this entry pertains to is the 2nd element in the list of 7520 // arguments, hence MEMBER_OF(2) 7521 // 7522 // map(from: s.f[:22], s.s) map(to: ps->p[:33]) 7523 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM 7524 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM 7525 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM 7526 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7527 // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*) 7528 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO 7529 // (*) the struct this entry pertains to is the 4th element in the list 7530 // of arguments, hence MEMBER_OF(4) 7531 7532 // Track if the map information being generated is the first for a capture. 7533 bool IsCaptureFirstInfo = IsFirstComponentList; 7534 // When the variable is on a declare target link or in a to clause with 7535 // unified memory, a reference is needed to hold the host/device address 7536 // of the variable. 7537 bool RequiresReference = false; 7538 7539 // Scan the components from the base to the complete expression. 7540 auto CI = Components.rbegin(); 7541 auto CE = Components.rend(); 7542 auto I = CI; 7543 7544 // Track if the map information being generated is the first for a list of 7545 // components. 7546 bool IsExpressionFirstInfo = true; 7547 bool FirstPointerInComplexData = false; 7548 Address BP = Address::invalid(); 7549 const Expr *AssocExpr = I->getAssociatedExpression(); 7550 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr); 7551 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7552 const auto *OAShE = dyn_cast<OMPArrayShapingExpr>(AssocExpr); 7553 7554 if (isa<MemberExpr>(AssocExpr)) { 7555 // The base is the 'this' pointer. The content of the pointer is going 7556 // to be the base of the field being mapped. 7557 BP = CGF.LoadCXXThisAddress(); 7558 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) || 7559 (OASE && 7560 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) { 7561 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7562 } else if (OAShE && 7563 isa<CXXThisExpr>(OAShE->getBase()->IgnoreParenCasts())) { 7564 BP = Address( 7565 CGF.EmitScalarExpr(OAShE->getBase()), 7566 CGF.getContext().getTypeAlignInChars(OAShE->getBase()->getType())); 7567 } else { 7568 // The base is the reference to the variable. 7569 // BP = &Var. 7570 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7571 if (const auto *VD = 7572 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) { 7573 if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7574 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) { 7575 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 7576 (*Res == OMPDeclareTargetDeclAttr::MT_To && 7577 CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) { 7578 RequiresReference = true; 7579 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 7580 } 7581 } 7582 } 7583 7584 // If the variable is a pointer and is being dereferenced (i.e. is not 7585 // the last component), the base has to be the pointer itself, not its 7586 // reference. References are ignored for mapping purposes. 7587 QualType Ty = 7588 I->getAssociatedDeclaration()->getType().getNonReferenceType(); 7589 if (Ty->isAnyPointerType() && std::next(I) != CE) { 7590 // No need to generate individual map information for the pointer, it 7591 // can be associated with the combined storage if shared memory mode is 7592 // active or the base declaration is not global variable. 7593 const auto *VD = dyn_cast<VarDecl>(I->getAssociatedDeclaration()); 7594 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 7595 !VD || VD->hasLocalStorage()) 7596 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7597 else 7598 FirstPointerInComplexData = true; 7599 ++I; 7600 } 7601 } 7602 7603 // Track whether a component of the list should be marked as MEMBER_OF some 7604 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry 7605 // in a component list should be marked as MEMBER_OF, all subsequent entries 7606 // do not belong to the base struct. E.g. 7607 // struct S2 s; 7608 // s.ps->ps->ps->f[:] 7609 // (1) (2) (3) (4) 7610 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a 7611 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3) 7612 // is the pointee of ps(2) which is not member of struct s, so it should not 7613 // be marked as such (it is still PTR_AND_OBJ). 7614 // The variable is initialized to false so that PTR_AND_OBJ entries which 7615 // are not struct members are not considered (e.g. array of pointers to 7616 // data). 7617 bool ShouldBeMemberOf = false; 7618 7619 // Variable keeping track of whether or not we have encountered a component 7620 // in the component list which is a member expression. Useful when we have a 7621 // pointer or a final array section, in which case it is the previous 7622 // component in the list which tells us whether we have a member expression. 7623 // E.g. X.f[:] 7624 // While processing the final array section "[:]" it is "f" which tells us 7625 // whether we are dealing with a member of a declared struct. 7626 const MemberExpr *EncounteredME = nullptr; 7627 7628 // Track for the total number of dimension. Start from one for the dummy 7629 // dimension. 7630 uint64_t DimSize = 1; 7631 7632 bool IsNonContiguous = CombinedInfo.NonContigInfo.IsNonContiguous; 7633 7634 for (; I != CE; ++I) { 7635 // If the current component is member of a struct (parent struct) mark it. 7636 if (!EncounteredME) { 7637 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression()); 7638 // If we encounter a PTR_AND_OBJ entry from now on it should be marked 7639 // as MEMBER_OF the parent struct. 7640 if (EncounteredME) { 7641 ShouldBeMemberOf = true; 7642 // Do not emit as complex pointer if this is actually not array-like 7643 // expression. 7644 if (FirstPointerInComplexData) { 7645 QualType Ty = std::prev(I) 7646 ->getAssociatedDeclaration() 7647 ->getType() 7648 .getNonReferenceType(); 7649 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7650 FirstPointerInComplexData = false; 7651 } 7652 } 7653 } 7654 7655 auto Next = std::next(I); 7656 7657 // We need to generate the addresses and sizes if this is the last 7658 // component, if the component is a pointer or if it is an array section 7659 // whose length can't be proved to be one. If this is a pointer, it 7660 // becomes the base address for the following components. 7661 7662 // A final array section, is one whose length can't be proved to be one. 7663 // If the map item is non-contiguous then we don't treat any array section 7664 // as final array section. 7665 bool IsFinalArraySection = 7666 !IsNonContiguous && 7667 isFinalArraySectionExpression(I->getAssociatedExpression()); 7668 7669 // Get information on whether the element is a pointer. Have to do a 7670 // special treatment for array sections given that they are built-in 7671 // types. 7672 const auto *OASE = 7673 dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression()); 7674 const auto *OAShE = 7675 dyn_cast<OMPArrayShapingExpr>(I->getAssociatedExpression()); 7676 const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression()); 7677 const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression()); 7678 bool IsPointer = 7679 OAShE || 7680 (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE) 7681 .getCanonicalType() 7682 ->isAnyPointerType()) || 7683 I->getAssociatedExpression()->getType()->isAnyPointerType(); 7684 bool IsNonDerefPointer = IsPointer && !UO && !BO && !IsNonContiguous; 7685 7686 if (OASE) 7687 ++DimSize; 7688 7689 if (Next == CE || IsNonDerefPointer || IsFinalArraySection) { 7690 // If this is not the last component, we expect the pointer to be 7691 // associated with an array expression or member expression. 7692 assert((Next == CE || 7693 isa<MemberExpr>(Next->getAssociatedExpression()) || 7694 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) || 7695 isa<OMPArraySectionExpr>(Next->getAssociatedExpression()) || 7696 isa<OMPArrayShapingExpr>(Next->getAssociatedExpression()) || 7697 isa<UnaryOperator>(Next->getAssociatedExpression()) || 7698 isa<BinaryOperator>(Next->getAssociatedExpression())) && 7699 "Unexpected expression"); 7700 7701 Address LB = Address::invalid(); 7702 if (OAShE) { 7703 LB = Address(CGF.EmitScalarExpr(OAShE->getBase()), 7704 CGF.getContext().getTypeAlignInChars( 7705 OAShE->getBase()->getType())); 7706 } else { 7707 LB = CGF.EmitOMPSharedLValue(I->getAssociatedExpression()) 7708 .getAddress(CGF); 7709 } 7710 7711 // If this component is a pointer inside the base struct then we don't 7712 // need to create any entry for it - it will be combined with the object 7713 // it is pointing to into a single PTR_AND_OBJ entry. 7714 bool IsMemberPointerOrAddr = 7715 (IsPointer || ForDeviceAddr) && EncounteredME && 7716 (dyn_cast<MemberExpr>(I->getAssociatedExpression()) == 7717 EncounteredME); 7718 if (!OverlappedElements.empty()) { 7719 // Handle base element with the info for overlapped elements. 7720 assert(!PartialStruct.Base.isValid() && "The base element is set."); 7721 assert(Next == CE && 7722 "Expected last element for the overlapped elements."); 7723 assert(!IsPointer && 7724 "Unexpected base element with the pointer type."); 7725 // Mark the whole struct as the struct that requires allocation on the 7726 // device. 7727 PartialStruct.LowestElem = {0, LB}; 7728 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars( 7729 I->getAssociatedExpression()->getType()); 7730 Address HB = CGF.Builder.CreateConstGEP( 7731 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB, 7732 CGF.VoidPtrTy), 7733 TypeSize.getQuantity() - 1); 7734 PartialStruct.HighestElem = { 7735 std::numeric_limits<decltype( 7736 PartialStruct.HighestElem.first)>::max(), 7737 HB}; 7738 PartialStruct.Base = BP; 7739 // Emit data for non-overlapped data. 7740 OpenMPOffloadMappingFlags Flags = 7741 OMP_MAP_MEMBER_OF | 7742 getMapTypeBits(MapType, MapModifiers, MotionModifiers, IsImplicit, 7743 /*AddPtrFlag=*/false, 7744 /*AddIsTargetParamFlag=*/false, IsNonContiguous); 7745 LB = BP; 7746 llvm::Value *Size = nullptr; 7747 // Do bitcopy of all non-overlapped structure elements. 7748 for (OMPClauseMappableExprCommon::MappableExprComponentListRef 7749 Component : OverlappedElements) { 7750 Address ComponentLB = Address::invalid(); 7751 for (const OMPClauseMappableExprCommon::MappableComponent &MC : 7752 Component) { 7753 if (MC.getAssociatedDeclaration()) { 7754 ComponentLB = 7755 CGF.EmitOMPSharedLValue(MC.getAssociatedExpression()) 7756 .getAddress(CGF); 7757 Size = CGF.Builder.CreatePtrDiff( 7758 CGF.EmitCastToVoidPtr(ComponentLB.getPointer()), 7759 CGF.EmitCastToVoidPtr(LB.getPointer())); 7760 break; 7761 } 7762 } 7763 assert(Size && "Failed to determine structure size"); 7764 CombinedInfo.BasePointers.push_back(BP.getPointer()); 7765 CombinedInfo.Pointers.push_back(LB.getPointer()); 7766 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 7767 Size, CGF.Int64Ty, /*isSigned=*/true)); 7768 CombinedInfo.Types.push_back(Flags); 7769 CombinedInfo.Mappers.push_back(nullptr); 7770 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize 7771 : 1); 7772 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1); 7773 } 7774 CombinedInfo.BasePointers.push_back(BP.getPointer()); 7775 CombinedInfo.Pointers.push_back(LB.getPointer()); 7776 Size = CGF.Builder.CreatePtrDiff( 7777 CGF.EmitCastToVoidPtr( 7778 CGF.Builder.CreateConstGEP(HB, 1).getPointer()), 7779 CGF.EmitCastToVoidPtr(LB.getPointer())); 7780 CombinedInfo.Sizes.push_back( 7781 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7782 CombinedInfo.Types.push_back(Flags); 7783 CombinedInfo.Mappers.push_back(nullptr); 7784 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize 7785 : 1); 7786 break; 7787 } 7788 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression()); 7789 if (!IsMemberPointerOrAddr || 7790 (Next == CE && MapType != OMPC_MAP_unknown)) { 7791 CombinedInfo.BasePointers.push_back(BP.getPointer()); 7792 CombinedInfo.Pointers.push_back(LB.getPointer()); 7793 CombinedInfo.Sizes.push_back( 7794 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7795 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize 7796 : 1); 7797 7798 // If Mapper is valid, the last component inherits the mapper. 7799 bool HasMapper = Mapper && Next == CE; 7800 CombinedInfo.Mappers.push_back(HasMapper ? Mapper : nullptr); 7801 7802 // We need to add a pointer flag for each map that comes from the 7803 // same expression except for the first one. We also need to signal 7804 // this map is the first one that relates with the current capture 7805 // (there is a set of entries for each capture). 7806 OpenMPOffloadMappingFlags Flags = getMapTypeBits( 7807 MapType, MapModifiers, MotionModifiers, IsImplicit, 7808 !IsExpressionFirstInfo || RequiresReference || 7809 FirstPointerInComplexData, 7810 IsCaptureFirstInfo && !RequiresReference, IsNonContiguous); 7811 7812 if (!IsExpressionFirstInfo) { 7813 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well, 7814 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags. 7815 if (IsPointer) 7816 Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS | 7817 OMP_MAP_DELETE | OMP_MAP_CLOSE); 7818 7819 if (ShouldBeMemberOf) { 7820 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag 7821 // should be later updated with the correct value of MEMBER_OF. 7822 Flags |= OMP_MAP_MEMBER_OF; 7823 // From now on, all subsequent PTR_AND_OBJ entries should not be 7824 // marked as MEMBER_OF. 7825 ShouldBeMemberOf = false; 7826 } 7827 } 7828 7829 CombinedInfo.Types.push_back(Flags); 7830 } 7831 7832 // If we have encountered a member expression so far, keep track of the 7833 // mapped member. If the parent is "*this", then the value declaration 7834 // is nullptr. 7835 if (EncounteredME) { 7836 const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl()); 7837 unsigned FieldIndex = FD->getFieldIndex(); 7838 7839 // Update info about the lowest and highest elements for this struct 7840 if (!PartialStruct.Base.isValid()) { 7841 PartialStruct.LowestElem = {FieldIndex, LB}; 7842 if (IsFinalArraySection) { 7843 Address HB = 7844 CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false) 7845 .getAddress(CGF); 7846 PartialStruct.HighestElem = {FieldIndex, HB}; 7847 } else { 7848 PartialStruct.HighestElem = {FieldIndex, LB}; 7849 } 7850 PartialStruct.Base = BP; 7851 } else if (FieldIndex < PartialStruct.LowestElem.first) { 7852 PartialStruct.LowestElem = {FieldIndex, LB}; 7853 } else if (FieldIndex > PartialStruct.HighestElem.first) { 7854 PartialStruct.HighestElem = {FieldIndex, LB}; 7855 } 7856 } 7857 7858 // Need to emit combined struct for array sections. 7859 if (IsFinalArraySection || IsNonContiguous) 7860 PartialStruct.IsArraySection = true; 7861 7862 // If we have a final array section, we are done with this expression. 7863 if (IsFinalArraySection) 7864 break; 7865 7866 // The pointer becomes the base for the next element. 7867 if (Next != CE) 7868 BP = LB; 7869 7870 IsExpressionFirstInfo = false; 7871 IsCaptureFirstInfo = false; 7872 FirstPointerInComplexData = false; 7873 } else if (FirstPointerInComplexData) { 7874 BP = CGF.EmitOMPSharedLValue(I->getAssociatedExpression()) 7875 .getAddress(CGF); 7876 FirstPointerInComplexData = false; 7877 } 7878 } 7879 7880 if (!IsNonContiguous) 7881 return; 7882 7883 const ASTContext &Context = CGF.getContext(); 7884 7885 // For supporting stride in array section, we need to initialize the first 7886 // dimension size as 1, first offset as 0, and first count as 1 7887 MapValuesArrayTy CurOffsets = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 0)}; 7888 MapValuesArrayTy CurCounts = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)}; 7889 MapValuesArrayTy CurStrides; 7890 MapValuesArrayTy DimSizes{llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)}; 7891 uint64_t ElementTypeSize; 7892 7893 // Collect Size information for each dimension and get the element size as 7894 // the first Stride. For example, for `int arr[10][10]`, the DimSizes 7895 // should be [10, 10] and the first stride is 4 btyes. 7896 for (const OMPClauseMappableExprCommon::MappableComponent &Component : 7897 Components) { 7898 const Expr *AssocExpr = Component.getAssociatedExpression(); 7899 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7900 7901 if (!OASE) 7902 continue; 7903 7904 QualType Ty = OMPArraySectionExpr::getBaseOriginalType(OASE->getBase()); 7905 auto *CAT = Context.getAsConstantArrayType(Ty); 7906 auto *VAT = Context.getAsVariableArrayType(Ty); 7907 7908 // We need all the dimension size except for the last dimension. 7909 assert((VAT || CAT || &Component == &*Components.begin()) && 7910 "Should be either ConstantArray or VariableArray if not the " 7911 "first Component"); 7912 7913 // Get element size if CurStrides is empty. 7914 if (CurStrides.empty()) { 7915 const Type *ElementType = nullptr; 7916 if (CAT) 7917 ElementType = CAT->getElementType().getTypePtr(); 7918 else if (VAT) 7919 ElementType = VAT->getElementType().getTypePtr(); 7920 else 7921 assert(&Component == &*Components.begin() && 7922 "Only expect pointer (non CAT or VAT) when this is the " 7923 "first Component"); 7924 // If ElementType is null, then it means the base is a pointer 7925 // (neither CAT nor VAT) and we'll attempt to get ElementType again 7926 // for next iteration. 7927 if (ElementType) { 7928 // For the case that having pointer as base, we need to remove one 7929 // level of indirection. 7930 if (&Component != &*Components.begin()) 7931 ElementType = ElementType->getPointeeOrArrayElementType(); 7932 ElementTypeSize = 7933 Context.getTypeSizeInChars(ElementType).getQuantity(); 7934 CurStrides.push_back( 7935 llvm::ConstantInt::get(CGF.Int64Ty, ElementTypeSize)); 7936 } 7937 } 7938 // Get dimension value except for the last dimension since we don't need 7939 // it. 7940 if (DimSizes.size() < Components.size() - 1) { 7941 if (CAT) 7942 DimSizes.push_back(llvm::ConstantInt::get( 7943 CGF.Int64Ty, CAT->getSize().getZExtValue())); 7944 else if (VAT) 7945 DimSizes.push_back(CGF.Builder.CreateIntCast( 7946 CGF.EmitScalarExpr(VAT->getSizeExpr()), CGF.Int64Ty, 7947 /*IsSigned=*/false)); 7948 } 7949 } 7950 7951 // Skip the dummy dimension since we have already have its information. 7952 auto DI = DimSizes.begin() + 1; 7953 // Product of dimension. 7954 llvm::Value *DimProd = 7955 llvm::ConstantInt::get(CGF.CGM.Int64Ty, ElementTypeSize); 7956 7957 // Collect info for non-contiguous. Notice that offset, count, and stride 7958 // are only meaningful for array-section, so we insert a null for anything 7959 // other than array-section. 7960 // Also, the size of offset, count, and stride are not the same as 7961 // pointers, base_pointers, sizes, or dims. Instead, the size of offset, 7962 // count, and stride are the same as the number of non-contiguous 7963 // declaration in target update to/from clause. 7964 for (const OMPClauseMappableExprCommon::MappableComponent &Component : 7965 Components) { 7966 const Expr *AssocExpr = Component.getAssociatedExpression(); 7967 7968 if (const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr)) { 7969 llvm::Value *Offset = CGF.Builder.CreateIntCast( 7970 CGF.EmitScalarExpr(AE->getIdx()), CGF.Int64Ty, 7971 /*isSigned=*/false); 7972 CurOffsets.push_back(Offset); 7973 CurCounts.push_back(llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/1)); 7974 CurStrides.push_back(CurStrides.back()); 7975 continue; 7976 } 7977 7978 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7979 7980 if (!OASE) 7981 continue; 7982 7983 // Offset 7984 const Expr *OffsetExpr = OASE->getLowerBound(); 7985 llvm::Value *Offset = nullptr; 7986 if (!OffsetExpr) { 7987 // If offset is absent, then we just set it to zero. 7988 Offset = llvm::ConstantInt::get(CGF.Int64Ty, 0); 7989 } else { 7990 Offset = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(OffsetExpr), 7991 CGF.Int64Ty, 7992 /*isSigned=*/false); 7993 } 7994 CurOffsets.push_back(Offset); 7995 7996 // Count 7997 const Expr *CountExpr = OASE->getLength(); 7998 llvm::Value *Count = nullptr; 7999 if (!CountExpr) { 8000 // In Clang, once a high dimension is an array section, we construct all 8001 // the lower dimension as array section, however, for case like 8002 // arr[0:2][2], Clang construct the inner dimension as an array section 8003 // but it actually is not in an array section form according to spec. 8004 if (!OASE->getColonLocFirst().isValid() && 8005 !OASE->getColonLocSecond().isValid()) { 8006 Count = llvm::ConstantInt::get(CGF.Int64Ty, 1); 8007 } else { 8008 // OpenMP 5.0, 2.1.5 Array Sections, Description. 8009 // When the length is absent it defaults to ⌈(size − 8010 // lower-bound)/stride⌉, where size is the size of the array 8011 // dimension. 8012 const Expr *StrideExpr = OASE->getStride(); 8013 llvm::Value *Stride = 8014 StrideExpr 8015 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr), 8016 CGF.Int64Ty, /*isSigned=*/false) 8017 : nullptr; 8018 if (Stride) 8019 Count = CGF.Builder.CreateUDiv( 8020 CGF.Builder.CreateNUWSub(*DI, Offset), Stride); 8021 else 8022 Count = CGF.Builder.CreateNUWSub(*DI, Offset); 8023 } 8024 } else { 8025 Count = CGF.EmitScalarExpr(CountExpr); 8026 } 8027 Count = CGF.Builder.CreateIntCast(Count, CGF.Int64Ty, /*isSigned=*/false); 8028 CurCounts.push_back(Count); 8029 8030 // Stride_n' = Stride_n * (D_0 * D_1 ... * D_n-1) * Unit size 8031 // Take `int arr[5][5][5]` and `arr[0:2:2][1:2:1][0:2:2]` as an example: 8032 // Offset Count Stride 8033 // D0 0 1 4 (int) <- dummy dimension 8034 // D1 0 2 8 (2 * (1) * 4) 8035 // D2 1 2 20 (1 * (1 * 5) * 4) 8036 // D3 0 2 200 (2 * (1 * 5 * 4) * 4) 8037 const Expr *StrideExpr = OASE->getStride(); 8038 llvm::Value *Stride = 8039 StrideExpr 8040 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr), 8041 CGF.Int64Ty, /*isSigned=*/false) 8042 : nullptr; 8043 DimProd = CGF.Builder.CreateNUWMul(DimProd, *(DI - 1)); 8044 if (Stride) 8045 CurStrides.push_back(CGF.Builder.CreateNUWMul(DimProd, Stride)); 8046 else 8047 CurStrides.push_back(DimProd); 8048 if (DI != DimSizes.end()) 8049 ++DI; 8050 } 8051 8052 CombinedInfo.NonContigInfo.Offsets.push_back(CurOffsets); 8053 CombinedInfo.NonContigInfo.Counts.push_back(CurCounts); 8054 CombinedInfo.NonContigInfo.Strides.push_back(CurStrides); 8055 } 8056 8057 /// Return the adjusted map modifiers if the declaration a capture refers to 8058 /// appears in a first-private clause. This is expected to be used only with 8059 /// directives that start with 'target'. 8060 MappableExprsHandler::OpenMPOffloadMappingFlags 8061 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const { 8062 assert(Cap.capturesVariable() && "Expected capture by reference only!"); 8063 8064 // A first private variable captured by reference will use only the 8065 // 'private ptr' and 'map to' flag. Return the right flags if the captured 8066 // declaration is known as first-private in this handler. 8067 if (FirstPrivateDecls.count(Cap.getCapturedVar())) { 8068 if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) && 8069 Cap.getCaptureKind() == CapturedStmt::VCK_ByRef) 8070 return MappableExprsHandler::OMP_MAP_ALWAYS | 8071 MappableExprsHandler::OMP_MAP_TO; 8072 if (Cap.getCapturedVar()->getType()->isAnyPointerType()) 8073 return MappableExprsHandler::OMP_MAP_TO | 8074 MappableExprsHandler::OMP_MAP_PTR_AND_OBJ; 8075 return MappableExprsHandler::OMP_MAP_PRIVATE | 8076 MappableExprsHandler::OMP_MAP_TO; 8077 } 8078 return MappableExprsHandler::OMP_MAP_TO | 8079 MappableExprsHandler::OMP_MAP_FROM; 8080 } 8081 8082 static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) { 8083 // Rotate by getFlagMemberOffset() bits. 8084 return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1) 8085 << getFlagMemberOffset()); 8086 } 8087 8088 static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags, 8089 OpenMPOffloadMappingFlags MemberOfFlag) { 8090 // If the entry is PTR_AND_OBJ but has not been marked with the special 8091 // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be 8092 // marked as MEMBER_OF. 8093 if ((Flags & OMP_MAP_PTR_AND_OBJ) && 8094 ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF)) 8095 return; 8096 8097 // Reset the placeholder value to prepare the flag for the assignment of the 8098 // proper MEMBER_OF value. 8099 Flags &= ~OMP_MAP_MEMBER_OF; 8100 Flags |= MemberOfFlag; 8101 } 8102 8103 void getPlainLayout(const CXXRecordDecl *RD, 8104 llvm::SmallVectorImpl<const FieldDecl *> &Layout, 8105 bool AsBase) const { 8106 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD); 8107 8108 llvm::StructType *St = 8109 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType(); 8110 8111 unsigned NumElements = St->getNumElements(); 8112 llvm::SmallVector< 8113 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4> 8114 RecordLayout(NumElements); 8115 8116 // Fill bases. 8117 for (const auto &I : RD->bases()) { 8118 if (I.isVirtual()) 8119 continue; 8120 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8121 // Ignore empty bases. 8122 if (Base->isEmpty() || CGF.getContext() 8123 .getASTRecordLayout(Base) 8124 .getNonVirtualSize() 8125 .isZero()) 8126 continue; 8127 8128 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base); 8129 RecordLayout[FieldIndex] = Base; 8130 } 8131 // Fill in virtual bases. 8132 for (const auto &I : RD->vbases()) { 8133 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8134 // Ignore empty bases. 8135 if (Base->isEmpty()) 8136 continue; 8137 unsigned FieldIndex = RL.getVirtualBaseIndex(Base); 8138 if (RecordLayout[FieldIndex]) 8139 continue; 8140 RecordLayout[FieldIndex] = Base; 8141 } 8142 // Fill in all the fields. 8143 assert(!RD->isUnion() && "Unexpected union."); 8144 for (const auto *Field : RD->fields()) { 8145 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we 8146 // will fill in later.) 8147 if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) { 8148 unsigned FieldIndex = RL.getLLVMFieldNo(Field); 8149 RecordLayout[FieldIndex] = Field; 8150 } 8151 } 8152 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *> 8153 &Data : RecordLayout) { 8154 if (Data.isNull()) 8155 continue; 8156 if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>()) 8157 getPlainLayout(Base, Layout, /*AsBase=*/true); 8158 else 8159 Layout.push_back(Data.get<const FieldDecl *>()); 8160 } 8161 } 8162 8163 public: 8164 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF) 8165 : CurDir(&Dir), CGF(CGF) { 8166 // Extract firstprivate clause information. 8167 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>()) 8168 for (const auto *D : C->varlists()) 8169 FirstPrivateDecls.try_emplace( 8170 cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit()); 8171 // Extract implicit firstprivates from uses_allocators clauses. 8172 for (const auto *C : Dir.getClausesOfKind<OMPUsesAllocatorsClause>()) { 8173 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) { 8174 OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I); 8175 if (const auto *DRE = dyn_cast_or_null<DeclRefExpr>(D.AllocatorTraits)) 8176 FirstPrivateDecls.try_emplace(cast<VarDecl>(DRE->getDecl()), 8177 /*Implicit=*/true); 8178 else if (const auto *VD = dyn_cast<VarDecl>( 8179 cast<DeclRefExpr>(D.Allocator->IgnoreParenImpCasts()) 8180 ->getDecl())) 8181 FirstPrivateDecls.try_emplace(VD, /*Implicit=*/true); 8182 } 8183 } 8184 // Extract device pointer clause information. 8185 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>()) 8186 for (auto L : C->component_lists()) 8187 DevPointersMap[std::get<0>(L)].push_back(std::get<1>(L)); 8188 } 8189 8190 /// Constructor for the declare mapper directive. 8191 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF) 8192 : CurDir(&Dir), CGF(CGF) {} 8193 8194 /// Generate code for the combined entry if we have a partially mapped struct 8195 /// and take care of the mapping flags of the arguments corresponding to 8196 /// individual struct members. 8197 void emitCombinedEntry(MapCombinedInfoTy &CombinedInfo, 8198 MapFlagsArrayTy &CurTypes, 8199 const StructRangeInfoTy &PartialStruct, 8200 bool NotTargetParams = false) const { 8201 if (CurTypes.size() == 1 && 8202 ((CurTypes.back() & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF) && 8203 !PartialStruct.IsArraySection) 8204 return; 8205 // Base is the base of the struct 8206 CombinedInfo.BasePointers.push_back(PartialStruct.Base.getPointer()); 8207 // Pointer is the address of the lowest element 8208 llvm::Value *LB = PartialStruct.LowestElem.second.getPointer(); 8209 CombinedInfo.Pointers.push_back(LB); 8210 // There should not be a mapper for a combined entry. 8211 CombinedInfo.Mappers.push_back(nullptr); 8212 // Size is (addr of {highest+1} element) - (addr of lowest element) 8213 llvm::Value *HB = PartialStruct.HighestElem.second.getPointer(); 8214 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1); 8215 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy); 8216 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy); 8217 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr); 8218 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty, 8219 /*isSigned=*/false); 8220 CombinedInfo.Sizes.push_back(Size); 8221 // Map type is always TARGET_PARAM, if generate info for captures. 8222 CombinedInfo.Types.push_back(NotTargetParams ? OMP_MAP_NONE 8223 : OMP_MAP_TARGET_PARAM); 8224 // If any element has the present modifier, then make sure the runtime 8225 // doesn't attempt to allocate the struct. 8226 if (CurTypes.end() != 8227 llvm::find_if(CurTypes, [](OpenMPOffloadMappingFlags Type) { 8228 return Type & OMP_MAP_PRESENT; 8229 })) 8230 CombinedInfo.Types.back() |= OMP_MAP_PRESENT; 8231 // Remove TARGET_PARAM flag from the first element if any. 8232 if (!CurTypes.empty()) 8233 CurTypes.front() &= ~OMP_MAP_TARGET_PARAM; 8234 8235 // All other current entries will be MEMBER_OF the combined entry 8236 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8237 // 0xFFFF in the MEMBER_OF field). 8238 OpenMPOffloadMappingFlags MemberOfFlag = 8239 getMemberOfFlag(CombinedInfo.BasePointers.size() - 1); 8240 for (auto &M : CurTypes) 8241 setCorrectMemberOfFlag(M, MemberOfFlag); 8242 } 8243 8244 /// Generate all the base pointers, section pointers, sizes, map types, and 8245 /// mappers for the extracted mappable expressions (all included in \a 8246 /// CombinedInfo). Also, for each item that relates with a device pointer, a 8247 /// pair of the relevant declaration and index where it occurs is appended to 8248 /// the device pointers info array. 8249 void generateAllInfo( 8250 MapCombinedInfoTy &CombinedInfo, bool NotTargetParams = false, 8251 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet = 8252 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const { 8253 // We have to process the component lists that relate with the same 8254 // declaration in a single chunk so that we can generate the map flags 8255 // correctly. Therefore, we organize all lists in a map. 8256 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8257 8258 // Helper function to fill the information map for the different supported 8259 // clauses. 8260 auto &&InfoGen = 8261 [&Info, &SkipVarSet]( 8262 const ValueDecl *D, 8263 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8264 OpenMPMapClauseKind MapType, 8265 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8266 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, 8267 bool ReturnDevicePointer, bool IsImplicit, const ValueDecl *Mapper, 8268 bool ForDeviceAddr = false) { 8269 const ValueDecl *VD = 8270 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8271 if (SkipVarSet.count(VD)) 8272 return; 8273 Info[VD].emplace_back(L, MapType, MapModifiers, MotionModifiers, 8274 ReturnDevicePointer, IsImplicit, Mapper, 8275 ForDeviceAddr); 8276 }; 8277 8278 assert(CurDir.is<const OMPExecutableDirective *>() && 8279 "Expect a executable directive"); 8280 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8281 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) 8282 for (const auto L : C->component_lists()) { 8283 InfoGen(std::get<0>(L), std::get<1>(L), C->getMapType(), 8284 C->getMapTypeModifiers(), llvm::None, 8285 /*ReturnDevicePointer=*/false, C->isImplicit(), std::get<2>(L)); 8286 } 8287 for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>()) 8288 for (const auto L : C->component_lists()) { 8289 InfoGen(std::get<0>(L), std::get<1>(L), OMPC_MAP_to, llvm::None, 8290 C->getMotionModifiers(), /*ReturnDevicePointer=*/false, 8291 C->isImplicit(), std::get<2>(L)); 8292 } 8293 for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>()) 8294 for (const auto L : C->component_lists()) { 8295 InfoGen(std::get<0>(L), std::get<1>(L), OMPC_MAP_from, llvm::None, 8296 C->getMotionModifiers(), /*ReturnDevicePointer=*/false, 8297 C->isImplicit(), std::get<2>(L)); 8298 } 8299 8300 // Look at the use_device_ptr clause information and mark the existing map 8301 // entries as such. If there is no map information for an entry in the 8302 // use_device_ptr list, we create one with map type 'alloc' and zero size 8303 // section. It is the user fault if that was not mapped before. If there is 8304 // no map information and the pointer is a struct member, then we defer the 8305 // emission of that entry until the whole struct has been processed. 8306 llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>> 8307 DeferredInfo; 8308 MapCombinedInfoTy UseDevicePtrCombinedInfo; 8309 8310 for (const auto *C : 8311 CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) { 8312 for (const auto L : C->component_lists()) { 8313 OMPClauseMappableExprCommon::MappableExprComponentListRef Components = 8314 std::get<1>(L); 8315 assert(!Components.empty() && 8316 "Not expecting empty list of components!"); 8317 const ValueDecl *VD = Components.back().getAssociatedDeclaration(); 8318 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 8319 const Expr *IE = Components.back().getAssociatedExpression(); 8320 // If the first component is a member expression, we have to look into 8321 // 'this', which maps to null in the map of map information. Otherwise 8322 // look directly for the information. 8323 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 8324 8325 // We potentially have map information for this declaration already. 8326 // Look for the first set of components that refer to it. 8327 if (It != Info.end()) { 8328 auto *CI = llvm::find_if(It->second, [VD](const MapInfo &MI) { 8329 return MI.Components.back().getAssociatedDeclaration() == VD; 8330 }); 8331 // If we found a map entry, signal that the pointer has to be returned 8332 // and move on to the next declaration. 8333 // Exclude cases where the base pointer is mapped as array subscript, 8334 // array section or array shaping. The base address is passed as a 8335 // pointer to base in this case and cannot be used as a base for 8336 // use_device_ptr list item. 8337 if (CI != It->second.end()) { 8338 auto PrevCI = std::next(CI->Components.rbegin()); 8339 const auto *VarD = dyn_cast<VarDecl>(VD); 8340 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 8341 isa<MemberExpr>(IE) || 8342 !VD->getType().getNonReferenceType()->isPointerType() || 8343 PrevCI == CI->Components.rend() || 8344 isa<MemberExpr>(PrevCI->getAssociatedExpression()) || !VarD || 8345 VarD->hasLocalStorage()) { 8346 CI->ReturnDevicePointer = true; 8347 continue; 8348 } 8349 } 8350 } 8351 8352 // We didn't find any match in our map information - generate a zero 8353 // size array section - if the pointer is a struct member we defer this 8354 // action until the whole struct has been processed. 8355 if (isa<MemberExpr>(IE)) { 8356 // Insert the pointer into Info to be processed by 8357 // generateInfoForComponentList. Because it is a member pointer 8358 // without a pointee, no entry will be generated for it, therefore 8359 // we need to generate one after the whole struct has been processed. 8360 // Nonetheless, generateInfoForComponentList must be called to take 8361 // the pointer into account for the calculation of the range of the 8362 // partial struct. 8363 InfoGen(nullptr, Components, OMPC_MAP_unknown, llvm::None, llvm::None, 8364 /*ReturnDevicePointer=*/false, C->isImplicit(), nullptr); 8365 DeferredInfo[nullptr].emplace_back(IE, VD, /*ForDeviceAddr=*/false); 8366 } else { 8367 llvm::Value *Ptr = 8368 CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc()); 8369 UseDevicePtrCombinedInfo.BasePointers.emplace_back(Ptr, VD); 8370 UseDevicePtrCombinedInfo.Pointers.push_back(Ptr); 8371 UseDevicePtrCombinedInfo.Sizes.push_back( 8372 llvm::Constant::getNullValue(CGF.Int64Ty)); 8373 UseDevicePtrCombinedInfo.Types.push_back( 8374 OMP_MAP_RETURN_PARAM | 8375 (NotTargetParams ? OMP_MAP_NONE : OMP_MAP_TARGET_PARAM)); 8376 UseDevicePtrCombinedInfo.Mappers.push_back(nullptr); 8377 } 8378 } 8379 } 8380 8381 // Look at the use_device_addr clause information and mark the existing map 8382 // entries as such. If there is no map information for an entry in the 8383 // use_device_addr list, we create one with map type 'alloc' and zero size 8384 // section. It is the user fault if that was not mapped before. If there is 8385 // no map information and the pointer is a struct member, then we defer the 8386 // emission of that entry until the whole struct has been processed. 8387 llvm::SmallDenseSet<CanonicalDeclPtr<const Decl>, 4> Processed; 8388 for (const auto *C : 8389 CurExecDir->getClausesOfKind<OMPUseDeviceAddrClause>()) { 8390 for (const auto L : C->component_lists()) { 8391 assert(!std::get<1>(L).empty() && 8392 "Not expecting empty list of components!"); 8393 const ValueDecl *VD = std::get<1>(L).back().getAssociatedDeclaration(); 8394 if (!Processed.insert(VD).second) 8395 continue; 8396 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 8397 const Expr *IE = std::get<1>(L).back().getAssociatedExpression(); 8398 // If the first component is a member expression, we have to look into 8399 // 'this', which maps to null in the map of map information. Otherwise 8400 // look directly for the information. 8401 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 8402 8403 // We potentially have map information for this declaration already. 8404 // Look for the first set of components that refer to it. 8405 if (It != Info.end()) { 8406 auto *CI = llvm::find_if(It->second, [VD](const MapInfo &MI) { 8407 return MI.Components.back().getAssociatedDeclaration() == VD; 8408 }); 8409 // If we found a map entry, signal that the pointer has to be returned 8410 // and move on to the next declaration. 8411 if (CI != It->second.end()) { 8412 CI->ReturnDevicePointer = true; 8413 continue; 8414 } 8415 } 8416 8417 // We didn't find any match in our map information - generate a zero 8418 // size array section - if the pointer is a struct member we defer this 8419 // action until the whole struct has been processed. 8420 if (isa<MemberExpr>(IE)) { 8421 // Insert the pointer into Info to be processed by 8422 // generateInfoForComponentList. Because it is a member pointer 8423 // without a pointee, no entry will be generated for it, therefore 8424 // we need to generate one after the whole struct has been processed. 8425 // Nonetheless, generateInfoForComponentList must be called to take 8426 // the pointer into account for the calculation of the range of the 8427 // partial struct. 8428 InfoGen(nullptr, std::get<1>(L), OMPC_MAP_unknown, llvm::None, 8429 llvm::None, /*ReturnDevicePointer=*/false, C->isImplicit(), 8430 nullptr, /*ForDeviceAddr=*/true); 8431 DeferredInfo[nullptr].emplace_back(IE, VD, /*ForDeviceAddr=*/true); 8432 } else { 8433 llvm::Value *Ptr; 8434 if (IE->isGLValue()) 8435 Ptr = CGF.EmitLValue(IE).getPointer(CGF); 8436 else 8437 Ptr = CGF.EmitScalarExpr(IE); 8438 CombinedInfo.BasePointers.emplace_back(Ptr, VD); 8439 CombinedInfo.Pointers.push_back(Ptr); 8440 CombinedInfo.Sizes.push_back( 8441 llvm::Constant::getNullValue(CGF.Int64Ty)); 8442 CombinedInfo.Types.push_back( 8443 OMP_MAP_RETURN_PARAM | 8444 (NotTargetParams ? OMP_MAP_NONE : OMP_MAP_TARGET_PARAM)); 8445 CombinedInfo.Mappers.push_back(nullptr); 8446 } 8447 } 8448 } 8449 8450 for (const auto &M : Info) { 8451 // We need to know when we generate information for the first component 8452 // associated with a capture, because the mapping flags depend on it. 8453 bool IsFirstComponentList = !NotTargetParams; 8454 8455 // Temporary generated information. 8456 MapCombinedInfoTy CurInfo; 8457 StructRangeInfoTy PartialStruct; 8458 8459 for (const MapInfo &L : M.second) { 8460 assert(!L.Components.empty() && 8461 "Not expecting declaration with no component lists."); 8462 8463 // Remember the current base pointer index. 8464 unsigned CurrentBasePointersIdx = CurInfo.BasePointers.size(); 8465 CurInfo.NonContigInfo.IsNonContiguous = 8466 L.Components.back().isNonContiguous(); 8467 generateInfoForComponentList(L.MapType, L.MapModifiers, 8468 L.MotionModifiers, L.Components, CurInfo, 8469 PartialStruct, IsFirstComponentList, 8470 L.IsImplicit, L.Mapper, L.ForDeviceAddr); 8471 8472 // If this entry relates with a device pointer, set the relevant 8473 // declaration and add the 'return pointer' flag. 8474 if (L.ReturnDevicePointer) { 8475 assert(CurInfo.BasePointers.size() > CurrentBasePointersIdx && 8476 "Unexpected number of mapped base pointers."); 8477 8478 const ValueDecl *RelevantVD = 8479 L.Components.back().getAssociatedDeclaration(); 8480 assert(RelevantVD && 8481 "No relevant declaration related with device pointer??"); 8482 8483 CurInfo.BasePointers[CurrentBasePointersIdx].setDevicePtrDecl( 8484 RelevantVD); 8485 CurInfo.Types[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM; 8486 } 8487 IsFirstComponentList = false; 8488 } 8489 8490 // Append any pending zero-length pointers which are struct members and 8491 // used with use_device_ptr or use_device_addr. 8492 auto CI = DeferredInfo.find(M.first); 8493 if (CI != DeferredInfo.end()) { 8494 for (const DeferredDevicePtrEntryTy &L : CI->second) { 8495 llvm::Value *BasePtr; 8496 llvm::Value *Ptr; 8497 if (L.ForDeviceAddr) { 8498 if (L.IE->isGLValue()) 8499 Ptr = this->CGF.EmitLValue(L.IE).getPointer(CGF); 8500 else 8501 Ptr = this->CGF.EmitScalarExpr(L.IE); 8502 BasePtr = Ptr; 8503 // Entry is RETURN_PARAM. Also, set the placeholder value 8504 // MEMBER_OF=FFFF so that the entry is later updated with the 8505 // correct value of MEMBER_OF. 8506 CurInfo.Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_MEMBER_OF); 8507 } else { 8508 BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF); 8509 Ptr = this->CGF.EmitLoadOfScalar(this->CGF.EmitLValue(L.IE), 8510 L.IE->getExprLoc()); 8511 // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder 8512 // value MEMBER_OF=FFFF so that the entry is later updated with the 8513 // correct value of MEMBER_OF. 8514 CurInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM | 8515 OMP_MAP_MEMBER_OF); 8516 } 8517 CurInfo.BasePointers.emplace_back(BasePtr, L.VD); 8518 CurInfo.Pointers.push_back(Ptr); 8519 CurInfo.Sizes.push_back( 8520 llvm::Constant::getNullValue(this->CGF.Int64Ty)); 8521 CurInfo.Mappers.push_back(nullptr); 8522 } 8523 } 8524 8525 // If there is an entry in PartialStruct it means we have a struct with 8526 // individual members mapped. Emit an extra combined entry. 8527 if (PartialStruct.Base.isValid()) 8528 emitCombinedEntry(CombinedInfo, CurInfo.Types, PartialStruct, 8529 NotTargetParams); 8530 8531 // We need to append the results of this capture to what we already have. 8532 CombinedInfo.append(CurInfo); 8533 } 8534 // Append data for use_device_ptr clauses. 8535 CombinedInfo.append(UseDevicePtrCombinedInfo); 8536 } 8537 8538 /// Generate all the base pointers, section pointers, sizes, map types, and 8539 /// mappers for the extracted map clauses of user-defined mapper (all included 8540 /// in \a CombinedInfo). 8541 void generateAllInfoForMapper(MapCombinedInfoTy &CombinedInfo) const { 8542 assert(CurDir.is<const OMPDeclareMapperDecl *>() && 8543 "Expect a declare mapper directive"); 8544 const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>(); 8545 // We have to process the component lists that relate with the same 8546 // declaration in a single chunk so that we can generate the map flags 8547 // correctly. Therefore, we organize all lists in a map. 8548 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8549 8550 // Fill the information map for map clauses. 8551 for (const auto *C : CurMapperDir->clauselists()) { 8552 const auto *MC = cast<OMPMapClause>(C); 8553 for (const auto L : MC->component_lists()) { 8554 const ValueDecl *VD = 8555 std::get<0>(L) ? cast<ValueDecl>(std::get<0>(L)->getCanonicalDecl()) 8556 : nullptr; 8557 // Get the corresponding user-defined mapper. 8558 Info[VD].emplace_back(std::get<1>(L), MC->getMapType(), 8559 MC->getMapTypeModifiers(), llvm::None, 8560 /*ReturnDevicePointer=*/false, MC->isImplicit(), 8561 std::get<2>(L)); 8562 } 8563 } 8564 8565 for (const auto &M : Info) { 8566 // We need to know when we generate information for the first component 8567 // associated with a capture, because the mapping flags depend on it. 8568 bool IsFirstComponentList = true; 8569 8570 // Temporary generated information. 8571 MapCombinedInfoTy CurInfo; 8572 StructRangeInfoTy PartialStruct; 8573 8574 for (const MapInfo &L : M.second) { 8575 assert(!L.Components.empty() && 8576 "Not expecting declaration with no component lists."); 8577 generateInfoForComponentList(L.MapType, L.MapModifiers, 8578 L.MotionModifiers, L.Components, CurInfo, 8579 PartialStruct, IsFirstComponentList, 8580 L.IsImplicit, L.Mapper, L.ForDeviceAddr); 8581 IsFirstComponentList = false; 8582 } 8583 8584 // If there is an entry in PartialStruct it means we have a struct with 8585 // individual members mapped. Emit an extra combined entry. 8586 if (PartialStruct.Base.isValid()) { 8587 CurInfo.NonContigInfo.Dims.push_back(0); 8588 emitCombinedEntry(CombinedInfo, CurInfo.Types, PartialStruct); 8589 } 8590 8591 // We need to append the results of this capture to what we already have. 8592 CombinedInfo.append(CurInfo); 8593 } 8594 } 8595 8596 /// Emit capture info for lambdas for variables captured by reference. 8597 void generateInfoForLambdaCaptures( 8598 const ValueDecl *VD, llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo, 8599 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const { 8600 const auto *RD = VD->getType() 8601 .getCanonicalType() 8602 .getNonReferenceType() 8603 ->getAsCXXRecordDecl(); 8604 if (!RD || !RD->isLambda()) 8605 return; 8606 Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD)); 8607 LValue VDLVal = CGF.MakeAddrLValue( 8608 VDAddr, VD->getType().getCanonicalType().getNonReferenceType()); 8609 llvm::DenseMap<const VarDecl *, FieldDecl *> Captures; 8610 FieldDecl *ThisCapture = nullptr; 8611 RD->getCaptureFields(Captures, ThisCapture); 8612 if (ThisCapture) { 8613 LValue ThisLVal = 8614 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture); 8615 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture); 8616 LambdaPointers.try_emplace(ThisLVal.getPointer(CGF), 8617 VDLVal.getPointer(CGF)); 8618 CombinedInfo.BasePointers.push_back(ThisLVal.getPointer(CGF)); 8619 CombinedInfo.Pointers.push_back(ThisLValVal.getPointer(CGF)); 8620 CombinedInfo.Sizes.push_back( 8621 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8622 CGF.Int64Ty, /*isSigned=*/true)); 8623 CombinedInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8624 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8625 CombinedInfo.Mappers.push_back(nullptr); 8626 } 8627 for (const LambdaCapture &LC : RD->captures()) { 8628 if (!LC.capturesVariable()) 8629 continue; 8630 const VarDecl *VD = LC.getCapturedVar(); 8631 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType()) 8632 continue; 8633 auto It = Captures.find(VD); 8634 assert(It != Captures.end() && "Found lambda capture without field."); 8635 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second); 8636 if (LC.getCaptureKind() == LCK_ByRef) { 8637 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second); 8638 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8639 VDLVal.getPointer(CGF)); 8640 CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF)); 8641 CombinedInfo.Pointers.push_back(VarLValVal.getPointer(CGF)); 8642 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 8643 CGF.getTypeSize( 8644 VD->getType().getCanonicalType().getNonReferenceType()), 8645 CGF.Int64Ty, /*isSigned=*/true)); 8646 } else { 8647 RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation()); 8648 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8649 VDLVal.getPointer(CGF)); 8650 CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF)); 8651 CombinedInfo.Pointers.push_back(VarRVal.getScalarVal()); 8652 CombinedInfo.Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0)); 8653 } 8654 CombinedInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8655 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8656 CombinedInfo.Mappers.push_back(nullptr); 8657 } 8658 } 8659 8660 /// Set correct indices for lambdas captures. 8661 void adjustMemberOfForLambdaCaptures( 8662 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers, 8663 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 8664 MapFlagsArrayTy &Types) const { 8665 for (unsigned I = 0, E = Types.size(); I < E; ++I) { 8666 // Set correct member_of idx for all implicit lambda captures. 8667 if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8668 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT)) 8669 continue; 8670 llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]); 8671 assert(BasePtr && "Unable to find base lambda address."); 8672 int TgtIdx = -1; 8673 for (unsigned J = I; J > 0; --J) { 8674 unsigned Idx = J - 1; 8675 if (Pointers[Idx] != BasePtr) 8676 continue; 8677 TgtIdx = Idx; 8678 break; 8679 } 8680 assert(TgtIdx != -1 && "Unable to find parent lambda."); 8681 // All other current entries will be MEMBER_OF the combined entry 8682 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8683 // 0xFFFF in the MEMBER_OF field). 8684 OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx); 8685 setCorrectMemberOfFlag(Types[I], MemberOfFlag); 8686 } 8687 } 8688 8689 /// Generate the base pointers, section pointers, sizes, map types, and 8690 /// mappers associated to a given capture (all included in \a CombinedInfo). 8691 void generateInfoForCapture(const CapturedStmt::Capture *Cap, 8692 llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo, 8693 StructRangeInfoTy &PartialStruct) const { 8694 assert(!Cap->capturesVariableArrayType() && 8695 "Not expecting to generate map info for a variable array type!"); 8696 8697 // We need to know when we generating information for the first component 8698 const ValueDecl *VD = Cap->capturesThis() 8699 ? nullptr 8700 : Cap->getCapturedVar()->getCanonicalDecl(); 8701 8702 // If this declaration appears in a is_device_ptr clause we just have to 8703 // pass the pointer by value. If it is a reference to a declaration, we just 8704 // pass its value. 8705 if (DevPointersMap.count(VD)) { 8706 CombinedInfo.BasePointers.emplace_back(Arg, VD); 8707 CombinedInfo.Pointers.push_back(Arg); 8708 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 8709 CGF.getTypeSize(CGF.getContext().VoidPtrTy), CGF.Int64Ty, 8710 /*isSigned=*/true)); 8711 CombinedInfo.Types.push_back( 8712 (Cap->capturesVariable() ? OMP_MAP_TO : OMP_MAP_LITERAL) | 8713 OMP_MAP_TARGET_PARAM); 8714 CombinedInfo.Mappers.push_back(nullptr); 8715 return; 8716 } 8717 8718 using MapData = 8719 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef, 8720 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool, 8721 const ValueDecl *>; 8722 SmallVector<MapData, 4> DeclComponentLists; 8723 assert(CurDir.is<const OMPExecutableDirective *>() && 8724 "Expect a executable directive"); 8725 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8726 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8727 for (const auto L : C->decl_component_lists(VD)) { 8728 const ValueDecl *VDecl, *Mapper; 8729 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8730 std::tie(VDecl, Components, Mapper) = L; 8731 assert(VDecl == VD && "We got information for the wrong declaration??"); 8732 assert(!Components.empty() && 8733 "Not expecting declaration with no component lists."); 8734 DeclComponentLists.emplace_back(Components, C->getMapType(), 8735 C->getMapTypeModifiers(), 8736 C->isImplicit(), Mapper); 8737 } 8738 } 8739 8740 // Find overlapping elements (including the offset from the base element). 8741 llvm::SmallDenseMap< 8742 const MapData *, 8743 llvm::SmallVector< 8744 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>, 8745 4> 8746 OverlappedData; 8747 size_t Count = 0; 8748 for (const MapData &L : DeclComponentLists) { 8749 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8750 OpenMPMapClauseKind MapType; 8751 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8752 bool IsImplicit; 8753 const ValueDecl *Mapper; 8754 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper) = L; 8755 ++Count; 8756 for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) { 8757 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1; 8758 std::tie(Components1, MapType, MapModifiers, IsImplicit, Mapper) = L1; 8759 auto CI = Components.rbegin(); 8760 auto CE = Components.rend(); 8761 auto SI = Components1.rbegin(); 8762 auto SE = Components1.rend(); 8763 for (; CI != CE && SI != SE; ++CI, ++SI) { 8764 if (CI->getAssociatedExpression()->getStmtClass() != 8765 SI->getAssociatedExpression()->getStmtClass()) 8766 break; 8767 // Are we dealing with different variables/fields? 8768 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration()) 8769 break; 8770 } 8771 // Found overlapping if, at least for one component, reached the head of 8772 // the components list. 8773 if (CI == CE || SI == SE) { 8774 assert((CI != CE || SI != SE) && 8775 "Unexpected full match of the mapping components."); 8776 const MapData &BaseData = CI == CE ? L : L1; 8777 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData = 8778 SI == SE ? Components : Components1; 8779 auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData); 8780 OverlappedElements.getSecond().push_back(SubData); 8781 } 8782 } 8783 } 8784 // Sort the overlapped elements for each item. 8785 llvm::SmallVector<const FieldDecl *, 4> Layout; 8786 if (!OverlappedData.empty()) { 8787 if (const auto *CRD = 8788 VD->getType().getCanonicalType()->getAsCXXRecordDecl()) 8789 getPlainLayout(CRD, Layout, /*AsBase=*/false); 8790 else { 8791 const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl(); 8792 Layout.append(RD->field_begin(), RD->field_end()); 8793 } 8794 } 8795 for (auto &Pair : OverlappedData) { 8796 llvm::sort( 8797 Pair.getSecond(), 8798 [&Layout]( 8799 OMPClauseMappableExprCommon::MappableExprComponentListRef First, 8800 OMPClauseMappableExprCommon::MappableExprComponentListRef 8801 Second) { 8802 auto CI = First.rbegin(); 8803 auto CE = First.rend(); 8804 auto SI = Second.rbegin(); 8805 auto SE = Second.rend(); 8806 for (; CI != CE && SI != SE; ++CI, ++SI) { 8807 if (CI->getAssociatedExpression()->getStmtClass() != 8808 SI->getAssociatedExpression()->getStmtClass()) 8809 break; 8810 // Are we dealing with different variables/fields? 8811 if (CI->getAssociatedDeclaration() != 8812 SI->getAssociatedDeclaration()) 8813 break; 8814 } 8815 8816 // Lists contain the same elements. 8817 if (CI == CE && SI == SE) 8818 return false; 8819 8820 // List with less elements is less than list with more elements. 8821 if (CI == CE || SI == SE) 8822 return CI == CE; 8823 8824 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration()); 8825 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration()); 8826 if (FD1->getParent() == FD2->getParent()) 8827 return FD1->getFieldIndex() < FD2->getFieldIndex(); 8828 const auto It = 8829 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) { 8830 return FD == FD1 || FD == FD2; 8831 }); 8832 return *It == FD1; 8833 }); 8834 } 8835 8836 // Associated with a capture, because the mapping flags depend on it. 8837 // Go through all of the elements with the overlapped elements. 8838 for (const auto &Pair : OverlappedData) { 8839 const MapData &L = *Pair.getFirst(); 8840 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8841 OpenMPMapClauseKind MapType; 8842 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8843 bool IsImplicit; 8844 const ValueDecl *Mapper; 8845 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper) = L; 8846 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 8847 OverlappedComponents = Pair.getSecond(); 8848 bool IsFirstComponentList = true; 8849 generateInfoForComponentList( 8850 MapType, MapModifiers, llvm::None, Components, CombinedInfo, 8851 PartialStruct, IsFirstComponentList, IsImplicit, Mapper, 8852 /*ForDeviceAddr=*/false, OverlappedComponents); 8853 } 8854 // Go through other elements without overlapped elements. 8855 bool IsFirstComponentList = OverlappedData.empty(); 8856 for (const MapData &L : DeclComponentLists) { 8857 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8858 OpenMPMapClauseKind MapType; 8859 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8860 bool IsImplicit; 8861 const ValueDecl *Mapper; 8862 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper) = L; 8863 auto It = OverlappedData.find(&L); 8864 if (It == OverlappedData.end()) 8865 generateInfoForComponentList(MapType, MapModifiers, llvm::None, 8866 Components, CombinedInfo, PartialStruct, 8867 IsFirstComponentList, IsImplicit, Mapper); 8868 IsFirstComponentList = false; 8869 } 8870 } 8871 8872 /// Generate the default map information for a given capture \a CI, 8873 /// record field declaration \a RI and captured value \a CV. 8874 void generateDefaultMapInfo(const CapturedStmt::Capture &CI, 8875 const FieldDecl &RI, llvm::Value *CV, 8876 MapCombinedInfoTy &CombinedInfo) const { 8877 bool IsImplicit = true; 8878 // Do the default mapping. 8879 if (CI.capturesThis()) { 8880 CombinedInfo.BasePointers.push_back(CV); 8881 CombinedInfo.Pointers.push_back(CV); 8882 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr()); 8883 CombinedInfo.Sizes.push_back( 8884 CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()), 8885 CGF.Int64Ty, /*isSigned=*/true)); 8886 // Default map type. 8887 CombinedInfo.Types.push_back(OMP_MAP_TO | OMP_MAP_FROM); 8888 } else if (CI.capturesVariableByCopy()) { 8889 CombinedInfo.BasePointers.push_back(CV); 8890 CombinedInfo.Pointers.push_back(CV); 8891 if (!RI.getType()->isAnyPointerType()) { 8892 // We have to signal to the runtime captures passed by value that are 8893 // not pointers. 8894 CombinedInfo.Types.push_back(OMP_MAP_LITERAL); 8895 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 8896 CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true)); 8897 } else { 8898 // Pointers are implicitly mapped with a zero size and no flags 8899 // (other than first map that is added for all implicit maps). 8900 CombinedInfo.Types.push_back(OMP_MAP_NONE); 8901 CombinedInfo.Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8902 } 8903 const VarDecl *VD = CI.getCapturedVar(); 8904 auto I = FirstPrivateDecls.find(VD); 8905 if (I != FirstPrivateDecls.end()) 8906 IsImplicit = I->getSecond(); 8907 } else { 8908 assert(CI.capturesVariable() && "Expected captured reference."); 8909 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr()); 8910 QualType ElementType = PtrTy->getPointeeType(); 8911 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 8912 CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true)); 8913 // The default map type for a scalar/complex type is 'to' because by 8914 // default the value doesn't have to be retrieved. For an aggregate 8915 // type, the default is 'tofrom'. 8916 CombinedInfo.Types.push_back(getMapModifiersForPrivateClauses(CI)); 8917 const VarDecl *VD = CI.getCapturedVar(); 8918 auto I = FirstPrivateDecls.find(VD); 8919 if (I != FirstPrivateDecls.end() && 8920 VD->getType().isConstant(CGF.getContext())) { 8921 llvm::Constant *Addr = 8922 CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD); 8923 // Copy the value of the original variable to the new global copy. 8924 CGF.Builder.CreateMemCpy( 8925 CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(CGF), 8926 Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)), 8927 CombinedInfo.Sizes.back(), /*IsVolatile=*/false); 8928 // Use new global variable as the base pointers. 8929 CombinedInfo.BasePointers.push_back(Addr); 8930 CombinedInfo.Pointers.push_back(Addr); 8931 } else { 8932 CombinedInfo.BasePointers.push_back(CV); 8933 if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) { 8934 Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue( 8935 CV, ElementType, CGF.getContext().getDeclAlign(VD), 8936 AlignmentSource::Decl)); 8937 CombinedInfo.Pointers.push_back(PtrAddr.getPointer()); 8938 } else { 8939 CombinedInfo.Pointers.push_back(CV); 8940 } 8941 } 8942 if (I != FirstPrivateDecls.end()) 8943 IsImplicit = I->getSecond(); 8944 } 8945 // Every default map produces a single argument which is a target parameter. 8946 CombinedInfo.Types.back() |= OMP_MAP_TARGET_PARAM; 8947 8948 // Add flag stating this is an implicit map. 8949 if (IsImplicit) 8950 CombinedInfo.Types.back() |= OMP_MAP_IMPLICIT; 8951 8952 // No user-defined mapper for default mapping. 8953 CombinedInfo.Mappers.push_back(nullptr); 8954 } 8955 }; 8956 } // anonymous namespace 8957 8958 static void emitNonContiguousDescriptor( 8959 CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo, 8960 CGOpenMPRuntime::TargetDataInfo &Info) { 8961 CodeGenModule &CGM = CGF.CGM; 8962 MappableExprsHandler::MapCombinedInfoTy::StructNonContiguousInfo 8963 &NonContigInfo = CombinedInfo.NonContigInfo; 8964 8965 // Build an array of struct descriptor_dim and then assign it to 8966 // offload_args. 8967 // 8968 // struct descriptor_dim { 8969 // uint64_t offset; 8970 // uint64_t count; 8971 // uint64_t stride 8972 // }; 8973 ASTContext &C = CGF.getContext(); 8974 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 8975 RecordDecl *RD; 8976 RD = C.buildImplicitRecord("descriptor_dim"); 8977 RD->startDefinition(); 8978 addFieldToRecordDecl(C, RD, Int64Ty); 8979 addFieldToRecordDecl(C, RD, Int64Ty); 8980 addFieldToRecordDecl(C, RD, Int64Ty); 8981 RD->completeDefinition(); 8982 QualType DimTy = C.getRecordType(RD); 8983 8984 enum { OffsetFD = 0, CountFD, StrideFD }; 8985 // We need two index variable here since the size of "Dims" is the same as the 8986 // size of Components, however, the size of offset, count, and stride is equal 8987 // to the size of base declaration that is non-contiguous. 8988 for (unsigned I = 0, L = 0, E = NonContigInfo.Dims.size(); I < E; ++I) { 8989 // Skip emitting ir if dimension size is 1 since it cannot be 8990 // non-contiguous. 8991 if (NonContigInfo.Dims[I] == 1) 8992 continue; 8993 llvm::APInt Size(/*numBits=*/32, NonContigInfo.Dims[I]); 8994 QualType ArrayTy = 8995 C.getConstantArrayType(DimTy, Size, nullptr, ArrayType::Normal, 0); 8996 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 8997 for (unsigned II = 0, EE = NonContigInfo.Dims[I]; II < EE; ++II) { 8998 unsigned RevIdx = EE - II - 1; 8999 LValue DimsLVal = CGF.MakeAddrLValue( 9000 CGF.Builder.CreateConstArrayGEP(DimsAddr, II), DimTy); 9001 // Offset 9002 LValue OffsetLVal = CGF.EmitLValueForField( 9003 DimsLVal, *std::next(RD->field_begin(), OffsetFD)); 9004 CGF.EmitStoreOfScalar(NonContigInfo.Offsets[L][RevIdx], OffsetLVal); 9005 // Count 9006 LValue CountLVal = CGF.EmitLValueForField( 9007 DimsLVal, *std::next(RD->field_begin(), CountFD)); 9008 CGF.EmitStoreOfScalar(NonContigInfo.Counts[L][RevIdx], CountLVal); 9009 // Stride 9010 LValue StrideLVal = CGF.EmitLValueForField( 9011 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 9012 CGF.EmitStoreOfScalar(NonContigInfo.Strides[L][RevIdx], StrideLVal); 9013 } 9014 // args[I] = &dims 9015 Address DAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9016 DimsAddr, CGM.Int8PtrTy); 9017 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 9018 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9019 Info.PointersArray, 0, I); 9020 Address PAddr(P, CGF.getPointerAlign()); 9021 CGF.Builder.CreateStore(DAddr.getPointer(), PAddr); 9022 ++L; 9023 } 9024 } 9025 9026 /// Emit the arrays used to pass the captures and map information to the 9027 /// offloading runtime library. If there is no map or capture information, 9028 /// return nullptr by reference. 9029 static void 9030 emitOffloadingArrays(CodeGenFunction &CGF, 9031 MappableExprsHandler::MapCombinedInfoTy &CombinedInfo, 9032 CGOpenMPRuntime::TargetDataInfo &Info, 9033 bool IsNonContiguous = false) { 9034 CodeGenModule &CGM = CGF.CGM; 9035 ASTContext &Ctx = CGF.getContext(); 9036 9037 // Reset the array information. 9038 Info.clearArrayInfo(); 9039 Info.NumberOfPtrs = CombinedInfo.BasePointers.size(); 9040 9041 if (Info.NumberOfPtrs) { 9042 // Detect if we have any capture size requiring runtime evaluation of the 9043 // size so that a constant array could be eventually used. 9044 bool hasRuntimeEvaluationCaptureSize = false; 9045 for (llvm::Value *S : CombinedInfo.Sizes) 9046 if (!isa<llvm::Constant>(S)) { 9047 hasRuntimeEvaluationCaptureSize = true; 9048 break; 9049 } 9050 9051 llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true); 9052 QualType PointerArrayType = Ctx.getConstantArrayType( 9053 Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal, 9054 /*IndexTypeQuals=*/0); 9055 9056 Info.BasePointersArray = 9057 CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer(); 9058 Info.PointersArray = 9059 CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer(); 9060 Address MappersArray = 9061 CGF.CreateMemTemp(PointerArrayType, ".offload_mappers"); 9062 Info.MappersArray = MappersArray.getPointer(); 9063 9064 // If we don't have any VLA types or other types that require runtime 9065 // evaluation, we can use a constant array for the map sizes, otherwise we 9066 // need to fill up the arrays as we do for the pointers. 9067 QualType Int64Ty = 9068 Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 9069 if (hasRuntimeEvaluationCaptureSize) { 9070 QualType SizeArrayType = Ctx.getConstantArrayType( 9071 Int64Ty, PointerNumAP, nullptr, ArrayType::Normal, 9072 /*IndexTypeQuals=*/0); 9073 Info.SizesArray = 9074 CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer(); 9075 } else { 9076 // We expect all the sizes to be constant, so we collect them to create 9077 // a constant array. 9078 SmallVector<llvm::Constant *, 16> ConstSizes; 9079 for (unsigned I = 0, E = CombinedInfo.Sizes.size(); I < E; ++I) { 9080 if (IsNonContiguous && 9081 (CombinedInfo.Types[I] & MappableExprsHandler::OMP_MAP_NON_CONTIG)) { 9082 ConstSizes.push_back(llvm::ConstantInt::get( 9083 CGF.Int64Ty, CombinedInfo.NonContigInfo.Dims[I])); 9084 } else { 9085 ConstSizes.push_back(cast<llvm::Constant>(CombinedInfo.Sizes[I])); 9086 } 9087 } 9088 9089 auto *SizesArrayInit = llvm::ConstantArray::get( 9090 llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes); 9091 std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"}); 9092 auto *SizesArrayGbl = new llvm::GlobalVariable( 9093 CGM.getModule(), SizesArrayInit->getType(), 9094 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 9095 SizesArrayInit, Name); 9096 SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 9097 Info.SizesArray = SizesArrayGbl; 9098 } 9099 9100 // The map types are always constant so we don't need to generate code to 9101 // fill arrays. Instead, we create an array constant. 9102 SmallVector<uint64_t, 4> Mapping(CombinedInfo.Types.size(), 0); 9103 llvm::copy(CombinedInfo.Types, Mapping.begin()); 9104 llvm::Constant *MapTypesArrayInit = 9105 llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping); 9106 std::string MaptypesName = 9107 CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 9108 auto *MapTypesArrayGbl = new llvm::GlobalVariable( 9109 CGM.getModule(), MapTypesArrayInit->getType(), 9110 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 9111 MapTypesArrayInit, MaptypesName); 9112 MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 9113 Info.MapTypesArray = MapTypesArrayGbl; 9114 9115 // If there's a present map type modifier, it must not be applied to the end 9116 // of a region, so generate a separate map type array in that case. 9117 if (Info.separateBeginEndCalls()) { 9118 bool EndMapTypesDiffer = false; 9119 for (uint64_t &Type : Mapping) { 9120 if (Type & MappableExprsHandler::OMP_MAP_PRESENT) { 9121 Type &= ~MappableExprsHandler::OMP_MAP_PRESENT; 9122 EndMapTypesDiffer = true; 9123 } 9124 } 9125 if (EndMapTypesDiffer) { 9126 MapTypesArrayInit = 9127 llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping); 9128 MaptypesName = CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 9129 MapTypesArrayGbl = new llvm::GlobalVariable( 9130 CGM.getModule(), MapTypesArrayInit->getType(), 9131 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 9132 MapTypesArrayInit, MaptypesName); 9133 MapTypesArrayGbl->setUnnamedAddr( 9134 llvm::GlobalValue::UnnamedAddr::Global); 9135 Info.MapTypesArrayEnd = MapTypesArrayGbl; 9136 } 9137 } 9138 9139 for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) { 9140 llvm::Value *BPVal = *CombinedInfo.BasePointers[I]; 9141 llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32( 9142 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9143 Info.BasePointersArray, 0, I); 9144 BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9145 BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0)); 9146 Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 9147 CGF.Builder.CreateStore(BPVal, BPAddr); 9148 9149 if (Info.requiresDevicePointerInfo()) 9150 if (const ValueDecl *DevVD = 9151 CombinedInfo.BasePointers[I].getDevicePtrDecl()) 9152 Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr); 9153 9154 llvm::Value *PVal = CombinedInfo.Pointers[I]; 9155 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 9156 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9157 Info.PointersArray, 0, I); 9158 P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9159 P, PVal->getType()->getPointerTo(/*AddrSpace=*/0)); 9160 Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 9161 CGF.Builder.CreateStore(PVal, PAddr); 9162 9163 if (hasRuntimeEvaluationCaptureSize) { 9164 llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32( 9165 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 9166 Info.SizesArray, 9167 /*Idx0=*/0, 9168 /*Idx1=*/I); 9169 Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty)); 9170 CGF.Builder.CreateStore(CGF.Builder.CreateIntCast(CombinedInfo.Sizes[I], 9171 CGM.Int64Ty, 9172 /*isSigned=*/true), 9173 SAddr); 9174 } 9175 9176 // Fill up the mapper array. 9177 llvm::Value *MFunc = llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 9178 if (CombinedInfo.Mappers[I]) { 9179 MFunc = CGM.getOpenMPRuntime().getOrCreateUserDefinedMapperFunc( 9180 cast<OMPDeclareMapperDecl>(CombinedInfo.Mappers[I])); 9181 MFunc = CGF.Builder.CreatePointerCast(MFunc, CGM.VoidPtrTy); 9182 Info.HasMapper = true; 9183 } 9184 Address MAddr = CGF.Builder.CreateConstArrayGEP(MappersArray, I); 9185 CGF.Builder.CreateStore(MFunc, MAddr); 9186 } 9187 } 9188 9189 if (!IsNonContiguous || CombinedInfo.NonContigInfo.Offsets.empty() || 9190 Info.NumberOfPtrs == 0) 9191 return; 9192 9193 emitNonContiguousDescriptor(CGF, CombinedInfo, Info); 9194 } 9195 9196 namespace { 9197 /// Additional arguments for emitOffloadingArraysArgument function. 9198 struct ArgumentsOptions { 9199 bool ForEndCall = false; 9200 ArgumentsOptions() = default; 9201 ArgumentsOptions(bool ForEndCall) : ForEndCall(ForEndCall) {} 9202 }; 9203 } // namespace 9204 9205 /// Emit the arguments to be passed to the runtime library based on the 9206 /// arrays of base pointers, pointers, sizes, map types, and mappers. If 9207 /// ForEndCall, emit map types to be passed for the end of the region instead of 9208 /// the beginning. 9209 static void emitOffloadingArraysArgument( 9210 CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg, 9211 llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg, 9212 llvm::Value *&MapTypesArrayArg, llvm::Value *&MappersArrayArg, 9213 CGOpenMPRuntime::TargetDataInfo &Info, 9214 const ArgumentsOptions &Options = ArgumentsOptions()) { 9215 assert((!Options.ForEndCall || Info.separateBeginEndCalls()) && 9216 "expected region end call to runtime only when end call is separate"); 9217 CodeGenModule &CGM = CGF.CGM; 9218 if (Info.NumberOfPtrs) { 9219 BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9220 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9221 Info.BasePointersArray, 9222 /*Idx0=*/0, /*Idx1=*/0); 9223 PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9224 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9225 Info.PointersArray, 9226 /*Idx0=*/0, 9227 /*Idx1=*/0); 9228 SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9229 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray, 9230 /*Idx0=*/0, /*Idx1=*/0); 9231 MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9232 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 9233 Options.ForEndCall && Info.MapTypesArrayEnd ? Info.MapTypesArrayEnd 9234 : Info.MapTypesArray, 9235 /*Idx0=*/0, 9236 /*Idx1=*/0); 9237 // If there is no user-defined mapper, set the mapper array to nullptr to 9238 // avoid an unnecessary data privatization 9239 if (!Info.HasMapper) 9240 MappersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9241 else 9242 MappersArrayArg = 9243 CGF.Builder.CreatePointerCast(Info.MappersArray, CGM.VoidPtrPtrTy); 9244 } else { 9245 BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9246 PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9247 SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 9248 MapTypesArrayArg = 9249 llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 9250 MappersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9251 } 9252 } 9253 9254 /// Check for inner distribute directive. 9255 static const OMPExecutableDirective * 9256 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) { 9257 const auto *CS = D.getInnermostCapturedStmt(); 9258 const auto *Body = 9259 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 9260 const Stmt *ChildStmt = 9261 CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 9262 9263 if (const auto *NestedDir = 9264 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 9265 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind(); 9266 switch (D.getDirectiveKind()) { 9267 case OMPD_target: 9268 if (isOpenMPDistributeDirective(DKind)) 9269 return NestedDir; 9270 if (DKind == OMPD_teams) { 9271 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers( 9272 /*IgnoreCaptured=*/true); 9273 if (!Body) 9274 return nullptr; 9275 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 9276 if (const auto *NND = 9277 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 9278 DKind = NND->getDirectiveKind(); 9279 if (isOpenMPDistributeDirective(DKind)) 9280 return NND; 9281 } 9282 } 9283 return nullptr; 9284 case OMPD_target_teams: 9285 if (isOpenMPDistributeDirective(DKind)) 9286 return NestedDir; 9287 return nullptr; 9288 case OMPD_target_parallel: 9289 case OMPD_target_simd: 9290 case OMPD_target_parallel_for: 9291 case OMPD_target_parallel_for_simd: 9292 return nullptr; 9293 case OMPD_target_teams_distribute: 9294 case OMPD_target_teams_distribute_simd: 9295 case OMPD_target_teams_distribute_parallel_for: 9296 case OMPD_target_teams_distribute_parallel_for_simd: 9297 case OMPD_parallel: 9298 case OMPD_for: 9299 case OMPD_parallel_for: 9300 case OMPD_parallel_master: 9301 case OMPD_parallel_sections: 9302 case OMPD_for_simd: 9303 case OMPD_parallel_for_simd: 9304 case OMPD_cancel: 9305 case OMPD_cancellation_point: 9306 case OMPD_ordered: 9307 case OMPD_threadprivate: 9308 case OMPD_allocate: 9309 case OMPD_task: 9310 case OMPD_simd: 9311 case OMPD_sections: 9312 case OMPD_section: 9313 case OMPD_single: 9314 case OMPD_master: 9315 case OMPD_critical: 9316 case OMPD_taskyield: 9317 case OMPD_barrier: 9318 case OMPD_taskwait: 9319 case OMPD_taskgroup: 9320 case OMPD_atomic: 9321 case OMPD_flush: 9322 case OMPD_depobj: 9323 case OMPD_scan: 9324 case OMPD_teams: 9325 case OMPD_target_data: 9326 case OMPD_target_exit_data: 9327 case OMPD_target_enter_data: 9328 case OMPD_distribute: 9329 case OMPD_distribute_simd: 9330 case OMPD_distribute_parallel_for: 9331 case OMPD_distribute_parallel_for_simd: 9332 case OMPD_teams_distribute: 9333 case OMPD_teams_distribute_simd: 9334 case OMPD_teams_distribute_parallel_for: 9335 case OMPD_teams_distribute_parallel_for_simd: 9336 case OMPD_target_update: 9337 case OMPD_declare_simd: 9338 case OMPD_declare_variant: 9339 case OMPD_begin_declare_variant: 9340 case OMPD_end_declare_variant: 9341 case OMPD_declare_target: 9342 case OMPD_end_declare_target: 9343 case OMPD_declare_reduction: 9344 case OMPD_declare_mapper: 9345 case OMPD_taskloop: 9346 case OMPD_taskloop_simd: 9347 case OMPD_master_taskloop: 9348 case OMPD_master_taskloop_simd: 9349 case OMPD_parallel_master_taskloop: 9350 case OMPD_parallel_master_taskloop_simd: 9351 case OMPD_requires: 9352 case OMPD_unknown: 9353 default: 9354 llvm_unreachable("Unexpected directive."); 9355 } 9356 } 9357 9358 return nullptr; 9359 } 9360 9361 /// Emit the user-defined mapper function. The code generation follows the 9362 /// pattern in the example below. 9363 /// \code 9364 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle, 9365 /// void *base, void *begin, 9366 /// int64_t size, int64_t type) { 9367 /// // Allocate space for an array section first. 9368 /// if (size > 1 && !maptype.IsDelete) 9369 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9370 /// size*sizeof(Ty), clearToFrom(type)); 9371 /// // Map members. 9372 /// for (unsigned i = 0; i < size; i++) { 9373 /// // For each component specified by this mapper: 9374 /// for (auto c : all_components) { 9375 /// if (c.hasMapper()) 9376 /// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size, 9377 /// c.arg_type); 9378 /// else 9379 /// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base, 9380 /// c.arg_begin, c.arg_size, c.arg_type); 9381 /// } 9382 /// } 9383 /// // Delete the array section. 9384 /// if (size > 1 && maptype.IsDelete) 9385 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9386 /// size*sizeof(Ty), clearToFrom(type)); 9387 /// } 9388 /// \endcode 9389 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D, 9390 CodeGenFunction *CGF) { 9391 if (UDMMap.count(D) > 0) 9392 return; 9393 ASTContext &C = CGM.getContext(); 9394 QualType Ty = D->getType(); 9395 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 9396 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 9397 auto *MapperVarDecl = 9398 cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl()); 9399 SourceLocation Loc = D->getLocation(); 9400 CharUnits ElementSize = C.getTypeSizeInChars(Ty); 9401 9402 // Prepare mapper function arguments and attributes. 9403 ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9404 C.VoidPtrTy, ImplicitParamDecl::Other); 9405 ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 9406 ImplicitParamDecl::Other); 9407 ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9408 C.VoidPtrTy, ImplicitParamDecl::Other); 9409 ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 9410 ImplicitParamDecl::Other); 9411 ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 9412 ImplicitParamDecl::Other); 9413 FunctionArgList Args; 9414 Args.push_back(&HandleArg); 9415 Args.push_back(&BaseArg); 9416 Args.push_back(&BeginArg); 9417 Args.push_back(&SizeArg); 9418 Args.push_back(&TypeArg); 9419 const CGFunctionInfo &FnInfo = 9420 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 9421 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 9422 SmallString<64> TyStr; 9423 llvm::raw_svector_ostream Out(TyStr); 9424 CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out); 9425 std::string Name = getName({"omp_mapper", TyStr, D->getName()}); 9426 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 9427 Name, &CGM.getModule()); 9428 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 9429 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 9430 // Start the mapper function code generation. 9431 CodeGenFunction MapperCGF(CGM); 9432 MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 9433 // Compute the starting and end addreses of array elements. 9434 llvm::Value *Size = MapperCGF.EmitLoadOfScalar( 9435 MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false, 9436 C.getPointerType(Int64Ty), Loc); 9437 // Convert the size in bytes into the number of array elements. 9438 Size = MapperCGF.Builder.CreateExactUDiv( 9439 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 9440 llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast( 9441 MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(), 9442 CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy))); 9443 llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size); 9444 llvm::Value *MapType = MapperCGF.EmitLoadOfScalar( 9445 MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false, 9446 C.getPointerType(Int64Ty), Loc); 9447 // Prepare common arguments for array initiation and deletion. 9448 llvm::Value *Handle = MapperCGF.EmitLoadOfScalar( 9449 MapperCGF.GetAddrOfLocalVar(&HandleArg), 9450 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9451 llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar( 9452 MapperCGF.GetAddrOfLocalVar(&BaseArg), 9453 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9454 llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar( 9455 MapperCGF.GetAddrOfLocalVar(&BeginArg), 9456 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9457 9458 // Emit array initiation if this is an array section and \p MapType indicates 9459 // that memory allocation is required. 9460 llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head"); 9461 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9462 ElementSize, HeadBB, /*IsInit=*/true); 9463 9464 // Emit a for loop to iterate through SizeArg of elements and map all of them. 9465 9466 // Emit the loop header block. 9467 MapperCGF.EmitBlock(HeadBB); 9468 llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body"); 9469 llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done"); 9470 // Evaluate whether the initial condition is satisfied. 9471 llvm::Value *IsEmpty = 9472 MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty"); 9473 MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 9474 llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock(); 9475 9476 // Emit the loop body block. 9477 MapperCGF.EmitBlock(BodyBB); 9478 llvm::BasicBlock *LastBB = BodyBB; 9479 llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI( 9480 PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent"); 9481 PtrPHI->addIncoming(PtrBegin, EntryBB); 9482 Address PtrCurrent = 9483 Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg) 9484 .getAlignment() 9485 .alignmentOfArrayElement(ElementSize)); 9486 // Privatize the declared variable of mapper to be the current array element. 9487 CodeGenFunction::OMPPrivateScope Scope(MapperCGF); 9488 Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() { 9489 return MapperCGF 9490 .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>()) 9491 .getAddress(MapperCGF); 9492 }); 9493 (void)Scope.Privatize(); 9494 9495 // Get map clause information. Fill up the arrays with all mapped variables. 9496 MappableExprsHandler::MapCombinedInfoTy Info; 9497 MappableExprsHandler MEHandler(*D, MapperCGF); 9498 MEHandler.generateAllInfoForMapper(Info); 9499 9500 // Call the runtime API __tgt_mapper_num_components to get the number of 9501 // pre-existing components. 9502 llvm::Value *OffloadingArgs[] = {Handle}; 9503 llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall( 9504 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 9505 OMPRTL___tgt_mapper_num_components), 9506 OffloadingArgs); 9507 llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl( 9508 PreviousSize, 9509 MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset())); 9510 9511 // Fill up the runtime mapper handle for all components. 9512 for (unsigned I = 0; I < Info.BasePointers.size(); ++I) { 9513 llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast( 9514 *Info.BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9515 llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast( 9516 Info.Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9517 llvm::Value *CurSizeArg = Info.Sizes[I]; 9518 9519 // Extract the MEMBER_OF field from the map type. 9520 llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member"); 9521 MapperCGF.EmitBlock(MemberBB); 9522 llvm::Value *OriMapType = MapperCGF.Builder.getInt64(Info.Types[I]); 9523 llvm::Value *Member = MapperCGF.Builder.CreateAnd( 9524 OriMapType, 9525 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF)); 9526 llvm::BasicBlock *MemberCombineBB = 9527 MapperCGF.createBasicBlock("omp.member.combine"); 9528 llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type"); 9529 llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member); 9530 MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB); 9531 // Add the number of pre-existing components to the MEMBER_OF field if it 9532 // is valid. 9533 MapperCGF.EmitBlock(MemberCombineBB); 9534 llvm::Value *CombinedMember = 9535 MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize); 9536 // Do nothing if it is not a member of previous components. 9537 MapperCGF.EmitBlock(TypeBB); 9538 llvm::PHINode *MemberMapType = 9539 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype"); 9540 MemberMapType->addIncoming(OriMapType, MemberBB); 9541 MemberMapType->addIncoming(CombinedMember, MemberCombineBB); 9542 9543 // Combine the map type inherited from user-defined mapper with that 9544 // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM 9545 // bits of the \a MapType, which is the input argument of the mapper 9546 // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM 9547 // bits of MemberMapType. 9548 // [OpenMP 5.0], 1.2.6. map-type decay. 9549 // | alloc | to | from | tofrom | release | delete 9550 // ---------------------------------------------------------- 9551 // alloc | alloc | alloc | alloc | alloc | release | delete 9552 // to | alloc | to | alloc | to | release | delete 9553 // from | alloc | alloc | from | from | release | delete 9554 // tofrom | alloc | to | from | tofrom | release | delete 9555 llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd( 9556 MapType, 9557 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO | 9558 MappableExprsHandler::OMP_MAP_FROM)); 9559 llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc"); 9560 llvm::BasicBlock *AllocElseBB = 9561 MapperCGF.createBasicBlock("omp.type.alloc.else"); 9562 llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to"); 9563 llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else"); 9564 llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from"); 9565 llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end"); 9566 llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom); 9567 MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB); 9568 // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM. 9569 MapperCGF.EmitBlock(AllocBB); 9570 llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd( 9571 MemberMapType, 9572 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9573 MappableExprsHandler::OMP_MAP_FROM))); 9574 MapperCGF.Builder.CreateBr(EndBB); 9575 MapperCGF.EmitBlock(AllocElseBB); 9576 llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ( 9577 LeftToFrom, 9578 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO)); 9579 MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB); 9580 // In case of to, clear OMP_MAP_FROM. 9581 MapperCGF.EmitBlock(ToBB); 9582 llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd( 9583 MemberMapType, 9584 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM)); 9585 MapperCGF.Builder.CreateBr(EndBB); 9586 MapperCGF.EmitBlock(ToElseBB); 9587 llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ( 9588 LeftToFrom, 9589 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM)); 9590 MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB); 9591 // In case of from, clear OMP_MAP_TO. 9592 MapperCGF.EmitBlock(FromBB); 9593 llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd( 9594 MemberMapType, 9595 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO)); 9596 // In case of tofrom, do nothing. 9597 MapperCGF.EmitBlock(EndBB); 9598 LastBB = EndBB; 9599 llvm::PHINode *CurMapType = 9600 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype"); 9601 CurMapType->addIncoming(AllocMapType, AllocBB); 9602 CurMapType->addIncoming(ToMapType, ToBB); 9603 CurMapType->addIncoming(FromMapType, FromBB); 9604 CurMapType->addIncoming(MemberMapType, ToElseBB); 9605 9606 llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg, 9607 CurSizeArg, CurMapType}; 9608 if (Info.Mappers[I]) { 9609 // Call the corresponding mapper function. 9610 llvm::Function *MapperFunc = getOrCreateUserDefinedMapperFunc( 9611 cast<OMPDeclareMapperDecl>(Info.Mappers[I])); 9612 assert(MapperFunc && "Expect a valid mapper function is available."); 9613 MapperCGF.EmitNounwindRuntimeCall(MapperFunc, OffloadingArgs); 9614 } else { 9615 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9616 // data structure. 9617 MapperCGF.EmitRuntimeCall( 9618 OMPBuilder.getOrCreateRuntimeFunction( 9619 CGM.getModule(), OMPRTL___tgt_push_mapper_component), 9620 OffloadingArgs); 9621 } 9622 } 9623 9624 // Update the pointer to point to the next element that needs to be mapped, 9625 // and check whether we have mapped all elements. 9626 llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32( 9627 PtrPHI, /*Idx0=*/1, "omp.arraymap.next"); 9628 PtrPHI->addIncoming(PtrNext, LastBB); 9629 llvm::Value *IsDone = 9630 MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone"); 9631 llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit"); 9632 MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB); 9633 9634 MapperCGF.EmitBlock(ExitBB); 9635 // Emit array deletion if this is an array section and \p MapType indicates 9636 // that deletion is required. 9637 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9638 ElementSize, DoneBB, /*IsInit=*/false); 9639 9640 // Emit the function exit block. 9641 MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true); 9642 MapperCGF.FinishFunction(); 9643 UDMMap.try_emplace(D, Fn); 9644 if (CGF) { 9645 auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn); 9646 Decls.second.push_back(D); 9647 } 9648 } 9649 9650 /// Emit the array initialization or deletion portion for user-defined mapper 9651 /// code generation. First, it evaluates whether an array section is mapped and 9652 /// whether the \a MapType instructs to delete this section. If \a IsInit is 9653 /// true, and \a MapType indicates to not delete this array, array 9654 /// initialization code is generated. If \a IsInit is false, and \a MapType 9655 /// indicates to not this array, array deletion code is generated. 9656 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel( 9657 CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base, 9658 llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType, 9659 CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) { 9660 StringRef Prefix = IsInit ? ".init" : ".del"; 9661 9662 // Evaluate if this is an array section. 9663 llvm::BasicBlock *IsDeleteBB = 9664 MapperCGF.createBasicBlock(getName({"omp.array", Prefix, ".evaldelete"})); 9665 llvm::BasicBlock *BodyBB = 9666 MapperCGF.createBasicBlock(getName({"omp.array", Prefix})); 9667 llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE( 9668 Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray"); 9669 MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB); 9670 9671 // Evaluate if we are going to delete this section. 9672 MapperCGF.EmitBlock(IsDeleteBB); 9673 llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd( 9674 MapType, 9675 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE)); 9676 llvm::Value *DeleteCond; 9677 if (IsInit) { 9678 DeleteCond = MapperCGF.Builder.CreateIsNull( 9679 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9680 } else { 9681 DeleteCond = MapperCGF.Builder.CreateIsNotNull( 9682 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9683 } 9684 MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB); 9685 9686 MapperCGF.EmitBlock(BodyBB); 9687 // Get the array size by multiplying element size and element number (i.e., \p 9688 // Size). 9689 llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul( 9690 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 9691 // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves 9692 // memory allocation/deletion purpose only. 9693 llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd( 9694 MapType, 9695 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9696 MappableExprsHandler::OMP_MAP_FROM))); 9697 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9698 // data structure. 9699 llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg}; 9700 MapperCGF.EmitRuntimeCall( 9701 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 9702 OMPRTL___tgt_push_mapper_component), 9703 OffloadingArgs); 9704 } 9705 9706 llvm::Function *CGOpenMPRuntime::getOrCreateUserDefinedMapperFunc( 9707 const OMPDeclareMapperDecl *D) { 9708 auto I = UDMMap.find(D); 9709 if (I != UDMMap.end()) 9710 return I->second; 9711 emitUserDefinedMapper(D); 9712 return UDMMap.lookup(D); 9713 } 9714 9715 void CGOpenMPRuntime::emitTargetNumIterationsCall( 9716 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9717 llvm::Value *DeviceID, 9718 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9719 const OMPLoopDirective &D)> 9720 SizeEmitter) { 9721 OpenMPDirectiveKind Kind = D.getDirectiveKind(); 9722 const OMPExecutableDirective *TD = &D; 9723 // Get nested teams distribute kind directive, if any. 9724 if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) 9725 TD = getNestedDistributeDirective(CGM.getContext(), D); 9726 if (!TD) 9727 return; 9728 const auto *LD = cast<OMPLoopDirective>(TD); 9729 auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF, 9730 PrePostActionTy &) { 9731 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) { 9732 llvm::Value *Args[] = {DeviceID, NumIterations}; 9733 CGF.EmitRuntimeCall( 9734 OMPBuilder.getOrCreateRuntimeFunction( 9735 CGM.getModule(), OMPRTL___kmpc_push_target_tripcount), 9736 Args); 9737 } 9738 }; 9739 emitInlinedDirective(CGF, OMPD_unknown, CodeGen); 9740 } 9741 9742 void CGOpenMPRuntime::emitTargetCall( 9743 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9744 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 9745 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 9746 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9747 const OMPLoopDirective &D)> 9748 SizeEmitter) { 9749 if (!CGF.HaveInsertPoint()) 9750 return; 9751 9752 assert(OutlinedFn && "Invalid outlined function!"); 9753 9754 const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() || 9755 D.hasClausesOfKind<OMPNowaitClause>(); 9756 llvm::SmallVector<llvm::Value *, 16> CapturedVars; 9757 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 9758 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF, 9759 PrePostActionTy &) { 9760 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9761 }; 9762 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen); 9763 9764 CodeGenFunction::OMPTargetDataInfo InputInfo; 9765 llvm::Value *MapTypesArray = nullptr; 9766 // Fill up the pointer arrays and transfer execution to the device. 9767 auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo, 9768 &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars, 9769 SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) { 9770 if (Device.getInt() == OMPC_DEVICE_ancestor) { 9771 // Reverse offloading is not supported, so just execute on the host. 9772 if (RequiresOuterTask) { 9773 CapturedVars.clear(); 9774 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9775 } 9776 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9777 return; 9778 } 9779 9780 // On top of the arrays that were filled up, the target offloading call 9781 // takes as arguments the device id as well as the host pointer. The host 9782 // pointer is used by the runtime library to identify the current target 9783 // region, so it only has to be unique and not necessarily point to 9784 // anything. It could be the pointer to the outlined function that 9785 // implements the target region, but we aren't using that so that the 9786 // compiler doesn't need to keep that, and could therefore inline the host 9787 // function if proven worthwhile during optimization. 9788 9789 // From this point on, we need to have an ID of the target region defined. 9790 assert(OutlinedFnID && "Invalid outlined function ID!"); 9791 9792 // Emit device ID if any. 9793 llvm::Value *DeviceID; 9794 if (Device.getPointer()) { 9795 assert((Device.getInt() == OMPC_DEVICE_unknown || 9796 Device.getInt() == OMPC_DEVICE_device_num) && 9797 "Expected device_num modifier."); 9798 llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer()); 9799 DeviceID = 9800 CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true); 9801 } else { 9802 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 9803 } 9804 9805 // Emit the number of elements in the offloading arrays. 9806 llvm::Value *PointerNum = 9807 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 9808 9809 // Return value of the runtime offloading call. 9810 llvm::Value *Return; 9811 9812 llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D); 9813 llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D); 9814 9815 // Emit tripcount for the target loop-based directive. 9816 emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter); 9817 9818 bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 9819 // The target region is an outlined function launched by the runtime 9820 // via calls __tgt_target() or __tgt_target_teams(). 9821 // 9822 // __tgt_target() launches a target region with one team and one thread, 9823 // executing a serial region. This master thread may in turn launch 9824 // more threads within its team upon encountering a parallel region, 9825 // however, no additional teams can be launched on the device. 9826 // 9827 // __tgt_target_teams() launches a target region with one or more teams, 9828 // each with one or more threads. This call is required for target 9829 // constructs such as: 9830 // 'target teams' 9831 // 'target' / 'teams' 9832 // 'target teams distribute parallel for' 9833 // 'target parallel' 9834 // and so on. 9835 // 9836 // Note that on the host and CPU targets, the runtime implementation of 9837 // these calls simply call the outlined function without forking threads. 9838 // The outlined functions themselves have runtime calls to 9839 // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by 9840 // the compiler in emitTeamsCall() and emitParallelCall(). 9841 // 9842 // In contrast, on the NVPTX target, the implementation of 9843 // __tgt_target_teams() launches a GPU kernel with the requested number 9844 // of teams and threads so no additional calls to the runtime are required. 9845 if (NumTeams) { 9846 // If we have NumTeams defined this means that we have an enclosed teams 9847 // region. Therefore we also expect to have NumThreads defined. These two 9848 // values should be defined in the presence of a teams directive, 9849 // regardless of having any clauses associated. If the user is using teams 9850 // but no clauses, these two values will be the default that should be 9851 // passed to the runtime library - a 32-bit integer with the value zero. 9852 assert(NumThreads && "Thread limit expression should be available along " 9853 "with number of teams."); 9854 llvm::Value *OffloadingArgs[] = {DeviceID, 9855 OutlinedFnID, 9856 PointerNum, 9857 InputInfo.BasePointersArray.getPointer(), 9858 InputInfo.PointersArray.getPointer(), 9859 InputInfo.SizesArray.getPointer(), 9860 MapTypesArray, 9861 InputInfo.MappersArray.getPointer(), 9862 NumTeams, 9863 NumThreads}; 9864 Return = CGF.EmitRuntimeCall( 9865 OMPBuilder.getOrCreateRuntimeFunction( 9866 CGM.getModule(), HasNowait 9867 ? OMPRTL___tgt_target_teams_nowait_mapper 9868 : OMPRTL___tgt_target_teams_mapper), 9869 OffloadingArgs); 9870 } else { 9871 llvm::Value *OffloadingArgs[] = {DeviceID, 9872 OutlinedFnID, 9873 PointerNum, 9874 InputInfo.BasePointersArray.getPointer(), 9875 InputInfo.PointersArray.getPointer(), 9876 InputInfo.SizesArray.getPointer(), 9877 MapTypesArray, 9878 InputInfo.MappersArray.getPointer()}; 9879 Return = CGF.EmitRuntimeCall( 9880 OMPBuilder.getOrCreateRuntimeFunction( 9881 CGM.getModule(), HasNowait ? OMPRTL___tgt_target_nowait_mapper 9882 : OMPRTL___tgt_target_mapper), 9883 OffloadingArgs); 9884 } 9885 9886 // Check the error code and execute the host version if required. 9887 llvm::BasicBlock *OffloadFailedBlock = 9888 CGF.createBasicBlock("omp_offload.failed"); 9889 llvm::BasicBlock *OffloadContBlock = 9890 CGF.createBasicBlock("omp_offload.cont"); 9891 llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return); 9892 CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock); 9893 9894 CGF.EmitBlock(OffloadFailedBlock); 9895 if (RequiresOuterTask) { 9896 CapturedVars.clear(); 9897 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9898 } 9899 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9900 CGF.EmitBranch(OffloadContBlock); 9901 9902 CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true); 9903 }; 9904 9905 // Notify that the host version must be executed. 9906 auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars, 9907 RequiresOuterTask](CodeGenFunction &CGF, 9908 PrePostActionTy &) { 9909 if (RequiresOuterTask) { 9910 CapturedVars.clear(); 9911 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9912 } 9913 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9914 }; 9915 9916 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 9917 &CapturedVars, RequiresOuterTask, 9918 &CS](CodeGenFunction &CGF, PrePostActionTy &) { 9919 // Fill up the arrays with all the captured variables. 9920 MappableExprsHandler::MapCombinedInfoTy CombinedInfo; 9921 9922 // Get mappable expression information. 9923 MappableExprsHandler MEHandler(D, CGF); 9924 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers; 9925 llvm::DenseSet<CanonicalDeclPtr<const Decl>> MappedVarSet; 9926 9927 auto RI = CS.getCapturedRecordDecl()->field_begin(); 9928 auto CV = CapturedVars.begin(); 9929 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(), 9930 CE = CS.capture_end(); 9931 CI != CE; ++CI, ++RI, ++CV) { 9932 MappableExprsHandler::MapCombinedInfoTy CurInfo; 9933 MappableExprsHandler::StructRangeInfoTy PartialStruct; 9934 9935 // VLA sizes are passed to the outlined region by copy and do not have map 9936 // information associated. 9937 if (CI->capturesVariableArrayType()) { 9938 CurInfo.BasePointers.push_back(*CV); 9939 CurInfo.Pointers.push_back(*CV); 9940 CurInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 9941 CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true)); 9942 // Copy to the device as an argument. No need to retrieve it. 9943 CurInfo.Types.push_back(MappableExprsHandler::OMP_MAP_LITERAL | 9944 MappableExprsHandler::OMP_MAP_TARGET_PARAM | 9945 MappableExprsHandler::OMP_MAP_IMPLICIT); 9946 CurInfo.Mappers.push_back(nullptr); 9947 } else { 9948 // If we have any information in the map clause, we use it, otherwise we 9949 // just do a default mapping. 9950 MEHandler.generateInfoForCapture(CI, *CV, CurInfo, PartialStruct); 9951 if (!CI->capturesThis()) 9952 MappedVarSet.insert(CI->getCapturedVar()); 9953 else 9954 MappedVarSet.insert(nullptr); 9955 if (CurInfo.BasePointers.empty() && !PartialStruct.Base.isValid()) 9956 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurInfo); 9957 // Generate correct mapping for variables captured by reference in 9958 // lambdas. 9959 if (CI->capturesVariable()) 9960 MEHandler.generateInfoForLambdaCaptures(CI->getCapturedVar(), *CV, 9961 CurInfo, LambdaPointers); 9962 } 9963 // We expect to have at least an element of information for this capture. 9964 assert((!CurInfo.BasePointers.empty() || PartialStruct.Base.isValid()) && 9965 "Non-existing map pointer for capture!"); 9966 assert(CurInfo.BasePointers.size() == CurInfo.Pointers.size() && 9967 CurInfo.BasePointers.size() == CurInfo.Sizes.size() && 9968 CurInfo.BasePointers.size() == CurInfo.Types.size() && 9969 CurInfo.BasePointers.size() == CurInfo.Mappers.size() && 9970 "Inconsistent map information sizes!"); 9971 9972 // If there is an entry in PartialStruct it means we have a struct with 9973 // individual members mapped. Emit an extra combined entry. 9974 if (PartialStruct.Base.isValid()) 9975 MEHandler.emitCombinedEntry(CombinedInfo, CurInfo.Types, PartialStruct); 9976 9977 // We need to append the results of this capture to what we already have. 9978 CombinedInfo.append(CurInfo); 9979 } 9980 // Adjust MEMBER_OF flags for the lambdas captures. 9981 MEHandler.adjustMemberOfForLambdaCaptures( 9982 LambdaPointers, CombinedInfo.BasePointers, CombinedInfo.Pointers, 9983 CombinedInfo.Types); 9984 // Map any list items in a map clause that were not captures because they 9985 // weren't referenced within the construct. 9986 MEHandler.generateAllInfo(CombinedInfo, /*NotTargetParams=*/true, 9987 MappedVarSet); 9988 9989 TargetDataInfo Info; 9990 // Fill up the arrays and create the arguments. 9991 emitOffloadingArrays(CGF, CombinedInfo, Info); 9992 emitOffloadingArraysArgument( 9993 CGF, Info.BasePointersArray, Info.PointersArray, Info.SizesArray, 9994 Info.MapTypesArray, Info.MappersArray, Info, {/*ForEndTask=*/false}); 9995 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 9996 InputInfo.BasePointersArray = 9997 Address(Info.BasePointersArray, CGM.getPointerAlign()); 9998 InputInfo.PointersArray = 9999 Address(Info.PointersArray, CGM.getPointerAlign()); 10000 InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign()); 10001 InputInfo.MappersArray = Address(Info.MappersArray, CGM.getPointerAlign()); 10002 MapTypesArray = Info.MapTypesArray; 10003 if (RequiresOuterTask) 10004 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 10005 else 10006 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 10007 }; 10008 10009 auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask]( 10010 CodeGenFunction &CGF, PrePostActionTy &) { 10011 if (RequiresOuterTask) { 10012 CodeGenFunction::OMPTargetDataInfo InputInfo; 10013 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo); 10014 } else { 10015 emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen); 10016 } 10017 }; 10018 10019 // If we have a target function ID it means that we need to support 10020 // offloading, otherwise, just execute on the host. We need to execute on host 10021 // regardless of the conditional in the if clause if, e.g., the user do not 10022 // specify target triples. 10023 if (OutlinedFnID) { 10024 if (IfCond) { 10025 emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen); 10026 } else { 10027 RegionCodeGenTy ThenRCG(TargetThenGen); 10028 ThenRCG(CGF); 10029 } 10030 } else { 10031 RegionCodeGenTy ElseRCG(TargetElseGen); 10032 ElseRCG(CGF); 10033 } 10034 } 10035 10036 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S, 10037 StringRef ParentName) { 10038 if (!S) 10039 return; 10040 10041 // Codegen OMP target directives that offload compute to the device. 10042 bool RequiresDeviceCodegen = 10043 isa<OMPExecutableDirective>(S) && 10044 isOpenMPTargetExecutionDirective( 10045 cast<OMPExecutableDirective>(S)->getDirectiveKind()); 10046 10047 if (RequiresDeviceCodegen) { 10048 const auto &E = *cast<OMPExecutableDirective>(S); 10049 unsigned DeviceID; 10050 unsigned FileID; 10051 unsigned Line; 10052 getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID, 10053 FileID, Line); 10054 10055 // Is this a target region that should not be emitted as an entry point? If 10056 // so just signal we are done with this target region. 10057 if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID, 10058 ParentName, Line)) 10059 return; 10060 10061 switch (E.getDirectiveKind()) { 10062 case OMPD_target: 10063 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName, 10064 cast<OMPTargetDirective>(E)); 10065 break; 10066 case OMPD_target_parallel: 10067 CodeGenFunction::EmitOMPTargetParallelDeviceFunction( 10068 CGM, ParentName, cast<OMPTargetParallelDirective>(E)); 10069 break; 10070 case OMPD_target_teams: 10071 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction( 10072 CGM, ParentName, cast<OMPTargetTeamsDirective>(E)); 10073 break; 10074 case OMPD_target_teams_distribute: 10075 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction( 10076 CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E)); 10077 break; 10078 case OMPD_target_teams_distribute_simd: 10079 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction( 10080 CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E)); 10081 break; 10082 case OMPD_target_parallel_for: 10083 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction( 10084 CGM, ParentName, cast<OMPTargetParallelForDirective>(E)); 10085 break; 10086 case OMPD_target_parallel_for_simd: 10087 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction( 10088 CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E)); 10089 break; 10090 case OMPD_target_simd: 10091 CodeGenFunction::EmitOMPTargetSimdDeviceFunction( 10092 CGM, ParentName, cast<OMPTargetSimdDirective>(E)); 10093 break; 10094 case OMPD_target_teams_distribute_parallel_for: 10095 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction( 10096 CGM, ParentName, 10097 cast<OMPTargetTeamsDistributeParallelForDirective>(E)); 10098 break; 10099 case OMPD_target_teams_distribute_parallel_for_simd: 10100 CodeGenFunction:: 10101 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction( 10102 CGM, ParentName, 10103 cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E)); 10104 break; 10105 case OMPD_parallel: 10106 case OMPD_for: 10107 case OMPD_parallel_for: 10108 case OMPD_parallel_master: 10109 case OMPD_parallel_sections: 10110 case OMPD_for_simd: 10111 case OMPD_parallel_for_simd: 10112 case OMPD_cancel: 10113 case OMPD_cancellation_point: 10114 case OMPD_ordered: 10115 case OMPD_threadprivate: 10116 case OMPD_allocate: 10117 case OMPD_task: 10118 case OMPD_simd: 10119 case OMPD_sections: 10120 case OMPD_section: 10121 case OMPD_single: 10122 case OMPD_master: 10123 case OMPD_critical: 10124 case OMPD_taskyield: 10125 case OMPD_barrier: 10126 case OMPD_taskwait: 10127 case OMPD_taskgroup: 10128 case OMPD_atomic: 10129 case OMPD_flush: 10130 case OMPD_depobj: 10131 case OMPD_scan: 10132 case OMPD_teams: 10133 case OMPD_target_data: 10134 case OMPD_target_exit_data: 10135 case OMPD_target_enter_data: 10136 case OMPD_distribute: 10137 case OMPD_distribute_simd: 10138 case OMPD_distribute_parallel_for: 10139 case OMPD_distribute_parallel_for_simd: 10140 case OMPD_teams_distribute: 10141 case OMPD_teams_distribute_simd: 10142 case OMPD_teams_distribute_parallel_for: 10143 case OMPD_teams_distribute_parallel_for_simd: 10144 case OMPD_target_update: 10145 case OMPD_declare_simd: 10146 case OMPD_declare_variant: 10147 case OMPD_begin_declare_variant: 10148 case OMPD_end_declare_variant: 10149 case OMPD_declare_target: 10150 case OMPD_end_declare_target: 10151 case OMPD_declare_reduction: 10152 case OMPD_declare_mapper: 10153 case OMPD_taskloop: 10154 case OMPD_taskloop_simd: 10155 case OMPD_master_taskloop: 10156 case OMPD_master_taskloop_simd: 10157 case OMPD_parallel_master_taskloop: 10158 case OMPD_parallel_master_taskloop_simd: 10159 case OMPD_requires: 10160 case OMPD_unknown: 10161 default: 10162 llvm_unreachable("Unknown target directive for OpenMP device codegen."); 10163 } 10164 return; 10165 } 10166 10167 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) { 10168 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt()) 10169 return; 10170 10171 scanForTargetRegionsFunctions(E->getRawStmt(), ParentName); 10172 return; 10173 } 10174 10175 // If this is a lambda function, look into its body. 10176 if (const auto *L = dyn_cast<LambdaExpr>(S)) 10177 S = L->getBody(); 10178 10179 // Keep looking for target regions recursively. 10180 for (const Stmt *II : S->children()) 10181 scanForTargetRegionsFunctions(II, ParentName); 10182 } 10183 10184 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) { 10185 // If emitting code for the host, we do not process FD here. Instead we do 10186 // the normal code generation. 10187 if (!CGM.getLangOpts().OpenMPIsDevice) { 10188 if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) { 10189 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 10190 OMPDeclareTargetDeclAttr::getDeviceType(FD); 10191 // Do not emit device_type(nohost) functions for the host. 10192 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost) 10193 return true; 10194 } 10195 return false; 10196 } 10197 10198 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl()); 10199 // Try to detect target regions in the function. 10200 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) { 10201 StringRef Name = CGM.getMangledName(GD); 10202 scanForTargetRegionsFunctions(FD->getBody(), Name); 10203 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 10204 OMPDeclareTargetDeclAttr::getDeviceType(FD); 10205 // Do not emit device_type(nohost) functions for the host. 10206 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host) 10207 return true; 10208 } 10209 10210 // Do not to emit function if it is not marked as declare target. 10211 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) && 10212 AlreadyEmittedTargetDecls.count(VD) == 0; 10213 } 10214 10215 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 10216 if (!CGM.getLangOpts().OpenMPIsDevice) 10217 return false; 10218 10219 // Check if there are Ctors/Dtors in this declaration and look for target 10220 // regions in it. We use the complete variant to produce the kernel name 10221 // mangling. 10222 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType(); 10223 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) { 10224 for (const CXXConstructorDecl *Ctor : RD->ctors()) { 10225 StringRef ParentName = 10226 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete)); 10227 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName); 10228 } 10229 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) { 10230 StringRef ParentName = 10231 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete)); 10232 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName); 10233 } 10234 } 10235 10236 // Do not to emit variable if it is not marked as declare target. 10237 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10238 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration( 10239 cast<VarDecl>(GD.getDecl())); 10240 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 10241 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10242 HasRequiresUnifiedSharedMemory)) { 10243 DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl())); 10244 return true; 10245 } 10246 return false; 10247 } 10248 10249 llvm::Constant * 10250 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF, 10251 const VarDecl *VD) { 10252 assert(VD->getType().isConstant(CGM.getContext()) && 10253 "Expected constant variable."); 10254 StringRef VarName; 10255 llvm::Constant *Addr; 10256 llvm::GlobalValue::LinkageTypes Linkage; 10257 QualType Ty = VD->getType(); 10258 SmallString<128> Buffer; 10259 { 10260 unsigned DeviceID; 10261 unsigned FileID; 10262 unsigned Line; 10263 getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID, 10264 FileID, Line); 10265 llvm::raw_svector_ostream OS(Buffer); 10266 OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID) 10267 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 10268 VarName = OS.str(); 10269 } 10270 Linkage = llvm::GlobalValue::InternalLinkage; 10271 Addr = 10272 getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName, 10273 getDefaultFirstprivateAddressSpace()); 10274 cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage); 10275 CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty); 10276 CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr)); 10277 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 10278 VarName, Addr, VarSize, 10279 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage); 10280 return Addr; 10281 } 10282 10283 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD, 10284 llvm::Constant *Addr) { 10285 if (CGM.getLangOpts().OMPTargetTriples.empty() && 10286 !CGM.getLangOpts().OpenMPIsDevice) 10287 return; 10288 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10289 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 10290 if (!Res) { 10291 if (CGM.getLangOpts().OpenMPIsDevice) { 10292 // Register non-target variables being emitted in device code (debug info 10293 // may cause this). 10294 StringRef VarName = CGM.getMangledName(VD); 10295 EmittedNonTargetVariables.try_emplace(VarName, Addr); 10296 } 10297 return; 10298 } 10299 // Register declare target variables. 10300 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags; 10301 StringRef VarName; 10302 CharUnits VarSize; 10303 llvm::GlobalValue::LinkageTypes Linkage; 10304 10305 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10306 !HasRequiresUnifiedSharedMemory) { 10307 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10308 VarName = CGM.getMangledName(VD); 10309 if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) { 10310 VarSize = CGM.getContext().getTypeSizeInChars(VD->getType()); 10311 assert(!VarSize.isZero() && "Expected non-zero size of the variable"); 10312 } else { 10313 VarSize = CharUnits::Zero(); 10314 } 10315 Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false); 10316 // Temp solution to prevent optimizations of the internal variables. 10317 if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) { 10318 std::string RefName = getName({VarName, "ref"}); 10319 if (!CGM.GetGlobalValue(RefName)) { 10320 llvm::Constant *AddrRef = 10321 getOrCreateInternalVariable(Addr->getType(), RefName); 10322 auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef); 10323 GVAddrRef->setConstant(/*Val=*/true); 10324 GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage); 10325 GVAddrRef->setInitializer(Addr); 10326 CGM.addCompilerUsedGlobal(GVAddrRef); 10327 } 10328 } 10329 } else { 10330 assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 10331 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10332 HasRequiresUnifiedSharedMemory)) && 10333 "Declare target attribute must link or to with unified memory."); 10334 if (*Res == OMPDeclareTargetDeclAttr::MT_Link) 10335 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink; 10336 else 10337 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10338 10339 if (CGM.getLangOpts().OpenMPIsDevice) { 10340 VarName = Addr->getName(); 10341 Addr = nullptr; 10342 } else { 10343 VarName = getAddrOfDeclareTargetVar(VD).getName(); 10344 Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer()); 10345 } 10346 VarSize = CGM.getPointerSize(); 10347 Linkage = llvm::GlobalValue::WeakAnyLinkage; 10348 } 10349 10350 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 10351 VarName, Addr, VarSize, Flags, Linkage); 10352 } 10353 10354 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) { 10355 if (isa<FunctionDecl>(GD.getDecl()) || 10356 isa<OMPDeclareReductionDecl>(GD.getDecl())) 10357 return emitTargetFunctions(GD); 10358 10359 return emitTargetGlobalVariable(GD); 10360 } 10361 10362 void CGOpenMPRuntime::emitDeferredTargetDecls() const { 10363 for (const VarDecl *VD : DeferredGlobalVariables) { 10364 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10365 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 10366 if (!Res) 10367 continue; 10368 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10369 !HasRequiresUnifiedSharedMemory) { 10370 CGM.EmitGlobal(VD); 10371 } else { 10372 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link || 10373 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10374 HasRequiresUnifiedSharedMemory)) && 10375 "Expected link clause or to clause with unified memory."); 10376 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 10377 } 10378 } 10379 } 10380 10381 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas( 10382 CodeGenFunction &CGF, const OMPExecutableDirective &D) const { 10383 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) && 10384 " Expected target-based directive."); 10385 } 10386 10387 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) { 10388 for (const OMPClause *Clause : D->clauselists()) { 10389 if (Clause->getClauseKind() == OMPC_unified_shared_memory) { 10390 HasRequiresUnifiedSharedMemory = true; 10391 } else if (const auto *AC = 10392 dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) { 10393 switch (AC->getAtomicDefaultMemOrderKind()) { 10394 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel: 10395 RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease; 10396 break; 10397 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst: 10398 RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent; 10399 break; 10400 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed: 10401 RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic; 10402 break; 10403 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown: 10404 break; 10405 } 10406 } 10407 } 10408 } 10409 10410 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const { 10411 return RequiresAtomicOrdering; 10412 } 10413 10414 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD, 10415 LangAS &AS) { 10416 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>()) 10417 return false; 10418 const auto *A = VD->getAttr<OMPAllocateDeclAttr>(); 10419 switch(A->getAllocatorType()) { 10420 case OMPAllocateDeclAttr::OMPNullMemAlloc: 10421 case OMPAllocateDeclAttr::OMPDefaultMemAlloc: 10422 // Not supported, fallback to the default mem space. 10423 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc: 10424 case OMPAllocateDeclAttr::OMPCGroupMemAlloc: 10425 case OMPAllocateDeclAttr::OMPHighBWMemAlloc: 10426 case OMPAllocateDeclAttr::OMPLowLatMemAlloc: 10427 case OMPAllocateDeclAttr::OMPThreadMemAlloc: 10428 case OMPAllocateDeclAttr::OMPConstMemAlloc: 10429 case OMPAllocateDeclAttr::OMPPTeamMemAlloc: 10430 AS = LangAS::Default; 10431 return true; 10432 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc: 10433 llvm_unreachable("Expected predefined allocator for the variables with the " 10434 "static storage."); 10435 } 10436 return false; 10437 } 10438 10439 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const { 10440 return HasRequiresUnifiedSharedMemory; 10441 } 10442 10443 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII( 10444 CodeGenModule &CGM) 10445 : CGM(CGM) { 10446 if (CGM.getLangOpts().OpenMPIsDevice) { 10447 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal; 10448 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false; 10449 } 10450 } 10451 10452 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() { 10453 if (CGM.getLangOpts().OpenMPIsDevice) 10454 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal; 10455 } 10456 10457 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) { 10458 if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal) 10459 return true; 10460 10461 const auto *D = cast<FunctionDecl>(GD.getDecl()); 10462 // Do not to emit function if it is marked as declare target as it was already 10463 // emitted. 10464 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) { 10465 if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) { 10466 if (auto *F = dyn_cast_or_null<llvm::Function>( 10467 CGM.GetGlobalValue(CGM.getMangledName(GD)))) 10468 return !F->isDeclaration(); 10469 return false; 10470 } 10471 return true; 10472 } 10473 10474 return !AlreadyEmittedTargetDecls.insert(D).second; 10475 } 10476 10477 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() { 10478 // If we don't have entries or if we are emitting code for the device, we 10479 // don't need to do anything. 10480 if (CGM.getLangOpts().OMPTargetTriples.empty() || 10481 CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice || 10482 (OffloadEntriesInfoManager.empty() && 10483 !HasEmittedDeclareTargetRegion && 10484 !HasEmittedTargetRegion)) 10485 return nullptr; 10486 10487 // Create and register the function that handles the requires directives. 10488 ASTContext &C = CGM.getContext(); 10489 10490 llvm::Function *RequiresRegFn; 10491 { 10492 CodeGenFunction CGF(CGM); 10493 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 10494 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 10495 std::string ReqName = getName({"omp_offloading", "requires_reg"}); 10496 RequiresRegFn = CGM.CreateGlobalInitOrCleanUpFunction(FTy, ReqName, FI); 10497 CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {}); 10498 OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE; 10499 // TODO: check for other requires clauses. 10500 // The requires directive takes effect only when a target region is 10501 // present in the compilation unit. Otherwise it is ignored and not 10502 // passed to the runtime. This avoids the runtime from throwing an error 10503 // for mismatching requires clauses across compilation units that don't 10504 // contain at least 1 target region. 10505 assert((HasEmittedTargetRegion || 10506 HasEmittedDeclareTargetRegion || 10507 !OffloadEntriesInfoManager.empty()) && 10508 "Target or declare target region expected."); 10509 if (HasRequiresUnifiedSharedMemory) 10510 Flags = OMP_REQ_UNIFIED_SHARED_MEMORY; 10511 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 10512 CGM.getModule(), OMPRTL___tgt_register_requires), 10513 llvm::ConstantInt::get(CGM.Int64Ty, Flags)); 10514 CGF.FinishFunction(); 10515 } 10516 return RequiresRegFn; 10517 } 10518 10519 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF, 10520 const OMPExecutableDirective &D, 10521 SourceLocation Loc, 10522 llvm::Function *OutlinedFn, 10523 ArrayRef<llvm::Value *> CapturedVars) { 10524 if (!CGF.HaveInsertPoint()) 10525 return; 10526 10527 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10528 CodeGenFunction::RunCleanupsScope Scope(CGF); 10529 10530 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn); 10531 llvm::Value *Args[] = { 10532 RTLoc, 10533 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 10534 CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())}; 10535 llvm::SmallVector<llvm::Value *, 16> RealArgs; 10536 RealArgs.append(std::begin(Args), std::end(Args)); 10537 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 10538 10539 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction( 10540 CGM.getModule(), OMPRTL___kmpc_fork_teams); 10541 CGF.EmitRuntimeCall(RTLFn, RealArgs); 10542 } 10543 10544 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 10545 const Expr *NumTeams, 10546 const Expr *ThreadLimit, 10547 SourceLocation Loc) { 10548 if (!CGF.HaveInsertPoint()) 10549 return; 10550 10551 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10552 10553 llvm::Value *NumTeamsVal = 10554 NumTeams 10555 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams), 10556 CGF.CGM.Int32Ty, /* isSigned = */ true) 10557 : CGF.Builder.getInt32(0); 10558 10559 llvm::Value *ThreadLimitVal = 10560 ThreadLimit 10561 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit), 10562 CGF.CGM.Int32Ty, /* isSigned = */ true) 10563 : CGF.Builder.getInt32(0); 10564 10565 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit) 10566 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal, 10567 ThreadLimitVal}; 10568 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 10569 CGM.getModule(), OMPRTL___kmpc_push_num_teams), 10570 PushNumTeamsArgs); 10571 } 10572 10573 void CGOpenMPRuntime::emitTargetDataCalls( 10574 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10575 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 10576 if (!CGF.HaveInsertPoint()) 10577 return; 10578 10579 // Action used to replace the default codegen action and turn privatization 10580 // off. 10581 PrePostActionTy NoPrivAction; 10582 10583 // Generate the code for the opening of the data environment. Capture all the 10584 // arguments of the runtime call by reference because they are used in the 10585 // closing of the region. 10586 auto &&BeginThenGen = [this, &D, Device, &Info, 10587 &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) { 10588 // Fill up the arrays with all the mapped variables. 10589 MappableExprsHandler::MapCombinedInfoTy CombinedInfo; 10590 10591 // Get map clause information. 10592 MappableExprsHandler MEHandler(D, CGF); 10593 MEHandler.generateAllInfo(CombinedInfo); 10594 10595 // Fill up the arrays and create the arguments. 10596 emitOffloadingArrays(CGF, CombinedInfo, Info, /*IsNonContiguous=*/true); 10597 10598 llvm::Value *BasePointersArrayArg = nullptr; 10599 llvm::Value *PointersArrayArg = nullptr; 10600 llvm::Value *SizesArrayArg = nullptr; 10601 llvm::Value *MapTypesArrayArg = nullptr; 10602 llvm::Value *MappersArrayArg = nullptr; 10603 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10604 SizesArrayArg, MapTypesArrayArg, 10605 MappersArrayArg, Info); 10606 10607 // Emit device ID if any. 10608 llvm::Value *DeviceID = nullptr; 10609 if (Device) { 10610 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10611 CGF.Int64Ty, /*isSigned=*/true); 10612 } else { 10613 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10614 } 10615 10616 // Emit the number of elements in the offloading arrays. 10617 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10618 10619 llvm::Value *OffloadingArgs[] = { 10620 DeviceID, PointerNum, BasePointersArrayArg, PointersArrayArg, 10621 SizesArrayArg, MapTypesArrayArg, MappersArrayArg}; 10622 CGF.EmitRuntimeCall( 10623 OMPBuilder.getOrCreateRuntimeFunction( 10624 CGM.getModule(), OMPRTL___tgt_target_data_begin_mapper), 10625 OffloadingArgs); 10626 10627 // If device pointer privatization is required, emit the body of the region 10628 // here. It will have to be duplicated: with and without privatization. 10629 if (!Info.CaptureDeviceAddrMap.empty()) 10630 CodeGen(CGF); 10631 }; 10632 10633 // Generate code for the closing of the data region. 10634 auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF, 10635 PrePostActionTy &) { 10636 assert(Info.isValid() && "Invalid data environment closing arguments."); 10637 10638 llvm::Value *BasePointersArrayArg = nullptr; 10639 llvm::Value *PointersArrayArg = nullptr; 10640 llvm::Value *SizesArrayArg = nullptr; 10641 llvm::Value *MapTypesArrayArg = nullptr; 10642 llvm::Value *MappersArrayArg = nullptr; 10643 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10644 SizesArrayArg, MapTypesArrayArg, 10645 MappersArrayArg, Info, {/*ForEndCall=*/true}); 10646 10647 // Emit device ID if any. 10648 llvm::Value *DeviceID = nullptr; 10649 if (Device) { 10650 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10651 CGF.Int64Ty, /*isSigned=*/true); 10652 } else { 10653 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10654 } 10655 10656 // Emit the number of elements in the offloading arrays. 10657 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10658 10659 llvm::Value *OffloadingArgs[] = { 10660 DeviceID, PointerNum, BasePointersArrayArg, PointersArrayArg, 10661 SizesArrayArg, MapTypesArrayArg, MappersArrayArg}; 10662 CGF.EmitRuntimeCall( 10663 OMPBuilder.getOrCreateRuntimeFunction( 10664 CGM.getModule(), OMPRTL___tgt_target_data_end_mapper), 10665 OffloadingArgs); 10666 }; 10667 10668 // If we need device pointer privatization, we need to emit the body of the 10669 // region with no privatization in the 'else' branch of the conditional. 10670 // Otherwise, we don't have to do anything. 10671 auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF, 10672 PrePostActionTy &) { 10673 if (!Info.CaptureDeviceAddrMap.empty()) { 10674 CodeGen.setAction(NoPrivAction); 10675 CodeGen(CGF); 10676 } 10677 }; 10678 10679 // We don't have to do anything to close the region if the if clause evaluates 10680 // to false. 10681 auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {}; 10682 10683 if (IfCond) { 10684 emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen); 10685 } else { 10686 RegionCodeGenTy RCG(BeginThenGen); 10687 RCG(CGF); 10688 } 10689 10690 // If we don't require privatization of device pointers, we emit the body in 10691 // between the runtime calls. This avoids duplicating the body code. 10692 if (Info.CaptureDeviceAddrMap.empty()) { 10693 CodeGen.setAction(NoPrivAction); 10694 CodeGen(CGF); 10695 } 10696 10697 if (IfCond) { 10698 emitIfClause(CGF, IfCond, EndThenGen, EndElseGen); 10699 } else { 10700 RegionCodeGenTy RCG(EndThenGen); 10701 RCG(CGF); 10702 } 10703 } 10704 10705 void CGOpenMPRuntime::emitTargetDataStandAloneCall( 10706 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10707 const Expr *Device) { 10708 if (!CGF.HaveInsertPoint()) 10709 return; 10710 10711 assert((isa<OMPTargetEnterDataDirective>(D) || 10712 isa<OMPTargetExitDataDirective>(D) || 10713 isa<OMPTargetUpdateDirective>(D)) && 10714 "Expecting either target enter, exit data, or update directives."); 10715 10716 CodeGenFunction::OMPTargetDataInfo InputInfo; 10717 llvm::Value *MapTypesArray = nullptr; 10718 // Generate the code for the opening of the data environment. 10719 auto &&ThenGen = [this, &D, Device, &InputInfo, 10720 &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) { 10721 // Emit device ID if any. 10722 llvm::Value *DeviceID = nullptr; 10723 if (Device) { 10724 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10725 CGF.Int64Ty, /*isSigned=*/true); 10726 } else { 10727 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10728 } 10729 10730 // Emit the number of elements in the offloading arrays. 10731 llvm::Constant *PointerNum = 10732 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 10733 10734 llvm::Value *OffloadingArgs[] = {DeviceID, 10735 PointerNum, 10736 InputInfo.BasePointersArray.getPointer(), 10737 InputInfo.PointersArray.getPointer(), 10738 InputInfo.SizesArray.getPointer(), 10739 MapTypesArray, 10740 InputInfo.MappersArray.getPointer()}; 10741 10742 // Select the right runtime function call for each standalone 10743 // directive. 10744 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 10745 RuntimeFunction RTLFn; 10746 switch (D.getDirectiveKind()) { 10747 case OMPD_target_enter_data: 10748 RTLFn = HasNowait ? OMPRTL___tgt_target_data_begin_nowait_mapper 10749 : OMPRTL___tgt_target_data_begin_mapper; 10750 break; 10751 case OMPD_target_exit_data: 10752 RTLFn = HasNowait ? OMPRTL___tgt_target_data_end_nowait_mapper 10753 : OMPRTL___tgt_target_data_end_mapper; 10754 break; 10755 case OMPD_target_update: 10756 RTLFn = HasNowait ? OMPRTL___tgt_target_data_update_nowait_mapper 10757 : OMPRTL___tgt_target_data_update_mapper; 10758 break; 10759 case OMPD_parallel: 10760 case OMPD_for: 10761 case OMPD_parallel_for: 10762 case OMPD_parallel_master: 10763 case OMPD_parallel_sections: 10764 case OMPD_for_simd: 10765 case OMPD_parallel_for_simd: 10766 case OMPD_cancel: 10767 case OMPD_cancellation_point: 10768 case OMPD_ordered: 10769 case OMPD_threadprivate: 10770 case OMPD_allocate: 10771 case OMPD_task: 10772 case OMPD_simd: 10773 case OMPD_sections: 10774 case OMPD_section: 10775 case OMPD_single: 10776 case OMPD_master: 10777 case OMPD_critical: 10778 case OMPD_taskyield: 10779 case OMPD_barrier: 10780 case OMPD_taskwait: 10781 case OMPD_taskgroup: 10782 case OMPD_atomic: 10783 case OMPD_flush: 10784 case OMPD_depobj: 10785 case OMPD_scan: 10786 case OMPD_teams: 10787 case OMPD_target_data: 10788 case OMPD_distribute: 10789 case OMPD_distribute_simd: 10790 case OMPD_distribute_parallel_for: 10791 case OMPD_distribute_parallel_for_simd: 10792 case OMPD_teams_distribute: 10793 case OMPD_teams_distribute_simd: 10794 case OMPD_teams_distribute_parallel_for: 10795 case OMPD_teams_distribute_parallel_for_simd: 10796 case OMPD_declare_simd: 10797 case OMPD_declare_variant: 10798 case OMPD_begin_declare_variant: 10799 case OMPD_end_declare_variant: 10800 case OMPD_declare_target: 10801 case OMPD_end_declare_target: 10802 case OMPD_declare_reduction: 10803 case OMPD_declare_mapper: 10804 case OMPD_taskloop: 10805 case OMPD_taskloop_simd: 10806 case OMPD_master_taskloop: 10807 case OMPD_master_taskloop_simd: 10808 case OMPD_parallel_master_taskloop: 10809 case OMPD_parallel_master_taskloop_simd: 10810 case OMPD_target: 10811 case OMPD_target_simd: 10812 case OMPD_target_teams_distribute: 10813 case OMPD_target_teams_distribute_simd: 10814 case OMPD_target_teams_distribute_parallel_for: 10815 case OMPD_target_teams_distribute_parallel_for_simd: 10816 case OMPD_target_teams: 10817 case OMPD_target_parallel: 10818 case OMPD_target_parallel_for: 10819 case OMPD_target_parallel_for_simd: 10820 case OMPD_requires: 10821 case OMPD_unknown: 10822 default: 10823 llvm_unreachable("Unexpected standalone target data directive."); 10824 break; 10825 } 10826 CGF.EmitRuntimeCall( 10827 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), RTLFn), 10828 OffloadingArgs); 10829 }; 10830 10831 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray]( 10832 CodeGenFunction &CGF, PrePostActionTy &) { 10833 // Fill up the arrays with all the mapped variables. 10834 MappableExprsHandler::MapCombinedInfoTy CombinedInfo; 10835 10836 // Get map clause information. 10837 MappableExprsHandler MEHandler(D, CGF); 10838 MEHandler.generateAllInfo(CombinedInfo); 10839 10840 TargetDataInfo Info; 10841 // Fill up the arrays and create the arguments. 10842 emitOffloadingArrays(CGF, CombinedInfo, Info, /*IsNonContiguous=*/true); 10843 bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() || 10844 D.hasClausesOfKind<OMPNowaitClause>(); 10845 emitOffloadingArraysArgument( 10846 CGF, Info.BasePointersArray, Info.PointersArray, Info.SizesArray, 10847 Info.MapTypesArray, Info.MappersArray, Info, {/*ForEndTask=*/false}); 10848 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 10849 InputInfo.BasePointersArray = 10850 Address(Info.BasePointersArray, CGM.getPointerAlign()); 10851 InputInfo.PointersArray = 10852 Address(Info.PointersArray, CGM.getPointerAlign()); 10853 InputInfo.SizesArray = 10854 Address(Info.SizesArray, CGM.getPointerAlign()); 10855 InputInfo.MappersArray = Address(Info.MappersArray, CGM.getPointerAlign()); 10856 MapTypesArray = Info.MapTypesArray; 10857 if (RequiresOuterTask) 10858 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 10859 else 10860 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 10861 }; 10862 10863 if (IfCond) { 10864 emitIfClause(CGF, IfCond, TargetThenGen, 10865 [](CodeGenFunction &CGF, PrePostActionTy &) {}); 10866 } else { 10867 RegionCodeGenTy ThenRCG(TargetThenGen); 10868 ThenRCG(CGF); 10869 } 10870 } 10871 10872 namespace { 10873 /// Kind of parameter in a function with 'declare simd' directive. 10874 enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector }; 10875 /// Attribute set of the parameter. 10876 struct ParamAttrTy { 10877 ParamKindTy Kind = Vector; 10878 llvm::APSInt StrideOrArg; 10879 llvm::APSInt Alignment; 10880 }; 10881 } // namespace 10882 10883 static unsigned evaluateCDTSize(const FunctionDecl *FD, 10884 ArrayRef<ParamAttrTy> ParamAttrs) { 10885 // Every vector variant of a SIMD-enabled function has a vector length (VLEN). 10886 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument 10887 // of that clause. The VLEN value must be power of 2. 10888 // In other case the notion of the function`s "characteristic data type" (CDT) 10889 // is used to compute the vector length. 10890 // CDT is defined in the following order: 10891 // a) For non-void function, the CDT is the return type. 10892 // b) If the function has any non-uniform, non-linear parameters, then the 10893 // CDT is the type of the first such parameter. 10894 // c) If the CDT determined by a) or b) above is struct, union, or class 10895 // type which is pass-by-value (except for the type that maps to the 10896 // built-in complex data type), the characteristic data type is int. 10897 // d) If none of the above three cases is applicable, the CDT is int. 10898 // The VLEN is then determined based on the CDT and the size of vector 10899 // register of that ISA for which current vector version is generated. The 10900 // VLEN is computed using the formula below: 10901 // VLEN = sizeof(vector_register) / sizeof(CDT), 10902 // where vector register size specified in section 3.2.1 Registers and the 10903 // Stack Frame of original AMD64 ABI document. 10904 QualType RetType = FD->getReturnType(); 10905 if (RetType.isNull()) 10906 return 0; 10907 ASTContext &C = FD->getASTContext(); 10908 QualType CDT; 10909 if (!RetType.isNull() && !RetType->isVoidType()) { 10910 CDT = RetType; 10911 } else { 10912 unsigned Offset = 0; 10913 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) { 10914 if (ParamAttrs[Offset].Kind == Vector) 10915 CDT = C.getPointerType(C.getRecordType(MD->getParent())); 10916 ++Offset; 10917 } 10918 if (CDT.isNull()) { 10919 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10920 if (ParamAttrs[I + Offset].Kind == Vector) { 10921 CDT = FD->getParamDecl(I)->getType(); 10922 break; 10923 } 10924 } 10925 } 10926 } 10927 if (CDT.isNull()) 10928 CDT = C.IntTy; 10929 CDT = CDT->getCanonicalTypeUnqualified(); 10930 if (CDT->isRecordType() || CDT->isUnionType()) 10931 CDT = C.IntTy; 10932 return C.getTypeSize(CDT); 10933 } 10934 10935 static void 10936 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn, 10937 const llvm::APSInt &VLENVal, 10938 ArrayRef<ParamAttrTy> ParamAttrs, 10939 OMPDeclareSimdDeclAttr::BranchStateTy State) { 10940 struct ISADataTy { 10941 char ISA; 10942 unsigned VecRegSize; 10943 }; 10944 ISADataTy ISAData[] = { 10945 { 10946 'b', 128 10947 }, // SSE 10948 { 10949 'c', 256 10950 }, // AVX 10951 { 10952 'd', 256 10953 }, // AVX2 10954 { 10955 'e', 512 10956 }, // AVX512 10957 }; 10958 llvm::SmallVector<char, 2> Masked; 10959 switch (State) { 10960 case OMPDeclareSimdDeclAttr::BS_Undefined: 10961 Masked.push_back('N'); 10962 Masked.push_back('M'); 10963 break; 10964 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10965 Masked.push_back('N'); 10966 break; 10967 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10968 Masked.push_back('M'); 10969 break; 10970 } 10971 for (char Mask : Masked) { 10972 for (const ISADataTy &Data : ISAData) { 10973 SmallString<256> Buffer; 10974 llvm::raw_svector_ostream Out(Buffer); 10975 Out << "_ZGV" << Data.ISA << Mask; 10976 if (!VLENVal) { 10977 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs); 10978 assert(NumElts && "Non-zero simdlen/cdtsize expected"); 10979 Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts); 10980 } else { 10981 Out << VLENVal; 10982 } 10983 for (const ParamAttrTy &ParamAttr : ParamAttrs) { 10984 switch (ParamAttr.Kind){ 10985 case LinearWithVarStride: 10986 Out << 's' << ParamAttr.StrideOrArg; 10987 break; 10988 case Linear: 10989 Out << 'l'; 10990 if (ParamAttr.StrideOrArg != 1) 10991 Out << ParamAttr.StrideOrArg; 10992 break; 10993 case Uniform: 10994 Out << 'u'; 10995 break; 10996 case Vector: 10997 Out << 'v'; 10998 break; 10999 } 11000 if (!!ParamAttr.Alignment) 11001 Out << 'a' << ParamAttr.Alignment; 11002 } 11003 Out << '_' << Fn->getName(); 11004 Fn->addFnAttr(Out.str()); 11005 } 11006 } 11007 } 11008 11009 // This are the Functions that are needed to mangle the name of the 11010 // vector functions generated by the compiler, according to the rules 11011 // defined in the "Vector Function ABI specifications for AArch64", 11012 // available at 11013 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi. 11014 11015 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI. 11016 /// 11017 /// TODO: Need to implement the behavior for reference marked with a 11018 /// var or no linear modifiers (1.b in the section). For this, we 11019 /// need to extend ParamKindTy to support the linear modifiers. 11020 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) { 11021 QT = QT.getCanonicalType(); 11022 11023 if (QT->isVoidType()) 11024 return false; 11025 11026 if (Kind == ParamKindTy::Uniform) 11027 return false; 11028 11029 if (Kind == ParamKindTy::Linear) 11030 return false; 11031 11032 // TODO: Handle linear references with modifiers 11033 11034 if (Kind == ParamKindTy::LinearWithVarStride) 11035 return false; 11036 11037 return true; 11038 } 11039 11040 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI. 11041 static bool getAArch64PBV(QualType QT, ASTContext &C) { 11042 QT = QT.getCanonicalType(); 11043 unsigned Size = C.getTypeSize(QT); 11044 11045 // Only scalars and complex within 16 bytes wide set PVB to true. 11046 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128) 11047 return false; 11048 11049 if (QT->isFloatingType()) 11050 return true; 11051 11052 if (QT->isIntegerType()) 11053 return true; 11054 11055 if (QT->isPointerType()) 11056 return true; 11057 11058 // TODO: Add support for complex types (section 3.1.2, item 2). 11059 11060 return false; 11061 } 11062 11063 /// Computes the lane size (LS) of a return type or of an input parameter, 11064 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI. 11065 /// TODO: Add support for references, section 3.2.1, item 1. 11066 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) { 11067 if (!getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) { 11068 QualType PTy = QT.getCanonicalType()->getPointeeType(); 11069 if (getAArch64PBV(PTy, C)) 11070 return C.getTypeSize(PTy); 11071 } 11072 if (getAArch64PBV(QT, C)) 11073 return C.getTypeSize(QT); 11074 11075 return C.getTypeSize(C.getUIntPtrType()); 11076 } 11077 11078 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the 11079 // signature of the scalar function, as defined in 3.2.2 of the 11080 // AAVFABI. 11081 static std::tuple<unsigned, unsigned, bool> 11082 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) { 11083 QualType RetType = FD->getReturnType().getCanonicalType(); 11084 11085 ASTContext &C = FD->getASTContext(); 11086 11087 bool OutputBecomesInput = false; 11088 11089 llvm::SmallVector<unsigned, 8> Sizes; 11090 if (!RetType->isVoidType()) { 11091 Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C)); 11092 if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {})) 11093 OutputBecomesInput = true; 11094 } 11095 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 11096 QualType QT = FD->getParamDecl(I)->getType().getCanonicalType(); 11097 Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C)); 11098 } 11099 11100 assert(!Sizes.empty() && "Unable to determine NDS and WDS."); 11101 // The LS of a function parameter / return value can only be a power 11102 // of 2, starting from 8 bits, up to 128. 11103 assert(std::all_of(Sizes.begin(), Sizes.end(), 11104 [](unsigned Size) { 11105 return Size == 8 || Size == 16 || Size == 32 || 11106 Size == 64 || Size == 128; 11107 }) && 11108 "Invalid size"); 11109 11110 return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)), 11111 *std::max_element(std::begin(Sizes), std::end(Sizes)), 11112 OutputBecomesInput); 11113 } 11114 11115 /// Mangle the parameter part of the vector function name according to 11116 /// their OpenMP classification. The mangling function is defined in 11117 /// section 3.5 of the AAVFABI. 11118 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) { 11119 SmallString<256> Buffer; 11120 llvm::raw_svector_ostream Out(Buffer); 11121 for (const auto &ParamAttr : ParamAttrs) { 11122 switch (ParamAttr.Kind) { 11123 case LinearWithVarStride: 11124 Out << "ls" << ParamAttr.StrideOrArg; 11125 break; 11126 case Linear: 11127 Out << 'l'; 11128 // Don't print the step value if it is not present or if it is 11129 // equal to 1. 11130 if (ParamAttr.StrideOrArg != 1) 11131 Out << ParamAttr.StrideOrArg; 11132 break; 11133 case Uniform: 11134 Out << 'u'; 11135 break; 11136 case Vector: 11137 Out << 'v'; 11138 break; 11139 } 11140 11141 if (!!ParamAttr.Alignment) 11142 Out << 'a' << ParamAttr.Alignment; 11143 } 11144 11145 return std::string(Out.str()); 11146 } 11147 11148 // Function used to add the attribute. The parameter `VLEN` is 11149 // templated to allow the use of "x" when targeting scalable functions 11150 // for SVE. 11151 template <typename T> 11152 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, 11153 char ISA, StringRef ParSeq, 11154 StringRef MangledName, bool OutputBecomesInput, 11155 llvm::Function *Fn) { 11156 SmallString<256> Buffer; 11157 llvm::raw_svector_ostream Out(Buffer); 11158 Out << Prefix << ISA << LMask << VLEN; 11159 if (OutputBecomesInput) 11160 Out << "v"; 11161 Out << ParSeq << "_" << MangledName; 11162 Fn->addFnAttr(Out.str()); 11163 } 11164 11165 // Helper function to generate the Advanced SIMD names depending on 11166 // the value of the NDS when simdlen is not present. 11167 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, 11168 StringRef Prefix, char ISA, 11169 StringRef ParSeq, StringRef MangledName, 11170 bool OutputBecomesInput, 11171 llvm::Function *Fn) { 11172 switch (NDS) { 11173 case 8: 11174 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 11175 OutputBecomesInput, Fn); 11176 addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName, 11177 OutputBecomesInput, Fn); 11178 break; 11179 case 16: 11180 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 11181 OutputBecomesInput, Fn); 11182 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 11183 OutputBecomesInput, Fn); 11184 break; 11185 case 32: 11186 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 11187 OutputBecomesInput, Fn); 11188 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 11189 OutputBecomesInput, Fn); 11190 break; 11191 case 64: 11192 case 128: 11193 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 11194 OutputBecomesInput, Fn); 11195 break; 11196 default: 11197 llvm_unreachable("Scalar type is too wide."); 11198 } 11199 } 11200 11201 /// Emit vector function attributes for AArch64, as defined in the AAVFABI. 11202 static void emitAArch64DeclareSimdFunction( 11203 CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN, 11204 ArrayRef<ParamAttrTy> ParamAttrs, 11205 OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName, 11206 char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) { 11207 11208 // Get basic data for building the vector signature. 11209 const auto Data = getNDSWDS(FD, ParamAttrs); 11210 const unsigned NDS = std::get<0>(Data); 11211 const unsigned WDS = std::get<1>(Data); 11212 const bool OutputBecomesInput = std::get<2>(Data); 11213 11214 // Check the values provided via `simdlen` by the user. 11215 // 1. A `simdlen(1)` doesn't produce vector signatures, 11216 if (UserVLEN == 1) { 11217 unsigned DiagID = CGM.getDiags().getCustomDiagID( 11218 DiagnosticsEngine::Warning, 11219 "The clause simdlen(1) has no effect when targeting aarch64."); 11220 CGM.getDiags().Report(SLoc, DiagID); 11221 return; 11222 } 11223 11224 // 2. Section 3.3.1, item 1: user input must be a power of 2 for 11225 // Advanced SIMD output. 11226 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) { 11227 unsigned DiagID = CGM.getDiags().getCustomDiagID( 11228 DiagnosticsEngine::Warning, "The value specified in simdlen must be a " 11229 "power of 2 when targeting Advanced SIMD."); 11230 CGM.getDiags().Report(SLoc, DiagID); 11231 return; 11232 } 11233 11234 // 3. Section 3.4.1. SVE fixed lengh must obey the architectural 11235 // limits. 11236 if (ISA == 's' && UserVLEN != 0) { 11237 if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) { 11238 unsigned DiagID = CGM.getDiags().getCustomDiagID( 11239 DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit " 11240 "lanes in the architectural constraints " 11241 "for SVE (min is 128-bit, max is " 11242 "2048-bit, by steps of 128-bit)"); 11243 CGM.getDiags().Report(SLoc, DiagID) << WDS; 11244 return; 11245 } 11246 } 11247 11248 // Sort out parameter sequence. 11249 const std::string ParSeq = mangleVectorParameters(ParamAttrs); 11250 StringRef Prefix = "_ZGV"; 11251 // Generate simdlen from user input (if any). 11252 if (UserVLEN) { 11253 if (ISA == 's') { 11254 // SVE generates only a masked function. 11255 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 11256 OutputBecomesInput, Fn); 11257 } else { 11258 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 11259 // Advanced SIMD generates one or two functions, depending on 11260 // the `[not]inbranch` clause. 11261 switch (State) { 11262 case OMPDeclareSimdDeclAttr::BS_Undefined: 11263 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 11264 OutputBecomesInput, Fn); 11265 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 11266 OutputBecomesInput, Fn); 11267 break; 11268 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 11269 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 11270 OutputBecomesInput, Fn); 11271 break; 11272 case OMPDeclareSimdDeclAttr::BS_Inbranch: 11273 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 11274 OutputBecomesInput, Fn); 11275 break; 11276 } 11277 } 11278 } else { 11279 // If no user simdlen is provided, follow the AAVFABI rules for 11280 // generating the vector length. 11281 if (ISA == 's') { 11282 // SVE, section 3.4.1, item 1. 11283 addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName, 11284 OutputBecomesInput, Fn); 11285 } else { 11286 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 11287 // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or 11288 // two vector names depending on the use of the clause 11289 // `[not]inbranch`. 11290 switch (State) { 11291 case OMPDeclareSimdDeclAttr::BS_Undefined: 11292 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 11293 OutputBecomesInput, Fn); 11294 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 11295 OutputBecomesInput, Fn); 11296 break; 11297 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 11298 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 11299 OutputBecomesInput, Fn); 11300 break; 11301 case OMPDeclareSimdDeclAttr::BS_Inbranch: 11302 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 11303 OutputBecomesInput, Fn); 11304 break; 11305 } 11306 } 11307 } 11308 } 11309 11310 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD, 11311 llvm::Function *Fn) { 11312 ASTContext &C = CGM.getContext(); 11313 FD = FD->getMostRecentDecl(); 11314 // Map params to their positions in function decl. 11315 llvm::DenseMap<const Decl *, unsigned> ParamPositions; 11316 if (isa<CXXMethodDecl>(FD)) 11317 ParamPositions.try_emplace(FD, 0); 11318 unsigned ParamPos = ParamPositions.size(); 11319 for (const ParmVarDecl *P : FD->parameters()) { 11320 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos); 11321 ++ParamPos; 11322 } 11323 while (FD) { 11324 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) { 11325 llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size()); 11326 // Mark uniform parameters. 11327 for (const Expr *E : Attr->uniforms()) { 11328 E = E->IgnoreParenImpCasts(); 11329 unsigned Pos; 11330 if (isa<CXXThisExpr>(E)) { 11331 Pos = ParamPositions[FD]; 11332 } else { 11333 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11334 ->getCanonicalDecl(); 11335 Pos = ParamPositions[PVD]; 11336 } 11337 ParamAttrs[Pos].Kind = Uniform; 11338 } 11339 // Get alignment info. 11340 auto NI = Attr->alignments_begin(); 11341 for (const Expr *E : Attr->aligneds()) { 11342 E = E->IgnoreParenImpCasts(); 11343 unsigned Pos; 11344 QualType ParmTy; 11345 if (isa<CXXThisExpr>(E)) { 11346 Pos = ParamPositions[FD]; 11347 ParmTy = E->getType(); 11348 } else { 11349 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11350 ->getCanonicalDecl(); 11351 Pos = ParamPositions[PVD]; 11352 ParmTy = PVD->getType(); 11353 } 11354 ParamAttrs[Pos].Alignment = 11355 (*NI) 11356 ? (*NI)->EvaluateKnownConstInt(C) 11357 : llvm::APSInt::getUnsigned( 11358 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy)) 11359 .getQuantity()); 11360 ++NI; 11361 } 11362 // Mark linear parameters. 11363 auto SI = Attr->steps_begin(); 11364 auto MI = Attr->modifiers_begin(); 11365 for (const Expr *E : Attr->linears()) { 11366 E = E->IgnoreParenImpCasts(); 11367 unsigned Pos; 11368 // Rescaling factor needed to compute the linear parameter 11369 // value in the mangled name. 11370 unsigned PtrRescalingFactor = 1; 11371 if (isa<CXXThisExpr>(E)) { 11372 Pos = ParamPositions[FD]; 11373 } else { 11374 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11375 ->getCanonicalDecl(); 11376 Pos = ParamPositions[PVD]; 11377 if (auto *P = dyn_cast<PointerType>(PVD->getType())) 11378 PtrRescalingFactor = CGM.getContext() 11379 .getTypeSizeInChars(P->getPointeeType()) 11380 .getQuantity(); 11381 } 11382 ParamAttrTy &ParamAttr = ParamAttrs[Pos]; 11383 ParamAttr.Kind = Linear; 11384 // Assuming a stride of 1, for `linear` without modifiers. 11385 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(1); 11386 if (*SI) { 11387 Expr::EvalResult Result; 11388 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) { 11389 if (const auto *DRE = 11390 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) { 11391 if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) { 11392 ParamAttr.Kind = LinearWithVarStride; 11393 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned( 11394 ParamPositions[StridePVD->getCanonicalDecl()]); 11395 } 11396 } 11397 } else { 11398 ParamAttr.StrideOrArg = Result.Val.getInt(); 11399 } 11400 } 11401 // If we are using a linear clause on a pointer, we need to 11402 // rescale the value of linear_step with the byte size of the 11403 // pointee type. 11404 if (Linear == ParamAttr.Kind) 11405 ParamAttr.StrideOrArg = ParamAttr.StrideOrArg * PtrRescalingFactor; 11406 ++SI; 11407 ++MI; 11408 } 11409 llvm::APSInt VLENVal; 11410 SourceLocation ExprLoc; 11411 const Expr *VLENExpr = Attr->getSimdlen(); 11412 if (VLENExpr) { 11413 VLENVal = VLENExpr->EvaluateKnownConstInt(C); 11414 ExprLoc = VLENExpr->getExprLoc(); 11415 } 11416 OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState(); 11417 if (CGM.getTriple().isX86()) { 11418 emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State); 11419 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) { 11420 unsigned VLEN = VLENVal.getExtValue(); 11421 StringRef MangledName = Fn->getName(); 11422 if (CGM.getTarget().hasFeature("sve")) 11423 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 11424 MangledName, 's', 128, Fn, ExprLoc); 11425 if (CGM.getTarget().hasFeature("neon")) 11426 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 11427 MangledName, 'n', 128, Fn, ExprLoc); 11428 } 11429 } 11430 FD = FD->getPreviousDecl(); 11431 } 11432 } 11433 11434 namespace { 11435 /// Cleanup action for doacross support. 11436 class DoacrossCleanupTy final : public EHScopeStack::Cleanup { 11437 public: 11438 static const int DoacrossFinArgs = 2; 11439 11440 private: 11441 llvm::FunctionCallee RTLFn; 11442 llvm::Value *Args[DoacrossFinArgs]; 11443 11444 public: 11445 DoacrossCleanupTy(llvm::FunctionCallee RTLFn, 11446 ArrayRef<llvm::Value *> CallArgs) 11447 : RTLFn(RTLFn) { 11448 assert(CallArgs.size() == DoacrossFinArgs); 11449 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 11450 } 11451 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 11452 if (!CGF.HaveInsertPoint()) 11453 return; 11454 CGF.EmitRuntimeCall(RTLFn, Args); 11455 } 11456 }; 11457 } // namespace 11458 11459 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF, 11460 const OMPLoopDirective &D, 11461 ArrayRef<Expr *> NumIterations) { 11462 if (!CGF.HaveInsertPoint()) 11463 return; 11464 11465 ASTContext &C = CGM.getContext(); 11466 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 11467 RecordDecl *RD; 11468 if (KmpDimTy.isNull()) { 11469 // Build struct kmp_dim { // loop bounds info casted to kmp_int64 11470 // kmp_int64 lo; // lower 11471 // kmp_int64 up; // upper 11472 // kmp_int64 st; // stride 11473 // }; 11474 RD = C.buildImplicitRecord("kmp_dim"); 11475 RD->startDefinition(); 11476 addFieldToRecordDecl(C, RD, Int64Ty); 11477 addFieldToRecordDecl(C, RD, Int64Ty); 11478 addFieldToRecordDecl(C, RD, Int64Ty); 11479 RD->completeDefinition(); 11480 KmpDimTy = C.getRecordType(RD); 11481 } else { 11482 RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl()); 11483 } 11484 llvm::APInt Size(/*numBits=*/32, NumIterations.size()); 11485 QualType ArrayTy = 11486 C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0); 11487 11488 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 11489 CGF.EmitNullInitialization(DimsAddr, ArrayTy); 11490 enum { LowerFD = 0, UpperFD, StrideFD }; 11491 // Fill dims with data. 11492 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) { 11493 LValue DimsLVal = CGF.MakeAddrLValue( 11494 CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy); 11495 // dims.upper = num_iterations; 11496 LValue UpperLVal = CGF.EmitLValueForField( 11497 DimsLVal, *std::next(RD->field_begin(), UpperFD)); 11498 llvm::Value *NumIterVal = CGF.EmitScalarConversion( 11499 CGF.EmitScalarExpr(NumIterations[I]), NumIterations[I]->getType(), 11500 Int64Ty, NumIterations[I]->getExprLoc()); 11501 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal); 11502 // dims.stride = 1; 11503 LValue StrideLVal = CGF.EmitLValueForField( 11504 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 11505 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1), 11506 StrideLVal); 11507 } 11508 11509 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, 11510 // kmp_int32 num_dims, struct kmp_dim * dims); 11511 llvm::Value *Args[] = { 11512 emitUpdateLocation(CGF, D.getBeginLoc()), 11513 getThreadID(CGF, D.getBeginLoc()), 11514 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()), 11515 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11516 CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(), 11517 CGM.VoidPtrTy)}; 11518 11519 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction( 11520 CGM.getModule(), OMPRTL___kmpc_doacross_init); 11521 CGF.EmitRuntimeCall(RTLFn, Args); 11522 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = { 11523 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())}; 11524 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction( 11525 CGM.getModule(), OMPRTL___kmpc_doacross_fini); 11526 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11527 llvm::makeArrayRef(FiniArgs)); 11528 } 11529 11530 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 11531 const OMPDependClause *C) { 11532 QualType Int64Ty = 11533 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 11534 llvm::APInt Size(/*numBits=*/32, C->getNumLoops()); 11535 QualType ArrayTy = CGM.getContext().getConstantArrayType( 11536 Int64Ty, Size, nullptr, ArrayType::Normal, 0); 11537 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr"); 11538 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) { 11539 const Expr *CounterVal = C->getLoopData(I); 11540 assert(CounterVal); 11541 llvm::Value *CntVal = CGF.EmitScalarConversion( 11542 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty, 11543 CounterVal->getExprLoc()); 11544 CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I), 11545 /*Volatile=*/false, Int64Ty); 11546 } 11547 llvm::Value *Args[] = { 11548 emitUpdateLocation(CGF, C->getBeginLoc()), 11549 getThreadID(CGF, C->getBeginLoc()), 11550 CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()}; 11551 llvm::FunctionCallee RTLFn; 11552 if (C->getDependencyKind() == OMPC_DEPEND_source) { 11553 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 11554 OMPRTL___kmpc_doacross_post); 11555 } else { 11556 assert(C->getDependencyKind() == OMPC_DEPEND_sink); 11557 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 11558 OMPRTL___kmpc_doacross_wait); 11559 } 11560 CGF.EmitRuntimeCall(RTLFn, Args); 11561 } 11562 11563 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc, 11564 llvm::FunctionCallee Callee, 11565 ArrayRef<llvm::Value *> Args) const { 11566 assert(Loc.isValid() && "Outlined function call location must be valid."); 11567 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 11568 11569 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) { 11570 if (Fn->doesNotThrow()) { 11571 CGF.EmitNounwindRuntimeCall(Fn, Args); 11572 return; 11573 } 11574 } 11575 CGF.EmitRuntimeCall(Callee, Args); 11576 } 11577 11578 void CGOpenMPRuntime::emitOutlinedFunctionCall( 11579 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, 11580 ArrayRef<llvm::Value *> Args) const { 11581 emitCall(CGF, Loc, OutlinedFn, Args); 11582 } 11583 11584 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) { 11585 if (const auto *FD = dyn_cast<FunctionDecl>(D)) 11586 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD)) 11587 HasEmittedDeclareTargetRegion = true; 11588 } 11589 11590 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF, 11591 const VarDecl *NativeParam, 11592 const VarDecl *TargetParam) const { 11593 return CGF.GetAddrOfLocalVar(NativeParam); 11594 } 11595 11596 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF, 11597 const VarDecl *VD) { 11598 if (!VD) 11599 return Address::invalid(); 11600 Address UntiedAddr = Address::invalid(); 11601 Address UntiedRealAddr = Address::invalid(); 11602 auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn); 11603 if (It != FunctionToUntiedTaskStackMap.end()) { 11604 const UntiedLocalVarsAddressesMap &UntiedData = 11605 UntiedLocalVarsStack[It->second]; 11606 auto I = UntiedData.find(VD); 11607 if (I != UntiedData.end()) { 11608 UntiedAddr = I->second.first; 11609 UntiedRealAddr = I->second.second; 11610 } 11611 } 11612 const VarDecl *CVD = VD->getCanonicalDecl(); 11613 if (CVD->hasAttr<OMPAllocateDeclAttr>()) { 11614 // Use the default allocation. 11615 if (!isAllocatableDecl(VD)) 11616 return UntiedAddr; 11617 llvm::Value *Size; 11618 CharUnits Align = CGM.getContext().getDeclAlign(CVD); 11619 if (CVD->getType()->isVariablyModifiedType()) { 11620 Size = CGF.getTypeSize(CVD->getType()); 11621 // Align the size: ((size + align - 1) / align) * align 11622 Size = CGF.Builder.CreateNUWAdd( 11623 Size, CGM.getSize(Align - CharUnits::fromQuantity(1))); 11624 Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align)); 11625 Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align)); 11626 } else { 11627 CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType()); 11628 Size = CGM.getSize(Sz.alignTo(Align)); 11629 } 11630 llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc()); 11631 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 11632 assert(AA->getAllocator() && 11633 "Expected allocator expression for non-default allocator."); 11634 llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator()); 11635 // According to the standard, the original allocator type is a enum 11636 // (integer). Convert to pointer type, if required. 11637 Allocator = CGF.EmitScalarConversion( 11638 Allocator, AA->getAllocator()->getType(), CGF.getContext().VoidPtrTy, 11639 AA->getAllocator()->getExprLoc()); 11640 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 11641 11642 llvm::Value *Addr = 11643 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 11644 CGM.getModule(), OMPRTL___kmpc_alloc), 11645 Args, getName({CVD->getName(), ".void.addr"})); 11646 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction( 11647 CGM.getModule(), OMPRTL___kmpc_free); 11648 QualType Ty = CGM.getContext().getPointerType(CVD->getType()); 11649 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11650 Addr, CGF.ConvertTypeForMem(Ty), getName({CVD->getName(), ".addr"})); 11651 if (UntiedAddr.isValid()) 11652 CGF.EmitStoreOfScalar(Addr, UntiedAddr, /*Volatile=*/false, Ty); 11653 11654 // Cleanup action for allocate support. 11655 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup { 11656 llvm::FunctionCallee RTLFn; 11657 unsigned LocEncoding; 11658 Address Addr; 11659 const Expr *Allocator; 11660 11661 public: 11662 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn, unsigned LocEncoding, 11663 Address Addr, const Expr *Allocator) 11664 : RTLFn(RTLFn), LocEncoding(LocEncoding), Addr(Addr), 11665 Allocator(Allocator) {} 11666 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 11667 if (!CGF.HaveInsertPoint()) 11668 return; 11669 llvm::Value *Args[3]; 11670 Args[0] = CGF.CGM.getOpenMPRuntime().getThreadID( 11671 CGF, SourceLocation::getFromRawEncoding(LocEncoding)); 11672 Args[1] = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11673 Addr.getPointer(), CGF.VoidPtrTy); 11674 llvm::Value *AllocVal = CGF.EmitScalarExpr(Allocator); 11675 // According to the standard, the original allocator type is a enum 11676 // (integer). Convert to pointer type, if required. 11677 AllocVal = CGF.EmitScalarConversion(AllocVal, Allocator->getType(), 11678 CGF.getContext().VoidPtrTy, 11679 Allocator->getExprLoc()); 11680 Args[2] = AllocVal; 11681 11682 CGF.EmitRuntimeCall(RTLFn, Args); 11683 } 11684 }; 11685 Address VDAddr = 11686 UntiedRealAddr.isValid() ? UntiedRealAddr : Address(Addr, Align); 11687 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>( 11688 NormalAndEHCleanup, FiniRTLFn, CVD->getLocation().getRawEncoding(), 11689 VDAddr, AA->getAllocator()); 11690 if (UntiedRealAddr.isValid()) 11691 if (auto *Region = 11692 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 11693 Region->emitUntiedSwitch(CGF); 11694 return VDAddr; 11695 } 11696 return UntiedAddr; 11697 } 11698 11699 bool CGOpenMPRuntime::isLocalVarInUntiedTask(CodeGenFunction &CGF, 11700 const VarDecl *VD) const { 11701 auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn); 11702 if (It == FunctionToUntiedTaskStackMap.end()) 11703 return false; 11704 return UntiedLocalVarsStack[It->second].count(VD) > 0; 11705 } 11706 11707 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII( 11708 CodeGenModule &CGM, const OMPLoopDirective &S) 11709 : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) { 11710 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11711 if (!NeedToPush) 11712 return; 11713 NontemporalDeclsSet &DS = 11714 CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back(); 11715 for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) { 11716 for (const Stmt *Ref : C->private_refs()) { 11717 const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts(); 11718 const ValueDecl *VD; 11719 if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) { 11720 VD = DRE->getDecl(); 11721 } else { 11722 const auto *ME = cast<MemberExpr>(SimpleRefExpr); 11723 assert((ME->isImplicitCXXThis() || 11724 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) && 11725 "Expected member of current class."); 11726 VD = ME->getMemberDecl(); 11727 } 11728 DS.insert(VD); 11729 } 11730 } 11731 } 11732 11733 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() { 11734 if (!NeedToPush) 11735 return; 11736 CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back(); 11737 } 11738 11739 CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::UntiedTaskLocalDeclsRAII( 11740 CodeGenFunction &CGF, 11741 const llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, 11742 std::pair<Address, Address>> &LocalVars) 11743 : CGM(CGF.CGM), NeedToPush(!LocalVars.empty()) { 11744 if (!NeedToPush) 11745 return; 11746 CGM.getOpenMPRuntime().FunctionToUntiedTaskStackMap.try_emplace( 11747 CGF.CurFn, CGM.getOpenMPRuntime().UntiedLocalVarsStack.size()); 11748 CGM.getOpenMPRuntime().UntiedLocalVarsStack.push_back(LocalVars); 11749 } 11750 11751 CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::~UntiedTaskLocalDeclsRAII() { 11752 if (!NeedToPush) 11753 return; 11754 CGM.getOpenMPRuntime().UntiedLocalVarsStack.pop_back(); 11755 } 11756 11757 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const { 11758 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11759 11760 return llvm::any_of( 11761 CGM.getOpenMPRuntime().NontemporalDeclsStack, 11762 [VD](const NontemporalDeclsSet &Set) { return Set.count(VD) > 0; }); 11763 } 11764 11765 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis( 11766 const OMPExecutableDirective &S, 11767 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled) 11768 const { 11769 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs; 11770 // Vars in target/task regions must be excluded completely. 11771 if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) || 11772 isOpenMPTaskingDirective(S.getDirectiveKind())) { 11773 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 11774 getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind()); 11775 const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front()); 11776 for (const CapturedStmt::Capture &Cap : CS->captures()) { 11777 if (Cap.capturesVariable() || Cap.capturesVariableByCopy()) 11778 NeedToCheckForLPCs.insert(Cap.getCapturedVar()); 11779 } 11780 } 11781 // Exclude vars in private clauses. 11782 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) { 11783 for (const Expr *Ref : C->varlists()) { 11784 if (!Ref->getType()->isScalarType()) 11785 continue; 11786 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11787 if (!DRE) 11788 continue; 11789 NeedToCheckForLPCs.insert(DRE->getDecl()); 11790 } 11791 } 11792 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) { 11793 for (const Expr *Ref : C->varlists()) { 11794 if (!Ref->getType()->isScalarType()) 11795 continue; 11796 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11797 if (!DRE) 11798 continue; 11799 NeedToCheckForLPCs.insert(DRE->getDecl()); 11800 } 11801 } 11802 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11803 for (const Expr *Ref : C->varlists()) { 11804 if (!Ref->getType()->isScalarType()) 11805 continue; 11806 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11807 if (!DRE) 11808 continue; 11809 NeedToCheckForLPCs.insert(DRE->getDecl()); 11810 } 11811 } 11812 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) { 11813 for (const Expr *Ref : C->varlists()) { 11814 if (!Ref->getType()->isScalarType()) 11815 continue; 11816 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11817 if (!DRE) 11818 continue; 11819 NeedToCheckForLPCs.insert(DRE->getDecl()); 11820 } 11821 } 11822 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) { 11823 for (const Expr *Ref : C->varlists()) { 11824 if (!Ref->getType()->isScalarType()) 11825 continue; 11826 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11827 if (!DRE) 11828 continue; 11829 NeedToCheckForLPCs.insert(DRE->getDecl()); 11830 } 11831 } 11832 for (const Decl *VD : NeedToCheckForLPCs) { 11833 for (const LastprivateConditionalData &Data : 11834 llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) { 11835 if (Data.DeclToUniqueName.count(VD) > 0) { 11836 if (!Data.Disabled) 11837 NeedToAddForLPCsAsDisabled.insert(VD); 11838 break; 11839 } 11840 } 11841 } 11842 } 11843 11844 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11845 CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal) 11846 : CGM(CGF.CGM), 11847 Action((CGM.getLangOpts().OpenMP >= 50 && 11848 llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(), 11849 [](const OMPLastprivateClause *C) { 11850 return C->getKind() == 11851 OMPC_LASTPRIVATE_conditional; 11852 })) 11853 ? ActionToDo::PushAsLastprivateConditional 11854 : ActionToDo::DoNotPush) { 11855 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11856 if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush) 11857 return; 11858 assert(Action == ActionToDo::PushAsLastprivateConditional && 11859 "Expected a push action."); 11860 LastprivateConditionalData &Data = 11861 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11862 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11863 if (C->getKind() != OMPC_LASTPRIVATE_conditional) 11864 continue; 11865 11866 for (const Expr *Ref : C->varlists()) { 11867 Data.DeclToUniqueName.insert(std::make_pair( 11868 cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(), 11869 SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref)))); 11870 } 11871 } 11872 Data.IVLVal = IVLVal; 11873 Data.Fn = CGF.CurFn; 11874 } 11875 11876 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11877 CodeGenFunction &CGF, const OMPExecutableDirective &S) 11878 : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) { 11879 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11880 if (CGM.getLangOpts().OpenMP < 50) 11881 return; 11882 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled; 11883 tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled); 11884 if (!NeedToAddForLPCsAsDisabled.empty()) { 11885 Action = ActionToDo::DisableLastprivateConditional; 11886 LastprivateConditionalData &Data = 11887 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11888 for (const Decl *VD : NeedToAddForLPCsAsDisabled) 11889 Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>())); 11890 Data.Fn = CGF.CurFn; 11891 Data.Disabled = true; 11892 } 11893 } 11894 11895 CGOpenMPRuntime::LastprivateConditionalRAII 11896 CGOpenMPRuntime::LastprivateConditionalRAII::disable( 11897 CodeGenFunction &CGF, const OMPExecutableDirective &S) { 11898 return LastprivateConditionalRAII(CGF, S); 11899 } 11900 11901 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() { 11902 if (CGM.getLangOpts().OpenMP < 50) 11903 return; 11904 if (Action == ActionToDo::DisableLastprivateConditional) { 11905 assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11906 "Expected list of disabled private vars."); 11907 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11908 } 11909 if (Action == ActionToDo::PushAsLastprivateConditional) { 11910 assert( 11911 !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11912 "Expected list of lastprivate conditional vars."); 11913 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11914 } 11915 } 11916 11917 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF, 11918 const VarDecl *VD) { 11919 ASTContext &C = CGM.getContext(); 11920 auto I = LastprivateConditionalToTypes.find(CGF.CurFn); 11921 if (I == LastprivateConditionalToTypes.end()) 11922 I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first; 11923 QualType NewType; 11924 const FieldDecl *VDField; 11925 const FieldDecl *FiredField; 11926 LValue BaseLVal; 11927 auto VI = I->getSecond().find(VD); 11928 if (VI == I->getSecond().end()) { 11929 RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional"); 11930 RD->startDefinition(); 11931 VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType()); 11932 FiredField = addFieldToRecordDecl(C, RD, C.CharTy); 11933 RD->completeDefinition(); 11934 NewType = C.getRecordType(RD); 11935 Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName()); 11936 BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl); 11937 I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal); 11938 } else { 11939 NewType = std::get<0>(VI->getSecond()); 11940 VDField = std::get<1>(VI->getSecond()); 11941 FiredField = std::get<2>(VI->getSecond()); 11942 BaseLVal = std::get<3>(VI->getSecond()); 11943 } 11944 LValue FiredLVal = 11945 CGF.EmitLValueForField(BaseLVal, FiredField); 11946 CGF.EmitStoreOfScalar( 11947 llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)), 11948 FiredLVal); 11949 return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF); 11950 } 11951 11952 namespace { 11953 /// Checks if the lastprivate conditional variable is referenced in LHS. 11954 class LastprivateConditionalRefChecker final 11955 : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> { 11956 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM; 11957 const Expr *FoundE = nullptr; 11958 const Decl *FoundD = nullptr; 11959 StringRef UniqueDeclName; 11960 LValue IVLVal; 11961 llvm::Function *FoundFn = nullptr; 11962 SourceLocation Loc; 11963 11964 public: 11965 bool VisitDeclRefExpr(const DeclRefExpr *E) { 11966 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11967 llvm::reverse(LPM)) { 11968 auto It = D.DeclToUniqueName.find(E->getDecl()); 11969 if (It == D.DeclToUniqueName.end()) 11970 continue; 11971 if (D.Disabled) 11972 return false; 11973 FoundE = E; 11974 FoundD = E->getDecl()->getCanonicalDecl(); 11975 UniqueDeclName = It->second; 11976 IVLVal = D.IVLVal; 11977 FoundFn = D.Fn; 11978 break; 11979 } 11980 return FoundE == E; 11981 } 11982 bool VisitMemberExpr(const MemberExpr *E) { 11983 if (!CodeGenFunction::IsWrappedCXXThis(E->getBase())) 11984 return false; 11985 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11986 llvm::reverse(LPM)) { 11987 auto It = D.DeclToUniqueName.find(E->getMemberDecl()); 11988 if (It == D.DeclToUniqueName.end()) 11989 continue; 11990 if (D.Disabled) 11991 return false; 11992 FoundE = E; 11993 FoundD = E->getMemberDecl()->getCanonicalDecl(); 11994 UniqueDeclName = It->second; 11995 IVLVal = D.IVLVal; 11996 FoundFn = D.Fn; 11997 break; 11998 } 11999 return FoundE == E; 12000 } 12001 bool VisitStmt(const Stmt *S) { 12002 for (const Stmt *Child : S->children()) { 12003 if (!Child) 12004 continue; 12005 if (const auto *E = dyn_cast<Expr>(Child)) 12006 if (!E->isGLValue()) 12007 continue; 12008 if (Visit(Child)) 12009 return true; 12010 } 12011 return false; 12012 } 12013 explicit LastprivateConditionalRefChecker( 12014 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM) 12015 : LPM(LPM) {} 12016 std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *> 12017 getFoundData() const { 12018 return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn); 12019 } 12020 }; 12021 } // namespace 12022 12023 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF, 12024 LValue IVLVal, 12025 StringRef UniqueDeclName, 12026 LValue LVal, 12027 SourceLocation Loc) { 12028 // Last updated loop counter for the lastprivate conditional var. 12029 // int<xx> last_iv = 0; 12030 llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType()); 12031 llvm::Constant *LastIV = 12032 getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"})); 12033 cast<llvm::GlobalVariable>(LastIV)->setAlignment( 12034 IVLVal.getAlignment().getAsAlign()); 12035 LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType()); 12036 12037 // Last value of the lastprivate conditional. 12038 // decltype(priv_a) last_a; 12039 llvm::Constant *Last = getOrCreateInternalVariable( 12040 CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName); 12041 cast<llvm::GlobalVariable>(Last)->setAlignment( 12042 LVal.getAlignment().getAsAlign()); 12043 LValue LastLVal = 12044 CGF.MakeAddrLValue(Last, LVal.getType(), LVal.getAlignment()); 12045 12046 // Global loop counter. Required to handle inner parallel-for regions. 12047 // iv 12048 llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc); 12049 12050 // #pragma omp critical(a) 12051 // if (last_iv <= iv) { 12052 // last_iv = iv; 12053 // last_a = priv_a; 12054 // } 12055 auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal, 12056 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 12057 Action.Enter(CGF); 12058 llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc); 12059 // (last_iv <= iv) ? Check if the variable is updated and store new 12060 // value in global var. 12061 llvm::Value *CmpRes; 12062 if (IVLVal.getType()->isSignedIntegerType()) { 12063 CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal); 12064 } else { 12065 assert(IVLVal.getType()->isUnsignedIntegerType() && 12066 "Loop iteration variable must be integer."); 12067 CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal); 12068 } 12069 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then"); 12070 llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit"); 12071 CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB); 12072 // { 12073 CGF.EmitBlock(ThenBB); 12074 12075 // last_iv = iv; 12076 CGF.EmitStoreOfScalar(IVVal, LastIVLVal); 12077 12078 // last_a = priv_a; 12079 switch (CGF.getEvaluationKind(LVal.getType())) { 12080 case TEK_Scalar: { 12081 llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc); 12082 CGF.EmitStoreOfScalar(PrivVal, LastLVal); 12083 break; 12084 } 12085 case TEK_Complex: { 12086 CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc); 12087 CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false); 12088 break; 12089 } 12090 case TEK_Aggregate: 12091 llvm_unreachable( 12092 "Aggregates are not supported in lastprivate conditional."); 12093 } 12094 // } 12095 CGF.EmitBranch(ExitBB); 12096 // There is no need to emit line number for unconditional branch. 12097 (void)ApplyDebugLocation::CreateEmpty(CGF); 12098 CGF.EmitBlock(ExitBB, /*IsFinished=*/true); 12099 }; 12100 12101 if (CGM.getLangOpts().OpenMPSimd) { 12102 // Do not emit as a critical region as no parallel region could be emitted. 12103 RegionCodeGenTy ThenRCG(CodeGen); 12104 ThenRCG(CGF); 12105 } else { 12106 emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc); 12107 } 12108 } 12109 12110 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF, 12111 const Expr *LHS) { 12112 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 12113 return; 12114 LastprivateConditionalRefChecker Checker(LastprivateConditionalStack); 12115 if (!Checker.Visit(LHS)) 12116 return; 12117 const Expr *FoundE; 12118 const Decl *FoundD; 12119 StringRef UniqueDeclName; 12120 LValue IVLVal; 12121 llvm::Function *FoundFn; 12122 std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) = 12123 Checker.getFoundData(); 12124 if (FoundFn != CGF.CurFn) { 12125 // Special codegen for inner parallel regions. 12126 // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1; 12127 auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD); 12128 assert(It != LastprivateConditionalToTypes[FoundFn].end() && 12129 "Lastprivate conditional is not found in outer region."); 12130 QualType StructTy = std::get<0>(It->getSecond()); 12131 const FieldDecl* FiredDecl = std::get<2>(It->getSecond()); 12132 LValue PrivLVal = CGF.EmitLValue(FoundE); 12133 Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 12134 PrivLVal.getAddress(CGF), 12135 CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy))); 12136 LValue BaseLVal = 12137 CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl); 12138 LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl); 12139 CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get( 12140 CGF.ConvertTypeForMem(FiredDecl->getType()), 1)), 12141 FiredLVal, llvm::AtomicOrdering::Unordered, 12142 /*IsVolatile=*/true, /*isInit=*/false); 12143 return; 12144 } 12145 12146 // Private address of the lastprivate conditional in the current context. 12147 // priv_a 12148 LValue LVal = CGF.EmitLValue(FoundE); 12149 emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal, 12150 FoundE->getExprLoc()); 12151 } 12152 12153 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional( 12154 CodeGenFunction &CGF, const OMPExecutableDirective &D, 12155 const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) { 12156 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 12157 return; 12158 auto Range = llvm::reverse(LastprivateConditionalStack); 12159 auto It = llvm::find_if( 12160 Range, [](const LastprivateConditionalData &D) { return !D.Disabled; }); 12161 if (It == Range.end() || It->Fn != CGF.CurFn) 12162 return; 12163 auto LPCI = LastprivateConditionalToTypes.find(It->Fn); 12164 assert(LPCI != LastprivateConditionalToTypes.end() && 12165 "Lastprivates must be registered already."); 12166 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 12167 getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind()); 12168 const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back()); 12169 for (const auto &Pair : It->DeclToUniqueName) { 12170 const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl()); 12171 if (!CS->capturesVariable(VD) || IgnoredDecls.count(VD) > 0) 12172 continue; 12173 auto I = LPCI->getSecond().find(Pair.first); 12174 assert(I != LPCI->getSecond().end() && 12175 "Lastprivate must be rehistered already."); 12176 // bool Cmp = priv_a.Fired != 0; 12177 LValue BaseLVal = std::get<3>(I->getSecond()); 12178 LValue FiredLVal = 12179 CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond())); 12180 llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc()); 12181 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res); 12182 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then"); 12183 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done"); 12184 // if (Cmp) { 12185 CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB); 12186 CGF.EmitBlock(ThenBB); 12187 Address Addr = CGF.GetAddrOfLocalVar(VD); 12188 LValue LVal; 12189 if (VD->getType()->isReferenceType()) 12190 LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(), 12191 AlignmentSource::Decl); 12192 else 12193 LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(), 12194 AlignmentSource::Decl); 12195 emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal, 12196 D.getBeginLoc()); 12197 auto AL = ApplyDebugLocation::CreateArtificial(CGF); 12198 CGF.EmitBlock(DoneBB, /*IsFinal=*/true); 12199 // } 12200 } 12201 } 12202 12203 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate( 12204 CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD, 12205 SourceLocation Loc) { 12206 if (CGF.getLangOpts().OpenMP < 50) 12207 return; 12208 auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD); 12209 assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() && 12210 "Unknown lastprivate conditional variable."); 12211 StringRef UniqueName = It->second; 12212 llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName); 12213 // The variable was not updated in the region - exit. 12214 if (!GV) 12215 return; 12216 LValue LPLVal = CGF.MakeAddrLValue( 12217 GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment()); 12218 llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc); 12219 CGF.EmitStoreOfScalar(Res, PrivLVal); 12220 } 12221 12222 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction( 12223 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 12224 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 12225 llvm_unreachable("Not supported in SIMD-only mode"); 12226 } 12227 12228 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction( 12229 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 12230 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 12231 llvm_unreachable("Not supported in SIMD-only mode"); 12232 } 12233 12234 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction( 12235 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 12236 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 12237 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 12238 bool Tied, unsigned &NumberOfParts) { 12239 llvm_unreachable("Not supported in SIMD-only mode"); 12240 } 12241 12242 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF, 12243 SourceLocation Loc, 12244 llvm::Function *OutlinedFn, 12245 ArrayRef<llvm::Value *> CapturedVars, 12246 const Expr *IfCond) { 12247 llvm_unreachable("Not supported in SIMD-only mode"); 12248 } 12249 12250 void CGOpenMPSIMDRuntime::emitCriticalRegion( 12251 CodeGenFunction &CGF, StringRef CriticalName, 12252 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, 12253 const Expr *Hint) { 12254 llvm_unreachable("Not supported in SIMD-only mode"); 12255 } 12256 12257 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF, 12258 const RegionCodeGenTy &MasterOpGen, 12259 SourceLocation Loc) { 12260 llvm_unreachable("Not supported in SIMD-only mode"); 12261 } 12262 12263 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 12264 SourceLocation Loc) { 12265 llvm_unreachable("Not supported in SIMD-only mode"); 12266 } 12267 12268 void CGOpenMPSIMDRuntime::emitTaskgroupRegion( 12269 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, 12270 SourceLocation Loc) { 12271 llvm_unreachable("Not supported in SIMD-only mode"); 12272 } 12273 12274 void CGOpenMPSIMDRuntime::emitSingleRegion( 12275 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, 12276 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars, 12277 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs, 12278 ArrayRef<const Expr *> AssignmentOps) { 12279 llvm_unreachable("Not supported in SIMD-only mode"); 12280 } 12281 12282 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF, 12283 const RegionCodeGenTy &OrderedOpGen, 12284 SourceLocation Loc, 12285 bool IsThreads) { 12286 llvm_unreachable("Not supported in SIMD-only mode"); 12287 } 12288 12289 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF, 12290 SourceLocation Loc, 12291 OpenMPDirectiveKind Kind, 12292 bool EmitChecks, 12293 bool ForceSimpleCall) { 12294 llvm_unreachable("Not supported in SIMD-only mode"); 12295 } 12296 12297 void CGOpenMPSIMDRuntime::emitForDispatchInit( 12298 CodeGenFunction &CGF, SourceLocation Loc, 12299 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 12300 bool Ordered, const DispatchRTInput &DispatchValues) { 12301 llvm_unreachable("Not supported in SIMD-only mode"); 12302 } 12303 12304 void CGOpenMPSIMDRuntime::emitForStaticInit( 12305 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, 12306 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) { 12307 llvm_unreachable("Not supported in SIMD-only mode"); 12308 } 12309 12310 void CGOpenMPSIMDRuntime::emitDistributeStaticInit( 12311 CodeGenFunction &CGF, SourceLocation Loc, 12312 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) { 12313 llvm_unreachable("Not supported in SIMD-only mode"); 12314 } 12315 12316 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 12317 SourceLocation Loc, 12318 unsigned IVSize, 12319 bool IVSigned) { 12320 llvm_unreachable("Not supported in SIMD-only mode"); 12321 } 12322 12323 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF, 12324 SourceLocation Loc, 12325 OpenMPDirectiveKind DKind) { 12326 llvm_unreachable("Not supported in SIMD-only mode"); 12327 } 12328 12329 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF, 12330 SourceLocation Loc, 12331 unsigned IVSize, bool IVSigned, 12332 Address IL, Address LB, 12333 Address UB, Address ST) { 12334 llvm_unreachable("Not supported in SIMD-only mode"); 12335 } 12336 12337 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 12338 llvm::Value *NumThreads, 12339 SourceLocation Loc) { 12340 llvm_unreachable("Not supported in SIMD-only mode"); 12341 } 12342 12343 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF, 12344 ProcBindKind ProcBind, 12345 SourceLocation Loc) { 12346 llvm_unreachable("Not supported in SIMD-only mode"); 12347 } 12348 12349 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 12350 const VarDecl *VD, 12351 Address VDAddr, 12352 SourceLocation Loc) { 12353 llvm_unreachable("Not supported in SIMD-only mode"); 12354 } 12355 12356 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition( 12357 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, 12358 CodeGenFunction *CGF) { 12359 llvm_unreachable("Not supported in SIMD-only mode"); 12360 } 12361 12362 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate( 12363 CodeGenFunction &CGF, QualType VarType, StringRef Name) { 12364 llvm_unreachable("Not supported in SIMD-only mode"); 12365 } 12366 12367 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF, 12368 ArrayRef<const Expr *> Vars, 12369 SourceLocation Loc, 12370 llvm::AtomicOrdering AO) { 12371 llvm_unreachable("Not supported in SIMD-only mode"); 12372 } 12373 12374 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 12375 const OMPExecutableDirective &D, 12376 llvm::Function *TaskFunction, 12377 QualType SharedsTy, Address Shareds, 12378 const Expr *IfCond, 12379 const OMPTaskDataTy &Data) { 12380 llvm_unreachable("Not supported in SIMD-only mode"); 12381 } 12382 12383 void CGOpenMPSIMDRuntime::emitTaskLoopCall( 12384 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, 12385 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, 12386 const Expr *IfCond, const OMPTaskDataTy &Data) { 12387 llvm_unreachable("Not supported in SIMD-only mode"); 12388 } 12389 12390 void CGOpenMPSIMDRuntime::emitReduction( 12391 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates, 12392 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 12393 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) { 12394 assert(Options.SimpleReduction && "Only simple reduction is expected."); 12395 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs, 12396 ReductionOps, Options); 12397 } 12398 12399 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit( 12400 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 12401 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 12402 llvm_unreachable("Not supported in SIMD-only mode"); 12403 } 12404 12405 void CGOpenMPSIMDRuntime::emitTaskReductionFini(CodeGenFunction &CGF, 12406 SourceLocation Loc, 12407 bool IsWorksharingReduction) { 12408 llvm_unreachable("Not supported in SIMD-only mode"); 12409 } 12410 12411 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 12412 SourceLocation Loc, 12413 ReductionCodeGen &RCG, 12414 unsigned N) { 12415 llvm_unreachable("Not supported in SIMD-only mode"); 12416 } 12417 12418 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF, 12419 SourceLocation Loc, 12420 llvm::Value *ReductionsPtr, 12421 LValue SharedLVal) { 12422 llvm_unreachable("Not supported in SIMD-only mode"); 12423 } 12424 12425 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 12426 SourceLocation Loc) { 12427 llvm_unreachable("Not supported in SIMD-only mode"); 12428 } 12429 12430 void CGOpenMPSIMDRuntime::emitCancellationPointCall( 12431 CodeGenFunction &CGF, SourceLocation Loc, 12432 OpenMPDirectiveKind CancelRegion) { 12433 llvm_unreachable("Not supported in SIMD-only mode"); 12434 } 12435 12436 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF, 12437 SourceLocation Loc, const Expr *IfCond, 12438 OpenMPDirectiveKind CancelRegion) { 12439 llvm_unreachable("Not supported in SIMD-only mode"); 12440 } 12441 12442 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction( 12443 const OMPExecutableDirective &D, StringRef ParentName, 12444 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 12445 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 12446 llvm_unreachable("Not supported in SIMD-only mode"); 12447 } 12448 12449 void CGOpenMPSIMDRuntime::emitTargetCall( 12450 CodeGenFunction &CGF, const OMPExecutableDirective &D, 12451 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 12452 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 12453 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 12454 const OMPLoopDirective &D)> 12455 SizeEmitter) { 12456 llvm_unreachable("Not supported in SIMD-only mode"); 12457 } 12458 12459 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) { 12460 llvm_unreachable("Not supported in SIMD-only mode"); 12461 } 12462 12463 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 12464 llvm_unreachable("Not supported in SIMD-only mode"); 12465 } 12466 12467 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) { 12468 return false; 12469 } 12470 12471 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF, 12472 const OMPExecutableDirective &D, 12473 SourceLocation Loc, 12474 llvm::Function *OutlinedFn, 12475 ArrayRef<llvm::Value *> CapturedVars) { 12476 llvm_unreachable("Not supported in SIMD-only mode"); 12477 } 12478 12479 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 12480 const Expr *NumTeams, 12481 const Expr *ThreadLimit, 12482 SourceLocation Loc) { 12483 llvm_unreachable("Not supported in SIMD-only mode"); 12484 } 12485 12486 void CGOpenMPSIMDRuntime::emitTargetDataCalls( 12487 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12488 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 12489 llvm_unreachable("Not supported in SIMD-only mode"); 12490 } 12491 12492 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall( 12493 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12494 const Expr *Device) { 12495 llvm_unreachable("Not supported in SIMD-only mode"); 12496 } 12497 12498 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF, 12499 const OMPLoopDirective &D, 12500 ArrayRef<Expr *> NumIterations) { 12501 llvm_unreachable("Not supported in SIMD-only mode"); 12502 } 12503 12504 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 12505 const OMPDependClause *C) { 12506 llvm_unreachable("Not supported in SIMD-only mode"); 12507 } 12508 12509 const VarDecl * 12510 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD, 12511 const VarDecl *NativeParam) const { 12512 llvm_unreachable("Not supported in SIMD-only mode"); 12513 } 12514 12515 Address 12516 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF, 12517 const VarDecl *NativeParam, 12518 const VarDecl *TargetParam) const { 12519 llvm_unreachable("Not supported in SIMD-only mode"); 12520 } 12521